#!/usr/bin/env python3
"""
ports_cpe_matcher.py - find CPE candidates for FreeBSD ports and audit existing CPE data.

Workflow
--------
  fetch     Clone/update the git repositories and download the raw data sources
            into the cache directory. Run it whenever you want fresher data.
  extract   Run `make -V` for every port of a ports tree and store the result in
            the cache. Slow (one make per port); rerun after updating the tree.
  match     Match the extracted ports against whatever sources are in the cache
            and write candidates.csv, cpe_issues.csv and report.json. Never
            touches the network.

Cache directory
---------------
$PORTS_CPE_MATCHER_CACHE, else $XDG_CACHE_HOME/ports-cpe-matcher, else
~/.cache/ports-cpe-matcher (or --cache-dir):

  state.json                      fetch bookkeeping (timestamps, HTTP validators)
  ports.jsonl                     output of extract
  report/                         output of match
  nvd/nvdcpe-2.0.zip              NVD CPE Dictionary 2.0 feed
  nvd/cve/nvdcve-2.0-YYYY.json.gz NVD CVE 2.0 yearly feeds
  osv/cpe_product_to_repo.json    osv.dev CPE-to-repository map
  repology/repology.json          Repology projects packaged in FreeBSD
  git/gentoo, git/buildroot,      shallow partial clones with sparse checkouts,
  git/openembedded-core,          so only the needed files are downloaded
  git/meta-openembedded,
  git/cvelistV5
  nixpkgs/packages.json           not fetched; create it yourself if you have Nix,
                                  e.g. nix-env -f '<nixpkgs>' -qa --meta --json '*'
                                  (untested; the loader reads meta.identifiers)

`fetch` needs git for the repositories. Repology requires bulk clients to send
a User-Agent that links to their source repository: pass --user-agent or set
PORTS_CPE_MATCHER_USER_AGENT, otherwise the repology source is skipped.

Scoring
-------
Every piece of evidence belongs to a family (gentoo, buildroot, nvd-cpe-refs,
name, ...). A candidate's score combines the strongest weight per family with
a noisy-OR, score = 1 - prod(1 - w), so repeated hits from one source cannot
inflate confidence. "name", "vendor-hint" and "version" are weak families: a
candidate supported only by them is always "low", as is any candidate NVD does
not know. Weights are defined at the top of this file.

Only the Python standard library is used (Python 3.8+).
"""
from __future__ import annotations

import argparse
import csv
import gzip
import hashlib
import json
import os
import re
import shutil
import signal
import subprocess
import sys
import tarfile
import time
import urllib.error
import urllib.parse
import urllib.request
import xml.etree.ElementTree as ET
import zipfile
import zlib
from collections import Counter, defaultdict
from concurrent.futures import ThreadPoolExecutor, as_completed
from dataclasses import dataclass, field
from datetime import datetime, timezone
from pathlib import Path
from typing import Iterator

# ---------------------------------------------------------------------------
# Evidence weights and thresholds
# ---------------------------------------------------------------------------
W_DISTRO_REPOLOGY = 0.55  # other distro's CPE, joined via the same Repology project
W_DISTRO_URL = 0.50       # other distro's CPE, joined via the same upstream repo/project
W_DISTRO_HOST = 0.30      # other distro's CPE, joined via the same homepage host
W_DISTRO_NAME = 0.20      # other distro's CPE, joined by package name only
W_OSV_REPO = 0.35         # osv.dev CPE->repo map (derived from NVD references)
W_REF_STRONG = 0.45       # NVD dictionary reference to the port's repo/project
W_REF_HOST = 0.25         # NVD dictionary reference to the port's homepage host
W_CVE_REF = 0.25          # CVE references to the port's repo (+ up to 0.15 for many CVEs)
W_NAME_UNIQUE = 0.15      # NVD product name equals port name, single vendor
W_NAME_SHARED = 0.08      # NVD product name equals port name, several vendors
W_VENDOR_HINT = 0.10      # vendor equals GH account, homepage domain, <name>_project ...
W_VERSION = 0.15          # the port's version is a known version of the product

HIGH, MEDIUM = 0.70, 0.40
WEAK_FAMILIES = {"name", "vendor-hint", "version"}
MAX_FANOUT = 40           # ignore URL keys shared by more products than this
SEVERITY_ORDER = {"high": 0, "medium": 1, "low": 2}


def log(msg: str) -> None:
    print(msg, file=sys.stderr, flush=True)


def norm(s: str | None) -> str:
    """Normalise a name for fuzzy comparison: lowercase alphanumerics only."""
    return re.sub(r"[^a-z0-9]+", "", (s or "").lower())


# ---------------------------------------------------------------------------
# CPE helpers
# ---------------------------------------------------------------------------
CPE23_SPLIT = re.compile(r"(?<!\\):")
CPE23_FIELDS = ["cpe", "2.3", "part", "vendor", "product", "version", "update",
                "edition", "language", "sw_edition", "target_sw", "target_hw", "other"]


@dataclass(frozen=True)
class Cpe:
    part: str
    vendor: str
    product: str
    version: str = ""
    update: str = ""

    @property
    def vp(self) -> str:
        return vp_join(self.vendor, self.product)


def _unescape(value: str) -> str:
    return re.sub(r"\\(.)", r"\1", value or "").lower()


def vp_join(vendor: str, product: str) -> str:
    """Key for a vendor/product pair; colons inside components stay escaped."""
    return f"{vendor.replace(':', chr(92) + ':')}:{product.replace(':', chr(92) + ':')}"


def vp_split(vp: str) -> tuple[str, str]:
    vendor, product = (CPE23_SPLIT.split(vp, maxsplit=1) + [""])[:2]
    return vendor.replace(chr(92) + ":", ":"), product.replace(chr(92) + ":", ":")


def _blank(value: str) -> str:
    return "" if value in ("*", "-") else value


def parse_cpe(s: str) -> Cpe | None:
    """Parse a CPE 2.3 formatted string or a CPE 2.2 URI (cpe:/a:vendor:product)."""
    if not isinstance(s, str):
        return None
    s = s.strip()
    if s.startswith("cpe:2.3:"):
        f = CPE23_SPLIT.split(s)
        if len(f) < 5:
            return None
        f += [""] * (7 - len(f))
        part, vendor, product = f[2].lower(), _unescape(f[3]), _unescape(f[4])
        version, update = _blank(_unescape(f[5])), _blank(_unescape(f[6]))
    elif s.startswith("cpe:/"):
        f = s[5:].split(":")
        f += [""] * (5 - len(f))
        f = [urllib.parse.unquote(x).lower() for x in f]
        part, vendor, product, version, update = (f[0] or "a"), f[1], f[2], _blank(f[3]), _blank(f[4])
    else:
        return None
    if vendor in ("", "*", "-") or product in ("", "*", "-"):
        return None
    return Cpe(part, vendor, product, version, update)


def cpe_str_problems(s: str) -> list[str]:
    """Syntax problems in a port's CPE_STR."""
    if not s:
        return ["USES=cpe is set but CPE_STR is empty"]
    if not s.startswith("cpe:2.3:"):
        return [f"CPE_STR does not start with 'cpe:2.3:' ({s})"]
    f = CPE23_SPLIT.split(s)
    problems = []
    if len(f) > 13:
        problems.append(f"{len(f)} fields instead of at most 13 (unescaped ':' inside a component?)")
    if len(f) < 5:
        return problems + ["vendor or product is missing"]
    if f[2] not in ("a", "o", "h"):
        problems.append(f"part '{f[2]}' is not one of a/o/h")
    for idx in (3, 4):
        if f[idx] in ("", "*", "-"):
            problems.append(f"{CPE23_FIELDS[idx]} is empty or a wildcard")
    for idx, comp in enumerate(f[3:13], start=3):
        if comp in ("*", "-", ""):
            continue
        name = CPE23_FIELDS[idx] if idx < len(CPE23_FIELDS) else f"field{idx}"
        if any(ch.isupper() for ch in comp):
            problems.append(f"{name} '{comp}' contains uppercase letters (NVD names are lowercase)")
        bad = set(re.sub(r"\\.|[A-Za-z0-9._-]", "", comp))
        if bad:
            problems.append(f"{name} '{comp}' contains unescaped character(s) {''.join(sorted(bad))!r}")
    return problems


# ---------------------------------------------------------------------------
# URL keys: turn URLs into comparable identifiers
#   repo:<host>/<owner>/<name>   source repositories on forges
#   proj:<platform>/<name>       projects on PyPI, SourceForge, GNU, CPAN, ...
#   host:<host>                  plain homepage host (weak)
# ---------------------------------------------------------------------------
GENERIC_HOSTS = {
    "github.com", "codeload.github.com", "raw.githubusercontent.com", "api.github.com",
    "objects.githubusercontent.com", "gist.github.com", "gitlab.com", "codeberg.org",
    "bitbucket.org", "git.sr.ht", "sr.ht", "gitea.com", "sourceforge.net", "sf.net",
    "downloads.sourceforge.net", "download.sourceforge.net", "prdownloads.sourceforge.net",
    "pypi.org", "pypi.python.org", "files.pythonhosted.org", "pypi.io", "rubygems.org",
    "npmjs.com", "registry.npmjs.org", "crates.io", "static.crates.io", "cpan.org",
    "metacpan.org", "cpan.metacpan.org", "search.cpan.org", "hackage.haskell.org",
    "gnu.org", "ftp.gnu.org", "ftpmirror.gnu.org", "alpha.gnu.org", "savannah.gnu.org",
    "savannah.nongnu.org", "download.savannah.gnu.org", "download.savannah.nongnu.org",
    "apache.org", "archive.apache.org", "downloads.apache.org", "dlcdn.apache.org",
    "dist.apache.org", "download.gnome.org", "gitlab.gnome.org", "download.kde.org",
    "invent.kde.org", "gitlab.freedesktop.org", "launchpad.net", "code.launchpad.net",
    "archive.org", "web.archive.org", "code.google.com", "storage.googleapis.com",
    "freebsd.org", "people.freebsd.org", "distcache.freebsd.org", "pkg.freebsd.org",
    "fossies.org", "ctan.org", "mirrors.ctan.org", "salsa.debian.org", "deb.debian.org",
    "debian.org", "golang.org", "pkg.go.dev", "proxy.golang.org", "go.dev",
    "nvd.nist.gov", "cve.mitre.org", "cve.org", "readthedocs.io", "readthedocs.org",
}
OTHER_GITLAB_HOSTS = {"invent.kde.org", "salsa.debian.org", "framagit.org", "code.videolan.org"}
GITHUB_RESERVED = {"advisories", "orgs", "users", "topics", "sponsors", "marketplace",
                   "features", "collections", "enterprise", "settings", "notifications",
                   "login", "about", "site", "security", "search", "apps", "pulls", "issues"}
PLATFORM_LABELS = {"github", "gitlab", "sourceforge", "sf", "pypi", "pythonhosted", "codeberg",
                   "bitbucket", "freebsd", "googlecode", "google", "launchpad", "metacpan",
                   "cpan", "rubygems", "npmjs", "crates", "readthedocs", "co", "com", "org",
                   "net", "ac", "gov", "edu", "archive"}


def _pep503(name: str) -> str:
    return re.sub(r"[-_.]+", "-", name).lower()


def url_keys(url: str) -> set[str]:
    if not isinstance(url, str):
        return set()
    url = url.strip().strip("<>\"'()[],;")
    m = re.match(r"^(.*/):[A-Za-z0-9_,]+$", url)  # ports MASTER_SITES group suffix
    if m:
        url = m.group(1)
    if "://" not in url:
        return set()
    try:
        u = urllib.parse.urlsplit(url)
        host = (u.hostname or "").lower()
    except ValueError:
        return set()
    if not host:
        return set()
    if host.startswith("www."):
        host = host[4:]
    parts = [urllib.parse.unquote(p).strip().lower() for p in u.path.split("/") if p]
    keys: set[str] = set()

    def repo(h: str, owner: str, name: str) -> None:
        if name.endswith(".git"):
            name = name[:-4]
        if owner and name:
            keys.add(f"repo:{h}/{owner}/{name}")

    if host in ("github.com", "codeload.github.com", "raw.githubusercontent.com"):
        if len(parts) >= 2 and parts[0] not in GITHUB_RESERVED:
            repo("github.com", parts[0], parts[1])
    elif host == "api.github.com":
        if len(parts) >= 3 and parts[0] == "repos":
            repo("github.com", parts[1], parts[2])
    elif host.endswith(".github.io"):
        owner = host[: -len(".github.io")]
        if parts:
            repo("github.com", owner, parts[0])
        keys.add(f"host:{host}")
    elif host == "gitlab.com" or host.startswith("gitlab.") or host in OTHER_GITLAB_HOSTS:
        path = parts[: parts.index("-")] if "-" in parts else parts[:2]
        if len(path) >= 2:
            last = path[-1][:-4] if path[-1].endswith(".git") else path[-1]
            keys.add(f"repo:{host}/{'/'.join(path[:-1] + [last])}")
            if host == "gitlab.gnome.org":
                keys.add(f"proj:gnome/{last}")
            elif host == "invent.kde.org":
                keys.add(f"proj:kde/{last}")
    elif host in ("codeberg.org", "bitbucket.org", "gitea.com", "notabug.org"):
        if len(parts) >= 2:
            repo(host, parts[0], parts[1])
    elif host == "git.sr.ht":
        if len(parts) >= 2:
            repo(host, parts[0] if parts[0].startswith("~") else "~" + parts[0], parts[1])
    elif host in ("sourceforge.net", "sf.net"):
        if len(parts) >= 2 and parts[0] in ("projects", "p", "project"):
            keys.add(f"proj:sourceforge/{parts[1]}")
    elif host in ("downloads.sourceforge.net", "download.sourceforge.net",
                  "prdownloads.sourceforge.net") or host.endswith(".dl.sourceforge.net"):
        p = parts[1:] if parts and parts[0] in ("project", "projects", "sourceforge") else parts
        if p:
            keys.add(f"proj:sourceforge/{p[0]}")
    elif host.endswith((".sourceforge.net", ".sourceforge.io", ".sf.net")):
        keys.add(f"proj:sourceforge/{host.split('.')[0]}")
    elif host in ("pypi.org", "pypi.python.org"):
        if len(parts) >= 2 and parts[0] in ("project", "pypi", "simple"):
            keys.add(f"proj:pypi/{_pep503(parts[1])}")
    elif host in ("files.pythonhosted.org", "pypi.io"):
        if len(parts) >= 4 and parts[:2] == ["packages", "source"]:
            keys.add(f"proj:pypi/{_pep503(parts[3])}")
    elif host == "rubygems.org":
        if len(parts) >= 2 and parts[0] == "gems":
            keys.add(f"proj:rubygems/{parts[1]}")
    elif host in ("npmjs.com", "registry.npmjs.org"):
        p = parts[1:] if parts and parts[0] == "package" else parts
        if p:
            keys.add(f"proj:npm/{'/'.join(p[:2]) if p[0].startswith('@') and len(p) > 1 else p[0]}")
    elif host == "crates.io":
        if len(parts) >= 2 and parts[0] == "crates":
            keys.add(f"proj:crates/{parts[1]}")
    elif host in ("metacpan.org", "search.cpan.org"):
        if len(parts) >= 2 and parts[0] in ("dist", "release", "pod"):
            keys.add(f"proj:cpan/{parts[1].replace('::', '-')}")
    elif host == "hackage.haskell.org":
        if len(parts) >= 2 and parts[0] == "package":
            keys.add(f"proj:hackage/{re.sub(r'-[0-9][0-9.]*$', '', parts[1])}")
    elif host in ("launchpad.net", "code.launchpad.net"):
        if parts and not parts[0].startswith(("~", "+")):
            keys.add(f"proj:launchpad/{parts[0]}")
    elif host in ("savannah.gnu.org", "savannah.nongnu.org"):
        if len(parts) >= 2 and parts[0] == "projects":
            keys.add(f"proj:gnu/{parts[1]}")
    elif host in ("download.savannah.gnu.org", "download.savannah.nongnu.org"):
        if len(parts) >= 2 and parts[0] == "releases":
            keys.add(f"proj:gnu/{parts[1]}")
    elif host == "ftpmirror.gnu.org":
        if parts:
            keys.add(f"proj:gnu/{parts[0]}")
    elif host in ("archive.apache.org", "downloads.apache.org", "dlcdn.apache.org",
                  "dist.apache.org", "apache.org"):
        p = parts[1:] if parts and parts[0] == "dist" else parts
        if p and host != "apache.org":
            keys.add(f"host:{p[0]}.apache.org")
    elif host == "download.gnome.org":
        if len(parts) >= 2 and parts[0] == "sources":
            keys.add(f"proj:gnome/{parts[1]}")
    elif host == "download.kde.org":
        if len(parts) >= 2 and parts[0] in ("stable", "unstable"):
            keys.add(f"proj:kde/{parts[1]}")

    # GNU projects on any mirror: .../gnu/<project>/...
    if "gnu" in parts[:3]:
        i = parts.index("gnu")
        if i + 1 < len(parts) and host != "gnu.org":
            keys.add(f"proj:gnu/{parts[i + 1]}")
    if host in ("gnu.org",) and len(parts) >= 2 and parts[0] == "software":
        keys.add(f"proj:gnu/{parts[1]}")

    if host not in GENERIC_HOSTS and not host.endswith((".sourceforge.net", ".sourceforge.io",
                                                         ".dl.sourceforge.net", ".sf.net")):
        keys.add(f"host:{host}")
    return keys


def is_strong(key: str) -> bool:
    return not key.startswith("host:")


# ---------------------------------------------------------------------------
# JSON document reader for feeds (files, gzip, zip, tar, directories)
# ---------------------------------------------------------------------------
def _is_json_name(name: str) -> bool:
    return name.lower().endswith((".json", ".json.gz"))


def _loads(data: bytes, name: str):
    if name.lower().endswith(".gz"):
        data = gzip.decompress(data)
    return json.loads(data)


def iter_json_documents(path: Path) -> Iterator[tuple[str, object]]:
    if path.is_dir():
        for p in sorted(path.rglob("*")):
            if p.is_file() and (_is_json_name(p.name) or p.suffix in (".zip",)):
                yield from iter_json_documents(p)
        return
    name = path.name.lower()
    try:
        if name.endswith(".zip"):
            with zipfile.ZipFile(path) as z:
                for info in z.infolist():
                    if _is_json_name(info.filename):
                        yield info.filename, _loads(z.read(info), info.filename)
        elif re.search(r"\.(tar|tar\.gz|tgz|tar\.xz|tar\.bz2)$", name):
            with tarfile.open(path) as t:
                for member in t:
                    if member.isfile() and _is_json_name(member.name):
                        fh = t.extractfile(member)
                        if fh:
                            yield member.name, _loads(fh.read(), member.name)
        elif _is_json_name(name):
            yield str(path), _loads(path.read_bytes(), name)
        else:
            log(f"warning: don't know how to read {path}")
    except (OSError, ValueError, zipfile.BadZipFile, tarfile.TarError) as exc:
        log(f"warning: could not read {path}: {exc}")


# ---------------------------------------------------------------------------
# Ports tree extraction
# ---------------------------------------------------------------------------
PORT_VARS = [
    "PKGNAME", "PORTNAME", "PORTVERSION", "DISTVERSION", "PKGNAMEPREFIX", "PKGNAMESUFFIX",
    "CATEGORIES", "MAINTAINER", "WWW", "MASTER_SITES", "USES",
    "USE_GITHUB", "GH_ACCOUNT", "GH_PROJECT", "GH_TAGNAME",
    "USE_GITLAB", "GL_SITE", "GL_ACCOUNT", "GL_PROJECT",
    "CPE_STR", "CPE_PART", "CPE_VENDOR", "CPE_PRODUCT", "CPE_VERSION", "CPE_UPDATE",
    "DEPRECATED",
]
SPECIAL_TOP_DIRS = {"Mk", "Templates", "Tools", "Keywords", "distfiles", "packages", ".git", ".hooks"}
SUBDIR_RE = re.compile(r"^\s*SUBDIR\s*\+?=\s*(.+?)\s*$", re.M)
EXPLICIT_CPE_RE = re.compile(r"^\s*CPE_(VENDOR|PRODUCT|VERSION|UPDATE|STR|PART)\s*[?:!+]?=", re.M)


def _subdirs(makefile: Path) -> list[str]:
    try:
        text = makefile.read_text(errors="replace")
    except OSError:
        return []
    out: list[str] = []
    for m in SUBDIR_RE.finditer(text):
        out += [s for s in m.group(1).split() if not s.startswith("#")]
    return out


def discover_origins(tree: Path, categories: list[str] | None = None) -> list[str]:
    cats = categories or _subdirs(tree / "Makefile")
    if not cats:
        cats = sorted(p.name for p in tree.iterdir()
                      if p.is_dir() and p.name not in SPECIAL_TOP_DIRS and (p / "Makefile").exists())
    origins = []
    for cat in cats:
        ports = _subdirs(tree / cat / "Makefile")
        if not ports and (tree / cat).is_dir():
            ports = sorted(p.name for p in (tree / cat).iterdir() if (p / "Makefile").exists())
        origins += [f"{cat}/{p}" for p in ports if (tree / cat / p / "Makefile").exists()]
    return origins


def extract_port(tree: Path, origin: str, make: str, env: dict, timeout: int) -> dict:
    portdir = tree / origin
    cmd = [make, "-C", str(portdir)]
    for var in PORT_VARS:
        cmd += ["-V", var]
    try:
        proc = subprocess.run(cmd, env=env, capture_output=True, text=True, timeout=timeout)
    except subprocess.TimeoutExpired:
        return {"origin": origin, "error": "timeout"}
    except OSError as exc:
        return {"origin": origin, "error": str(exc)}
    lines = proc.stdout.split("\n")
    if lines and lines[-1] == "":
        lines.pop()
    if proc.returncode != 0 or len(lines) != len(PORT_VARS):
        detail = (proc.stderr.strip().splitlines() or [f"exit {proc.returncode}, {len(lines)} lines"])[-1]
        return {"origin": origin, "error": detail[:300]}
    rec = {"origin": origin}
    rec.update({var.lower(): val.strip() for var, val in zip(PORT_VARS, lines)})
    try:
        makefile = (portdir / "Makefile").read_text(errors="replace")
        rec["cpe_explicit"] = sorted({m.group(1) for m in EXPLICIT_CPE_RE.finditer(makefile)})
    except OSError:
        rec["cpe_explicit"] = []
    rec["uses_cpe"] = any(u.split(":")[0] == "cpe" for u in rec.get("uses", "").split())
    return rec


def cmd_extract(args: argparse.Namespace) -> int:
    tree = Path(args.ports_tree).resolve()
    if not (tree / "Mk" / "bsd.port.mk").exists():
        log(f"error: {tree} does not look like a ports tree (no Mk/bsd.port.mk)")
        return 2
    if not sys.platform.startswith("freebsd") and args.make == "make":
        log("warning: not running on FreeBSD; `make` is probably not bmake and will fail")
    origins = args.origin or discover_origins(tree, args.category)
    final = args.cache_dir / PORTS_FILE
    partial = final.with_name(final.name + ".part")
    final.parent.mkdir(parents=True, exist_ok=True)
    if args.resume:
        out, mode = (partial if partial.exists() else final), "a"
    else:
        out, mode = partial, "w"  # the previous ports.jsonl stays usable until we finish
    done: set[str] = set()
    if args.resume and out.exists():
        for rec in read_jsonl(out):
            if not rec.get("error"):
                done.add(rec["origin"])
    todo = [o for o in origins if o not in done]
    log(f"{len(origins)} ports found, {len(done)} already extracted, {len(todo)} to go")

    env = dict(os.environ, PORTSDIR=str(tree), BATCH="yes")
    if not args.use_make_conf:
        env["__MAKE_CONF"] = "/dev/null"  # keep local OPTIONS/overrides out of the data

    started, errors = time.time(), 0
    pool = ThreadPoolExecutor(max_workers=args.jobs)
    futures = [pool.submit(extract_port, tree, o, args.make, env, args.timeout) for o in todo]
    try:
        with out.open(mode) as fh:
            for n, fut in enumerate(as_completed(futures), 1):
                rec = fut.result()
                errors += bool(rec.get("error"))
                fh.write(json.dumps(rec, sort_keys=True) + "\n")
                if n % 500 == 0 or n == len(futures):
                    rate = n / max(time.time() - started, 1e-6)
                    log(f"  {n}/{len(futures)} ports, {errors} errors, {rate:.1f} ports/s")
    except KeyboardInterrupt:
        for fut in futures:
            fut.cancel()
        log("interrupted; continue later with: extract --resume")
        sys.stderr.flush()
        os._exit(130)  # records are flushed and the file closed; don't wait for running make jobs
    pool.shutdown()
    if out == partial:
        os.replace(partial, final)
    log(f"wrote {final}")
    return 0


def read_jsonl(path: Path) -> Iterator[dict]:
    with Path(path).open() as fh:
        for line in fh:
            line = line.strip()
            if line:
                yield json.loads(line)


# ---------------------------------------------------------------------------
# NVD CPE dictionary
# ---------------------------------------------------------------------------
@dataclass
class ProductInfo:
    title: str = ""
    versions: set = field(default_factory=set)
    updates: set = field(default_factory=set)
    n_entries: int = 0
    n_deprecated: int = 0
    deprecated_by: set = field(default_factory=set)
    ref_keys: set = field(default_factory=set)


class NvdCpeDictionary:
    def __init__(self) -> None:
        self.products: dict[str, ProductInfo] = {}
        self.ref_index: dict[str, set[str]] = defaultdict(set)

    def load(self, path: Path) -> None:
        count = 0
        for _, doc in iter_json_documents(path):
            items = doc.get("products") if isinstance(doc, dict) else None
            for item in items or []:
                cpe = item.get("cpe", item) if isinstance(item, dict) else {}
                c = parse_cpe(cpe.get("cpeName", ""))
                if c is None or c.part != "a":
                    continue
                info = self.products.get(c.vp)
                if info is None:
                    info = self.products[c.vp] = ProductInfo()
                info.n_entries += 1
                count += 1
                if c.version:
                    info.versions.add(c.version)
                if c.update:
                    info.updates.add(c.update)
                if cpe.get("deprecated"):
                    info.n_deprecated += 1
                    for dep in cpe.get("deprecatedBy") or []:
                        d = parse_cpe(dep.get("cpeName", "")) if isinstance(dep, dict) else None
                        if d and d.vp != c.vp:
                            info.deprecated_by.add(d.vp)
                if not info.title:
                    for t in cpe.get("titles") or []:
                        if str(t.get("lang", "en")).startswith("en"):
                            info.title = t.get("title", "")
                            break
                for ref in cpe.get("refs") or []:
                    for key in url_keys(ref.get("ref", "")):
                        if key not in info.ref_keys:
                            info.ref_keys.add(key)
                            self.ref_index[key].add(c.vp)
        log(f"NVD CPE dictionary {path}: {count} entries, {len(self.products)} products so far")


# ---------------------------------------------------------------------------
# CVE data (NVD CVE feeds and CVEList v5)
# ---------------------------------------------------------------------------
class CveData:
    def __init__(self) -> None:
        self.products: dict[str, set[str]] = {}
        self.versions: dict[str, set[str]] = defaultdict(set)
        self.ref_index: dict[str, dict[str, set[str]]] = defaultdict(lambda: defaultdict(set))
        self.records = 0

    def _add(self, cve_id: str, cpes: list[Cpe], ref_urls: list[str]) -> None:
        vps = set()
        for c in cpes:
            if c.part != "a":
                continue
            vps.add(c.vp)
            if c.version:
                self.versions[c.vp].add(c.version)
        if not vps:
            return
        self.records += 1
        for vp in vps:
            self.products.setdefault(vp, set()).add(cve_id)
        keys = {k for u in ref_urls for k in url_keys(u) if is_strong(k)}
        for key in keys:
            for vp in vps:
                self.ref_index[key][vp].add(cve_id)

    def load_nvd(self, path: Path) -> None:
        before = self.records
        for _, doc in iter_json_documents(path):
            if not isinstance(doc, dict):
                continue
            for item in doc.get("vulnerabilities") or []:
                cve = item.get("cve") or {}
                cve_id = cve.get("id")
                if not cve_id or str(cve.get("vulnStatus", "")).lower() == "rejected":
                    continue
                cpes = []
                for conf in cve.get("configurations") or []:
                    for node in conf.get("nodes") or []:
                        for match in node.get("cpeMatch") or []:
                            if match.get("vulnerable") is False:
                                continue  # platform the vulnerable software runs on
                            c = parse_cpe(match.get("criteria", ""))
                            if c:
                                cpes.append(c)
                self._add(cve_id, cpes, [r.get("url", "") for r in cve.get("references") or []])
        log(f"NVD CVE feed {path}: {self.records - before} CVEs with application CPEs")

    def load_cvelist(self, root: Path) -> None:
        before = self.records
        for dirpath, _, files in os.walk(root):
            for fn in files:
                if not (fn.startswith("CVE-") and fn.endswith(".json")):
                    continue
                try:
                    rec = json.loads(Path(dirpath, fn).read_bytes())
                except (OSError, ValueError):
                    continue
                meta = rec.get("cveMetadata") or {}
                if meta.get("state") == "REJECTED":
                    continue
                containers = rec.get("containers") or {}
                found: set[str] = set()
                _collect_cpe_strings(containers, found)
                refs = []
                for cont in [containers.get("cna") or {}] + list(containers.get("adp") or []):
                    refs += [r.get("url", "") for r in cont.get("references") or []]
                self._add(meta.get("cveId") or fn[:-5], [c for c in map(parse_cpe, found) if c], refs)
        log(f"CVEList v5 {root}: {self.records - before} CVEs with application CPEs")


def _collect_cpe_strings(obj, out: set[str]) -> None:
    if isinstance(obj, str):
        if obj.startswith(("cpe:2.3:", "cpe:/")):
            out.add(obj)
    elif isinstance(obj, dict):
        for value in obj.values():
            _collect_cpe_strings(value, out)
    elif isinstance(obj, list):
        for value in obj:
            _collect_cpe_strings(value, out)


# ---------------------------------------------------------------------------
# Other distributions' CPE mappings
# ---------------------------------------------------------------------------
@dataclass
class DistroEntry:
    source: str                      # gentoo, buildroot, openembedded, nixpkgs, osv-cpe-repos
    pkg_id: str                      # native identifier, e.g. dev-libs/expat
    name: str                        # package name for name joins
    cpes: list                       # [(vendor or None, product)]
    url_keys: set = field(default_factory=set)


class DistroIndex:
    def __init__(self) -> None:
        self.by_url: dict[str, list[DistroEntry]] = defaultdict(list)
        self.by_name: dict[str, list[DistroEntry]] = defaultdict(list)
        self.by_pkg: dict[tuple[str, str], list[DistroEntry]] = defaultdict(list)
        self.counts: Counter = Counter()

    def add(self, entry: DistroEntry) -> None:
        if not entry.cpes:
            return
        self.counts[entry.source] += 1
        for key in entry.url_keys:
            self.by_url[key].append(entry)
        if norm(entry.name):
            self.by_name[norm(entry.name)].append(entry)
        ids = {entry.pkg_id.lower(), entry.pkg_id.lower().rsplit("/", 1)[-1], entry.name.lower()}
        for ident in ids:
            self.by_pkg[(entry.source, ident)].append(entry)


def _dedupe(items: list) -> list:
    return list(dict.fromkeys(items))


GENTOO_REMOTE_URL = {
    "github": "https://github.com/{}", "gitlab": "https://gitlab.com/{}",
    "codeberg": "https://codeberg.org/{}", "bitbucket": "https://bitbucket.org/{}",
    "sourcehut": "https://git.sr.ht/{}", "sourceforge": "https://sourceforge.net/projects/{}",
    "pypi": "https://pypi.org/project/{}", "rubygems": "https://rubygems.org/gems/{}",
    "cpan": "https://metacpan.org/dist/{}", "crates-io": "https://crates.io/crates/{}",
    "npm": "https://www.npmjs.com/package/{}", "hackage": "https://hackage.haskell.org/package/{}",
    "launchpad": "https://launchpad.net/{}", "savannah": "https://savannah.gnu.org/projects/{}",
    "savannah-nongnu": "https://savannah.nongnu.org/projects/{}",
    "freedesktop-gitlab": "https://gitlab.freedesktop.org/{}",
    "gnome-gitlab": "https://gitlab.gnome.org/{}", "kde-invent": "https://invent.kde.org/{}",
}


def load_gentoo(root: Path) -> Iterator[DistroEntry]:
    for md in sorted(root.glob("*/*/metadata.xml")):
        try:
            tree = ET.parse(md)
        except (ET.ParseError, OSError):
            continue
        cpes, keys = [], set()
        for rid in tree.iter("remote-id"):
            kind, value = rid.get("type", ""), (rid.text or "").strip()
            if not value:
                continue
            if kind == "cpe":
                c = parse_cpe(value)
                if c:
                    cpes.append((c.vendor, c.product))
            elif kind in GENTOO_REMOTE_URL:
                keys |= url_keys(GENTOO_REMOTE_URL[kind].format(value))
        if cpes:
            cat, pkg = md.parent.parent.name, md.parent.name
            yield DistroEntry("gentoo", f"{cat}/{pkg}", pkg, _dedupe(cpes), keys)


BR_ASSIGN = re.compile(r"^\s*([A-Z0-9_]+)\s*[:?+]?=\s*(.*?)\s*$", re.M)


def load_buildroot(root: Path) -> Iterator[DistroEntry]:
    pkgdir = root / "package" if (root / "package").is_dir() else root
    for mk in sorted(pkgdir.rglob("*.mk")):
        name = mk.parent.name
        if mk.stem != name:
            continue
        prefix = re.sub(r"[^A-Z0-9]", "_", name.upper()) + "_"
        try:
            text = re.sub(r"\\\n[ \t]*", " ", mk.read_text(errors="replace"))  # join continuation lines
        except OSError:
            continue
        values: dict[str, str] = {}
        for var, val in BR_ASSIGN.findall(text):
            if var.startswith(prefix):
                values.setdefault(var[len(prefix):], val)
        cpe_vars = {k: v for k, v in values.items() if k.startswith("CPE_ID_")}
        if not cpe_vars:
            continue
        if set(cpe_vars) == {"CPE_ID_VALID"} and cpe_vars["CPE_ID_VALID"].upper() != "YES":
            continue
        # Buildroot defaults once any CPE_ID variable is set (package/pkg-generic.mk)
        vendor = cpe_vars.get("CPE_ID_VENDOR") or f"{name}_project"
        product = cpe_vars.get("CPE_ID_PRODUCT") or name
        if "$" in vendor or "$" in product:
            continue
        # make turns "\\\\+" into "\\+"; the result is CPE formatted-string escaping
        vendor, product = (_unescape(x.replace("\\\\", "\\")) for x in (vendor, product))
        keys: set[str] = set()
        site = values.get("SITE", "")
        m = re.search(r"\$\(call\s+(github|gitlab)\s*,\s*([^,\s)]+)\s*,\s*([^,\s)]+)", site)
        if m:
            keys |= url_keys(f"https://{m.group(1)}.com/{m.group(2)}/{m.group(3)}")
        elif site:
            keys |= url_keys(site.split("$(")[0])
        yield DistroEntry("buildroot", name, name, [(vendor, product)], keys)


OE_CVE = re.compile(
    r'^\s*CVE_PRODUCT(?::append|:prepend)?\s*(\?\?=|\?=|\+=|=\+|\.=|=\.|:=|=)\s*"([^"]*)"', re.M)
OE_URL = re.compile(r'(?:https?|git|gitsm)://[^\s;"\\]+')


def load_openembedded(root: Path) -> Iterator[DistroEntry]:
    for path in sorted(root.rglob("*")):
        if path.suffix not in (".bb", ".inc", ".bbappend") or ".git" in path.parts or not path.is_file():
            continue
        if "/classes" in path.as_posix():
            continue  # class-level defaults such as CVE_PRODUCT ??= "${BPN}"
        try:
            text = path.read_text(errors="replace")
        except OSError:
            continue
        if "CVE_PRODUCT" not in text:
            continue
        recipe = re.split(r"[_%]", path.stem, maxsplit=1)[0]
        cpes = []
        for op, value in OE_CVE.findall(text):
            if op == "??=":
                continue
            value = value.replace("${BPN}", recipe).replace("${PN}", recipe)
            for item in value.split():
                if "$" in item:
                    continue
                if ":" in item:
                    vendor, product = item.split(":", 1)
                    cpes.append((vendor.lower(), product.lower()))
                else:
                    cpes.append((None, item.lower()))
        if not cpes:
            continue
        keys: set[str] = set()
        for line in text.splitlines():
            if not line.lstrip().startswith("#"):
                for u in OE_URL.findall(line):
                    keys |= url_keys(u)
        yield DistroEntry("openembedded", path.relative_to(root).as_posix(), recipe, _dedupe(cpes), keys)


def load_nixpkgs_json(path: Path) -> Iterator[DistroEntry]:
    for _, doc in iter_json_documents(path):
        if not isinstance(doc, dict):
            continue
        packages = doc.get("packages", doc)
        for attr, pkg in packages.items():
            if not isinstance(pkg, dict):
                continue
            meta = pkg.get("meta") or {}
            ident = meta.get("identifiers") or {}
            parts = ident.get("cpeParts") or {}
            product = (parts.get("product") or pkg.get("pname") or "").lower()
            vendor = (parts.get("vendor") or "").lower()
            cpes = []
            if vendor and vendor != "*" and product:
                cpes.append((vendor, product))  # nixpkgs has no default vendor: set by a maintainer
            elif ident.get("cpe"):
                c = parse_cpe(ident["cpe"])
                if c:
                    cpes.append((c.vendor, c.product))
            if not cpes:
                continue
            keys: set[str] = set()
            homepage = meta.get("homepage")
            for u in homepage if isinstance(homepage, list) else [homepage]:
                keys |= url_keys(u or "")
            name = pkg.get("pname") or re.sub(r"-\d.*$", "", str(pkg.get("name", attr)))
            yield DistroEntry("nixpkgs", attr, name, cpes, keys)


def load_osv_cpe_repos(path: Path) -> Iterator[DistroEntry]:
    for _, doc in iter_json_documents(path):
        if not isinstance(doc, dict):
            continue
        for key, value in doc.items():
            urls = value if isinstance(value, list) else (
                [u for v in value.values() for u in (v if isinstance(v, list) else [v])]
                if isinstance(value, dict) else [value])
            keys = {k for u in urls if isinstance(u, str) for k in url_keys(u) if is_strong(k)}
            if not keys:
                continue
            vendor, product = key.lower().split(":", 1) if ":" in key else (None, key.lower())
            yield DistroEntry("osv-cpe-repos", key, product, [(vendor, product)], keys)


# ---------------------------------------------------------------------------
# Repology
# ---------------------------------------------------------------------------
def repology_family(repo: str) -> str | None:
    repo = repo.lower()
    if repo.startswith("gentoo"):
        return "gentoo"
    if repo.startswith("buildroot"):
        return "buildroot"
    if repo.startswith("nix"):
        return "nixpkgs"
    if repo.startswith(("openembedded", "yocto")):
        return "openembedded"
    return None


class RepologyIndex:
    def __init__(self, data: dict) -> None:
        self.origin_to_project: dict[str, str] = {}
        self.project_ids: dict[str, set[tuple[str, str]]] = defaultdict(set)
        for project, packages in (data.get("projects") or {}).items():
            for pkg in packages:
                repo = pkg.get("repo", "")
                if repo == data.get("repo", "freebsd") and pkg.get("srcname"):
                    self.origin_to_project[pkg["srcname"]] = project
                family = repology_family(repo)
                if family:
                    for fld in ("srcname", "binname", "visiblename"):
                        value = str(pkg.get(fld) or "").lower()
                        if value:
                            self.project_ids[project].add((family, value))
                            self.project_ids[project].add((family, value.rsplit("/", 1)[-1]))

    @classmethod
    def load(cls, path: Path) -> "RepologyIndex":
        with Path(path).open() as fh:
            index = cls(json.load(fh))
        log(f"Repology {path}: {len(index.origin_to_project)} FreeBSD origins mapped to projects")
        return index

    def lookup(self, origin: str) -> tuple[str | None, set[tuple[str, str]]]:
        project = self.origin_to_project.get(origin)
        return project, (self.project_ids.get(project, set()) if project else set())


# ---------------------------------------------------------------------------
# Cache directory and data sources
# ---------------------------------------------------------------------------
APP_NAME = "ports-cpe-matcher"
APP_VERSION = "0.2"
DEFAULT_USER_AGENT = f"{APP_NAME}/{APP_VERSION} (Python-urllib)"
PORTS_FILE = "ports.jsonl"
REPORT_DIR = "report"
STATE_FILE = "state.json"
STALE_HOURS = 7 * 24
NVD_FIRST_YEAR = 2002
NVD_DELAY = 1.0          # seconds between NVD feed downloads
REPOLOGY_DELAY = 1.1     # Repology allows at most one request per second


def default_cache_dir() -> Path:
    if os.environ.get("PORTS_CPE_MATCHER_CACHE"):
        return Path(os.environ["PORTS_CPE_MATCHER_CACHE"]).expanduser()
    return Path(os.environ.get("XDG_CACHE_HOME") or "~/.cache").expanduser() / APP_NAME


@dataclass(frozen=True)
class Source:
    name: str
    kind: str              # git, http, nvd-cve, repology, manual
    path: str              # relative to the cache directory
    description: str
    url: str = ""
    sparse: tuple = ()     # sparse-checkout patterns (git)
    partial: bool = True   # clone without blobs, fetch only what the checkout needs (git)


_RECIPES = ("*.bb", "*.bbappend", "*.inc")
SOURCES = [
    Source("nvd-cpe", "http", "nvd/nvdcpe-2.0.zip", "NVD CPE Dictionary 2.0 feed",
           "https://nvd.nist.gov/feeds/json/cpe/2.0/nvdcpe-2.0.zip"),
    Source("nvd-cve", "nvd-cve", "nvd/cve", "NVD CVE 2.0 yearly feeds",
           "https://nvd.nist.gov/feeds/json/cve/2.0/"),
    Source("cvelist", "git", "git/cvelistV5", "CVEList v5 with CNA/ADP CPEs (largest download)",
           "https://github.com/CVEProject/cvelistV5", partial=False),
    Source("gentoo", "git", "git/gentoo", "Gentoo repository (metadata.xml files only)",
           "https://github.com/gentoo/gentoo", sparse=("/*/*/metadata.xml",)),
    Source("buildroot", "git", "git/buildroot", "Buildroot package makefiles",
           "https://github.com/buildroot/buildroot", sparse=("/package/**/*.mk",)),
    Source("openembedded-core", "git", "git/openembedded-core", "OpenEmbedded-Core recipes",
           "https://github.com/openembedded/openembedded-core", sparse=_RECIPES),
    Source("meta-openembedded", "git", "git/meta-openembedded", "meta-openembedded recipes",
           "https://github.com/openembedded/meta-openembedded", sparse=_RECIPES),
    Source("osv-cpe-repos", "http", "osv/cpe_product_to_repo.json",
           "osv.dev CPE-to-repository map (undocumented, may be unavailable)",
           "https://storage.googleapis.com/osv-test-cve-osv-conversion/cpe_repos/cpe_product_to_repo.json"),
    Source("repology", "repology", "repology/repology.json", "Repology projects packaged in FreeBSD",
           "https://repology.org/api/v1/projects/"),
    Source("nixpkgs", "manual", "nixpkgs/packages.json", "nixpkgs metadata JSON (not fetched, see --help)"),
]
SOURCE_BY_NAME = {s.name: s for s in SOURCES}
FETCH_GROUPS = {
    "all": [s.name for s in SOURCES if s.kind != "manual"],
    "git": [s.name for s in SOURCES if s.kind == "git"],
    "data": [s.name for s in SOURCES if s.kind in ("http", "nvd-cve", "repology")],
}


class FetchError(Exception):
    pass


class State:
    """Fetch bookkeeping: timestamps, HTTP validators and NVD .meta contents."""

    def __init__(self, cache: Path) -> None:
        self.path = cache / STATE_FILE
        try:
            self.data = json.loads(self.path.read_text())
        except (OSError, ValueError):
            self.data = {}

    def source(self, name: str) -> dict:
        return self.data.setdefault(name, {})

    def save(self) -> None:
        self.path.parent.mkdir(parents=True, exist_ok=True)
        tmp = self.path.with_name(self.path.name + ".tmp")
        tmp.write_text(json.dumps(self.data, indent=1, sort_keys=True))
        os.replace(tmp, self.path)


def _now() -> str:
    return datetime.now(timezone.utc).isoformat(timespec="seconds")


def _hours_since(iso: str | None) -> float | None:
    try:
        return (datetime.now(timezone.utc) - datetime.fromisoformat(iso)).total_seconds() / 3600
    except (TypeError, ValueError):
        return None


def _fmt_age(hours: float | None) -> str:
    if hours is None:
        return "never"
    if hours < 1:
        return f"{hours * 60:.0f} min ago"
    return f"{hours:.1f} h ago" if hours < 48 else f"{hours / 24:.0f} days ago"


def source_present(cache: Path, source: Source) -> bool:
    path = cache / source.path
    if source.kind == "git":
        return (path / ".git").is_dir()
    if path.is_dir():
        return any(path.iterdir())
    return path.is_file() and path.stat().st_size > 0


def source_age_hours(cache: Path, state: State, source: Source) -> float | None:
    hours = _hours_since(state.data.get(source.name, {}).get("checked"))
    if hours is None and source_present(cache, source):
        hours = (time.time() - (cache / source.path).stat().st_mtime) / 3600
    return hours


# ---------------------------------------------------------------------------
# fetch: HTTP helpers
# ---------------------------------------------------------------------------
def http_open(url: str, user_agent: str, headers: dict | None = None, timeout: int = 300):
    """urlopen with retries for rate limiting and transient server/network errors."""
    request = urllib.request.Request(url, headers={"User-Agent": user_agent, **(headers or {})})
    wait = 5
    for attempt in range(1, 5):
        try:
            return urllib.request.urlopen(request, timeout=timeout)
        except urllib.error.HTTPError as exc:
            if exc.code not in (429, 500, 502, 503, 504) or attempt == 4:
                raise
            retry_after = str((exc.headers or {}).get("Retry-After", ""))
            delay = int(retry_after) if retry_after.isdigit() else wait
        except (urllib.error.URLError, TimeoutError, ConnectionError):
            if attempt == 4:
                raise
            delay = wait
        log(f"    temporary failure for {url}, retrying in {delay}s")
        time.sleep(delay)
        wait *= 2
    raise FetchError(f"giving up on {url}")  # not reached


def http_download(url: str, dest: Path, user_agent: str, validate=None,
                  validators: dict | None = None) -> tuple[bool, dict]:
    """Download atomically. With validators (ETag/Last-Modified from an earlier
    download) the request is conditional. Returns (changed, validators)."""
    headers = {}
    if validators and dest.exists():
        if validators.get("etag"):
            headers["If-None-Match"] = validators["etag"]
        if validators.get("last_modified"):
            headers["If-Modified-Since"] = validators["last_modified"]
    try:
        response = http_open(url, user_agent, headers)
    except urllib.error.HTTPError as exc:
        if exc.code == 304:
            return False, validators or {}
        raise
    dest.parent.mkdir(parents=True, exist_ok=True)
    tmp = dest.with_name(dest.name + ".part")
    try:
        with response, tmp.open("wb") as fh:
            shutil.copyfileobj(response, fh, 1 << 20)
            new_validators = {"etag": response.headers.get("ETag"),
                              "last_modified": response.headers.get("Last-Modified")}
        if validate:
            validate(tmp)
    except BaseException:
        tmp.unlink(missing_ok=True)
        raise
    os.replace(tmp, dest)
    return True, new_validators


def _validate_zip(path: Path) -> None:
    try:
        with zipfile.ZipFile(path) as z:
            if not any(_is_json_name(name) for name in z.namelist()):
                raise FetchError("archive contains no JSON files")
    except zipfile.BadZipFile as exc:
        raise FetchError(f"not a valid zip archive: {exc}") from exc


def _validate_json(path: Path) -> None:
    try:
        with path.open("rb") as fh:
            json.load(fh)
    except ValueError as exc:
        raise FetchError(f"not valid JSON: {exc}") from exc


def _nvd_gz_validator(expected_sha256: str):
    def check(path: Path) -> None:
        raw, plain = hashlib.sha256(), hashlib.sha256()
        with path.open("rb") as fh:
            for block in iter(lambda: fh.read(1 << 20), b""):
                raw.update(block)
        try:
            with gzip.open(path, "rb") as fh:
                for block in iter(lambda: fh.read(1 << 20), b""):
                    plain.update(block)
        except (OSError, EOFError, zlib.error) as exc:
            raise FetchError(f"corrupt gzip data: {exc}") from exc
        # .meta checksums have described the uncompressed JSON; accept either form
        if expected_sha256 and expected_sha256 not in (raw.hexdigest(), plain.hexdigest()):
            raise FetchError("SHA-256 does not match the .meta file")
    return check


# ---------------------------------------------------------------------------
# fetch: individual source kinds
# ---------------------------------------------------------------------------
def fetch_http(source: Source, cache: Path, state: State, user_agent: str, force: bool) -> tuple[str, bool]:
    st = state.source(source.name)
    dest = cache / source.path
    validate = _validate_zip if dest.suffix == ".zip" else _validate_json
    try:
        changed, validators = http_download(source.url, dest, user_agent, validate,
                                            None if force else st.get("http"))
    except urllib.error.HTTPError as exc:
        if exc.code in (401, 403, 404, 410):
            raise FetchError(f"{source.url} answered HTTP {exc.code}") from exc
        raise
    st["http"] = validators
    size = dest.stat().st_size / 1e6
    return (f"downloaded {size:.1f} MB" if changed else f"not modified ({size:.1f} MB cached)"), changed


def fetch_nvd_cve(source: Source, cache: Path, state: State, user_agent: str, force: bool) -> tuple[str, bool]:
    """Yearly feeds; a feed is only downloaded when its .meta file changed."""
    metas = state.source(source.name).setdefault("meta", {})
    outdir = cache / source.path
    outdir.mkdir(parents=True, exist_ok=True)
    years = list(range(NVD_FIRST_YEAR, datetime.now(timezone.utc).year + 1))
    downloaded, unchanged, unpublished, failed = [], 0, [], []
    for year in years:
        name = f"nvdcve-2.0-{year}"
        dest = outdir / f"{name}.json.gz"
        try:
            with http_open(f"{source.url}{name}.meta", user_agent, timeout=60) as resp:
                meta = resp.read().decode("utf-8", "replace").strip()
        except urllib.error.HTTPError as exc:
            if exc.code == 404:
                unpublished.append(year)
                continue
            raise FetchError(f"{source.url}{name}.meta answered HTTP {exc.code}") from exc
        if not force and dest.exists() and metas.get(name) == meta:
            unchanged += 1
            continue
        fields = dict(line.split(":", 1) for line in meta.splitlines() if ":" in line)
        log(f"    {name}.json.gz")
        try:
            http_download(f"{source.url}{name}.json.gz", dest, user_agent,
                          _nvd_gz_validator(fields.get("sha256", "").strip().lower()))
        except (FetchError, OSError) as exc:  # keep going; the old file (if any) stays in place
            log(f"    {name}: {exc}")
            failed.append(year)
            continue
        metas[name] = meta
        downloaded.append(year)
        state.save()  # progress survives an interruption
        time.sleep(NVD_DELAY)
    if len(unpublished) == len(years):
        raise FetchError(f"no feed found below {source.url}; has the location changed?")
    detail = f"{len(downloaded)} yearly feed(s) downloaded, {unchanged} unchanged"
    if unpublished:
        detail += f", not published: {', '.join(map(str, unpublished))}"
    if failed:
        raise FetchError(f"{detail}; failed: {', '.join(map(str, failed))} (retried on the next fetch)")
    return detail, bool(downloaded)


def fetch_repology(source: Source, cache: Path, user_agent: str | None) -> tuple[str, bool]:
    if not user_agent or "http" not in user_agent:
        raise FetchError("Repology requires a User-Agent linking to your source repository; pass "
                         "--user-agent 'ports-cpe-matcher/0.2 (+https://github.com/<you>/<repo>)' "
                         "or set PORTS_CPE_MATCHER_USER_AGENT")
    projects: dict[str, list] = {}
    start, pages = "", 0
    while True:
        query = urllib.parse.urlencode({"inrepo": "freebsd"})
        url = source.url + (urllib.parse.quote(start, safe="") + "/" if start else "") + "?" + query
        with http_open(url, user_agent, timeout=120) as resp:
            page = json.load(resp)
        pages += 1
        new = [name for name in page if name not in projects]
        projects.update(page)
        if pages % 10 == 0 or not new:
            log(f"    {len(projects)} projects after {pages} requests")
        if not new:
            break
        start = max(page)
        time.sleep(REPOLOGY_DELAY)
    dest = cache / source.path
    dest.parent.mkdir(parents=True, exist_ok=True)
    tmp = dest.with_name(dest.name + ".part")
    tmp.write_text(json.dumps({"fetched": _now(), "repo": "freebsd", "projects": projects}))
    os.replace(tmp, dest)
    return f"{len(projects)} projects", True


def _git(args: list[str], cwd: Path | None = None) -> str:
    proc = subprocess.run(["git", *args], cwd=cwd, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True)
    if proc.returncode != 0:
        lines = [line.strip() for line in proc.stderr.splitlines() if line.strip()]
        reason = next((line for line in lines if line.startswith(("fatal:", "error:"))),
                      lines[-1] if lines else f"exit status {proc.returncode}")
        raise FetchError(f"git {' '.join(args[:2])} failed: {reason}")
    return proc.stdout.strip()


def fetch_git(source: Source, cache: Path, gc: bool) -> tuple[str, bool]:
    """Shallow clone (partial and sparse where configured), or update to the remote HEAD."""
    if shutil.which("git") is None:
        raise FetchError("git is not installed (pkg install git)")
    dest = cache / source.path
    filter_args = ["--filter=blob:none"] if source.partial else []
    before = None
    if (dest / ".git").is_dir():
        before = _git(["rev-parse", "HEAD"], dest)
        # Fetch the remote HEAD into the tracking branch created by the clone. Moving that
        # ref (instead of only FETCH_HEAD) is what lets --gc release the old snapshots.
        tracking = f"refs/remotes/origin/{_git(['symbolic-ref', '--short', 'HEAD'], dest)}"
        _git(["fetch", "--quiet", "--depth", "1", *filter_args, "origin", f"+HEAD:{tracking}"], dest)
        _git(["reset", "--quiet", "--hard", tracking], dest)
        if source.sparse:  # re-apply, in case the patterns changed in a newer version of this script
            _git(["sparse-checkout", "set", "--no-cone", *source.sparse], dest)
    else:
        if dest.exists():
            raise FetchError(f"{dest} exists but is not a git checkout; remove it and retry")
        tmp = dest.with_name(dest.name + ".partial")
        if tmp.exists():
            shutil.rmtree(tmp)
        tmp.parent.mkdir(parents=True, exist_ok=True)
        log(f"    cloning {source.url}")
        _git(["clone", "--quiet", "--depth", "1", "--single-branch", "--no-checkout",
              *filter_args, source.url, str(tmp)])
        if source.sparse:
            _git(["sparse-checkout", "set", "--no-cone", *source.sparse], tmp)
        _git(["checkout", "--quiet"], tmp)
        os.replace(tmp, dest)
    if gc:
        for ref in _git(["for-each-ref", "--format=%(refname)", "refs/remotes/origin/"], dest).split():
            if ref != f"refs/remotes/origin/{_git(['symbolic-ref', '--short', 'HEAD'], dest)}":
                _git(["update-ref", "-d", ref], dest)  # stale tracking refs pin old objects
        _git(["reflog", "expire", "--expire=now", "--all"], dest)
        _git(["gc", "--quiet", "--prune=now"], dest)
    head = _git(["log", "-1", "--format=%h %cs"], dest)
    changed = before != _git(["rev-parse", "HEAD"], dest)
    verb = "cloned at" if before is None else ("updated to" if changed else "already at")
    return f"{verb} {head}", changed


def print_source_table(cache: Path, state: State) -> None:
    print(f"cache directory: {cache}\n")
    print(f"{'source':18} {'present':7}  {'checked':14} {'changed':14} details")
    for source in SOURCES:
        st = state.data.get(source.name, {})
        present = "yes" if source_present(cache, source) else "no"
        print(f"{source.name:18} {present:7}  {_fmt_age(_hours_since(st.get('checked'))):14} "
              f"{_fmt_age(_hours_since(st.get('changed'))):14} {st.get('detail') or source.description}")
    ports = cache / PORTS_FILE
    extracted = (f"extracted {_fmt_age((time.time() - ports.stat().st_mtime) / 3600)}"
                 if ports.exists() else "missing (run extract)")
    print(f"\nports data: {extracted}")


def cmd_fetch(args: argparse.Namespace) -> int:
    cache: Path = args.cache_dir
    state = State(cache)
    if args.list:
        print_source_table(cache, state)
        return 0
    selected: list[str] = []
    for token in args.sources or ["all"]:
        if token in FETCH_GROUPS:
            names = FETCH_GROUPS[token]
        elif token in SOURCE_BY_NAME:
            if SOURCE_BY_NAME[token].kind == "manual":
                log(f"{token} cannot be fetched: {SOURCE_BY_NAME[token].description}")
                continue
            names = [token]
        else:
            log(f"error: unknown source '{token}'; choose from {', '.join(list(FETCH_GROUPS) + FETCH_GROUPS['all'])}")
            return 2
        selected += [n for n in names if n not in selected and n not in (args.skip or [])]

    cache.mkdir(parents=True, exist_ok=True)
    custom_agent = args.user_agent or os.environ.get("PORTS_CPE_MATCHER_USER_AGENT")
    user_agent = custom_agent or DEFAULT_USER_AGENT
    failures, skipped = [], []
    for source in SOURCES:
        if source.name not in selected:
            continue
        st = state.source(source.name)
        log(f"==> {source.name}: {source.description}")
        age = _hours_since(st.get("checked"))
        if (source.kind != "git" and not args.force and age is not None and age < args.max_age
                and source_present(cache, source)):
            log(f"    checked {_fmt_age(age)}, skipped (--max-age {args.max_age:g}, or --force)")
            continue
        if source.kind == "repology" and not (custom_agent and "http" in custom_agent):
            log("    skipped: Repology requires --user-agent (or PORTS_CPE_MATCHER_USER_AGENT) "
                "with a link to your source repository")
            skipped.append(source.name)
            continue
        try:
            if source.kind == "git":
                detail, changed = fetch_git(source, cache, args.gc)
            elif source.kind == "http":
                detail, changed = fetch_http(source, cache, state, user_agent, args.force)
            elif source.kind == "nvd-cve":
                detail, changed = fetch_nvd_cve(source, cache, state, user_agent, args.force)
            else:
                detail, changed = fetch_repology(source, cache, custom_agent)
        except KeyboardInterrupt:
            state.save()
            log("interrupted")
            return 130
        except (FetchError, OSError, ValueError) as exc:  # OSError covers URLError and HTTPError
            failures.append(source.name)
            log(f"    FAILED: {exc}")
            state.save()
            continue
        st["checked"] = _now()
        if changed:
            st["changed"] = st["checked"]
        st["detail"] = detail
        state.save()
        log(f"    {detail}")
    log(f"\ncache: {cache}")
    if skipped:
        log(f"skipped: {', '.join(skipped)}")
    if failures:
        log(f"failed: {', '.join(failures)}")
        return 1
    return 0


# ---------------------------------------------------------------------------
# Matching
# ---------------------------------------------------------------------------
@dataclass
class Evidence:
    family: str
    weight: float
    detail: str


class Candidate:
    def __init__(self, vp: str) -> None:
        self.vp = vp
        self.evidence: list[Evidence] = []
        self.in_nvd: bool | None = None  # None: no NVD data loaded

    def add(self, family: str, weight: float, detail: str) -> None:
        if weight >= 0.01:
            self.evidence.append(Evidence(family, round(weight, 3), detail))

    def best_by_family(self) -> dict[str, Evidence]:
        best: dict[str, Evidence] = {}
        for ev in self.evidence:
            if ev.family not in best or ev.weight > best[ev.family].weight:
                best[ev.family] = ev
        return best

    @property
    def score(self) -> float:
        remaining = 1.0
        for ev in self.best_by_family().values():
            remaining *= 1 - min(ev.weight, 0.95)
        return round(1 - remaining, 3)

    @property
    def independent(self) -> list[str]:
        return sorted(f for f in self.best_by_family() if f not in WEAK_FAMILIES)

    @property
    def level(self) -> str:
        if not self.independent or self.in_nvd is False:
            return "low"  # weak evidence only, or a CPE NVD does not know (useless for matching)
        if self.score >= HIGH:
            return "high"
        return "medium" if self.score >= MEDIUM else "low"

    def sources_str(self) -> str:
        best = sorted(self.best_by_family().items(), key=lambda kv: -kv[1].weight)
        return "; ".join(f"{fam}({ev.weight:.2f})" for fam, ev in best)

    def evidence_str(self) -> str:
        return " | ".join(f"{ev.family}: {ev.detail}"
                          for ev in sorted(self.evidence, key=lambda e: (-e.weight, e.family)))


class MatchContext:
    def __init__(self, cpe_dict: NvdCpeDictionary | None, cves: CveData | None,
                 distros: DistroIndex, repology: RepologyIndex | None, name_all_products: bool) -> None:
        self.cpe_dict, self.cves, self.distros, self.repology = cpe_dict, cves, distros, repology
        self.name_all_products = name_all_products
        self.product_vendors: dict[str, set[str]] = defaultdict(set)
        self.name_index: dict[str, set[str]] = defaultdict(set)
        vps: set[str] = set()
        if cpe_dict:
            vps |= cpe_dict.products.keys()
        if cves:
            vps |= cves.products.keys()
        for vp in vps:
            product = vp_split(vp)[1]
            self.product_vendors[product].add(vp)
            if norm(product):
                self.name_index[norm(product)].add(vp)
        self.has_nvd = bool(vps)

    def known(self, vp: str) -> bool:
        return bool((self.cpe_dict and vp in self.cpe_dict.products) or
                    (self.cves and vp in self.cves.products))

    def cve_count(self, vp: str) -> int:
        return len(self.cves.products.get(vp, ())) if self.cves else 0

    def versions(self, vp: str) -> set[str]:
        out: set[str] = set()
        if self.cpe_dict and vp in self.cpe_dict.products:
            out |= self.cpe_dict.products[vp].versions
        if self.cves:
            out |= self.cves.versions.get(vp, set())
        return out

    def updates(self, vp: str) -> set[str]:
        info = self.cpe_dict.products.get(vp) if self.cpe_dict else None
        return info.updates if info else set()

    def title(self, vp: str) -> str:
        info = self.cpe_dict.products.get(vp) if self.cpe_dict else None
        return info.title if info else ""

    def resolve(self, vendor: str | None, product: str) -> list[tuple[str, float, str]]:
        """Map a (vendor, product) claim from another source to NVD vendor:product keys."""
        if vendor:
            vp = vp_join(vendor, product)
            if not self.has_nvd or self.known(vp):
                return [(vp, 1.0, "")]
            return [(vp, 0.5, " [not found in loaded NVD data]")]
        vps = self.product_vendors.get(product, set())
        if not vps:
            return [] if self.has_nvd else [(vp_join("*", product), 0.5, " [no vendor given]")]
        if len(vps) == 1:
            return [(next(iter(vps)), 0.9, " [vendor inferred from NVD]")]
        return [(vp, 0.5, f" [vendor ambiguous, {len(vps)} NVD vendors]") for vp in sorted(vps)]

    def rank_vps(self, vps, limit: int = 5) -> str:
        ranked = sorted(vps, key=lambda vp: (-self.cve_count(vp), vp))
        shown = [f"{vp} ({self.cve_count(vp)} CVEs)" if self.cves else vp for vp in ranked[:limit]]
        return ", ".join(shown) + (f", +{len(ranked) - limit} more" if len(ranked) > limit else "")

    def version_format_hint(self, vp: str, version: str, update: str) -> tuple[str, str] | None:
        known = self.versions(vp)
        if len(known) < 3 or not version or version in known:
            return None
        alternatives = []
        if version.startswith("v"):
            alternatives.append((version[1:], update))
        m = re.match(r"^(\d+(?:\.\d+)*)[._-]?((?:rc|alpha|beta|pre|p|r)\d+|[a-z])$", version)
        if m:
            alternatives.append((m.group(1), m.group(2)))
        alternatives.append((version.replace("_", "."), update))
        alternatives.append((version.split(",")[0].split("_")[0], update))
        for alt_version, alt_update in alternatives:
            if alt_version == version or alt_version not in known:
                continue
            if alt_update and alt_update != update and alt_update not in self.updates(vp):
                continue  # NVD does not use the update field like that for this product
            detail = f"CPE_VERSION '{version}' is unknown to NVD, but '{alt_version}' is known"
            suggestion = f"CPE_VERSION={alt_version}"
            if alt_update and alt_update != update:
                suggestion += f" CPE_UPDATE={alt_update}"
            return detail, suggestion
        return None


@dataclass
class PortView:
    rec: dict
    origin: str
    pkgname: str
    portname: str
    maintainer: str
    uses_cpe: bool
    cpe_str: str
    current: Cpe | None
    explicit: set
    versions: set
    strong_keys: set
    host_keys: set
    name_keys: set
    vendor_hints: set


def build_port_view(rec: dict) -> PortView:
    strong: set[str] = set()
    hosts: set[str] = set()

    def take(url: str, with_host: bool) -> None:
        for key in url_keys(url):
            if is_strong(key):
                strong.add(key)
            elif with_host:
                hosts.add(key)

    for url in rec.get("www", "").split():
        take(url, True)
    for i, url in enumerate(rec.get("master_sites", "").split()):
        take(url, i == 0)  # only the primary site's host; the rest are usually mirrors
    use_gh = rec.get("use_github", "") not in ("", "nodefault")
    use_gl = rec.get("use_gitlab", "") not in ("", "nodefault")
    if use_gh and rec.get("gh_account") and rec.get("gh_project"):
        take(f"https://github.com/{rec['gh_account']}/{rec['gh_project']}", False)
    if use_gl and rec.get("gl_account") and rec.get("gl_project"):
        site = (rec.get("gl_site") or "https://gitlab.com").rstrip("/")
        take(f"{site}/{rec['gl_account']}/{rec['gl_project']}", False)

    portname = rec.get("portname", "")
    names = {norm(portname)}
    slotless = re.sub(r"\d+$", "", portname)  # lua54, postgresql17, ...
    if len(norm(slotless)) >= 3:
        names.add(norm(slotless))
    if use_gh:
        names.add(norm(rec.get("gh_project")))
    if use_gl:
        names.add(norm(rec.get("gl_project")))
    names.discard("")

    hints = {norm(portname), norm(portname) + "project"}
    if use_gh:
        hints.add(norm(rec.get("gh_account")))
    if use_gl:
        hints.add(norm(rec.get("gl_account")))
    for key in hosts:
        labels = key[5:].split(".")
        if len(labels) >= 2 and labels[-2] not in PLATFORM_LABELS:
            hints.add(norm(labels[-2]))
    for key in strong:
        for platform in ("gnu", "gnome", "kde"):
            if key.startswith(f"proj:{platform}/"):
                hints.add(platform)
        if key.startswith("repo:"):
            hints.add(norm(key.split("/")[1]))
    hints.discard("")

    uses_cpe = bool(rec.get("uses_cpe"))
    current = None
    if uses_cpe:
        current = parse_cpe(rec.get("cpe_str", ""))
        if current is None and rec.get("cpe_vendor") and rec.get("cpe_product"):
            current = Cpe((rec.get("cpe_part") or "a").lower(), _unescape(rec["cpe_vendor"]),
                          _unescape(rec["cpe_product"]), (rec.get("cpe_version") or "").lower(),
                          (rec.get("cpe_update") or "").lower())
    return PortView(
        rec=rec, origin=rec["origin"], pkgname=rec.get("pkgname", ""), portname=portname,
        maintainer=rec.get("maintainer", ""), uses_cpe=uses_cpe, cpe_str=rec.get("cpe_str", ""),
        current=current, explicit=set(rec.get("cpe_explicit") or []),
        versions={v.lower() for v in (rec.get("distversion"), rec.get("portversion")) if v},
        strong_keys=strong, host_keys=hosts, name_keys=names, vendor_hints=hints,
    )


@dataclass
class Join:
    entry: DistroEntry
    weight: float
    how: str


def find_candidates(pv: PortView, ctx: MatchContext) -> tuple[dict[str, Candidate], list[Join]]:
    cands: dict[str, Candidate] = {}

    def add(vp: str, family: str, weight: float, detail: str) -> None:
        if vp not in cands:
            cands[vp] = Candidate(vp)
        cands[vp].add(family, weight, detail)

    # 1. Other distributions (and osv.dev's map), keeping the best join per entry
    joins: dict[int, Join] = {}

    def join(entry: DistroEntry, weight: float, how: str) -> None:
        if entry.source == "osv-cpe-repos":
            weight = min(weight, W_OSV_REPO)
        if id(entry) not in joins or weight > joins[id(entry)].weight:
            joins[id(entry)] = Join(entry, weight, how)

    for key in pv.strong_keys:
        for entry in ctx.distros.by_url.get(key, ()):
            join(entry, W_DISTRO_URL, f"{entry.pkg_id} shares {key}")
    for key in pv.host_keys:
        entries = ctx.distros.by_url.get(key, ())
        if 0 < len(entries) <= 3:
            for entry in entries:
                join(entry, W_DISTRO_HOST, f"{entry.pkg_id} shares {key}")
    if ctx.repology:
        project, ids = ctx.repology.lookup(pv.origin)
        for ident in ids:
            for entry in ctx.distros.by_pkg.get(ident, ()):
                join(entry, W_DISTRO_REPOLOGY, f"{entry.pkg_id} is in Repology project '{project}'")
    for name in pv.name_keys:
        for entry in ctx.distros.by_name.get(name, ()):
            if entry.source != "osv-cpe-repos":
                join(entry, W_DISTRO_NAME, f"{entry.pkg_id} has the same name")
    for j in joins.values():
        for vendor, product in j.entry.cpes:
            for vp, factor, note in ctx.resolve(vendor, product):
                add(vp, j.entry.source, j.weight * factor, j.how + note)

    # 2. NVD CPE dictionary references
    if ctx.cpe_dict:
        for key in pv.strong_keys | pv.host_keys:
            vps = ctx.cpe_dict.ref_index.get(key)
            if not vps or len(vps) > MAX_FANOUT:
                continue
            base = W_REF_STRONG if is_strong(key) else W_REF_HOST
            shared = f" (shared by {len(vps)} products)" if len(vps) > 1 else ""
            for vp in vps:
                add(vp, "nvd-cpe-refs", base / len(vps), f"dictionary references {key}{shared}")

    # 3. CVE references
    if ctx.cves:
        for key in pv.strong_keys:
            hits = ctx.cves.ref_index.get(key)
            if not hits or len(hits) > MAX_FANOUT:
                continue
            total = sum(len(ids) for ids in hits.values())
            for vp, ids in hits.items():
                count = len(ids)
                weight = (W_CVE_REF + 0.05 * min(count - 1, 3)) * count / total
                add(vp, "nvd-cve-refs", weight, f"{count} CVE(s) for this product reference {key}")

    # 4. Name heuristics
    for name in pv.name_keys:
        vps = ctx.name_index.get(name, set())
        if not vps or len(vps) > 25:
            continue
        weight = W_NAME_UNIQUE if len(vps) == 1 else W_NAME_SHARED
        for vp in sorted(vps):
            if (not ctx.name_all_products and ctx.cves and vp not in cands
                    and not ctx.cves.products.get(vp)):
                continue  # name-only candidates must at least have CVEs
            suffix = f" ({len(vps)} vendors use this product name)" if len(vps) > 1 else ""
            add(vp, "name", weight, f"product name matches port name '{name}'{suffix}")

    # 5. Corroboration of every candidate
    for vp, cand in cands.items():
        cand.in_nvd = ctx.known(vp) if ctx.has_nvd else None
        vendor = vp_split(vp)[0]
        if vendor != "*" and norm(vendor) in pv.vendor_hints:
            cand.add("vendor-hint", W_VENDOR_HINT, f"vendor '{vendor}' matches port/upstream naming")
        hit = sorted(pv.versions & ctx.versions(vp))
        if hit:
            cand.add("version", W_VERSION, f"port version {hit[0]} is a known version of this product")
    return cands, list(joins.values())


@dataclass
class Issue:
    issue: str
    severity: str
    details: str
    suggestion: str = ""


def audit_port(pv: PortView, cands: dict[str, Candidate], ranked: list[Candidate],
               joins: list[Join], ctx: MatchContext) -> list[Issue]:
    """Check a port's existing CPE. Each root cause yields one issue; supporting
    observations (distro disagreement, best candidate) are folded into it."""
    if not pv.uses_cpe:
        return []
    issues: list[Issue] = []
    problems = cpe_str_problems(pv.cpe_str)
    if problems:
        issues.append(Issue("malformed-cpe", "high", "; ".join(problems)))
    cur = pv.current
    if cur is None:
        return issues
    vp = cur.vp
    cur_cand = cands.get(vp)
    cur_score = cur_cand.score if cur_cand else 0.0
    best = next((c for c in ranked if c.independent and c.in_nvd is not False), None)
    best_hint = f"{best.vp} [{best.level}, {best.sources_str()}]" if best and best.vp != vp else ""

    # Other distributions that are firmly joined to this port but list a different,
    # NVD-known CPE with another vendor (same vendor, other product = sub-package, ignored)
    disagreements = []
    for j in joins:
        if j.weight < W_DISTRO_URL or j.entry.source == "osv-cpe-repos":
            continue
        claims = [(v, p) for v, p in j.entry.cpes]
        if any((v in (cur.vendor, None)) and p == cur.product for v, p in claims):
            continue
        if any(v == cur.vendor for v, _ in claims):
            continue
        rivals = [vp_join(v, p) for v, p in claims if v and (not ctx.has_nvd or ctx.known(vp_join(v, p)))]
        if rivals:
            disagreements.append((j.entry, rivals))
    disagreement_note = "; ".join(f"{e.source} {e.pkg_id} lists {', '.join(r)}" for e, r in disagreements[:4])

    def with_context(details: str) -> str:
        return details + (f" (also: {disagreement_note})" if disagreement_note else "")

    existence_issue = False
    if ctx.has_nvd and cur.part == "a":
        if not ctx.known(vp):
            existence_issue = True
            others = ctx.product_vendors.get(cur.product, set()) - {vp}
            if others:
                suggestion = ctx.rank_vps(others) + (f"; best evidence: {best_hint}" if best_hint else "")
                issues.append(Issue("vendor-mismatch", "high", with_context(
                    f"'{vp}' is not in NVD, but product '{cur.product}' exists under other vendor(s)"),
                    suggestion))
            elif ctx.cpe_dict:
                issues.append(Issue("unknown-cpe", "medium", with_context(
                    f"'{vp}' is neither in the NVD CPE dictionary nor in CVE data"), best_hint))
            else:
                issues.append(Issue("no-cve-data", "low", with_context(
                    f"'{vp}' appears in no loaded CVE (fine if the software never had one; "
                    f"load --nvd-cpe to check the name itself)"), best_hint))
        else:
            info = ctx.cpe_dict.products.get(vp) if ctx.cpe_dict else None
            if info and info.n_entries and info.n_deprecated == info.n_entries:
                existence_issue = True
                issues.append(Issue("deprecated-cpe", "high", with_context(
                    f"all {info.n_entries} dictionary entries for '{vp}' are deprecated"),
                    ", ".join(sorted(info.deprecated_by)) or best_hint))
            hint = ctx.version_format_hint(vp, cur.version, cur.update)
            if hint:
                issues.append(Issue("version-format", "medium", hint[0], hint[1]))

    if existence_issue:
        return issues

    conflict = (best is not None and best.vp != vp and best.level in ("high", "medium")
                and best.score >= cur_score + 0.2)
    if conflict:
        severity = "high" if best.level == "high" and cur_score < 0.2 else "medium"
        issues.append(Issue("evidence-conflict", severity, with_context(
            f"best-supported candidate {best.vp} scores {best.score:.2f}, current {vp} scores "
            f"{cur_score:.2f}"), best_hint))
    elif disagreements:
        cur_cves = ctx.cve_count(vp)
        rival_cves = max(ctx.cve_count(r) for _, rs in disagreements for r in rs)
        severity = "medium" if rival_cves > cur_cves else "low"
        issues.append(Issue("distro-disagreement", severity,
                            f"{disagreement_note} (current {vp}: {cur_cves} CVEs, "
                            f"rival: up to {rival_cves})", best_hint))

    if (not conflict and ctx.has_nvd and "VENDOR" not in pv.explicit and "STR" not in pv.explicit
            and not (cur_cand and cur_cand.independent)):
        rivals = {o for o in ctx.product_vendors.get(cur.product, set())
                  if o != vp and (ctx.cve_count(o) > 0 or not ctx.cves)}
        if rivals:
            issues.append(Issue("ambiguous-default-vendor", "low",
                                f"vendor '{cur.vendor}' comes from the USES=cpe default and nothing "
                                f"corroborates it; {len(rivals)} other vendor(s) use product "
                                f"'{cur.product}'", ctx.rank_vps(rivals)))
    return issues


# ---------------------------------------------------------------------------
# match command
# ---------------------------------------------------------------------------
def cmd_match(args: argparse.Namespace) -> int:
    cache: Path = args.cache_dir
    ports_file = cache / PORTS_FILE
    if not ports_file.exists():
        log(f"error: {ports_file} not found; run the extract command first")
        return 2
    latest: dict[str, dict] = {}
    for rec in read_jsonl(ports_file):
        latest[rec["origin"]] = rec  # after extract --resume the last record for an origin wins
    records = sorted(latest.values(), key=lambda r: r["origin"])
    failed = [r for r in records if r.get("error")]
    records = [r for r in records if not r.get("error")]
    if args.origin:
        wanted = set(args.origin)
        records = [r for r in records if r["origin"] in wanted]
    ports_age = (time.time() - ports_file.stat().st_mtime) / 3600
    log(f"cache: {cache}")
    log(f"ports: {len(records)} ports extracted {_fmt_age(ports_age)} "
        f"({len(failed)} extraction errors skipped)")

    state = State(cache)
    excluded = set(args.exclude or [])
    present: dict[str, Path] = {}
    log("sources:")
    for source in SOURCES:
        if source.name in excluded:
            log(f"  {source.name:18} excluded")
            continue
        if not source_present(cache, source):
            hint = "optional, see --help" if source.kind == "manual" else f"run: fetch {source.name}"
            log(f"  {source.name:18} missing ({hint})")
            continue
        age = source_age_hours(cache, state, source)
        stale = "  <- stale, consider running fetch" if age is not None and age > STALE_HOURS else ""
        log(f"  {source.name:18} checked {_fmt_age(age)}{stale}")
        present[source.name] = cache / source.path

    cpe_dict = None
    if "nvd-cpe" in present:
        cpe_dict = NvdCpeDictionary()
        cpe_dict.load(present["nvd-cpe"])
    cves = None
    if "nvd-cve" in present or "cvelist" in present:
        cves = CveData()
        if "nvd-cve" in present:
            cves.load_nvd(present["nvd-cve"])
        if "cvelist" in present:
            root = present["cvelist"]
            cves.load_cvelist(root / "cves" if (root / "cves").is_dir() else root)
    distros = DistroIndex()
    distro_loaders = [("gentoo", load_gentoo), ("buildroot", load_buildroot),
                      ("openembedded-core", load_openembedded), ("meta-openembedded", load_openembedded),
                      ("nixpkgs", load_nixpkgs_json), ("osv-cpe-repos", load_osv_cpe_repos)]
    for name, loader in distro_loaders:
        if name in present:
            count = 0
            for entry in loader(present[name]):
                distros.add(entry)
                count += 1
            log(f"{name}: {count} entries with CPEs")
    repology = RepologyIndex.load(present["repology"]) if "repology" in present else None
    ctx = MatchContext(cpe_dict, cves, distros, repology, args.name_all_products)
    if not ctx.has_nvd:
        log("warning: no NVD data available; candidates cannot be validated and existing CPEs "
            "cannot be checked for existence (run: fetch nvd-cpe nvd-cve)")

    out = args.out or cache / REPORT_DIR
    out.mkdir(parents=True, exist_ok=True)
    out.mkdir(parents=True, exist_ok=True)
    cand_rows, issue_rows, report_ports = [], [], []
    stats: Counter = Counter()
    issue_stats: Counter = Counter()

    for rec in records:
        pv = build_port_view(rec)
        cands, joins = find_candidates(pv, ctx)
        ranked = sorted(cands.values(), key=lambda c: (-c.score, c.vp))
        issues = audit_port(pv, cands, ranked, joins, ctx)
        cur_vp = pv.current.vp if pv.current else ""
        shown = [c for c in ranked if c.score >= args.min_score][: args.max_candidates]
        if cur_vp in cands and cands[cur_vp] not in shown:
            shown.append(cands[cur_vp])

        stats["ports"] += 1
        stats["uses_cpe"] += pv.uses_cpe
        best_level = next((c.level for c in ranked if c.vp != cur_vp), None)
        if not pv.uses_cpe and best_level in ("high", "medium"):
            stats[f"without_cpe_{best_level}_candidate"] += 1

        for cand in shown:
            cand_rows.append({
                "origin": pv.origin, "pkgname": pv.pkgname, "maintainer": pv.maintainer,
                "current_cpe": cur_vp, "candidate": cand.vp,
                "is_current": "yes" if cand.vp == cur_vp else "",
                "confidence": cand.level, "score": f"{cand.score:.3f}",
                "independent_sources": len(cand.independent), "sources": cand.sources_str(),
                "evidence": cand.evidence_str(),
                "in_nvd": {True: "yes", False: "no", None: "unknown"}[cand.in_nvd],
                "cve_count": ctx.cve_count(cand.vp) if cves else "", "nvd_title": ctx.title(cand.vp),
                "_best": shown[0].score if shown else 0.0,
            })
        for issue in issues:
            issue_stats[(issue.issue, issue.severity)] += 1
            issue_rows.append({
                "origin": pv.origin, "pkgname": pv.pkgname, "maintainer": pv.maintainer,
                "current_cpe": cur_vp, "cpe_str": pv.cpe_str, "issue": issue.issue,
                "severity": issue.severity, "details": issue.details, "suggestion": issue.suggestion,
            })
        if shown or issues:
            report_ports.append({
                "origin": pv.origin, "pkgname": pv.pkgname, "maintainer": pv.maintainer,
                "uses_cpe": pv.uses_cpe, "cpe_str": pv.cpe_str, "current": cur_vp,
                "candidates": [{
                    "cpe": c.vp, "confidence": c.level, "score": c.score,
                    "independent_sources": c.independent,
                    "evidence": [ev.__dict__ for ev in sorted(c.evidence, key=lambda e: -e.weight)],
                } for c in shown],
                "issues": [i.__dict__ for i in issues],
            })

    cand_rows.sort(key=lambda r: (-r["_best"], r["origin"], -float(r["score"])))
    for row in cand_rows:
        del row["_best"]
    issue_rows.sort(key=lambda r: (SEVERITY_ORDER[r["severity"]], r["issue"], r["origin"]))
    _write_csv(out / "candidates.csv", cand_rows,
               ["origin", "pkgname", "maintainer", "current_cpe", "candidate", "is_current",
                "confidence", "score", "independent_sources", "sources", "evidence", "in_nvd",
                "cve_count", "nvd_title"])
    _write_csv(out / "cpe_issues.csv", issue_rows,
               ["origin", "pkgname", "maintainer", "current_cpe", "cpe_str", "issue", "severity",
                "details", "suggestion"])
    summary = {
        "ports": stats["ports"], "ports_with_uses_cpe": stats["uses_cpe"],
        "ports_without_cpe_with_high_candidate": stats["without_cpe_high_candidate"],
        "ports_without_cpe_with_medium_candidate": stats["without_cpe_medium_candidate"],
        "extraction_errors": len(failed),
        "issues": {f"{k[0]}/{k[1]}": v for k, v in sorted(issue_stats.items())},
        "sources": {
            "nvd_cpe_products": len(cpe_dict.products) if cpe_dict else 0,
            "cve_records": cves.records if cves else 0,
            "distro_entries": dict(distros.counts),
            "repology_origins": len(repology.origin_to_project) if repology else 0,
        },
    }
    with (out / "report.json").open("w") as fh:
        json.dump({"generated": datetime.now(timezone.utc).isoformat(), "summary": summary,
                   "ports": report_ports}, fh, indent=1)
    log(json.dumps(summary, indent=2))
    log(f"wrote {out/'candidates.csv'}, {out/'cpe_issues.csv'}, {out/'report.json'}")
    return 0


def _write_csv(path: Path, rows: list[dict], columns: list[str]) -> None:
    with path.open("w", newline="") as fh:
        writer = csv.DictWriter(fh, fieldnames=columns)
        writer.writeheader()
        writer.writerows(rows)


# ---------------------------------------------------------------------------
# CLI
# ---------------------------------------------------------------------------
def main(argv: list[str] | None = None) -> int:
    fmt = argparse.RawDescriptionHelpFormatter
    common = argparse.ArgumentParser(add_help=False)
    common.add_argument("--cache-dir", type=lambda s: Path(s).expanduser(), default=default_cache_dir(),
                        help=f"cache directory (default: {default_cache_dir()})")
    parser = argparse.ArgumentParser(description=__doc__, formatter_class=fmt)
    sub = parser.add_subparsers(dest="command", required=True)
    fetchable = [s.name for s in SOURCES if s.kind != "manual"]

    p = sub.add_parser("fetch", parents=[common], formatter_class=fmt,
                       help="clone/update repositories and download data into the cache",
                       description="Clone or update the git repositories and download the raw data.\n\n"
                                   "sources:\n" + "\n".join(f"  {s.name:18} {s.description}" for s in SOURCES)
                                   + "\n\ngroups:\n  all                everything fetchable (default)"
                                   "\n  git                all git repositories\n  data               all downloaded data")
    p.add_argument("sources", nargs="*", metavar="SOURCE", help="sources or groups to fetch (default: all)")
    p.add_argument("--skip", action="append", choices=fetchable, metavar="SOURCE", help="skip a source (repeatable)")
    p.add_argument("--list", action="store_true", help="show sources and cache status, fetch nothing")
    p.add_argument("--force", action="store_true", help="ignore --max-age and re-download unchanged data")
    p.add_argument("--max-age", type=float, default=12.0, metavar="HOURS",
                   help="skip downloaded sources checked more recently than this (default: 12); "
                        "git repositories are always updated")
    p.add_argument("--user-agent", help="User-Agent with a link to your source repository "
                                        "(required by Repology; env PORTS_CPE_MATCHER_USER_AGENT)")
    p.add_argument("--gc", action="store_true", help="prune old objects in the git repositories afterwards")
    p.set_defaults(func=cmd_fetch)

    p = sub.add_parser("extract", parents=[common], help="extract port metadata with make -V into the cache")
    p.add_argument("--ports-tree", default="/usr/ports")
    p.add_argument("--jobs", type=int, default=os.cpu_count() or 4)
    p.add_argument("--make", default="make", help="make binary (bmake on non-FreeBSD hosts)")
    p.add_argument("--timeout", type=int, default=120, help="seconds per port")
    p.add_argument("--category", action="append", help="only these categories (repeatable)")
    p.add_argument("--origin", action="append", help="only these origins, e.g. ftp/curl (repeatable)")
    p.add_argument("--resume", action="store_true",
                   help="continue an interrupted extraction, or retry the ports that failed")
    p.add_argument("--use-make-conf", action="store_true", help="honour /etc/make.conf")
    p.set_defaults(func=cmd_extract)

    p = sub.add_parser("match", parents=[common], help="match cached ports against cached sources (offline)")
    p.add_argument("--out", type=lambda s: Path(s).expanduser(), help="report directory (default: <cache>/report)")
    p.add_argument("--exclude", action="append", choices=[s.name for s in SOURCES], metavar="SOURCE",
                   help="ignore a cached source (repeatable)")
    p.add_argument("--origin", action="append", help="only these origins (repeatable)")
    p.add_argument("--min-score", type=float, default=0.15)
    p.add_argument("--max-candidates", type=int, default=5)
    p.add_argument("--name-all-products", action="store_true",
                   help="also propose name-only candidates for products without CVEs")
    p.set_defaults(func=cmd_match)

    args = parser.parse_args(argv)
    if hasattr(signal, "SIGPIPE"):
        signal.signal(signal.SIGPIPE, signal.SIG_DFL)  # `... | head` should not print a traceback
    return args.func(args)


if __name__ == "__main__":
    sys.exit(main())
