Skip to content

Commit 9acfffa

Browse files
committed
feat(scanner): CVE pipeline, suppressions, baseline, incremental scan (PR-12)
- cli/dsgai_cve.py (stdlib urllib): CVE enrichment in the CLI so the LLM never transcribes CVE data. OSV /v1/querybatch is the per-version source; NVD enriches CVSS by cveId only (no keywordSearch). Cached at ~/.dsgai/cve-cache/ (24h TTL, --refresh-cve); classification is CVSS-aware (>=7 or CRITICAL/HIGH = EXPLOITABLE) and CVSS is cached so online and offline runs are byte-identical. - Suppressions: inline '# dsgai-ignore: P##.# reason="..."' (reason required) move findings to a visible checkpoint 'suppressed' list, never silent. Each directive suppresses exactly one finding (same line, else the line below). - Baseline: 'baseline' subcommand writes fingerprints; --baseline gates only on NEW findings (baselined ones carry baselined: true). - Incremental: --diff <ref> scans only git-changed files; scan_scope becomes 'diff:<ref>' and the report is labelled INCREMENTAL. - New 'cve' subcommand; --exclude/--diff/--baseline wired through the skill and the Action (now fetches dsgai_cve.py). Exploitable CVEs also gate the build. - schemas/dsgai-scan.schema.json extended (suppressed[], baselined, cve shape), still forbidding match content on findings AND suppressed entries. Verified: fixture findings unchanged (29, schema-valid); baseline round-trip (16 fps -> exit 0); suppression suppresses exactly one; langchain==0.1.0 yields real OSV advisories incl. EXPLOITABLE with a populated cache; offline re-run identical. pytest 16 passed (+1 live-CVE test, opt-in via DSGAI_CVE_LIVE).
1 parent 0f3385c commit 9acfffa

7 files changed

Lines changed: 504 additions & 15 deletions

File tree

‎dsgai_scanner_tool/CHANGES_v0.3.md‎

Lines changed: 13 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -9,6 +9,19 @@ dates are ISO-8601. The previous line is recorded in [`CHANGES_v0.2.md`](CHANGES
99
## [Unreleased]
1010

1111
### Added
12+
- **CVE pipeline, suppressions, baseline, incremental scanning** (PR-12).
13+
- CVE fetching moved into the CLI (`cli/dsgai_cve.py`, stdlib urllib): OSV
14+
`querybatch` is the per-version source, NVD enriches CVSS by `cveId` only (no
15+
`keywordSearch`). Cached at `~/.dsgai/cve-cache/` (24h TTL, `--refresh-cve`);
16+
online and offline runs are byte-identical. **The LLM never transcribes CVE data.**
17+
- Inline `# dsgai-ignore: P##.# reason="…"` suppressions — surfaced in a visible
18+
`suppressed` section, never silently dropped; a reason is required.
19+
- `baseline` subcommand + `--baseline` — gate only on findings not in the baseline.
20+
- `--diff <ref>` incremental scans (files changed vs a ref), labelled
21+
"INCREMENTAL — not a full assessment".
22+
- New `cve` subcommand; `--exclude`/`--diff`/`--baseline` wired through the skill and
23+
the Action (which now fetches `dsgai_cve.py`). CVE enrichment reaches CI Job 1 with
24+
no WebFetch. `langchain==0.1.0` yields real OSV advisories incl. EXPLOITABLE.
1225
- Contributor infrastructure: `[scanner]` GitHub issue-form templates (false-positive,
1326
false-negative, new-rule, bug), scanner `CONTRIBUTING.md`, public `ROADMAP.md`, and
1427
this changelog scaffold. (PR-01)
Lines changed: 203 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,203 @@
1+
#!/usr/bin/env python3
2+
"""DSGAI CVE enrichment — deterministic, cached, stdlib-only.
3+
4+
The CLI fetches CVE data so the LLM never transcribes it (hallucinated CVEs
5+
become impossible by construction — the model renders what this fetched).
6+
7+
Sources:
8+
- OSV (https://osv.dev) — the only per-version source, via POST /v1/querybatch.
9+
- NVD — used SOLELY to enrich a known CVE id with CVSS via ?cveId= (never
10+
keywordSearch, which returns junk for names like "ai"/"instructor").
11+
12+
Cache: ~/.dsgai/cve-cache/<ecosystem>/<package>@<version>.json, 24h TTL.
13+
`--refresh-cve` ignores the cache; `offline=True` uses cache only (no network).
14+
"""
15+
import json
16+
import os
17+
import re
18+
import time
19+
import urllib.error
20+
import urllib.request
21+
22+
OSV_BATCH = "https://api.osv.dev/v1/querybatch"
23+
OSV_VULN = "https://api.osv.dev/v1/vulns/"
24+
NVD_CVE = "https://services.nvd.nist.gov/rest/json/cves/2.0"
25+
CACHE_TTL = 24 * 3600
26+
CACHE_DIR = os.path.join(os.path.expanduser("~"), ".dsgai", "cve-cache")
27+
28+
# Minimal ecosystem detection from manifest filename → OSV ecosystem.
29+
ECOSYSTEMS = {
30+
"requirements.txt": "PyPI", "requirements": "PyPI", "pyproject.toml": "PyPI",
31+
"package.json": "npm", "go.mod": "Go", "Cargo.toml": "crates.io",
32+
"Gemfile.lock": "RubyGems",
33+
}
34+
35+
_REQ_RE = re.compile(r'^\s*([A-Za-z0-9_.\-]+)\s*==\s*([A-Za-z0-9_.\-]+)')
36+
37+
38+
def _cache_path(eco, pkg, ver):
39+
safe = re.sub(r'[^A-Za-z0-9_.\-]', '_', f"{pkg}@{ver}")
40+
return os.path.join(CACHE_DIR, re.sub(r'[^A-Za-z0-9_.\-]', '_', eco), safe + ".json")
41+
42+
43+
def _cache_get(eco, pkg, ver, refresh):
44+
if refresh:
45+
return None
46+
p = _cache_path(eco, pkg, ver)
47+
try:
48+
if time.time() - os.path.getmtime(p) <= CACHE_TTL:
49+
return json.load(open(p, encoding="utf-8"))
50+
except (OSError, ValueError):
51+
return None
52+
return None
53+
54+
55+
def _cache_put(eco, pkg, ver, data):
56+
p = _cache_path(eco, pkg, ver)
57+
os.makedirs(os.path.dirname(p), exist_ok=True)
58+
with open(p, "w", encoding="utf-8", newline="\n") as fh:
59+
json.dump(data, fh, sort_keys=True)
60+
61+
62+
def parse_dependencies(discovered):
63+
"""Return [{ecosystem, package, version}] from pinned manifest entries.
64+
65+
`discovered` is a list of (abs_path, rel_path). Only exact-pinned entries are
66+
queried (OSV needs a concrete version).
67+
"""
68+
deps, seen = [], set()
69+
for ap, rel in discovered:
70+
base = os.path.basename(rel)
71+
eco = ECOSYSTEMS.get(base)
72+
if eco == "PyPI" and base.startswith("requirements"):
73+
try:
74+
for line in open(ap, encoding="utf-8", errors="replace"):
75+
if line.lstrip().startswith("#"):
76+
continue
77+
m = _REQ_RE.match(line)
78+
if m:
79+
key = (eco, m.group(1).lower(), m.group(2))
80+
if key not in seen:
81+
seen.add(key)
82+
deps.append({"ecosystem": eco, "package": m.group(1),
83+
"version": m.group(2)})
84+
except OSError:
85+
continue
86+
return deps
87+
88+
89+
def _http_json(url, data=None, timeout=20):
90+
headers = {"Content-Type": "application/json", "User-Agent": "dsgai-scan"}
91+
req = urllib.request.Request(
92+
url, data=json.dumps(data).encode() if data is not None else None,
93+
headers=headers, method="POST" if data is not None else "GET")
94+
with urllib.request.urlopen(req, timeout=timeout) as resp:
95+
return json.loads(resp.read().decode())
96+
97+
98+
def _severity_of(detail):
99+
"""Extract a coarse severity + CVSS score from an OSV vuln detail."""
100+
score = None
101+
for sev in detail.get("severity", []) or []:
102+
if sev.get("type", "").startswith("CVSS"):
103+
# OSV gives a vector; we keep the label from DB-specific fields below.
104+
score = sev.get("score")
105+
# database_specific severity label (GHSA gives HIGH/CRITICAL/…)
106+
label = (detail.get("database_specific", {}) or {}).get("severity")
107+
return label, score
108+
109+
110+
def classify(label, cvss):
111+
"""EXPLOITABLE for high/critical (CVSS >= 7 or CRITICAL/HIGH label);
112+
VULNERABLE for anything else with a severity; INFO if wholly unknown."""
113+
if (cvss is not None and cvss >= 7.0) or label in ("CRITICAL", "HIGH"):
114+
return "EXPLOITABLE"
115+
if (cvss is not None) or label in ("MODERATE", "MEDIUM", "LOW"):
116+
return "VULNERABLE"
117+
return "INFO"
118+
119+
120+
def enrich_cvss_nvd(cve_id, offline, timeout=20):
121+
"""Fetch CVSS base score for a CVE id from NVD (cveId only). None on failure."""
122+
if offline or not cve_id.startswith("CVE-"):
123+
return None
124+
try:
125+
data = _http_json(f"{NVD_CVE}?cveId={cve_id}", timeout=timeout)
126+
for v in data.get("vulnerabilities", []):
127+
metrics = v.get("cve", {}).get("metrics", {})
128+
for key in ("cvssMetricV31", "cvssMetricV30", "cvssMetricV2"):
129+
if metrics.get(key):
130+
return metrics[key][0]["cvssData"].get("baseScore")
131+
except (urllib.error.URLError, ValueError, KeyError, TimeoutError):
132+
return None
133+
return None
134+
135+
136+
def enrich(discovered, offline=False, refresh=False):
137+
"""Return a list of CVE finding dicts (deterministic order).
138+
139+
Each: {package, version, ecosystem, id, aliases, summary, status, cvss}.
140+
Cached per {ecosystem, package, version}; second run needs no network.
141+
"""
142+
deps = parse_dependencies(discovered)
143+
results = []
144+
uncached = []
145+
cache_map = {}
146+
for d in deps:
147+
c = _cache_get(d["ecosystem"], d["package"], d["version"], refresh)
148+
if c is not None:
149+
cache_map[(d["ecosystem"], d["package"].lower(), d["version"])] = c
150+
else:
151+
uncached.append(d)
152+
153+
fetched = {}
154+
if uncached and not offline:
155+
try:
156+
queries = [{"package": {"name": d["package"], "ecosystem": d["ecosystem"]},
157+
"version": d["version"]} for d in uncached]
158+
batch = _http_json(OSV_BATCH, {"queries": queries})
159+
for d, res in zip(uncached, batch.get("results", [])):
160+
vulns = []
161+
for v in res.get("vulns", []) or []:
162+
try:
163+
detail = _http_json(OSV_VULN + v["id"])
164+
except (urllib.error.URLError, ValueError, TimeoutError):
165+
detail = {"id": v["id"]}
166+
aliases = sorted(detail.get("aliases", []) or [])
167+
label, _ = _severity_of(detail)
168+
# NVD enriches CVSS for known CVE ids; stored in cache so
169+
# offline re-runs are byte-identical to the online run.
170+
cvss = None
171+
for aid in [detail.get("id", v["id"])] + aliases:
172+
if aid.startswith("CVE-"):
173+
cvss = enrich_cvss_nvd(aid, offline=False)
174+
if cvss is not None:
175+
break
176+
vulns.append({
177+
"id": detail.get("id", v["id"]),
178+
"aliases": aliases,
179+
"summary": (detail.get("summary") or "")[:300],
180+
"cvss": cvss,
181+
"status": classify(label, cvss),
182+
})
183+
data = {"vulns": vulns}
184+
_cache_put(d["ecosystem"], d["package"], d["version"], data)
185+
fetched[(d["ecosystem"], d["package"].lower(), d["version"])] = data
186+
except (urllib.error.URLError, ValueError, TimeoutError):
187+
pass # network failure → whatever is cached still renders
188+
189+
merged = dict(cache_map)
190+
merged.update(fetched)
191+
for d in deps:
192+
key = (d["ecosystem"], d["package"].lower(), d["version"])
193+
for v in merged.get(key, {}).get("vulns", []):
194+
# cvss/status come from the cache (populated at fetch time) so an
195+
# offline re-run is byte-identical to the online run that seeded it.
196+
results.append({
197+
"package": d["package"], "version": d["version"],
198+
"ecosystem": d["ecosystem"], "id": v["id"],
199+
"aliases": v.get("aliases", []), "summary": v.get("summary", ""),
200+
"status": v.get("status", "INFO"), "cvss": v.get("cvss"),
201+
})
202+
results.sort(key=lambda x: (x["package"], x["version"], x["id"]))
203+
return results

0 commit comments

Comments
 (0)