729 lines
30 KiB
Python
Executable File
729 lines
30 KiB
Python
Executable File
"""
|
|
utils/enricher.py — Markdown artifact enrichment.
|
|
|
|
Scans a markdown report for security-relevant artifacts (CVEs, IPv4 addresses,
|
|
hostnames, emails, GitHub repos / commits, raw URLs, vendor product names) and:
|
|
|
|
1. Rewrites each occurrence inline as a clickable markdown link pointing at
|
|
the most useful primary lookup (NVD for CVEs, Shodan for IPs, github.com
|
|
for repos, etc.).
|
|
2. Appends an "INTEL APPENDIX" section grouping every unique artifact with
|
|
the FULL set of enrichment links (Shodan + Censys + VirusTotal for IPs,
|
|
NVD + GitHub PoC search + exploit-db + nuclei templates for CVEs, etc.).
|
|
|
|
Code fences, existing markdown links and inline code spans are preserved
|
|
unmodified so the original report stays readable.
|
|
|
|
Public API:
|
|
enrich_markdown(text: str) -> tuple[str, Stats]
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import ipaddress
|
|
import re
|
|
from dataclasses import dataclass, field
|
|
from typing import Callable
|
|
from urllib.parse import quote, quote_plus
|
|
|
|
# ──────────────────────────────────────────────────────────────────────────
|
|
# Regex patterns
|
|
# ──────────────────────────────────────────────────────────────────────────
|
|
# CVE: CVE-YYYY-NNNN(N+) — case insensitive, normalised to uppercase.
|
|
CVE_RE = re.compile(r"\bCVE-\d{4}-\d{4,7}\b", re.IGNORECASE)
|
|
|
|
# IPv4 with optional CIDR; we filter private/invalid in post-processing.
|
|
IPV4_RE = re.compile(
|
|
r"\b(?:25[0-5]|2[0-4]\d|[01]?\d\d?)"
|
|
r"(?:\.(?:25[0-5]|2[0-4]\d|[01]?\d\d?)){3}"
|
|
r"(?:/\d{1,2})?\b"
|
|
)
|
|
|
|
# Emails — RFC-ish, good enough for report text.
|
|
EMAIL_RE = re.compile(r"\b[\w.+-]+@[\w-]+\.[\w.-]{2,}\b")
|
|
|
|
# Raw http(s) URLs — used both for "wrap as link" and for hostname extraction.
|
|
URL_RE = re.compile(r"https?://[^\s<>()\[\]\"']+", re.IGNORECASE)
|
|
|
|
# GitHub repo refs like org/repo, optionally followed by @sha or #commit-sha.
|
|
# Restrictive: 1st segment 1-39 chars (GitHub login limit), 2nd 1-100 chars,
|
|
# both must contain at least one letter to avoid matching numeric paths.
|
|
GH_REPO_RE = re.compile(
|
|
r"(?<![\w/])"
|
|
r"([A-Za-z0-9](?:[A-Za-z0-9-]{0,38}))"
|
|
r"/"
|
|
r"([A-Za-z0-9_][A-Za-z0-9_.-]{0,99})"
|
|
r"(?:[@#]([0-9a-f]{7,40}))?"
|
|
r"(?![\w/])"
|
|
)
|
|
|
|
# Bare hostnames (FQDNs not part of a URL). We extract from the full text but
|
|
# skip ones already inside URLs to avoid double-linking.
|
|
HOSTNAME_RE = re.compile(
|
|
r"(?<![\w/@.-])"
|
|
r"(?:[a-z0-9](?:[a-z0-9-]{0,61}[a-z0-9])?\.){1,}"
|
|
r"(?:[a-z]{2,24})"
|
|
r"(?![\w-])",
|
|
re.IGNORECASE,
|
|
)
|
|
|
|
# Git commit SHA (7-40 hex). Only enriched when paired with a known repo.
|
|
SHA_RE = re.compile(r"\b([0-9a-f]{7,40})\b")
|
|
|
|
# Code fence / inline code stripper — protects these regions from rewriting.
|
|
FENCED_BLOCK_RE = re.compile(r"```.*?```", re.DOTALL)
|
|
INLINE_CODE_RE = re.compile(r"`[^`\n]+`")
|
|
EXISTING_LINK_RE = re.compile(r"\[[^\]]+\]\([^)]+\)")
|
|
MD_TABLE_DIVIDER_RE = re.compile(r"^\s*\|?\s*:?-{3,}", re.MULTILINE)
|
|
|
|
# Vendors / products → vendor pages. Map of regex → (display, url).
|
|
# Keys are matched case-insensitively as whole "words" (word boundaries).
|
|
VENDORS: dict[str, tuple[str, str]] = {
|
|
r"Blackboard Learn": (
|
|
"Blackboard Learn",
|
|
"https://help.blackboard.com/Learn",
|
|
),
|
|
r"OmniUpdate(?: CMS)?|Modern Campus CMS|OU Campus": (
|
|
"Modern Campus CMS (OmniUpdate)",
|
|
"https://moderncampus.com/cms/",
|
|
),
|
|
r"Apache TomEE": (
|
|
"Apache TomEE",
|
|
"https://tomee.apache.org/security/",
|
|
),
|
|
r"Sitefinity": (
|
|
"Progress Sitefinity",
|
|
"https://www.progress.com/sitefinity-cms",
|
|
),
|
|
r"Anthology Engage(?:CMS)?": (
|
|
"Anthology Engage",
|
|
"https://www.anthology.com/products/engagement/engage",
|
|
),
|
|
r"PeopleSoft": (
|
|
"Oracle PeopleSoft",
|
|
"https://www.oracle.com/applications/peoplesoft/",
|
|
),
|
|
r"CourseLeaf": (
|
|
"CourseLeaf (Leepfrog)",
|
|
"https://courseleaf.com/",
|
|
),
|
|
r"\bInfor\b": (
|
|
"Infor",
|
|
"https://www.infor.com/",
|
|
),
|
|
r"Kaltura": (
|
|
"Kaltura",
|
|
"https://corp.kaltura.com/",
|
|
),
|
|
r"TeamDynamix": (
|
|
"TeamDynamix",
|
|
"https://www.teamdynamix.com/",
|
|
),
|
|
r"FortiGuard": (
|
|
"FortiGuard Labs",
|
|
"https://www.fortiguard.com/",
|
|
),
|
|
r"Cisco SCCP|Cisco SIP|Skinny Call Control Protocol": (
|
|
"Cisco SCCP/SIP",
|
|
"https://www.cisco.com/c/en/us/products/unified-communications/index.html",
|
|
),
|
|
r"New Relic": (
|
|
"New Relic",
|
|
"https://newrelic.com/",
|
|
),
|
|
r"Supabase": (
|
|
"Supabase",
|
|
"https://supabase.com/docs",
|
|
),
|
|
r"Vercel": (
|
|
"Vercel",
|
|
"https://vercel.com/",
|
|
),
|
|
r"ProtonMail": (
|
|
"Proton Mail",
|
|
"https://proton.me/mail",
|
|
),
|
|
}
|
|
|
|
# Pre-compile vendor regex (case-insensitive, word boundaries where sensible).
|
|
_VENDOR_PATTERNS: list[tuple[re.Pattern[str], str, str]] = [
|
|
(re.compile(rf"\b(?:{pat})\b", re.IGNORECASE), disp, url)
|
|
for pat, (disp, url) in VENDORS.items()
|
|
]
|
|
|
|
|
|
# ──────────────────────────────────────────────────────────────────────────
|
|
# Enrichment link factories
|
|
# ──────────────────────────────────────────────────────────────────────────
|
|
def cve_links(cve: str) -> list[tuple[str, str]]:
|
|
"""Return (label, url) tuples for a CVE id."""
|
|
q = quote_plus(cve)
|
|
return [
|
|
("NVD", f"https://nvd.nist.gov/vuln/detail/{cve}"),
|
|
("MITRE", f"https://cve.mitre.org/cgi-bin/cvename.cgi?name={cve}"),
|
|
("GitHub Advisory", f"https://github.com/advisories?query={q}"),
|
|
("GitHub PoC search", f"https://github.com/search?q={q}&type=repositories"),
|
|
("exploit-db", f"https://www.exploit-db.com/search?cve={cve.removeprefix('CVE-')}"),
|
|
("Nuclei templates", f"https://github.com/projectdiscovery/nuclei-templates/search?q={q}"),
|
|
("trickest/cve", f"https://github.com/trickest/cve/tree/main/{cve.split('-')[1]}"),
|
|
("Vulncheck", f"https://vulncheck.com/browse/cve/{cve}"),
|
|
]
|
|
|
|
|
|
def ipv4_links(ip: str) -> list[tuple[str, str]]:
|
|
"""Return (label, url) tuples for an IPv4 address (no CIDR suffix)."""
|
|
return [
|
|
("Shodan", f"https://www.shodan.io/host/{ip}"),
|
|
("Censys", f"https://search.censys.io/hosts/{ip}"),
|
|
("VirusTotal", f"https://www.virustotal.com/gui/ip-address/{ip}"),
|
|
("GreyNoise", f"https://viz.greynoise.io/ip/{ip}"),
|
|
("AbuseIPDB", f"https://www.abuseipdb.com/check/{ip}"),
|
|
("ipinfo.io", f"https://ipinfo.io/{ip}"),
|
|
]
|
|
|
|
|
|
def hostname_links(host: str) -> list[tuple[str, str]]:
|
|
"""Return (label, url) tuples for a hostname / FQDN."""
|
|
host_q = quote(host)
|
|
return [
|
|
("Open", f"https://{host}"),
|
|
("crt.sh", f"https://crt.sh/?q=%25.{host_q}"),
|
|
("urlscan.io", f"https://urlscan.io/domain/{host_q}"),
|
|
("Shodan", f"https://www.shodan.io/search?query=hostname%3A{host_q}"),
|
|
("VirusTotal", f"https://www.virustotal.com/gui/domain/{host_q}"),
|
|
("SecurityTrails", f"https://securitytrails.com/domain/{host_q}/dns"),
|
|
("Wayback", f"https://web.archive.org/web/*/{host_q}/*"),
|
|
]
|
|
|
|
|
|
def email_links(email: str) -> list[tuple[str, str]]:
|
|
"""Return (label, url) tuples for an email address."""
|
|
local, _, domain = email.partition("@")
|
|
return [
|
|
("mailto", f"mailto:{email}"),
|
|
("HIBP", f"https://haveibeenpwned.com/account/{quote(email)}"),
|
|
("Hunter (domain)", f"https://hunter.io/search/{quote(domain)}"),
|
|
("Epieos", f"https://epieos.com/?q={quote(email)}"),
|
|
]
|
|
|
|
|
|
def github_links(org: str, repo: str, sha: str | None) -> list[tuple[str, str]]:
|
|
"""Return (label, url) tuples for a GitHub repo, optionally at a commit."""
|
|
base = f"https://github.com/{org}/{repo}"
|
|
out = [
|
|
("Repo", base),
|
|
("Commits", f"{base}/commits"),
|
|
("Issues", f"{base}/issues"),
|
|
("Secret scan (TruffleHog)", "https://github.com/trufflesecurity/trufflehog"),
|
|
("GH advisories", f"https://github.com/{org}/{repo}/security/advisories"),
|
|
]
|
|
if sha:
|
|
out.insert(1, (f"Commit {sha[:7]}", f"{base}/commit/{sha}"))
|
|
return out
|
|
|
|
|
|
# ──────────────────────────────────────────────────────────────────────────
|
|
# Stats container
|
|
# ──────────────────────────────────────────────────────────────────────────
|
|
@dataclass
|
|
class Stats:
|
|
"""Counts + unique artifact sets for the enrichment appendix."""
|
|
cves: set[str] = field(default_factory=set)
|
|
ipv4s: set[str] = field(default_factory=set)
|
|
cidrs: set[str] = field(default_factory=set)
|
|
hostnames: set[str] = field(default_factory=set)
|
|
emails: set[str] = field(default_factory=set)
|
|
repos: set[tuple[str, str]] = field(default_factory=set) # (org, repo)
|
|
commits: set[tuple[str, str, str]] = field(default_factory=set) # (org, repo, sha)
|
|
urls: set[str] = field(default_factory=set)
|
|
vendors: set[str] = field(default_factory=set)
|
|
|
|
def total(self) -> int:
|
|
return (
|
|
len(self.cves) + len(self.ipv4s) + len(self.cidrs) + len(self.hostnames)
|
|
+ len(self.emails) + len(self.repos) + len(self.commits) + len(self.urls)
|
|
+ len(self.vendors)
|
|
)
|
|
|
|
def as_summary(self) -> dict[str, int]:
|
|
return {
|
|
"CVEs": len(self.cves),
|
|
"IPv4s": len(self.ipv4s),
|
|
"CIDRs": len(self.cidrs),
|
|
"Hostnames": len(self.hostnames),
|
|
"Emails": len(self.emails),
|
|
"GitHub repos": len(self.repos),
|
|
"Git commits": len(self.commits),
|
|
"URLs": len(self.urls),
|
|
"Vendors": len(self.vendors),
|
|
}
|
|
|
|
|
|
# ──────────────────────────────────────────────────────────────────────────
|
|
# Protect/restore code spans + existing links
|
|
# ──────────────────────────────────────────────────────────────────────────
|
|
_PLACEHOLDER = "\x00ENRICH{}\x00"
|
|
|
|
|
|
def _stash(text: str, stats: Stats) -> tuple[str, list[str]]:
|
|
"""Replace code fences / inline code / existing links with placeholders.
|
|
|
|
Returns (modified_text, stash) where placeholders look like \\x00ENRICH<i>\\x00
|
|
so the rewrite passes won't touch them.
|
|
|
|
Inline backticks (`x`) get *special* handling: if `x` is a single, clean
|
|
artifact (CVE, IP, IPv4/CIDR, email, FQDN, org/repo, URL), the WHOLE
|
|
backtick span is replaced with `[\`x\`](url)` so the artifact is clickable
|
|
while still rendering as code. Otherwise the span is stashed unchanged.
|
|
"""
|
|
stash: list[str] = []
|
|
|
|
def stash_repl(m: re.Match[str]) -> str:
|
|
stash.append(m.group(0))
|
|
return _PLACEHOLDER.format(len(stash) - 1)
|
|
|
|
def inline_code_repl(m: re.Match[str]) -> str:
|
|
raw = m.group(0) # e.g. "`devcolor/codebenders-datathon`"
|
|
inner = raw[1:-1].strip() # strip the backticks
|
|
link = _single_artifact_link(inner, stats)
|
|
if link is None:
|
|
# Plain code — keep the original span out of further passes.
|
|
stash.append(raw)
|
|
return _PLACEHOLDER.format(len(stash) - 1)
|
|
# Replace with a markdown link wrapping the code, then stash the
|
|
# whole thing so subsequent passes don't try to re-match the inner.
|
|
replacement = f"[{raw}]({link})"
|
|
stash.append(replacement)
|
|
return _PLACEHOLDER.format(len(stash) - 1)
|
|
|
|
text = FENCED_BLOCK_RE.sub(stash_repl, text)
|
|
text = INLINE_CODE_RE.sub(inline_code_repl, text)
|
|
text = EXISTING_LINK_RE.sub(stash_repl, text)
|
|
return text, stash
|
|
|
|
|
|
def _single_artifact_link(content: str, stats: Stats) -> str | None:
|
|
"""If `content` is exactly one recognisable artifact, return its primary
|
|
enrichment URL and update stats. Otherwise return None.
|
|
"""
|
|
s = content.strip()
|
|
if not s:
|
|
return None
|
|
|
|
# CVE
|
|
if CVE_RE.fullmatch(s):
|
|
cve = s.upper()
|
|
stats.cves.add(cve)
|
|
return f"https://nvd.nist.gov/vuln/detail/{cve}"
|
|
|
|
# URL
|
|
if URL_RE.fullmatch(s):
|
|
url = s.rstrip(".,;:)")
|
|
stats.urls.add(url)
|
|
host_match = re.match(r"https?://([^/:]+)", url, re.IGNORECASE)
|
|
if host_match and _looks_like_hostname(host_match.group(1).lower()):
|
|
stats.hostnames.add(host_match.group(1).lower())
|
|
return url
|
|
|
|
# Email
|
|
if EMAIL_RE.fullmatch(s):
|
|
stats.emails.add(s)
|
|
return f"mailto:{s}"
|
|
|
|
# GitHub org/repo (optionally @sha) — full-string match is strong signal
|
|
# that the user intended it as a repo ref.
|
|
gh = GH_REPO_RE.fullmatch(s)
|
|
if gh:
|
|
org, repo, sha = gh.group(1), gh.group(2), gh.group(3)
|
|
if _looks_like_repo(org, repo) and "." not in org:
|
|
stats.repos.add((org, repo))
|
|
base = f"https://github.com/{org}/{repo}"
|
|
if sha:
|
|
stats.commits.add((org, repo, sha))
|
|
return f"{base}/commit/{sha}"
|
|
return base
|
|
|
|
# IPv4 (with optional CIDR)
|
|
if IPV4_RE.fullmatch(s):
|
|
ip, _, cidr = s.partition("/")
|
|
if cidr:
|
|
stats.cidrs.add(s)
|
|
return f"https://www.shodan.io/search?query=net%3A{quote(s)}"
|
|
if _is_public_ipv4(ip):
|
|
stats.ipv4s.add(ip)
|
|
return f"https://www.shodan.io/host/{ip}"
|
|
return None
|
|
|
|
# Hostname
|
|
if HOSTNAME_RE.fullmatch(s) and _looks_like_hostname(s):
|
|
host = s.lower()
|
|
stats.hostnames.add(host)
|
|
return f"https://{host}"
|
|
|
|
return None
|
|
|
|
|
|
def _restore(text: str, stash: list[str]) -> str:
|
|
for i, original in enumerate(stash):
|
|
text = text.replace(_PLACEHOLDER.format(i), original)
|
|
return text
|
|
|
|
|
|
# ──────────────────────────────────────────────────────────────────────────
|
|
# Hostname filtering — skip noisy false positives
|
|
# ──────────────────────────────────────────────────────────────────────────
|
|
_HOST_SKIP_TLDS = {
|
|
# Common file extensions that look like TLDs in our regex.
|
|
"md", "txt", "log", "csv", "json", "yml", "yaml", "ini", "cfg",
|
|
"py", "js", "ts", "go", "rs", "java", "html", "css", "xml", "sh",
|
|
"ps1", "bat", "exe", "dll", "so", "dylib", "png", "jpg", "jpeg",
|
|
"gif", "svg", "pdf", "zip", "tar", "gz", "rar", "7z", "iso",
|
|
"bak", "old", "tmp", "swp", "db", "sql",
|
|
# Common code/JS identifiers mis-detected as TLDs.
|
|
"redirect", "then", "catch", "call", "apply", "bind",
|
|
"prototype", "constructor", "length", "value", "name",
|
|
}
|
|
|
|
_REPO_SKIP_SUFFIXES = (
|
|
".py", ".js", ".ts", ".java", ".rb", ".go", ".rs", ".c", ".cpp",
|
|
".h", ".hpp", ".cs", ".php", ".html", ".css", ".json", ".yml",
|
|
".yaml", ".xml", ".sh", ".md", ".txt", ".log", ".sql", ".bak",
|
|
)
|
|
|
|
|
|
# Recognised gTLDs / common ccTLDs. Anything outside this set OR a 2-letter
|
|
# all-alpha ccTLD gets rejected as not-a-real-hostname (catches JS method
|
|
# names like jquery.parseHtml, internal aliases like kctcsfin.ps, etc.).
|
|
_KNOWN_TLDS = {
|
|
# gTLDs
|
|
"com", "net", "org", "edu", "gov", "mil", "int", "info", "biz",
|
|
"name", "pro", "mobi", "aero", "coop", "museum",
|
|
# Modern / brand
|
|
"io", "co", "dev", "app", "ai", "cloud", "tech", "xyz", "online",
|
|
"site", "store", "shop", "blog", "page", "wiki", "news", "today",
|
|
"news", "live", "world", "group", "club", "fun", "studio", "design",
|
|
"art", "video", "tv", "fm", "gg", "ly", "sh", "me", "to", "vc",
|
|
"cc", "ws", "asia", "jobs", "travel", "post", "tel", "cat",
|
|
# Software / infra
|
|
"software", "services", "agency", "network", "social", "life",
|
|
"academy", "consulting", "solutions", "systems", "support", "ventures",
|
|
"finance", "money", "security", "wtf",
|
|
}
|
|
|
|
|
|
def _looks_like_hostname(host: str) -> bool:
|
|
"""Heuristic: must look like a real internet hostname, not a code
|
|
identifier (`jquery.parseHtml`) or internal alias (`kctcsfin.ps`).
|
|
"""
|
|
host = host.strip(".")
|
|
if "." not in host:
|
|
return False
|
|
parts = host.split(".")
|
|
if any(p == "" for p in parts):
|
|
return False
|
|
tld = parts[-1].lower()
|
|
if tld in _HOST_SKIP_TLDS:
|
|
return False
|
|
if not tld.isalpha():
|
|
return False
|
|
# TLD must be either a known gTLD/brand OR a 2-letter all-letter ccTLD.
|
|
if not (tld in _KNOWN_TLDS or len(tld) == 2):
|
|
return False
|
|
# Reject if any segment has internal mixed case (camelCase identifier).
|
|
for seg in parts:
|
|
if re.search(r"[a-z][A-Z]", seg):
|
|
return False
|
|
return True
|
|
|
|
|
|
def _looks_like_repo(org: str, repo: str) -> bool:
|
|
"""Heuristic for GH org/repo refs (used by both inline and backtick passes)."""
|
|
if not (re.search(r"[A-Za-z]", org) and re.search(r"[A-Za-z]", repo)):
|
|
return False
|
|
if org.isdigit() or repo.isdigit():
|
|
return False
|
|
rl = repo.lower()
|
|
if any(rl.endswith(suf) for suf in _REPO_SKIP_SUFFIXES):
|
|
return False
|
|
return True
|
|
|
|
|
|
def _is_public_ipv4(ip: str) -> bool:
|
|
"""True if the IP is globally routable (skip RFC1918, loopback, etc.)."""
|
|
try:
|
|
addr = ipaddress.IPv4Address(ip)
|
|
except (ipaddress.AddressValueError, ValueError):
|
|
return False
|
|
return not (addr.is_private or addr.is_loopback or addr.is_link_local
|
|
or addr.is_multicast or addr.is_unspecified or addr.is_reserved)
|
|
|
|
|
|
# ──────────────────────────────────────────────────────────────────────────
|
|
# Inline rewrite passes
|
|
# ──────────────────────────────────────────────────────────────────────────
|
|
def _rewrite_cves(text: str, stats: Stats) -> str:
|
|
def repl(m: re.Match[str]) -> str:
|
|
cve = m.group(0).upper()
|
|
stats.cves.add(cve)
|
|
return f"[{cve}](https://nvd.nist.gov/vuln/detail/{cve})"
|
|
return CVE_RE.sub(repl, text)
|
|
|
|
|
|
def _rewrite_ipv4s(text: str, stats: Stats) -> str:
|
|
def repl(m: re.Match[str]) -> str:
|
|
raw = m.group(0)
|
|
ip, _, cidr = raw.partition("/")
|
|
if cidr:
|
|
stats.cidrs.add(raw)
|
|
# Link CIDR to Shodan net: query.
|
|
return f"[{raw}](https://www.shodan.io/search?query=net%3A{quote(raw)})"
|
|
if not _is_public_ipv4(ip):
|
|
return raw
|
|
stats.ipv4s.add(ip)
|
|
return f"[{ip}](https://www.shodan.io/host/{ip})"
|
|
return IPV4_RE.sub(repl, text)
|
|
|
|
|
|
def _rewrite_emails(text: str, stats: Stats) -> str:
|
|
def repl(m: re.Match[str]) -> str:
|
|
email = m.group(0)
|
|
stats.emails.add(email)
|
|
return f"[{email}](mailto:{email})"
|
|
return EMAIL_RE.sub(repl, text)
|
|
|
|
|
|
def _rewrite_urls(text: str, stats: Stats) -> str:
|
|
def repl(m: re.Match[str]) -> str:
|
|
url = m.group(0).rstrip(".,;:)")
|
|
trailing = m.group(0)[len(url):]
|
|
stats.urls.add(url)
|
|
# Extract hostname for the appendix too.
|
|
host_match = re.match(r"https?://([^/:]+)", url, re.IGNORECASE)
|
|
if host_match:
|
|
host = host_match.group(1).lower()
|
|
if _looks_like_hostname(host):
|
|
stats.hostnames.add(host)
|
|
return f"[{url}]({url}){trailing}"
|
|
return URL_RE.sub(repl, text)
|
|
|
|
|
|
def _rewrite_hostnames(text: str, stats: Stats) -> str:
|
|
def repl(m: re.Match[str]) -> str:
|
|
host = m.group(0).lower()
|
|
if not _looks_like_hostname(host):
|
|
return m.group(0)
|
|
stats.hostnames.add(host)
|
|
return f"[{host}](https://{host})"
|
|
return HOSTNAME_RE.sub(repl, text)
|
|
|
|
|
|
def _rewrite_github(text: str, stats: Stats) -> str:
|
|
"""Rewrite `org/repo` and `org/repo@sha` refs as GitHub links.
|
|
|
|
`org/repo` is an extremely ambiguous shape ("voice/data", "OWA/Exchange",
|
|
"443/tcp", "Dev/test"). To avoid false positives we require ALL of:
|
|
- both segments contain at least one letter
|
|
- neither segment is a known protocol/port-style stopword
|
|
- both segments contain a hyphen (a strong GitHub-style convention), AND
|
|
- a context word (github / git history / repo / commit) appears within
|
|
80 chars before the match.
|
|
Single-hyphen / no-hyphen refs that are genuine repos (e.g. `shadcn/ui` or
|
|
`devcolor/codebenders-datathon`) are expected to appear in backticks in
|
|
the source; those are already handled by the inline-code pass in _stash().
|
|
"""
|
|
org_skip = {"http", "https", "ftp", "mailto", "tcp", "udp", "icmp"}
|
|
|
|
def repl(m: re.Match[str]) -> str:
|
|
org, repo, sha = m.group(1), m.group(2), m.group(3)
|
|
if org.lower() in org_skip or repo.lower() in org_skip:
|
|
return m.group(0)
|
|
if not _looks_like_repo(org, repo):
|
|
return m.group(0)
|
|
if not ("-" in org and "-" in repo):
|
|
return m.group(0)
|
|
start = m.start()
|
|
window = text[max(0, start - 80):start].lower()
|
|
if not re.search(r"github|git history|\bgit \b|\brepo\b|\bcommit\b", window):
|
|
return m.group(0)
|
|
stats.repos.add((org, repo))
|
|
base = f"https://github.com/{org}/{repo}"
|
|
if sha:
|
|
stats.commits.add((org, repo, sha))
|
|
return f"[{org}/{repo}@{sha[:7]}]({base}/commit/{sha})"
|
|
return f"[{org}/{repo}]({base})"
|
|
return GH_REPO_RE.sub(repl, text)
|
|
|
|
|
|
def _rewrite_vendors(text: str, stats: Stats) -> str:
|
|
"""Wrap vendor/product mentions with their docs link (first occurrence only).
|
|
|
|
To avoid noisy repetition, we only link the FIRST hit per vendor in the
|
|
document; subsequent occurrences stay plain text.
|
|
"""
|
|
seen: set[str] = set()
|
|
for pat, display, url in _VENDOR_PATTERNS:
|
|
if display in seen:
|
|
continue
|
|
# Only the first occurrence: use sub with count=1 and skip if already done.
|
|
def repl(m: re.Match[str], _disp=display, _url=url) -> str:
|
|
stats.vendors.add(_disp)
|
|
seen.add(_disp)
|
|
return f"[{m.group(0)}]({_url})"
|
|
text, n = pat.subn(repl, text, count=1)
|
|
if n:
|
|
seen.add(display)
|
|
return text
|
|
|
|
|
|
# ──────────────────────────────────────────────────────────────────────────
|
|
# Appendix builder
|
|
# ──────────────────────────────────────────────────────────────────────────
|
|
def _link_orphan_commits(text: str, stats: Stats) -> str:
|
|
"""After other passes have run, find lines that contain a GitHub repo
|
|
link plus one or more unlinked backticked SHAs, and link those SHAs to
|
|
that repo's commit URL.
|
|
|
|
Matches both `[\`abc1234\`]` (already wrapped via single-artifact pass that
|
|
failed because no repo context) and a bare ` `abc1234` ` code span.
|
|
"""
|
|
repo_link_re = re.compile(r"https?://github\.com/([\w.-]+)/([\w.-]+?)(?:[)/#?]|$)")
|
|
sha_code_re = re.compile(r"`([0-9a-f]{7,40})`")
|
|
|
|
out_lines: list[str] = []
|
|
for line in text.splitlines(keepends=True):
|
|
m = repo_link_re.search(line)
|
|
if not m:
|
|
out_lines.append(line)
|
|
continue
|
|
org, repo = m.group(1), m.group(2)
|
|
|
|
def repl(sm: re.Match[str], _org=org, _repo=repo) -> str:
|
|
sha = sm.group(1)
|
|
stats.commits.add((_org, _repo, sha))
|
|
return f"[`{sha}`](https://github.com/{_org}/{_repo}/commit/{sha})"
|
|
|
|
out_lines.append(sha_code_re.sub(repl, line))
|
|
return "".join(out_lines)
|
|
|
|
|
|
def _format_link_row(label: str, links: list[tuple[str, str]]) -> str:
|
|
bits = " · ".join(f"[{name}]({url})" for name, url in links)
|
|
return f"- **{label}** — {bits}"
|
|
|
|
|
|
def _build_appendix(stats: Stats) -> str:
|
|
if stats.total() == 0:
|
|
return ""
|
|
|
|
lines: list[str] = []
|
|
lines.append("\n---\n")
|
|
lines.append("## INTEL APPENDIX — Artifact Enrichment\n")
|
|
lines.append(
|
|
"_Auto-generated by VulnForge enricher. Each artifact below is paired with "
|
|
"external recon / threat-intel lookups for rapid pivot._\n"
|
|
)
|
|
|
|
if stats.cves:
|
|
lines.append("### CVEs\n")
|
|
for cve in sorted(stats.cves):
|
|
lines.append(_format_link_row(cve, cve_links(cve)))
|
|
lines.append("")
|
|
|
|
if stats.ipv4s:
|
|
lines.append("### IPv4 Hosts\n")
|
|
for ip in sorted(stats.ipv4s, key=lambda x: tuple(int(p) for p in x.split("."))):
|
|
lines.append(_format_link_row(ip, ipv4_links(ip)))
|
|
lines.append("")
|
|
|
|
if stats.cidrs:
|
|
lines.append("### IPv4 Ranges (CIDR)\n")
|
|
for cidr in sorted(stats.cidrs):
|
|
net_q = quote(cidr)
|
|
links = [
|
|
("Shodan net:", f"https://www.shodan.io/search?query=net%3A{net_q}"),
|
|
("Censys", f"https://search.censys.io/search?resource=hosts&q=ip%3A%5B{quote(cidr.split('/')[0])}%2F{cidr.split('/')[1]}%5D"),
|
|
("BGP.tools", f"https://bgp.tools/prefix/{net_q}"),
|
|
]
|
|
lines.append(_format_link_row(cidr, links))
|
|
lines.append("")
|
|
|
|
if stats.hostnames:
|
|
lines.append("### Hostnames / Domains\n")
|
|
for host in sorted(stats.hostnames):
|
|
lines.append(_format_link_row(host, hostname_links(host)))
|
|
lines.append("")
|
|
|
|
if stats.emails:
|
|
lines.append("### Emails\n")
|
|
for email in sorted(stats.emails):
|
|
lines.append(_format_link_row(email, email_links(email)))
|
|
lines.append("")
|
|
|
|
if stats.repos or stats.commits:
|
|
lines.append("### Source Code (GitHub)\n")
|
|
# First emit repos; commits get their own bullet under the parent repo.
|
|
repo_commits: dict[tuple[str, str], list[str]] = {}
|
|
for org, repo, sha in stats.commits:
|
|
repo_commits.setdefault((org, repo), []).append(sha)
|
|
for org, repo in sorted(stats.repos):
|
|
lines.append(_format_link_row(f"{org}/{repo}", github_links(org, repo, None)))
|
|
for sha in sorted(repo_commits.get((org, repo), [])):
|
|
commit_url = f"https://github.com/{org}/{repo}/commit/{sha}"
|
|
lines.append(f" - commit [`{sha[:12]}`]({commit_url})")
|
|
# Orphan commits (commit seen but repo wasn't matched) — rare.
|
|
for (org, repo), shas in repo_commits.items():
|
|
if (org, repo) not in stats.repos:
|
|
lines.append(_format_link_row(f"{org}/{repo}", github_links(org, repo, None)))
|
|
for sha in sorted(shas):
|
|
commit_url = f"https://github.com/{org}/{repo}/commit/{sha}"
|
|
lines.append(f" - commit [`{sha[:12]}`]({commit_url})")
|
|
lines.append("")
|
|
|
|
if stats.vendors:
|
|
lines.append("### Vendors / Products\n")
|
|
for vendor in sorted(stats.vendors):
|
|
# Find the URL we used for this vendor.
|
|
for pat, disp, url in _VENDOR_PATTERNS:
|
|
if disp == vendor:
|
|
lines.append(f"- **{vendor}** — [docs]({url})")
|
|
break
|
|
lines.append("")
|
|
|
|
return "\n".join(lines)
|
|
|
|
|
|
# ──────────────────────────────────────────────────────────────────────────
|
|
# Public entrypoint
|
|
# ──────────────────────────────────────────────────────────────────────────
|
|
def enrich_markdown(text: str) -> tuple[str, Stats]:
|
|
"""Rewrite artifacts inline and append an INTEL APPENDIX.
|
|
|
|
Code fences, inline code, and existing markdown links are preserved.
|
|
Returns the new text and a Stats object describing what was found.
|
|
"""
|
|
stats = Stats()
|
|
body, stash = _stash(text, stats)
|
|
|
|
# Order matters: URLs before hostnames (URL pass also collects hostnames
|
|
# from its URL strings into stats so they appear in the appendix). CVEs
|
|
# before any other text replacement so the "CVE-..." literal becomes the
|
|
# link text. GitHub repo pass before hostnames so "org/repo" isn't eaten.
|
|
passes: list[Callable[[str, Stats], str]] = [
|
|
_rewrite_urls,
|
|
_rewrite_cves,
|
|
_rewrite_emails,
|
|
_rewrite_github,
|
|
_rewrite_ipv4s,
|
|
_rewrite_hostnames,
|
|
_rewrite_vendors,
|
|
]
|
|
for fn in passes:
|
|
body = fn(body, stats)
|
|
|
|
body = _restore(body, stash)
|
|
# Post-pass: any backticked commit SHA on a line that also contains a
|
|
# github.com/<org>/<repo> link gets rewritten to that commit URL.
|
|
body = _link_orphan_commits(body, stats)
|
|
appendix = _build_appendix(stats)
|
|
return body + appendix, stats
|