diff --git a/redflare/core/models.py b/redflare/core/models.py
new file mode 100644
index 0000000..a8047d1
--- /dev/null
+++ b/redflare/core/models.py
@@ -0,0 +1,71 @@
+from __future__ import annotations
+
+from dataclasses import asdict, dataclass, field
+from datetime import datetime, timezone
+from hashlib import sha256
+import json
+from typing import Any
+
+
+def utc_now() -> str:
+ return datetime.now(timezone.utc).isoformat()
+
+
+@dataclass(frozen=True)
+class Target:
+ url: str
+ host: str
+ scheme: str
+ port: int
+
+ @property
+ def authority(self) -> str:
+ default = (self.scheme == "http" and self.port == 80) or (
+ self.scheme == "https" and self.port == 443
+ )
+ return self.host if default else f"{self.host}:{self.port}"
+
+
+@dataclass
+class Finding:
+ run_id: str
+ target: str
+ module: str
+ category: str
+ title: str
+ severity: str
+ confidence: float
+ description: str
+ evidence: dict[str, Any] = field(default_factory=dict)
+ tags: list[str] = field(default_factory=list)
+ timestamp: str = field(default_factory=utc_now)
+ id: str = ""
+ test_id: str = ""
+ standards: dict[str, list[dict[str, str]]] = field(default_factory=dict)
+ remediation: str = ""
+
+ def __post_init__(self) -> None:
+ if not self.id:
+ evidence = json.dumps(self.evidence, sort_keys=True, default=str)
+ material = "\x00".join([self.target, self.module, self.category, self.title, evidence])
+ self.id = sha256(material.encode()).hexdigest()[:16]
+
+ def to_dict(self) -> dict[str, Any]:
+ return asdict(self)
+
+
+@dataclass
+class ModuleResult:
+ module: str
+ target: str
+ status: str = "completed"
+ findings: list[Finding] = field(default_factory=list)
+ observations: dict[str, Any] = field(default_factory=dict)
+ artifacts: list[str] = field(default_factory=list)
+ errors: list[str] = field(default_factory=list)
+ duration_seconds: float = 0.0
+
+ def to_dict(self) -> dict[str, Any]:
+ value = asdict(self)
+ value["findings"] = [finding.to_dict() for finding in self.findings]
+ return value
diff --git a/redflare/core/runner.py b/redflare/core/runner.py
new file mode 100644
index 0000000..e0758f2
--- /dev/null
+++ b/redflare/core/runner.py
@@ -0,0 +1,82 @@
+from __future__ import annotations
+
+from concurrent.futures import ThreadPoolExecutor, as_completed
+from dataclasses import asdict
+from datetime import datetime, timezone
+
+from redflare.modules.base import Module, ModuleContext
+from .correlation import correlate
+from .models import ModuleResult, Target
+from .standards import enrich_finding, registry_document
+from .storage import RunStore
+
+
+class Runner:
+ def __init__(self, store: RunStore, modules: list[Module], context: ModuleContext, workers: int = 2):
+ self.store = store
+ self.modules = modules
+ self.context = context
+ self.workers = max(1, workers)
+
+ def run(self, targets: list[Target]) -> tuple[list[ModuleResult], dict]:
+ self.store.write_manifest(
+ {
+ "run_id": self.store.run_id,
+ "created_at": datetime.now(timezone.utc).isoformat(),
+ "targets": [asdict(target) for target in targets],
+ "modules": [module.name for module in self.modules],
+ "authorized_acknowledgement": True,
+ }
+ )
+ results: list[ModuleResult] = []
+ with ThreadPoolExecutor(max_workers=self.workers) as executor:
+ future_map = {
+ executor.submit(self._run_target_pipeline, target): target for target in targets
+ }
+ for future in as_completed(future_map):
+ target = future_map[future]
+ try:
+ target_results = future.result()
+ except Exception as exc:
+ target_results = [ModuleResult(
+ "pipeline",
+ target.url,
+ status="error",
+ errors=[f"{type(exc).__name__}: {exc}"],
+ )]
+ for result in target_results:
+ for finding in result.findings:
+ enrich_finding(finding)
+ results.append(result)
+ self.store.write_result(result)
+ correlated = correlate(self.store.run_id, results)
+ for result in correlated:
+ for finding in result.findings:
+ enrich_finding(finding)
+ results.append(result)
+ self.store.write_result(result)
+ surface_graph = self.context.surface_graph.snapshot()
+ self.store.write_surface_graph(surface_graph)
+ self.store.write_test_registry(registry_document())
+ return results, self.store.finalize(results, surface_graph=surface_graph)
+
+ def _run_target_pipeline(self, target: Target) -> list[ModuleResult]:
+ results = []
+ for module in self.modules:
+ self.context.emit(target.url, module.name, "start", module.description)
+ try:
+ result = module.run(target, self.context)
+ except Exception as exc:
+ result = ModuleResult(
+ module.name,
+ target.url,
+ status="error",
+ errors=[f"{type(exc).__name__}: {exc}"],
+ )
+ kind = "success" if result.status == "completed" else result.status
+ message = f"finished in {result.duration_seconds:.2f}s; findings={len(result.findings)}"
+ if result.errors:
+ message += f"; {result.errors[-1]}"
+ self.context.emit(target.url, module.name, kind, message)
+ results.append(result)
+ return results
diff --git a/redflare/core/scope.py b/redflare/core/scope.py
new file mode 100644
index 0000000..b800cd0
--- /dev/null
+++ b/redflare/core/scope.py
@@ -0,0 +1,68 @@
+from __future__ import annotations
+
+import ipaddress
+import json
+import socket
+from dataclasses import dataclass, field
+from pathlib import Path
+from urllib.parse import urlparse
+
+from .models import Target
+
+
+class ScopeError(ValueError):
+ pass
+
+
+@dataclass
+class ScopePolicy:
+ allowed_hosts: set[str] = field(default_factory=set)
+ allow_public: bool = False
+
+ @classmethod
+ def from_file(cls, path: str | None, allow_public: bool = False) -> "ScopePolicy":
+ if not path:
+ return cls(allow_public=allow_public)
+ data = json.loads(Path(path).read_text(encoding="utf-8"))
+ return cls(
+ allowed_hosts={str(host).lower() for host in data.get("allowed_hosts", [])},
+ allow_public=allow_public or bool(data.get("allow_public", False)),
+ )
+
+ def validate(self, target: Target) -> None:
+ host = target.host.lower()
+ if self.allowed_hosts and host not in self.allowed_hosts:
+ raise ScopeError(f"{host} is not listed in allowed_hosts")
+ if not self.allow_public and is_public_host(host):
+ raise ScopeError(
+ f"{host} resolves to public address space; use --allow-public only for authorized scope"
+ )
+
+
+def normalize_target(value: str) -> Target:
+ value = value.strip()
+ if not value:
+ raise ScopeError("target cannot be empty")
+ if "://" not in value:
+ value = "https://" + value
+ parsed = urlparse(value)
+ if parsed.scheme not in {"http", "https"} or not parsed.hostname:
+ raise ScopeError(f"unsupported target: {value}")
+ port = parsed.port or (443 if parsed.scheme == "https" else 80)
+ normalized = f"{parsed.scheme}://{parsed.hostname}"
+ if port not in {80, 443}:
+ normalized += f":{port}"
+ normalized += parsed.path.rstrip("/") or ""
+ return Target(normalized, parsed.hostname, parsed.scheme, port)
+
+
+def is_public_host(host: str) -> bool:
+ try:
+ addresses = {item[4][0] for item in socket.getaddrinfo(host, None)}
+ except socket.gaierror as exc:
+ raise ScopeError(f"cannot resolve {host}: {exc}") from exc
+ for address in addresses:
+ ip = ipaddress.ip_address(address)
+ if not (ip.is_private or ip.is_loopback or ip.is_link_local):
+ return True
+ return False
diff --git a/redflare/core/storage.py b/redflare/core/storage.py
new file mode 100644
index 0000000..8822346
--- /dev/null
+++ b/redflare/core/storage.py
@@ -0,0 +1,205 @@
+from __future__ import annotations
+
+import html
+import json
+import re
+from dataclasses import replace
+from datetime import datetime
+from pathlib import Path
+from typing import Iterable
+
+from .models import Finding, ModuleResult
+
+
+def target_run_id(targets: Iterable[object]) -> str:
+ values = list(targets)
+ first = values[0] if values else None
+ authority = str(getattr(first, "authority", None) or getattr(first, "host", None) or "assessment")
+ slug = re.sub(r"[^A-Za-z0-9._-]+", "_", authority).strip("._-") or "assessment"
+ extra = f"_plus_{len(values) - 1}" if len(values) > 1 else ""
+ stamp = datetime.now().strftime("%Y%m%d_%H%M%S")
+ return f"scan_{slug}{extra}_{stamp}"
+
+
+class RunStore:
+ def __init__(self, base: str = "runs", run_id: str | None = None):
+ stamp = datetime.now().strftime("%Y%m%d_%H%M%S")
+ self.run_id = run_id or f"run_{stamp}"
+ root = Path(base)
+ candidate = root / self.run_id
+ sequence = 1
+ while candidate.exists():
+ candidate = root / f"{self.run_id}_{sequence:02d}"
+ sequence += 1
+ self.root = candidate
+ self.modules = self.root / "modules"
+ self.artifacts = self.root / "artifacts"
+ self.modules.mkdir(parents=True)
+ self.artifacts.mkdir(parents=True)
+ self.findings_file = self.root / "findings.jsonl"
+ self.findings_file.touch()
+
+ def write_manifest(self, manifest: dict) -> None:
+ self._write_json(self.root / "manifest.json", manifest)
+
+ def write_result(self, result: ModuleResult) -> None:
+ safe_target = result.target.replace("://", "_").replace("/", "_").replace(":", "_")
+ self._write_json(self.modules / f"{safe_target}__{result.module}.json", result.to_dict())
+ with self.findings_file.open("a", encoding="utf-8") as handle:
+ for finding in result.findings:
+ handle.write(json.dumps(finding.to_dict(), sort_keys=True) + "\n")
+
+ def write_surface_graph(self, graph: dict) -> None:
+ self._write_json(self.root / "attack_surface.json", graph)
+
+ def write_test_registry(self, registry: dict) -> None:
+ self._write_json(self.root / "test_registry.json", registry)
+
+ def finalize(self, results: Iterable[ModuleResult], surface_graph: dict | None = None) -> dict:
+ results = list(results)
+ findings = deduplicate_findings(
+ [finding for result in results for finding in result.findings]
+ )
+ with self.findings_file.open("w", encoding="utf-8") as handle:
+ for finding in findings:
+ handle.write(json.dumps(finding.to_dict(), sort_keys=True) + "\n")
+ summary = {
+ "run_id": self.run_id,
+ "module_results": len(results),
+ "completed": sum(result.status == "completed" for result in results),
+ "errors": sum(result.status == "error" for result in results),
+ "findings": len(findings),
+ "by_severity": count_by(finding.severity for finding in findings),
+ "by_module": count_by(finding.module for finding in findings),
+ "by_category": count_by(finding.category for finding in findings),
+ "sensitive_exposures": sum(
+ finding.category == "sensitive-data-exposure" for finding in findings
+ ),
+ "known_cves": sum(
+ finding.category == "known-vulnerable-component" for finding in findings
+ ),
+ "attack_surface": (surface_graph or {}).get("summary", {}),
+ }
+ self._write_json(self.root / "summary.json", summary)
+ (self.root / "report.html").write_text(
+ render_html(summary, findings, results), encoding="utf-8"
+ )
+ return summary
+
+ @staticmethod
+ def _write_json(path: Path, value: dict) -> None:
+ path.write_text(json.dumps(value, indent=2, sort_keys=True), encoding="utf-8")
+
+
+def count_by(values: Iterable[str]) -> dict[str, int]:
+ counts: dict[str, int] = {}
+ for value in values:
+ counts[value] = counts.get(value, 0) + 1
+ return dict(sorted(counts.items()))
+
+
+def deduplicate_findings(findings: list[Finding]) -> list[Finding]:
+ canonical_categories = {
+ "browser-security-headers": "security-headers",
+ }
+ selected: dict[tuple[str, str], Finding] = {}
+ sources: dict[tuple[str, str], set[str]] = {}
+ for finding in findings:
+ category = canonical_categories.get(finding.category, finding.category)
+ key = (
+ finding.target,
+ category if category == "security-headers" else finding.id,
+ )
+ sources.setdefault(key, set()).add(finding.module)
+ current = selected.get(key)
+ if current is None or finding.confidence > current.confidence:
+ selected[key] = replace(
+ finding,
+ category=category,
+ evidence=dict(finding.evidence),
+ )
+ for key, finding in selected.items():
+ modules = sorted(sources[key])
+ if len(modules) > 1:
+ finding.evidence["corroborated_by"] = modules
+ finding.tags = sorted(set(finding.tags + ["corroborated"]))
+ return list(selected.values())
+
+
+def render_html(
+ summary: dict,
+ findings: list[Finding],
+ results: list[ModuleResult] | None = None,
+) -> str:
+ rows = []
+ for finding in sorted(findings, key=lambda item: (item.severity, item.target), reverse=True):
+ rows.append(
+ "
"
+ f"| {html.escape(finding.severity)} | "
+ f"{html.escape(finding.test_id)} | "
+ f"{html.escape(finding.module)} | "
+ f"{html.escape(finding.target)} | "
+ f"{html.escape(finding.title)} | "
+ f"{html.escape(finding.description)} | "
+ "
"
+ )
+ assessments = []
+ for result in sorted(results or [], key=lambda item: (item.target, item.module)):
+ module_findings = "".join(
+ ""
+ f"[{html.escape(finding.severity.upper())}] {html.escape(finding.title)}
"
+ f"Test ID: {html.escape(finding.test_id or 'unmapped')}
"
+ f"{html.escape(finding.description)}
"
+ f"{render_reference_links(finding)}"
+ f"{html.escape(json.dumps(finding.evidence, indent=2, default=str))}"
+ f"Remediation: {html.escape(finding.remediation or 'Review and remediate based on verified impact.')}
"
+ ""
+ for finding in result.findings
+ ) or "No module findings.
"
+ errors = "".join(f"{html.escape(error)}" for error in result.errors) or "None"
+ artifacts = "".join(
+ f"{html.escape(artifact)}" for artifact in result.artifacts
+ ) or "None"
+ assessments.append(
+ ""
+ f"{html.escape(result.module.replace('_', ' ').upper())} ASSESSMENT
"
+ f"Target: {html.escape(result.target)}
"
+ f"Status: {html.escape(result.status.upper())} "
+ f"Duration: {result.duration_seconds:.2f}s "
+ f"Findings: {len(result.findings)}
"
+ "Complete observations
"
+ f"{html.escape(json.dumps(result.observations, indent=2, default=str))}"
+ f"Module findings
{module_findings}"
+ f"Errors
"
+ f"Artifacts
"
+ ""
+ )
+ return f"""REDflare {html.escape(summary['run_id'])}
+
+REDflare Final Assessment Report
Run summary
{html.escape(json.dumps(summary, indent=2))}
+{''.join(assessments)}
+Assessment artifacts
attack_surface.jsontest_registry.json
+Consolidated findings
| Severity | Test ID | Module | Target | Finding | Description |
{''.join(rows)}
"""
+
+
+def render_reference_links(finding: Finding) -> str:
+ links = []
+ seen = set()
+ for family, references in finding.standards.items():
+ for reference in references:
+ url = str(reference.get("url") or "")
+ if not url or url in seen:
+ continue
+ seen.add(url)
+ label = str(reference.get("id") or family)
+ links.append(
+ f''
+ f'{html.escape(label)}'
+ )
+ return f"References: {' · '.join(links)}
" if links else ""
diff --git a/redflare/core/surface_graph.py b/redflare/core/surface_graph.py
new file mode 100644
index 0000000..6b98907
--- /dev/null
+++ b/redflare/core/surface_graph.py
@@ -0,0 +1,205 @@
+from __future__ import annotations
+
+from dataclasses import dataclass, field
+import re
+from threading import RLock
+from typing import Any
+from urllib.parse import parse_qsl, urlencode, urlsplit, urlunsplit
+
+
+def canonical_url(url: str) -> str:
+ parsed = urlsplit(url)
+ path = parsed.path or "/"
+ return urlunsplit((parsed.scheme.lower(), parsed.netloc.lower(), path, "", ""))
+
+
+@dataclass
+class SurfaceParameter:
+ name: str
+ location: str = "unknown"
+ required: bool = False
+ data_type: str = "unknown"
+ sources: set[str] = field(default_factory=set)
+
+ def merge(self, *, required: bool, data_type: str, source: str) -> None:
+ self.required = self.required or required
+ if self.data_type == "unknown" and data_type:
+ self.data_type = data_type
+ self.sources.add(source)
+
+ def to_dict(self) -> dict[str, Any]:
+ return {
+ "name": self.name,
+ "location": self.location,
+ "required": self.required,
+ "data_type": self.data_type,
+ "sources": sorted(self.sources),
+ }
+
+
+@dataclass
+class SurfaceEndpoint:
+ url: str
+ observed_requests: set[tuple[str, str]] = field(default_factory=set)
+ methods: set[str] = field(default_factory=set)
+ content_types: set[str] = field(default_factory=set)
+ authentication: set[str] = field(default_factory=set)
+ sources: set[str] = field(default_factory=set)
+ status_codes: set[int] = field(default_factory=set)
+ parameters: dict[str, SurfaceParameter] = field(default_factory=dict)
+ metadata: dict[str, Any] = field(default_factory=dict)
+
+ def to_dict(self) -> dict[str, Any]:
+ return {
+ "url": self.url,
+ "observed_urls": sorted({redact_url(value) for _, value in self.observed_requests}),
+ "methods": sorted(self.methods),
+ "content_types": sorted(self.content_types),
+ "authentication": sorted(self.authentication),
+ "sources": sorted(self.sources),
+ "status_codes": sorted(self.status_codes),
+ "parameters": [
+ item.to_dict()
+ for item in sorted(self.parameters.values(), key=lambda value: (value.location, value.name))
+ ],
+ "metadata": self.metadata,
+ }
+
+
+class AttackSurfaceGraph:
+ """Thread-safe endpoint inventory shared by every module in a run."""
+
+ def __init__(self) -> None:
+ self._lock = RLock()
+ self._endpoints: dict[str, dict[str, SurfaceEndpoint]] = {}
+ self._edges: dict[str, set[tuple[str, str, str]]] = {}
+ self._documents: dict[str, list[dict[str, Any]]] = {}
+
+ def add_endpoint(
+ self,
+ target: str,
+ url: str,
+ *,
+ method: str = "GET",
+ source: str,
+ content_type: str | None = None,
+ authentication: str | None = None,
+ status: int | None = None,
+ parameters: list[dict[str, Any]] | None = None,
+ metadata: dict[str, Any] | None = None,
+ ) -> SurfaceEndpoint:
+ normalized = canonical_url(url)
+ with self._lock:
+ endpoint = self._endpoints.setdefault(target, {}).setdefault(
+ normalized, SurfaceEndpoint(normalized)
+ )
+ endpoint.methods.add(method.upper())
+ endpoint.observed_requests.add(
+ (method.upper(), urlunsplit((*urlsplit(url)[:4], "")))
+ )
+ endpoint.sources.add(source)
+ if content_type:
+ endpoint.content_types.add(content_type.split(";", 1)[0].strip().lower())
+ if authentication:
+ endpoint.authentication.add(authentication)
+ if status is not None:
+ endpoint.status_codes.add(int(status))
+ for name, _ in parse_qsl(urlsplit(url).query, keep_blank_values=True):
+ self._merge_parameter(endpoint, name, "query", False, "string", source)
+ for parameter in parameters or []:
+ name = str(parameter.get("name") or "").strip()
+ if name:
+ self._merge_parameter(
+ endpoint,
+ name,
+ str(parameter.get("location") or "unknown"),
+ bool(parameter.get("required", False)),
+ str(parameter.get("data_type") or parameter.get("type") or "unknown"),
+ source,
+ )
+ if metadata:
+ endpoint.metadata.update(metadata)
+ return endpoint
+
+ @staticmethod
+ def _merge_parameter(
+ endpoint: SurfaceEndpoint,
+ name: str,
+ location: str,
+ required: bool,
+ data_type: str,
+ source: str,
+ ) -> None:
+ key = f"{location}:{name}"
+ parameter = endpoint.parameters.setdefault(key, SurfaceParameter(name, location))
+ parameter.merge(required=required, data_type=data_type, source=source)
+
+ def add_edge(self, target: str, source_url: str, destination_url: str, relation: str) -> None:
+ with self._lock:
+ self._edges.setdefault(target, set()).add(
+ (canonical_url(source_url), canonical_url(destination_url), relation)
+ )
+
+ def add_document(self, target: str, document: dict[str, Any]) -> None:
+ with self._lock:
+ documents = self._documents.setdefault(target, [])
+ identity = (document.get("kind"), document.get("url"))
+ if not any((item.get("kind"), item.get("url")) == identity for item in documents):
+ documents.append(document)
+
+ def endpoint_urls(self, target: str) -> list[str]:
+ with self._lock:
+ return sorted(self._endpoints.get(target, {}))
+
+ def request_urls(self, target: str, method: str = "GET") -> list[str]:
+ with self._lock:
+ values = {
+ observed
+ for endpoint in self._endpoints.get(target, {}).values()
+ for observed_method, observed in endpoint.observed_requests
+ if method.upper() == observed_method
+ }
+ return sorted(values)
+
+ def snapshot(self) -> dict[str, Any]:
+ with self._lock:
+ targets: dict[str, Any] = {}
+ for target, endpoints in sorted(self._endpoints.items()):
+ edges = [
+ {"source": source, "destination": destination, "relation": relation}
+ for source, destination, relation in sorted(self._edges.get(target, set()))
+ ]
+ targets[target] = {
+ "endpoints": [endpoints[url].to_dict() for url in sorted(endpoints)],
+ "edges": edges,
+ "documents": list(self._documents.get(target, [])),
+ }
+ endpoint_count = sum(len(value["endpoints"]) for value in targets.values())
+ parameter_count = sum(
+ len(endpoint["parameters"])
+ for value in targets.values()
+ for endpoint in value["endpoints"]
+ )
+ return {
+ "schema_version": "1.0",
+ "summary": {
+ "targets": len(targets),
+ "endpoints": endpoint_count,
+ "parameters": parameter_count,
+ "edges": sum(len(value["edges"]) for value in targets.values()),
+ "documents": sum(len(value["documents"]) for value in targets.values()),
+ },
+ "targets": targets,
+ }
+
+
+def redact_url(url: str) -> str:
+ parsed = urlsplit(url)
+ sensitive = re.compile(r"(?:token|secret|password|passwd|api[_-]?key|auth|session|jwt)", re.I)
+ query = urlencode(
+ [
+ (name, "" if sensitive.search(name) else value)
+ for name, value in parse_qsl(parsed.query, keep_blank_values=True)
+ ]
+ )
+ return urlunsplit((parsed.scheme, parsed.netloc, parsed.path, query, ""))