Phase-accurate webrunner cost/time estimator with masscan-Tor OPSEC warnings

Estimator rewrite:
- Phase-decomposed estimate_scan_hours: masscan, nmap-only, nmap-fingerprint,
  probe, nuclei phases each timed separately
- Accepts 6 tuning params (nmap_timing, nmap_workers, nuclei_rate,
  nuclei_concurrency, hit_rate, use_tor) — was previously ignoring all of them
- nmap-only mode fixed: was underestimated ~100x (treated nmap like packet
  scanning); now uses NMAP_TIME_BY_TIMING table per timing template
- masscan+nuclei mode now properly modeled (was falling through to
  masscan-only formula, missing the nuclei phase entirely)
- Tor multiplier scoped per-phase: applied to nmap/nuclei/probe only.
  Masscan uses raw sockets and Tor cannot proxy raw sockets — was
  previously applying Tor multiplier to masscan rate (incorrect)
- 5min/node provisioning overhead added (matters vs Linode 1h billing min)
- build_estimate_table accepts tuning dict, threads through to engine

Operator-facing changes (deploy_webrunner.py):
- Tuning section moved BEFORE estimate display so numbers reflect actual
  operator choices, not defaults
- Estimate header now shows active tuning values when set
- Caveats block printed below estimate explaining empirical limits
  (hit rate range, nuclei template variance, nmap timing tradeoffs,
  provisioning, billing minimums)
- CRITICAL OPSEC banner displayed when Tor enabled with masscan modes:
  loud red warning that masscan SYN packets bypass Tor and reveal cloud
  node IP. Includes confirmation gate — operator must affirm they
  understand before proceeding. Suggests nmap-only mode for full Tor
  coverage if attribution risk is unacceptable

Why: operator was getting estimates off by 100x for nmap-only and 0%
accurate for masscan+nuclei. Tuning controls were invisible to the
estimator. Tor + masscan was silently broken-by-design (raw sockets
bypass proxychains) without operator awareness.
This commit is contained in:
n0mad1k
2026-05-03 06:45:36 -04:00
parent 13ce8a9b8a
commit 61251876a6
2 changed files with 185 additions and 37 deletions
+79 -17
View File
@@ -208,8 +208,53 @@ def _get_tuning_params(scan_mode: str, default_rate: int, vars_overrides: dict)
return tuning
def _show_estimate_table(total_ips: int, n_ports: int, providers: list[str], scan_mode: str, use_tor: bool = False):
rows = build_estimate_table(total_ips, n_ports, providers, scan_mode, use_tor=use_tor)
def _show_masscan_tor_warning():
R = COLORS['RED']
Y = COLORS['YELLOW']
W = COLORS['WHITE']
Z = COLORS['RESET']
print(f"\n{R}{'' * 70}{Z}")
print(f"{R}█ ⚠ CRITICAL OPSEC WARNING — MASSCAN BYPASSES TOR █{Z}")
print(f"{R}{'' * 70}{Z}")
print(f"{W} masscan uses raw sockets and CANNOT be tunneled through Tor or")
print(f" proxychains. Every SYN packet sent during the masscan phase reveals")
print(f" THIS CLOUD NODE'S IP to the targets and any monitoring along the path.{Z}")
print()
print(f"{Y} Tor will protect: {Z}{W}nmap fingerprinting, nuclei requests, probe banners{Z}")
print(f"{Y} Tor will NOT protect: {Z}{R}masscan SYN scan (bypasses the proxy entirely){Z}")
print()
print(f"{W} If you need full Tor coverage, use 'nmap-only' mode (slow but fully")
print(f" proxied via -sT). The masscan phase is fundamentally incompatible.{Z}\n")
def _show_caveats_block(scan_mode: str, use_tor: bool):
Y = COLORS['YELLOW']
W = COLORS['WHITE']
R = COLORS['RED']
Z = COLORS['RESET']
print(f"{Y} Caveats — empirical estimates, real scans vary ±30%:{Z}")
print(f"{W} • Hit rate assumes ~1% of IPs have open ports on selected ports.")
print(f" Real range: 0.5%5% (higher for SSH/HTTP/HTTPS, lower for niche")
print(f" ports). Override via vars file: hit_rate: 0.03{Z}")
if scan_mode in ('masscan+nuclei',):
print(f"{W} • Nuclei estimate assumes 5s/target (typical CVE template, 13 HTTP")
print(f" requests). Heavy templates with many requests/matchers run 25×")
print(f" longer.{Z}")
if scan_mode in ('masscan+nmap', 'geo-scout', 'nmap-only'):
print(f"{W} • nmap timing: T1=60s/host, T2=30s, T3=15s, T4=10s. Lower T values")
print(f" evade rate-limit detection but multiply scan time accordingly.{Z}")
if use_tor:
print(f"{R} • Tor adds 5× latency to nmap/nuclei/probe phases. Real overhead")
print(f" varies 3×–10× depending on circuit quality. Masscan is NOT")
print(f" proxied — see the warning banner above.{Z}")
print(f"{W} • Provisioning: ~5 min/node included. Add 510 min for first-time")
print(f" cloud-provider auth or new region.")
print(f" • Cost: Linode/FlokiNET bill minimum 1h per node. AWS bills per")
print(f" second. Short scans still incur the per-provider minimum.{Z}\n")
def _show_estimate_table(total_ips: int, n_ports: int, providers: list[str], scan_mode: str, use_tor: bool = False, tuning: dict | None = None):
rows = build_estimate_table(total_ips, n_ports, providers, scan_mode, use_tor=use_tor, tuning=tuning)
C = COLORS['CYAN']
W = COLORS['WHITE']
@@ -217,11 +262,25 @@ def _show_estimate_table(total_ips: int, n_ports: int, providers: list[str], sca
G = COLORS['GREEN']
R = COLORS['RESET']
tor_note = " [Tor: estimates reflect nmap/probe latency]" if use_tor else ""
masscan_modes = ('masscan-only', 'masscan+nmap', 'geo-scout', 'masscan+nuclei')
if use_tor and scan_mode in masscan_modes:
_show_masscan_tor_warning()
tor_note = " [Tor: estimates reflect nmap/probe latency only]" if use_tor else ""
print(f"\n{C}{'' * 70}{R}")
print(f"{C} WEBRUNNER — Cost & Time Estimate{R}")
print(f"{W} Total: {fmt_ip_count(total_ips)} IPs {n_ports} ports Mode: {scan_mode}{R}")
print(f"{W} Providers: {', '.join(PROVIDER_LABELS[p] for p in providers)}{Y}{tor_note}{R}")
if tuning:
bits = []
if 'masscan_rate' in tuning:
bits.append(f"masscan={tuning['masscan_rate']}pps")
if 'nmap_timing' in tuning:
bits.append(f"nmap=T{tuning['nmap_timing']}/{tuning.get('nmap_workers', 10)}w")
if 'nuclei_rate' in tuning and scan_mode == 'masscan+nuclei':
bits.append(f"nuclei={tuning['nuclei_rate']}rps/{tuning.get('nuclei_concurrency', 25)}c")
if bits:
print(f"{W} Tuning: {' · '.join(bits)}{R}")
print(f"{C}{'' * 70}{R}")
print(f" {'Preset':<14} {'Nodes':>6} {'IPs/Node':>10} {'Time/Node':>12} {'Total $':>10}")
print(f" {'' * 56}")
@@ -239,6 +298,7 @@ def _show_estimate_table(total_ips: int, n_ports: int, providers: list[str], sca
print(f"{C}{'' * 70}{R}")
print(f"{Y} * = recommended (Pareto-optimal: speed vs. billing minimum){R}\n")
_show_caveats_block(scan_mode, use_tor)
# ── parameter gathering ───────────────────────────────────────────────────────
@@ -335,16 +395,23 @@ def gather_webrunner_parameters() -> dict | None:
# Tor routing — ask before estimate so table reflects the slowdown
tor_raw = input(f"\nRoute scans through Tor? [y/N]: ").strip().lower()
config['use_tor'] = tor_raw in ['y', 'yes']
if config['use_tor']:
if scan_mode == 'masscan-only':
print(f"{COLORS['YELLOW']} Warning: masscan uses raw sockets and bypasses proxychains — Tor has no effect in masscan-only mode.{COLORS['RESET']}")
elif scan_mode in ('geo-scout', 'masscan+nmap'):
print(f"{COLORS['YELLOW']} Note: masscan phase bypasses Tor (raw sockets); nmap and probe phases will use Tor.{COLORS['RESET']}")
else:
print(f"{COLORS['YELLOW']} Tor routing enabled — all nmap traffic will use proxychains → SOCKS5 9050.{COLORS['RESET']}")
masscan_modes = ('masscan-only', 'masscan+nmap', 'geo-scout', 'masscan+nuclei')
if config['use_tor'] and scan_mode in masscan_modes:
_show_masscan_tor_warning()
confirm = input(f"{COLORS['RED']} Continue with Tor enabled, knowing masscan will leak this node's IP? [y/N]: {COLORS['RESET']}").strip().lower()
if confirm not in ['y', 'yes']:
print(f"{COLORS['YELLOW']} Tor disabled. To get full Tor coverage, re-run with scan mode 'nmap-only'.{COLORS['RESET']}")
config['use_tor'] = False
elif config['use_tor']:
print(f"{COLORS['GREEN']} Tor routing enabled — all nmap traffic will use proxychains → SOCKS5 9050.{COLORS['RESET']}")
# Cost / time estimate reflects Tor throughput impact
_show_estimate_table(total_ips, len(ports), providers, scan_mode, use_tor=config['use_tor'])
# Advanced tuning — gather BEFORE estimate so table reflects operator choices
default_rate = config['masscan_rate']
tuning = _get_tuning_params(scan_mode, default_rate, vars_overrides)
config.update(tuning)
# Cost / time estimate — reflects tuning + Tor throughput impact
_show_estimate_table(total_ips, len(ports), providers, scan_mode, use_tor=config['use_tor'], tuning=tuning)
# Preset selection
print(f"{COLORS['BLUE']}Select deployment preset:{COLORS['RESET']}")
@@ -438,11 +505,6 @@ def gather_webrunner_parameters() -> dict | None:
if config['operator_ip']:
print(f"{COLORS['GREEN']}Operator IP: {config['operator_ip']}{COLORS['RESET']}")
# Advanced tuning
default_rate = config['masscan_rate']
tuning = _get_tuning_params(scan_mode, default_rate, vars_overrides)
config.update(tuning)
# OPSEC / teardown options
print(f"\n{COLORS['BLUE']}Deployment Options:{COLORS['RESET']}")
+106 -20
View File
@@ -48,20 +48,101 @@ BILLING_MINIMUM = {
}
NMAP_TIME_PER_HOST_SEC = 10 # empirical: nmap -sV -T4 per host, seconds
MASSCAN_HIT_RATE = 0.01 # fraction of scanned IPs that have open ports
_NMAP_WORKERS = 10 # must match WEBRUNNER_NMAP_WORKERS default
# ── Empirical timing constants ────────────────────────────────────────────────
# Per-host nmap time by timing template (seconds, -sV --version-intensity 5)
NMAP_TIME_BY_TIMING = {
1: 60.0, # T1 sneaky
2: 30.0, # T2 polite
3: 15.0, # T3 normal
4: 10.0, # T4 aggressive (baseline)
}
# Average per-target time for typical nuclei CVE template (13 HTTP requests).
# Heavy templates with many requests/matchers will run 25× longer.
NUCLEI_TIME_PER_TARGET_SEC = 5.0
# TCP banner grab time (geo-scout probe phase)
PROBE_TIME_PER_HOST_SEC = 3.0
# Default fraction of scanned IPs with open ports on common ports.
# Real-world range: 0.5%5% depending on ports & geography.
MASSCAN_HIT_RATE = 0.01
# Defaults — must match WEBRUNNER tuning defaults
_NMAP_WORKERS = 10
_NUCLEI_RATE = 150
_NUCLEI_CONCURRENCY = 25
# Tor latency multiplier — applies ONLY to TCP probe phases (nmap/nuclei/probe).
# Masscan uses raw sockets and bypasses proxychains entirely — Tor cannot
# protect masscan SYN packets. The cloud node's IP is exposed to every
# masscan target regardless of this setting.
TOR_PHASE_MULTIPLIER = 5.0
# Per-node provisioning overhead (apt install, optional nuclei download).
# Nodes provision in parallel so this is roughly constant.
PROVISION_OVERHEAD_HOURS = 5 / 60
def estimate_scan_hours(ip_count: int, n_ports: int, rate: int, scan_mode: str = 'masscan-only') -> float:
def estimate_scan_hours(
ip_count: int,
n_ports: int,
rate: int,
scan_mode: str = 'masscan-only',
*,
nmap_timing: int = 4,
nmap_workers: int = _NMAP_WORKERS,
nuclei_rate: int = _NUCLEI_RATE,
nuclei_concurrency: int = _NUCLEI_CONCURRENCY,
hit_rate: float = MASSCAN_HIT_RATE,
use_tor: bool = False,
) -> float:
"""Phase-decomposed scan time estimate. Returns hours per node."""
if rate <= 0:
return 0.0
masscan_hours = (ip_count * n_ports / rate) / 3600
return PROVISION_OVERHEAD_HOURS
nmap_workers = max(nmap_workers, 1)
seconds = 0.0
# masscan phase: raw sockets, no Tor penalty (Tor cannot proxy raw sockets)
if scan_mode in ('masscan-only', 'masscan+nmap', 'geo-scout', 'masscan+nuclei'):
seconds += ip_count * n_ports / rate
# nmap-only phase: every IP gets full -sV fingerprint
if scan_mode == 'nmap-only':
per_host = NMAP_TIME_BY_TIMING.get(nmap_timing, 10.0)
if use_tor:
per_host *= TOR_PHASE_MULTIPLIER
seconds += (ip_count * per_host) / nmap_workers
# nmap fingerprint phase: only on masscan hits
if scan_mode in ('masscan+nmap', 'geo-scout'):
nmap_hosts = ip_count * MASSCAN_HIT_RATE
nmap_hours = (nmap_hosts * NMAP_TIME_PER_HOST_SEC) / (_NMAP_WORKERS * 3600)
return masscan_hours + nmap_hours
return masscan_hours
nmap_hosts = ip_count * hit_rate
per_host = NMAP_TIME_BY_TIMING.get(nmap_timing, 10.0)
if use_tor:
per_host *= TOR_PHASE_MULTIPLIER
seconds += (nmap_hosts * per_host) / nmap_workers
# probe banner phase: geo-scout only, on masscan hits
if scan_mode == 'geo-scout':
probe_hosts = ip_count * hit_rate
per_probe = PROBE_TIME_PER_HOST_SEC
if use_tor:
per_probe *= TOR_PHASE_MULTIPLIER
seconds += (probe_hosts * per_probe) / nmap_workers
# nuclei phase: only on masscan-discovered ip:port pairs
if scan_mode == 'masscan+nuclei':
nuclei_targets = ip_count * hit_rate
per_target = NUCLEI_TIME_PER_TARGET_SEC
if use_tor:
per_target *= TOR_PHASE_MULTIPLIER
rate_throughput = float(nuclei_rate)
parallel_throughput = nuclei_concurrency / per_target
effective_rps = max(min(rate_throughput, parallel_throughput), 1.0)
seconds += nuclei_targets / effective_rps
return seconds / 3600 + PROVISION_OVERHEAD_HOURS
def node_cost(provider: str, instance_type: str, scan_hours: float) -> float:
@@ -84,30 +165,35 @@ def fmt_ip_count(n: int) -> str:
return str(n)
# Tor adds ~5x latency overhead for TCP probes; masscan bypasses proxychains (raw sockets)
# so only nmap/probe phases are affected — modes that include masscan see partial impact
TOR_RATE_MULTIPLIER = 0.2
def build_estimate_table(
total_ips: int,
n_ports: int,
providers: list[str],
scan_mode: str,
use_tor: bool = False,
tuning: dict | None = None,
) -> list[dict]:
mode_rate = SCAN_MODES.get(scan_mode, {'rate': 3000})['rate']
if use_tor and scan_mode != 'masscan-only':
mode_rate = int(mode_rate * TOR_RATE_MULTIPLIER)
tuning = tuning or {}
masscan_rate = int(tuning.get('masscan_rate') or SCAN_MODES.get(scan_mode, {'rate': 3000})['rate'])
kwargs = dict(
nmap_timing=int(tuning.get('nmap_timing', 4)),
nmap_workers=int(tuning.get('nmap_workers', _NMAP_WORKERS)),
nuclei_rate=int(tuning.get('nuclei_rate', _NUCLEI_RATE)),
nuclei_concurrency=int(tuning.get('nuclei_concurrency', _NUCLEI_CONCURRENCY)),
hit_rate=float(tuning.get('hit_rate', MASSCAN_HIT_RATE)),
use_tor=use_tor,
)
rows = []
for preset_key, preset in PRESETS.items():
n_chunks = max(1, math.ceil(total_ips / preset['chunk_size']))
typical_ips = min(preset['chunk_size'], total_ips)
hours_per_node = estimate_scan_hours(typical_ips, n_ports, mode_rate, scan_mode)
hours_per_node = estimate_scan_hours(typical_ips, n_ports, masscan_rate, scan_mode, **kwargs)
total_cost = 0.0
for i in range(n_chunks):
chunk_ips = min(preset['chunk_size'], total_ips - i * preset['chunk_size'])
chunk_hours = estimate_scan_hours(chunk_ips, n_ports, mode_rate, scan_mode)
chunk_hours = estimate_scan_hours(chunk_ips, n_ports, masscan_rate, scan_mode, **kwargs)
provider = providers[i % len(providers)]
instance = DEFAULT_INSTANCE.get(provider, 'g6-standard-2')
total_cost += node_cost(provider, instance, chunk_hours)