feat: identify decoded KEY .sub files (real-world coverage 58%→93%)

KEY .sub files (~35% of real captures) returned zero identification
because every path bailed on `not has_raw_data`, despite the file
already carrying a decoded Protocol name. Route those by protocol name
instead:

- category_router: add route_by_protocol() — device-specific brands
  (CAME/Nice/KeeLoq/Security+/Honeywell) map to one confident category;
  generic shared encoders (Princeton/EV1527/Holtek/Intertechno) map to a
  broad allowed family set, since the same silicon spans
  remote/doorbell/fan/gate.
- pattern_decoder.decode: emit a decoded_key DeviceMatch (details carry
  predicted_category) for KEY files instead of [].
- device_identifier.identify: only bail when there is neither RAW data
  nor a protocol name; gate statistical scoring on has_raw_data.

Adds scripts/benchmark_realworld.py — validates the real pipeline
against the real UberGuidoZ corpus (folder = ground-truth device type),
reporting top-1, routed (truth in allowed set), and coverage.

Measured on n=344 (seed 42): coverage 58%→93%, routed 59.3%→65.2%;
generalizes on seed 7 (93%/63.6%). RAW path byte-identical (no
regression); synthetic phase-0 gate still passes (top-3 67%).

Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
This commit is contained in:
leetcrypt
2026-07-19 11:51:19 -07:00
parent 11d8957a49
commit 9f32e448ab
4 changed files with 410 additions and 5 deletions
+273
View File
@@ -0,0 +1,273 @@
#!/usr/bin/env python3
"""
Real-World Device-Type Validation
==================================
Unlike ``benchmark_phase0.py`` (which *fabricates* .sub files from the protocol
DB's own timing parameters and therefore only measures an upper bound), this
harness runs the **real identification pipeline** against **real community
Flipper Zero captures** — the UberGuidoZ Sub-GHz corpus — where the folder name
is the ground-truth device type.
It measures what actually matters for "what kind of device is this?":
1. Category routing — does the router put the capture in the right device
family? Reported two ways:
top1 : router's single best category == ground truth
routed : ground truth ∈ router's allowed_categories (the searched set)
2. Coverage — % of files the pipeline can even act on (RAW-parseable,
timing-extractable, ≥1 device match).
Run:
python scripts/benchmark_realworld.py # default sample
python scripts/benchmark_realworld.py --per-category 80 --out /tmp/rw.json
"""
import argparse
import json
import random
import sys
import time
from collections import Counter, defaultdict
from pathlib import Path
sys.path.insert(0, str(Path(__file__).parent.parent))
from src.parser.sub_parser import parse_sub_file
from src.matcher.pattern_decoder import get_pattern_decoder
# ── Ground truth: UberGuidoZ folder name -> GigLez router category ──────────
# Only folders with an unambiguous mapping onto a category the router can emit
# are included. Ambiguous grab-bags (Misc, Jamming, Settings, Pocsag, ...) are
# intentionally excluded so the denominator stays honest.
FOLDER_TO_CATEGORY = {
"Doorbells": "Doorbell",
"Garages": "Garage Door Opener",
"Gates": "Garage Door Opener",
"Weather_stations": "Weather Sensor",
"Ceiling_Fans": "Fan Controller",
"Fans": "Fan Controller",
"Motion_Sensors": "Security Sensor",
"Smoke_Alarm": "Security Sensor",
"Vehicles": "Remote Control",
"Smart_Home_Remotes": "Remote Control",
"Remote_Outlet_Switches": "Remote Control",
}
DEFAULT_DATASET = (
Path(__file__).parent.parent
/ "data/rf_test_datasets/UberGuidoZ_Flipper/Sub-GHz"
)
def collect_files(dataset_root: Path, per_category: int, seed: int):
"""Return list of (path, ground_truth_category, folder) sampled per folder."""
rng = random.Random(seed)
out = []
for folder, category in FOLDER_TO_CATEGORY.items():
folder_path = dataset_root / folder
if not folder_path.is_dir():
continue
subs = sorted(folder_path.rglob("*.sub"))
rng.shuffle(subs)
for p in subs[:per_category]:
out.append((p, category, folder))
return out
def evaluate(files, decoder):
"""Run the routing + decode pipeline over the sampled files."""
ta = decoder.timing_analyzer
pd = decoder.preamble_detector
router = decoder.category_router
results = []
for path, gt_category, folder in files:
rec = {
"file": str(path),
"folder": folder,
"ground_truth": gt_category,
"status": None, # ok | no_raw | no_timing | parse_error
"file_format": None,
"protocol": None, # Flipper Protocol: field (present on KEY files)
"predicted_top1": None,
"allowed_categories": [],
"routed_hit": False,
"top1_hit": False,
"n_device_matches": 0,
"top_device": None,
"top_confidence": None,
}
try:
meta = parse_sub_file(str(path))
rec["file_format"] = getattr(meta, "file_format", None)
rec["protocol"] = getattr(meta, "protocol", None)
if not getattr(meta, "has_raw_data", False):
# Decoded KEY file: no pulses to time, but a Protocol name is
# itself identifying — route by name.
if not getattr(meta, "protocol", None):
rec["status"] = "no_raw" # nothing to go on
results.append(rec)
continue
pred = router.route_by_protocol(meta.protocol, meta.frequency)
rec["status"] = "ok_key"
rec["predicted_top1"] = pred.primary_category
rec["allowed_categories"] = list(pred.allowed_categories or [])
rec["top1_hit"] = (pred.primary_category == gt_category)
rec["routed_hit"] = (
gt_category in rec["allowed_categories"] or pred.use_full_db
)
matches = decoder.decode(meta)
rec["n_device_matches"] = len(matches)
if matches:
rec["top_device"] = matches[0].name
rec["top_confidence"] = round(matches[0].confidence, 3)
results.append(rec)
continue
pulses = meta.raw_data
timing = ta.extract_timing(pulses)
short, long = timing.short_pulse_us, timing.long_pulse_us
if short == 0 or long == 0:
rec["status"] = "no_timing"
results.append(rec)
continue
detected = pd.detect(pulses, short, long)
ptype = detected.type if detected else "none"
pred = router.predict(
frequency=meta.frequency,
short_pulse_us=short,
long_pulse_us=long,
pulse_count=len(pulses),
preamble_type=ptype,
)
rec["status"] = "ok"
rec["predicted_top1"] = pred.primary_category
rec["allowed_categories"] = list(pred.allowed_categories or [])
rec["top1_hit"] = (pred.primary_category == gt_category)
rec["routed_hit"] = (
gt_category in rec["allowed_categories"]
or (pred.use_full_db) # full-DB fallback searches everything
)
matches = decoder.decode(meta)
rec["n_device_matches"] = len(matches)
if matches:
rec["top_device"] = matches[0].name
rec["top_confidence"] = round(matches[0].confidence, 3)
except Exception as e: # noqa: BLE001 — want to bucket, not crash
rec["status"] = "parse_error"
rec["error"] = str(e)[:200]
results.append(rec)
return results
def report(results):
total = len(results)
status_counts = Counter(r["status"] for r in results)
fmt_counts = Counter(r["file_format"] for r in results)
routable = [r for r in results if r["status"] in ("ok", "ok_key")]
n_routable = len(routable)
n_raw = sum(1 for r in routable if r["status"] == "ok")
n_key = sum(1 for r in routable if r["status"] == "ok_key")
top1_hits = sum(r["top1_hit"] for r in routable)
routed_hits = sum(r["routed_hit"] for r in routable)
with_device = sum(1 for r in routable if r["n_device_matches"] > 0)
print("=" * 74)
print("REAL-WORLD DEVICE-TYPE VALIDATION (UberGuidoZ Sub-GHz corpus)")
print("=" * 74)
print(f"Files sampled : {total}")
print(f" status breakdown : {dict(status_counts)}")
print(f" file formats : {dict(fmt_counts)}")
print(f"Routable (RAW + KEY) : {n_routable} "
f"({n_routable/total:.0%} of sampled) "
f"[RAW timing={n_raw}, KEY protocol={n_key}]")
print()
if n_routable:
print("── Category accuracy (over routable files) ──")
print(f" top-1 (best == truth) : {top1_hits}/{n_routable} "
f"= {top1_hits/n_routable:.1%}")
print(f" routed (truth ∈ allowed set): {routed_hits}/{n_routable} "
f"= {routed_hits/n_routable:.1%}")
print(f" device match coverage : {with_device}/{n_routable} "
f"= {with_device/n_routable:.1%}")
print()
# End-to-end (routable AND top-1 correct) over ALL sampled files — the
# number a user actually experiences on an arbitrary upload.
print("── End-to-end over ALL sampled (incl. unparseable) ──")
print(f" top-1 : {top1_hits}/{total} = {top1_hits/total:.1%}")
print(f" routed : {routed_hits}/{total} = {routed_hits/total:.1%}")
print()
# Per-category breakdown
print("── Per-category (routable only) ──")
by_cat = defaultdict(list)
for r in routable:
by_cat[r["ground_truth"]].append(r)
print(f" {'category':22} {'n':>4} {'top1':>7} {'routed':>7}")
for cat in sorted(by_cat):
rs = by_cat[cat]
n = len(rs)
t1 = sum(x["top1_hit"] for x in rs) / n
rt = sum(x["routed_hit"] for x in rs) / n
print(f" {cat:22} {n:>4} {t1:>6.0%} {rt:>6.0%}")
print()
# Confusion: where did top-1 send the misses?
print("── Top-1 confusion (ground_truth -> predicted, misses only) ──")
conf = Counter()
for r in routable:
if not r["top1_hit"]:
conf[(r["ground_truth"], r["predicted_top1"])] += 1
for (gt, pred), c in conf.most_common(15):
print(f" {gt:22} -> {str(pred):22} x{c}")
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--dataset", type=Path, default=DEFAULT_DATASET)
ap.add_argument("--per-category", type=int, default=50,
help="max files sampled per folder (0 = all)")
ap.add_argument("--seed", type=int, default=42)
ap.add_argument("--out", type=Path, default=None,
help="write per-file JSON results here")
args = ap.parse_args()
if not args.dataset.is_dir():
print(f"Dataset not found: {args.dataset}", file=sys.stderr)
sys.exit(1)
per_cat = args.per_category or 10**9
files = collect_files(args.dataset, per_cat, args.seed)
if not files:
print("No .sub files found under mapped folders.", file=sys.stderr)
sys.exit(1)
print(f"Loading decoder + protocol DB ...")
decoder = get_pattern_decoder()
t0 = time.time()
results = evaluate(files, decoder)
dt = time.time() - t0
print(f"Evaluated {len(files)} files in {dt:.1f}s "
f"({dt/len(files)*1000:.0f} ms/file)\n")
report(results)
if args.out:
args.out.write_text(json.dumps(results, indent=2))
print(f"\nPer-file results -> {args.out}")
if __name__ == "__main__":
main()
+96
View File
@@ -36,6 +36,36 @@ class DeviceCategory:
UNKNOWN = "Unknown"
# ── Decoded-protocol-name routing (KEY .sub files) ─────────────────────────
# A KEY file already carries a Flipper `Protocol:` name. For device-specific
# brands that name alone pins the category. Checked in order; first hit wins.
# (keywords, category, confidence)
_PROTOCOL_BRAND_RULES = [
(("honeywell", "2gig", "magellan", "scher", "khan", "mastercode"),
DeviceCategory.SECURITY_SENSOR, 0.80),
(("came", "nice", "faac", "hormann", "somfy", "keeloq", "gate", "megacode",
"linear", "chamberlain", "liftmaster", "marantec", "doorhan", "an-motors",
"an_motors", "aprimatic", "beninca", "bft", "security+", "secplus", "genie",
"clemsa", "ansonic", "sommer", "novoferm", "dooya", "alutech", "elmes",
"nero", "hcs101", "starline", "star_line"),
DeviceCategory.GARAGE_DOOR, 0.80),
(("nexus", "lacrosse", "acurite", "oregon", "ambient", "infactory", "auriol",
"bresser", "thermo", "gt-wt", "gt_wt", "wt450", "tx8300", "tx_8300",
"wendox", "kedsum"),
DeviceCategory.WEATHER_SENSOR, 0.75),
(("tpms", "schrader", "pmv"),
DeviceCategory.TPMS, 0.80),
]
# Shared line-encoder silicon reused across remotes/doorbells/fans/gates. The
# name does NOT determine device type, so we route to the whole family rather
# than dishonestly narrowing to one category.
_GENERIC_ENCODERS = (
"princeton", "ev1527", "pt2262", "pt2264", "holtek", "ht12", "ht6",
"smc5326", "intertechno", "rcswitch", "rc-switch", "x10", "power_smart",
)
@dataclass
class CategoryPrediction:
"""Result of category routing"""
@@ -136,6 +166,72 @@ class CategoryRouter:
use_full_db=True,
)
def route_by_protocol(
self, protocol: Optional[str], frequency: int
) -> CategoryPrediction:
"""
Route a *decoded* KEY-file to a category from its Protocol name.
KEY .sub files have no RAW pulses to time, but they DO carry a Flipper
`Protocol:` name. Device-specific brands (CAME, Nice, KeeLoq, Security+,
Honeywell, ...) pin one category. Generic shared encoders (Princeton,
EV1527, Holtek, Intertechno, ...) are the same silicon in remotes,
doorbells, fans and gates — so we route to the whole family (a broad
allowed set) rather than guessing a single type.
Args:
protocol: Decoded Flipper protocol name (e.g. "CAME", "Princeton")
frequency: Carrier frequency in Hz
Returns:
CategoryPrediction
"""
name = (protocol or "").strip().lower()
freq_mhz = (frequency or 0) / 1_000_000
if not name or name in ("raw", "binraw", "unknown"):
return CategoryPrediction(
primary_category=DeviceCategory.UNKNOWN,
confidence=0.0,
allowed_categories=[],
reasoning="no decoded protocol name",
use_full_db=True,
)
# Device-specific brand → single confident category
for keywords, category, conf in _PROTOCOL_BRAND_RULES:
if any(k in name for k in keywords):
return CategoryPrediction(
primary_category=category,
confidence=conf,
allowed_categories=[category],
reasoning=f"decoded protocol '{protocol}'{category}",
)
# Generic shared encoder → broad family (honest 'routed' signal)
if any(k in name for k in _GENERIC_ENCODERS):
allowed = [DeviceCategory.REMOTE_CONTROL, DeviceCategory.DOORBELL,
DeviceCategory.FAN_CONTROLLER, DeviceCategory.GARAGE_DOOR]
# 315/390 bands skew toward garage/gate; 433 is the free-for-all.
primary = (DeviceCategory.GARAGE_DOOR if 300 <= freq_mhz <= 392
else DeviceCategory.REMOTE_CONTROL)
return CategoryPrediction(
primary_category=primary,
confidence=0.45,
allowed_categories=allowed,
reasoning=(f"generic encoder '{protocol}' @ {freq_mhz:.1f} MHz "
"→ shared-silicon family (remote/doorbell/fan/gate)"),
)
# Unrecognised name — don't guess, search everything.
return CategoryPrediction(
primary_category=DeviceCategory.UNKNOWN,
confidence=0.0,
allowed_categories=[],
reasoning=f"unrecognised protocol '{protocol}'",
use_full_db=True,
)
# ── Band-specific helpers ──────────────────────────────────────────────
def _route_300mhz_band(
+6 -5
View File
@@ -117,16 +117,17 @@ class DeviceIdentifier:
t0 = time.time()
result = IdentificationResult(signal_metadata=signal_data)
if not signal_data.has_raw_data:
# No raw data - cannot identify
if not signal_data.has_raw_data and not signal_data.protocol:
# No raw data AND no decoded protocol name - cannot identify
result.method = "none"
return result
# Step 1: Pattern-based decoding (heuristic)
# Step 1: Pattern-based decoding (heuristic). For KEY files this routes
# by the decoded protocol name; for RAW files it times the pulses.
heuristic_matches = self.pattern_decoder.decode(signal_data)
# Step 2: Statistical classification (if enabled and sufficient data)
if use_statistical and heuristic_matches:
# Step 2: Statistical classification (needs RAW pulses; skip on KEY)
if use_statistical and heuristic_matches and signal_data.has_raw_data:
statistical_matches = self._apply_statistical_scoring(
signal_data,
heuristic_matches
+35
View File
@@ -103,6 +103,10 @@ class PatternDecoder:
List of DeviceMatch sorted by confidence (highest first)
"""
if not metadata.has_raw_data:
# KEY files carry no pulses to time, but a decoded Protocol name is
# itself identifying — route it instead of returning nothing.
if metadata.protocol:
return self._decode_from_key(metadata)
return []
pulses = metadata.raw_data
@@ -125,6 +129,37 @@ class PatternDecoder:
# Deduplicate and rank by confidence
return self._rank_matches(matches)
def _decode_from_key(self, metadata: SignalMetadata) -> List[DeviceMatch]:
"""
Identify a decoded (KEY) .sub file from its Protocol name.
KEY files have no RAW pulses but carry a decoded `Protocol:` name,
frequency and key. We route the protocol name to a device category and
emit a match surfacing the real Flipper protocol + routed category, so
decoded uploads are no longer silently unidentifiable.
"""
pred = self.category_router.route_by_protocol(
metadata.protocol, metadata.frequency
)
signature = ProtocolSignature(
name=metadata.protocol,
category=pred.primary_category,
frequency=metadata.frequency or 433920000,
)
details = {
"predicted_category": pred.primary_category,
"allowed_categories": pred.allowed_categories,
"routing_reason": pred.reasoning,
"source": "decoded_key",
"bit_length": metadata.bit_length,
}
return [DeviceMatch(
protocol=signature,
confidence=pred.confidence,
match_method="decoded_key",
details=details,
)]
def _identify_pulse_widths(self, pulses: List[int]) -> Tuple[int, int, int, int]:
"""
Identify SHORT/LONG pulse and gap durations using robust timing analyzer