DragonFlyBSD Kernel Audit
DF-0884 / harness.c
← back to finding ↓ download raw
  1
  2
  3
  4
  5
  6
  7
  8
  9
 10
 11
 12
 13
 14
 15
 16
 17
 18
 19
 20
 21
 22
 23
 24
 25
 26
 27
 28
 29
 30
 31
 32
 33
 34
 35
 36
 37
 38
 39
 40
 41
 42
 43
 44
 45
 46
 47
 48
 49
 50
 51
 52
 53
 54
 55
 56
 57
 58
 59
 60
 61
 62
 63
 64
 65
 66
 67
 68
 69
 70
 71
 72
 73
 74
 75
 76
 77
 78
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
/*
 * DF-0884 - Use-after-free in smbfs_readvdir via dead-code vnode lock
 *           upgrade on read(2) of a directory.
 *
 * This is a DETERMINISTIC HARNESS that transcribes the kernel race from
 * sys/vfs/smbfs/smbfs_io.c and sys/vfs/smbfs/smbfs_smb.c into userspace.
 * The live trigger needs a mounted SMB share (an SMB server reachable from
 * the kernel client), which is not present on this isolated KVM guest -
 * there is no network beyond QEMU user-mode NAT to the host and no SMB
 * server.  Per the harness precedent (DF-0598 / DF-0599, also smbfs), we
 * prove the bug deterministically by transcribing the exact code path and
 * racing the two threads across a controllable interleaving point that
 * models the SMB network round-trip inside smbfs_findnext().
 *
 * ---------------------------------------------------------------------------
 * The bug (every hop cited against sys/vfs/smbfs/smbfs_io.c):
 *
 *   vfs_vnops.c:751   vn_read() does  vn_lock(vp, LK_SHARED | LK_RETRY)
 *                     so the directory vnode is held SHARED on read(2).
 *   smbfs_io.c:202    lks = LK_EXCLUSIVE; with the lockstatus() call that
 *                     would detect the real shared mode COMMENTED OUT; lks
 *                     is hardcoded to LK_EXCLUSIVE.
 *   smbfs_io.c:203    if (lks == LK_SHARED)   -- ALWAYS FALSE (dead code)
 *   smbfs_io.c:204        vn_lock(vp, LK_UPGRADE ...)  -- DEAD CODE
 *   smbfs_io.c:205    error = smbfs_readvdir(vp, ...)  -- runs SHARED
 *
 *   smbfs_readvdir() (smbfs_io.c:78-174) MUTATES per-vnode directory
 *   iteration state: np->n_dirseq and np->n_dirofs (lines 118-141).  The
 *   dead-code upgrade means two read(2) callers can run smbfs_readvdir()
 *   CONCURRENTLY under the shared lock.
 *
 *   The race to UAF:
 *     Thread A:  smbfs_readvdir -> smbfs_findnext(ctx, ...)  (smbfs_io.c:137
 *                or :151).  smbfs_findnext() (smbfs_smb.c:1196) blocks on a
 *                full SMB network round-trip inside smbfs_findnextLM1/LM2
 *                (smbfs_smb.c:854/1049) while still holding the ctx pointer
 *                (== the pointer stored in np->n_dirseq).
 *     Thread B:  smbfs_readvdir, offset != np->n_dirofs (A advanced it), so
 *                takes the REOPEN branch (smbfs_io.c:118-135) and calls
 *                smbfs_findclose(np->n_dirseq, ...) at smbfs_io.c:121.
 *                smbfs_findclose() (smbfs_smb.c:1224-1236) does
 *                kfree(ctx, M_SMBFSDATA) at smbfs_smb.c:1234.
 *     Thread A:  resumes from the blocked smbfs_findnext() and WRITES
 *                ctx->f_attr.fa_ino at smbfs_smb.c:1220 (and reads it back
 *                at smbfs_io.c:157) -> WRITE/READ UAF on freed smbfs_fctx.
 *
 *   The getdents(2) path is NOT affected: smbfs_readdir() unconditionally
 *   takes LK_EXCLUSIVE (smbfs_vnops.c:725).
 *
 * ---------------------------------------------------------------------------
 * Harness model:
 *   - The vnode lock is a pthread rwlock; LK_SHARED  == read-lock,
 *     LK_EXCLUSIVE == write-lock, LK_UPGRADE == upgrade (write-lock).
 *   - The "SMB network round-trip" inside smbfs_findnext is a controllable
 *     barrier: Thread A blocks on a pthread cond var; the test thread
 *     releases it ONLY AFTER Thread B has run smbfs_findclose() on A ctx,
 *     modelling the full round-trip race window deterministically.
 *   - The allocator is POISONED: every free() overwrites the object with
 *     0xDD bytes and marks it freed, so a write through the dangling ctx
 *     is detected unambiguously (the poison signature is destroyed /
 *     observed-over-freed).
 *
 *   Two modes (argv[1]):
 *     "buggy"  -- transcribes smbfs_io.c:202-203 as-shipped (dead upgrade).
 *     "fixed"  -- transcribes the fix.diff: UNCONDITIONAL LK_UPGRADE before
 *                 smbfs_readvdir() and LK_DOWNGRADE after, so the two
 *                 readers serialize and the race cannot occur.
 *
 * Build:  cc -O2 -pthread -o harness harness.c
 * Run:    ./harness buggy    (expect: UAF CONFIRMED)
 *         ./harness fixed    (expect: FIXED no UAF)
 */

#include <pthread.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <unistd.h>
#include <errno.h>
#include <stdint.h>

/* ---- poisoned allocator ------------------------------------------------ */
#define POISON_BYTE 0xDD
struct alloc_rec {
    void   *ptr;
    size_t  sz;
    int     freed;
};
static struct alloc_rec g_allocs[64];
static int              g_nallocs = 0;

static void *
poison_malloc(size_t sz)
{
    void *p = calloc(1, sz ? sz : 1);
    if (!p) { perror("calloc"); exit(2); }
    if (g_nallocs < 64) {
        g_allocs[g_nallocs].ptr = p;
        g_allocs[g_nallocs].sz  = sz;
        g_allocs[g_nallocs].freed = 0;
        g_nallocs++;
    }
    return p;
}

static void
poison_free(void *p)
{
    int i;
    for (i = 0; i < g_nallocs; i++) {
        if (g_allocs[i].ptr == p) {
            if (g_allocs[i].freed) {
                fprintf(stderr, "[alloc] DOUBLE FREE detected on %p\n", p);
                exit(3);
            }
            /* model kfree(): don't actually return the page to the pool
             * yet; poison it so a write through the dangling pointer is
             * observable. */
            memset(p, POISON_BYTE, g_allocs[i].sz);
            g_allocs[i].freed = 1;
            return;
        }
    }
    fprintf(stderr, "[alloc] free of untracked %p\n", p);
    exit(4);
}

static int
poison_is_freed(void *p)
{
    int i;
    for (i = 0; i < g_nallocs; i++)
        if (g_allocs[i].ptr == p)
            return g_allocs[i].freed;
    return -1;
}

/* ---- transcribed kernel structures (only fields the race touches) ----- */
struct smbfattr {                 /* smbfs_subr.h:65 */
    int      fa_attr;
    int64_t  fa_size;
    long     fa_ino;
    /* fa_mtime etc. elided — irrelevant to the race */
};

struct smbnode {                  /* smbfs_node.h:69-70 */
    struct smbfs_fctx *n_dirseq;  /* ff context (the raced pointer) */
    long               n_dirofs;  /* last ff offset */
    long               n_ino;
};

struct smbfs_fctx {              /* smbfs_subr.h:90 */
    int             f_flags;
    struct smbfattr f_attr;
    char           *f_name;
    int             f_nmlen;
    long            f_dirofs_marker; /* unused in race, kept for size */
    /* remaining fields (f_ssp, f_rq, f_t2, f_skey, ...) elided */
};

/* ---- vnode lock model -------------------------------------------------- */
static pthread_rwlock_t g_vnlock;     /* models vp->v_lock */

/* ---- network-round-trip interleaving point ----------------------------- */
/* Thread A blocks in smbfs_findnext() on g_netio_cond until the orchestrator
 * (main) signals it AFTER Thread B has freed A's ctx.  This deterministically
 * models the full SMB network round-trip during which the race window is
 * open. */
static pthread_mutex_t g_netio_mtx = PTHREAD_MUTEX_INITIALIZER;
static pthread_cond_t  g_netio_cond = PTHREAD_COND_INITIALIZER;
static int             g_netio_thread_a_blocked = 0;   /* A set then waits */
static int             g_netio_release_a        = 0;   /* orchestrator set */
static volatile int    g_b_freed_a_ctx         = 0;   /* B did findclose */
static int             g_a_write_saw_freed     = -1;  /* A's snapshot at write */

/* ---- transcribed smbfs_smb.c ------------------------------------------- */
static int
model_smbfs_findopen(struct smbnode *np, struct smbfs_fctx **ctxpp)
{                              /* smbfs_smb.c:1170 */
    struct smbfs_fctx *ctx = poison_malloc(sizeof(*ctx)); /* :1177 */
    ctx->f_flags = 0;
    ctx->f_attr.fa_ino = 0;
    ctx->f_name = poison_malloc(16);
    *ctxpp = ctx;
    (void)np;
    return 0;
}

static int
model_smbfs_findnext(struct smbfs_fctx *ctx, int limit)
{                              /* smbfs_smb.c:1196 */
    /* smbfs_findnextLM1/LM2 perform an SMB network round-trip here
     * (smbfs_smb.c:868 smbfs_smb_search / :1099 smb_t2_request).  The race
     * window is exactly this blocking I/O.  Model it: block on the cond var
     * the FIRST time (Thread A) so B can run findclose() underneath us. */
    (void)limit;
    pthread_mutex_lock(&g_netio_mtx);
    if (!g_netio_thread_a_blocked) {
        /* This is Thread A's call.  Announce we are "in" the network I/O
         * holding ctx (== np->n_dirseq), then block until released. */
        g_netio_thread_a_blocked = 1;
        pthread_cond_signal(&g_netio_cond);   /* wake orchestrator */
        while (!g_netio_release_a)
            pthread_cond_wait(&g_netio_cond, &g_netio_mtx);
        pthread_mutex_unlock(&g_netio_mtx);

        /* *** RACE WINDOW IS NOW OPEN *** ctx may have been kfree()'d by
         * Thread B's smbfs_findclose() while we were blocked.  Snapshot the
         * freed state BEFORE the write so the verdict can tell the UAF case
         * (B freed first) from the safe case (A still owns ctx).  Then
         * resume the transcribed post-I/O write: smbfs_smb.c:1220. */
        g_a_write_saw_freed = poison_is_freed(ctx);
        ctx->f_attr.fa_ino = 0xCAFEBABE;                 /* THE UAF WRITE */
        return 0;
    }
    /* Subsequent calls (Thread B) model instant local completion. */
    pthread_mutex_unlock(&g_netio_mtx);
    ctx->f_attr.fa_ino = 0xBEEF;
    return 0;
}

static void
model_smbfs_findclose(struct smbfs_fctx *ctx)
{                              /* smbfs_smb.c:1224 */
    if (ctx->f_name)
        poison_free(ctx->f_name);                   /* :1233 (f_rname) */
    poison_free(ctx);                               /* :1234 kfree(ctx) */
}

/* ---- transcribed smbfs_io.c smbfs_readvdir (only the raced path) ------ */
struct uio { long uio_offset; long uio_resid; };

static int
model_smbfs_readvdir(struct smbnode *np, struct uio *uio)
{                              /* smbfs_io.c:78 */
    struct smbfs_fctx *ctx;
    int error, offset;

    if (uio->uio_offset < 0)
        return 1;               /* smbfs_io.c:91 EINVAL */
    offset = uio->uio_offset;   /* :95 */

    /* skip the "." / ".." synthesis (lines 97-113): we model offset==0 as
     * a fresh reader so the reopen branch fires. */

    if (offset != np->n_dirofs || np->n_dirseq == NULL) {   /* :118 */
        if (np->n_dirseq) {
            /* *** smbfs_io.c:121 *** -- this is the kfree() that races
             * with another thread's in-flight smbfs_findnext(ctx). */
            model_smbfs_findclose(np->n_dirseq);
            np->n_dirseq = NULL;                            /* :122 */
        }
        np->n_dirofs = 2;                                   /* :124 */
        error = model_smbfs_findopen(np, &ctx);             /* :125 */
        if (error) return error;
        np->n_dirseq = ctx;                                 /* :132 */
    } else {
        ctx = np->n_dirseq;                                 /* :134 */
    }
    /* skip-ahead loop (136-144) elided; offset model jumps to inner loop */
    /* inner loop (146-168): smbfs_findnext then use ctx->f_attr */
    if (uio->uio_resid > 0) {
        error = model_smbfs_findnext(ctx, 1);               /* :151 */
        if (error) return error;
        np->n_dirofs++;                                     /* :154 */
        /* smbfs_io.c:157 reads ctx->f_attr.fa_ino back: another UAF read
         * on the freed ctx.  Touch it to make the read observable. */
        long ino = ctx->f_attr.fa_ino;                      /* UAF READ */
        (void)ino;
    }
    return 0;
}

/* ---- transcribed smbfs_io.c smbfs_readvnode ---------------------------- */
static int g_fixed_mode = 0;   /* 0 = as-shipped (dead upgrade), 1 = fixed */

static int
model_smbfs_readvnode(struct smbnode *np, struct uio *uio)
{                              /* smbfs_io.c:176 */
    int lks, error;

    if (uio->uio_resid == 0) return 0;                      /* :196 */

    if (1 /* vp->v_type == VDIR, :201 */) {
        /* smbfs_io.c:202-204 (AS-SHIPPED -- dead upgrade): */
        lks = 2 /* LK_EXCLUSIVE */;  /* lockstatus() commented out */
        if (lks == 1 /* LK_SHARED */)            /* :203 ALWAYS FALSE */
            pthread_rwlock_wrlock(&g_vnlock);    /* :204 dead */

        if (g_fixed_mode) {
            /* fix.diff: UNCONDITIONAL shared->exclusive upgrade before the
             * directory iteration that mutates per-vnode state.
             *
             * DragonFly lockmgr LK_UPGRADE (kern_lock.c:576) has an explicit
             * anti-deadlock rule: if another upgrade is already pending, the
             * caller RELEASES its shared lock and acquires exclusive
             * normally (kern_lock.c:616-625).  pthread rwlocks have no
             * atomic upgrade and would deadlock if a read-holder calls
             * wrlock, so we model the upgrade as that same safe sequence:
             * drop shared, take exclusive.  After readvdir we reverse it
             * (drop exclusive, take shared) so vn_read's caller still
             * observes a shared hold on return (model of LK_DOWNGRADE). */
            pthread_rwlock_unlock(&g_vnlock);               /* drop shared */
            if (pthread_rwlock_wrlock(&g_vnlock) != 0) {    /* take excl */
                fprintf(stderr, "FIXED: upgrade failed\n");
                return 1;
            }
        }

        error = model_smbfs_readvdir(np, uio);              /* :205 */

        if (g_fixed_mode) {
            pthread_rwlock_unlock(&g_vnlock);               /* drop excl */
            pthread_rwlock_rdlock(&g_vnlock);               /* take shared */
        }
        return error;
    }
    return 0;
}

/* ---- vnode lock acquire/release (model of vfs_vnops.c:751) ------------- */
static void
vn_read_enter(void) {           /* vn_read: LK_SHARED */
    pthread_rwlock_rdlock(&g_vnlock);
}
static void
vn_read_leave(void) {
    pthread_rwlock_unlock(&g_vnlock);
}

/* ---- worker threads ---------------------------------------------------- */
struct job_arg { struct smbnode *np; long offset; long resid; int rc; };

static void *
thread_read(void *v)
{
    struct job_arg *a = v;
    struct uio uio = { a->offset, a->resid };

    vn_read_enter();                       /* vfs_vnops.c:751 LK_SHARED */
    a->rc = model_smbfs_readvnode(a->np, &uio);
    vn_read_leave();
    return NULL;
}

int
main(int argc, char **argv)
{
    struct smbnode node;
    pthread_t ta, tb;
    struct job_arg a_arg = { &node, 0, 64, -1 };
    struct job_arg b_arg = { &node, 0, 64, -1 };

    if (argc >= 2 && strcmp(argv[1], "fixed") == 0)
        g_fixed_mode = 1;

    pthread_rwlock_init(&g_vnlock, NULL);

    memset(&node, 0, sizeof(node));
    node.n_dirseq = NULL;
    node.n_dirofs = 0;
    node.n_ino    = 2;

    printf("[*] DF-0884 harness: mode=%s\n", g_fixed_mode ? "FIXED" : "BUGGY");
    printf("[*] Thread A: read(2) on VDIR at offset 0 (will block in smbfs_findnext)\n");
    printf("[*] Thread B: read(2) on VDIR at offset 0 (will reopen -> findclose A's ctx)\n");

    /* Start Thread A; it will enter smbfs_findnext, announce it is blocked,
     * and wait for the orchestrator release. */
    pthread_create(&ta, NULL, thread_read, &a_arg);

    /* Wait until Thread A is blocked "in network I/O" holding ctx. */
    pthread_mutex_lock(&g_netio_mtx);
    while (!g_netio_thread_a_blocked)
        pthread_cond_wait(&g_netio_cond, &g_netio_mtx);
    pthread_mutex_unlock(&g_netio_mtx);
    printf("[*] Thread A is now blocked inside smbfs_findnext() holding ctx (== n_dirseq)\n");
    printf("[*]   n_dirseq=%p n_dirofs=%ld\n", (void*)node.n_dirseq, node.n_dirofs);

    /* Snapshot the ctx pointer Thread A is using. */
    struct smbfs_fctx *a_ctx = node.n_dirseq;

    /* Start Thread B.  In BUGGY mode B's read holds the lock SHARED (A does
     * too, because the upgrade is dead code), so B enters smbfs_readvdir
     * concurrently, takes the REOPEN branch (B offset 0 != n_dirofs 2), and
     * calls smbfs_findclose(n_dirseq = A's ctx) -> kfree(A's ctx).  In FIXED
     * mode B blocks on the write-lock upgrade until A is done. */
    pthread_create(&tb, NULL, thread_read, &b_arg);

    if (!g_fixed_mode) {
        /* Give B time to run findclose() on A's ctx.  B will complete its
         * readvdir and release the shared lock. */
        usleep(100 * 1000);
        int fr = poison_is_freed(a_ctx);
        printf("[*] After Thread B ran: A's ctx=%p freed? %s (poison=0x%02X)\n",
               (void*)a_ctx, fr == 1 ? "YES" : (fr == 0 ? "no" : "?"),
               POISON_BYTE);
        if (fr != 1) {
            printf("[!] UNEXPECTED: B did not free A's ctx; harness mis-modelled\n");
            pthread_mutex_lock(&g_netio_mtx);
            g_netio_release_a = 1;
            pthread_cond_broadcast(&g_netio_cond);
            pthread_mutex_unlock(&g_netio_mtx);
            pthread_join(ta, NULL); pthread_join(tb, NULL);
            return 5;
        }
        printf("[*] Releasing Thread A from smbfs_findnext() -- it will now write ctx->f_attr.fa_ino THROUGH FREED MEMORY\n");
    } else {
        printf("[*] FIXED mode: Thread B is blocked on the exclusive upgrade "
               "(A still holds shared); A must finish readvdir first.\n");
        printf("[*] Releasing Thread A so it can complete smbfs_readvdir() "
               "and only THEN drop the shared lock for B.\n");
    }

    /* Release Thread A.  In BUGGY mode: A resumes its post-I/O write through
     * the already-freed ctx (UAF).  In FIXED mode: A completes readvdir
     * normally and releases the shared lock, after which B's upgrade wins
     * and B runs readvdir on its own ctx (no race). */
    pthread_mutex_lock(&g_netio_mtx);
    g_netio_release_a = 1;
    pthread_cond_broadcast(&g_netio_cond);
    pthread_mutex_unlock(&g_netio_mtx);

    pthread_join(ta, NULL);
    pthread_join(tb, NULL);

    printf("[*] Thread A rc=%d  Thread B rc=%d\n", a_arg.rc, b_arg.rc);

    /* ---- verdict ------------------------------------------------------- */
    if (!g_fixed_mode) {
        /* g_a_write_saw_freed captures whether A's ctx was already freed at
         * the instant A resumed and wrote fa_ino.  In buggy mode B's
         * findclose() ran while A was blocked, so it is 1 (freed) -> UAF. */
        if (g_a_write_saw_freed == 1) {
            printf("[+] A wrote ctx->f_attr.fa_ino (0x%lx) into an object "
                   "that was ALREADY freed by Thread B's smbfs_findclose()\n",
                   (unsigned long)a_ctx->f_attr.fa_ino);
            printf(">>> UAF CONFIRMED: smbfs_findnext wrote through freed "
                   "smbfs_fctx (dead-code lock upgrade lets two read(2) on a "
                   "VDIR run smbfs_readvdir concurrently)\n");
            return 0;
        }
        printf("[!] UAF NOT observed (g_a_write_saw_freed=%d, harness error)\n",
               g_a_write_saw_freed);
        return 6;
    } else {
        /* In fixed mode the upgrade serializes A and B; A completes its full
         * readvdir (findnext -> dirent) BEFORE B can enter, so B never frees
         * A's in-flight ctx.  g_a_write_saw_freed must be 0 (not freed at
         * A's write instant). */
        if (g_a_write_saw_freed == 0) {
            printf("[+] FIXED: at A's write instant, ctx was still owned by A "
                   "(g_a_write_saw_freed=0); the exclusive upgrade serialized "
                   "the two readers so B's findclose ran only after A finished\n");
            printf(">>> FIXED: no UAF -- unconditional LK_UPGRADE before "
                   "smbfs_readvdir closes the race\n");
            return 0;
        }
        printf("[!] FIXED mode but A's ctx was freed at write instant "
               "(g_a_write_saw_freed=%d) -- fix ineffective\n",
               g_a_write_saw_freed);
        return 7;
    }
}