@samitouri / QOSamiQemu / commits / a94a1d7699

fuse: Manually process requests (without libfuse)

Manually read requests from the /dev/fuse FD and process them, without using libfuse. This allows us to safely add parallel request processing in coroutines later, without having to worry about libfuse internals. (Technically, we already have exactly that problem with read_from_fuse_export()/read_from_fuse_fd() nesting.) We will continue to use libfuse for mounting the filesystem; fusermount3 is a effectively a helper program of libfuse, so it should know best how to interact with it. (Doing it manually without libfuse, while doable, is a bit of a pain, and it is not clear to me how stable the "protocol" actually is.) Take this opportunity of quite a major rewrite to update the Copyright line with corrected information that has surfaced in the meantime. Here are some benchmarks from before this patch (4k, iodepth=16, libaio; except 'sync', which are iodepth=1 and pvsync2): file: read: seq aio: 99.8k ±1.5k IOPS rand aio: 50.5k ±1.0k seq sync: 36.1k ±1.1k rand sync: 10.0k ±0.1k write: seq aio: 72.0k ±9.3k rand aio: 70.6k ±2.5k seq sync: 30.6k ±0.8k rand sync: 30.1k ±1.0k null: read: seq aio: 157.9k ±4.7k rand aio: 158.7k ±4.8k seq sync: 80.2k ±2.8k rand sync: 77.5k ±3.8k write: seq aio: 154.3k ±3.6k rand aio: 154.3k ±4.2k seq sync: 76.1k ±5.2k rand sync: 72.9k ±4.0k And with this patch applied: file: read: seq aio: 106.8k ±1.9k (+7%) rand aio: 48.3k ±8.8k (-4%) seq sync: 35.5k ±1.4k (-2%) rand sync: 10.0k ±0.2k (±0%) write: seq aio: 76.3k ±6.6k (+6%) rand aio: 76.4k ±1.5k (+8%) seq sync: 31.6k ±0.6k (+3%) rand sync: 30.9k ±0.8k (+3%) null: read: seq aio: 161.7k ±6.0k (+2%) rand aio: 165.6k ±7.1k (+4%) seq sync: 80.5k ±3.0k (±0%) rand sync: 78.5k ±3.1k (+1%) write: seq aio: 185.1k ±3.3k (+20%) rand aio: 186.7k ±4.8k (+21%) seq sync: 82.5k ±4.2k (+8%) rand sync: 78.7k ±3.2k (+8%) So not much difference, aside from write AIO to a null-co export getting a bit better. Signed-off-by: Hanna Czenczek <hreitz@redhat.com> Message-ID: <20260309150856.26800-18-hreitz@redhat.com> Reviewed-by: Kevin Wolf <kwolf@redhat.com> Signed-off-by: Kevin Wolf <kwolf@redhat.com>

Hanna Czenczek committed Mar 9, 2026 at 16:08 UTC a94a1d76990bdc25eeb43f22888d306fd57fcdb6
1 file changed +721 -224
block/export/fuse.c
+721 -224
@@ -1,7 +1,7 @@
1 /*
2 * Present a block device as a raw image through FUSE
3 *
4 - * Copyright (c) 2020 Max Reitz <mreitz@redhat.com>
4 + * Copyright (c) 2020, 2025 Hanna Czenczek <hreitz@redhat.com>
5 *
6 * This program is free software; you can redistribute it and/or modify
7 * it under the terms of the GNU General Public License as published by
@@ -27,12 +27,15 @@
27 #include "block/qapi.h"
28 #include "qapi/error.h"
29 #include "qapi/qapi-commands-block.h"
30 +#include "qemu/error-report.h"
31 #include "qemu/main-loop.h"
32 #include "system/block-backend.h"
33
34 #include <fuse.h>
35 #include <fuse_lowlevel.h>
36
37 +#include "standard-headers/linux/fuse.h"
38 +
39 #if defined(CONFIG_FALLOCATE_ZERO_RANGE)
40 #include <linux/falloc.h>
41 #endif
@@ -42,17 +45,101 @@
45 #endif
46
47 /* Prevent overly long bounce buffer allocations */
45 -#define FUSE_MAX_BOUNCE_BYTES (MIN(BDRV_REQUEST_MAX_BYTES, 64 * 1024 * 1024))
48 +#define FUSE_MAX_READ_BYTES (MIN(BDRV_REQUEST_MAX_BYTES, 64 * 1024 * 1024))
49 +#define FUSE_MAX_WRITE_BYTES (64 * 1024)
50 +
51 +/*
52 + * fuse_init_in structure before 7.36. We don't need the flags2 field added
53 + * there, so we can work with the smaller older structure to stay compatible
54 + * with older kernels.
55 + */
56 +struct fuse_init_in_compat {
57 + uint32_t major;
58 + uint32_t minor;
59 + uint32_t max_readahead;
60 + uint32_t flags;
61 +};
62 +
63 +typedef struct FuseRequestInHeader {
64 + struct fuse_in_header common;
65 + /* All supported requests */
66 + union {
67 + struct fuse_init_in_compat init;
68 + struct fuse_open_in open;
69 + struct fuse_setattr_in setattr;
70 + struct fuse_read_in read;
71 + struct fuse_write_in write;
72 + struct fuse_fallocate_in fallocate;
73 +#ifdef CONFIG_FUSE_LSEEK
74 + struct fuse_lseek_in lseek;
75 +#endif
76 + };
77 +} FuseRequestInHeader;
78 +
79 +typedef struct FuseRequestOutHeader {
80 + struct fuse_out_header common;
81 + /* All supported requests */
82 + union {
83 + struct fuse_init_out init;
84 + struct fuse_statfs_out statfs;
85 + struct fuse_open_out open;
86 + struct fuse_attr_out attr;
87 + struct fuse_write_out write;
88 +#ifdef CONFIG_FUSE_LSEEK
89 + struct fuse_lseek_out lseek;
90 +#endif
91 + };
92 +} FuseRequestOutHeader;
93 +
94 +typedef union FuseRequestInHeaderBuf {
95 + struct FuseRequestInHeader structured;
96 + struct {
97 + /*
98 + * Part of the request header that is filled for write requests
99 + * (Needed because we want the data to go into a different buffer, to
100 + * avoid having to use a bounce buffer)
101 + */
102 + char head[sizeof(struct fuse_in_header) +
103 + sizeof(struct fuse_write_in)];
104 + /*
105 + * Rest of the request header for requests that have a longer header
106 + * than write requests
107 + */
108 + char tail[sizeof(FuseRequestInHeader) -
109 + (sizeof(struct fuse_in_header) +
110 + sizeof(struct fuse_write_in))];
111 + };
112 +} FuseRequestInHeaderBuf;
113
114 +QEMU_BUILD_BUG_ON(sizeof(FuseRequestInHeaderBuf) !=
115 + sizeof(FuseRequestInHeader));
116 +QEMU_BUILD_BUG_ON(sizeof(((FuseRequestInHeaderBuf *)0)->head) +
117 + sizeof(((FuseRequestInHeaderBuf *)0)->tail) !=
118 + sizeof(FuseRequestInHeader));
119
120 typedef struct FuseExport {
121 BlockExport common;
122
123 struct fuse_session *fuse_session;
52 - struct fuse_buf fuse_buf;
124 unsigned int in_flight; /* atomic */
125 bool mounted, fd_handler_set_up;
126
127 + /*
128 + * Cached buffer to receive the data of WRITE requests. Cached because:
129 + * To read requests, we put a FuseRequestInHeaderBuf (FRIHB) object on the
130 + * stack, and a (WRITE data) buffer on the heap. We pass FRIHB.head and the
131 + * data buffer to readv(). This way, for WRITE requests, we get exactly
132 + * their data in the data buffer and can avoid bounce buffering.
133 + * However, for non-WRITE requests, some of the header may end up in the
134 + * data buffer, so we will need to copy that back into the FRIHB object, and
135 + * then we don't need the heap buffer anymore. That is why we cache it, so
136 + * we can trivially reuse it between non-WRITE requests.
137 + *
138 + * Note that these data buffers and thus req_write_data_cached are allocated
139 + * via blk_blockalign() and thus need to be freed via qemu_vfree().
140 + */
141 + void *req_write_data_cached;
142 +
143 /*
144 * Set when there was an unrecoverable error and no requests should be read
145 * from the device anymore (basically only in case of something we would
@@ -60,6 +147,8 @@ typedef struct FuseExport {
147 */
148 bool halted;
149
150 + int fuse_fd;
151 +
152 char *mountpoint;
153 bool writable;
154 bool growable;
@@ -71,20 +160,31 @@ typedef struct FuseExport {
160 gid_t st_gid;
161 } FuseExport;
162
163 +/*
164 + * Verify that the size of FuseRequestInHeaderBuf.head plus the data
165 + * buffer are big enough to be accepted by the FUSE kernel driver.
166 + */
167 +QEMU_BUILD_BUG_ON(sizeof(((FuseRequestInHeaderBuf *)0)->head) +
168 + FUSE_MAX_WRITE_BYTES <
169 + FUSE_MIN_READ_BUFFER);
170 +
171 static GHashTable *exports;
75 -static const struct fuse_lowlevel_ops fuse_ops;
172
173 static void fuse_export_shutdown(BlockExport *exp);
174 static void fuse_export_delete(BlockExport *exp);
79 -static void fuse_export_halt(FuseExport *exp) G_GNUC_UNUSED;
175 +static void fuse_export_halt(FuseExport *exp);
176
177 static void init_exports_table(void);
178
179 static int mount_fuse_export(FuseExport *exp, Error **errp);
84 -static void read_from_fuse_export(void *opaque);
180
181 static bool is_regular_file(const char *path, Error **errp);
182
183 +static void read_from_fuse_fd(void *opaque);
184 +static void fuse_process_request(FuseExport *exp,
185 + const FuseRequestInHeader *in_hdr,
186 + const void *data_buffer);
187 +static int fuse_write_err(int fd, const struct fuse_in_header *in_hdr, int err);
188
189 static void fuse_inc_in_flight(FuseExport *exp)
190 {
@@ -105,22 +205,26 @@ static void fuse_dec_in_flight(FuseExport *exp)
205 }
206 }
207
208 +/**
209 + * Attach FUSE FD read handler.
210 + */
211 static void fuse_attach_handlers(FuseExport *exp)
212 {
213 if (qatomic_read(&exp->halted)) {
214 return;
215 }
216
114 - aio_set_fd_handler(exp->common.ctx,
115 - fuse_session_fd(exp->fuse_session),
116 - read_from_fuse_export, NULL, NULL, NULL, exp);
217 + aio_set_fd_handler(exp->common.ctx, exp->fuse_fd,
218 + read_from_fuse_fd, NULL, NULL, NULL, exp);
219 exp->fd_handler_set_up = true;
220 }
221
222 +/**
223 + * Detach FUSE FD read handler.
224 + */
225 static void fuse_detach_handlers(FuseExport *exp)
226 {
122 - aio_set_fd_handler(exp->common.ctx,
123 - fuse_session_fd(exp->fuse_session),
227 + aio_set_fd_handler(exp->common.ctx, exp->fuse_fd,
228 NULL, NULL, NULL, NULL, NULL);
229 exp->fd_handler_set_up = false;
230 }
@@ -247,6 +351,13 @@ static int fuse_export_create(BlockExport *blk_exp,
351
352 g_hash_table_insert(exports, g_strdup(exp->mountpoint), NULL);
353
354 + exp->fuse_fd = fuse_session_fd(exp->fuse_session);
355 + ret = qemu_fcntl_addfl(exp->fuse_fd, O_NONBLOCK);
356 + if (ret < 0) {
357 + error_setg_errno(errp, -ret, "Failed to make FUSE FD non-blocking");
358 + goto fail;
359 + }
360 +
361 fuse_attach_handlers(exp);
362 return 0;
363
@@ -278,6 +389,17 @@ static int mount_fuse_export(FuseExport *exp, Error **errp)
389 char *mount_opts;
390 struct fuse_args fuse_args;
391 int ret;
392 + /*
393 + * We just create the session for mounting/unmounting, no need to provide
394 + * any operations. However, since libfuse commit 52a633a5d, we have to
395 + * provide some op struct and cannot just pass NULL (even though the commit
396 + * message ("allow passing ops as NULL") seems to imply the exact opposite,
397 + * as does the comment added to fuse_session_new_fn() ("To create a no-op
398 + * session just for mounting pass op as NULL.").
399 + * This is how said libfuse commit implements a no-op session internally, so
400 + * do it the same way.
401 + */
402 + static const struct fuse_lowlevel_ops null_ops = { 0 };
403
404 /*
405 * Note that these mount options differ from what we would pass to a direct
@@ -292,7 +414,7 @@ static int mount_fuse_export(FuseExport *exp, Error **errp)
414 mount_opts = g_strdup_printf("%s,nosuid,nodev,noatime,max_read=%zu,"
415 "default_permissions%s",
416 exp->writable ? "rw" : "ro",
295 - FUSE_MAX_BOUNCE_BYTES,
417 + FUSE_MAX_READ_BYTES,
418 exp->allow_other ? ",allow_other" : "");
419
420 fuse_argv[0] = ""; /* Dummy program name */
@@ -301,8 +423,8 @@ static int mount_fuse_export(FuseExport *exp, Error **errp)
423 fuse_argv[3] = NULL;
424 fuse_args = (struct fuse_args)FUSE_ARGS_INIT(3, (char **)fuse_argv);
425
304 - exp->fuse_session = fuse_session_new(&fuse_args, &fuse_ops,
305 - sizeof(fuse_ops), exp);
426 + exp->fuse_session = fuse_session_new(&fuse_args, &null_ops,
427 + sizeof(null_ops), NULL);
428 g_free(mount_opts);
429 if (!exp->fuse_session) {
430 error_setg(errp, "Failed to set up FUSE session");
@@ -326,36 +448,163 @@ fail:
448 }
449
450 /**
329 - * Callback to be invoked when the FUSE session FD can be read from.
330 - * (This is basically the FUSE event loop.)
451 + * Allocate a buffer to receive WRITE data, or take the cached one.
452 */
332 -static void read_from_fuse_export(void *opaque)
453 +static void *get_write_data_buffer(FuseExport *exp)
454 {
334 - FuseExport *exp = opaque;
335 - int ret;
455 + if (exp->req_write_data_cached) {
456 + void *cached = exp->req_write_data_cached;
457 + exp->req_write_data_cached = NULL;
458 + return cached;
459 + } else {
460 + return blk_blockalign(exp->common.blk, FUSE_MAX_WRITE_BYTES);
461 + }
462 +}
463
337 - if (unlikely(qatomic_read(&exp->halted))) {
464 +/**
465 + * Release a WRITE data buffer, possibly reusing it for a subsequent request.
466 + */
467 +static void release_write_data_buffer(FuseExport *exp, void **buffer)
468 +{
469 + if (!*buffer) {
470 return;
471 }
472
473 + if (!exp->req_write_data_cached) {
474 + exp->req_write_data_cached = *buffer;
475 + } else {
476 + qemu_vfree(*buffer);
477 + }
478 + *buffer = NULL;
479 +}
480 +
481 +/**
482 + * Return the length of the specific operation's own in_header.
483 + * Return -ENOSYS if the operation is not supported.
484 + */
485 +static ssize_t req_op_hdr_len(const FuseRequestInHeader *in_hdr)
486 +{
487 + switch (in_hdr->common.opcode) {
488 + case FUSE_INIT:
489 + return sizeof(in_hdr->init);
490 + case FUSE_OPEN:
491 + return sizeof(in_hdr->open);
492 + case FUSE_SETATTR:
493 + return sizeof(in_hdr->setattr);
494 + case FUSE_READ:
495 + return sizeof(in_hdr->read);
496 + case FUSE_WRITE:
497 + return sizeof(in_hdr->write);
498 + case FUSE_FALLOCATE:
499 + return sizeof(in_hdr->fallocate);
500 +#ifdef CONFIG_FUSE_LSEEK
501 + case FUSE_LSEEK:
502 + return sizeof(in_hdr->lseek);
503 +#endif
504 + case FUSE_DESTROY:
505 + case FUSE_STATFS:
506 + case FUSE_RELEASE:
507 + case FUSE_LOOKUP:
508 + case FUSE_FORGET:
509 + case FUSE_BATCH_FORGET:
510 + case FUSE_GETATTR:
511 + case FUSE_FSYNC:
512 + case FUSE_FLUSH:
513 + /* These requests don't have their own header or we don't care */
514 + return 0;
515 + default:
516 + return -ENOSYS;
517 + }
518 +}
519 +
520 +/**
521 + * Try to read and process a single request from the FUSE FD.
522 + */
523 +static void read_from_fuse_fd(void *opaque)
524 +{
525 + FuseExport *exp = opaque;
526 + int fuse_fd = exp->fuse_fd;
527 + ssize_t ret;
528 + FuseRequestInHeaderBuf in_hdr_buf;
529 + const FuseRequestInHeader *in_hdr;
530 + void *data_buffer = NULL;
531 + struct iovec iov[2];
532 + ssize_t op_hdr_len;
533 +
534 fuse_inc_in_flight(exp);
535
343 - do {
344 - ret = fuse_session_receive_buf(exp->fuse_session, &exp->fuse_buf);
345 - } while (ret == -EINTR);
346 - if (ret < 0) {
347 - goto out;
536 + if (unlikely(qatomic_read(&exp->halted))) {
537 + goto no_request;
538 + }
539 +
540 + data_buffer = get_write_data_buffer(exp);
541 +
542 + /* Construct the I/O vector to hold the FUSE request */
543 + iov[0] = (struct iovec) { &in_hdr_buf.head, sizeof(in_hdr_buf.head) };
544 + iov[1] = (struct iovec) { data_buffer, FUSE_MAX_WRITE_BYTES };
545 + ret = RETRY_ON_EINTR(readv(fuse_fd, iov, ARRAY_SIZE(iov)));
546 + if (ret < 0 && errno == EAGAIN) {
547 + /* No request available */
548 + goto no_request;
549 + } else if (unlikely(ret < 0)) {
550 + error_report("Failed to read from FUSE device: %s", strerror(errno));
551 + goto no_request;
552 + }
553 +
554 + if (unlikely(ret < sizeof(in_hdr->common))) {
555 + error_report("Incomplete read from FUSE device, expected at least %zu "
556 + "bytes, read %zi bytes; cannot trust subsequent "
557 + "requests, halting the export",
558 + sizeof(in_hdr->common), ret);
559 + fuse_export_halt(exp);
560 + goto no_request;
561 + }
562 + in_hdr = &in_hdr_buf.structured;
563 +
564 + if (unlikely(ret != in_hdr->common.len)) {
565 + error_report("Number of bytes read from FUSE device does not match "
566 + "request size, expected %" PRIu32 " bytes, read %zi "
567 + "bytes; cannot trust subsequent requests, halting the "
568 + "export",
569 + in_hdr->common.len, ret);
570 + fuse_export_halt(exp);
571 + goto no_request;
572 + }
573 +
574 + op_hdr_len = req_op_hdr_len(in_hdr);
575 + if (op_hdr_len < 0) {
576 + fuse_write_err(fuse_fd, &in_hdr->common, op_hdr_len);
577 + goto no_request;
578 + }
579 +
580 + if (unlikely(ret < sizeof(in_hdr->common) + op_hdr_len)) {
581 + error_report("FUSE request truncated, expected %zu bytes, read %zi "
582 + "bytes",
583 + sizeof(in_hdr->common) + op_hdr_len, ret);
584 + fuse_write_err(fuse_fd, &in_hdr->common, -EINVAL);
585 + goto no_request;
586 }
587
588 /*
351 - * Note that aio_poll() in any request-processing function can lead to a
352 - * nested read_from_fuse_export() call, which will overwrite the contents of
353 - * exp->fuse_buf. Anything that takes a buffer needs to take care that the
354 - * content is copied before potentially polling via aio_poll().
589 + * Only WRITE uses the write data buffer, so for non-WRITE requests longer
590 + * than .head, we need to copy any data that spilled into data_buffer into
591 + * .tail. Then we can release the write data buffer.
592 */
356 - fuse_session_process_buf(exp->fuse_session, &exp->fuse_buf);
593 + if (in_hdr->common.opcode != FUSE_WRITE) {
594 + if (ret > sizeof(in_hdr_buf.head)) {
595 + size_t len;
596 + /* Limit size to prevent overflow */
597 + len = MIN(ret - sizeof(in_hdr_buf.head), sizeof(in_hdr_buf.tail));
598 + memcpy(in_hdr_buf.tail, data_buffer, len);
599 + }
600 +
601 + release_write_data_buffer(exp, &data_buffer);
602 + }
603
358 -out:
604 + fuse_process_request(exp, in_hdr, data_buffer);
605 +
606 +no_request:
607 + release_write_data_buffer(exp, &data_buffer);
608 fuse_dec_in_flight(exp);
609 }
610
@@ -363,18 +612,14 @@ static void fuse_export_shutdown(BlockExport *blk_exp)
612 {
613 FuseExport *exp = container_of(blk_exp, FuseExport, common);
614
366 - if (exp->fuse_session) {
367 - fuse_session_exit(exp->fuse_session);
368 -
369 - if (exp->fd_handler_set_up) {
370 - fuse_detach_handlers(exp);
371 - }
615 + if (exp->fd_handler_set_up) {
616 + fuse_detach_handlers(exp);
617 }
618
619 if (exp->mountpoint) {
620 /*
376 - * Safe to drop now, because we will not handle any requests
377 - * for this export anymore anyway.
621 + * Safe to drop now, because we will not handle any requests for this
622 + * export anymore anyway (at least not from the main thread).
623 */
624 g_hash_table_remove(exports, exp->mountpoint);
625 }
@@ -392,7 +637,7 @@ static void fuse_export_delete(BlockExport *blk_exp)
637 fuse_session_destroy(exp->fuse_session);
638 }
639
395 - free(exp->fuse_buf.mem);
640 + qemu_vfree(exp->req_write_data_cached);
641 g_free(exp->mountpoint);
642 }
643
@@ -434,46 +679,101 @@ static bool is_regular_file(const char *path, Error **errp)
679 }
680
681 /**
437 - * A chance to set change some parameters supplied to FUSE_INIT.
682 + * Process FUSE INIT.
683 + * Return the number of bytes written to *out on success, and -errno on error.
684 */
439 -static void fuse_init(void *userdata, struct fuse_conn_info *conn)
685 +static ssize_t fuse_init(FuseExport *exp, struct fuse_init_out *out,
686 + const struct fuse_init_in_compat *in)
687 {
688 + const uint32_t supported_flags = FUSE_ASYNC_READ | FUSE_ASYNC_DIO;
689 +
690 + if (in->major != 7) {
691 + error_report("FUSE major version mismatch: We have 7, but kernel has %"
692 + PRIu32, in->major);
693 + return -EINVAL;
694 + }
695 +
696 + /* 2007's 7.9 added fuse_attr.blksize; working around that would be hard */
697 + if (in->minor < 9) {
698 + error_report("FUSE minor version too old: 9 required, but kernel has %"
699 + PRIu32, in->minor);
700 + return -EINVAL;
701 + }
702 +
703 + *out = (struct fuse_init_out) {
704 + .major = 7,
705 + .minor = MIN(FUSE_KERNEL_MINOR_VERSION, in->minor),
706 + .max_readahead = in->max_readahead,
707 + .max_write = FUSE_MAX_WRITE_BYTES,
708 + .flags = in->flags & supported_flags,
709 + .flags2 = 0,
710 +
711 + /* libfuse maximum: 2^16 - 1 */
712 + .max_background = UINT16_MAX,
713 +
714 + /* libfuse default: max_background * 3 / 4 */
715 + .congestion_threshold = (int)UINT16_MAX * 3 / 4,
716 +
717 + /* libfuse default: 1 */
718 + .time_gran = 1,
719 +
720 + /*
721 + * probably unneeded without FUSE_MAX_PAGES, but this would be the
722 + * libfuse default
723 + */
724 + .max_pages = DIV_ROUND_UP(FUSE_MAX_WRITE_BYTES,
725 + qemu_real_host_page_size()),
726 +
727 + /* Only needed for mappings (i.e. DAX) */
728 + .map_alignment = 0,
729 + };
730 +
731 /*
442 - * MIN_NON_ZERO() would not be wrong here, but what we set here
443 - * must equal what has been passed to fuse_session_new().
444 - * Therefore, as long as max_read must be passed as a mount option
445 - * (which libfuse claims will be changed at some point), we have
446 - * to set max_read to a fixed value here.
732 + * Before 7.23, fuse_init_out is shorter.
733 + * Drop the tail (time_gran, max_pages, map_alignment).
734 */
448 - conn->max_read = FUSE_MAX_BOUNCE_BYTES;
449 -
450 - conn->max_write = MIN_NON_ZERO(BDRV_REQUEST_MAX_BYTES, conn->max_write);
735 + return out->minor >= 23 ? sizeof(*out) : FUSE_COMPAT_22_INIT_OUT_SIZE;
736 }
737
738 /**
454 - * Let clients look up files. Always return ENOENT because we only
455 - * care about the mountpoint itself.
739 + * Return some filesystem information, just to not break e.g. `df`.
740 */
457 -static void fuse_lookup(fuse_req_t req, fuse_ino_t parent, const char *name)
741 +static ssize_t fuse_statfs(FuseExport *exp, struct fuse_statfs_out *out)
742 {
459 - fuse_reply_err(req, ENOENT);
743 + BlockDriverState *root_bs;
744 + uint32_t opt_transfer = 512;
745 +
746 + root_bs = blk_bs(exp->common.blk);
747 + if (root_bs) {
748 + opt_transfer = root_bs->bl.opt_transfer;
749 + if (!opt_transfer) {
750 + opt_transfer = root_bs->bl.request_alignment;
751 + }
752 + opt_transfer = MAX(opt_transfer, 512);
753 + }
754 +
755 + *out = (struct fuse_statfs_out) {
756 + /* These are the fields libfuse sets by default */
757 + .st = {
758 + .namelen = 255,
759 + .bsize = opt_transfer,
760 + },
761 + };
762 + return sizeof(*out);
763 }
764
765 /**
766 * Let clients get file attributes (i.e., stat() the file).
767 + * Return the number of bytes written to *out on success, and -errno on error.
768 */
465 -static void fuse_getattr(fuse_req_t req, fuse_ino_t inode,
466 - struct fuse_file_info *fi)
769 +static ssize_t fuse_getattr(FuseExport *exp, struct fuse_attr_out *out)
770 {
468 - struct stat statbuf;
771 int64_t length, allocated_blocks;
772 time_t now = time(NULL);
471 - FuseExport *exp = fuse_req_userdata(req);
773
774 length = blk_getlength(exp->common.blk);
775 if (length < 0) {
475 - fuse_reply_err(req, -length);
476 - return;
776 + return length;
777 }
778
779 allocated_blocks = bdrv_get_allocated_file_size(blk_bs(exp->common.blk));
@@ -483,21 +783,24 @@ static void fuse_getattr(fuse_req_t req, fuse_ino_t inode,
783 allocated_blocks = DIV_ROUND_UP(allocated_blocks, 512);
784 }
785
486 - statbuf = (struct stat) {
487 - .st_ino = 1,
488 - .st_mode = exp->st_mode,
489 - .st_nlink = 1,
490 - .st_uid = exp->st_uid,
491 - .st_gid = exp->st_gid,
492 - .st_size = length,
493 - .st_blksize = blk_bs(exp->common.blk)->bl.request_alignment,
494 - .st_blocks = allocated_blocks,
495 - .st_atime = now,
496 - .st_mtime = now,
497 - .st_ctime = now,
786 + *out = (struct fuse_attr_out) {
787 + .attr_valid = 1,
788 + .attr = {
789 + .ino = 1,
790 + .mode = exp->st_mode,
791 + .nlink = 1,
792 + .uid = exp->st_uid,
793 + .gid = exp->st_gid,
794 + .size = length,
795 + .blksize = blk_bs(exp->common.blk)->bl.request_alignment,
796 + .blocks = allocated_blocks,
797 + .atime = now,
798 + .mtime = now,
799 + .ctime = now,
800 + },
801 };
802
500 - fuse_reply_attr(req, &statbuf, 1.);
803 + return sizeof(*out);
804 }
805
806 static int fuse_do_truncate(const FuseExport *exp, int64_t size,
@@ -520,101 +823,99 @@ static int fuse_do_truncate(const FuseExport *exp, int64_t size,
823 * permit access: Read-only exports cannot be given +w, and exports
824 * without allow_other cannot be given a different UID or GID, and
825 * they cannot be given non-owner access.
826 + * Return the number of bytes written to *out on success, and -errno on error.
827 */
524 -static void fuse_setattr(fuse_req_t req, fuse_ino_t inode, struct stat *statbuf,
525 - int to_set, struct fuse_file_info *fi)
828 +static ssize_t fuse_setattr(FuseExport *exp, struct fuse_attr_out *out,
829 + uint32_t to_set, uint64_t size, uint32_t mode,
830 + uint32_t uid, uint32_t gid)
831 {
527 - FuseExport *exp = fuse_req_userdata(req);
832 int supported_attrs;
833 int ret;
834
531 - supported_attrs = FUSE_SET_ATTR_SIZE | FUSE_SET_ATTR_MODE;
835 + /* SIZE and MODE are actually supported, the others can be safely ignored */
836 + supported_attrs = FATTR_SIZE | FATTR_MODE |
837 + FATTR_FH | FATTR_LOCKOWNER | FATTR_KILL_SUIDGID;
838 if (exp->allow_other) {
533 - supported_attrs |= FUSE_SET_ATTR_UID | FUSE_SET_ATTR_GID;
839 + supported_attrs |= FATTR_UID | FATTR_GID;
840 }
841
842 if (to_set & ~supported_attrs) {
537 - fuse_reply_err(req, ENOTSUP);
538 - return;
843 + return -ENOTSUP;
844 }
845
846 /* Do some argument checks first before committing to anything */
542 - if (to_set & FUSE_SET_ATTR_MODE) {
847 + if (to_set & FATTR_MODE) {
848 /*
849 * Without allow_other, non-owners can never access the export, so do
850 * not allow setting permissions for them
851 */
547 - if (!exp->allow_other &&
548 - (statbuf->st_mode & (S_IRWXG | S_IRWXO)) != 0)
549 - {
550 - fuse_reply_err(req, EPERM);
551 - return;
852 + if (!exp->allow_other && (mode & (S_IRWXG | S_IRWXO)) != 0) {
853 + return -EPERM;
854 }
855
856 /* +w for read-only exports makes no sense, disallow it */
555 - if (!exp->writable &&
556 - (statbuf->st_mode & (S_IWUSR | S_IWGRP | S_IWOTH)) != 0)
557 - {
558 - fuse_reply_err(req, EROFS);
559 - return;
857 + if (!exp->writable && (mode & (S_IWUSR | S_IWGRP | S_IWOTH)) != 0) {
858 + return -EROFS;
859 }
860 }
861
563 - if (to_set & FUSE_SET_ATTR_SIZE) {
862 + if (to_set & FATTR_SIZE) {
863 if (!exp->writable) {
565 - fuse_reply_err(req, EACCES);
566 - return;
864 + return -EACCES;
865 }
866
569 - ret = fuse_do_truncate(exp, statbuf->st_size, true, PREALLOC_MODE_OFF);
867 + ret = fuse_do_truncate(exp, size, true, PREALLOC_MODE_OFF);
868 if (ret < 0) {
571 - fuse_reply_err(req, -ret);
572 - return;
869 + return ret;
870 }
871 }
872
576 - if (to_set & FUSE_SET_ATTR_MODE) {
873 + if (to_set & FATTR_MODE) {
874 /* Ignore FUSE-supplied file type, only change the mode */
578 - exp->st_mode = (statbuf->st_mode & 07777) | S_IFREG;
875 + exp->st_mode = (mode & 07777) | S_IFREG;
876 }
877
581 - if (to_set & FUSE_SET_ATTR_UID) {
582 - exp->st_uid = statbuf->st_uid;
878 + if (to_set & FATTR_UID) {
879 + exp->st_uid = uid;
880 }
881
585 - if (to_set & FUSE_SET_ATTR_GID) {
586 - exp->st_gid = statbuf->st_gid;
882 + if (to_set & FATTR_GID) {
883 + exp->st_gid = gid;
884 }
885
589 - fuse_getattr(req, inode, fi);
886 + return fuse_getattr(exp, out);
887 }
888
889 /**
593 - * Let clients open a file (i.e., the exported image).
890 + * Open an inode. We only have a single inode in our exported filesystem, so we
891 + * just acknowledge the request.
892 + * Return the number of bytes written to *out on success, and -errno on error.
893 */
595 -static void fuse_open(fuse_req_t req, fuse_ino_t inode,
596 - struct fuse_file_info *fi)
894 +static ssize_t fuse_open(FuseExport *exp, struct fuse_open_out *out)
895 {
598 - fi->direct_io = true;
599 - fi->parallel_direct_writes = true;
600 - fuse_reply_open(req, fi);
896 + *out = (struct fuse_open_out) {
897 + .open_flags = FOPEN_DIRECT_IO | FOPEN_PARALLEL_DIRECT_WRITES,
898 + };
899 + return sizeof(*out);
900 }
901
902 /**
604 - * Handle client reads from the exported image.
903 + * Handle client reads from the exported image. Allocates *bufptr and reads
904 + * data from the block device into that buffer.
905 + * Returns the buffer (read) size on success, and -errno on error.
906 + * Note: If the returned size is 0, *bufptr will be set to NULL.
907 + * After use, *bufptr must be freed via qemu_vfree().
908 */
606 -static void fuse_read(fuse_req_t req, fuse_ino_t inode,
607 - size_t size, off_t offset, struct fuse_file_info *fi)
909 +static ssize_t fuse_read(FuseExport *exp, void **bufptr,
910 + uint64_t offset, uint32_t size)
911 {
609 - FuseExport *exp = fuse_req_userdata(req);
912 int64_t blk_len;
913 void *buf;
914 int ret;
915
916 /* Limited by max_read, should not happen */
615 - if (size > FUSE_MAX_BOUNCE_BYTES) {
616 - fuse_reply_err(req, EINVAL);
617 - return;
917 + if (size > FUSE_MAX_READ_BYTES) {
918 + return -EINVAL;
919 }
920
921 /**
@@ -623,18 +924,13 @@ static void fuse_read(fuse_req_t req, fuse_ino_t inode,
924 */
925 blk_len = blk_getlength(exp->common.blk);
926 if (blk_len < 0) {
626 - fuse_reply_err(req, -blk_len);
627 - return;
927 + return blk_len;
928 }
929
930 if (offset >= blk_len) {
631 - /*
632 - * Technically libfuse does not allow returning a zero error code for
633 - * read requests, but in practice this is a 0-length read (and a future
634 - * commit will change this code anyway)
635 - */
636 - fuse_reply_err(req, 0);
637 - return;
931 + /* Explicitly set to NULL because we return success here */
932 + *bufptr = NULL;
933 + return 0;
934 }
935
936 if (offset + size > blk_len) {
@@ -643,108 +939,96 @@ static void fuse_read(fuse_req_t req, fuse_ino_t inode,
939
940 buf = qemu_try_blockalign(blk_bs(exp->common.blk), size);
941 if (!buf) {
646 - fuse_reply_err(req, ENOMEM);
647 - return;
942 + return -ENOMEM;
943 }
944
945 ret = blk_pread(exp->common.blk, offset, size, buf, 0);
651 - if (ret >= 0) {
652 - fuse_reply_buf(req, buf, size);
653 - } else {
654 - fuse_reply_err(req, -ret);
946 + if (ret < 0) {
947 + qemu_vfree(buf);
948 + return ret;
949 }
950
657 - qemu_vfree(buf);
951 + *bufptr = buf;
952 + return size;
953 }
954
955 /**
661 - * Handle client writes to the exported image.
956 + * Handle client writes to the exported image. @buf has the data to be written.
957 + * Return the number of bytes written to *out on success, and -errno on error.
958 */
663 -static void fuse_write(fuse_req_t req, fuse_ino_t inode, const char *buf,
664 - size_t size, off_t offset, struct fuse_file_info *fi)
959 +static ssize_t fuse_write(FuseExport *exp, struct fuse_write_out *out,
960 + uint64_t offset, uint32_t size, const void *buf)
961 {
666 - FuseExport *exp = fuse_req_userdata(req);
667 - QEMU_AUTO_VFREE void *copied = NULL;
962 int64_t blk_len;
963 int ret;
964
965 + QEMU_BUILD_BUG_ON(FUSE_MAX_WRITE_BYTES > BDRV_REQUEST_MAX_BYTES);
966 /* Limited by max_write, should not happen */
672 - if (size > BDRV_REQUEST_MAX_BYTES) {
673 - fuse_reply_err(req, EINVAL);
674 - return;
967 + if (size > FUSE_MAX_WRITE_BYTES) {
968 + return -EINVAL;
969 }
970
971 if (!exp->writable) {
678 - fuse_reply_err(req, EACCES);
679 - return;
972 + return -EACCES;
973 }
974
682 - /*
683 - * Heed the note on read_from_fuse_export(): If we call aio_poll() (which
684 - * any blk_*() I/O function may do), read_from_fuse_export() may be nested,
685 - * overwriting the request buffer content. Therefore, we must copy it here.
686 - */
687 - copied = blk_blockalign(exp->common.blk, size);
688 - memcpy(copied, buf, size);
689 -
975 /**
976 * Clients will expect short writes at EOF, so we have to limit
977 * offset+size to the image length.
978 */
979 blk_len = blk_getlength(exp->common.blk);
980 if (blk_len < 0) {
696 - fuse_reply_err(req, -blk_len);
697 - return;
981 + return blk_len;
982 }
983
984 if (offset >= blk_len && !exp->growable) {
701 - fuse_reply_write(req, 0);
702 - return;
985 + *out = (struct fuse_write_out) {
986 + .size = 0,
987 + };
988 + return sizeof(*out);
989 }
990
991 if (offset + size < offset) {
706 - fuse_reply_err(req, EINVAL);
707 - return;
992 + return -EINVAL;
993 } else if (offset + size > blk_len) {
994 if (exp->growable) {
995 ret = fuse_do_truncate(exp, offset + size, true, PREALLOC_MODE_OFF);
996 if (ret < 0) {
712 - fuse_reply_err(req, -ret);
713 - return;
997 + return ret;
998 }
999 } else {
1000 size = blk_len - offset;
1001 }
1002 }
1003
720 - ret = blk_pwrite(exp->common.blk, offset, size, copied, 0);
721 - if (ret >= 0) {
722 - fuse_reply_write(req, size);
723 - } else {
724 - fuse_reply_err(req, -ret);
1004 + ret = blk_pwrite(exp->common.blk, offset, size, buf, 0);
1005 + if (ret < 0) {
1006 + return ret;
1007 }
1008 +
1009 + *out = (struct fuse_write_out) {
1010 + .size = size,
1011 + };
1012 + return sizeof(*out);
1013 }
1014
1015 /**
1016 * Let clients perform various fallocate() operations.
1017 + * Return 0 on success (no 'out' object), and -errno on error.
1018 */
731 -static void fuse_fallocate(fuse_req_t req, fuse_ino_t inode, int mode,
732 - off_t offset, off_t length,
733 - struct fuse_file_info *fi)
1019 +static ssize_t fuse_fallocate(FuseExport *exp, uint64_t offset, uint64_t length,
1020 + uint32_t mode)
1021 {
735 - FuseExport *exp = fuse_req_userdata(req);
1022 int64_t blk_len;
1023 int ret;
1024
1025 if (!exp->writable) {
740 - fuse_reply_err(req, EACCES);
741 - return;
1026 + return -EACCES;
1027 }
1028
1029 blk_len = blk_getlength(exp->common.blk);
1030 if (blk_len < 0) {
746 - fuse_reply_err(req, -blk_len);
747 - return;
1031 + return blk_len;
1032 }
1033
1034 #ifdef CONFIG_FALLOCATE_PUNCH_HOLE
@@ -756,16 +1040,14 @@ static void fuse_fallocate(fuse_req_t req, fuse_ino_t inode, int mode,
1040 if (!mode) {
1041 /* We can only fallocate at the EOF with a truncate */
1042 if (offset < blk_len) {
759 - fuse_reply_err(req, EOPNOTSUPP);
760 - return;
1043 + return -EOPNOTSUPP;
1044 }
1045
1046 if (offset > blk_len) {
1047 /* No preallocation needed here */
1048 ret = fuse_do_truncate(exp, offset, true, PREALLOC_MODE_OFF);
1049 if (ret < 0) {
767 - fuse_reply_err(req, -ret);
768 - return;
1050 + return ret;
1051 }
1052 }
1053
@@ -775,8 +1057,7 @@ static void fuse_fallocate(fuse_req_t req, fuse_ino_t inode, int mode,
1057 #ifdef CONFIG_FALLOCATE_PUNCH_HOLE
1058 else if (mode & FALLOC_FL_PUNCH_HOLE) {
1059 if (!(mode & FALLOC_FL_KEEP_SIZE)) {
778 - fuse_reply_err(req, EINVAL);
779 - return;
1060 + return -EINVAL;
1061 }
1062
1063 do {
@@ -804,8 +1085,7 @@ static void fuse_fallocate(fuse_req_t req, fuse_ino_t inode, int mode,
1085 ret = fuse_do_truncate(exp, offset + length, false,
1086 PREALLOC_MODE_OFF);
1087 if (ret < 0) {
807 - fuse_reply_err(req, -ret);
808 - return;
1088 + return ret;
1089 }
1090 }
1091
@@ -823,44 +1103,38 @@ static void fuse_fallocate(fuse_req_t req, fuse_ino_t inode, int mode,
1103 ret = -EOPNOTSUPP;
1104 }
1105
826 - fuse_reply_err(req, ret < 0 ? -ret : 0);
1106 + return ret < 0 ? ret : 0;
1107 }
1108
1109 /**
1110 * Let clients fsync the exported image.
1111 + * Return 0 on success (no 'out' object), and -errno on error.
1112 */
832 -static void fuse_fsync(fuse_req_t req, fuse_ino_t inode, int datasync,
833 - struct fuse_file_info *fi)
1113 +static ssize_t fuse_fsync(FuseExport *exp)
1114 {
835 - FuseExport *exp = fuse_req_userdata(req);
836 - int ret;
837 -
838 - ret = blk_flush(exp->common.blk);
839 - fuse_reply_err(req, ret < 0 ? -ret : 0);
1115 + return blk_flush(exp->common.blk);
1116 }
1117
1118 /**
1119 * Called before an FD to the exported image is closed. (libfuse
1120 * notes this to be a way to return last-minute errors.)
1121 + * Return 0 on success (no 'out' object), and -errno on error.
1122 */
846 -static void fuse_flush(fuse_req_t req, fuse_ino_t inode,
847 - struct fuse_file_info *fi)
1123 +static ssize_t fuse_flush(FuseExport *exp)
1124 {
849 - fuse_fsync(req, inode, 1, fi);
1125 + return blk_flush(exp->common.blk);
1126 }
1127
1128 #ifdef CONFIG_FUSE_LSEEK
1129 /**
1130 * Let clients inquire allocation status.
1131 + * Return the number of bytes written to *out on success, and -errno on error.
1132 */
856 -static void fuse_lseek(fuse_req_t req, fuse_ino_t inode, off_t offset,
857 - int whence, struct fuse_file_info *fi)
1133 +static ssize_t fuse_lseek(FuseExport *exp, struct fuse_lseek_out *out,
1134 + uint64_t offset, uint32_t whence)
1135 {
859 - FuseExport *exp = fuse_req_userdata(req);
860 -
1136 if (whence != SEEK_HOLE && whence != SEEK_DATA) {
862 - fuse_reply_err(req, EINVAL);
863 - return;
1137 + return -EINVAL;
1138 }
1139
1140 while (true) {
@@ -870,8 +1144,7 @@ static void fuse_lseek(fuse_req_t req, fuse_ino_t inode, off_t offset,
1144 ret = bdrv_block_status_above(blk_bs(exp->common.blk), NULL,
1145 offset, INT64_MAX, &pnum, NULL, NULL);
1146 if (ret < 0) {
873 - fuse_reply_err(req, -ret);
874 - return;
1147 + return ret;
1148 }
1149
1150 if (!pnum && (ret & BDRV_BLOCK_EOF)) {
@@ -888,34 +1161,38 @@ static void fuse_lseek(fuse_req_t req, fuse_ino_t inode, off_t offset,
1161
1162 blk_len = blk_getlength(exp->common.blk);
1163 if (blk_len < 0) {
891 - fuse_reply_err(req, -blk_len);
892 - return;
1164 + return blk_len;
1165 }
1166
1167 if (offset > blk_len || whence == SEEK_DATA) {
896 - fuse_reply_err(req, ENXIO);
897 - } else {
898 - fuse_reply_lseek(req, offset);
1168 + return -ENXIO;
1169 }
900 - return;
1170 +
1171 + *out = (struct fuse_lseek_out) {
1172 + .offset = offset,
1173 + };
1174 + return sizeof(*out);
1175 }
1176
1177 if (ret & BDRV_BLOCK_DATA) {
1178 if (whence == SEEK_DATA) {
905 - fuse_reply_lseek(req, offset);
906 - return;
1179 + *out = (struct fuse_lseek_out) {
1180 + .offset = offset,
1181 + };
1182 + return sizeof(*out);
1183 }
1184 } else {
1185 if (whence == SEEK_HOLE) {
910 - fuse_reply_lseek(req, offset);
911 - return;
1186 + *out = (struct fuse_lseek_out) {
1187 + .offset = offset,
1188 + };
1189 + return sizeof(*out);
1190 }
1191 }
1192
1193 /* Safety check against infinite loops */
1194 if (!pnum) {
917 - fuse_reply_err(req, ENXIO);
918 - return;
1195 + return -ENXIO;
1196 }
1197
1198 offset += pnum;
@@ -923,21 +1200,241 @@ static void fuse_lseek(fuse_req_t req, fuse_ino_t inode, off_t offset,
1200 }
1201 #endif
1202
926 -static const struct fuse_lowlevel_ops fuse_ops = {
927 - .init = fuse_init,
928 - .lookup = fuse_lookup,
929 - .getattr = fuse_getattr,
930 - .setattr = fuse_setattr,
931 - .open = fuse_open,
932 - .read = fuse_read,
933 - .write = fuse_write,
934 - .fallocate = fuse_fallocate,
935 - .flush = fuse_flush,
936 - .fsync = fuse_fsync,
1203 +/**
1204 + * Write a FUSE response to the given @fd.
1205 + *
1206 + * Effectively, writes out_hdr->common.len bytes of the buffer that is *out_hdr.
1207 + *
1208 + * @fd: FUSE file descriptor
1209 + * @out_hdr: Request response header and request-specific response data
1210 + */
1211 +static int fuse_write_response(int fd, FuseRequestOutHeader *out_hdr)
1212 +{
1213 + size_t to_write = out_hdr->common.len;
1214 + ssize_t ret;
1215 +
1216 + /* Must at least write fuse_out_header */
1217 + assert(to_write >= sizeof(out_hdr->common));
1218 +
1219 + ret = RETRY_ON_EINTR(write(fd, out_hdr, to_write));
1220 + if (ret < 0) {
1221 + ret = -errno;
1222 + error_report("Failed to write to FUSE device: %s", strerror(-ret));
1223 + return ret;
1224 + }
1225 +
1226 + /* Short writes are unexpected, treat them as errors */
1227 + if (ret != to_write) {
1228 + error_report("Short write to FUSE device, wrote %zi of %zu bytes",
1229 + ret, to_write);
1230 + return -EIO;
1231 + }
1232 +
1233 + return 0;
1234 +}
1235 +
1236 +/**
1237 + * Write a FUSE error response to @fd.
1238 + *
1239 + * @fd: FUSE file descriptor
1240 + * @in_hdr: Incoming request header to which to respond
1241 + * @err: Error code (-errno, must be negative!)
1242 + */
1243 +static int fuse_write_err(int fd, const struct fuse_in_header *in_hdr, int err)
1244 +{
1245 + FuseRequestOutHeader out_hdr = {
1246 + .common = {
1247 + .len = sizeof(out_hdr.common),
1248 + /* FUSE expects negative error values */
1249 + .error = err,
1250 + .unique = in_hdr->unique,
1251 + },
1252 + };
1253 +
1254 + return fuse_write_response(fd, &out_hdr);
1255 +}
1256 +
1257 +/**
1258 + * Write a FUSE response to the given @fd, using separate buffers for the
1259 + * response header and data.
1260 + *
1261 + * In contrast to fuse_write_response(), this function cannot return a full
1262 + * FuseRequestOutHeader (i.e. including request-specific response structs),
1263 + * but only FuseRequestOutHeader.common. The remaining data must be in
1264 + * *buf.
1265 + *
1266 + * (Total length must be set in out_hdr->len.)
1267 + *
1268 + * @fd: FUSE file descriptor
1269 + * @out_hdr: Request response header
1270 + * @buf: Pointer to response data
1271 + */
1272 +static int fuse_write_buf_response(int fd,
1273 + const struct fuse_out_header *out_hdr,
1274 + const void *buf)
1275 +{
1276 + size_t to_write = out_hdr->len;
1277 + struct iovec iov[2] = {
1278 + { (void *)out_hdr, sizeof(*out_hdr) },
1279 + { (void *)buf, to_write - sizeof(*out_hdr) },
1280 + };
1281 + ssize_t ret;
1282 +
1283 + /* *buf length must not be negative */
1284 + assert(to_write >= sizeof(*out_hdr));
1285 +
1286 + ret = RETRY_ON_EINTR(writev(fd, iov, ARRAY_SIZE(iov)));
1287 + if (ret < 0) {
1288 + ret = -errno;
1289 + error_report("Failed to write to FUSE device: %s", strerror(-ret));
1290 + return ret;
1291 + }
1292 +
1293 + /* Short writes are unexpected, treat them as errors */
1294 + if (ret != to_write) {
1295 + error_report("Short write to FUSE device, wrote %zi of %zu bytes",
1296 + ret, to_write);
1297 + return -EIO;
1298 + }
1299 +
1300 + return 0;
1301 +}
1302 +
1303 +/**
1304 + * Process a FUSE request, incl. writing the response.
1305 + */
1306 +static void fuse_process_request(FuseExport *exp,
1307 + const FuseRequestInHeader *in_hdr,
1308 + const void *data_buffer)
1309 +{
1310 + FuseRequestOutHeader out_hdr;
1311 + /* For read requests: Data to be returned */
1312 + void *out_data_buffer = NULL;
1313 + ssize_t ret;
1314 +
1315 + switch (in_hdr->common.opcode) {
1316 + case FUSE_INIT:
1317 + ret = fuse_init(exp, &out_hdr.init, &in_hdr->init);
1318 + break;
1319 +
1320 + case FUSE_DESTROY:
1321 + ret = 0;
1322 + break;
1323 +
1324 + case FUSE_STATFS:
1325 + ret = fuse_statfs(exp, &out_hdr.statfs);
1326 + break;
1327 +
1328 + case FUSE_OPEN:
1329 + ret = fuse_open(exp, &out_hdr.open);
1330 + break;
1331 +
1332 + case FUSE_RELEASE:
1333 + ret = 0;
1334 + break;
1335 +
1336 + case FUSE_LOOKUP:
1337 + ret = -ENOENT; /* There is no node but the root node */
1338 + break;
1339 +
1340 + case FUSE_FORGET:
1341 + case FUSE_BATCH_FORGET:
1342 + /* These have no response, and there is nothing we need to do */
1343 + return;
1344 +
1345 + case FUSE_GETATTR:
1346 + ret = fuse_getattr(exp, &out_hdr.attr);
1347 + break;
1348 +
1349 + case FUSE_SETATTR: {
1350 + const struct fuse_setattr_in *in = &in_hdr->setattr;
1351 + ret = fuse_setattr(exp, &out_hdr.attr,
1352 + in->valid, in->size, in->mode, in->uid, in->gid);
1353 + break;
1354 + }
1355 +
1356 + case FUSE_READ: {
1357 + const struct fuse_read_in *in = &in_hdr->read;
1358 + ret = fuse_read(exp, &out_data_buffer, in->offset, in->size);
1359 + break;
1360 + }
1361 +
1362 + case FUSE_WRITE: {
1363 + const struct fuse_write_in *in = &in_hdr->write;
1364 + uint32_t req_len = in_hdr->common.len;
1365 +
1366 + if (unlikely(req_len < sizeof(in_hdr->common) + sizeof(*in) +
1367 + in->size)) {
1368 + warn_report("FUSE WRITE truncated; received %zu bytes of %" PRIu32,
1369 + req_len - sizeof(in_hdr->common) - sizeof(*in),
1370 + in->size);
1371 + ret = -EINVAL;
1372 + break;
1373 + }
1374 +
1375 + /*
1376 + * read_from_fuse_fd() has checked that in_hdr->len matches the number
1377 + * of bytes read, which cannot exceed the max_write value we set
1378 + * (FUSE_MAX_WRITE_BYTES). So we know that FUSE_MAX_WRITE_BYTES >=
1379 + * in_hdr->len >= in->size + X, so this assertion must hold.
1380 + */
1381 + assert(in->size <= FUSE_MAX_WRITE_BYTES);
1382 +
1383 + ret = fuse_write(exp, &out_hdr.write,
1384 + in->offset, in->size, data_buffer);
1385 + break;
1386 + }
1387 +
1388 + case FUSE_FALLOCATE: {
1389 + const struct fuse_fallocate_in *in = &in_hdr->fallocate;
1390 + ret = fuse_fallocate(exp, in->offset, in->length, in->mode);
1391 + break;
1392 + }
1393 +
1394 + case FUSE_FSYNC:
1395 + ret = fuse_fsync(exp);
1396 + break;
1397 +
1398 + case FUSE_FLUSH:
1399 + ret = fuse_flush(exp);
1400 + break;
1401 +
1402 #ifdef CONFIG_FUSE_LSEEK
938 - .lseek = fuse_lseek,
1403 + case FUSE_LSEEK: {
1404 + const struct fuse_lseek_in *in = &in_hdr->lseek;
1405 + ret = fuse_lseek(exp, &out_hdr.lseek, in->offset, in->whence);
1406 + break;
1407 + }
1408 #endif
940 -};
1409 +
1410 + default:
1411 + ret = -ENOSYS;
1412 + }
1413 +
1414 + if (ret >= 0) {
1415 + out_hdr.common = (struct fuse_out_header) {
1416 + .len = sizeof(out_hdr.common) + ret,
1417 + .unique = in_hdr->common.unique,
1418 + };
1419 + } else {
1420 + /* fuse_read() must not return a buffer in case of error */
1421 + assert(out_data_buffer == NULL);
1422 +
1423 + out_hdr.common = (struct fuse_out_header) {
1424 + .len = sizeof(out_hdr.common),
1425 + /* FUSE expects negative errno values */
1426 + .error = ret,
1427 + .unique = in_hdr->common.unique,
1428 + };
1429 + }
1430 +
1431 + if (out_data_buffer) {
1432 + fuse_write_buf_response(exp->fuse_fd, &out_hdr.common, out_data_buffer);
1433 + qemu_vfree(out_data_buffer);
1434 + } else {
1435 + fuse_write_response(exp->fuse_fd, &out_hdr);
1436 + }
1437 +}
1438
1439 const BlockExportDriver blk_exp_fuse = {
1440 .type = BLOCK_EXPORT_TYPE_FUSE,