include/boost/corosio/native/detail/uring/uring_descriptor.hpp
92.4% Lines (182 / 197)
100.0% Functions (27 / 27)
Functions (27)
Function
Calls
Lines
Blocks
boost::corosio::detail::uring_descriptor_read_op::uring_descriptor_read_op()
:119
83x
100.0%
100.0%
boost::corosio::detail::uring_descriptor_read_op::do_prep(boost::corosio::detail::uring_op*, io_uring_sqe*)
:124
28x
100.0%
100.0%
boost::corosio::detail::uring_descriptor_write_op::uring_descriptor_write_op()
:160
74x
100.0%
100.0%
boost::corosio::detail::uring_descriptor_write_op::do_prep(boost::corosio::detail::uring_op*, io_uring_sqe*)
:165
12x
88.9%
62.0%
boost::corosio::detail::fd_is_pollable(int)
:209
54x
87.5%
92.0%
boost::corosio::detail::uring_descriptor::uring_descriptor(boost::corosio::detail::uring_scheduler&)
:242
71x
100.0%
100.0%
boost::corosio::detail::uring_descriptor::~uring_descriptor()
:246
68x
100.0%
100.0%
boost::corosio::detail::uring_descriptor::read_some(std::__n4861::coroutine_handle<void>, boost::capy::executor_ref, boost::corosio::buffer_param, std::stop_token, std::error_code*, unsigned long*)
:253
26x
73.3%
83.0%
boost::corosio::detail::uring_descriptor::write_some(std::__n4861::coroutine_handle<void>, boost::capy::executor_ref, boost::corosio::buffer_param, std::stop_token, std::error_code*, unsigned long*)
:286
11x
60.0%
66.0%
boost::corosio::detail::uring_descriptor::wait(std::__n4861::coroutine_handle<void>, boost::capy::executor_ref, boost::corosio::wait_type, std::stop_token, std::error_code*)
:320
15x
93.8%
85.0%
boost::corosio::detail::uring_descriptor::native_handle() const
:377
128x
100.0%
100.0%
boost::corosio::detail::uring_descriptor::release_descriptor()
:382
2x
100.0%
100.0%
boost::corosio::detail::uring_descriptor::cancel()
:396
8x
100.0%
100.0%
boost::corosio::detail::uring_descriptor::epoch() const
:407
64x
100.0%
68.0%
boost::corosio::detail::uring_descriptor::set_descriptor(int, bool)
:420
57x
100.0%
100.0%
boost::corosio::detail::uring_descriptor::pollable() const
:428
2x
100.0%
100.0%
boost::corosio::detail::uring_descriptor::close_file()
:434
156x
100.0%
100.0%
boost::corosio::detail::uring_descriptor::close_descriptor()
:441
225x
100.0%
100.0%
void boost::corosio::detail::uring_descriptor::arm_slot<boost::corosio::detail::uring_descriptor_read_op>(boost::corosio::detail::uring_descriptor_read_op&)
:463
26x
100.0%
67.0%
void boost::corosio::detail::uring_descriptor::arm_slot<boost::corosio::detail::uring_descriptor_write_op>(boost::corosio::detail::uring_descriptor_write_op&)
:463
11x
100.0%
67.0%
boost::corosio::detail::uring_descriptor::push_completed(boost::corosio::detail::scheduler_op*)
:472
3x
100.0%
100.0%
bool boost::corosio::detail::uring_descriptor_abandoned<boost::corosio::detail::uring_descriptor_read_op>(boost::corosio::detail::uring_descriptor_read_op const&)
:483
37x
100.0%
100.0%
bool boost::corosio::detail::uring_descriptor_abandoned<boost::corosio::detail::uring_descriptor_write_op>(boost::corosio::detail::uring_descriptor_write_op const&)
:483
13x
100.0%
88.0%
bool boost::corosio::detail::uring_descriptor_continue<boost::corosio::detail::uring_descriptor_read_op>(boost::corosio::detail::uring_descriptor_read_op&)
:491
37x
100.0%
100.0%
bool boost::corosio::detail::uring_descriptor_continue<boost::corosio::detail::uring_descriptor_write_op>(boost::corosio::detail::uring_descriptor_write_op&)
:491
10x
56.5%
51.0%
boost::corosio::detail::uring_descriptor_read_op::do_handler(void*, boost::corosio::detail::scheduler_op*, unsigned int, unsigned int)
:544
27x
100.0%
94.0%
boost::corosio::detail::uring_descriptor_write_op::do_handler(void*, boost::corosio::detail::scheduler_op*, unsigned int, unsigned int)
:568
9x
92.3%
88.0%
| Line | TLA | Hits | Source Code |
|---|---|---|---|
| 1 | // | ||
| 2 | // Copyright (c) 2026 Michael Vandeberg | ||
| 3 | // | ||
| 4 | // Distributed under the Boost Software License, Version 1.0. (See accompanying | ||
| 5 | // file LICENSE_1_0.txt or copy at http://www.boost.org/LICENSE_1_0.txt) | ||
| 6 | // | ||
| 7 | // Official repository: https://github.com/cppalliance/corosio | ||
| 8 | // | ||
| 9 | |||
| 10 | #ifndef BOOST_COROSIO_NATIVE_DETAIL_URING_URING_DESCRIPTOR_HPP | ||
| 11 | #define BOOST_COROSIO_NATIVE_DETAIL_URING_URING_DESCRIPTOR_HPP | ||
| 12 | |||
| 13 | #include <boost/corosio/detail/platform.hpp> | ||
| 14 | |||
| 15 | #if BOOST_COROSIO_HAS_URING | ||
| 16 | |||
| 17 | #include <boost/corosio/posix_stream_descriptor.hpp> | ||
| 18 | #include <boost/corosio/wait_type.hpp> | ||
| 19 | #include <boost/corosio/detail/intrusive.hpp> | ||
| 20 | #include <boost/corosio/native/detail/uring/uring_file_ops.hpp> | ||
| 21 | #include <boost/corosio/native/detail/uring/uring_scheduler.hpp> | ||
| 22 | #include <boost/corosio/native/detail/uring/uring_socket_ops.hpp> | ||
| 23 | #include <boost/corosio/native/detail/validate_fd.hpp> | ||
| 24 | |||
| 25 | #include <atomic> | ||
| 26 | #include <coroutine> | ||
| 27 | #include <cstddef> | ||
| 28 | #include <cstdint> | ||
| 29 | #include <memory> | ||
| 30 | #include <system_error> | ||
| 31 | |||
| 32 | #include <errno.h> | ||
| 33 | #include <poll.h> | ||
| 34 | #include <sys/epoll.h> | ||
| 35 | #include <unistd.h> | ||
| 36 | |||
| 37 | /* io_uring-backed implementation of posix_stream_descriptor. | ||
| 38 | |||
| 39 | Three things differ from the reactor backends and from the other | ||
| 40 | io_uring services: | ||
| 41 | |||
| 42 | Transfers submit READV/WRITEV at offset -1 so the kernel uses (and | ||
| 43 | advances) the descriptor's own file position. The file services | ||
| 44 | pass a real offset; a pipe, tty or character device has none. | ||
| 45 | |||
| 46 | The descriptor's flags are never modified, as in asio. A transfer | ||
| 47 | on a blocking fd the kernel can poll parks in its internal poll and | ||
| 48 | ASYNC_CANCEL removes it. One the kernel punts to an io-wq worker | ||
| 49 | (no poll support, or no FMODE_NOWAIT) holds that worker; the cancel | ||
| 50 | interrupts it only if the driver's wait is interruptible, and the | ||
| 51 | resulting -EINTR is reported as canceled. A caller who made the fd | ||
| 52 | non-blocking gets EAGAIN completions, which the two-phase shape | ||
| 53 | below turns into a poll and a retry. A file with no poll support | ||
| 54 | is ready to every poll, so assign probes for it (fd_is_pollable) | ||
| 55 | and its EAGAIN fails with EOPNOTSUPP instead, as on the reactors. | ||
| 56 | |||
| 57 | An O_NONBLOCK descriptor the kernel cannot retry internally | ||
| 58 | completes with -EAGAIN; the op then re-arms itself as a poll_add on | ||
| 59 | the same descriptor and re-submits the transfer when the poll says | ||
| 60 | ready. The handler makes that decision *before* coro_drain_if_shutdown, | ||
| 61 | which disarms stop_cb: an op going round again keeps its | ||
| 62 | cancellation wiring, so a stop_token firing between phases still | ||
| 63 | reaches the kernel. | ||
| 64 | |||
| 65 | The gap between a CQE and its dispatch is the whole difficulty of | ||
| 66 | that shape, and one epoch counter closes it. Nothing of a transfer | ||
| 67 | waiting in that gap is in the ring, so cancel-by-fd cannot reach | ||
| 68 | it. cancel() and every descriptor change bump epoch_ before they | ||
| 69 | take ring_mutex_, and do_prep, which runs under ring_mutex_, | ||
| 70 | compares the op's snapshot against it: an op that no longer | ||
| 71 | matches preps a NOP instead of a transfer and completes canceled. | ||
| 72 | Either the prep sees the bump, or the cancel's SQE queues behind | ||
| 73 | the transfer's and finds it. The epoch is also what makes the | ||
| 74 | staleness check exact rather than heuristic: a closed fd number | ||
| 75 | the next assign() gets back would satisfy a bare fd comparison. | ||
| 76 | |||
| 77 | There is no adopt-time registration. assign() validates and takes | ||
| 78 | the descriptor; a kernel that refuses it says so at the first | ||
| 79 | operation, as the public docstring promises. | ||
| 80 | */ | ||
| 81 | |||
| 82 | namespace boost::corosio::detail { | ||
| 83 | |||
| 84 | class uring_descriptor; | ||
| 85 | |||
| 86 | /** Advance a two-phase transfer op, or report that it is finished. | ||
| 87 | |||
| 88 | A kernel `EAGAIN` becomes a `poll_add` on the same descriptor, and | ||
| 89 | the poll's completion re-submits the transfer. Every path that | ||
| 90 | stops instead leaves @a op carrying a result the completion decode | ||
| 91 | can read as terminal. | ||
| 92 | |||
| 93 | @param op The op whose CQE just arrived. | ||
| 94 | @return True when a fresh SQE was submitted, in which case the | ||
| 95 | caller must neither complete nor disarm the op. | ||
| 96 | */ | ||
| 97 | template<class Op> | ||
| 98 | bool uring_descriptor_continue(Op& op) noexcept; | ||
| 99 | |||
| 100 | /// True when @a op no longer belongs to its owner's current intent. | ||
| 101 | /// Called from do_prep, under ring_mutex_. | ||
| 102 | template<class Op> | ||
| 103 | bool uring_descriptor_abandoned(Op const& op) noexcept; | ||
| 104 | |||
| 105 | /** Scatter read via `IORING_OP_READV` at the descriptor's own offset. | ||
| 106 | |||
| 107 | @see uring_descriptor_continue for the `polling` phase. | ||
| 108 | */ | ||
| 109 | struct uring_descriptor_read_op final : uring_file_read_op_base | ||
| 110 | { | ||
| 111 | uring_descriptor* desc = nullptr; | ||
| 112 | /// True while the submitted SQE is the readiness poll, not the read. | ||
| 113 | bool polling = false; | ||
| 114 | /// Owner epoch snapshotted by arm_slot; see uring_descriptor_continue. | ||
| 115 | std::uint32_t epoch = 0; | ||
| 116 | /// Set by do_prep when the op was abandoned before its SQE was built. | ||
| 117 | bool abandoned = false; | ||
| 118 | |||
| 119 | 83x | uring_descriptor_read_op() noexcept : uring_file_read_op_base(&do_handler) | |
| 120 | { | ||
| 121 | 83x | prep_func = &do_prep; | |
| 122 | 83x | } | |
| 123 | |||
| 124 | 28x | static void do_prep(uring_op* base, ::io_uring_sqe* sqe) noexcept | |
| 125 | { | ||
| 126 | 28x | auto* self = static_cast<uring_descriptor_read_op*>(base); | |
| 127 | // Decided here because every cancel path records its intent | ||
| 128 | // before taking ring_mutex_, which this runs under: either the | ||
| 129 | // intent is visible now, or the cancel SQE queues behind ours. | ||
| 130 | 28x | if (uring_descriptor_abandoned(*self)) | |
| 131 | { | ||
| 132 | 1x | self->abandoned = true; | |
| 133 | 1x | ::io_uring_prep_nop(sqe); | |
| 134 | 1x | return; | |
| 135 | } | ||
| 136 | 27x | if (self->polling) | |
| 137 | 1x | ::io_uring_prep_poll_add(sqe, self->fd, POLLIN); | |
| 138 | else | ||
| 139 | 26x | uring_file_read_op_base::do_prep(base, sqe); | |
| 140 | } | ||
| 141 | |||
| 142 | static void do_handler( | ||
| 143 | void* owner, | ||
| 144 | scheduler_op* base, | ||
| 145 | std::uint32_t bytes, | ||
| 146 | std::uint32_t error) noexcept; | ||
| 147 | }; | ||
| 148 | |||
| 149 | /// Gather write via `IORING_OP_WRITEV` at the descriptor's own offset. | ||
| 150 | struct uring_descriptor_write_op final : uring_file_write_op_base | ||
| 151 | { | ||
| 152 | uring_descriptor* desc = nullptr; | ||
| 153 | /// True while the submitted SQE is the readiness poll, not the write. | ||
| 154 | bool polling = false; | ||
| 155 | /// Owner epoch snapshotted by arm_slot; see uring_descriptor_continue. | ||
| 156 | std::uint32_t epoch = 0; | ||
| 157 | /// Set by do_prep when the op was abandoned before its SQE was built. | ||
| 158 | bool abandoned = false; | ||
| 159 | |||
| 160 | 74x | uring_descriptor_write_op() noexcept : uring_file_write_op_base(&do_handler) | |
| 161 | { | ||
| 162 | 74x | prep_func = &do_prep; | |
| 163 | 74x | } | |
| 164 | |||
| 165 | 12x | static void do_prep(uring_op* base, ::io_uring_sqe* sqe) noexcept | |
| 166 | { | ||
| 167 | 12x | auto* self = static_cast<uring_descriptor_write_op*>(base); | |
| 168 | // Decided here because every cancel path records its intent | ||
| 169 | // before taking ring_mutex_, which this runs under: either the | ||
| 170 | // intent is visible now, or the cancel SQE queues behind ours. | ||
| 171 | 12x | if (uring_descriptor_abandoned(*self)) | |
| 172 | { | ||
| 173 | 1x | self->abandoned = true; | |
| 174 | 1x | ::io_uring_prep_nop(sqe); | |
| 175 | 1x | return; | |
| 176 | } | ||
| 177 | 11x | if (self->polling) | |
| 178 | ✗ | ::io_uring_prep_poll_add(sqe, self->fd, POLLOUT); | |
| 179 | else | ||
| 180 | 11x | uring_file_write_op_base::do_prep(base, sqe); | |
| 181 | } | ||
| 182 | |||
| 183 | static void do_handler( | ||
| 184 | void* owner, | ||
| 185 | scheduler_op* base, | ||
| 186 | std::uint32_t bytes, | ||
| 187 | std::uint32_t error) noexcept; | ||
| 188 | }; | ||
| 189 | |||
| 190 | /** Native io_uring implementation of @ref posix_stream_descriptor. | ||
| 191 | |||
| 192 | Holds the adopted descriptor and the five embedded op slots: one | ||
| 193 | transfer per direction, and one wait per direction. | ||
| 194 | |||
| 195 | @par Thread Safety | ||
| 196 | Distinct objects: Safe.@n | ||
| 197 | Shared objects: Unsafe. Each slot carries a single pending | ||
| 198 | operation, so a descriptor must not have two operations of the | ||
| 199 | same kind in flight. | ||
| 200 | */ | ||
| 201 | /** Return whether the kernel can poll @p fd. | ||
| 202 | |||
| 203 | epoll refuses a file with no poll support with EPERM; io_uring | ||
| 204 | instead reports such a file ready to every poll. Nothing is left | ||
| 205 | registered and @p fd is not modified. When the probe itself fails, | ||
| 206 | the descriptor is assumed pollable. | ||
| 207 | */ | ||
| 208 | inline bool | ||
| 209 | 54x | fd_is_pollable(int fd) noexcept | |
| 210 | { | ||
| 211 | 54x | int ep = ::epoll_create1(EPOLL_CLOEXEC); | |
| 212 | 54x | if (ep < 0) | |
| 213 | ✗ | return true; | |
| 214 | 54x | ::epoll_event ev{}; | |
| 215 | bool const pollable = | ||
| 216 | 54x | ::epoll_ctl(ep, EPOLL_CTL_ADD, fd, &ev) == 0 || errno != EPERM; | |
| 217 | 54x | ::close(ep); | |
| 218 | 54x | return pollable; | |
| 219 | } | ||
| 220 | |||
| 221 | class BOOST_COROSIO_DECL uring_descriptor final | ||
| 222 | : public posix_stream_descriptor::implementation | ||
| 223 | , public std::enable_shared_from_this<uring_descriptor> | ||
| 224 | , public intrusive_list<uring_descriptor>::node | ||
| 225 | { | ||
| 226 | uring_scheduler* sched_ = nullptr; | ||
| 227 | int fd_ = -1; | ||
| 228 | bool pollable_ = true; | ||
| 229 | |||
| 230 | // Bumped by cancel() and by every descriptor change. A transfer op | ||
| 231 | // between its EAGAIN CQE and its dispatch is invisible to the ring; | ||
| 232 | // its snapshot of this is what stops it re-arming. | ||
| 233 | std::atomic<std::uint32_t> epoch_{0}; | ||
| 234 | |||
| 235 | uring_descriptor_read_op rd_; | ||
| 236 | uring_descriptor_write_op wr_; | ||
| 237 | uring_wait_op wait_rd_; | ||
| 238 | uring_wait_op wait_wr_; | ||
| 239 | uring_wait_op wait_er_; | ||
| 240 | |||
| 241 | public: | ||
| 242 | 71x | explicit uring_descriptor(uring_scheduler& sched) noexcept : sched_(&sched) | |
| 243 | { | ||
| 244 | 71x | } | |
| 245 | |||
| 246 | 68x | ~uring_descriptor() override | |
| 247 | 68x | { | |
| 248 | 68x | close_descriptor(); | |
| 249 | 68x | } | |
| 250 | |||
| 251 | // -- io_stream::implementation -- | ||
| 252 | |||
| 253 | 26x | std::coroutine_handle<> read_some( | |
| 254 | std::coroutine_handle<> h, | ||
| 255 | capy::executor_ref ex, | ||
| 256 | buffer_param buffers, | ||
| 257 | std::stop_token token, | ||
| 258 | std::error_code* ec, | ||
| 259 | std::size_t* bytes) override | ||
| 260 | { | ||
| 261 | 26x | rd_.prepare( | |
| 262 | h, ex, ec, bytes, fd_, /*file_offset=*/-1, sched_, | ||
| 263 | 52x | shared_from_this(), buffers, token); | |
| 264 | 26x | arm_slot(rd_); | |
| 265 | 26x | sched_->work_started(); | |
| 266 | |||
| 267 | // Closed-object contract outranks the zero-length no-op. | ||
| 268 | 26x | if (fd_ < 0) | |
| 269 | { | ||
| 270 | ✗ | rd_.empty_buffer = false; | |
| 271 | ✗ | rd_.res = -EBADF; | |
| 272 | ✗ | push_completed(&rd_); | |
| 273 | ✗ | return std::noop_coroutine(); | |
| 274 | } | ||
| 275 | |||
| 276 | 26x | if (rd_.empty_buffer || rd_.cancelled.load(std::memory_order_acquire)) | |
| 277 | { | ||
| 278 | 1x | push_completed(&rd_); | |
| 279 | 1x | return std::noop_coroutine(); | |
| 280 | } | ||
| 281 | |||
| 282 | 25x | uring_submit_op(*sched_, &rd_); | |
| 283 | 25x | return std::noop_coroutine(); | |
| 284 | } | ||
| 285 | |||
| 286 | 11x | std::coroutine_handle<> write_some( | |
| 287 | std::coroutine_handle<> h, | ||
| 288 | capy::executor_ref ex, | ||
| 289 | buffer_param buffers, | ||
| 290 | std::stop_token token, | ||
| 291 | std::error_code* ec, | ||
| 292 | std::size_t* bytes) override | ||
| 293 | { | ||
| 294 | 11x | wr_.prepare( | |
| 295 | h, ex, ec, bytes, fd_, /*file_offset=*/-1, sched_, | ||
| 296 | 22x | shared_from_this(), buffers, token); | |
| 297 | 11x | arm_slot(wr_); | |
| 298 | 11x | sched_->work_started(); | |
| 299 | |||
| 300 | 11x | if (fd_ < 0) | |
| 301 | { | ||
| 302 | ✗ | wr_.empty_buffer = false; | |
| 303 | ✗ | wr_.res = -EBADF; | |
| 304 | ✗ | push_completed(&wr_); | |
| 305 | ✗ | return std::noop_coroutine(); | |
| 306 | } | ||
| 307 | |||
| 308 | 11x | if (wr_.empty_buffer || wr_.cancelled.load(std::memory_order_acquire)) | |
| 309 | { | ||
| 310 | ✗ | push_completed(&wr_); | |
| 311 | ✗ | return std::noop_coroutine(); | |
| 312 | } | ||
| 313 | |||
| 314 | 11x | uring_submit_op(*sched_, &wr_); | |
| 315 | 11x | return std::noop_coroutine(); | |
| 316 | } | ||
| 317 | |||
| 318 | // -- posix_stream_descriptor::implementation -- | ||
| 319 | |||
| 320 | 15x | std::coroutine_handle<> wait( | |
| 321 | std::coroutine_handle<> h, | ||
| 322 | capy::executor_ref ex, | ||
| 323 | wait_type w, | ||
| 324 | std::stop_token token, | ||
| 325 | std::error_code* ec) override | ||
| 326 | { | ||
| 327 | 15x | uring_wait_op* op = nullptr; | |
| 328 | 15x | int poll_flags = 0; | |
| 329 | 15x | switch (w) | |
| 330 | { | ||
| 331 | 7x | case wait_type::read: | |
| 332 | 7x | op = &wait_rd_; | |
| 333 | 7x | poll_flags = POLLIN; | |
| 334 | 7x | break; | |
| 335 | 3x | case wait_type::write: | |
| 336 | 3x | op = &wait_wr_; | |
| 337 | 3x | poll_flags = POLLOUT; | |
| 338 | 3x | break; | |
| 339 | 5x | case wait_type::error: | |
| 340 | 5x | op = &wait_er_; | |
| 341 | // POLLERR, POLLHUP and POLLNVAL are reported whether or not | ||
| 342 | // they are asked for, so the error wait names only POLLPRI. | ||
| 343 | 5x | poll_flags = POLLPRI; | |
| 344 | 5x | break; | |
| 345 | } | ||
| 346 | |||
| 347 | 15x | op->prepare( | |
| 348 | 30x | h, ex, ec, fd_, sched_, shared_from_this(), poll_flags, token); | |
| 349 | 15x | sched_->work_started(); | |
| 350 | |||
| 351 | 15x | if (fd_ < 0) | |
| 352 | { | ||
| 353 | 1x | op->res = -EBADF; | |
| 354 | 1x | push_completed(op); | |
| 355 | 1x | return std::noop_coroutine(); | |
| 356 | } | ||
| 357 | |||
| 358 | // Without poll support there is no error condition to watch; | ||
| 359 | // the kernel would refuse the poll with EINVAL. | ||
| 360 | 14x | if (w == wait_type::error && !pollable_) | |
| 361 | { | ||
| 362 | 1x | op->res = -EOPNOTSUPP; | |
| 363 | 1x | push_completed(op); | |
| 364 | 1x | return std::noop_coroutine(); | |
| 365 | } | ||
| 366 | |||
| 367 | 13x | if (op->cancelled.load(std::memory_order_acquire)) | |
| 368 | { | ||
| 369 | ✗ | push_completed(op); | |
| 370 | ✗ | return std::noop_coroutine(); | |
| 371 | } | ||
| 372 | |||
| 373 | 13x | uring_submit_op(*sched_, op); | |
| 374 | 13x | return std::noop_coroutine(); | |
| 375 | } | ||
| 376 | |||
| 377 | 128x | native_handle_type native_handle() const noexcept override | |
| 378 | { | ||
| 379 | 128x | return fd_; | |
| 380 | } | ||
| 381 | |||
| 382 | 2x | native_handle_type release_descriptor() noexcept override | |
| 383 | { | ||
| 384 | // Bump before the flush, as close_descriptor does. Flush the | ||
| 385 | // cancel while the fd is still open so the kernel resolves it | ||
| 386 | // before the caller can close and recycle the number. Do NOT | ||
| 387 | // close -- the caller takes ownership. | ||
| 388 | 2x | epoch_.fetch_add(1, std::memory_order_release); | |
| 389 | 2x | if (fd_ >= 0) | |
| 390 | 2x | sched_->cancel_and_flush(fd_); | |
| 391 | 2x | native_handle_type released = fd_; | |
| 392 | 2x | fd_ = -1; | |
| 393 | 2x | return released; | |
| 394 | } | ||
| 395 | |||
| 396 | 8x | void cancel() noexcept override | |
| 397 | { | ||
| 398 | // Bump before the SQE: cancel-by-fd reaches only what the ring | ||
| 399 | // currently holds, and an op waiting for its handler to run | ||
| 400 | // holds nothing there. The epoch is what that op consults. | ||
| 401 | 8x | epoch_.fetch_add(1, std::memory_order_release); | |
| 402 | 8x | if (fd_ >= 0) | |
| 403 | 2x | sched_->submit_cancel_by_fd(fd_); | |
| 404 | 8x | } | |
| 405 | |||
| 406 | /// Epoch bumped by every @ref cancel and every descriptor change. | ||
| 407 | 64x | std::uint32_t epoch() const noexcept | |
| 408 | { | ||
| 409 | 128x | return epoch_.load(std::memory_order_acquire); | |
| 410 | } | ||
| 411 | |||
| 412 | // -- Service-facing (non-virtual) -- | ||
| 413 | |||
| 414 | /** Adopt an already-validated descriptor. | ||
| 415 | |||
| 416 | @param fd The descriptor to adopt. | ||
| 417 | @param pollable Whether the kernel can poll @p fd; see | ||
| 418 | @ref fd_is_pollable. | ||
| 419 | */ | ||
| 420 | 57x | void set_descriptor(int fd, bool pollable = true) noexcept | |
| 421 | { | ||
| 422 | 57x | epoch_.fetch_add(1, std::memory_order_release); | |
| 423 | 57x | fd_ = fd; | |
| 424 | 57x | pollable_ = pollable; | |
| 425 | 57x | } | |
| 426 | |||
| 427 | /// Whether the kernel can poll the held descriptor. | ||
| 428 | 2x | bool pollable() const noexcept | |
| 429 | { | ||
| 430 | 2x | return pollable_; | |
| 431 | } | ||
| 432 | |||
| 433 | /// Teardown hook named by uring_file_service_base. | ||
| 434 | 156x | void close_file() noexcept | |
| 435 | { | ||
| 436 | 156x | close_descriptor(); | |
| 437 | 156x | } | |
| 438 | |||
| 439 | /// Cancel pending operations and close the descriptor. No-op when | ||
| 440 | /// already closed. | ||
| 441 | 225x | void close_descriptor() noexcept | |
| 442 | { | ||
| 443 | 225x | if (fd_ < 0) | |
| 444 | 172x | return; | |
| 445 | // Bump before the flush: an op prepping concurrently must see | ||
| 446 | // the change, or its SQE could land on a recycled fd number. | ||
| 447 | 53x | epoch_.fetch_add(1, std::memory_order_release); | |
| 448 | // Both kernel entries below can run a queued pipe write as task | ||
| 449 | // work; with the reader already gone that raises SIGPIPE. | ||
| 450 | 53x | scoped_sigpipe_block no_sigpipe; | |
| 451 | 53x | sched_->cancel_and_flush(fd_); | |
| 452 | 53x | ::close(fd_); | |
| 453 | 53x | fd_ = -1; | |
| 454 | 53x | } | |
| 455 | |||
| 456 | private: | ||
| 457 | /** Bind a transfer slot to this descriptor for a fresh submission. | ||
| 458 | |||
| 459 | The epoch snapshot taken here is what every later prep compares | ||
| 460 | against; see uring_descriptor_continue. | ||
| 461 | */ | ||
| 462 | template<class Op> | ||
| 463 | 37x | void arm_slot(Op& op) noexcept | |
| 464 | { | ||
| 465 | 37x | op.desc = this; | |
| 466 | 37x | op.polling = false; | |
| 467 | 37x | op.abandoned = false; | |
| 468 | 37x | op.epoch = epoch_.load(std::memory_order_acquire); | |
| 469 | 37x | } | |
| 470 | |||
| 471 | /// Queue an already-counted op for the next dispatch cycle. | ||
| 472 | 3x | void push_completed(scheduler_op* op) noexcept | |
| 473 | { | ||
| 474 | 3x | uring_scheduler::lock_type lock(sched_->dispatch_mutex()); | |
| 475 | 3x | sched_->push_completed_locked(op); | |
| 476 | 3x | } | |
| 477 | }; | ||
| 478 | |||
| 479 | // --- Deferred implementations (need uring_descriptor complete) --- | ||
| 480 | |||
| 481 | template<class Op> | ||
| 482 | bool | ||
| 483 | 50x | uring_descriptor_abandoned(Op const& op) noexcept | |
| 484 | { | ||
| 485 | 99x | return op.cancelled.load(std::memory_order_acquire) || | |
| 486 | 99x | op.desc->epoch() != op.epoch; | |
| 487 | } | ||
| 488 | |||
| 489 | template<class Op> | ||
| 490 | bool | ||
| 491 | 47x | uring_descriptor_continue(Op& op) noexcept | |
| 492 | { | ||
| 493 | // A NOP completed: the op was abandoned at prep. | ||
| 494 | 47x | if (op.abandoned) | |
| 495 | { | ||
| 496 | 1x | op.res = -ECANCELED; | |
| 497 | 1x | return false; | |
| 498 | } | ||
| 499 | |||
| 500 | // A failed poll, a transfer error or a byte count is the operation's | ||
| 501 | // answer, and bytes beat a cancel that landed after them. | ||
| 502 | 92x | bool const rearm = op.polling | |
| 503 | 86x | ? op.res >= 0 | |
| 504 | 40x | : (op.res == -EAGAIN || op.res == -EWOULDBLOCK); | |
| 505 | 46x | if (!rearm) | |
| 506 | { | ||
| 507 | // A poll the full SQ never took comes back as -EAGAIN, and a | ||
| 508 | // transfer the cancel interrupted in an io-wq worker as -EINTR; | ||
| 509 | // either one abandoned meanwhile is owed canceled, as above. | ||
| 510 | 39x | bool const interrupted = | |
| 511 | 39x | op.polling ? op.res == -EAGAIN : op.res == -EINTR; | |
| 512 | 39x | if (interrupted && uring_descriptor_abandoned(op)) | |
| 513 | 2x | op.res = -ECANCELED; | |
| 514 | 39x | return false; | |
| 515 | } | ||
| 516 | |||
| 517 | // Abandoned since the CQE arrived: the caller is owed canceled, | ||
| 518 | // never the poll's revents or the transfer's EAGAIN. | ||
| 519 | 7x | if (uring_descriptor_abandoned(op)) | |
| 520 | { | ||
| 521 | 4x | op.res = -ECANCELED; | |
| 522 | 4x | return false; | |
| 523 | } | ||
| 524 | |||
| 525 | // A file with no poll support is ready to every poll, so arming | ||
| 526 | // one would retry the refused transfer on a CPU forever. The | ||
| 527 | // reactors report the same refusal. | ||
| 528 | 3x | if (!op.polling && !op.desc->pollable()) | |
| 529 | { | ||
| 530 | 1x | op.res = -EOPNOTSUPP; | |
| 531 | 1x | return false; | |
| 532 | } | ||
| 533 | 2x | op.polling = !op.polling; | |
| 534 | |||
| 535 | // do_one spends a work_finished() on every op it dispatches, so an | ||
| 536 | // op going round again has to be counted again. Nothing may touch | ||
| 537 | // op after the submit: another thread can complete and free it. | ||
| 538 | 2x | op.sched_->work_started(); | |
| 539 | 2x | uring_submit_op(*op.sched_, &op); | |
| 540 | 2x | return true; | |
| 541 | } | ||
| 542 | |||
| 543 | inline void | ||
| 544 | 27x | uring_descriptor_read_op::do_handler( | |
| 545 | void* owner, | ||
| 546 | scheduler_op* base, | ||
| 547 | std::uint32_t /*bytes*/, | ||
| 548 | std::uint32_t /*error*/) noexcept | ||
| 549 | { | ||
| 550 | 27x | auto* self = static_cast<uring_descriptor_read_op*>(base); | |
| 551 | 27x | if (owner != nullptr && uring_descriptor_continue(*self)) | |
| 552 | 1x | return; | |
| 553 | |||
| 554 | 26x | if (coro_drain_if_shutdown(owner, self)) | |
| 555 | 2x | return; | |
| 556 | |||
| 557 | 24x | if (self->sched_) | |
| 558 | 24x | self->sched_->reset_inline_budget(); | |
| 559 | |||
| 560 | 24x | uring_set_result(self, /*is_read=*/true, self->empty_buffer); | |
| 561 | 24x | if (self->bytes_out) | |
| 562 | 24x | *self->bytes_out = | |
| 563 | 24x | self->res >= 0 ? static_cast<std::size_t>(self->res) : 0u; | |
| 564 | 24x | coro_resume(self); | |
| 565 | } | ||
| 566 | |||
| 567 | inline void | ||
| 568 | 9x | uring_descriptor_write_op::do_handler( | |
| 569 | void* owner, | ||
| 570 | scheduler_op* base, | ||
| 571 | std::uint32_t /*bytes*/, | ||
| 572 | std::uint32_t /*error*/) noexcept | ||
| 573 | { | ||
| 574 | 9x | auto* self = static_cast<uring_descriptor_write_op*>(base); | |
| 575 | 9x | if (owner != nullptr && uring_descriptor_continue(*self)) | |
| 576 | ✗ | return; | |
| 577 | |||
| 578 | 9x | if (coro_drain_if_shutdown(owner, self)) | |
| 579 | 1x | return; | |
| 580 | |||
| 581 | 8x | if (self->sched_) | |
| 582 | 8x | self->sched_->reset_inline_budget(); | |
| 583 | |||
| 584 | 8x | uring_set_result(self, /*is_read=*/false, self->empty_buffer); | |
| 585 | 8x | if (self->bytes_out) | |
| 586 | 8x | *self->bytes_out = | |
| 587 | 8x | self->res >= 0 ? static_cast<std::size_t>(self->res) : 0u; | |
| 588 | 8x | coro_resume(self); | |
| 589 | } | ||
| 590 | |||
| 591 | } // namespace boost::corosio::detail | ||
| 592 | |||
| 593 | #endif // BOOST_COROSIO_HAS_URING | ||
| 594 | |||
| 595 | #endif // BOOST_COROSIO_NATIVE_DETAIL_URING_URING_DESCRIPTOR_HPP | ||
| 596 |