include/boost/corosio/native/detail/select/select_scheduler.hpp

99.4% Lines (173 / 174, 1 excl) 100.0% Functions (12 / 12)
select_scheduler.hpp
f(x) Functions (12)
Line TLA Hits Source Code
1 //
2 // Copyright (c) 2026 Steve Gerbino
3 // Copyright (c) 2026 Michael Vandeberg
4 //
5 // Distributed under the Boost Software License, Version 1.0. (See accompanying
6 // file LICENSE_1_0.txt or copy at http://www.boost.org/LICENSE_1_0.txt)
7 //
8 // Official repository: https://github.com/cppalliance/corosio
9 //
10
11 #ifndef BOOST_COROSIO_NATIVE_DETAIL_SELECT_SELECT_SCHEDULER_HPP
12 #define BOOST_COROSIO_NATIVE_DETAIL_SELECT_SELECT_SCHEDULER_HPP
13
14 #include <boost/corosio/detail/platform.hpp>
15
16 #if BOOST_COROSIO_HAS_SELECT
17
18 #include <boost/corosio/detail/config.hpp>
19 #include <boost/capy/ex/execution_context.hpp>
20
21 #include <boost/corosio/native/detail/reactor/reactor_scheduler.hpp>
22 #include <boost/corosio/native/detail/reactor/reactor_signal_pipe.hpp>
23
24 #include <boost/corosio/native/detail/select/select_traits.hpp>
25 #include <boost/corosio/detail/timer_service.hpp>
26 #include <boost/corosio/native/detail/make_err.hpp>
27
28 #include <boost/corosio/detail/except.hpp>
29
30 #include <sys/select.h>
31 #include <unistd.h>
32 #include <errno.h>
33 #include <fcntl.h>
34
35 #include <atomic>
36 #include <chrono>
37 #include <cstdint>
38 #include <limits>
39 #include <mutex>
40 #include <new>
41 #include <unordered_map>
42
43 namespace boost::corosio::detail {
44
45 struct select_op;
46
47 /** POSIX scheduler using select() for I/O multiplexing.
48
49 This scheduler implements the scheduler interface using the POSIX select()
50 call for I/O event notification. It inherits the shared reactor threading
51 model from reactor_scheduler: signal state machine, inline completion
52 budget, work counting, and the do_one event loop.
53
54 The design mirrors epoll_scheduler for behavioral consistency:
55 - Same single-reactor thread coordination model
56 - Same deferred I/O pattern (reactor marks ready; workers do I/O)
57 - Same timer integration pattern
58
59 Known Limitations:
60 - FD_SETSIZE (~1024) limits maximum concurrent connections
61 - O(n) scanning: rebuilds fd_sets each iteration
62 - Level-triggered only (no edge-triggered mode)
63
64 @par Thread Safety
65 All public member functions are thread-safe.
66 */
67 class BOOST_COROSIO_DECL select_scheduler final : public reactor_scheduler
68 {
69 public:
70 /** Construct the scheduler.
71
72 Creates a self-pipe for reactor interruption.
73
74 @param ctx Reference to the owning execution_context.
75 @param concurrency_hint Hint for expected thread count (unused).
76 */
77 select_scheduler(capy::execution_context& ctx, int concurrency_hint = -1);
78
79 /// Destroy the scheduler.
80 ~select_scheduler() override;
81
82 select_scheduler(select_scheduler const&) = delete;
83 select_scheduler& operator=(select_scheduler const&) = delete;
84
85 /// Shut down the scheduler, draining pending operations.
86 void shutdown() override;
87
88 /** Return the maximum file descriptor value supported.
89
90 Returns FD_SETSIZE - 1, the maximum fd value that can be
91 monitored by select(). Operations with fd >= FD_SETSIZE
92 will fail with EINVAL.
93
94 @return The maximum supported file descriptor value.
95 */
96 static constexpr int max_fd() noexcept
97 {
98 return FD_SETSIZE - 1;
99 }
100
101 /** Register a descriptor for persistent monitoring.
102
103 The fd is added to the registered_descs_ map and will be
104 included in subsequent select() calls. The reactor is
105 interrupted so a blocked select() rebuilds its fd_sets.
106
107 @param fd The file descriptor to register.
108 @param desc Pointer to descriptor state for this fd.
109
110 @return The error if the fd cannot be tracked, otherwise a
111 default constructed error code.
112 */
113 std::error_code
114 register_descriptor(int fd, reactor_descriptor_state* desc) const;
115
116 /// No-op: write readiness is watched from registration on.
117 std::error_code
118 2330x ensure_write_registered(int, reactor_descriptor_state*) const noexcept
119 {
120 2330x return {};
121 }
122
123 /** Deregister a persistently registered descriptor.
124
125 @param fd The file descriptor to deregister.
126 */
127 void deregister_descriptor(int fd) const;
128
129 /** Interrupt the reactor so it rebuilds its fd_sets.
130
131 Called when a write, connect, or write-wait op is registered
132 after the reactor's snapshot was taken. Without this,
133 select() may block not watching for writability on the fd.
134 */
135 void notify_reactor() const;
136
137 /// Watch the read end of the POSIX signal self-pipe (see scheduler.hpp).
138 61x [[nodiscard]] std::error_code register_signal_reader(int read_fd) override
139 {
140 61x return register_descriptor(read_fd, signal_pipe_reader_.arm());
141 }
142
143 private:
144 void run_task(lock_type& lock, context_type& ctx, long timeout_us) override;
145 void interrupt_reactor() const override;
146 long calculate_timeout(long requested_timeout_us) const;
147
148 // Watches the global signal self-pipe's read end (armed lazily by
149 // register_signal_reader on the first signal registration).
150 reactor_signal_pipe_reader signal_pipe_reader_;
151
152 // Self-pipe for interrupting select()
153 int pipe_fds_[2]; // [0]=read, [1]=write
154
155 // Per-fd tracking for fd_set building
156 mutable std::unordered_map<int, reactor_descriptor_state*>
157 registered_descs_;
158 mutable int max_fd_ = -1;
159 };
160
161 1237x inline select_scheduler::select_scheduler(capy::execution_context& ctx, int)
162 1237x : pipe_fds_{-1, -1}
163 1237x , max_fd_(-1)
164 {
165 1237x if (::pipe(pipe_fds_) < 0)
166 1x detail::throw_system_error(make_err(errno), "pipe");
167
168 3699x for (int i = 0; i < 2; ++i)
169 {
170 2469x int flags = ::fcntl(pipe_fds_[i], F_GETFL, 0);
171 2469x if (flags == -1)
172 {
173 2x int errn = errno;
174 2x ::close(pipe_fds_[0]);
175 2x ::close(pipe_fds_[1]);
176 2x detail::throw_system_error(make_err(errn), "fcntl F_GETFL");
177 }
178 2467x if (::fcntl(pipe_fds_[i], F_SETFL, flags | O_NONBLOCK) == -1)
179 {
180 2x int errn = errno;
181 2x ::close(pipe_fds_[0]);
182 2x ::close(pipe_fds_[1]);
183 2x detail::throw_system_error(make_err(errn), "fcntl F_SETFL");
184 }
185 2465x if (::fcntl(pipe_fds_[i], F_SETFD, FD_CLOEXEC) == -1)
186 {
187 2x int errn = errno;
188 2x ::close(pipe_fds_[0]);
189 2x ::close(pipe_fds_[1]);
190 2x detail::throw_system_error(make_err(errn), "fcntl F_SETFD");
191 }
192 }
193
194 1230x timer_svc_ = &get_timer_service(ctx, *this);
195 1230x timer_svc_->set_on_earliest_changed(
196 4113x timer_service::callback(this, [](void* p) {
197 2883x static_cast<select_scheduler*>(p)->interrupt_reactor();
198 2883x }));
199
200 1230x completed_ops_.push(&task_op_);
201 1251x }
202
203 2460x inline select_scheduler::~select_scheduler()
204 {
205 1230x if (pipe_fds_[0] >= 0)
206 1230x ::close(pipe_fds_[0]);
207 1230x if (pipe_fds_[1] >= 0)
208 1230x ::close(pipe_fds_[1]);
209 2460x }
210
211 inline void
212 1230x select_scheduler::shutdown()
213 {
214 1230x shutdown_drain();
215
216 1230x if (pipe_fds_[1] >= 0)
217 1230x interrupt_reactor();
218 1230x }
219
220 inline std::error_code
221 5254x select_scheduler::register_descriptor(
222 int fd, reactor_descriptor_state* desc) const
223 {
224 5254x if (fd < 0 || fd >= FD_SETSIZE)
225 1x return make_err(EMFILE);
226
227 5253x desc->registered_events = reactor_event_read | reactor_event_write;
228 5253x desc->unpollable = false; // the state is reused across adoptions
229 5253x desc->fd = fd;
230 5253x desc->scheduler_ = this;
231 5253x desc->mutex.set_enabled(reactor_io_locking_);
232 5253x desc->ready_events_.store(0, std::memory_order_relaxed);
233
234 {
235 5253x conditionally_enabled_mutex::scoped_lock lock(desc->mutex);
236 5253x desc->impl_ref_.reset();
237 5253x desc->read_ready = false;
238 5253x desc->write_ready = false;
239 5253x }
240
241 {
242 5253x mutex_type::scoped_lock lock(mutex_);
243 try
244 {
245 5253x registered_descs_[fd] = desc;
246 }
247 1x catch (std::bad_alloc const&)
248 {
249 1x return make_err(ENOMEM);
250 1x }
251 5252x if (fd > max_fd_)
252 5203x max_fd_ = fd;
253 5253x }
254
255 5252x interrupt_reactor();
256 5252x return {};
257 }
258
259 inline void
260 5192x select_scheduler::deregister_descriptor(int fd) const
261 {
262 5192x mutex_type::scoped_lock lock(mutex_);
263
264 5192x auto it = registered_descs_.find(fd);
265 5192x if (it == registered_descs_.end())
266 ✗ return;
267
268 5192x registered_descs_.erase(it);
269
270 5192x if (fd == max_fd_)
271 {
272 4843x max_fd_ = pipe_fds_[0];
273 9226x for (auto& [registered_fd, state] : registered_descs_)
274 {
275 4383x if (registered_fd > max_fd_)
276 4285x max_fd_ = registered_fd;
277 }
278 }
279 5192x }
280
281 inline void
282 5092x select_scheduler::notify_reactor() const
283 {
284 5092x interrupt_reactor();
285 5092x }
286
287 inline void
288 16623x select_scheduler::interrupt_reactor() const
289 {
290 16623x char byte = 1;
291 16623x [[maybe_unused]] auto r = ::write(pipe_fds_[1], &byte, 1);
292 16623x }
293
294 inline long
295 9175x select_scheduler::calculate_timeout(long requested_timeout_us) const
296 {
297 9175x if (requested_timeout_us == 0)
298 − return 0; // LCOV_EXCL_LINE run_task passes 0 via task_interrupted_, never through this argument
299
300 9175x auto nearest = timer_svc_->nearest_expiry();
301 9175x if (nearest == timer_service::time_point::max())
302 2597x return requested_timeout_us;
303
304 6578x auto now = std::chrono::steady_clock::now();
305 6578x if (nearest <= now)
306 362x return 0;
307
308 auto timer_timeout_us =
309 6216x std::chrono::duration_cast<std::chrono::microseconds>(nearest - now)
310 6216x .count();
311
312 6216x constexpr auto long_max =
313 static_cast<long long>((std::numeric_limits<long>::max)());
314 auto capped_timer_us =
315 6216x (std::min)((std::max)(static_cast<long long>(timer_timeout_us),
316 6216x static_cast<long long>(0)),
317 6216x long_max);
318
319 6216x if (requested_timeout_us < 0)
320 6214x return static_cast<long>(capped_timer_us);
321
322 return static_cast<long>(
323 2x (std::min)(static_cast<long long>(requested_timeout_us),
324 2x capped_timer_us));
325 }
326
327 inline void
328 38602x select_scheduler::run_task(lock_type& lock, context_type& ctx, long timeout_us)
329 {
330 long effective_timeout_us =
331 38602x task_interrupted_ ? 0 : calculate_timeout(timeout_us);
332
333 // Snapshot registered descriptors while holding lock.
334 // Record which directions each fd needs monitored to avoid a hot
335 // loop: select is level-triggered, so a writable socket (nearly
336 // always writable) or an always-readable fd (/dev/zero, a pipe at
337 // EOF) would return select() immediately every iteration if
338 // unconditionally added. Membership in both sets is opt-in: a
339 // parked op or wait in a direction opts that direction in. The
340 // exceptional set is opt-in too, for any parked op or wait:
341 // Darwin reports a character device (/dev/zero, /dev/null) as
342 // exceptional on every call.
343 struct fd_entry
344 {
345 int fd;
346 reactor_descriptor_state* desc;
347 std::uint32_t want;
348 };
349 fd_entry snapshot[FD_SETSIZE];
350 38602x int snapshot_count = 0;
351
352 129403x for (auto& [fd, desc] : registered_descs_)
353 {
354 90801x if (snapshot_count < FD_SETSIZE)
355 {
356 90801x conditionally_enabled_mutex::scoped_lock desc_lock(desc->mutex);
357 90801x snapshot[snapshot_count].fd = fd;
358 90801x snapshot[snapshot_count].desc = desc;
359 90801x snapshot[snapshot_count].want =
360 90801x ((desc->read_op || desc->wait_read_op) ? reactor_event_read
361 90801x : 0) |
362 90452x ((desc->write_op || desc->connect_op || desc->wait_write_op)
363 181253x ? reactor_event_write
364 90801x : 0) |
365 90801x (desc->wait_error_op ? reactor_event_error : 0);
366 90801x ++snapshot_count;
367 90801x }
368 }
369
370 38602x if (lock.owns_lock())
371 9176x lock.unlock();
372
373 38602x task_cleanup on_exit{this, &lock, ctx};
374
375 fd_set read_fds, write_fds, except_fds;
376 656234x FD_ZERO(&read_fds);
377 656234x FD_ZERO(&write_fds);
378 656234x FD_ZERO(&except_fds);
379
380 38602x FD_SET(pipe_fds_[0], &read_fds);
381 38602x int nfds = pipe_fds_[0];
382
383 129403x for (int i = 0; i < snapshot_count; ++i)
384 {
385 90801x int fd = snapshot[i].fd;
386 90801x if (snapshot[i].want & reactor_event_read)
387 11335x FD_SET(fd, &read_fds);
388 90801x if (snapshot[i].want & reactor_event_write)
389 2581x FD_SET(fd, &write_fds);
390 90801x if (snapshot[i].want != 0)
391 14206x FD_SET(fd, &except_fds);
392 90801x if (fd > nfds)
393 36287x nfds = fd;
394 }
395
396 struct timeval tv;
397 38602x struct timeval* tv_ptr = nullptr;
398 38602x if (effective_timeout_us >= 0)
399 {
400 36398x tv.tv_sec = effective_timeout_us / 1000000;
401 36398x tv.tv_usec = effective_timeout_us % 1000000;
402 36398x tv_ptr = &tv;
403 }
404
405 38602x int ready = ::select(nfds + 1, &read_fds, &write_fds, &except_fds, tv_ptr);
406
407 // EINTR: signal interrupted select(), just retry.
408 // EBADF: an fd was closed between snapshot and select(); retry
409 // with a fresh snapshot from registered_descs_.
410 // Both fall through with no ready descriptors rather than
411 // returning: the caller handed this function an owned lock that
412 // only the epilogue below re-acquires.
413 38602x if (ready < 0)
414 {
415 3x if (errno != EINTR && errno != EBADF)
416 1x detail::throw_system_error(make_err(errno), "select");
417 2x ready = 0;
418 }
419
420 // Process timers outside the lock
421 38601x timer_svc_->process_expired();
422
423 38601x ready_queue local_ops;
424
425 38601x if (ready > 0)
426 {
427 8306x if (FD_ISSET(pipe_fds_[0], &read_fds))
428 {
429 char buf[256];
430 13776x while (::read(pipe_fds_[0], buf, sizeof(buf)) > 0)
431 {
432 }
433 }
434
435 19826x for (int i = 0; i < snapshot_count; ++i)
436 {
437 11520x int fd = snapshot[i].fd;
438 11520x reactor_descriptor_state* desc = snapshot[i].desc;
439
440 11520x std::uint32_t flags = 0;
441 11520x if (FD_ISSET(fd, &read_fds))
442 3788x flags |= reactor_event_read;
443 11520x if (FD_ISSET(fd, &write_fds))
444 2318x flags |= reactor_event_write;
445 11520x if (FD_ISSET(fd, &except_fds))
446 6x flags |= reactor_event_error;
447
448 11520x if (flags == 0)
449 5412x continue;
450
451 6108x desc->add_ready_events(flags);
452
453 6108x bool expected = false;
454 6108x if (desc->is_enqueued_.compare_exchange_strong(
455 expected, true, std::memory_order_release,
456 std::memory_order_relaxed))
457 {
458 4944x local_ops.push(desc);
459 }
460 }
461 }
462
463 38601x lock.lock();
464
465 38601x completed_ops_.splice(local_ops);
466 38602x }
467
468 } // namespace boost::corosio::detail
469
470 #endif // BOOST_COROSIO_HAS_SELECT
471
472 #endif // BOOST_COROSIO_NATIVE_DETAIL_SELECT_SELECT_SCHEDULER_HPP
473