include/boost/corosio/native/detail/select/select_scheduler.hpp

99.4% Lines (167 / 168, 1 excl) 100.0% Functions (11 / 11)
select_scheduler.hpp
f(x) Functions (11)
Line TLA Hits Source Code
1 //
2 // Copyright (c) 2026 Steve Gerbino
3 // Copyright (c) 2026 Michael Vandeberg
4 //
5 // Distributed under the Boost Software License, Version 1.0. (See accompanying
6 // file LICENSE_1_0.txt or copy at http://www.boost.org/LICENSE_1_0.txt)
7 //
8 // Official repository: https://github.com/cppalliance/corosio
9 //
10
11 #ifndef BOOST_COROSIO_NATIVE_DETAIL_SELECT_SELECT_SCHEDULER_HPP
12 #define BOOST_COROSIO_NATIVE_DETAIL_SELECT_SELECT_SCHEDULER_HPP
13
14 #include <boost/corosio/detail/platform.hpp>
15
16 #if BOOST_COROSIO_HAS_SELECT
17
18 #include <boost/corosio/detail/config.hpp>
19 #include <boost/capy/ex/execution_context.hpp>
20
21 #include <boost/corosio/native/detail/reactor/reactor_scheduler.hpp>
22 #include <boost/corosio/native/detail/reactor/reactor_signal_pipe.hpp>
23
24 #include <boost/corosio/native/detail/select/select_traits.hpp>
25 #include <boost/corosio/detail/timer_service.hpp>
26 #include <boost/corosio/native/detail/make_err.hpp>
27 #include <boost/corosio/native/detail/posix/posix_resolver_service.hpp>
28 #include <boost/corosio/native/detail/posix/posix_signal_service.hpp>
29 #include <boost/corosio/native/detail/posix/posix_stream_file_service.hpp>
30 #include <boost/corosio/native/detail/posix/posix_random_access_file_service.hpp>
31
32 #include <boost/corosio/detail/except.hpp>
33
34 #include <sys/select.h>
35 #include <unistd.h>
36 #include <errno.h>
37 #include <fcntl.h>
38
39 #include <atomic>
40 #include <chrono>
41 #include <cstdint>
42 #include <limits>
43 #include <mutex>
44 #include <new>
45 #include <unordered_map>
46
47 namespace boost::corosio::detail {
48
49 struct select_op;
50
51 /** POSIX scheduler using select() for I/O multiplexing.
52
53 This scheduler implements the scheduler interface using the POSIX select()
54 call for I/O event notification. It inherits the shared reactor threading
55 model from reactor_scheduler: signal state machine, inline completion
56 budget, work counting, and the do_one event loop.
57
58 The design mirrors epoll_scheduler for behavioral consistency:
59 - Same single-reactor thread coordination model
60 - Same deferred I/O pattern (reactor marks ready; workers do I/O)
61 - Same timer integration pattern
62
63 Known Limitations:
64 - FD_SETSIZE (~1024) limits maximum concurrent connections
65 - O(n) scanning: rebuilds fd_sets each iteration
66 - Level-triggered only (no edge-triggered mode)
67
68 @par Thread Safety
69 All public member functions are thread-safe.
70 */
71 class BOOST_COROSIO_DECL select_scheduler final : public reactor_scheduler
72 {
73 public:
74 /** Construct the scheduler.
75
76 Creates a self-pipe for reactor interruption.
77
78 @param ctx Reference to the owning execution_context.
79 @param concurrency_hint Hint for expected thread count (unused).
80 */
81 select_scheduler(capy::execution_context& ctx, int concurrency_hint = -1);
82
83 /// Destroy the scheduler.
84 ~select_scheduler() override;
85
86 select_scheduler(select_scheduler const&) = delete;
87 select_scheduler& operator=(select_scheduler const&) = delete;
88
89 /// Shut down the scheduler, draining pending operations.
90 void shutdown() override;
91
92 /** Return the maximum file descriptor value supported.
93
94 Returns FD_SETSIZE - 1, the maximum fd value that can be
95 monitored by select(). Operations with fd >= FD_SETSIZE
96 will fail with EINVAL.
97
98 @return The maximum supported file descriptor value.
99 */
100 static constexpr int max_fd() noexcept
101 {
102 return FD_SETSIZE - 1;
103 }
104
105 /** Register a descriptor for persistent monitoring.
106
107 The fd is added to the registered_descs_ map and will be
108 included in subsequent select() calls. The reactor is
109 interrupted so a blocked select() rebuilds its fd_sets.
110
111 @param fd The file descriptor to register.
112 @param desc Pointer to descriptor state for this fd.
113
114 @return The error if the fd cannot be tracked, otherwise a
115 default constructed error code.
116 */
117 std::error_code
118 register_descriptor(int fd, reactor_descriptor_state* desc) const;
119
120 /** Deregister a persistently registered descriptor.
121
122 @param fd The file descriptor to deregister.
123 */
124 void deregister_descriptor(int fd) const;
125
126 /** Interrupt the reactor so it rebuilds its fd_sets.
127
128 Called when a write, connect, or write-wait op is registered
129 after the reactor's snapshot was taken. Without this,
130 select() may block not watching for writability on the fd.
131 */
132 void notify_reactor() const;
133
134 /// Watch the read end of the POSIX signal self-pipe (see scheduler.hpp).
135 55x [[nodiscard]] std::error_code register_signal_reader(int read_fd) override
136 {
137 55x return register_descriptor(read_fd, signal_pipe_reader_.arm());
138 }
139
140 private:
141 void run_task(lock_type& lock, context_type& ctx, long timeout_us) override;
142 void interrupt_reactor() const override;
143 long calculate_timeout(long requested_timeout_us) const;
144
145 // Watches the global signal self-pipe's read end (armed lazily by
146 // register_signal_reader on the first signal registration).
147 reactor_signal_pipe_reader signal_pipe_reader_;
148
149 // Self-pipe for interrupting select()
150 int pipe_fds_[2]; // [0]=read, [1]=write
151
152 // Per-fd tracking for fd_set building
153 mutable std::unordered_map<int, reactor_descriptor_state*>
154 registered_descs_;
155 mutable int max_fd_ = -1;
156 };
157
158 886x inline select_scheduler::select_scheduler(capy::execution_context& ctx, int)
159 886x : pipe_fds_{-1, -1}
160 886x , max_fd_(-1)
161 {
162 886x if (::pipe(pipe_fds_) < 0)
163 1x detail::throw_system_error(make_err(errno), "pipe");
164
165 2646x for (int i = 0; i < 2; ++i)
166 {
167 1767x int flags = ::fcntl(pipe_fds_[i], F_GETFL, 0);
168 1767x if (flags == -1)
169 {
170 2x int errn = errno;
171 2x ::close(pipe_fds_[0]);
172 2x ::close(pipe_fds_[1]);
173 2x detail::throw_system_error(make_err(errn), "fcntl F_GETFL");
174 }
175 1765x if (::fcntl(pipe_fds_[i], F_SETFL, flags | O_NONBLOCK) == -1)
176 {
177 2x int errn = errno;
178 2x ::close(pipe_fds_[0]);
179 2x ::close(pipe_fds_[1]);
180 2x detail::throw_system_error(make_err(errn), "fcntl F_SETFL");
181 }
182 1763x if (::fcntl(pipe_fds_[i], F_SETFD, FD_CLOEXEC) == -1)
183 {
184 2x int errn = errno;
185 2x ::close(pipe_fds_[0]);
186 2x ::close(pipe_fds_[1]);
187 2x detail::throw_system_error(make_err(errn), "fcntl F_SETFD");
188 }
189 }
190
191 879x timer_svc_ = &get_timer_service(ctx, *this);
192 879x timer_svc_->set_on_earliest_changed(
193 3436x timer_service::callback(this, [](void* p) {
194 2557x static_cast<select_scheduler*>(p)->interrupt_reactor();
195 2557x }));
196
197 879x get_resolver_service(ctx, *this);
198 879x get_signal_service(ctx, *this);
199 879x get_stream_file_service(ctx, *this);
200 879x get_random_access_file_service(ctx, *this);
201
202 879x completed_ops_.push(&task_op_);
203 900x }
204
205 1758x inline select_scheduler::~select_scheduler()
206 {
207 879x if (pipe_fds_[0] >= 0)
208 879x ::close(pipe_fds_[0]);
209 879x if (pipe_fds_[1] >= 0)
210 879x ::close(pipe_fds_[1]);
211 1758x }
212
213 inline void
214 879x select_scheduler::shutdown()
215 {
216 879x shutdown_drain();
217
218 879x if (pipe_fds_[1] >= 0)
219 879x interrupt_reactor();
220 879x }
221
222 inline std::error_code
223 4842x select_scheduler::register_descriptor(
224 int fd, reactor_descriptor_state* desc) const
225 {
226 4842x if (fd < 0 || fd >= FD_SETSIZE)
227 1x return make_err(EMFILE);
228
229 4841x desc->registered_events = reactor_event_read | reactor_event_write;
230 4841x desc->fd = fd;
231 4841x desc->scheduler_ = this;
232 4841x desc->mutex.set_enabled(reactor_io_locking_);
233 4841x desc->ready_events_.store(0, std::memory_order_relaxed);
234
235 {
236 4841x conditionally_enabled_mutex::scoped_lock lock(desc->mutex);
237 4841x desc->impl_ref_.reset();
238 4841x desc->read_ready = false;
239 4841x desc->write_ready = false;
240 4841x }
241
242 {
243 4841x mutex_type::scoped_lock lock(mutex_);
244 try
245 {
246 4841x registered_descs_[fd] = desc;
247 }
248 1x catch (std::bad_alloc const&)
249 {
250 1x return make_err(ENOMEM);
251 1x }
252 4840x if (fd > max_fd_)
253 4786x max_fd_ = fd;
254 4841x }
255
256 4840x interrupt_reactor();
257 4840x return {};
258 }
259
260 inline void
261 4786x select_scheduler::deregister_descriptor(int fd) const
262 {
263 4786x mutex_type::scoped_lock lock(mutex_);
264
265 4786x auto it = registered_descs_.find(fd);
266 4786x if (it == registered_descs_.end())
267 return;
268
269 4786x registered_descs_.erase(it);
270
271 4786x if (fd == max_fd_)
272 {
273 4449x max_fd_ = pipe_fds_[0];
274 8496x for (auto& [registered_fd, state] : registered_descs_)
275 {
276 4047x if (registered_fd > max_fd_)
277 3954x max_fd_ = registered_fd;
278 }
279 }
280 4786x }
281
282 inline void
283 2153x select_scheduler::notify_reactor() const
284 {
285 2153x interrupt_reactor();
286 2153x }
287
288 inline void
289 11223x select_scheduler::interrupt_reactor() const
290 {
291 11223x char byte = 1;
292 11223x [[maybe_unused]] auto r = ::write(pipe_fds_[1], &byte, 1);
293 11223x }
294
295 inline long
296 292405x select_scheduler::calculate_timeout(long requested_timeout_us) const
297 {
298 292405x if (requested_timeout_us == 0)
299 return 0; // LCOV_EXCL_LINE run_task passes 0 via task_interrupted_, never through this argument
300
301 292405x auto nearest = timer_svc_->nearest_expiry();
302 292405x if (nearest == timer_service::time_point::max())
303 726x return requested_timeout_us;
304
305 291679x auto now = std::chrono::steady_clock::now();
306 291679x if (nearest <= now)
307 691x return 0;
308
309 auto timer_timeout_us =
310 290988x std::chrono::duration_cast<std::chrono::microseconds>(nearest - now)
311 290988x .count();
312
313 290988x constexpr auto long_max =
314 static_cast<long long>((std::numeric_limits<long>::max)());
315 auto capped_timer_us =
316 290988x (std::min)((std::max)(static_cast<long long>(timer_timeout_us),
317 290988x static_cast<long long>(0)),
318 290988x long_max);
319
320 290988x if (requested_timeout_us < 0)
321 290986x return static_cast<long>(capped_timer_us);
322
323 return static_cast<long>(
324 2x (std::min)(static_cast<long long>(requested_timeout_us),
325 2x capped_timer_us));
326 }
327
328 inline void
329 315456x select_scheduler::run_task(lock_type& lock, context_type& ctx, long timeout_us)
330 {
331 long effective_timeout_us =
332 315456x task_interrupted_ ? 0 : calculate_timeout(timeout_us);
333
334 // Snapshot registered descriptors while holding lock.
335 // Record which fds need write monitoring to avoid a hot loop:
336 // select is level-triggered so writable sockets (nearly always
337 // writable) would cause select() to return immediately every
338 // iteration if unconditionally added to write_fds. Membership
339 // stays opt-in: a parked write wait opts in the same way a
340 // parked write or connect op does.
341 struct fd_entry
342 {
343 int fd;
344 reactor_descriptor_state* desc;
345 bool needs_write;
346 };
347 fd_entry snapshot[FD_SETSIZE];
348 315456x int snapshot_count = 0;
349
350 816306x for (auto& [fd, desc] : registered_descs_)
351 {
352 500850x if (snapshot_count < FD_SETSIZE)
353 {
354 500850x conditionally_enabled_mutex::scoped_lock desc_lock(desc->mutex);
355 500850x snapshot[snapshot_count].fd = fd;
356 500850x snapshot[snapshot_count].desc = desc;
357 500850x snapshot[snapshot_count].needs_write =
358 500850x (desc->write_op || desc->connect_op || desc->wait_write_op);
359 500850x ++snapshot_count;
360 500850x }
361 }
362
363 315456x if (lock.owns_lock())
364 292406x lock.unlock();
365
366 315456x task_cleanup on_exit{this, &lock, ctx};
367
368 fd_set read_fds, write_fds, except_fds;
369 5362752x FD_ZERO(&read_fds);
370 5362752x FD_ZERO(&write_fds);
371 5362752x FD_ZERO(&except_fds);
372
373 315456x FD_SET(pipe_fds_[0], &read_fds);
374 315456x int nfds = pipe_fds_[0];
375
376 816306x for (int i = 0; i < snapshot_count; ++i)
377 {
378 500850x int fd = snapshot[i].fd;
379 500850x FD_SET(fd, &read_fds);
380 500850x if (snapshot[i].needs_write)
381 12815x FD_SET(fd, &write_fds);
382 500850x FD_SET(fd, &except_fds);
383 500850x if (fd > nfds)
384 315058x nfds = fd;
385 }
386
387 struct timeval tv;
388 315456x struct timeval* tv_ptr = nullptr;
389 315456x if (effective_timeout_us >= 0)
390 {
391 314747x tv.tv_sec = effective_timeout_us / 1000000;
392 314747x tv.tv_usec = effective_timeout_us % 1000000;
393 314747x tv_ptr = &tv;
394 }
395
396 315456x int ready = ::select(nfds + 1, &read_fds, &write_fds, &except_fds, tv_ptr);
397
398 // EINTR: signal interrupted select(), just retry.
399 // EBADF: an fd was closed between snapshot and select(); retry
400 // with a fresh snapshot from registered_descs_.
401 // Both fall through with no ready descriptors rather than
402 // returning: the caller handed this function an owned lock that
403 // only the epilogue below re-acquires.
404 315456x if (ready < 0)
405 {
406 3x if (errno != EINTR && errno != EBADF)
407 1x detail::throw_system_error(make_err(errno), "select");
408 2x ready = 0;
409 }
410
411 // Process timers outside the lock
412 315455x timer_svc_->process_expired();
413
414 315455x ready_queue local_ops;
415
416 315455x if (ready > 0)
417 {
418 299643x if (FD_ISSET(pipe_fds_[0], &read_fds))
419 {
420 char buf[256];
421 10006x while (::read(pipe_fds_[0], buf, sizeof(buf)) > 0)
422 {
423 }
424 }
425
426 753259x for (int i = 0; i < snapshot_count; ++i)
427 {
428 453616x int fd = snapshot[i].fd;
429 453616x reactor_descriptor_state* desc = snapshot[i].desc;
430
431 453616x std::uint32_t flags = 0;
432 453616x if (FD_ISSET(fd, &read_fds))
433 299329x flags |= reactor_event_read;
434 453616x if (FD_ISSET(fd, &write_fds))
435 2145x flags |= reactor_event_write;
436 453616x if (FD_ISSET(fd, &except_fds))
437 16x flags |= reactor_event_error;
438
439 453616x if (flags == 0)
440 152151x continue;
441
442 301465x desc->add_ready_events(flags);
443
444 301465x bool expected = false;
445 301465x if (desc->is_enqueued_.compare_exchange_strong(
446 expected, true, std::memory_order_release,
447 std::memory_order_relaxed))
448 {
449 301465x local_ops.push(desc);
450 }
451 }
452 }
453
454 315455x lock.lock();
455
456 315455x completed_ops_.splice(local_ops);
457 315456x }
458
459 } // namespace boost::corosio::detail
460
461 #endif // BOOST_COROSIO_HAS_SELECT
462
463 #endif // BOOST_COROSIO_NATIVE_DETAIL_SELECT_SELECT_SCHEDULER_HPP
464