Files
KPN/tests/test_scheduler.cpp
T
dtourolle abbb2d4770 fix: submitting to a stopped pool must be refused, not fatal
ThreadPool::stop() ends with queues_.clear(), and submit() went straight to
queues_[target] with no check. A submission arriving after stop indexed an
empty vector and segfaulted.

This is not a contrived teardown ordering. A node's space callback fires from
whichever thread drained the channel, and that thread belongs to the
*consumer*; the callback it runs belongs to the *producer*. Stop the producer
first — which a sources-first shutdown does by design — and the consumer keeps
draining its backlog, firing the producer's space callback into a pool that
has already been torn down:

    ThreadPool::submit
      <- source node's space_callback
      <- Channel<int>::try_pop_now      (relay draining its input)
      <- relay fire_once

The static-network shutdown case in the next commit crashed about 12 runs in
20 on this. It survived until now because halt() stops in reverse topological
order — consumers first — so the producer whose callback might fire is always
still alive. shutdown() stops sources first and does not have that protection.

Reading stopped_ without a lock would not fix it: the window between the read
and the indexing is exactly where clear() runs. submit() takes a shared lock
and stop() an exclusive one, so submissions still proceed in parallel with
each other while being serialised against teardown. stop() sets the flag under
the lock, releases it to join — a worker's task may itself call submit, and
holding the lock across the join would deadlock against that — then retakes it
to destroy the queues.

Refusals are counted rather than silent. A teardown race is expected, but a
node repeatedly trying to run after its pool is gone is worth being able to
see. try_submit also checks stop_flag_ first, so a stopped node cannot claim
the submit gate and leave it held.

Not a smart-pointer problem, for anyone reading the crash: nothing here is
owned by a raw pointer. It is std::vector::operator[] on a vector that was
emptied by another thread.

Verified in both directions: with the guard removed the new scheduler cases
segfault; with it they pass.
2026-08-05 15:14:06 +02:00

287 lines
9.1 KiB
C++

#include <catch2/catch_test_macros.hpp>
#include <kpn/scheduler.hpp>
#include <atomic>
#include <chrono>
#include <thread>
#include <vector>
#include <mutex>
using namespace kpn;
using namespace std::chrono_literals;
// ── basic execution ───────────────────────────────────────────────────────────
TEST_CASE("scheduler runs submitted tasks", "[scheduler]") {
ThreadPool pool(2);
pool.start();
std::atomic<int> counter{0};
for (int i = 0; i < 100; ++i)
pool.submit([&counter]{ counter.fetch_add(1, std::memory_order_relaxed); });
pool.drain();
REQUIRE(counter.load() == 100);
pool.stop();
}
TEST_CASE("scheduler single thread executes all tasks", "[scheduler]") {
ThreadPool pool(1);
pool.start();
std::atomic<int> counter{0};
for (int i = 0; i < 50; ++i)
pool.submit([&counter]{ counter.fetch_add(1, std::memory_order_relaxed); });
pool.drain();
REQUIRE(counter.load() == 50);
pool.stop();
}
// ── drain ─────────────────────────────────────────────────────────────────────
TEST_CASE("drain returns immediately when pool is idle", "[scheduler]") {
ThreadPool pool(2);
pool.start();
pool.drain(); // nothing submitted — should return immediately
pool.stop();
}
TEST_CASE("drain waits for all tasks to complete", "[scheduler]") {
ThreadPool pool(4);
pool.start();
std::atomic<int> counter{0};
constexpr int N = 200;
for (int i = 0; i < N; ++i) {
pool.submit([&counter]{
std::this_thread::sleep_for(1ms);
counter.fetch_add(1, std::memory_order_relaxed);
});
}
pool.drain();
REQUIRE(counter.load() == N);
pool.stop();
}
TEST_CASE("drain is safe to call multiple times", "[scheduler]") {
ThreadPool pool(2);
pool.start();
std::atomic<int> counter{0};
pool.submit([&counter]{ counter.fetch_add(1, std::memory_order_relaxed); });
pool.drain();
REQUIRE(counter.load() == 1);
pool.submit([&counter]{ counter.fetch_add(1, std::memory_order_relaxed); });
pool.drain();
REQUIRE(counter.load() == 2);
pool.stop();
}
// ── priority ordering ─────────────────────────────────────────────────────────
TEST_CASE("higher priority tasks run before lower priority on single thread", "[scheduler]") {
// Single thread guarantees serial execution — we can observe order.
ThreadPool pool(1);
pool.start();
// Pause the worker so we can fill the queue before it drains.
std::mutex gate;
gate.lock();
pool.submit([&gate]{ std::lock_guard lg(gate); }); // blocks worker
std::vector<float> order;
std::mutex order_mx;
for (float p : {0.1f, 0.9f, 0.5f, 0.8f, 0.2f}) {
pool.submit([p, &order, &order_mx]{
std::lock_guard lg(order_mx);
order.push_back(p);
}, p);
}
gate.unlock(); // release the blocking task
pool.drain();
pool.stop();
// order should be descending by priority
REQUIRE(order.size() == 5);
for (std::size_t i = 1; i < order.size(); ++i)
REQUIRE(order[i - 1] >= order[i]);
}
TEST_CASE("equal priority tasks execute in FIFO order on single thread", "[scheduler]") {
ThreadPool pool(1);
pool.start();
std::mutex gate;
gate.lock();
pool.submit([&gate]{ std::lock_guard lg(gate); });
std::vector<int> order;
std::mutex order_mx;
for (int i = 0; i < 5; ++i) {
pool.submit([i, &order, &order_mx]{
std::lock_guard lg(order_mx);
order.push_back(i);
}, 0.5f); // all same priority
}
gate.unlock();
pool.drain();
pool.stop();
REQUIRE(order == std::vector<int>{0, 1, 2, 3, 4});
}
// ── total_ / active_ accounting ───────────────────────────────────────────────
TEST_CASE("snapshot queue depth and active counts are consistent", "[scheduler]") {
ThreadPool pool(2);
pool.start();
// While tasks are running, active should be > 0 and total >= active.
std::atomic<bool> running{false};
std::mutex gate;
gate.lock();
for (int i = 0; i < 4; ++i) {
pool.submit([&gate, &running]{
running.store(true, std::memory_order_relaxed);
std::lock_guard lg(gate);
});
}
// Spin until at least one task has started
while (!running.load(std::memory_order_relaxed))
std::this_thread::yield();
auto snap = pool.snapshot("test");
REQUIRE(snap.active_count > 0);
REQUIRE(snap.queue_depth + snap.active_count > 0);
gate.unlock();
pool.drain();
auto snap2 = pool.snapshot("test");
REQUIRE(snap2.active_count == 0);
REQUIRE(snap2.queue_depth == 0);
pool.stop();
}
TEST_CASE("submitted and completed counters are accurate", "[scheduler]") {
ThreadPool pool(3);
pool.start();
constexpr int N = 60;
for (int i = 0; i < N; ++i)
pool.submit([]{ std::this_thread::yield(); });
pool.drain();
auto snap = pool.snapshot("test");
REQUIRE(snap.tasks_submitted == static_cast<uint64_t>(N));
REQUIRE(snap.tasks_completed == static_cast<uint64_t>(N));
pool.stop();
}
// ── work stealing ─────────────────────────────────────────────────────────────
TEST_CASE("work stealing: all tasks complete with uneven initial distribution", "[scheduler]") {
// 4-thread pool. Submit a burst to ensure some threads start empty and must steal.
ThreadPool pool(4);
pool.start();
std::atomic<int> counter{0};
constexpr int N = 400;
for (int i = 0; i < N; ++i)
pool.submit([&counter]{
std::this_thread::sleep_for(100us);
counter.fetch_add(1, std::memory_order_relaxed);
});
pool.drain();
REQUIRE(counter.load() == N);
pool.stop();
}
TEST_CASE("work stealing: tasks complete with more threads than initial queue targets", "[scheduler]") {
// With round-robin, some threads may get no tasks initially and must steal.
constexpr std::size_t THREADS = 8;
ThreadPool pool(THREADS);
pool.start();
std::atomic<int> counter{0};
// Submit fewer tasks than threads so most threads must steal
for (int i = 0; i < 4; ++i)
pool.submit([&counter]{ counter.fetch_add(1, std::memory_order_relaxed); });
pool.drain();
REQUIRE(counter.load() == 4);
pool.stop();
}
// Regression: submitting to a stopped pool must be a no-op, not a segfault.
//
// stop() ends with queues_.clear(), and submit() went straight to
// queues_[target] with no check — so a submission arriving after stop indexed
// an empty vector.
//
// This is not a contrived teardown ordering; it happens on a normal path. A
// node's space callback fires from whichever thread drained the channel, and
// that thread belongs to the *consumer*. Stop the producer first — which a
// sources-first shutdown does by design — and the consumer keeps draining its
// backlog, firing the producer's space callback into a pool that has already
// been torn down. Before this fix the static-network shutdown case crashed
// about 12 runs in 20.
//
// Checking stopped_ without the lock would not be enough: the window between
// reading the flag and indexing the vector is exactly where clear() runs.
TEST_CASE("submitting to a stopped pool is refused, not fatal", "[scheduler]") {
ThreadPool pool(2);
pool.start();
pool.stop();
std::atomic<int> ran{0};
for (int i = 0; i < 10; ++i)
pool.submit([&] { ran.fetch_add(1, std::memory_order_relaxed); });
CHECK(ran.load(std::memory_order_relaxed) == 0);
CHECK(pool.rejected() == 10);
}
TEST_CASE("submitting while the pool stops does not crash", "[scheduler]") {
// The racing form of the case above: a producer thread submitting
// continuously while stop() runs underneath it. Nothing is asserted about
// how many tasks run — the point is that every submission either enqueues
// or is refused, and none touches a destroyed queue.
for (int rep = 0; rep < 20; ++rep) {
ThreadPool pool(4);
pool.start();
std::atomic<bool> go{false};
std::atomic<int> ran{0};
std::thread submitter([&] {
while (!go.load(std::memory_order_acquire)) {}
for (int i = 0; i < 2000; ++i)
pool.submit([&] { ran.fetch_add(1, std::memory_order_relaxed); });
});
go.store(true, std::memory_order_release);
std::this_thread::sleep_for(std::chrono::microseconds(200));
pool.stop();
submitter.join();
// Everything submitted was either executed or refused; nothing vanished
// into a queue that no longer existed.
CHECK(pool.rejected() + pool.snapshot("p").tasks_completed <= 2000);
}
}