|
|
@@ -94,22 +94,35 @@ void WebhookController::stop() {
|
|
|
|
|
|
// A worker that already popped a task is running it outside this lock,
|
|
|
// so notify_all above does not reach it - dispatchWorkerLoop only checks
|
|
|
- // dispatch_running_ between tasks. Give it a moment to reach
|
|
|
+ // dispatch_running_ between tasks. That worker still has to reach
|
|
|
// registerActiveTask (called right before the blocking gRPC call starts)
|
|
|
- // before looking for it below: 2 seconds is far more than that handful of
|
|
|
- // synchronous calls needs, comfortably inside systemd's 90 second default
|
|
|
- // TimeoutStopSec with room to spare for TryCancel to take effect and the
|
|
|
- // join after it, and short enough that an ordinary restart with nothing
|
|
|
- // in flight barely notices it.
|
|
|
- std::this_thread::sleep_for(std::chrono::seconds(2));
|
|
|
-
|
|
|
+ // before it becomes visible here.
|
|
|
{
|
|
|
- std::lock_guard<std::mutex> lock(active_mutex_);
|
|
|
+ std::unique_lock<std::mutex> lock(active_mutex_);
|
|
|
+ // Setting this under the same lock registerActiveTask takes is what
|
|
|
+ // makes the race deterministic rather than timing-dependent: a task
|
|
|
+ // that registers after this point - however late, no matter how
|
|
|
+ // loaded the machine is - sees shutting_down_ already true and
|
|
|
+ // cancels itself right there (see registerActiveTask), instead of
|
|
|
+ // depending on a cancellation pass here having already run by the
|
|
|
+ // time it arrives.
|
|
|
+ shutting_down_ = true;
|
|
|
for (const auto& active : active_tasks_) {
|
|
|
LOG_WARN("Cancelling in-flight immediate-mode form dispatch for workflow {} at shutdown",
|
|
|
active.workflow_id);
|
|
|
active.context->TryCancel();
|
|
|
}
|
|
|
+ // Wait for every active task to unregister (its worker returned from
|
|
|
+ // the gRPC call and is on its way back to dispatchWorkerLoop, which
|
|
|
+ // exits immediately once dispatch_running_ is false), rather than
|
|
|
+ // sleeping a flat 2 seconds regardless of whether anything was
|
|
|
+ // running at all. Still bounded by the same grace period as before -
|
|
|
+ // TryCancel is expected to unblock everything well inside it - so a
|
|
|
+ // pathological case does not turn shutdown unbounded, it just stops
|
|
|
+ // waiting and lets the join below find out how long it actually
|
|
|
+ // takes.
|
|
|
+ active_cv_.wait_for(lock, std::chrono::seconds(2),
|
|
|
+ [this] { return active_tasks_.empty(); });
|
|
|
}
|
|
|
|
|
|
// TryCancel unblocks this process's own client call quickly regardless
|
|
|
@@ -165,11 +178,22 @@ bool WebhookController::tryEnqueueDispatch(const std::string& workflow_id, std::
|
|
|
void WebhookController::registerActiveTask(::grpc::ClientContext* context, const std::string& workflow_id) {
|
|
|
std::lock_guard<std::mutex> lock(active_mutex_);
|
|
|
active_tasks_.push_back(ActiveTask{context, workflow_id});
|
|
|
+ if (shutting_down_) {
|
|
|
+ // stop()'s own cancellation pass already ran by the time this task
|
|
|
+ // reached here - it would otherwise sit uncancelled until the
|
|
|
+ // bounded wait in stop() gives up, or worse, until the hour-long
|
|
|
+ // deadline set on this context. Cancel it the moment it becomes
|
|
|
+ // visible instead.
|
|
|
+ LOG_WARN("Cancelling in-flight immediate-mode form dispatch for workflow {} at shutdown "
|
|
|
+ "(registered after shutdown began)", workflow_id);
|
|
|
+ context->TryCancel();
|
|
|
+ }
|
|
|
}
|
|
|
|
|
|
void WebhookController::unregisterActiveTask(::grpc::ClientContext* context) {
|
|
|
std::lock_guard<std::mutex> lock(active_mutex_);
|
|
|
std::erase_if(active_tasks_, [context](const ActiveTask& t) { return t.context == context; });
|
|
|
+ active_cv_.notify_all();
|
|
|
}
|
|
|
|
|
|
void WebhookController::registerRoutes(httplib::Server& server) {
|