|
@@ -17,6 +17,7 @@
|
|
|
#include "credentials/credential_store.hpp"
|
|
#include "credentials/credential_store.hpp"
|
|
|
#include "scheduler/workflow_scheduler.hpp"
|
|
#include "scheduler/workflow_scheduler.hpp"
|
|
|
#include "common/time_utils.hpp"
|
|
#include "common/time_utils.hpp"
|
|
|
|
|
+#include <unordered_set>
|
|
|
#include "common/config_defaults.hpp"
|
|
#include "common/config_defaults.hpp"
|
|
|
#include "logging/logger.hpp"
|
|
#include "logging/logger.hpp"
|
|
|
#include <grpcpp/grpcpp.h>
|
|
#include <grpcpp/grpcpp.h>
|
|
@@ -233,7 +234,11 @@ void WebServerService::setupRoutes() {
|
|
|
*node_store_, *auth_middleware_, *load_balancer_, &node_sync_server_->service());
|
|
*node_store_, *auth_middleware_, *load_balancer_, &node_sync_server_->service());
|
|
|
node_ctrl_->registerRoutes(server);
|
|
node_ctrl_->registerRoutes(server);
|
|
|
|
|
|
|
|
- runner_ctrl_ = std::make_unique<api::RunnerController>(*runner_registry_, *auth_middleware_);
|
|
|
|
|
|
|
+ runner_ctrl_ = std::make_unique<api::RunnerController>(
|
|
|
|
|
+ *runner_registry_, *auth_middleware_,
|
|
|
|
|
+ [this](const std::string& runner_id, const std::string& address) {
|
|
|
|
|
+ reconcileOrphanedExecutions(runner_id, address);
|
|
|
|
|
+ });
|
|
|
runner_ctrl_->registerRoutes(server);
|
|
runner_ctrl_->registerRoutes(server);
|
|
|
|
|
|
|
|
webhook_ctrl_ = std::make_unique<api::WebhookController>(
|
|
webhook_ctrl_ = std::make_unique<api::WebhookController>(
|
|
@@ -412,6 +417,75 @@ void WebServerService::noteExecutionOutcome(const std::string& workflow_id,
|
|
|
});
|
|
});
|
|
|
}
|
|
}
|
|
|
|
|
|
|
|
|
|
+void WebServerService::reconcileOrphanedExecutions(const std::string& runner_id,
|
|
|
|
|
+ const std::string& address) {
|
|
|
|
|
+ // Ask the runner what it is actually running rather than assuming. A runner
|
|
|
|
|
+ // that re-registers while working - a duplicate call, a flapping network -
|
|
|
|
|
+ // must not have its live executions closed underneath it, and only the
|
|
|
|
|
+ // runner knows which those are.
|
|
|
|
|
+ std::unordered_set<std::string> still_running;
|
|
|
|
|
+ {
|
|
|
|
|
+ auto channel = ::grpc::CreateChannel(address, ::grpc::InsecureChannelCredentials());
|
|
|
|
|
+ auto stub = proto::RunnerService::NewStub(channel);
|
|
|
|
|
+
|
|
|
|
|
+ proto::ListActiveExecutionsRequest request;
|
|
|
|
|
+ proto::ListActiveExecutionsResponse response;
|
|
|
|
|
+ ::grpc::ClientContext context;
|
|
|
|
|
+ context.set_deadline(std::chrono::system_clock::now() + std::chrono::seconds(10));
|
|
|
|
|
+
|
|
|
|
|
+ auto status = stub->ListActiveExecutions(&context, request, &response);
|
|
|
|
|
+ if (!status.ok()) {
|
|
|
|
|
+ // Without an answer there is no way to tell an orphan from a live
|
|
|
|
|
+ // run, and closing a live one is far worse than leaving a stale
|
|
|
|
|
+ // record for someone to press Stop on.
|
|
|
|
|
+ LOG_WARN("Runner {} could not say what it is running ({}), so nothing was reconciled",
|
|
|
|
|
+ runner_id, status.error_message());
|
|
|
|
|
+ return;
|
|
|
|
|
+ }
|
|
|
|
|
+ for (const auto& id : response.execution_ids()) {
|
|
|
|
|
+ still_running.insert(id);
|
|
|
|
|
+ }
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ storage::QueryOptions options;
|
|
|
|
|
+ options.filters.push_back({"runnerId", runner_id});
|
|
|
|
|
+ options.filters.push_back({"status", "running"});
|
|
|
|
|
+ options.page = 1;
|
|
|
|
|
+ options.page_size = 500;
|
|
|
|
|
+
|
|
|
|
|
+ auto found = storage_->query("executions", options);
|
|
|
|
|
+ if (found.failed()) {
|
|
|
|
|
+ LOG_WARN("Could not look for orphaned executions on runner {}: {}",
|
|
|
|
|
+ runner_id, found.error().message());
|
|
|
|
|
+ return;
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ int closed = 0;
|
|
|
|
|
+ for (const auto& record : found.value().documents) {
|
|
|
|
|
+ const std::string id = record.value("_id", "");
|
|
|
|
|
+ if (id.empty() || still_running.contains(id)) {
|
|
|
|
|
+ continue;
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ // Deliberately not touching Waiting. That is a run parked on a person,
|
|
|
|
|
+ // not on a runner, and it is meant to outlive one.
|
|
|
|
|
+ const nlohmann::json patch = {
|
|
|
|
|
+ {"status", "cancelled"},
|
|
|
|
|
+ {"error", "The runner restarted while this was running, so nothing was left to finish it"},
|
|
|
|
|
+ {"finishedAt", common::TimeUtils::nowMs()}
|
|
|
|
|
+ };
|
|
|
|
|
+ if (storage_->update("executions", id, patch, 0, true).ok()) {
|
|
|
|
|
+ ++closed;
|
|
|
|
|
+ ws_server_->broadcast("executions." + id + ".cancelled", {{"executionId", id}});
|
|
|
|
|
+ }
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ if (closed > 0) {
|
|
|
|
|
+ LOG_WARN("Closed {} execution(s) that runner {} was recorded as running but is not",
|
|
|
|
|
+ closed, runner_id);
|
|
|
|
|
+ }
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
void WebServerService::runErrorWorkflow(const std::string& failed_workflow_id,
|
|
void WebServerService::runErrorWorkflow(const std::string& failed_workflow_id,
|
|
|
const std::string& failed_execution_id,
|
|
const std::string& failed_execution_id,
|
|
|
const std::string& error_message) {
|
|
const std::string& error_message) {
|