runner_service.cpp 41 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642643644645646647648649650651652653654655656657658659660661662663664665666667668669670671672673674675676677678679680681682683684685686687688689690691692693694695696697698699700701702703704705706707708709710711712713714715716717718719720721722723724725726727728729730731732733734735736737738739740741742743744745746747748749750751752753754755756757758759760761762763764765766767768769770771772773774775776777778779780781782783784785786787788789790791792793794795796797798799800801802803804805806807808809810811812813814815816817818819820821822823824825826827828829830831832833834835836837838839840841842843844845846847848849850851852853854855856857858859860861862863864865866867868869870871872873874875876877878879880881882883884885886887888889890891892893894895896897898899900901902903904905906907908909910911912913914915916917918919920921922923924925926927928929930931932933934935936937938939940941942943944945946947948949950951952953954955956957958959960961962963964965966967968969970971972973974975976977978979980981982983984985986987988989990991992993994995996997998999100010011002100310041005100610071008100910101011101210131014101510161017101810191020
  1. #include "runner_service.hpp"
  2. #include "common/uuid.hpp"
  3. #include "common/time_utils.hpp"
  4. #include "logging/logger.hpp"
  5. #include "proto/runner.grpc.pb.h"
  6. #include <grpcpp/health_check_service_interface.h>
  7. #include <sys/resource.h>
  8. #include <fstream>
  9. #include <curl/curl.h>
  10. #ifdef MYSQL_SUPPORT
  11. #include "runner/mysql/mysql_client.hpp"
  12. #endif
  13. #ifdef POSTGRESQL_SUPPORT
  14. #include "runner/postgresql/postgresql_client.hpp"
  15. #endif
  16. namespace smartbotic::runner {
  17. using namespace common;
  18. // Maps the engine's ExecutionStatus onto the wire enum explicitly. The two
  19. // enums are not numerically aligned: proto::ExecutionStatus reserves 0 for
  20. // EXECUTION_STATUS_UNSPECIFIED, while ExecutionStatus::Pending is 0, so a
  21. // bare static_cast silently shifts every status by one. Deliberately no
  22. // default label, so an unhandled case is a compiler warning rather than a
  23. // silent mismatch.
  24. static proto::ExecutionStatus toProtoStatus(ExecutionStatus status) {
  25. switch (status) {
  26. case ExecutionStatus::Pending: return proto::EXECUTION_STATUS_PENDING;
  27. case ExecutionStatus::Running: return proto::EXECUTION_STATUS_RUNNING;
  28. case ExecutionStatus::Completed: return proto::EXECUTION_STATUS_COMPLETED;
  29. case ExecutionStatus::Failed: return proto::EXECUTION_STATUS_FAILED;
  30. case ExecutionStatus::Cancelled: return proto::EXECUTION_STATUS_CANCELLED;
  31. case ExecutionStatus::Waiting: return proto::EXECUTION_STATUS_WAITING;
  32. }
  33. return proto::EXECUTION_STATUS_UNSPECIFIED;
  34. }
  35. // RunnerServiceImpl implementation
  36. RunnerServiceImpl::RunnerServiceImpl(WorkflowEngine& engine, NodeRegistry& registry,
  37. storage::StorageClient& storage,
  38. ExecutionEventCallback event_callback)
  39. : engine_(engine), registry_(registry), storage_(storage), event_callback_(event_callback) {}
  40. grpc::Status RunnerServiceImpl::ExecuteWorkflow(grpc::ServerContext* context,
  41. const proto::ExecuteWorkflowRequest* request,
  42. proto::ExecuteWorkflowResponse* response) {
  43. LOG_INFO("ExecuteWorkflow called for workflow: {}", request->workflow_id());
  44. // An inline workflow runs as given, without being stored anywhere. This is
  45. // how the editor asks a node what it can offer - which models a server has,
  46. // say - while its config is still being edited.
  47. if (!request->inline_workflow().empty()) {
  48. nlohmann::json inline_doc;
  49. try {
  50. inline_doc = nlohmann::json::parse(request->inline_workflow());
  51. } catch (const std::exception& e) {
  52. response->set_status(proto::EXECUTION_STATUS_FAILED);
  53. auto* error = response->mutable_error();
  54. error->set_message(std::string("inline_workflow is not valid JSON: ") + e.what());
  55. return grpc::Status::OK;
  56. }
  57. // The id decides which credentials this run may read, so it comes from
  58. // the request rather than from the caller-supplied document.
  59. inline_doc["_id"] = request->workflow_id();
  60. auto inline_workflow = Workflow::fromJson(inline_doc);
  61. nlohmann::json inline_trigger;
  62. if (!request->trigger_data().empty()) {
  63. try {
  64. inline_trigger = nlohmann::json::parse(request->trigger_data());
  65. } catch (...) {}
  66. }
  67. auto inline_outcome = engine_.execute(inline_workflow, "manual", inline_trigger, nullptr);
  68. if (inline_outcome.failed()) {
  69. response->set_status(proto::EXECUTION_STATUS_FAILED);
  70. auto* error = response->mutable_error();
  71. error->set_message(inline_outcome.error().message());
  72. return grpc::Status::OK;
  73. }
  74. const auto& inline_result = inline_outcome.value();
  75. response->set_execution_id(inline_result.execution_id);
  76. response->set_status(toProtoStatus(inline_result.status));
  77. response->set_result(inline_result.toJson().dump());
  78. return grpc::Status::OK;
  79. }
  80. // Get workflow from database
  81. auto workflow_result = storage_.get("workflows", request->workflow_id());
  82. // A trigger runs what was published. The record itself is the draft.
  83. if (request->use_published() && workflow_result.ok()) {
  84. const int64_t published = workflow_result.value().value("publishedVersion", int64_t{0});
  85. const int64_t current = workflow_result.value().value("_version", int64_t{0});
  86. if (published > 0 && published != current) {
  87. auto pinned = storage_.getVersion("workflows", request->workflow_id(), published);
  88. if (pinned.ok()) {
  89. LOG_INFO("Workflow {} running published version {} (draft is {})",
  90. request->workflow_id(), published, current);
  91. // A stored version is the document as it was, and the database
  92. // keeps _id outside the document - so a version comes back
  93. // without one. Everything downstream identifies the run by it,
  94. // and without it the execution recorded an empty workflowId:
  95. // the run happened, but the workflow's own execution list never
  96. // showed it, which reads as "it did not run".
  97. pinned.value()["_id"] = request->workflow_id();
  98. workflow_result = pinned;
  99. } else {
  100. // Refusing would stop a live workflow because its history has
  101. // aged out, which is worse than running the draft - but it must
  102. // be said out loud, because the two can differ.
  103. LOG_WARN("Workflow {}: published version {} is no longer stored, "
  104. "running the current draft instead", request->workflow_id(), published);
  105. }
  106. }
  107. }
  108. if (workflow_result.failed()) {
  109. LOG_ERROR("Failed to get workflow from database: {}", workflow_result.error().message());
  110. response->set_status(proto::EXECUTION_STATUS_FAILED);
  111. auto* error = response->mutable_error();
  112. error->set_code(static_cast<int32_t>(workflow_result.error().code()));
  113. error->set_message(workflow_result.error().message());
  114. return grpc::Status::OK;
  115. }
  116. // Parse workflow
  117. auto workflow = Workflow::fromJson(workflow_result.value());
  118. // Parse trigger data
  119. nlohmann::json trigger_data;
  120. if (!request->trigger_data().empty()) {
  121. try {
  122. trigger_data = nlohmann::json::parse(request->trigger_data());
  123. } catch (...) {
  124. trigger_data = request->trigger_data();
  125. }
  126. }
  127. // Create callback to forward events
  128. ExecutionCallback callback;
  129. if (event_callback_) {
  130. callback = [this, &workflow](const std::string& event_type, const nlohmann::json& data) {
  131. nlohmann::json event_data = data;
  132. event_data["workflowId"] = workflow.id;
  133. event_callback_(event_type, event_data);
  134. };
  135. }
  136. // Execute workflow with callback
  137. LOG_INFO("Starting workflow execution...");
  138. auto result = engine_.execute(workflow, request->trigger_type(), trigger_data, callback);
  139. if (result.failed()) {
  140. LOG_ERROR("Workflow execution failed: {}", result.error().message());
  141. // Emit execution.failed event
  142. if (event_callback_) {
  143. event_callback_("execution.failed", {
  144. {"executionId", ""},
  145. {"workflowId", workflow.id},
  146. {"error", result.error().message()}
  147. });
  148. }
  149. response->set_status(proto::EXECUTION_STATUS_FAILED);
  150. auto* error = response->mutable_error();
  151. error->set_code(static_cast<int32_t>(result.error().code()));
  152. error->set_message(result.error().message());
  153. return grpc::Status::OK;
  154. }
  155. LOG_INFO("Workflow execution completed, execution_id: {}, status: {}",
  156. result.value().execution_id, executionStatusToString(result.value().status));
  157. // No terminal event is emitted here. The engine already emits one for
  158. // every finished run, on its own way out, and this was a second copy of
  159. // the same event: every completed run broadcast execution.completed twice
  160. // and every failed run broadcast execution.failed twice. Confirmed on the
  161. // wire before removing - two broadcasts a millisecond apart, from one
  162. // "Workflow execution completed" in the runner's log.
  163. //
  164. // The duplicate was not just noise. The webserver runs an error workflow
  165. // on execution.failed, so an error handler ran twice per failure, and it
  166. // releases the scheduler slot on the same event.
  167. //
  168. // The engine's is also the better of the two: it covers waiting and
  169. // cancelled rather than only the two statuses handled here, it carries the
  170. // status and error fields, it truncates a large final output instead of
  171. // posting the whole thing over HTTP, and it fires on the resume path too -
  172. // which is why a resumed run already emitted exactly once and made the
  173. // asymmetry visible.
  174. //
  175. // The failure branch above stays: it covers execute() itself returning an
  176. // error, where the engine produced no result and emitted nothing.
  177. response->set_execution_id(result.value().execution_id);
  178. response->set_status(toProtoStatus(result.value().status));
  179. if (request->wait_for_completion()) {
  180. if (!result.value().webhook_response.is_null()) {
  181. nlohmann::json envelope;
  182. envelope["_webhookResponse"] = result.value().webhook_response;
  183. response->set_result(envelope.dump());
  184. } else {
  185. response->set_result(result.value().final_output.dump());
  186. }
  187. }
  188. return grpc::Status::OK;
  189. }
  190. grpc::Status RunnerServiceImpl::CancelExecution(grpc::ServerContext* context,
  191. const proto::CancelExecutionRequest* request,
  192. proto::CancelExecutionResponse* response) {
  193. const auto outcome = engine_.cancelExecution(request->execution_id());
  194. response->set_was_running(outcome.was_running);
  195. response->set_marked(outcome.marked);
  196. return grpc::Status::OK;
  197. }
  198. grpc::Status RunnerServiceImpl::ListActiveExecutions(
  199. grpc::ServerContext* context,
  200. const proto::ListActiveExecutionsRequest* request,
  201. proto::ListActiveExecutionsResponse* response) {
  202. for (const auto& id : engine_.activeExecutionIds()) {
  203. response->add_execution_ids(id);
  204. }
  205. return grpc::Status::OK;
  206. }
  207. grpc::Status RunnerServiceImpl::ResumeExecution(grpc::ServerContext* context,
  208. const proto::ResumeExecutionRequest* request,
  209. proto::ExecuteWorkflowResponse* response) {
  210. nlohmann::json payload = nlohmann::json::object();
  211. if (!request->payload().empty()) {
  212. try {
  213. payload = nlohmann::json::parse(request->payload());
  214. } catch (const std::exception& e) {
  215. return grpc::Status(grpc::StatusCode::INVALID_ARGUMENT,
  216. std::string("payload is not JSON: ") + e.what());
  217. }
  218. }
  219. // The resumed half of the run needs the same event plumbing the initial
  220. // half gets in ExecuteWorkflow above, or the UI shows nothing for it and
  221. // execution.failed never reaches the handler that runs error workflows.
  222. // ExecuteWorkflow's callback stamps workflowId onto every event because
  223. // most of the engine's per-node events don't carry it themselves; this
  224. // does the same, reading workflowId from the execution record up front
  225. // since resume() (unlike execute()) is not handed a parsed Workflow by
  226. // its caller.
  227. std::string workflow_id;
  228. auto stored = storage_.get("executions", request->execution_id());
  229. if (stored.ok()) {
  230. workflow_id = stored.value().value("workflowId", "");
  231. }
  232. ExecutionCallback callback;
  233. if (event_callback_) {
  234. callback = [this, workflow_id](const std::string& event_type, const nlohmann::json& data) {
  235. nlohmann::json event_data = data;
  236. event_data["workflowId"] = workflow_id;
  237. event_callback_(event_type, event_data);
  238. };
  239. }
  240. auto result = engine_.resume(request->execution_id(), request->token(), payload, callback);
  241. if (result.failed()) {
  242. return grpc::Status(grpc::StatusCode::FAILED_PRECONDITION, result.error().message());
  243. }
  244. response->set_execution_id(result.value().execution_id);
  245. response->set_status(toProtoStatus(result.value().status));
  246. response->set_result(result.value().final_output.dump());
  247. return grpc::Status::OK;
  248. }
  249. grpc::Status RunnerServiceImpl::ListNodes(grpc::ServerContext* context,
  250. const proto::ListNodesRequest* request,
  251. proto::ListNodesResponse* response) {
  252. std::vector<NodeDefinition> nodes;
  253. if (request->category().empty()) {
  254. nodes = registry_.getAllNodes();
  255. } else {
  256. nodes = registry_.getNodesByCategory(request->category());
  257. }
  258. for (const auto& node : nodes) {
  259. auto* proto_node = response->add_nodes();
  260. proto_node->set_id(node.id);
  261. proto_node->set_name(node.name);
  262. proto_node->set_category(node.category);
  263. proto_node->set_version(node.version);
  264. proto_node->set_description(node.description);
  265. proto_node->set_icon(node.icon);
  266. proto_node->set_is_trigger(node.is_trigger);
  267. proto_node->set_config_schema(node.config_schema.dump());
  268. proto_node->set_input_schema(node.input_schema.dump());
  269. proto_node->set_output_schema(node.output_schema.dump());
  270. for (const auto& input : node.inputs) {
  271. auto* proto_input = proto_node->add_inputs();
  272. proto_input->set_name(input.name);
  273. proto_input->set_display_name(input.display_name);
  274. proto_input->set_type(input.type);
  275. proto_input->set_required(input.required);
  276. }
  277. for (const auto& output : node.outputs) {
  278. auto* proto_output = proto_node->add_outputs();
  279. proto_output->set_name(output.name);
  280. proto_output->set_display_name(output.display_name);
  281. proto_output->set_type(output.type);
  282. if (!output.color.empty()) {
  283. proto_output->set_color(output.color);
  284. }
  285. }
  286. }
  287. return grpc::Status::OK;
  288. }
  289. grpc::Status RunnerServiceImpl::ReloadNode(grpc::ServerContext* context,
  290. const proto::ReloadNodeRequest* request,
  291. proto::ReloadNodeResponse* response) {
  292. // Nodes are now synced automatically from webserver
  293. // Manual reload is no longer needed
  294. auto node = registry_.getNode(request->node_id());
  295. if (!node) {
  296. response->set_success(false);
  297. auto* error = response->mutable_error();
  298. error->set_code(404);
  299. error->set_message("Node not found: " + request->node_id());
  300. return grpc::Status::OK;
  301. }
  302. response->set_success(true);
  303. auto* proto_node = response->mutable_node();
  304. proto_node->set_id(node->id);
  305. proto_node->set_name(node->name);
  306. proto_node->set_version(node->version);
  307. return grpc::Status::OK;
  308. }
  309. grpc::Status RunnerServiceImpl::ExecuteNode(grpc::ServerContext* context,
  310. const proto::ExecuteNodeRequest* request,
  311. proto::ExecuteNodeResponse* response) {
  312. auto node_def = registry_.getNode(request->node_type());
  313. if (!node_def) {
  314. response->set_success(false);
  315. response->set_error("Node type not found: " + request->node_type());
  316. return grpc::Status::OK;
  317. }
  318. // Parse input and config
  319. nlohmann::json input, config;
  320. try {
  321. if (!request->input().empty()) {
  322. input = nlohmann::json::parse(request->input());
  323. }
  324. if (!request->config().empty()) {
  325. config = nlohmann::json::parse(request->config());
  326. }
  327. } catch (const std::exception& e) {
  328. response->set_success(false);
  329. response->set_error("Invalid JSON: " + std::string(e.what()));
  330. return grpc::Status::OK;
  331. }
  332. // Create execution context
  333. engine::ScriptContext ctx;
  334. ctx.execution_id = common::UUID::generate();
  335. ctx.node_id = request->node_type();
  336. ctx.input = input;
  337. ctx.config = config;
  338. // Execute
  339. engine::ScriptEnginePool pool(1);
  340. auto* engine = pool.acquire();
  341. auto result = engine->execute(node_def->code, ctx);
  342. pool.release(engine);
  343. response->set_success(result.success);
  344. response->set_output(result.output.dump());
  345. response->set_error(result.error);
  346. response->set_execution_time_ms(result.execution_time_ms);
  347. return grpc::Status::OK;
  348. }
  349. grpc::Status RunnerServiceImpl::GetNodeCode(grpc::ServerContext* context,
  350. const proto::GetNodeCodeRequest* request,
  351. proto::GetNodeCodeResponse* response) {
  352. auto result = registry_.getNodeCode(request->node_id());
  353. if (result.failed()) {
  354. response->set_success(false);
  355. auto* error = response->mutable_error();
  356. error->set_code(static_cast<int32_t>(result.error().code()));
  357. error->set_message(result.error().message());
  358. return grpc::Status::OK;
  359. }
  360. response->set_success(true);
  361. response->set_code(result.value());
  362. // file_path no longer applicable - nodes stored in database
  363. return grpc::Status::OK;
  364. }
  365. grpc::Status RunnerServiceImpl::SaveNodeCode(grpc::ServerContext* context,
  366. const proto::SaveNodeCodeRequest* request,
  367. proto::SaveNodeCodeResponse* response) {
  368. // Node modifications are now handled centrally by the webserver
  369. response->set_success(false);
  370. auto* error = response->mutable_error();
  371. error->set_code(501);
  372. error->set_message("Node modifications should be done through the webserver API");
  373. return grpc::Status::OK;
  374. }
  375. grpc::Status RunnerServiceImpl::CreateNode(grpc::ServerContext* context,
  376. const proto::CreateNodeRequest* request,
  377. proto::CreateNodeResponse* response) {
  378. // Node creation is now handled centrally by the webserver
  379. response->set_success(false);
  380. auto* error = response->mutable_error();
  381. error->set_code(501);
  382. error->set_message("Node creation should be done through the webserver API");
  383. return grpc::Status::OK;
  384. }
  385. grpc::Status RunnerServiceImpl::DeleteNode(grpc::ServerContext* context,
  386. const proto::DeleteNodeRequest* request,
  387. proto::DeleteNodeResponse* response) {
  388. // Node deletion is now handled centrally by the webserver
  389. response->set_success(false);
  390. auto* error = response->mutable_error();
  391. error->set_code(501);
  392. error->set_message("Node deletion should be done through the webserver API");
  393. return grpc::Status::OK;
  394. }
  395. // RunnerService implementation
  396. RunnerService::RunnerService(const RunnerServiceConfig& config)
  397. : config_(config) {
  398. // Initialize storage client
  399. storage::StorageClientConfig storage_config;
  400. storage_config.address = config_.database_address;
  401. storage_config.project = config_.database_project;
  402. storage_config.max_message_size_mb = config_.max_message_size_mb;
  403. storage_ = std::make_unique<storage::StorageClient>(storage_config);
  404. // Initialize credential client
  405. credentials::CredentialClientConfig cred_config;
  406. cred_config.address = config_.credential_service_address;
  407. credential_client_ = std::make_unique<credentials::CredentialClient>(cred_config);
  408. // Initialize node registry - now loads from webserver
  409. config_.node_registry_config.webserver_address = config_.node_sync_address;
  410. registry_ = std::make_unique<NodeRegistry>(config_.node_registry_config);
  411. // Initialize workflow engine
  412. config_.workflow_engine_config.max_concurrent_executions = config_.max_concurrent_executions;
  413. config_.workflow_engine_config.runner_id = config_.runner_id;
  414. engine_ = std::make_unique<WorkflowEngine>(*registry_, *storage_, config_.workflow_engine_config);
  415. // Set up credential auth callback for workflow engine
  416. engine_->setCredentialAuthCallback(
  417. [this](const std::string& credential_id, const std::string& workflow_id)
  418. -> common::Result<engine::CredentialAuth> {
  419. auto result = credential_client_->getHttpAuth(credential_id, workflow_id);
  420. if (result.failed()) {
  421. return result.error();
  422. }
  423. engine::CredentialAuth auth;
  424. auth.header_name = result.value().header_name;
  425. auth.header_value = result.value().header_value;
  426. return auth;
  427. });
  428. // Set up IMAP credential callback for workflow engine
  429. engine_->setImapCredentialCallback(
  430. [this](const std::string& credential_id, const std::string& workflow_id)
  431. -> common::Result<engine::ImapCredential> {
  432. auto result = credential_client_->getImapCredentials(credential_id, workflow_id);
  433. if (result.failed()) {
  434. return result.error();
  435. }
  436. engine::ImapCredential cred;
  437. cred.host = result.value().host;
  438. cred.port = result.value().port;
  439. cred.username = result.value().username;
  440. cred.password = result.value().password;
  441. cred.use_ssl = result.value().use_ssl;
  442. return cred;
  443. });
  444. // Set up SMTP credential callback for workflow engine
  445. engine_->setSmtpCredentialCallback(
  446. [this](const std::string& credential_id, const std::string& workflow_id)
  447. -> common::Result<engine::SmtpCredential> {
  448. auto result = credential_client_->getSmtpCredentials(credential_id, workflow_id);
  449. if (result.failed()) {
  450. return result.error();
  451. }
  452. engine::SmtpCredential cred;
  453. cred.host = result.value().host;
  454. cred.port = result.value().port;
  455. cred.username = result.value().username;
  456. cred.password = result.value().password;
  457. cred.security = result.value().security;
  458. cred.from_address = result.value().from_address;
  459. cred.from_name = result.value().from_name;
  460. return cred;
  461. });
  462. #ifdef MYSQL_SUPPORT
  463. // Set up MySQL credential callback for workflow engine
  464. engine_->setMysqlCredentialCallback(
  465. [this](const std::string& credential_id, const std::string& workflow_id)
  466. -> common::Result<engine::MysqlCredential> {
  467. auto result = credential_client_->getMysqlCredentials(credential_id, workflow_id);
  468. if (result.failed()) {
  469. return result.error();
  470. }
  471. engine::MysqlCredential cred;
  472. cred.host = result.value().host;
  473. cred.port = result.value().port;
  474. cred.username = result.value().username;
  475. cred.password = result.value().password;
  476. cred.database = result.value().database;
  477. cred.use_ssl = result.value().use_ssl;
  478. return cred;
  479. });
  480. // Set up MySQL query callback for workflow engine
  481. engine_->setMysqlQueryCallback(
  482. [this](const engine::MysqlQueryOptions& options, const std::string& workflow_id)
  483. -> engine::MysqlQueryResult {
  484. engine::MysqlQueryResult result;
  485. // Get MySQL credentials
  486. auto cred_result = credential_client_->getMysqlCredentials(options.credential_id, workflow_id);
  487. if (cred_result.failed()) {
  488. result.success = false;
  489. result.error = cred_result.error().message();
  490. return result;
  491. }
  492. // Build connection credentials
  493. mysql::MysqlCredentials creds;
  494. creds.host = cred_result.value().host;
  495. creds.port = cred_result.value().port;
  496. creds.username = cred_result.value().username;
  497. creds.password = cred_result.value().password;
  498. creds.database = options.database.empty() ? cred_result.value().database : options.database;
  499. creds.use_ssl = cred_result.value().use_ssl;
  500. // Create client and execute query
  501. mysql::MysqlClient client(creds);
  502. auto query_result = client.query(options.query, options.params);
  503. result.success = query_result.success;
  504. result.rows = query_result.rows;
  505. result.columns = query_result.columns;
  506. result.affected_rows = query_result.affected_rows;
  507. result.insert_id = query_result.insert_id;
  508. result.error = query_result.error;
  509. return result;
  510. });
  511. #endif
  512. #ifdef POSTGRESQL_SUPPORT
  513. // Set up PostgreSQL credential callback for workflow engine
  514. engine_->setPostgresqlCredentialCallback(
  515. [this](const std::string& credential_id, const std::string& workflow_id)
  516. -> common::Result<engine::PostgresqlCredential> {
  517. auto result = credential_client_->getPostgresqlCredentials(credential_id, workflow_id);
  518. if (result.failed()) {
  519. return result.error();
  520. }
  521. engine::PostgresqlCredential cred;
  522. cred.host = result.value().host;
  523. cred.port = result.value().port;
  524. cred.username = result.value().username;
  525. cred.password = result.value().password;
  526. cred.database = result.value().database;
  527. cred.use_ssl = result.value().use_ssl;
  528. return cred;
  529. });
  530. // Set up PostgreSQL query callback for workflow engine
  531. engine_->setPostgresqlQueryCallback(
  532. [this](const engine::PostgresqlQueryOptions& options, const std::string& workflow_id)
  533. -> engine::PostgresqlQueryResult {
  534. engine::PostgresqlQueryResult result;
  535. // Get PostgreSQL credentials
  536. auto cred_result = credential_client_->getPostgresqlCredentials(options.credential_id, workflow_id);
  537. if (cred_result.failed()) {
  538. result.success = false;
  539. result.error = cred_result.error().message();
  540. return result;
  541. }
  542. // Build connection credentials
  543. postgresql::PostgresqlCredentials creds;
  544. creds.host = cred_result.value().host;
  545. creds.port = cred_result.value().port;
  546. creds.username = cred_result.value().username;
  547. creds.password = cred_result.value().password;
  548. creds.database = options.database.empty() ? cred_result.value().database : options.database;
  549. creds.use_ssl = cred_result.value().use_ssl;
  550. // Create client and execute query
  551. postgresql::PostgresqlClient client(creds);
  552. auto query_result = client.query(options.query, options.params);
  553. result.success = query_result.success;
  554. result.rows = query_result.rows;
  555. result.columns = query_result.columns;
  556. result.affected_rows = query_result.affected_rows;
  557. result.insert_id = query_result.insert_id;
  558. result.error = query_result.error;
  559. return result;
  560. });
  561. #endif
  562. }
  563. RunnerService::~RunnerService() {
  564. stop();
  565. }
  566. RunnerServiceConfig RunnerService::loadConfig(const std::filesystem::path& path) {
  567. RunnerServiceConfig config;
  568. auto result = config::Config::fromFile(path);
  569. if (result.ok()) {
  570. auto& cfg = result.value();
  571. config.grpc_port = cfg.getOr<int>("grpc_port", 9003);
  572. config.runner_id = cfg.getOr<std::string>("runner_id", "runner-1");
  573. config.webserver_address = cfg.getOr<std::string>("webserver_address", "localhost:8080");
  574. config.node_sync_address = cfg.getOr<std::string>("node_sync_address", "localhost:9002");
  575. config.credential_service_address = cfg.getOr<std::string>("credential_service_address", "localhost:9003");
  576. config.database_address = cfg.getOr<std::string>("database_address", "localhost:9004");
  577. config.database_project = cfg.getOr<std::string>("database_project", "smartbotic-automation");
  578. config.max_message_size_mb = cfg.getOr<int>("max_message_size_mb", 64);
  579. // Node sync configuration
  580. config.node_registry_config.webserver_address = config.node_sync_address;
  581. config.node_registry_config.sync_enabled =
  582. cfg.getOr<bool>("node_sync.enabled", true);
  583. config.node_registry_config.reconnect_interval_ms =
  584. cfg.getOr<int>("node_sync.reconnect_interval_ms", 5000);
  585. config.heartbeat_interval_sec =
  586. cfg.getOr<int>("registration.heartbeat_interval_sec", 10);
  587. config.max_concurrent_executions =
  588. cfg.getOr<int>("registration.max_concurrent_executions", 10);
  589. config.workflow_engine_config.default_timeout_ms =
  590. cfg.getOr<int>("execution.default_timeout_ms", 60000);
  591. config.workflow_engine_config.script_config.max_memory_mb =
  592. cfg.getOr<int>("execution.max_memory_per_script_mb", 64);
  593. }
  594. return config;
  595. }
  596. void RunnerService::start() {
  597. if (running_) {
  598. return;
  599. }
  600. LOG_INFO("Starting runner service {}...", config_.runner_id);
  601. // Start node registry (loads nodes from webserver and subscribes to changes)
  602. registry_->start();
  603. // Create event callback to send events to webserver
  604. std::string webserver_url = "http://" + config_.webserver_address + "/api/v1/internal/execution-event";
  605. ExecutionEventCallback event_callback = [webserver_url](const std::string& event_type, const nlohmann::json& data) {
  606. LOG_DEBUG("Sending event {} to webserver", event_type);
  607. // Send event to webserver via HTTP POST (non-blocking in separate thread)
  608. std::thread([webserver_url, event_type, data]() {
  609. CURL* curl = curl_easy_init();
  610. if (!curl) {
  611. LOG_ERROR("Failed to init curl for event {}", event_type);
  612. return;
  613. }
  614. nlohmann::json payload;
  615. payload["event"] = event_type;
  616. payload["data"] = data;
  617. std::string body = payload.dump();
  618. curl_easy_setopt(curl, CURLOPT_URL, webserver_url.c_str());
  619. curl_easy_setopt(curl, CURLOPT_POST, 1L);
  620. curl_easy_setopt(curl, CURLOPT_POSTFIELDS, body.c_str());
  621. curl_easy_setopt(curl, CURLOPT_POSTFIELDSIZE, static_cast<long>(body.size()));
  622. struct curl_slist* headers = nullptr;
  623. headers = curl_slist_append(headers, "Content-Type: application/json");
  624. curl_easy_setopt(curl, CURLOPT_HTTPHEADER, headers);
  625. curl_easy_setopt(curl, CURLOPT_TIMEOUT, 5L);
  626. // Discard response body (don't write to stdout)
  627. curl_easy_setopt(curl, CURLOPT_WRITEFUNCTION, +[](char*, size_t size, size_t nmemb, void*) -> size_t {
  628. return size * nmemb;
  629. });
  630. CURLcode res = curl_easy_perform(curl);
  631. if (res != CURLE_OK) {
  632. LOG_ERROR("Failed to send event {} to webserver: {}", event_type, curl_easy_strerror(res));
  633. }
  634. curl_slist_free_all(headers);
  635. curl_easy_cleanup(curl);
  636. }).detach();
  637. };
  638. // Start gRPC server
  639. service_impl_ = std::make_unique<RunnerServiceImpl>(*engine_, *registry_, *storage_, event_callback);
  640. grpc::EnableDefaultHealthCheckService(true);
  641. grpc::ServerBuilder builder;
  642. builder.AddListeningPort("0.0.0.0:" + std::to_string(config_.grpc_port),
  643. grpc::InsecureServerCredentials());
  644. builder.RegisterService(service_impl_.get());
  645. // grpc++ defaults an unset receive limit to 4 MB, well under a webhook
  646. // body the webserver now accepts up to server.max_upload_mb (32 MB by
  647. // default) for. ExecuteWorkflow carries that body from the webserver to
  648. // this server as part of the request, so without raising this the
  649. // gRPC hop silently re-imposes a lower cap than the HTTP one already
  650. // passed. Reuse max_message_size_mb - already read from config above
  651. // and already applied to the database client - instead of adding a
  652. // second knob for the same idea.
  653. const int max_message_bytes = config_.max_message_size_mb * 1024 * 1024;
  654. builder.SetMaxReceiveMessageSize(max_message_bytes);
  655. builder.SetMaxSendMessageSize(max_message_bytes);
  656. server_ = builder.BuildAndStart();
  657. LOG_INFO("Runner gRPC server listening on port {}", config_.grpc_port);
  658. running_ = true;
  659. // Register with webserver
  660. registered_ = registerWithWebServer();
  661. // Start heartbeat, which also re-registers whenever the webserver forgets us
  662. heartbeat_thread_ = std::thread(&RunnerService::heartbeatLoop, this);
  663. LOG_INFO("Runner service {} started", config_.runner_id);
  664. }
  665. void RunnerService::stop() {
  666. if (!running_) {
  667. return;
  668. }
  669. LOG_INFO("Stopping runner service...");
  670. // Signal shutdown to background threads
  671. {
  672. std::lock_guard<std::mutex> lock(shutdown_mutex_);
  673. running_ = false;
  674. }
  675. shutdown_cv_.notify_all();
  676. // Unregister
  677. unregisterFromWebServer();
  678. // Stop heartbeat (will wake up immediately now)
  679. if (heartbeat_thread_.joinable()) {
  680. heartbeat_thread_.join();
  681. }
  682. // Stop node registry
  683. registry_->stop();
  684. // Stop gRPC server
  685. if (server_) {
  686. server_->Shutdown();
  687. }
  688. LOG_INFO("Runner service stopped");
  689. }
  690. bool RunnerService::registerWithWebServer() {
  691. // Use HTTP to register with webserver
  692. nlohmann::json body;
  693. body["id"] = config_.runner_id;
  694. body["address"] = "localhost:" + std::to_string(config_.grpc_port);
  695. nlohmann::json capabilities;
  696. std::vector<std::string> node_types;
  697. for (const auto& node : registry_->getAllNodes()) {
  698. node_types.push_back(node.id);
  699. }
  700. capabilities["nodeTypes"] = node_types;
  701. capabilities["maxMemoryPerScriptMb"] = config_.workflow_engine_config.script_config.max_memory_mb;
  702. capabilities["maxExecutionTimeoutSec"] = config_.workflow_engine_config.default_timeout_ms / 1000;
  703. body["capabilities"] = capabilities;
  704. std::string url = "http://" + config_.webserver_address + "/api/internal/runners/register";
  705. CURL* curl = curl_easy_init();
  706. if (!curl) {
  707. LOG_WARN("Failed to initialize curl for registration");
  708. return false;
  709. }
  710. std::string response_data;
  711. std::string body_str = body.dump();
  712. curl_easy_setopt(curl, CURLOPT_URL, url.c_str());
  713. curl_easy_setopt(curl, CURLOPT_POST, 1L);
  714. curl_easy_setopt(curl, CURLOPT_POSTFIELDS, body_str.c_str());
  715. curl_easy_setopt(curl, CURLOPT_TIMEOUT, 5L);
  716. struct curl_slist* headers = nullptr;
  717. headers = curl_slist_append(headers, "Content-Type: application/json");
  718. curl_easy_setopt(curl, CURLOPT_HTTPHEADER, headers);
  719. curl_easy_setopt(curl, CURLOPT_WRITEFUNCTION, +[](char* ptr, size_t size, size_t nmemb, void* userdata) -> size_t {
  720. auto* data = static_cast<std::string*>(userdata);
  721. data->append(ptr, size * nmemb);
  722. return size * nmemb;
  723. });
  724. curl_easy_setopt(curl, CURLOPT_WRITEDATA, &response_data);
  725. CURLcode res = curl_easy_perform(curl);
  726. long http_code = 0;
  727. if (res == CURLE_OK) {
  728. curl_easy_getinfo(curl, CURLINFO_RESPONSE_CODE, &http_code);
  729. }
  730. // A transport-level success is not enough: the webserver can still reject the
  731. // registration, and reporting that as success would hide a runner that is not
  732. // actually reachable for work.
  733. const bool ok = (res == CURLE_OK && http_code >= 200 && http_code < 300);
  734. if (ok) {
  735. LOG_INFO("Runner registered with webserver");
  736. } else if (res != CURLE_OK) {
  737. LOG_WARN("Failed to register with webserver: {}", curl_easy_strerror(res));
  738. } else {
  739. LOG_WARN("Webserver rejected registration: HTTP {}", http_code);
  740. }
  741. curl_slist_free_all(headers);
  742. curl_easy_cleanup(curl);
  743. return ok;
  744. }
  745. void RunnerService::heartbeatLoop() {
  746. std::string url = "http://" + config_.webserver_address + "/api/internal/runners/heartbeat";
  747. while (running_) {
  748. // Wait for shutdown signal or timeout
  749. {
  750. std::unique_lock<std::mutex> lock(shutdown_mutex_);
  751. if (shutdown_cv_.wait_for(lock,
  752. std::chrono::seconds(config_.heartbeat_interval_sec),
  753. [this] { return !running_.load(); })) {
  754. // Shutdown signaled, exit loop
  755. break;
  756. }
  757. }
  758. if (!running_) break;
  759. auto metrics = collectMetrics();
  760. nlohmann::json body;
  761. body["id"] = config_.runner_id;
  762. body["status"] = "online";
  763. body["metrics"] = {
  764. {"activeExecutions", metrics.active_executions},
  765. {"maxExecutions", metrics.max_executions},
  766. {"memoryUsedBytes", metrics.memory_used_bytes},
  767. {"memoryTotalBytes", metrics.memory_total_bytes},
  768. {"cpuPercent", metrics.cpu_percent}
  769. };
  770. CURL* curl = curl_easy_init();
  771. if (!curl) continue;
  772. std::string body_str = body.dump();
  773. std::string response_data;
  774. curl_easy_setopt(curl, CURLOPT_URL, url.c_str());
  775. curl_easy_setopt(curl, CURLOPT_POST, 1L);
  776. curl_easy_setopt(curl, CURLOPT_POSTFIELDS, body_str.c_str());
  777. curl_easy_setopt(curl, CURLOPT_TIMEOUT, 5L);
  778. struct curl_slist* headers = nullptr;
  779. headers = curl_slist_append(headers, "Content-Type: application/json");
  780. curl_easy_setopt(curl, CURLOPT_HTTPHEADER, headers);
  781. curl_easy_setopt(curl, CURLOPT_WRITEFUNCTION, +[](char* ptr, size_t size, size_t nmemb, void* userdata) -> size_t {
  782. auto* data = static_cast<std::string*>(userdata);
  783. data->append(ptr, size * nmemb);
  784. return size * nmemb;
  785. });
  786. curl_easy_setopt(curl, CURLOPT_WRITEDATA, &response_data);
  787. CURLcode res = curl_easy_perform(curl);
  788. long http_code = 0;
  789. if (res == CURLE_OK) {
  790. curl_easy_getinfo(curl, CURLINFO_RESPONSE_CODE, &http_code);
  791. }
  792. curl_slist_free_all(headers);
  793. curl_easy_cleanup(curl);
  794. if (res == CURLE_OK && http_code >= 200 && http_code < 300) {
  795. if (!registered_) {
  796. LOG_INFO("Reconnected to webserver");
  797. registered_ = true;
  798. }
  799. continue;
  800. }
  801. // The registry lives in webserver memory, so a webserver restart forgets
  802. // this runner while the runner itself stays healthy. It answers 404 for an
  803. // unknown runner; without re-registering here the runner would stay
  804. // invisible and every execution would fail with "no runners available".
  805. if (registered_) {
  806. if (res != CURLE_OK) {
  807. LOG_WARN("Heartbeat failed: {}", curl_easy_strerror(res));
  808. } else {
  809. LOG_WARN("Heartbeat rejected: HTTP {}", http_code);
  810. }
  811. registered_ = false;
  812. }
  813. if (registerWithWebServer()) {
  814. registered_ = true;
  815. }
  816. }
  817. }
  818. void RunnerService::unregisterFromWebServer() {
  819. std::string url = "http://" + config_.webserver_address + "/api/internal/runners/unregister";
  820. nlohmann::json body;
  821. body["id"] = config_.runner_id;
  822. CURL* curl = curl_easy_init();
  823. if (!curl) return;
  824. std::string body_str = body.dump();
  825. std::string response_data;
  826. curl_easy_setopt(curl, CURLOPT_URL, url.c_str());
  827. curl_easy_setopt(curl, CURLOPT_POST, 1L);
  828. curl_easy_setopt(curl, CURLOPT_POSTFIELDS, body_str.c_str());
  829. curl_easy_setopt(curl, CURLOPT_TIMEOUT, 5L);
  830. struct curl_slist* headers = nullptr;
  831. headers = curl_slist_append(headers, "Content-Type: application/json");
  832. curl_easy_setopt(curl, CURLOPT_HTTPHEADER, headers);
  833. curl_easy_setopt(curl, CURLOPT_WRITEFUNCTION, +[](char* ptr, size_t size, size_t nmemb, void* userdata) -> size_t {
  834. auto* data = static_cast<std::string*>(userdata);
  835. data->append(ptr, size * nmemb);
  836. return size * nmemb;
  837. });
  838. curl_easy_setopt(curl, CURLOPT_WRITEDATA, &response_data);
  839. curl_easy_perform(curl);
  840. curl_slist_free_all(headers);
  841. curl_easy_cleanup(curl);
  842. }
  843. RunnerMetrics RunnerService::collectMetrics() {
  844. RunnerMetrics metrics;
  845. metrics.active_executions = engine_->getActiveExecutionCount();
  846. metrics.max_executions = config_.max_concurrent_executions;
  847. // Get memory usage
  848. struct rusage usage;
  849. if (getrusage(RUSAGE_SELF, &usage) == 0) {
  850. metrics.memory_used_bytes = usage.ru_maxrss * 1024; // KB to bytes
  851. }
  852. // Get total memory from /proc/meminfo
  853. std::ifstream meminfo("/proc/meminfo");
  854. std::string line;
  855. while (std::getline(meminfo, line)) {
  856. if (line.starts_with("MemTotal:")) {
  857. std::istringstream iss(line);
  858. std::string label;
  859. int64_t kb;
  860. iss >> label >> kb;
  861. metrics.memory_total_bytes = kb * 1024;
  862. break;
  863. }
  864. }
  865. return metrics;
  866. }
  867. } // namespace smartbotic::runner