rss-reader.js 29 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642643644645646647648649650651652653654655656657658659660661662663664665666667668669670671672673674675676677678679680681682683684685686687688689690691692693694695696697698699700701702703704705706707708709710711712713714715716717718719720721722723724725726727728729730731732733734735736737738739740741742743744745746747748749750751752753754755756757758759760761762763764765766767768769770771772773774775776777778779780781782783784785786787788789790791792793794795796797798799800801802803804805806807808809810811812813814815816817818819820821822823824825826827828829830831832833834835836837838839840
  1. /**
  2. * @node rss-reader
  3. * @name RSS Reader
  4. * @category data
  5. * @version 1.2.0
  6. * @description Read and parse RSS/Atom feeds with optional filtering by keyword, date, or count. Supports stateful new item detection.
  7. * @icon rss
  8. */
  9. const configSchema = {
  10. type: 'object',
  11. properties: {
  12. url: {
  13. type: 'string',
  14. title: 'Feed URL',
  15. description: 'URL of the RSS or Atom feed'
  16. },
  17. filterKeyword: {
  18. type: 'string',
  19. title: 'Keyword Filter',
  20. description: 'Filter items by keyword (matches title and description)'
  21. },
  22. filterNewerThan: {
  23. type: 'string',
  24. title: 'Newer Than',
  25. description: 'Only include items newer than this date (ISO 8601 format, e.g., 2024-01-15T00:00:00Z)'
  26. },
  27. maxItems: {
  28. type: 'number',
  29. title: 'Max Items',
  30. description: 'Maximum number of items to return (0 = unlimited)',
  31. default: 0
  32. },
  33. userAgent: {
  34. type: 'string',
  35. title: 'User-Agent',
  36. description: 'Sent with the request. Several feeds refuse an unnamed client outright - Reddit answers 403 - and some rate-limit a browser string harder than a named one. Identify yourself here',
  37. default: 'smartbotic-rss/1.0 (+https://smartbotics.ai)'
  38. },
  39. timeout: {
  40. type: 'number',
  41. title: 'Timeout (ms)',
  42. default: 30000
  43. },
  44. detectNewItems: {
  45. type: 'boolean',
  46. title: 'Detect New Items',
  47. description: 'When enabled, only output items that have not been seen in previous executions',
  48. default: false
  49. },
  50. stateCollection: {
  51. type: 'string',
  52. title: 'State Collection',
  53. description: 'Select a read-write collection to store seen item identifiers',
  54. dynamicOptions: {
  55. source: 'storage.collections',
  56. labelField: 'name',
  57. valueField: 'name',
  58. filter: { access: 'read-write' }
  59. },
  60. showWhen: { field: 'detectNewItems', value: true }
  61. }
  62. },
  63. required: ['url']
  64. };
  65. const inputSchema = {
  66. type: 'object',
  67. properties: {
  68. url: {
  69. type: 'string',
  70. description: 'Override feed URL from input'
  71. },
  72. filterKeyword: {
  73. type: 'string',
  74. description: 'Override keyword filter from input'
  75. },
  76. filterNewerThan: {
  77. type: 'string',
  78. description: 'Override date filter from input'
  79. },
  80. maxItems: {
  81. type: 'number',
  82. description: 'Override max items from input'
  83. }
  84. }
  85. };
  86. const outputSchema = {
  87. type: 'object',
  88. properties: {
  89. feedTitle: { type: 'string', description: 'Title of the feed' },
  90. feedLink: { type: 'string', description: 'Link to the feed homepage' },
  91. feedDescription: { type: 'string', description: 'Description of the feed' },
  92. feedLanguage: { type: 'string', description: 'Language of the feed' },
  93. feedPubDate: { type: 'string', description: 'Publication date of the feed' },
  94. feedGenerator: { type: 'string', description: 'Generator of the feed' },
  95. feedImage: {
  96. type: 'object',
  97. description: 'Feed image/logo',
  98. properties: {
  99. url: { type: 'string' },
  100. title: { type: 'string' },
  101. link: { type: 'string' }
  102. }
  103. },
  104. itemCount: { type: 'number', description: 'Number of items returned' },
  105. newItemCount: { type: 'number', description: 'Number of new items (when detectNewItems is enabled)' },
  106. totalFeedItemCount: { type: 'number', description: 'Total items in feed before filtering (when detectNewItems is enabled)' },
  107. items: {
  108. type: 'array',
  109. description: 'Array of feed items',
  110. items: {
  111. type: 'object',
  112. properties: {
  113. title: { type: 'string', description: 'Item title' },
  114. link: { type: 'string', description: 'Item link/URL' },
  115. description: { type: 'string', description: 'Item summary/description' },
  116. content: { type: 'string', description: 'Full content (content:encoded)' },
  117. pubDate: { type: 'string', description: 'Publication date' },
  118. author: { type: 'string', description: 'Author name' },
  119. categories: { type: 'array', items: { type: 'string' }, description: 'Categories/tags' },
  120. guid: { type: 'string', description: 'Unique identifier' },
  121. comments: { type: 'string', description: 'Comments URL' },
  122. source: { type: 'string', description: 'Source attribution' },
  123. enclosure: {
  124. type: 'object',
  125. description: 'Media enclosure (attachment)',
  126. properties: {
  127. url: { type: 'string', description: 'Media URL' },
  128. type: { type: 'string', description: 'MIME type' },
  129. length: { type: 'number', description: 'File size in bytes' }
  130. }
  131. },
  132. media: {
  133. type: 'object',
  134. description: 'Media content (media:content)',
  135. properties: {
  136. url: { type: 'string' },
  137. type: { type: 'string' },
  138. width: { type: 'number' },
  139. height: { type: 'number' }
  140. }
  141. },
  142. thumbnail: {
  143. type: 'object',
  144. description: 'Thumbnail image (media:thumbnail)',
  145. properties: {
  146. url: { type: 'string' },
  147. width: { type: 'number' },
  148. height: { type: 'number' }
  149. }
  150. }
  151. }
  152. }
  153. }
  154. }
  155. };
  156. /**
  157. * Simple XML parser utilities for RSS/Atom feeds
  158. * Handles basic XML structure without external dependencies
  159. */
  160. function xmlGetTagContent(xml, tagName) {
  161. var pattern = new RegExp('<' + tagName + '[^>]*>([\\s\\S]*?)<\\/' + tagName + '>', 'i');
  162. var match = xml.match(pattern);
  163. if (match) {
  164. return match[1].trim();
  165. }
  166. return null;
  167. }
  168. function xmlGetAllTagContents(xml, tagName) {
  169. var results = [];
  170. var regex = new RegExp('<' + tagName + '[^>]*>([\\s\\S]*?)<\\/' + tagName + '>', 'gi');
  171. var match;
  172. while ((match = regex.exec(xml)) !== null) {
  173. results.push(match[1].trim());
  174. }
  175. return results;
  176. }
  177. function xmlGetTagAttributes(xml, tagName) {
  178. // Match self-closing tag or opening tag
  179. var regex = new RegExp('<' + tagName + '([^>]*?)\\/?>', 'i');
  180. var match = xml.match(regex);
  181. if (!match) return null;
  182. var attrString = match[1];
  183. var attrs = {};
  184. // Parse attributes: name="value" or name='value'
  185. var attrRegex = /(\w+)=["']([^"']*)["']/g;
  186. var attrMatch;
  187. while ((attrMatch = attrRegex.exec(attrString)) !== null) {
  188. attrs[attrMatch[1]] = attrMatch[2];
  189. }
  190. return attrs;
  191. }
  192. function xmlDecodeHtmlEntities(text) {
  193. if (!text) return '';
  194. return text
  195. .replace(/&lt;/g, '<')
  196. .replace(/&gt;/g, '>')
  197. .replace(/&amp;/g, '&')
  198. .replace(/&quot;/g, '"')
  199. .replace(/&apos;/g, "'")
  200. .replace(/&#(\d+);/g, function(_, dec) { return String.fromCharCode(dec); })
  201. .replace(/&#x([0-9a-f]+);/gi, function(_, hex) { return String.fromCharCode(parseInt(hex, 16)); });
  202. }
  203. function xmlStripCdata(text) {
  204. if (!text) return '';
  205. return text.replace(/<!\[CDATA\[([\s\S]*?)\]\]>/g, '$1');
  206. }
  207. function xmlCleanText(text) {
  208. if (!text) return '';
  209. return xmlDecodeHtmlEntities(xmlStripCdata(text)).trim();
  210. }
  211. /**
  212. * Parse RSS 2.0 feed
  213. */
  214. function parseRss(xml) {
  215. var channel = xmlGetTagContent(xml, 'channel');
  216. if (!channel) {
  217. throw new Error('Invalid RSS feed: no channel element found');
  218. }
  219. var feedTitle = xmlGetTagContent(channel, 'title');
  220. var feedLink = xmlGetTagContent(channel, 'link');
  221. var feedDesc = xmlGetTagContent(channel, 'description');
  222. var feedLanguage = xmlGetTagContent(channel, 'language');
  223. var feedPubDate = xmlGetTagContent(channel, 'pubDate');
  224. var feedLastBuildDate = xmlGetTagContent(channel, 'lastBuildDate');
  225. var feedGenerator = xmlGetTagContent(channel, 'generator');
  226. var feedImage = xmlGetTagContent(channel, 'image');
  227. var feed = {
  228. title: xmlCleanText(feedTitle) || '',
  229. link: xmlCleanText(feedLink) || '',
  230. description: xmlCleanText(feedDesc) || '',
  231. language: xmlCleanText(feedLanguage) || '',
  232. pubDate: xmlCleanText(feedPubDate) || '',
  233. lastBuildDate: xmlCleanText(feedLastBuildDate) || '',
  234. generator: xmlCleanText(feedGenerator) || '',
  235. items: []
  236. };
  237. // Parse feed image if present
  238. if (feedImage) {
  239. feed.image = {
  240. url: xmlCleanText(xmlGetTagContent(feedImage, 'url')) || '',
  241. title: xmlCleanText(xmlGetTagContent(feedImage, 'title')) || '',
  242. link: xmlCleanText(xmlGetTagContent(feedImage, 'link')) || ''
  243. };
  244. }
  245. var itemRegex = /<item[^>]*>([\s\S]*?)<\/item>/gi;
  246. var match;
  247. while ((match = itemRegex.exec(channel)) !== null) {
  248. var itemXml = match[1];
  249. var catArray = xmlGetAllTagContents(itemXml, 'category');
  250. var categories = [];
  251. for (var ci = 0; ci < catArray.length; ci++) {
  252. categories.push(xmlCleanText(catArray[ci]));
  253. }
  254. var item = {
  255. title: xmlCleanText(xmlGetTagContent(itemXml, 'title')) || '',
  256. link: xmlCleanText(xmlGetTagContent(itemXml, 'link')) || '',
  257. description: xmlCleanText(xmlGetTagContent(itemXml, 'description')) || '',
  258. content: xmlCleanText(xmlGetTagContent(itemXml, 'content:encoded')) || '',
  259. pubDate: xmlCleanText(xmlGetTagContent(itemXml, 'pubDate')) || '',
  260. author: xmlCleanText(xmlGetTagContent(itemXml, 'author') ||
  261. xmlGetTagContent(itemXml, 'dc:creator')) || '',
  262. categories: categories,
  263. guid: xmlCleanText(xmlGetTagContent(itemXml, 'guid')) || '',
  264. comments: xmlCleanText(xmlGetTagContent(itemXml, 'comments')) || '',
  265. source: xmlCleanText(xmlGetTagContent(itemXml, 'source')) || ''
  266. };
  267. // Parse enclosure (media attachment)
  268. var enclosureAttrs = xmlGetTagAttributes(itemXml, 'enclosure');
  269. if (enclosureAttrs) {
  270. item.enclosure = {
  271. url: enclosureAttrs.url || '',
  272. type: enclosureAttrs.type || '',
  273. length: enclosureAttrs.length ? parseInt(enclosureAttrs.length, 10) : 0
  274. };
  275. }
  276. // Parse media:content (common in media RSS)
  277. var mediaAttrs = xmlGetTagAttributes(itemXml, 'media:content');
  278. if (mediaAttrs) {
  279. item.media = {
  280. url: mediaAttrs.url || '',
  281. type: mediaAttrs.type || mediaAttrs.medium || '',
  282. width: mediaAttrs.width ? parseInt(mediaAttrs.width, 10) : 0,
  283. height: mediaAttrs.height ? parseInt(mediaAttrs.height, 10) : 0
  284. };
  285. }
  286. // Parse media:thumbnail
  287. var thumbAttrs = xmlGetTagAttributes(itemXml, 'media:thumbnail');
  288. if (thumbAttrs) {
  289. item.thumbnail = {
  290. url: thumbAttrs.url || '',
  291. width: thumbAttrs.width ? parseInt(thumbAttrs.width, 10) : 0,
  292. height: thumbAttrs.height ? parseInt(thumbAttrs.height, 10) : 0
  293. };
  294. }
  295. feed.items.push(item);
  296. }
  297. return feed;
  298. }
  299. /**
  300. * Parse Atom feed
  301. */
  302. function parseAtom(xml) {
  303. var feedContent = xmlGetTagContent(xml, 'feed');
  304. if (!feedContent) {
  305. throw new Error('Invalid Atom feed: no feed element found');
  306. }
  307. function getAtomLink(content, rel) {
  308. var linkRegex = /<link([^>]*)>/gi;
  309. var match;
  310. while ((match = linkRegex.exec(content)) !== null) {
  311. var attrs = match[1];
  312. var relMatch = attrs.match(/rel=["']([^"']*)["']/);
  313. var hrefMatch = attrs.match(/href=["']([^"']*)["']/);
  314. if (hrefMatch) {
  315. var linkRel = relMatch ? relMatch[1] : 'alternate';
  316. if (linkRel === rel || (!rel && linkRel === 'alternate')) {
  317. return hrefMatch[1];
  318. }
  319. }
  320. }
  321. return '';
  322. }
  323. function getAtomLinkByType(content, type) {
  324. var linkRegex = /<link([^>]*)>/gi;
  325. var match;
  326. while ((match = linkRegex.exec(content)) !== null) {
  327. var attrs = match[1];
  328. var typeMatch = attrs.match(/type=["']([^"']*)["']/);
  329. var hrefMatch = attrs.match(/href=["']([^"']*)["']/);
  330. if (hrefMatch && typeMatch && typeMatch[1].indexOf(type) !== -1) {
  331. return {
  332. url: hrefMatch[1],
  333. type: typeMatch[1]
  334. };
  335. }
  336. }
  337. return null;
  338. }
  339. var feed = {
  340. title: xmlCleanText(xmlGetTagContent(feedContent, 'title')) || '',
  341. link: getAtomLink(feedContent, 'alternate') || getAtomLink(feedContent, null) || '',
  342. description: xmlCleanText(xmlGetTagContent(feedContent, 'subtitle')) || '',
  343. language: '',
  344. pubDate: xmlCleanText(xmlGetTagContent(feedContent, 'updated')) || '',
  345. lastBuildDate: '',
  346. generator: xmlCleanText(xmlGetTagContent(feedContent, 'generator')) || '',
  347. items: []
  348. };
  349. // Parse feed icon/logo
  350. var feedIcon = xmlGetTagContent(feedContent, 'icon');
  351. var feedLogo = xmlGetTagContent(feedContent, 'logo');
  352. if (feedIcon || feedLogo) {
  353. feed.image = {
  354. url: xmlCleanText(feedLogo || feedIcon) || '',
  355. title: feed.title,
  356. link: feed.link
  357. };
  358. }
  359. var entryRegex = /<entry[^>]*>([\s\S]*?)<\/entry>/gi;
  360. var match;
  361. while ((match = entryRegex.exec(feedContent)) !== null) {
  362. var entryXml = match[1];
  363. var categories = [];
  364. var catRegex = /<category([^>]*)>/gi;
  365. var catMatch;
  366. while ((catMatch = catRegex.exec(entryXml)) !== null) {
  367. var termMatch = catMatch[1].match(/term=["']([^"']*)["']/);
  368. if (termMatch) {
  369. categories.push(xmlCleanText(termMatch[1]));
  370. }
  371. }
  372. var author = '';
  373. var authorContent = xmlGetTagContent(entryXml, 'author');
  374. if (authorContent) {
  375. author = xmlCleanText(xmlGetTagContent(authorContent, 'name')) || '';
  376. }
  377. var pubDate = xmlCleanText(
  378. xmlGetTagContent(entryXml, 'published') ||
  379. xmlGetTagContent(entryXml, 'updated')
  380. ) || '';
  381. var summary = xmlCleanText(xmlGetTagContent(entryXml, 'summary')) || '';
  382. var content = xmlCleanText(xmlGetTagContent(entryXml, 'content')) || '';
  383. var item = {
  384. title: xmlCleanText(xmlGetTagContent(entryXml, 'title')) || '',
  385. link: getAtomLink(entryXml, 'alternate') || getAtomLink(entryXml, null) || '',
  386. description: summary || content,
  387. content: content,
  388. pubDate: pubDate,
  389. author: author,
  390. categories: categories,
  391. guid: xmlCleanText(xmlGetTagContent(entryXml, 'id')) || '',
  392. comments: '',
  393. source: ''
  394. };
  395. // Check for enclosure link
  396. var enclosureLink = getAtomLinkByType(entryXml, 'image');
  397. if (!enclosureLink) {
  398. enclosureLink = getAtomLinkByType(entryXml, 'audio');
  399. }
  400. if (!enclosureLink) {
  401. enclosureLink = getAtomLinkByType(entryXml, 'video');
  402. }
  403. if (enclosureLink) {
  404. item.enclosure = {
  405. url: enclosureLink.url,
  406. type: enclosureLink.type,
  407. length: 0
  408. };
  409. }
  410. // Parse media:content
  411. var mediaAttrs = xmlGetTagAttributes(entryXml, 'media:content');
  412. if (mediaAttrs) {
  413. item.media = {
  414. url: mediaAttrs.url || '',
  415. type: mediaAttrs.type || mediaAttrs.medium || '',
  416. width: mediaAttrs.width ? parseInt(mediaAttrs.width, 10) : 0,
  417. height: mediaAttrs.height ? parseInt(mediaAttrs.height, 10) : 0
  418. };
  419. }
  420. // Parse media:thumbnail
  421. var thumbAttrs = xmlGetTagAttributes(entryXml, 'media:thumbnail');
  422. if (thumbAttrs) {
  423. item.thumbnail = {
  424. url: thumbAttrs.url || '',
  425. width: thumbAttrs.width ? parseInt(thumbAttrs.width, 10) : 0,
  426. height: thumbAttrs.height ? parseInt(thumbAttrs.height, 10) : 0
  427. };
  428. }
  429. feed.items.push(item);
  430. }
  431. return feed;
  432. }
  433. /**
  434. * Parse date string to timestamp
  435. */
  436. function parseDate(dateStr) {
  437. if (!dateStr) return 0;
  438. var timestamp = Date.parse(dateStr);
  439. if (!isNaN(timestamp)) {
  440. return timestamp;
  441. }
  442. var rfc822Regex = /(\d{1,2})\s+(\w{3})\s+(\d{4})\s+(\d{2}):(\d{2}):(\d{2})/;
  443. var match = dateStr.match(rfc822Regex);
  444. if (match) {
  445. var months = {
  446. Jan: 0, Feb: 1, Mar: 2, Apr: 3, May: 4, Jun: 5,
  447. Jul: 6, Aug: 7, Sep: 8, Oct: 9, Nov: 10, Dec: 11
  448. };
  449. var day = parseInt(match[1], 10);
  450. var month = months[match[2]];
  451. var year = parseInt(match[3], 10);
  452. var hour = parseInt(match[4], 10);
  453. var minute = parseInt(match[5], 10);
  454. var second = parseInt(match[6], 10);
  455. if (month !== undefined) {
  456. return new Date(year, month, day, hour, minute, second).getTime();
  457. }
  458. }
  459. return 0;
  460. }
  461. /**
  462. * Filter items by keyword
  463. */
  464. function filterByKeyword(items, keyword) {
  465. if (!keyword) return items;
  466. var lowerKeyword = keyword.toLowerCase();
  467. var result = [];
  468. for (var i = 0; i < items.length; i++) {
  469. var item = items[i];
  470. var title = (item.title || '').toLowerCase();
  471. var description = (item.description || '').toLowerCase();
  472. if (title.indexOf(lowerKeyword) !== -1 || description.indexOf(lowerKeyword) !== -1) {
  473. result.push(item);
  474. }
  475. }
  476. return result;
  477. }
  478. /**
  479. * Filter items by date
  480. */
  481. function filterByDate(items, newerThanStr) {
  482. if (!newerThanStr) return items;
  483. var threshold = parseDate(newerThanStr);
  484. if (threshold === 0) {
  485. smartbotic.log.warn('Could not parse date filter: ' + newerThanStr);
  486. return items;
  487. }
  488. var result = [];
  489. for (var i = 0; i < items.length; i++) {
  490. var item = items[i];
  491. var itemDate = parseDate(item.pubDate);
  492. if (itemDate > threshold) {
  493. result.push(item);
  494. }
  495. }
  496. return result;
  497. }
  498. /**
  499. * Limit number of items
  500. */
  501. function limitItems(items, maxItems) {
  502. if (!maxItems || maxItems <= 0) return items;
  503. return items.slice(0, maxItems);
  504. }
  505. /**
  506. * Get unique identifier for an RSS/Atom item
  507. * Prefers GUID, falls back to link
  508. */
  509. function getItemIdentifier(item) {
  510. return item.guid || item.link || '';
  511. }
  512. /**
  513. * Generate a stable document ID from feed URL for state storage
  514. */
  515. function generateStateDocId(feedUrl) {
  516. // Create a simple hash from the feed URL for a stable doc ID
  517. var hash = 0;
  518. for (var i = 0; i < feedUrl.length; i++) {
  519. var char = feedUrl.charCodeAt(i);
  520. hash = ((hash << 5) - hash) + char;
  521. hash = hash & hash; // Convert to 32-bit integer
  522. }
  523. return 'rss-state-' + Math.abs(hash).toString(36);
  524. }
  525. /**
  526. * Load seen item identifiers from storage
  527. */
  528. function loadSeenItems(collection, feedUrl) {
  529. var docId = generateStateDocId(feedUrl);
  530. try {
  531. var result = smartbotic.storage.get(collection, docId);
  532. if (result.found && result.document && result.document.seenIds) {
  533. // Use plain object as a set (for QuickJS compatibility)
  534. var seenObj = {};
  535. var storedIds = result.document.seenIds;
  536. // Handle both array format and object format (in case of legacy data)
  537. if (Array.isArray(storedIds)) {
  538. smartbotic.log.debug('Loaded ' + storedIds.length + ' seen item IDs from storage');
  539. for (var i = 0; i < storedIds.length; i++) {
  540. seenObj[storedIds[i]] = true;
  541. }
  542. } else if (typeof storedIds === 'object' && storedIds !== null) {
  543. // Object format - keys are the IDs
  544. var keys = Object.keys(storedIds);
  545. smartbotic.log.debug('Loaded ' + keys.length + ' seen item IDs from storage (object format)');
  546. for (var j = 0; j < keys.length; j++) {
  547. seenObj[keys[j]] = true;
  548. }
  549. }
  550. return {
  551. seenIds: seenObj,
  552. docId: docId,
  553. exists: true,
  554. version: result.document._version || 1
  555. };
  556. }
  557. } catch (error) {
  558. smartbotic.log.warn('Error loading seen items from storage: ' + error.message);
  559. }
  560. return {
  561. seenIds: {},
  562. docId: docId,
  563. exists: false,
  564. version: 0
  565. };
  566. }
  567. /**
  568. * Save seen item identifiers to storage
  569. */
  570. function saveSeenItems(collection, docId, seenIds, feedUrl, exists, version) {
  571. // Convert object keys to array
  572. var seenArray = Object.keys(seenIds);
  573. var documentData = {
  574. feedUrl: feedUrl,
  575. seenIds: seenArray,
  576. lastUpdated: new Date().toISOString(),
  577. itemCount: seenArray.length
  578. };
  579. try {
  580. if (exists) {
  581. // Update existing document
  582. var updateResult = smartbotic.storage.update(collection, docId, documentData, version, false);
  583. if (!updateResult.success) {
  584. smartbotic.log.error('Failed to update seen items: ' + updateResult.error);
  585. return false;
  586. }
  587. smartbotic.log.debug('Updated ' + seenArray.length + ' seen item IDs in storage');
  588. } else {
  589. // Insert new document
  590. var insertResult = smartbotic.storage.insert(collection, documentData, docId, 0);
  591. if (!insertResult.success) {
  592. // If insert fails because document exists, try update instead
  593. if (insertResult.error && insertResult.error.indexOf('already exists') !== -1) {
  594. smartbotic.log.debug('Document exists, trying update instead');
  595. var retryResult = smartbotic.storage.update(collection, docId, documentData, 0, false);
  596. if (!retryResult.success) {
  597. smartbotic.log.error('Failed to update seen items on retry: ' + retryResult.error);
  598. return false;
  599. }
  600. smartbotic.log.debug('Updated ' + seenArray.length + ' seen item IDs in storage (retry)');
  601. } else {
  602. smartbotic.log.error('Failed to save seen items: ' + insertResult.error);
  603. return false;
  604. }
  605. } else {
  606. smartbotic.log.debug('Saved ' + seenArray.length + ' seen item IDs to storage');
  607. }
  608. }
  609. return true;
  610. } catch (error) {
  611. smartbotic.log.error('Error saving seen items to storage: ' + error.message);
  612. return false;
  613. }
  614. }
  615. /**
  616. * Filter items to only include new (unseen) items
  617. */
  618. function filterNewItems(items, seenIds) {
  619. var newItems = [];
  620. for (var i = 0; i < items.length; i++) {
  621. var id = getItemIdentifier(items[i]);
  622. if (id && !seenIds[id]) {
  623. newItems.push(items[i]);
  624. }
  625. }
  626. return newItems;
  627. }
  628. async function execute(config, input) {
  629. var url = input.url || config.url;
  630. var filterKeyword = input.filterKeyword || config.filterKeyword;
  631. var filterNewerThan = input.filterNewerThan || config.filterNewerThan;
  632. var maxItems = input.maxItems !== undefined ? input.maxItems : config.maxItems;
  633. var timeout = config.timeout || 30000;
  634. var detectNewItems = config.detectNewItems || false;
  635. var stateCollection = config.stateCollection;
  636. if (!url) {
  637. throw new Error('Feed URL is required');
  638. }
  639. // Validate stateful mode configuration
  640. if (detectNewItems && !stateCollection) {
  641. throw new Error('State collection is required when "Detect New Items" is enabled. Please select a read-write collection in the workflow storage settings.');
  642. }
  643. // No User-Agent at all is what several feeds reject, Reddit among them, so
  644. // there is always one.
  645. var userAgent = String(config.userAgent || '').trim() ||
  646. 'smartbotic-rss/1.0 (+https://smartbotics.ai)';
  647. smartbotic.log.info('Fetching RSS/Atom feed from ' + url);
  648. var response = smartbotic.http.request({
  649. method: 'GET',
  650. url: url,
  651. timeout: timeout,
  652. headers: {
  653. 'Accept': 'application/rss+xml, application/atom+xml, application/xml, text/xml, */*',
  654. 'User-Agent': userAgent
  655. }
  656. });
  657. if (response.status < 200 || response.status >= 300) {
  658. var hint = '';
  659. if (response.status === 403) {
  660. hint = '. The feed refused this client - most feeds that do this want a descriptive ' +
  661. 'User-Agent, which is the User-Agent setting on this node';
  662. } else if (response.status === 429) {
  663. hint = '. The feed is rate limiting this address - poll it less often, or from ' +
  664. 'somewhere else. Reddit throttles hard and counts every request, including ' +
  665. 'failed ones';
  666. } else if (response.status === 404) {
  667. hint = '. Check the address: ' + url;
  668. }
  669. throw new Error('Failed to fetch feed: HTTP ' + response.status + hint);
  670. }
  671. var xml = typeof response.data === 'string' ? response.data : JSON.stringify(response.data);
  672. if (!xml || xml.trim().length === 0) {
  673. throw new Error('Empty response from feed URL');
  674. }
  675. var feed;
  676. if (xml.indexOf('<feed') !== -1 && xml.indexOf('xmlns') !== -1 && xml.indexOf('atom') !== -1) {
  677. smartbotic.log.debug('Detected Atom feed format');
  678. feed = parseAtom(xml);
  679. } else if (xml.indexOf('<rss') !== -1 || xml.indexOf('<channel') !== -1) {
  680. smartbotic.log.debug('Detected RSS feed format');
  681. feed = parseRss(xml);
  682. } else {
  683. throw new Error('Unknown feed format: neither RSS nor Atom detected');
  684. }
  685. var items = feed.items;
  686. var totalFeedItemCount = items.length;
  687. // Apply standard filters first (keyword, date)
  688. if (filterKeyword) {
  689. smartbotic.log.debug('Filtering by keyword: ' + filterKeyword);
  690. items = filterByKeyword(items, filterKeyword);
  691. }
  692. if (filterNewerThan) {
  693. smartbotic.log.debug('Filtering items newer than: ' + filterNewerThan);
  694. items = filterByDate(items, filterNewerThan);
  695. }
  696. // Stateful new item detection
  697. var newItemCount = items.length;
  698. var stateInfo = null;
  699. if (detectNewItems) {
  700. smartbotic.log.info('Detecting new items using collection: ' + stateCollection);
  701. // Load previously seen items
  702. stateInfo = loadSeenItems(stateCollection, url);
  703. // Filter to only new items
  704. var newItems = filterNewItems(items, stateInfo.seenIds);
  705. newItemCount = newItems.length;
  706. smartbotic.log.info('Found ' + newItemCount + ' new items out of ' + items.length + ' filtered items');
  707. // Update seen IDs with all current items (both new and previously seen)
  708. // This ensures we track all items from the current feed
  709. for (var i = 0; i < items.length; i++) {
  710. var id = getItemIdentifier(items[i]);
  711. if (id) {
  712. stateInfo.seenIds[id] = true;
  713. }
  714. }
  715. // Use only new items for output
  716. items = newItems;
  717. }
  718. // Apply max items limit last
  719. if (maxItems && maxItems > 0) {
  720. smartbotic.log.debug('Limiting to ' + maxItems + ' items');
  721. items = limitItems(items, maxItems);
  722. }
  723. smartbotic.log.info('Returning ' + items.length + ' items from feed');
  724. // Persist seen items state after successful processing
  725. if (detectNewItems && stateInfo) {
  726. var saved = saveSeenItems(
  727. stateCollection,
  728. stateInfo.docId,
  729. stateInfo.seenIds,
  730. url,
  731. stateInfo.exists,
  732. stateInfo.version
  733. );
  734. if (!saved) {
  735. smartbotic.log.warn('Failed to persist seen items state, but continuing with output');
  736. }
  737. }
  738. var result = {
  739. feedTitle: feed.title,
  740. feedLink: feed.link,
  741. feedDescription: feed.description,
  742. itemCount: items.length,
  743. items: items
  744. };
  745. // Add stateful mode info if enabled
  746. if (detectNewItems) {
  747. result.newItemCount = newItemCount;
  748. result.totalFeedItemCount = totalFeedItemCount;
  749. }
  750. return result;
  751. }
  752. module.exports = { configSchema, inputSchema, outputSchema, execute };