imap-extract-attachments.js 20 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525
  1. /**
  2. * @node imap-extract-attachments
  3. * @name IMAP Extract Attachments
  4. * @category email
  5. * @version 1.4.0
  6. * @description Extract attachments from email as binary files. Compatible with OCR Upload and other binary-accepting nodes.
  7. * @icon paperclip
  8. */
  9. const configSchema = {
  10. type: 'object',
  11. properties: {
  12. emailSource: {
  13. type: 'string',
  14. title: 'Email Source',
  15. description: 'Path to email data containing the raw body (e.g., {{item.body}} for loop iterations, or {{data.body}} for direct input)'
  16. },
  17. deduplicateByHash: {
  18. type: 'boolean',
  19. title: 'Deduplicate by Content',
  20. description: 'Skip storing attachments that already exist (based on content hash). Prevents duplicates when processing the same email multiple times.',
  21. default: true
  22. },
  23. filterMimeTypes: {
  24. type: 'string',
  25. title: 'Filter by MIME Types',
  26. description: 'Comma-separated list of MIME types to include (e.g. application/pdf, image/*). Leave empty for all attachments.'
  27. },
  28. filterExtensions: {
  29. type: 'string',
  30. title: 'Filter by Extensions',
  31. description: 'Comma-separated list of file extensions to include (e.g. pdf, jpg, png). Leave empty for all.'
  32. },
  33. maxSize: {
  34. type: 'integer',
  35. title: 'Max Size (bytes)',
  36. description: 'Maximum attachment size in bytes (0 = no limit)',
  37. default: 0
  38. },
  39. storeInDatabase: {
  40. type: 'boolean',
  41. title: 'Store in Database',
  42. description: 'Store extracted attachments in database for later retrieval',
  43. default: false
  44. },
  45. storageCollection: {
  46. type: 'string',
  47. title: 'Storage Collection',
  48. description: 'Select collection to store attachments',
  49. dynamicOptions: {
  50. source: 'storage.collections',
  51. labelField: 'name',
  52. valueField: 'name',
  53. filter: { access: 'read-write' }
  54. }
  55. },
  56. storageTtlHours: {
  57. type: 'number',
  58. title: 'TTL (hours)',
  59. default: 24,
  60. description: 'Auto-delete stored files after this many hours. Default is 24 hours; set to 0 to keep them forever.'
  61. },
  62. binaryStoragePath: {
  63. type: 'string',
  64. title: 'Binary Storage Path',
  65. default: './data/attachments',
  66. description: 'Base directory for storing binary files on disk'
  67. }
  68. }
  69. };
  70. const inputSchema = {
  71. type: 'object',
  72. properties: {
  73. raw: {
  74. type: 'string',
  75. description: 'Raw email content (from IMAP Fetch or IMAP Trigger with fetchContent enabled)'
  76. },
  77. email: {
  78. type: 'object',
  79. description: 'Email object containing raw property'
  80. }
  81. }
  82. };
  83. const outputSchema = {
  84. type: 'object',
  85. properties: {
  86. attachments: {
  87. type: 'array',
  88. description: 'Array of extracted attachments as binary file objects',
  89. items: {
  90. type: 'object',
  91. properties: {
  92. type: { type: 'string', const: 'binary' },
  93. data: { type: 'string', description: 'Base64-encoded content' },
  94. mimeType: { type: 'string' },
  95. filename: { type: 'string' },
  96. size: { type: 'number' },
  97. contentId: { type: 'string', description: 'Content-ID for inline attachments' },
  98. storage: {
  99. type: 'object',
  100. description: 'Storage reference (if storeInDatabase enabled)',
  101. properties: {
  102. collection: { type: 'string' },
  103. id: { type: 'string' }
  104. }
  105. }
  106. }
  107. }
  108. },
  109. count: { type: 'integer', description: 'Number of attachments extracted' },
  110. totalSize: { type: 'integer', description: 'Total size of all attachments in bytes' }
  111. }
  112. };
  113. const outputs = [
  114. { name: 'main', displayName: 'Attachments', type: 'object', color: '#f59e0b' }
  115. ];
  116. /**
  117. * Parse MIME multipart content and extract attachments
  118. */
  119. // A header value may be folded across lines: RFC 5322 continues one wherever a
  120. // line begins with whitespace. Every regex below reads a value with a class
  121. // that stops at \r or \n, so a folded value used to be captured only as far as
  122. // the first line - which for a long filename means losing the end of it,
  123. // including the extension.
  124. function unfoldHeaders(text) {
  125. return String(text || '').replace(/\r?\n[ \t]+/g, ' ');
  126. }
  127. // A filename with anything outside ASCII arrives as an RFC 2047 encoded word:
  128. // =?UTF-8?Q?S=C3=A1rk=C3=A1nyok_p=C3=A1rz=C3=A1si...?=
  129. // Undecoded, that string does not end in ".pdf", so an extension filter drops
  130. // the attachment and the workflow proceeds as though the mail had none. Any
  131. // attachment with an accented name has always been lost this way.
  132. function decodeEncodedWords(text) {
  133. var value = String(text || '');
  134. if (value.indexOf('=?') === -1) {
  135. return value;
  136. }
  137. // Adjacent encoded words are joined without the whitespace between them,
  138. // which is what the standard says to do when a long value was split.
  139. value = value.replace(/\?=\s+=\?/g, '?==?');
  140. return value.replace(/=\?([^?]+)\?([BbQq])\?([^?]*)\?=/g, function (all, charset, kind, payload) {
  141. try {
  142. if (kind === 'B' || kind === 'b') {
  143. return smartbotic.utils.base64Decode(payload);
  144. }
  145. // Q encoding: underscores are spaces, =XX is a byte.
  146. var text = payload.replace(/_/g, ' ');
  147. var bytes = text.replace(/=([0-9A-Fa-f]{2})/g, function (m, hex) {
  148. return String.fromCharCode(parseInt(hex, 16));
  149. });
  150. // The bytes are UTF-8; decodeURIComponent turns them into characters.
  151. try {
  152. return decodeURIComponent(escape(bytes));
  153. } catch (e) {
  154. return bytes;
  155. }
  156. } catch (e) {
  157. return all;
  158. }
  159. });
  160. }
  161. function parseMimeContent(rawEmail) {
  162. const attachments = [];
  163. // Find boundary from Content-Type header
  164. const boundaryMatch = rawEmail.match(/boundary=["']?([^"'\s;]+)["']?/i);
  165. if (!boundaryMatch) {
  166. // Not a multipart email, no attachments
  167. return attachments;
  168. }
  169. const boundary = boundaryMatch[1];
  170. const parts = rawEmail.split('--' + boundary);
  171. for (let i = 1; i < parts.length; i++) {
  172. const part = parts[i];
  173. // Skip the closing boundary marker
  174. if (part.trim() === '--' || part.trim().startsWith('--')) {
  175. continue;
  176. }
  177. // Parse headers and body from this part
  178. const headerEndIndex = part.indexOf('\r\n\r\n');
  179. const headerEndIndex2 = part.indexOf('\n\n');
  180. const actualHeaderEnd = headerEndIndex !== -1 ? headerEndIndex : headerEndIndex2;
  181. if (actualHeaderEnd === -1) continue;
  182. // Unfolded before anything reads a value out of it - see unfoldHeaders.
  183. const headers = unfoldHeaders(part.substring(0, actualHeaderEnd));
  184. const body = part.substring(actualHeaderEnd + (headerEndIndex !== -1 ? 4 : 2));
  185. // Check if this is an attachment
  186. const contentDisposition = headers.match(/Content-Disposition:\s*([^\r\n]+)/i);
  187. const contentType = headers.match(/Content-Type:\s*([^\r\n;]+)/i);
  188. const contentTransferEncoding = headers.match(/Content-Transfer-Encoding:\s*([^\r\n]+)/i);
  189. const contentId = headers.match(/Content-ID:\s*<?([^>\r\n]+)>?/i);
  190. // Extract filename from Content-Disposition or Content-Type
  191. let filename = null;
  192. const filenameMatch = headers.match(/filename=["']?([^"'\r\n;]+)["']?/i);
  193. const filenameStarMatch = headers.match(/filename\*=(?:UTF-8''|utf-8'')([^\r\n;]+)/i);
  194. const nameMatch = headers.match(/name=["']?([^"'\r\n;]+)["']?/i);
  195. if (filenameStarMatch) {
  196. try {
  197. filename = decodeURIComponent(filenameStarMatch[1]);
  198. } catch (e) {
  199. filename = filenameStarMatch[1];
  200. }
  201. } else if (filenameMatch) {
  202. filename = decodeEncodedWords(filenameMatch[1]);
  203. } else if (nameMatch) {
  204. filename = decodeEncodedWords(nameMatch[1]);
  205. }
  206. // Determine if this is an attachment (has filename or Content-Disposition: attachment)
  207. const isAttachment = filename ||
  208. (contentDisposition && contentDisposition[1].toLowerCase().includes('attachment'));
  209. // Skip if not an attachment and no filename
  210. if (!isAttachment) continue;
  211. // Get MIME type
  212. const mimeType = contentType ? contentType[1].trim() : 'application/octet-stream';
  213. // Skip text/plain and text/html parts (these are usually the email body)
  214. if (mimeType.startsWith('text/plain') || mimeType.startsWith('text/html')) {
  215. if (!filename) continue; // Only skip if no filename (body part)
  216. }
  217. // Decode content based on transfer encoding
  218. let data = body.trim();
  219. const encoding = contentTransferEncoding ? contentTransferEncoding[1].trim().toLowerCase() : '';
  220. if (encoding === 'base64') {
  221. // Already base64 encoded, just clean it up
  222. data = data.replace(/[\r\n\s]/g, '');
  223. } else if (encoding === 'quoted-printable') {
  224. // Decode quoted-printable and then base64 encode
  225. const decoded = decodeQuotedPrintable(data);
  226. data = smartbotic.utils.base64Encode(decoded);
  227. } else {
  228. // Raw or 7bit/8bit - base64 encode it
  229. data = smartbotic.utils.base64Encode(data);
  230. }
  231. // Generate filename if not present
  232. if (!filename) {
  233. const ext = getExtensionForMimeType(mimeType);
  234. filename = 'attachment_' + (attachments.length + 1) + ext;
  235. }
  236. // Calculate size (approximate from base64)
  237. const size = Math.floor(data.length * 3 / 4);
  238. attachments.push({
  239. type: 'binary',
  240. data: data,
  241. mimeType: mimeType,
  242. filename: filename,
  243. size: size,
  244. contentId: contentId ? contentId[1] : null
  245. });
  246. }
  247. return attachments;
  248. }
  249. /**
  250. * Decode quoted-printable encoding
  251. */
  252. function decodeQuotedPrintable(str) {
  253. return str
  254. .replace(/=\r?\n/g, '') // Remove soft line breaks
  255. .replace(/=([0-9A-Fa-f]{2})/g, (match, hex) => {
  256. return String.fromCharCode(parseInt(hex, 16));
  257. });
  258. }
  259. /**
  260. * Get file extension for common MIME types
  261. */
  262. function getExtensionForMimeType(mimeType) {
  263. const mimeMap = {
  264. 'application/pdf': '.pdf',
  265. 'application/msword': '.doc',
  266. 'application/vnd.openxmlformats-officedocument.wordprocessingml.document': '.docx',
  267. 'application/vnd.ms-excel': '.xls',
  268. 'application/vnd.openxmlformats-officedocument.spreadsheetml.sheet': '.xlsx',
  269. 'application/vnd.ms-powerpoint': '.ppt',
  270. 'application/vnd.openxmlformats-officedocument.presentationml.presentation': '.pptx',
  271. 'application/zip': '.zip',
  272. 'application/x-rar-compressed': '.rar',
  273. 'application/x-7z-compressed': '.7z',
  274. 'application/gzip': '.gz',
  275. 'image/jpeg': '.jpg',
  276. 'image/png': '.png',
  277. 'image/gif': '.gif',
  278. 'image/bmp': '.bmp',
  279. 'image/webp': '.webp',
  280. 'image/tiff': '.tiff',
  281. 'text/plain': '.txt',
  282. 'text/csv': '.csv',
  283. 'text/html': '.html',
  284. 'application/json': '.json',
  285. 'application/xml': '.xml'
  286. };
  287. return mimeMap[mimeType.toLowerCase()] || '.bin';
  288. }
  289. /**
  290. * Check if MIME type matches filter pattern (supports wildcards like "image/*")
  291. */
  292. function matchesMimeType(mimeType, patterns) {
  293. if (!patterns || patterns.length === 0) return true;
  294. const lowerMime = mimeType.toLowerCase();
  295. for (const pattern of patterns) {
  296. const lowerPattern = pattern.trim().toLowerCase();
  297. if (lowerPattern.endsWith('/*')) {
  298. const prefix = lowerPattern.slice(0, -1);
  299. if (lowerMime.startsWith(prefix)) return true;
  300. } else if (lowerMime === lowerPattern) {
  301. return true;
  302. }
  303. }
  304. return false;
  305. }
  306. /**
  307. * Check if filename matches extension filter
  308. */
  309. function matchesExtension(filename, extensions) {
  310. if (!extensions || extensions.length === 0) return true;
  311. const lowerFilename = filename.toLowerCase();
  312. for (const ext of extensions) {
  313. const lowerExt = ext.trim().toLowerCase();
  314. const dotExt = lowerExt.startsWith('.') ? lowerExt : '.' + lowerExt;
  315. if (lowerFilename.endsWith(dotExt)) return true;
  316. }
  317. return false;
  318. }
  319. module.exports = {
  320. configSchema,
  321. inputSchema,
  322. outputSchema,
  323. outputs,
  324. async execute(config, input, context) {
  325. // Get raw email content from input
  326. let rawEmail = null;
  327. // First priority: use configured emailSource if provided
  328. // The expression (e.g., {{item.body}}) is evaluated by the workflow engine,
  329. // so config.emailSource contains the actual email body string
  330. if (config.emailSource && typeof config.emailSource === 'string' && config.emailSource.length > 0) {
  331. rawEmail = config.emailSource;
  332. smartbotic.log.info('IMAP Extract Attachments: Using configured email source');
  333. }
  334. // Fallback: try to auto-detect from various input structures
  335. if (!rawEmail) {
  336. rawEmail = input.raw || input.body;
  337. if (!rawEmail && input.email) {
  338. rawEmail = input.email.raw || input.email.body;
  339. }
  340. if (!rawEmail && input.data) {
  341. rawEmail = input.data.raw || input.data.body;
  342. }
  343. // Support loop iteration input (item/currentItem from Loop node)
  344. if (!rawEmail && input.item) {
  345. rawEmail = input.item.raw || input.item.body;
  346. }
  347. if (!rawEmail && input.currentItem) {
  348. rawEmail = input.currentItem.raw || input.currentItem.body;
  349. }
  350. }
  351. if (!rawEmail) {
  352. // Log available keys for debugging
  353. const availableKeys = Object.keys(input || {}).join(', ');
  354. smartbotic.log.warn('IMAP Extract Attachments: No raw email found. Input keys: ' + availableKeys);
  355. throw new Error('No raw email content provided. Configure the "Email Source" field with an expression like {{item.body}} for loop iterations, or ensure the email data is properly connected.');
  356. }
  357. // Parse MIME content
  358. let attachments = parseMimeContent(rawEmail);
  359. // Apply filters
  360. const mimeFilters = config.filterMimeTypes
  361. ? config.filterMimeTypes.split(',').map(s => s.trim()).filter(s => s)
  362. : [];
  363. const extFilters = config.filterExtensions
  364. ? config.filterExtensions.split(',').map(s => s.trim()).filter(s => s)
  365. : [];
  366. const maxSize = config.maxSize || 0;
  367. attachments = attachments.filter(att => {
  368. // Filter by MIME type
  369. if (!matchesMimeType(att.mimeType, mimeFilters)) return false;
  370. // Filter by extension
  371. if (!matchesExtension(att.filename, extFilters)) return false;
  372. // Filter by size
  373. if (maxSize > 0 && att.size > maxSize) return false;
  374. return true;
  375. });
  376. // Store attachments - always write to disk, optionally store metadata in database
  377. const storagePath = config.binaryStoragePath || './data/attachments';
  378. const timestamp = Date.now();
  379. const deduplicateByHash = config.deduplicateByHash !== false; // Default to true
  380. for (const att of attachments) {
  381. // Generate file path - use content hash for deduplication if enabled
  382. const date = new Date(timestamp);
  383. const yearMonth = `${date.getFullYear()}-${String(date.getMonth() + 1).padStart(2, '0')}`;
  384. const safeFilename = att.filename.replace(/[^a-zA-Z0-9._-]/g, '_');
  385. let filePath;
  386. let skipWrite = false;
  387. if (deduplicateByHash) {
  388. // Calculate content hash for deduplication
  389. const contentHash = smartbotic.utils.hash(att.data);
  390. att.contentHash = contentHash;
  391. // Use hash-based path: basePath/YYYY-MM/hash_filename
  392. // Using first 16 chars of SHA-256 hash (64 bits) for reasonable uniqueness
  393. const shortHash = contentHash.substring(0, 16);
  394. filePath = `${storagePath}/${yearMonth}/${shortHash}_${safeFilename}`;
  395. // Check if file already exists
  396. if (smartbotic.fs.exists(filePath)) {
  397. smartbotic.log.info(`Attachment ${att.filename} already exists (hash: ${shortHash}), reusing: ${filePath}`);
  398. att.filePath = filePath;
  399. att.deduplicated = true;
  400. delete att.data; // Remove base64 data from memory
  401. skipWrite = true;
  402. }
  403. } else {
  404. // Use UUID-based path for unique storage
  405. const uuid = smartbotic.utils.uuid();
  406. filePath = `${storagePath}/${yearMonth}/${uuid}_${safeFilename}`;
  407. }
  408. if (!skipWrite) {
  409. // Write binary data to disk
  410. const writeResult = smartbotic.fs.writeFile(filePath, att.data, 'base64');
  411. if (writeResult.success) {
  412. // Replace base64 data with file path
  413. att.filePath = filePath;
  414. att.deduplicated = false;
  415. delete att.data; // Remove base64 data from memory
  416. smartbotic.log.info(`Stored attachment ${att.filename} to disk: ${filePath}`);
  417. } else {
  418. smartbotic.log.warn(`Failed to write attachment to disk: ${writeResult.error}`);
  419. // Keep base64 data as fallback
  420. }
  421. }
  422. }
  423. // Store metadata in database if configured (without binary data)
  424. if (config.storeInDatabase && attachments.length > 0) {
  425. const collection = config.storageCollection || 'email_attachments';
  426. const ttlHours = config.storageTtlHours ?? 24;
  427. const ttlMs = ttlHours > 0 ? ttlHours * 60 * 60 * 1000 : 0;
  428. for (const att of attachments) {
  429. // Store only metadata, not the binary data
  430. const storageDoc = {
  431. filename: att.filename,
  432. mimeType: att.mimeType,
  433. size: att.size,
  434. contentId: att.contentId,
  435. contentHash: att.contentHash, // For deduplication lookup
  436. deduplicated: att.deduplicated,
  437. extractedAt: timestamp,
  438. filePath: att.filePath // Reference to disk location
  439. };
  440. const insertResult = smartbotic.storage.insert(collection, storageDoc, null, ttlMs);
  441. if (insertResult.success) {
  442. att.storage = {
  443. collection: collection,
  444. id: insertResult.id
  445. };
  446. smartbotic.log.info(`Stored attachment metadata ${att.filename} in ${collection}/${insertResult.id}`);
  447. }
  448. }
  449. }
  450. // Calculate total size
  451. const totalSize = attachments.reduce((sum, att) => sum + att.size, 0);
  452. smartbotic.log.info(`Extracted ${attachments.length} attachment(s), total size: ${totalSize} bytes`);
  453. return {
  454. attachments,
  455. count: attachments.length,
  456. totalSize
  457. };
  458. }
  459. };