| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480 |
- /**
- * @node imap-extract-attachments
- * @name IMAP Extract Attachments
- * @category email
- * @version 1.4.0
- * @description Extract attachments from email as binary files. Compatible with OCR Upload and other binary-accepting nodes.
- * @icon paperclip
- */
- const configSchema = {
- type: 'object',
- properties: {
- emailSource: {
- type: 'string',
- title: 'Email Source',
- description: 'Path to email data containing the raw body (e.g., {{item.body}} for loop iterations, or {{data.body}} for direct input)'
- },
- deduplicateByHash: {
- type: 'boolean',
- title: 'Deduplicate by Content',
- description: 'Skip storing attachments that already exist (based on content hash). Prevents duplicates when processing the same email multiple times.',
- default: true
- },
- filterMimeTypes: {
- type: 'string',
- title: 'Filter by MIME Types',
- description: 'Comma-separated list of MIME types to include (e.g. application/pdf, image/*). Leave empty for all attachments.'
- },
- filterExtensions: {
- type: 'string',
- title: 'Filter by Extensions',
- description: 'Comma-separated list of file extensions to include (e.g. pdf, jpg, png). Leave empty for all.'
- },
- maxSize: {
- type: 'integer',
- title: 'Max Size (bytes)',
- description: 'Maximum attachment size in bytes (0 = no limit)',
- default: 0
- },
- storeInDatabase: {
- type: 'boolean',
- title: 'Store in Database',
- description: 'Store extracted attachments in database for later retrieval',
- default: false
- },
- storageCollection: {
- type: 'string',
- title: 'Storage Collection',
- description: 'Select collection to store attachments',
- dynamicOptions: {
- source: 'storage.collections',
- labelField: 'name',
- valueField: 'name',
- filter: { access: 'read-write' }
- }
- },
- storageTtlHours: {
- type: 'number',
- title: 'TTL (hours)',
- default: 24,
- description: 'Auto-delete stored files after hours (0 = never)'
- },
- binaryStoragePath: {
- type: 'string',
- title: 'Binary Storage Path',
- default: './data/attachments',
- description: 'Base directory for storing binary files on disk'
- }
- }
- };
- const inputSchema = {
- type: 'object',
- properties: {
- raw: {
- type: 'string',
- description: 'Raw email content (from IMAP Fetch or IMAP Trigger with fetchContent enabled)'
- },
- email: {
- type: 'object',
- description: 'Email object containing raw property'
- }
- }
- };
- const outputSchema = {
- type: 'object',
- properties: {
- attachments: {
- type: 'array',
- description: 'Array of extracted attachments as binary file objects',
- items: {
- type: 'object',
- properties: {
- type: { type: 'string', const: 'binary' },
- data: { type: 'string', description: 'Base64-encoded content' },
- mimeType: { type: 'string' },
- filename: { type: 'string' },
- size: { type: 'number' },
- contentId: { type: 'string', description: 'Content-ID for inline attachments' },
- storage: {
- type: 'object',
- description: 'Storage reference (if storeInDatabase enabled)',
- properties: {
- collection: { type: 'string' },
- id: { type: 'string' }
- }
- }
- }
- }
- },
- count: { type: 'integer', description: 'Number of attachments extracted' },
- totalSize: { type: 'integer', description: 'Total size of all attachments in bytes' }
- }
- };
- const outputs = [
- { name: 'main', displayName: 'Attachments', type: 'object', color: '#f59e0b' }
- ];
- /**
- * Parse MIME multipart content and extract attachments
- */
- function parseMimeContent(rawEmail) {
- const attachments = [];
- // Find boundary from Content-Type header
- const boundaryMatch = rawEmail.match(/boundary=["']?([^"'\s;]+)["']?/i);
- if (!boundaryMatch) {
- // Not a multipart email, no attachments
- return attachments;
- }
- const boundary = boundaryMatch[1];
- const parts = rawEmail.split('--' + boundary);
- for (let i = 1; i < parts.length; i++) {
- const part = parts[i];
- // Skip the closing boundary marker
- if (part.trim() === '--' || part.trim().startsWith('--')) {
- continue;
- }
- // Parse headers and body from this part
- const headerEndIndex = part.indexOf('\r\n\r\n');
- const headerEndIndex2 = part.indexOf('\n\n');
- const actualHeaderEnd = headerEndIndex !== -1 ? headerEndIndex : headerEndIndex2;
- if (actualHeaderEnd === -1) continue;
- const headers = part.substring(0, actualHeaderEnd);
- const body = part.substring(actualHeaderEnd + (headerEndIndex !== -1 ? 4 : 2));
- // Check if this is an attachment
- const contentDisposition = headers.match(/Content-Disposition:\s*([^\r\n]+)/i);
- const contentType = headers.match(/Content-Type:\s*([^\r\n;]+)/i);
- const contentTransferEncoding = headers.match(/Content-Transfer-Encoding:\s*([^\r\n]+)/i);
- const contentId = headers.match(/Content-ID:\s*<?([^>\r\n]+)>?/i);
- // Extract filename from Content-Disposition or Content-Type
- let filename = null;
- const filenameMatch = headers.match(/filename=["']?([^"'\r\n;]+)["']?/i);
- const filenameStarMatch = headers.match(/filename\*=(?:UTF-8''|utf-8'')([^\r\n;]+)/i);
- const nameMatch = headers.match(/name=["']?([^"'\r\n;]+)["']?/i);
- if (filenameStarMatch) {
- try {
- filename = decodeURIComponent(filenameStarMatch[1]);
- } catch (e) {
- filename = filenameStarMatch[1];
- }
- } else if (filenameMatch) {
- filename = filenameMatch[1];
- } else if (nameMatch) {
- filename = nameMatch[1];
- }
- // Determine if this is an attachment (has filename or Content-Disposition: attachment)
- const isAttachment = filename ||
- (contentDisposition && contentDisposition[1].toLowerCase().includes('attachment'));
- // Skip if not an attachment and no filename
- if (!isAttachment) continue;
- // Get MIME type
- const mimeType = contentType ? contentType[1].trim() : 'application/octet-stream';
- // Skip text/plain and text/html parts (these are usually the email body)
- if (mimeType.startsWith('text/plain') || mimeType.startsWith('text/html')) {
- if (!filename) continue; // Only skip if no filename (body part)
- }
- // Decode content based on transfer encoding
- let data = body.trim();
- const encoding = contentTransferEncoding ? contentTransferEncoding[1].trim().toLowerCase() : '';
- if (encoding === 'base64') {
- // Already base64 encoded, just clean it up
- data = data.replace(/[\r\n\s]/g, '');
- } else if (encoding === 'quoted-printable') {
- // Decode quoted-printable and then base64 encode
- const decoded = decodeQuotedPrintable(data);
- data = smartbotic.utils.base64Encode(decoded);
- } else {
- // Raw or 7bit/8bit - base64 encode it
- data = smartbotic.utils.base64Encode(data);
- }
- // Generate filename if not present
- if (!filename) {
- const ext = getExtensionForMimeType(mimeType);
- filename = 'attachment_' + (attachments.length + 1) + ext;
- }
- // Calculate size (approximate from base64)
- const size = Math.floor(data.length * 3 / 4);
- attachments.push({
- type: 'binary',
- data: data,
- mimeType: mimeType,
- filename: filename,
- size: size,
- contentId: contentId ? contentId[1] : null
- });
- }
- return attachments;
- }
- /**
- * Decode quoted-printable encoding
- */
- function decodeQuotedPrintable(str) {
- return str
- .replace(/=\r?\n/g, '') // Remove soft line breaks
- .replace(/=([0-9A-Fa-f]{2})/g, (match, hex) => {
- return String.fromCharCode(parseInt(hex, 16));
- });
- }
- /**
- * Get file extension for common MIME types
- */
- function getExtensionForMimeType(mimeType) {
- const mimeMap = {
- 'application/pdf': '.pdf',
- 'application/msword': '.doc',
- 'application/vnd.openxmlformats-officedocument.wordprocessingml.document': '.docx',
- 'application/vnd.ms-excel': '.xls',
- 'application/vnd.openxmlformats-officedocument.spreadsheetml.sheet': '.xlsx',
- 'application/vnd.ms-powerpoint': '.ppt',
- 'application/vnd.openxmlformats-officedocument.presentationml.presentation': '.pptx',
- 'application/zip': '.zip',
- 'application/x-rar-compressed': '.rar',
- 'application/x-7z-compressed': '.7z',
- 'application/gzip': '.gz',
- 'image/jpeg': '.jpg',
- 'image/png': '.png',
- 'image/gif': '.gif',
- 'image/bmp': '.bmp',
- 'image/webp': '.webp',
- 'image/tiff': '.tiff',
- 'text/plain': '.txt',
- 'text/csv': '.csv',
- 'text/html': '.html',
- 'application/json': '.json',
- 'application/xml': '.xml'
- };
- return mimeMap[mimeType.toLowerCase()] || '.bin';
- }
- /**
- * Check if MIME type matches filter pattern (supports wildcards like "image/*")
- */
- function matchesMimeType(mimeType, patterns) {
- if (!patterns || patterns.length === 0) return true;
- const lowerMime = mimeType.toLowerCase();
- for (const pattern of patterns) {
- const lowerPattern = pattern.trim().toLowerCase();
- if (lowerPattern.endsWith('/*')) {
- const prefix = lowerPattern.slice(0, -1);
- if (lowerMime.startsWith(prefix)) return true;
- } else if (lowerMime === lowerPattern) {
- return true;
- }
- }
- return false;
- }
- /**
- * Check if filename matches extension filter
- */
- function matchesExtension(filename, extensions) {
- if (!extensions || extensions.length === 0) return true;
- const lowerFilename = filename.toLowerCase();
- for (const ext of extensions) {
- const lowerExt = ext.trim().toLowerCase();
- const dotExt = lowerExt.startsWith('.') ? lowerExt : '.' + lowerExt;
- if (lowerFilename.endsWith(dotExt)) return true;
- }
- return false;
- }
- module.exports = {
- configSchema,
- inputSchema,
- outputSchema,
- outputs,
- async execute(config, input, context) {
- // Get raw email content from input
- let rawEmail = null;
- // First priority: use configured emailSource if provided
- // The expression (e.g., {{item.body}}) is evaluated by the workflow engine,
- // so config.emailSource contains the actual email body string
- if (config.emailSource && typeof config.emailSource === 'string' && config.emailSource.length > 0) {
- rawEmail = config.emailSource;
- smartbotic.log.info('IMAP Extract Attachments: Using configured email source');
- }
- // Fallback: try to auto-detect from various input structures
- if (!rawEmail) {
- rawEmail = input.raw || input.body;
- if (!rawEmail && input.email) {
- rawEmail = input.email.raw || input.email.body;
- }
- if (!rawEmail && input.data) {
- rawEmail = input.data.raw || input.data.body;
- }
- // Support loop iteration input (item/currentItem from Loop node)
- if (!rawEmail && input.item) {
- rawEmail = input.item.raw || input.item.body;
- }
- if (!rawEmail && input.currentItem) {
- rawEmail = input.currentItem.raw || input.currentItem.body;
- }
- }
- if (!rawEmail) {
- // Log available keys for debugging
- const availableKeys = Object.keys(input || {}).join(', ');
- smartbotic.log.warn('IMAP Extract Attachments: No raw email found. Input keys: ' + availableKeys);
- throw new Error('No raw email content provided. Configure the "Email Source" field with an expression like {{item.body}} for loop iterations, or ensure the email data is properly connected.');
- }
- // Parse MIME content
- let attachments = parseMimeContent(rawEmail);
- // Apply filters
- const mimeFilters = config.filterMimeTypes
- ? config.filterMimeTypes.split(',').map(s => s.trim()).filter(s => s)
- : [];
- const extFilters = config.filterExtensions
- ? config.filterExtensions.split(',').map(s => s.trim()).filter(s => s)
- : [];
- const maxSize = config.maxSize || 0;
- attachments = attachments.filter(att => {
- // Filter by MIME type
- if (!matchesMimeType(att.mimeType, mimeFilters)) return false;
- // Filter by extension
- if (!matchesExtension(att.filename, extFilters)) return false;
- // Filter by size
- if (maxSize > 0 && att.size > maxSize) return false;
- return true;
- });
- // Store attachments - always write to disk, optionally store metadata in database
- const storagePath = config.binaryStoragePath || './data/attachments';
- const timestamp = Date.now();
- const deduplicateByHash = config.deduplicateByHash !== false; // Default to true
- for (const att of attachments) {
- // Generate file path - use content hash for deduplication if enabled
- const date = new Date(timestamp);
- const yearMonth = `${date.getFullYear()}-${String(date.getMonth() + 1).padStart(2, '0')}`;
- const safeFilename = att.filename.replace(/[^a-zA-Z0-9._-]/g, '_');
- let filePath;
- let skipWrite = false;
- if (deduplicateByHash) {
- // Calculate content hash for deduplication
- const contentHash = smartbotic.utils.hash(att.data);
- att.contentHash = contentHash;
- // Use hash-based path: basePath/YYYY-MM/hash_filename
- // Using first 16 chars of SHA-256 hash (64 bits) for reasonable uniqueness
- const shortHash = contentHash.substring(0, 16);
- filePath = `${storagePath}/${yearMonth}/${shortHash}_${safeFilename}`;
- // Check if file already exists
- if (smartbotic.fs.exists(filePath)) {
- smartbotic.log.info(`Attachment ${att.filename} already exists (hash: ${shortHash}), reusing: ${filePath}`);
- att.filePath = filePath;
- att.deduplicated = true;
- delete att.data; // Remove base64 data from memory
- skipWrite = true;
- }
- } else {
- // Use UUID-based path for unique storage
- const uuid = smartbotic.utils.uuid();
- filePath = `${storagePath}/${yearMonth}/${uuid}_${safeFilename}`;
- }
- if (!skipWrite) {
- // Write binary data to disk
- const writeResult = smartbotic.fs.writeFile(filePath, att.data, 'base64');
- if (writeResult.success) {
- // Replace base64 data with file path
- att.filePath = filePath;
- att.deduplicated = false;
- delete att.data; // Remove base64 data from memory
- smartbotic.log.info(`Stored attachment ${att.filename} to disk: ${filePath}`);
- } else {
- smartbotic.log.warn(`Failed to write attachment to disk: ${writeResult.error}`);
- // Keep base64 data as fallback
- }
- }
- }
- // Store metadata in database if configured (without binary data)
- if (config.storeInDatabase && attachments.length > 0) {
- const collection = config.storageCollection || 'email_attachments';
- const ttlHours = config.storageTtlHours || 0;
- const ttlMs = ttlHours > 0 ? ttlHours * 60 * 60 * 1000 : 0;
- for (const att of attachments) {
- // Store only metadata, not the binary data
- const storageDoc = {
- filename: att.filename,
- mimeType: att.mimeType,
- size: att.size,
- contentId: att.contentId,
- contentHash: att.contentHash, // For deduplication lookup
- deduplicated: att.deduplicated,
- extractedAt: timestamp,
- filePath: att.filePath // Reference to disk location
- };
- const insertResult = smartbotic.storage.insert(collection, storageDoc, null, ttlMs);
- if (insertResult.success) {
- att.storage = {
- collection: collection,
- id: insertResult.id
- };
- smartbotic.log.info(`Stored attachment metadata ${att.filename} in ${collection}/${insertResult.id}`);
- }
- }
- }
- // Calculate total size
- const totalSize = attachments.reduce((sum, att) => sum + att.size, 0);
- smartbotic.log.info(`Extracted ${attachments.length} attachment(s), total size: ${totalSize} bytes`);
- return {
- attachments,
- count: attachments.length,
- totalSize
- };
- }
- };
|