🗂️ feat: Better Persistence for Code Execution Files Between Sessions (#11362)

* refactor: process code output files for re-use (WIP) * feat: file attachment handling with additional metadata for downloads * refactor: Update directory path logic for local file saving based on basePath * refactor: file attachment handling to support TFile type and improve data merging logic * feat: thread filtering of code-generated files - Introduced parentMessageId parameter in addedConvo and initialize functions to enhance thread management. - Updated related methods to utilize parentMessageId for retrieving messages and filtering code-generated files by conversation threads. - Enhanced type definitions to include parentMessageId in relevant interfaces for better clarity and usage. * chore: imports/params ordering * feat: update file model to use messageId for filtering and processing - Changed references from 'message' to 'messageId' in file-related methods for consistency. - Added messageId field to the file schema and updated related types. - Enhanced file processing logic to accommodate the new messageId structure. * feat: enhance file retrieval methods to support user-uploaded execute_code files - Added a new method `getUserCodeFiles` to retrieve user-uploaded execute_code files, excluding code-generated files. - Updated existing file retrieval methods to improve filtering logic and handle edge cases. - Enhanced thread data extraction to collect both message IDs and file IDs efficiently. - Integrated `getUserCodeFiles` into relevant endpoints for better file management in conversations. * chore: update @librechat/agents package version to 3.0.78 in package-lock.json and related package.json files * refactor: file processing and retrieval logic - Added a fallback mechanism for download URLs when files exceed size limits or cannot be processed locally. - Implemented a deduplication strategy for code-generated files based on conversationId and filename to optimize storage. - Updated file retrieval methods to ensure proper filtering by messageIds, preventing orphaned files from being included. - Introduced comprehensive tests for new thread data extraction functionality, covering edge cases and performance considerations. * fix: improve file retrieval tests and handling of optional properties - Updated tests to safely access optional properties using non-null assertions. - Modified test descriptions for clarity regarding the exclusion of execute_code files. - Ensured that the retrieval logic correctly reflects the expected outcomes for file queries. * test: add comprehensive unit tests for processCodeOutput functionality - Introduced a new test suite for the processCodeOutput function, covering various scenarios including file retrieval, creation, and processing for both image and non-image files. - Implemented mocks for dependencies such as axios, logger, and file models to isolate tests and ensure reliable outcomes. - Validated behavior for existing files, new file creation, and error handling, including size limits and fallback mechanisms. - Enhanced test coverage for metadata handling and usage increment logic, ensuring robust verification of file processing outcomes. * test: enhance file size limit enforcement in processCodeOutput tests - Introduced a configurable file size limit for tests to improve flexibility and coverage. - Mocked the `librechat-data-provider` to allow dynamic adjustment of file size limits during tests. - Updated the file size limit enforcement test to validate behavior when files exceed specified limits, ensuring proper fallback to download URLs. - Reset file size limit after tests to maintain isolation for subsequent test cases.
2026-02-28 13:24:10 +01:00 · 2026-01-16 10:06:24 -05:00 · 2026-01-16 10:06:24 -05:00 · 75c02a1a18
commit 75c02a1a18
parent c18dc0d894
22 changed files with 1362 additions and 81 deletions
--- a/packages/api/src/utils/message.spec.ts
+++ b/packages/api/src/utils/message.spec.ts
@ -1,4 +1,8 @@
-import { sanitizeFileForTransmit, sanitizeMessageForTransmit } from './message';
+import { Constants } from 'librechat-data-provider';
+import { sanitizeFileForTransmit, sanitizeMessageForTransmit, getThreadData } from './message';
+
+/** Cast to string for type compatibility with ThreadMessage */
+const NO_PARENT = Constants.NO_PARENT as string;

 describe('sanitizeFileForTransmit', () => {
  it('should remove text field from file', () => {
@ -120,3 +124,272 @@ describe('sanitizeMessageForTransmit', () => {
    expect(message.files[0].text).toBe('original text');
  });
 });
+
+describe('getThreadData', () => {
+  describe('edge cases - empty and null inputs', () => {
+    it('should return empty result for empty messages array', () => {
+      const result = getThreadData([], 'parent-123');
+
+      expect(result.messageIds).toEqual([]);
+      expect(result.fileIds).toEqual([]);
+    });
+
+    it('should return empty result for null parentMessageId', () => {
+      const messages = [
+        { messageId: 'msg-1', parentMessageId: null },
+        { messageId: 'msg-2', parentMessageId: 'msg-1' },
+      ];
+
+      const result = getThreadData(messages, null);
+
+      expect(result.messageIds).toEqual([]);
+      expect(result.fileIds).toEqual([]);
+    });
+
+    it('should return empty result for undefined parentMessageId', () => {
+      const messages = [{ messageId: 'msg-1', parentMessageId: null }];
+
+      const result = getThreadData(messages, undefined);
+
+      expect(result.messageIds).toEqual([]);
+      expect(result.fileIds).toEqual([]);
+    });
+
+    it('should return empty result when parentMessageId not found in messages', () => {
+      const messages = [
+        { messageId: 'msg-1', parentMessageId: null },
+        { messageId: 'msg-2', parentMessageId: 'msg-1' },
+      ];
+
+      const result = getThreadData(messages, 'non-existent');
+
+      expect(result.messageIds).toEqual([]);
+      expect(result.fileIds).toEqual([]);
+    });
+  });
+
+  describe('thread traversal', () => {
+    it('should traverse a simple linear thread', () => {
+      const messages = [
+        { messageId: 'msg-1', parentMessageId: NO_PARENT },
+        { messageId: 'msg-2', parentMessageId: 'msg-1' },
+        { messageId: 'msg-3', parentMessageId: 'msg-2' },
+      ];
+
+      const result = getThreadData(messages, 'msg-3');
+
+      expect(result.messageIds).toEqual(['msg-3', 'msg-2', 'msg-1']);
+      expect(result.fileIds).toEqual([]);
+    });
+
+    it('should stop at NO_PARENT constant', () => {
+      const messages = [
+        { messageId: 'msg-1', parentMessageId: NO_PARENT },
+        { messageId: 'msg-2', parentMessageId: 'msg-1' },
+      ];
+
+      const result = getThreadData(messages, 'msg-2');
+
+      expect(result.messageIds).toEqual(['msg-2', 'msg-1']);
+    });
+
+    it('should collect only messages in the thread branch', () => {
+      // Branched conversation: msg-1 -> msg-2 -> msg-3 (branch A)
+      //                       msg-1 -> msg-4 -> msg-5 (branch B)
+      const messages = [
+        { messageId: 'msg-1', parentMessageId: NO_PARENT },
+        { messageId: 'msg-2', parentMessageId: 'msg-1' },
+        { messageId: 'msg-3', parentMessageId: 'msg-2' },
+        { messageId: 'msg-4', parentMessageId: 'msg-1' },
+        { messageId: 'msg-5', parentMessageId: 'msg-4' },
+      ];
+
+      const resultBranchA = getThreadData(messages, 'msg-3');
+      expect(resultBranchA.messageIds).toEqual(['msg-3', 'msg-2', 'msg-1']);
+
+      const resultBranchB = getThreadData(messages, 'msg-5');
+      expect(resultBranchB.messageIds).toEqual(['msg-5', 'msg-4', 'msg-1']);
+    });
+
+    it('should handle single message thread', () => {
+      const messages = [{ messageId: 'msg-1', parentMessageId: NO_PARENT }];
+
+      const result = getThreadData(messages, 'msg-1');
+
+      expect(result.messageIds).toEqual(['msg-1']);
+      expect(result.fileIds).toEqual([]);
+    });
+  });
+
+  describe('circular reference protection', () => {
+    it('should handle circular references without infinite loop', () => {
+      // Malformed data: msg-2 points to msg-3 which points back to msg-2
+      const messages = [
+        { messageId: 'msg-1', parentMessageId: NO_PARENT },
+        { messageId: 'msg-2', parentMessageId: 'msg-3' },
+        { messageId: 'msg-3', parentMessageId: 'msg-2' },
+      ];
+
+      const result = getThreadData(messages, 'msg-2');
+
+      // Should stop when encountering a visited ID
+      expect(result.messageIds).toEqual(['msg-2', 'msg-3']);
+      expect(result.fileIds).toEqual([]);
+    });
+
+    it('should handle self-referencing message', () => {
+      const messages = [{ messageId: 'msg-1', parentMessageId: 'msg-1' }];
+
+      const result = getThreadData(messages, 'msg-1');
+
+      expect(result.messageIds).toEqual(['msg-1']);
+    });
+  });
+
+  describe('file ID collection', () => {
+    it('should collect file IDs from messages with files', () => {
+      const messages = [
+        {
+          messageId: 'msg-1',
+          parentMessageId: NO_PARENT,
+          files: [{ file_id: 'file-1' }, { file_id: 'file-2' }],
+        },
+        {
+          messageId: 'msg-2',
+          parentMessageId: 'msg-1',
+          files: [{ file_id: 'file-3' }],
+        },
+      ];
+
+      const result = getThreadData(messages, 'msg-2');
+
+      expect(result.messageIds).toEqual(['msg-2', 'msg-1']);
+      expect(result.fileIds).toContain('file-1');
+      expect(result.fileIds).toContain('file-2');
+      expect(result.fileIds).toContain('file-3');
+      expect(result.fileIds).toHaveLength(3);
+    });
+
+    it('should deduplicate file IDs across messages', () => {
+      const messages = [
+        {
+          messageId: 'msg-1',
+          parentMessageId: NO_PARENT,
+          files: [{ file_id: 'file-shared' }, { file_id: 'file-1' }],
+        },
+        {
+          messageId: 'msg-2',
+          parentMessageId: 'msg-1',
+          files: [{ file_id: 'file-shared' }, { file_id: 'file-2' }],
+        },
+      ];
+
+      const result = getThreadData(messages, 'msg-2');
+
+      expect(result.fileIds).toContain('file-shared');
+      expect(result.fileIds).toContain('file-1');
+      expect(result.fileIds).toContain('file-2');
+      expect(result.fileIds).toHaveLength(3);
+    });
+
+    it('should skip files without file_id', () => {
+      const messages = [
+        {
+          messageId: 'msg-1',
+          parentMessageId: NO_PARENT,
+          files: [{ file_id: 'file-1' }, { file_id: undefined }, { file_id: '' }],
+        },
+      ];
+
+      const result = getThreadData(messages, 'msg-1');
+
+      expect(result.fileIds).toEqual(['file-1']);
+    });
+
+    it('should handle messages with empty files array', () => {
+      const messages = [
+        {
+          messageId: 'msg-1',
+          parentMessageId: NO_PARENT,
+          files: [],
+        },
+        {
+          messageId: 'msg-2',
+          parentMessageId: 'msg-1',
+          files: [{ file_id: 'file-1' }],
+        },
+      ];
+
+      const result = getThreadData(messages, 'msg-2');
+
+      expect(result.messageIds).toEqual(['msg-2', 'msg-1']);
+      expect(result.fileIds).toEqual(['file-1']);
+    });
+
+    it('should handle messages without files property', () => {
+      const messages = [
+        { messageId: 'msg-1', parentMessageId: NO_PARENT },
+        {
+          messageId: 'msg-2',
+          parentMessageId: 'msg-1',
+          files: [{ file_id: 'file-1' }],
+        },
+      ];
+
+      const result = getThreadData(messages, 'msg-2');
+
+      expect(result.messageIds).toEqual(['msg-2', 'msg-1']);
+      expect(result.fileIds).toEqual(['file-1']);
+    });
+
+    it('should only collect files from messages in the thread', () => {
+      // msg-3 is not in the thread from msg-2
+      const messages = [
+        {
+          messageId: 'msg-1',
+          parentMessageId: NO_PARENT,
+          files: [{ file_id: 'file-1' }],
+        },
+        {
+          messageId: 'msg-2',
+          parentMessageId: 'msg-1',
+          files: [{ file_id: 'file-2' }],
+        },
+        {
+          messageId: 'msg-3',
+          parentMessageId: 'msg-1',
+          files: [{ file_id: 'file-3' }],
+        },
+      ];
+
+      const result = getThreadData(messages, 'msg-2');
+
+      expect(result.fileIds).toContain('file-1');
+      expect(result.fileIds).toContain('file-2');
+      expect(result.fileIds).not.toContain('file-3');
+    });
+  });
+
+  describe('performance - O(1) lookups', () => {
+    it('should handle large message arrays efficiently', () => {
+      // Create a linear thread of 1000 messages
+      const messages = [];
+      for (let i = 0; i < 1000; i++) {
+        messages.push({
+          messageId: `msg-${i}`,
+          parentMessageId: i === 0 ? NO_PARENT : `msg-${i - 1}`,
+          files: [{ file_id: `file-${i}` }],
+        });
+      }
+
+      const startTime = performance.now();
+      const result = getThreadData(messages, 'msg-999');
+      const endTime = performance.now();
+
+      expect(result.messageIds).toHaveLength(1000);
+      expect(result.fileIds).toHaveLength(1000);
+      // Should complete in reasonable time (< 100ms for 1000 messages)
+      expect(endTime - startTime).toBeLessThan(100);
+    });
+  });
+});
--- a/packages/api/src/utils/message.ts
+++ b/packages/api/src/utils/message.ts
@ -1,3 +1,4 @@
+import { Constants } from 'librechat-data-provider';
 import type { TFile, TMessage } from 'librechat-data-provider';

 /** Fields to strip from files before client transmission */
@ -66,3 +67,74 @@ export function sanitizeMessageForTransmit<T extends Partial<TMessage>>(

  return sanitized;
 }
+
+/** Minimal message shape for thread traversal */
+type ThreadMessage = {
+  messageId: string;
+  parentMessageId?: string | null;
+  files?: Array<{ file_id?: string }>;
+};
+
+/** Result of thread data extraction */
+export type ThreadData = {
+  messageIds: string[];
+  fileIds: string[];
+};
+
+/**
+ * Extracts thread message IDs and file IDs in a single O(n) pass.
+ * Builds a Map for O(1) lookups, then traverses the thread collecting both IDs.
+ *
+ * @param messages - All messages in the conversation (should be queried with select for efficiency)
+ * @param parentMessageId - The ID of the parent message to start traversal from
+ * @returns Object containing messageIds and fileIds arrays
+ */
+export function getThreadData(
+  messages: ThreadMessage[],
+  parentMessageId: string | null | undefined,
+): ThreadData {
+  const result: ThreadData = { messageIds: [], fileIds: [] };
+
+  if (!messages || messages.length === 0 || !parentMessageId) {
+    return result;
+  }
+
+  /** Build Map for O(1) lookups instead of O(n) .find() calls */
+  const messageMap = new Map<string, ThreadMessage>();
+  for (const msg of messages) {
+    messageMap.set(msg.messageId, msg);
+  }
+
+  const fileIdSet = new Set<string>();
+  const visitedIds = new Set<string>();
+  let currentId: string | null | undefined = parentMessageId;
+
+  /** Single traversal: collect message IDs and file IDs together */
+  while (currentId) {
+    if (visitedIds.has(currentId)) {
+      break;
+    }
+    visitedIds.add(currentId);
+
+    const message = messageMap.get(currentId);
+    if (!message) {
+      break;
+    }
+
+    result.messageIds.push(message.messageId);
+
+    /** Collect file IDs from this message */
+    if (message.files) {
+      for (const file of message.files) {
+        if (file.file_id) {
+          fileIdSet.add(file.file_id);
+        }
+      }
+    }
+
+    currentId = message.parentMessageId === Constants.NO_PARENT ? null : message.parentMessageId;
+  }
+
+  result.fileIds = Array.from(fileIdSet);
+  return result;
+}