diff --git a/src/backend/base/langflow/initial_setup/starter_projects/Vector Store RAG.json b/src/backend/base/langflow/initial_setup/starter_projects/Vector Store RAG.json index e321adbf34..d0aee7bf95 100644 --- a/src/backend/base/langflow/initial_setup/starter_projects/Vector Store RAG.json +++ b/src/backend/base/langflow/initial_setup/starter_projects/Vector Store RAG.json @@ -7,7 +7,7 @@ "data": { "sourceHandle": { "dataType": "ChatInput", - "id": "ChatInput-nEcic", + "id": "ChatInput-QY6i6", "name": "message", "output_types": [ "Message" @@ -15,7 +15,7 @@ }, "targetHandle": { "fieldName": "question", - "id": "Prompt-sgn73", + "id": "Prompt-POcpa", "inputTypes": [ "Message", "Text" @@ -23,12 +23,12 @@ "type": "str" } }, - "id": "reactflow__edge-ChatInput-nEcic{œdataTypeœ:œChatInputœ,œidœ:œChatInput-nEcicœ,œnameœ:œmessageœ,œoutput_typesœ:[œMessageœ]}-Prompt-sgn73{œfieldNameœ:œquestionœ,œidœ:œPrompt-sgn73œ,œinputTypesœ:[œMessageœ,œTextœ],œtypeœ:œstrœ}", + "id": "reactflow__edge-ChatInput-QY6i6{œdataTypeœ:œChatInputœ,œidœ:œChatInput-QY6i6œ,œnameœ:œmessageœ,œoutput_typesœ:[œMessageœ]}-Prompt-POcpa{œfieldNameœ:œquestionœ,œidœ:œPrompt-POcpaœ,œinputTypesœ:[œMessageœ,œTextœ],œtypeœ:œstrœ}", "selected": false, - "source": "ChatInput-nEcic", - "sourceHandle": "{œdataTypeœ: œChatInputœ, œidœ: œChatInput-nEcicœ, œnameœ: œmessageœ, œoutput_typesœ: [œMessageœ]}", - "target": "Prompt-sgn73", - "targetHandle": "{œfieldNameœ: œquestionœ, œidœ: œPrompt-sgn73œ, œinputTypesœ: [œMessageœ, œTextœ], œtypeœ: œstrœ}" + "source": "ChatInput-QY6i6", + "sourceHandle": "{œdataTypeœ: œChatInputœ, œidœ: œChatInput-QY6i6œ, œnameœ: œmessageœ, œoutput_typesœ: [œMessageœ]}", + "target": "Prompt-POcpa", + "targetHandle": "{œfieldNameœ: œquestionœ, œidœ: œPrompt-POcpaœ, œinputTypesœ: [œMessageœ, œTextœ], œtypeœ: œstrœ}" }, { "animated": false, @@ -36,7 +36,7 @@ "data": { "sourceHandle": { "dataType": "parser", - "id": "parser-conKB", + "id": "parser-2ehJ6", "name": "parsed_text", "output_types": [ "Message" @@ -44,7 +44,7 @@ }, "targetHandle": { "fieldName": "context", - "id": "Prompt-sgn73", + "id": "Prompt-POcpa", "inputTypes": [ "Message", "Text" @@ -52,44 +52,12 @@ "type": "str" } }, - "id": "reactflow__edge-parser-conKB{œdataTypeœ:œparserœ,œidœ:œparser-conKBœ,œnameœ:œparsed_textœ,œoutput_typesœ:[œMessageœ]}-Prompt-sgn73{œfieldNameœ:œcontextœ,œidœ:œPrompt-sgn73œ,œinputTypesœ:[œMessageœ,œTextœ],œtypeœ:œstrœ}", + "id": "reactflow__edge-parser-2ehJ6{œdataTypeœ:œparserœ,œidœ:œparser-2ehJ6œ,œnameœ:œparsed_textœ,œoutput_typesœ:[œMessageœ]}-Prompt-POcpa{œfieldNameœ:œcontextœ,œidœ:œPrompt-POcpaœ,œinputTypesœ:[œMessageœ,œTextœ],œtypeœ:œstrœ}", "selected": false, - "source": "parser-conKB", - "sourceHandle": "{œdataTypeœ: œparserœ, œidœ: œparser-conKBœ, œnameœ: œparsed_textœ, œoutput_typesœ: [œMessageœ]}", - "target": "Prompt-sgn73", - "targetHandle": "{œfieldNameœ: œcontextœ, œidœ: œPrompt-sgn73œ, œinputTypesœ: [œMessageœ, œTextœ], œtypeœ: œstrœ}" - }, - { - "animated": false, - "className": "", - "data": { - "sourceHandle": { - "dataType": "File", - "id": "File-7BcMh", - "name": "message", - "output_types": [ - "Message" - ] - }, - "targetHandle": { - "fieldName": "data_inputs", - "id": "SplitText-ZKezh", - "inputTypes": [ - "Data", - "JSON", - "DataFrame", - "Table", - "Message" - ], - "type": "other" - } - }, - "id": "reactflow__edge-File-7BcMh{œdataTypeœ:œFileœ,œidœ:œFile-7BcMhœ,œnameœ:œmessageœ,œoutput_typesœ:[œMessageœ]}-SplitText-ZKezh{œfieldNameœ:œdata_inputsœ,œidœ:œSplitText-ZKezhœ,œinputTypesœ:[œDataœ,œJSONœ,œDataFrameœ,œTableœ,œMessageœ],œtypeœ:œotherœ}", - "selected": false, - "source": "File-7BcMh", - "sourceHandle": "{œdataTypeœ: œFileœ, œidœ: œFile-7BcMhœ, œnameœ: œmessageœ, œoutput_typesœ: [œMessageœ]}", - "target": "SplitText-ZKezh", - "targetHandle": "{œfieldNameœ: œdata_inputsœ, œidœ: œSplitText-ZKezhœ, œinputTypesœ: [œDataœ, œJSONœ, œDataFrameœ, œTableœ, œMessageœ], œtypeœ: œotherœ}" + "source": "parser-2ehJ6", + "sourceHandle": "{œdataTypeœ: œparserœ, œidœ: œparser-2ehJ6œ, œnameœ: œparsed_textœ, œoutput_typesœ: [œMessageœ]}", + "target": "Prompt-POcpa", + "targetHandle": "{œfieldNameœ: œcontextœ, œidœ: œPrompt-POcpaœ, œinputTypesœ: [œMessageœ, œTextœ], œtypeœ: œstrœ}" }, { "animated": false, @@ -97,7 +65,7 @@ "data": { "sourceHandle": { "dataType": "Prompt", - "id": "Prompt-sgn73", + "id": "Prompt-POcpa", "name": "prompt", "output_types": [ "Message" @@ -105,19 +73,19 @@ }, "targetHandle": { "fieldName": "input_value", - "id": "LanguageModelComponent-lQgc6", + "id": "LanguageModelComponent-XSmrK", "inputTypes": [ "Message" ], "type": "str" } }, - "id": "reactflow__edge-Prompt-sgn73{œdataTypeœ:œPromptœ,œidœ:œPrompt-sgn73œ,œnameœ:œpromptœ,œoutput_typesœ:[œMessageœ]}-LanguageModelComponent-lQgc6{œfieldNameœ:œinput_valueœ,œidœ:œLanguageModelComponent-lQgc6œ,œinputTypesœ:[œMessageœ],œtypeœ:œstrœ}", + "id": "reactflow__edge-Prompt-POcpa{œdataTypeœ:œPromptœ,œidœ:œPrompt-POcpaœ,œnameœ:œpromptœ,œoutput_typesœ:[œMessageœ]}-LanguageModelComponent-XSmrK{œfieldNameœ:œinput_valueœ,œidœ:œLanguageModelComponent-XSmrKœ,œinputTypesœ:[œMessageœ],œtypeœ:œstrœ}", "selected": false, - "source": "Prompt-sgn73", - "sourceHandle": "{œdataTypeœ: œPromptœ, œidœ: œPrompt-sgn73œ, œnameœ: œpromptœ, œoutput_typesœ: [œMessageœ]}", - "target": "LanguageModelComponent-lQgc6", - "targetHandle": "{œfieldNameœ: œinput_valueœ, œidœ: œLanguageModelComponent-lQgc6œ, œinputTypesœ: [œMessageœ], œtypeœ: œstrœ}" + "source": "Prompt-POcpa", + "sourceHandle": "{œdataTypeœ: œPromptœ, œidœ: œPrompt-POcpaœ, œnameœ: œpromptœ, œoutput_typesœ: [œMessageœ]}", + "target": "LanguageModelComponent-XSmrK", + "targetHandle": "{œfieldNameœ: œinput_valueœ, œidœ: œLanguageModelComponent-XSmrKœ, œinputTypesœ: [œMessageœ], œtypeœ: œstrœ}" }, { "animated": false, @@ -125,7 +93,7 @@ "data": { "sourceHandle": { "dataType": "LanguageModelComponent", - "id": "LanguageModelComponent-lQgc6", + "id": "LanguageModelComponent-XSmrK", "name": "text_output", "output_types": [ "Message" @@ -133,7 +101,7 @@ }, "targetHandle": { "fieldName": "input_value", - "id": "ChatOutput-3IwvO", + "id": "ChatOutput-7BwOj", "inputTypes": [ "Data", "JSON", @@ -144,44 +112,12 @@ "type": "str" } }, - "id": "reactflow__edge-LanguageModelComponent-lQgc6{œdataTypeœ:œLanguageModelComponentœ,œidœ:œLanguageModelComponent-lQgc6œ,œnameœ:œtext_outputœ,œoutput_typesœ:[œMessageœ]}-ChatOutput-3IwvO{œfieldNameœ:œinput_valueœ,œidœ:œChatOutput-3IwvOœ,œinputTypesœ:[œDataœ,œJSONœ,œDataFrameœ,œTableœ,œMessageœ],œtypeœ:œstrœ}", + "id": "reactflow__edge-LanguageModelComponent-XSmrK{œdataTypeœ:œLanguageModelComponentœ,œidœ:œLanguageModelComponent-XSmrKœ,œnameœ:œtext_outputœ,œoutput_typesœ:[œMessageœ]}-ChatOutput-7BwOj{œfieldNameœ:œinput_valueœ,œidœ:œChatOutput-7BwOjœ,œinputTypesœ:[œDataœ,œJSONœ,œDataFrameœ,œTableœ,œMessageœ],œtypeœ:œstrœ}", "selected": false, - "source": "LanguageModelComponent-lQgc6", - "sourceHandle": "{œdataTypeœ: œLanguageModelComponentœ, œidœ: œLanguageModelComponent-lQgc6œ, œnameœ: œtext_outputœ, œoutput_typesœ: [œMessageœ]}", - "target": "ChatOutput-3IwvO", - "targetHandle": "{œfieldNameœ: œinput_valueœ, œidœ: œChatOutput-3IwvOœ, œinputTypesœ: [œDataœ, œJSONœ, œDataFrameœ, œTableœ, œMessageœ], œtypeœ: œstrœ}" - }, - { - "animated": false, - "className": "", - "data": { - "sourceHandle": { - "dataType": "SplitText", - "id": "SplitText-ZKezh", - "name": "dataframe", - "output_types": [ - "Table" - ] - }, - "targetHandle": { - "fieldName": "input_df", - "id": "KnowledgeIngestion-0tQZt", - "inputTypes": [ - "Message", - "Data", - "JSON", - "DataFrame", - "Table" - ], - "type": "other" - } - }, - "id": "xy-edge__SplitText-ZKezh{œdataTypeœ:œSplitTextœ,œidœ:œSplitText-ZKezhœ,œnameœ:œdataframeœ,œoutput_typesœ:[œTableœ]}-KnowledgeIngestion-0tQZt{œfieldNameœ:œinput_dfœ,œidœ:œKnowledgeIngestion-0tQZtœ,œinputTypesœ:[œMessageœ,œDataœ,œJSONœ,œDataFrameœ,œTableœ],œtypeœ:œotherœ}", - "selected": false, - "source": "SplitText-ZKezh", - "sourceHandle": "{œdataTypeœ: œSplitTextœ, œidœ: œSplitText-ZKezhœ, œnameœ: œdataframeœ, œoutput_typesœ: [œTableœ]}", - "target": "KnowledgeIngestion-0tQZt", - "targetHandle": "{œfieldNameœ: œinput_dfœ, œidœ: œKnowledgeIngestion-0tQZtœ, œinputTypesœ: [œMessageœ, œDataœ, œJSONœ, œDataFrameœ, œTableœ], œtypeœ: œotherœ}" + "source": "LanguageModelComponent-XSmrK", + "sourceHandle": "{œdataTypeœ: œLanguageModelComponentœ, œidœ: œLanguageModelComponent-XSmrKœ, œnameœ: œtext_outputœ, œoutput_typesœ: [œMessageœ]}", + "target": "ChatOutput-7BwOj", + "targetHandle": "{œfieldNameœ: œinput_valueœ, œidœ: œChatOutput-7BwOjœ, œinputTypesœ: [œDataœ, œJSONœ, œDataFrameœ, œTableœ, œMessageœ], œtypeœ: œstrœ}" }, { "animated": false, @@ -189,7 +125,7 @@ "data": { "sourceHandle": { "dataType": "ChatInput", - "id": "ChatInput-nEcic", + "id": "ChatInput-QY6i6", "name": "message", "output_types": [ "Message" @@ -197,19 +133,19 @@ }, "targetHandle": { "fieldName": "search_query", - "id": "KnowledgeBase-rTx4O", + "id": "KnowledgeBase-GjNLJ", "inputTypes": [ "Message" ], "type": "str" } }, - "id": "xy-edge__ChatInput-nEcic{œdataTypeœ:œChatInputœ,œidœ:œChatInput-nEcicœ,œnameœ:œmessageœ,œoutput_typesœ:[œMessageœ]}-KnowledgeBase-rTx4O{œfieldNameœ:œsearch_queryœ,œidœ:œKnowledgeBase-rTx4Oœ,œinputTypesœ:[œMessageœ],œtypeœ:œstrœ}", + "id": "reactflow__edge-ChatInput-QY6i6{œdataTypeœ:œChatInputœ,œidœ:œChatInput-QY6i6œ,œnameœ:œmessageœ,œoutput_typesœ:[œMessageœ]}-KnowledgeBase-GjNLJ{œfieldNameœ:œsearch_queryœ,œidœ:œKnowledgeBase-GjNLJœ,œinputTypesœ:[œMessageœ],œtypeœ:œstrœ}", "selected": false, - "source": "ChatInput-nEcic", - "sourceHandle": "{œdataTypeœ: œChatInputœ, œidœ: œChatInput-nEcicœ, œnameœ: œmessageœ, œoutput_typesœ: [œMessageœ]}", - "target": "KnowledgeBase-rTx4O", - "targetHandle": "{œfieldNameœ: œsearch_queryœ, œidœ: œKnowledgeBase-rTx4Oœ, œinputTypesœ: [œMessageœ], œtypeœ: œstrœ}" + "source": "ChatInput-QY6i6", + "sourceHandle": "{œdataTypeœ: œChatInputœ, œidœ: œChatInput-QY6i6œ, œnameœ: œmessageœ, œoutput_typesœ: [œMessageœ]}", + "target": "KnowledgeBase-GjNLJ", + "targetHandle": "{œfieldNameœ: œsearch_queryœ, œidœ: œKnowledgeBase-GjNLJœ, œinputTypesœ: [œMessageœ], œtypeœ: œstrœ}" }, { "animated": false, @@ -217,7 +153,7 @@ "data": { "sourceHandle": { "dataType": "KnowledgeBase", - "id": "KnowledgeBase-rTx4O", + "id": "KnowledgeBase-GjNLJ", "name": "retrieve_data", "output_types": [ "Table" @@ -225,7 +161,7 @@ }, "targetHandle": { "fieldName": "input_data", - "id": "parser-conKB", + "id": "parser-2ehJ6", "inputTypes": [ "DataFrame", "Table", @@ -235,12 +171,12 @@ "type": "other" } }, - "id": "xy-edge__KnowledgeBase-rTx4O{œdataTypeœ:œKnowledgeBaseœ,œidœ:œKnowledgeBase-rTx4Oœ,œnameœ:œretrieve_dataœ,œoutput_typesœ:[œTableœ]}-parser-conKB{œfieldNameœ:œinput_dataœ,œidœ:œparser-conKBœ,œinputTypesœ:[œDataFrameœ,œTableœ,œDataœ,œJSONœ],œtypeœ:œotherœ}", + "id": "reactflow__edge-KnowledgeBase-GjNLJ{œdataTypeœ:œKnowledgeBaseœ,œidœ:œKnowledgeBase-GjNLJœ,œnameœ:œretrieve_dataœ,œoutput_typesœ:[œTableœ]}-parser-2ehJ6{œfieldNameœ:œinput_dataœ,œidœ:œparser-2ehJ6œ,œinputTypesœ:[œDataFrameœ,œTableœ,œDataœ,œJSONœ],œtypeœ:œotherœ}", "selected": false, - "source": "KnowledgeBase-rTx4O", - "sourceHandle": "{œdataTypeœ: œKnowledgeBaseœ, œidœ: œKnowledgeBase-rTx4Oœ, œnameœ: œretrieve_dataœ, œoutput_typesœ: [œTableœ]}", - "target": "parser-conKB", - "targetHandle": "{œfieldNameœ: œinput_dataœ, œidœ: œparser-conKBœ, œinputTypesœ: [œDataFrameœ, œTableœ, œDataœ, œJSONœ], œtypeœ: œotherœ}" + "source": "KnowledgeBase-GjNLJ", + "sourceHandle": "{œdataTypeœ: œKnowledgeBaseœ, œidœ: œKnowledgeBase-GjNLJœ, œnameœ: œretrieve_dataœ, œoutput_typesœ: [œTableœ]}", + "target": "parser-2ehJ6", + "targetHandle": "{œfieldNameœ: œinput_dataœ, œidœ: œparser-2ehJ6œ, œinputTypesœ: [œDataFrameœ, œTableœ, œDataœ, œJSONœ], œtypeœ: œotherœ}" } ], "nodes": [ @@ -248,7 +184,7 @@ "data": { "description": "Get chat inputs from the Playground.", "display_name": "Chat Input", - "id": "ChatInput-nEcic", + "id": "ChatInput-QY6i6", "node": { "base_classes": [ "Message" @@ -487,14 +423,14 @@ "type": "ChatInput" }, "dragging": false, - "id": "ChatInput-nEcic", + "id": "ChatInput-QY6i6", "measured": { "height": 207, "width": 320 }, "position": { - "x": 827.4877596995269, - "y": 421.8759496538444 + "x": 803.9507719567438, + "y": 486.18094072679895 }, "positionAbsolute": { "x": 743.9745420290319, @@ -507,7 +443,7 @@ "data": { "description": "Create a prompt template with dynamic variables.", "display_name": "Prompt", - "id": "Prompt-sgn73", + "id": "Prompt-POcpa", "node": { "base_classes": [ "Message" @@ -702,14 +638,14 @@ "type": "Prompt" }, "dragging": false, - "id": "Prompt-sgn73", + "id": "Prompt-POcpa", "measured": { "height": 429, "width": 320 }, "position": { - "x": 1977.9097981422992, - "y": 640.5656416923846 + "x": 1958.7859956012876, + "y": 482.680940726799 }, "positionAbsolute": { "x": 1977.9097981422992, @@ -720,259 +656,11 @@ }, { "data": { - "description": "Split text into chunks based on specified criteria.", - "display_name": "Split Text", - "id": "SplitText-ZKezh", + "id": "note-GJegX", "node": { - "base_classes": [ - "Data", - "JSON" - ], - "beta": false, - "conditional_paths": [], - "custom_fields": {}, - "description": "Split text into chunks based on specified criteria.", - "display_name": "Split Text", - "documentation": "", - "edited": false, - "field_order": [ - "data_inputs", - "chunk_overlap", - "chunk_size", - "separator", - "text_key", - "keep_separator", - "clean_output" - ], - "frozen": false, - "icon": "scissors-line-dashed", - "legacy": false, - "lf_version": "1.9.0", - "metadata": { - "code_hash": "859adebdf672", - "dependencies": { - "dependencies": [ - { - "name": "langchain_text_splitters", - "version": "1.1.2" - }, - { - "name": "lfx", - "version": null - } - ], - "total_dependencies": 2 - }, - "module": "lfx.components.processing.split_text.SplitTextComponent" - }, - "output_types": [], - "outputs": [ - { - "allows_loop": false, - "cache": true, - "display_name": "Chunks", - "group_outputs": false, - "method": "split_text", - "name": "dataframe", - "selected": "Table", - "tool_mode": true, - "types": [ - "Table" - ], - "value": "__UNDEFINED__" - } - ], - "pinned": false, - "template": { - "_type": "Component", - "chunk_overlap": { - "advanced": false, - "display_name": "Chunk Overlap", - "dynamic": false, - "info": "Number of characters to overlap between chunks.", - "list": false, - "name": "chunk_overlap", - "placeholder": "", - "required": false, - "show": true, - "title_case": false, - "trace_as_metadata": true, - "type": "int", - "value": 200 - }, - "chunk_size": { - "advanced": false, - "display_name": "Chunk Size", - "dynamic": false, - "info": "The maximum length of each chunk. Text is first split by separator, then chunks are merged up to this size. Individual splits larger than this won't be further divided.", - "list": false, - "name": "chunk_size", - "placeholder": "", - "required": false, - "show": true, - "title_case": false, - "trace_as_metadata": true, - "type": "int", - "value": 1000 - }, - "clean_output": { - "_input_type": "BoolInput", - "advanced": true, - "display_name": "Clean Output", - "dynamic": false, - "info": "When enabled, only the text column is included in the output. Metadata columns are removed.", - "list": false, - "list_add_label": "Add More", - "name": "clean_output", - "override_skip": false, - "placeholder": "", - "required": false, - "show": true, - "title_case": false, - "tool_mode": false, - "trace_as_metadata": true, - "track_in_telemetry": true, - "type": "bool", - "value": false - }, - "code": { - "advanced": true, - "dynamic": true, - "fileTypes": [], - "file_path": "", - "info": "", - "list": false, - "load_from_db": false, - "multiline": true, - "name": "code", - "password": false, - "placeholder": "", - "required": true, - "show": true, - "title_case": false, - "type": "code", - "value": "from langchain_text_splitters import CharacterTextSplitter\n\nfrom lfx.custom.custom_component.component import Component\nfrom lfx.io import BoolInput, DropdownInput, HandleInput, IntInput, MessageTextInput, Output\nfrom lfx.schema.data import Data\nfrom lfx.schema.dataframe import DataFrame\nfrom lfx.schema.message import Message\nfrom lfx.utils.util import unescape_string\n\n\nclass SplitTextComponent(Component):\n display_name: str = \"Split Text\"\n description: str = \"Split text into chunks based on specified criteria.\"\n documentation: str = \"https://docs.langflow.org/split-text\"\n icon = \"scissors-line-dashed\"\n name = \"SplitText\"\n\n inputs = [\n HandleInput(\n name=\"data_inputs\",\n display_name=\"Input\",\n info=\"The data with texts to split in chunks.\",\n input_types=[\"Data\", \"JSON\", \"DataFrame\", \"Table\", \"Message\"],\n required=True,\n ),\n IntInput(\n name=\"chunk_overlap\",\n display_name=\"Chunk Overlap\",\n info=\"Number of characters to overlap between chunks.\",\n value=200,\n ),\n IntInput(\n name=\"chunk_size\",\n display_name=\"Chunk Size\",\n info=(\n \"The maximum length of each chunk. Text is first split by separator, \"\n \"then chunks are merged up to this size. \"\n \"Individual splits larger than this won't be further divided.\"\n ),\n value=1000,\n ),\n MessageTextInput(\n name=\"separator\",\n display_name=\"Separator\",\n info=(\n \"The character to split on. Use \\\\n for newline. \"\n \"Examples: \\\\n\\\\n for paragraphs, \\\\n for lines, . for sentences\"\n ),\n value=\"\\n\",\n ),\n MessageTextInput(\n name=\"text_key\",\n display_name=\"Text Key\",\n info=\"The key to use for the text column.\",\n value=\"text\",\n advanced=True,\n ),\n DropdownInput(\n name=\"keep_separator\",\n display_name=\"Keep Separator\",\n info=\"Whether to keep the separator in the output chunks and where to place it.\",\n options=[\"False\", \"True\", \"Start\", \"End\"],\n value=\"False\",\n advanced=True,\n ),\n BoolInput(\n name=\"clean_output\",\n display_name=\"Clean Output\",\n info=\"When enabled, only the text column is included in the output. Metadata columns are removed.\",\n value=False,\n advanced=True,\n ),\n ]\n\n outputs = [\n Output(display_name=\"Chunks\", name=\"dataframe\", method=\"split_text\"),\n ]\n\n def _docs_to_data(self, docs, *, clean: bool = False) -> list[Data]:\n return [\n Data(text=doc.page_content) if clean else Data(text=doc.page_content, data=doc.metadata) for doc in docs\n ]\n\n def _fix_separator(self, separator: str) -> str:\n \"\"\"Fix common separator issues and convert to proper format.\"\"\"\n if separator == \"/n\":\n return \"\\n\"\n if separator == \"/t\":\n return \"\\t\"\n return separator\n\n def split_text_base(self):\n separator = self._fix_separator(self.separator)\n separator = unescape_string(separator)\n\n if isinstance(self.data_inputs, DataFrame):\n if not len(self.data_inputs):\n msg = \"DataFrame is empty\"\n raise TypeError(msg)\n\n self.data_inputs.text_key = self.text_key\n try:\n documents = self.data_inputs.to_lc_documents()\n except Exception as e:\n msg = f\"Error converting DataFrame to documents: {e}\"\n raise TypeError(msg) from e\n elif isinstance(self.data_inputs, Message):\n self.data_inputs = [self.data_inputs.to_data()]\n return self.split_text_base()\n else:\n if not self.data_inputs:\n msg = \"No data inputs provided\"\n raise TypeError(msg)\n\n documents = []\n if isinstance(self.data_inputs, Data):\n self.data_inputs.text_key = self.text_key\n documents = [self.data_inputs.to_lc_document()]\n else:\n try:\n documents = [input_.to_lc_document() for input_ in self.data_inputs if isinstance(input_, Data)]\n if not documents:\n msg = f\"No valid Data inputs found in {type(self.data_inputs)}\"\n raise TypeError(msg)\n except AttributeError as e:\n msg = f\"Invalid input type in collection: {e}\"\n raise TypeError(msg) from e\n try:\n # Convert string 'False'/'True' to boolean\n keep_sep = self.keep_separator\n if isinstance(keep_sep, str):\n if keep_sep.lower() == \"false\":\n keep_sep = False\n elif keep_sep.lower() == \"true\":\n keep_sep = True\n # 'start' and 'end' are kept as strings\n\n splitter = CharacterTextSplitter(\n chunk_overlap=self.chunk_overlap,\n chunk_size=self.chunk_size,\n separator=separator,\n keep_separator=keep_sep,\n )\n return splitter.split_documents(documents)\n except Exception as e:\n msg = f\"Error splitting text: {e}\"\n raise TypeError(msg) from e\n\n def split_text(self) -> DataFrame:\n docs = self.split_text_base()\n df = DataFrame(self._docs_to_data(docs, clean=self.clean_output))\n return df if self.clean_output else df.smart_column_order()\n" - }, - "data_inputs": { - "advanced": false, - "display_name": "Input", - "dynamic": false, - "info": "The data with texts to split in chunks.", - "input_types": [ - "Data", - "JSON", - "DataFrame", - "Table", - "Message" - ], - "list": false, - "name": "data_inputs", - "placeholder": "", - "required": true, - "show": true, - "title_case": false, - "trace_as_metadata": true, - "type": "other", - "value": "" - }, - "keep_separator": { - "_input_type": "DropdownInput", - "advanced": true, - "combobox": false, - "dialog_inputs": {}, - "display_name": "Keep Separator", - "dynamic": false, - "info": "Whether to keep the separator in the output chunks and where to place it.", - "name": "keep_separator", - "options": [ - "False", - "True", - "Start", - "End" - ], - "options_metadata": [], - "placeholder": "", - "required": false, - "show": true, - "title_case": false, - "tool_mode": false, - "trace_as_metadata": true, - "type": "str", - "value": "False" - }, - "separator": { - "advanced": false, - "display_name": "Separator", - "dynamic": false, - "info": "The character to split on. Use \\n for newline. Examples: \\n\\n for paragraphs, \\n for lines, . for sentences", - "input_types": [ - "Message" - ], - "list": false, - "load_from_db": false, - "name": "separator", - "placeholder": "", - "required": false, - "show": true, - "title_case": false, - "trace_as_input": true, - "trace_as_metadata": true, - "type": "str", - "value": "\n" - }, - "text_key": { - "_input_type": "MessageTextInput", - "advanced": true, - "display_name": "Text Key", - "dynamic": false, - "info": "The key to use for the text column.", - "input_types": [ - "Message" - ], - "list": false, - "list_add_label": "Add More", - "load_from_db": false, - "name": "text_key", - "placeholder": "", - "required": false, - "show": true, - "title_case": false, - "tool_mode": false, - "trace_as_input": true, - "trace_as_metadata": true, - "type": "str", - "value": "text" - } - } - }, - "selected_output": "chunks", - "type": "SplitText" - }, - "dragging": false, - "id": "SplitText-ZKezh", - "measured": { - "height": 415, - "width": 320 - }, - "position": { - "x": 1752.5928994344388, - "y": 1342.7825043187643 - }, - "positionAbsolute": { - "x": 1683.4543896546102, - "y": 1350.7871623588553 - }, - "selected": false, - "type": "genericNode" - }, - { - "data": { - "id": "note-HcazY", - "node": { - "description": "## 🐕 2. Retriever Flow\n\nThis flow answers your questions with contextual data retrieved from your vector database.\n\nOpen the **Playground** and ask, \n\n```\nWhat is this document about?\n```\n", + "description": "# 📖 README\n Retrieval Augmented Generation (RAG) is a way of providing additional context to a Large Language Model (LLM) by preloading a vector store with embeddings for relevant content. When a user chats with the LLM, a _similarity search_ retrieves relevant content by comparing an embedding for the user's query against the embeddings in the vector\n store.\n For example, a RAG chatbot could be pre-loaded with product data, and then it can help customers find specific products based on their queries.\n\n ## Quick start\n 1. Configure your **Model Provider** with your API credentials.\n 2. Select a database or create a new one using the **[Knowledge Ingestion](http://localhost:7860/assets/knowledge-bases)** feature. \n 3. Open the **Playground** to start a chat with the 🐕 **Retriever** flow.\n\n\n\n ## Next steps\n Experiment by changing the prompt and the loaded data to see how the LLM's responses change.", "display_name": "", "documentation": "", - "i18n_key": "template_notes.vector_store_rag.f8f85922", "template": { "backgroundColor": "neutral" } @@ -980,53 +668,15 @@ "type": "note" }, "dragging": false, - "height": 324, - "id": "note-HcazY", + "height": 1107, + "id": "note-GJegX", "measured": { - "height": 324, - "width": 324 + "height": 1107, + "width": 404 }, "position": { - "x": 374.388314931542, - "y": 486.18094072679895 - }, - "positionAbsolute": { - "x": 374.388314931542, - "y": 486.18094072679895 - }, - "resizing": false, - "selected": false, - "style": { - "height": 324, - "width": 324 - }, - "type": "noteNode", - "width": 324 - }, - { - "data": { - "id": "note-lE4I6", - "node": { - "description": "# 📖 README\n Retrieval Augmented Generation (RAG) is a way of providing additional context to a Large Language Model (LLM) by preloading a vector store with embeddings for relevant content. When a user chats with the LLM, a _similarity search_ retrieves relevant content by comparing an embedding for the user's query against the embeddings in the vector\n store.\n For example, a RAG chatbot could be pre-loaded with product data, and then it can help customers find specific products based on their queries.\n This template has two sub-flows. One flow loads data into your knowledge base, and the other is the user-driven chat flow that compares a new query against the existing\n content in your knowledge base.\n\n ## Quick start\n 1. Configure your **Model Provider** with your API credentials.\n 2. Select a database or create a new one using the **Knowledge Ingestion** component.\n\n ## Run the flows\n 1. Load your data into the knowledge base with the 📚 **Load Data** flow. Select a file to upload in the **File** component, and then click **Play** ▶️ on the **Knowledge\n Ingestion** component to run the **Load Data** flow.\n 2. Open the **Playground** to start a chat with the 🐕 **Retriever** flow.\n\n Only run the **Load Data** flow when you need to populate your knowledge base with baseline content, such as product data.\n The **Retriever** flow is the user-facing chat flow. This flow generates an embedding from chat input, runs a similarity search against the knowledge base to retrieve relevant\n content, and then passes the original query and the retrieved content to the LLM, which produces the chat response sent to the user.\n\n ## Next steps\n Experiment by changing the prompt and the loaded data to see how the LLM's responses change.", - "display_name": "", - "documentation": "", - "i18n_key": "template_notes.vector_store_rag.1591280a", - "template": { - "backgroundColor": "neutral" - } - }, - "type": "note" - }, - "dragging": false, - "height": 725, - "id": "note-lE4I6", - "measured": { - "height": 725, - "width": 399 - }, - "position": { - "x": 191.12162720143235, - "y": 1157.6038620251531 + "x": 337.40222632114364, + "y": 488.378713524636 }, "positionAbsolute": { "x": 94.28986613312418, @@ -1039,13 +689,13 @@ "width": 324 }, "type": "noteNode", - "width": 399 + "width": 404 }, { "data": { "description": "Display a chat message in the Playground.", "display_name": "Chat Output", - "id": "ChatOutput-3IwvO", + "id": "ChatOutput-7BwOj", "node": { "base_classes": [ "Message" @@ -1305,14 +955,14 @@ "type": "ChatOutput" }, "dragging": false, - "id": "ChatOutput-3IwvO", + "id": "ChatOutput-7BwOj", "measured": { "height": 207, "width": 320 }, "position": { - "x": 2738.611008351098, - "y": 829.6219994149209 + "x": 2715.074020608315, + "y": 486.18094072679895 }, "positionAbsolute": { "x": 2734.385670401691, @@ -1323,50 +973,11 @@ }, { "data": { - "id": "note-SnIby", + "id": "note-H0ORp", "node": { - "description": "## 📚 1. Load Data Flow\n\nRun this first! Load data from a local file and embed it into the vector database.\n\nSelect a Database and a Collection, or create new ones. \n\nClick **Run component** on the **Knowledge Ingestion** component to load your data.\n\n\n### Next steps:\n Experiment by changing the prompt and the contextual data to see how the retrieval flow's responses change.", + "description": "### 💡 Configure your language model here👇", "display_name": "", "documentation": "", - "i18n_key": "template_notes.vector_store_rag.09dbcc90", - "template": { - "backgroundColor": "neutral" - } - }, - "type": "note" - }, - "dragging": false, - "height": 460, - "id": "note-SnIby", - "measured": { - "height": 460, - "width": 340 - }, - "position": { - "x": 913.9906853654297, - "y": 1523.8879126168624 - }, - "positionAbsolute": { - "x": 955.3277857006676, - "y": 1552.171191793604 - }, - "resizing": false, - "selected": false, - "style": { - "height": 324, - "width": 324 - }, - "type": "noteNode", - "width": 340 - }, - { - "data": { - "id": "note-lrteq", - "node": { - "description": "### 💡 Add your OpenAI API key here 👇", - "display_name": "", - "documentation": "", - "i18n_key": "template_notes.vector_store_rag.ec099fc5", "template": { "backgroundColor": "transparent" } @@ -1375,14 +986,14 @@ }, "dragging": false, "height": 324, - "id": "note-lrteq", + "id": "note-H0ORp", "measured": { "height": 324, "width": 324 }, "position": { - "x": 2350.297636215281, - "y": 577.4592910079571 + "x": 2342.463816175529, + "y": 393.9443397009448 }, "positionAbsolute": { "x": 2350.297636215281, @@ -1394,7 +1005,7 @@ }, { "data": { - "id": "parser-conKB", + "id": "parser-2ehJ6", "node": { "base_classes": [ "Message" @@ -1571,712 +1182,21 @@ "type": "parser" }, "dragging": false, - "id": "parser-conKB", + "id": "parser-2ehJ6", "measured": { "height": 331, "width": 320 }, "position": { - "x": 1583.5982144641368, - "y": 651.635660385082 + "x": 1570.3586588588214, + "y": 482.680940726799 }, "selected": false, "type": "genericNode" }, { "data": { - "id": "File-7BcMh", - "node": { - "base_classes": [ - "Message" - ], - "beta": false, - "conditional_paths": [], - "custom_fields": {}, - "description": "Loads and returns the content from uploaded files.", - "display_name": "File", - "documentation": "", - "edited": false, - "field_order": [ - "storage_location", - "path", - "file_path", - "separator", - "silent_errors", - "delete_server_file_after_processing", - "ignore_unsupported_extensions", - "ignore_unspecified_files", - "file_path_str", - "aws_access_key_id", - "aws_secret_access_key", - "bucket_name", - "aws_region", - "s3_file_key", - "service_account_key", - "file_id", - "advanced_mode", - "pipeline", - "ocr_engine", - "md_image_placeholder", - "md_page_break_placeholder", - "doc_key", - "use_multithreading", - "concurrency_multithreading", - "markdown" - ], - "frozen": false, - "icon": "file-text", - "last_updated": "2026-04-10T20:58:58.107Z", - "legacy": false, - "lf_version": "1.9.0", - "metadata": { - "code_hash": "c20646f04f8e", - "dependencies": { - "dependencies": [ - { - "name": "lfx", - "version": null - }, - { - "name": "langchain_core", - "version": "1.3.2" - }, - { - "name": "pydantic", - "version": "2.13.3" - }, - { - "name": "googleapiclient", - "version": "2.195.0" - } - ], - "total_dependencies": 4 - }, - "module": "lfx.components.files_and_knowledge.file.FileComponent" - }, - "minimized": false, - "output_types": [], - "outputs": [ - { - "allows_loop": false, - "cache": true, - "display_name": "Raw Content", - "group_outputs": false, - "method": "load_files_message", - "name": "message", - "selected": "Message", - "tool_mode": true, - "types": [ - "Message" - ], - "value": "__UNDEFINED__" - } - ], - "pinned": false, - "template": { - "_type": "Component", - "advanced_mode": { - "_input_type": "BoolInput", - "advanced": false, - "display_name": "Advanced Parser", - "dynamic": false, - "info": "Enable advanced document processing and export with Docling for PDFs, images, and office documents. Note that advanced document processing can consume significant resources.", - "list": false, - "list_add_label": "Add More", - "name": "advanced_mode", - "placeholder": "", - "real_time_refresh": true, - "required": false, - "show": true, - "title_case": false, - "tool_mode": false, - "trace_as_metadata": true, - "type": "bool", - "value": false - }, - "aws_access_key_id": { - "_input_type": "SecretStrInput", - "advanced": false, - "display_name": "AWS Access Key ID", - "dynamic": false, - "info": "AWS Access key ID.", - "input_types": [], - "load_from_db": false, - "name": "aws_access_key_id", - "override_skip": false, - "password": true, - "placeholder": "", - "required": true, - "show": false, - "title_case": false, - "track_in_telemetry": false, - "type": "str", - "value": "" - }, - "aws_region": { - "_input_type": "StrInput", - "advanced": false, - "display_name": "AWS Region", - "dynamic": false, - "info": "AWS region (e.g., us-east-1, eu-west-1).", - "list": false, - "list_add_label": "Add More", - "load_from_db": false, - "name": "aws_region", - "override_skip": false, - "placeholder": "", - "required": false, - "show": false, - "title_case": false, - "tool_mode": false, - "trace_as_metadata": true, - "track_in_telemetry": false, - "type": "str", - "value": "" - }, - "aws_secret_access_key": { - "_input_type": "SecretStrInput", - "advanced": false, - "display_name": "AWS Secret Key", - "dynamic": false, - "info": "AWS Secret Key.", - "input_types": [], - "load_from_db": false, - "name": "aws_secret_access_key", - "override_skip": false, - "password": true, - "placeholder": "", - "required": true, - "show": false, - "title_case": false, - "track_in_telemetry": false, - "type": "str", - "value": "" - }, - "bucket_name": { - "_input_type": "StrInput", - "advanced": false, - "display_name": "S3 Bucket Name", - "dynamic": false, - "info": "Enter the name of the S3 bucket.", - "list": false, - "list_add_label": "Add More", - "load_from_db": false, - "name": "bucket_name", - "override_skip": false, - "placeholder": "", - "required": true, - "show": false, - "title_case": false, - "tool_mode": false, - "trace_as_metadata": true, - "track_in_telemetry": false, - "type": "str", - "value": "" - }, - "code": { - "advanced": true, - "dynamic": true, - "fileTypes": [], - "file_path": "", - "info": "", - "list": false, - "load_from_db": false, - "multiline": true, - "name": "code", - "password": false, - "placeholder": "", - "required": true, - "show": true, - "title_case": false, - "type": "code", - "value": "\"\"\"Enhanced file component with Docling support and process isolation.\n\nNotes:\n-----\n- ALL Docling parsing/export runs in a separate OS process to prevent memory\n growth and native library state from impacting the main Langflow process.\n- Standard text/structured parsing continues to use existing BaseFileComponent\n utilities (and optional threading via `parallel_load_data`).\n\"\"\"\n\nfrom __future__ import annotations\n\nimport contextlib\nimport json\nimport subprocess\nimport sys\nimport textwrap\nimport time\nfrom copy import deepcopy\nfrom pathlib import Path\nfrom tempfile import NamedTemporaryFile\nfrom typing import Any\n\nfrom lfx.base.data.base_file import BaseFileComponent\nfrom lfx.base.data.storage_utils import parse_storage_path, read_file_bytes, validate_image_content_type\nfrom lfx.base.data.utils import TEXT_FILE_TYPES, parallel_load_data, parse_text_file_to_data\nfrom lfx.inputs import SortableListInput\nfrom lfx.inputs.inputs import DropdownInput, MessageTextInput, StrInput\nfrom lfx.io import BoolInput, FileInput, IntInput, Output, SecretStrInput\nfrom lfx.schema.data import Data\nfrom lfx.schema.dataframe import DataFrame # noqa: TC001\nfrom lfx.schema.message import Message\nfrom lfx.services.deps import get_settings_service, get_storage_service\nfrom lfx.utils.async_helpers import run_until_complete\nfrom lfx.utils.validate_cloud import is_astra_cloud_environment\n\n\ndef _get_storage_location_options():\n \"\"\"Get storage location options, filtering out Local if in Astra cloud environment.\"\"\"\n all_options = [{\"name\": \"AWS\", \"icon\": \"Amazon\"}, {\"name\": \"Google Drive\", \"icon\": \"google\"}]\n if is_astra_cloud_environment():\n return all_options\n return [{\"name\": \"Local\", \"icon\": \"hard-drive\"}, *all_options]\n\n\nclass FileComponent(BaseFileComponent):\n \"\"\"File component with optional Docling processing (isolated in a subprocess).\"\"\"\n\n display_name = \"Read File\"\n # description is now a dynamic property - see get_tool_description()\n _base_description = \"Loads content from one or more files.\"\n documentation: str = \"https://docs.langflow.org/read-file\"\n icon = \"file-text\"\n name = \"File\"\n add_tool_output = True # Enable tool mode toggle without requiring tool_mode inputs\n\n # Extensions that can be processed without Docling (using standard text parsing)\n TEXT_EXTENSIONS = TEXT_FILE_TYPES\n\n # Extensions that require Docling for processing (images, advanced office formats, etc.)\n DOCLING_ONLY_EXTENSIONS = [\n \"adoc\",\n \"asciidoc\",\n \"asc\",\n \"bmp\",\n \"dotx\",\n \"dotm\",\n \"docm\",\n \"jpg\",\n \"jpeg\",\n \"png\",\n \"potx\",\n \"ppsx\",\n \"pptm\",\n \"potm\",\n \"ppsm\",\n \"pptx\",\n \"tiff\",\n \"xls\",\n \"xlsx\",\n \"xhtml\",\n \"webp\",\n ]\n\n # Docling-supported/compatible extensions; TEXT_FILE_TYPES are supported by the base loader.\n VALID_EXTENSIONS = [\n *TEXT_EXTENSIONS,\n *DOCLING_ONLY_EXTENSIONS,\n ]\n\n # Fixed export settings used when markdown export is requested.\n EXPORT_FORMAT = \"Markdown\"\n IMAGE_MODE = \"placeholder\"\n\n _base_inputs = deepcopy(BaseFileComponent.get_base_inputs())\n\n for input_item in _base_inputs:\n if isinstance(input_item, FileInput) and input_item.name == \"path\":\n input_item.real_time_refresh = True\n input_item.tool_mode = False # Disable tool mode for file upload input\n input_item.required = False # Make it optional so it doesn't error in tool mode\n break\n\n inputs = [\n SortableListInput(\n name=\"storage_location\",\n display_name=\"Storage Location\",\n placeholder=\"Select Location\",\n info=\"Choose where to read the file from.\",\n options=_get_storage_location_options(),\n real_time_refresh=True,\n limit=1,\n value=[{\"name\": \"Local\", \"icon\": \"hard-drive\"}],\n advanced=True,\n ),\n *_base_inputs,\n StrInput(\n name=\"file_path_str\",\n display_name=\"File Path\",\n info=(\n \"Path to the file to read. Used when component is called as a tool. \"\n \"If not provided, will use the uploaded file from 'path' input.\"\n ),\n show=False,\n advanced=True,\n tool_mode=True, # Required for Toolset toggle, but _get_tools() ignores this parameter\n required=False,\n ),\n # AWS S3 specific inputs\n SecretStrInput(\n name=\"aws_access_key_id\",\n display_name=\"AWS Access Key ID\",\n info=\"AWS Access key ID.\",\n show=False,\n advanced=False,\n required=True,\n ),\n SecretStrInput(\n name=\"aws_secret_access_key\",\n display_name=\"AWS Secret Key\",\n info=\"AWS Secret Key.\",\n show=False,\n advanced=False,\n required=True,\n ),\n StrInput(\n name=\"bucket_name\",\n display_name=\"S3 Bucket Name\",\n info=\"Enter the name of the S3 bucket.\",\n show=False,\n advanced=False,\n required=True,\n ),\n StrInput(\n name=\"aws_region\",\n display_name=\"AWS Region\",\n info=\"AWS region (e.g., us-east-1, eu-west-1).\",\n show=False,\n advanced=False,\n ),\n StrInput(\n name=\"s3_file_key\",\n display_name=\"S3 File Key\",\n info=\"The key (path) of the file in S3 bucket.\",\n show=False,\n advanced=False,\n required=True,\n ),\n # Google Drive specific inputs\n SecretStrInput(\n name=\"service_account_key\",\n display_name=\"GCP Credentials Secret Key\",\n info=\"Your Google Cloud Platform service account JSON key as a secret string (complete JSON content).\",\n show=False,\n advanced=False,\n required=True,\n ),\n StrInput(\n name=\"file_id\",\n display_name=\"Google Drive File ID\",\n info=(\"The Google Drive file ID to read. The file must be shared with the service account email.\"),\n show=False,\n advanced=False,\n required=True,\n ),\n BoolInput(\n name=\"advanced_mode\",\n display_name=\"Advanced Parser\",\n value=False,\n real_time_refresh=True,\n info=(\n \"Enable advanced document processing and export with Docling for PDFs, images, and office documents. \"\n \"Note that advanced document processing can consume significant resources.\"\n ),\n # Disabled in cloud\n show=not is_astra_cloud_environment(),\n ),\n DropdownInput(\n name=\"pipeline\",\n display_name=\"Pipeline\",\n info=\"Docling pipeline to use\",\n options=[\"standard\", \"vlm\"],\n value=\"standard\",\n advanced=True,\n real_time_refresh=True,\n ),\n DropdownInput(\n name=\"ocr_engine\",\n display_name=\"OCR Engine\",\n info=\"OCR engine to use. Only available when pipeline is set to 'standard'.\",\n options=[\"None\", \"easyocr\"],\n value=\"easyocr\",\n show=False,\n advanced=True,\n ),\n StrInput(\n name=\"md_image_placeholder\",\n display_name=\"Image placeholder\",\n info=\"Specify the image placeholder for markdown exports.\",\n value=\"\",\n advanced=True,\n show=False,\n ),\n StrInput(\n name=\"md_page_break_placeholder\",\n display_name=\"Page break placeholder\",\n info=\"Add this placeholder between pages in the markdown output.\",\n value=\"\",\n advanced=True,\n show=False,\n ),\n MessageTextInput(\n name=\"doc_key\",\n display_name=\"Doc Key\",\n info=\"The key to use for the DoclingDocument column.\",\n value=\"doc\",\n advanced=True,\n show=False,\n ),\n # Deprecated input retained for backward-compatibility.\n BoolInput(\n name=\"use_multithreading\",\n display_name=\"[Deprecated] Use Multithreading\",\n advanced=True,\n value=True,\n info=\"Set 'Processing Concurrency' greater than 1 to enable multithreading.\",\n ),\n IntInput(\n name=\"concurrency_multithreading\",\n display_name=\"Processing Concurrency\",\n advanced=True,\n info=\"When multiple files are being processed, the number of files to process concurrently.\",\n value=1,\n ),\n BoolInput(\n name=\"markdown\",\n display_name=\"Markdown Export\",\n info=\"Export processed documents to Markdown format. Only available when advanced mode is enabled.\",\n value=False,\n show=False,\n ),\n ]\n\n outputs = [\n Output(display_name=\"Raw Content\", name=\"message\", method=\"load_files_message\", tool_mode=True),\n ]\n\n # ------------------------------ Tool description with file names --------------\n\n def get_tool_description(self) -> str:\n \"\"\"Return a dynamic description that includes the names of uploaded files.\n\n This helps the Agent understand which files are available to read.\n \"\"\"\n base_description = \"Loads and returns the content from uploaded files.\"\n\n # Get the list of uploaded file paths\n file_paths = getattr(self, \"path\", None)\n if not file_paths:\n return base_description\n\n # Ensure it's a list\n if not isinstance(file_paths, list):\n file_paths = [file_paths]\n\n # Extract just the file names from the paths\n file_names = []\n for fp in file_paths:\n if fp:\n name = Path(fp).name\n file_names.append(name)\n\n if file_names:\n files_str = \", \".join(file_names)\n return f\"{base_description} Available files: {files_str}. Call this tool to read these files.\"\n\n return base_description\n\n @property\n def description(self) -> str:\n \"\"\"Dynamic description property that includes uploaded file names.\"\"\"\n return self.get_tool_description()\n\n async def _get_tools(self) -> list:\n \"\"\"Override to create a tool without parameters.\n\n The Read File component should use the files already uploaded via UI,\n not accept file paths from the Agent (which wouldn't know the internal paths).\n \"\"\"\n from langchain_core.tools import StructuredTool\n from pydantic import BaseModel\n\n # Empty schema - no parameters needed\n class EmptySchema(BaseModel):\n \"\"\"No parameters required - uses pre-uploaded files.\"\"\"\n\n async def read_files_tool() -> str:\n \"\"\"Read the content of uploaded files.\"\"\"\n try:\n if getattr(self, \"advanced_mode\", False):\n # In advanced mode, use the markdown output path so that the\n # tool shares the same Docling processing as the advanced\n # outputs rather than triggering a second subprocess via\n # load_files_message.\n self.markdown = True\n result = self.load_files_markdown()\n else:\n result = self.load_files_message()\n if hasattr(result, \"get_text\"):\n return result.get_text()\n if hasattr(result, \"text\"):\n return result.text\n return str(result)\n except (FileNotFoundError, ValueError, OSError, RuntimeError) as e:\n return f\"Error reading files: {e}\"\n\n description = self.get_tool_description()\n\n tool = StructuredTool(\n name=\"load_files_message\",\n description=description,\n coroutine=read_files_tool,\n args_schema=EmptySchema,\n handle_tool_error=True,\n tags=[\"load_files_message\"],\n metadata={\n \"display_name\": \"Read File\",\n \"display_description\": description,\n },\n )\n\n return [tool]\n\n # ------------------------------ UI helpers --------------------------------------\n\n def _path_value(self, template: dict) -> list[str]:\n \"\"\"Return the list of currently selected file paths from the template.\"\"\"\n return template.get(\"path\", {}).get(\"file_path\", [])\n\n def _disable_docling_fields_in_cloud(self, build_config: dict[str, Any]) -> None:\n \"\"\"Disable all Docling-related fields in cloud environments.\"\"\"\n if \"advanced_mode\" in build_config:\n build_config[\"advanced_mode\"][\"show\"] = False\n build_config[\"advanced_mode\"][\"value\"] = False\n # Hide all Docling-related fields\n docling_fields = (\"pipeline\", \"ocr_engine\", \"doc_key\", \"md_image_placeholder\", \"md_page_break_placeholder\")\n for field in docling_fields:\n if field in build_config:\n build_config[field][\"show\"] = False\n # Also disable OCR engine specifically\n if \"ocr_engine\" in build_config:\n build_config[\"ocr_engine\"][\"value\"] = \"None\"\n\n def update_build_config(\n self,\n build_config: dict[str, Any],\n field_value: Any,\n field_name: str | None = None,\n ) -> dict[str, Any]:\n \"\"\"Show/hide Advanced Parser and related fields based on selection context.\"\"\"\n # Update storage location options dynamically based on cloud environment\n if \"storage_location\" in build_config:\n updated_options = _get_storage_location_options()\n build_config[\"storage_location\"][\"options\"] = updated_options\n\n # Handle storage location selection\n if field_name == \"storage_location\":\n # Extract selected storage location\n selected = [location[\"name\"] for location in field_value] if isinstance(field_value, list) else []\n\n # Hide all storage-specific fields first\n storage_fields = [\n \"aws_access_key_id\",\n \"aws_secret_access_key\",\n \"bucket_name\",\n \"aws_region\",\n \"s3_file_key\",\n \"service_account_key\",\n \"file_id\",\n ]\n\n for f_name in storage_fields:\n if f_name in build_config:\n build_config[f_name][\"show\"] = False\n\n # Show fields based on selected storage location\n if len(selected) == 1:\n location = selected[0]\n\n if location == \"Local\":\n # Show file upload input for local storage\n if \"path\" in build_config:\n build_config[\"path\"][\"show\"] = True\n\n elif location == \"AWS\":\n # Hide file upload input, show AWS fields\n if \"path\" in build_config:\n build_config[\"path\"][\"show\"] = False\n\n aws_fields = [\n \"aws_access_key_id\",\n \"aws_secret_access_key\",\n \"bucket_name\",\n \"aws_region\",\n \"s3_file_key\",\n ]\n for f_name in aws_fields:\n if f_name in build_config:\n build_config[f_name][\"show\"] = True\n build_config[f_name][\"advanced\"] = False\n\n elif location == \"Google Drive\":\n # Hide file upload input, show Google Drive fields\n if \"path\" in build_config:\n build_config[\"path\"][\"show\"] = False\n\n gdrive_fields = [\"service_account_key\", \"file_id\"]\n for f_name in gdrive_fields:\n if f_name in build_config:\n build_config[f_name][\"show\"] = True\n build_config[f_name][\"advanced\"] = False\n # No storage location selected - show file upload by default\n elif \"path\" in build_config:\n build_config[\"path\"][\"show\"] = True\n\n return build_config\n\n if field_name == \"path\":\n paths = self._path_value(build_config)\n\n # Disable in cloud environments\n if is_astra_cloud_environment():\n self._disable_docling_fields_in_cloud(build_config)\n else:\n # If all files can be processed by docling, do so\n allow_advanced = all(not file_path.endswith((\".csv\", \".xlsx\", \".parquet\")) for file_path in paths)\n build_config[\"advanced_mode\"][\"show\"] = allow_advanced\n if not allow_advanced:\n build_config[\"advanced_mode\"][\"value\"] = False\n docling_fields = (\n \"pipeline\",\n \"ocr_engine\",\n \"doc_key\",\n \"md_image_placeholder\",\n \"md_page_break_placeholder\",\n )\n for field in docling_fields:\n if field in build_config:\n build_config[field][\"show\"] = False\n\n # Docling Processing\n elif field_name == \"advanced_mode\":\n # Disable in cloud environments - don't show Docling fields even if advanced_mode is toggled\n if is_astra_cloud_environment():\n self._disable_docling_fields_in_cloud(build_config)\n else:\n docling_fields = (\n \"pipeline\",\n \"ocr_engine\",\n \"doc_key\",\n \"md_image_placeholder\",\n \"md_page_break_placeholder\",\n )\n for field in docling_fields:\n if field in build_config:\n build_config[field][\"show\"] = bool(field_value)\n if field == \"pipeline\":\n build_config[field][\"advanced\"] = not bool(field_value)\n\n elif field_name == \"pipeline\":\n # Disable in cloud environments - don't show OCR engine even if pipeline is changed\n if is_astra_cloud_environment():\n self._disable_docling_fields_in_cloud(build_config)\n elif field_value == \"standard\":\n build_config[\"ocr_engine\"][\"show\"] = True\n build_config[\"ocr_engine\"][\"value\"] = \"easyocr\"\n else:\n build_config[\"ocr_engine\"][\"show\"] = False\n build_config[\"ocr_engine\"][\"value\"] = \"None\"\n\n return build_config\n\n def update_outputs(self, frontend_node: dict[str, Any], field_name: str, field_value: Any) -> dict[str, Any]: # noqa: ARG002\n \"\"\"Dynamically show outputs based on file count/type and advanced mode.\"\"\"\n if field_name not in [\"path\", \"advanced_mode\", \"pipeline\"]:\n return frontend_node\n\n template = frontend_node.get(\"template\", {})\n paths = self._path_value(template)\n if not paths:\n return frontend_node\n\n frontend_node[\"outputs\"] = []\n if len(paths) == 1:\n file_path = paths[0] if field_name == \"path\" else frontend_node[\"template\"][\"path\"][\"file_path\"][0]\n if file_path.endswith((\".csv\", \".xlsx\", \".parquet\")):\n frontend_node[\"outputs\"].append(\n Output(\n display_name=\"Structured Content\",\n name=\"dataframe\",\n method=\"load_files_structured\",\n tool_mode=True,\n ),\n )\n elif file_path.endswith(\".json\"):\n frontend_node[\"outputs\"].append(\n Output(display_name=\"Structured Content\", name=\"json\", method=\"load_files_json\", tool_mode=True),\n )\n\n advanced_mode = frontend_node.get(\"template\", {}).get(\"advanced_mode\", {}).get(\"value\", False)\n if advanced_mode:\n frontend_node[\"outputs\"].append(\n Output(\n display_name=\"Structured Output\",\n name=\"advanced_dataframe\",\n method=\"load_files_dataframe\",\n tool_mode=True,\n ),\n )\n frontend_node[\"outputs\"].append(\n Output(\n display_name=\"Markdown\", name=\"advanced_markdown\", method=\"load_files_markdown\", tool_mode=True\n ),\n )\n frontend_node[\"outputs\"].append(\n Output(display_name=\"File Path\", name=\"path\", method=\"load_files_path\", tool_mode=True),\n )\n else:\n frontend_node[\"outputs\"].append(\n Output(display_name=\"Raw Content\", name=\"message\", method=\"load_files_message\", tool_mode=True),\n )\n frontend_node[\"outputs\"].append(\n Output(display_name=\"File Path\", name=\"path\", method=\"load_files_path\", tool_mode=True),\n )\n else:\n # Multiple files => DataFrame output; advanced parser disabled\n frontend_node[\"outputs\"].append(\n Output(display_name=\"Files\", name=\"dataframe\", method=\"load_files\", tool_mode=True)\n )\n\n return frontend_node\n\n # ------------------------------ Core processing ----------------------------------\n\n def _get_selected_storage_location(self) -> str:\n \"\"\"Get the selected storage location from the SortableListInput.\"\"\"\n if hasattr(self, \"storage_location\") and self.storage_location:\n if isinstance(self.storage_location, list) and len(self.storage_location) > 0:\n return self.storage_location[0].get(\"name\", \"\")\n if isinstance(self.storage_location, dict):\n return self.storage_location.get(\"name\", \"\")\n return \"Local\" # Default to Local if not specified\n\n def _validate_and_resolve_paths(self) -> list[BaseFileComponent.BaseFile]:\n \"\"\"Override to handle file_path_str input from tool mode and cloud storage.\n\n Priority:\n 1. Cloud storage (AWS/Google Drive) if selected\n 2. file_path_str (if provided by the tool call)\n 3. path (uploaded file from UI)\n \"\"\"\n storage_location = self._get_selected_storage_location()\n\n # Handle AWS S3\n if storage_location == \"AWS\":\n return self._read_from_aws_s3()\n\n # Handle Google Drive\n if storage_location == \"Google Drive\":\n return self._read_from_google_drive()\n\n # Handle Local storage\n # Check if file_path_str is provided (from tool mode)\n file_path_str = getattr(self, \"file_path_str\", None)\n if file_path_str:\n # Use the string path from tool mode\n from pathlib import Path\n\n from lfx.schema.data import Data\n\n # Use same resolution logic as BaseFileComponent (support storage paths)\n path_str = str(file_path_str)\n if parse_storage_path(path_str):\n try:\n resolved_path = Path(self.get_full_path(path_str))\n except (ValueError, AttributeError):\n resolved_path = Path(self.resolve_path(path_str))\n else:\n resolved_path = Path(self.resolve_path(path_str))\n\n if not resolved_path.exists():\n msg = f\"File or directory not found: {file_path_str}\"\n self.log(msg)\n if not self.silent_errors:\n raise ValueError(msg)\n return []\n\n data_obj = Data(data={self.SERVER_FILE_PATH_FIELDNAME: str(resolved_path)})\n return [BaseFileComponent.BaseFile(data_obj, resolved_path, delete_after_processing=False)]\n\n # Otherwise use the default implementation (uses path FileInput)\n return super()._validate_and_resolve_paths()\n\n def _read_from_aws_s3(self) -> list[BaseFileComponent.BaseFile]:\n \"\"\"Read file from AWS S3.\"\"\"\n from lfx.base.data.cloud_storage_utils import create_s3_client, validate_aws_credentials\n\n # Validate AWS credentials\n validate_aws_credentials(self)\n if not getattr(self, \"s3_file_key\", None):\n msg = \"S3 File Key is required\"\n raise ValueError(msg)\n\n # Create S3 client\n s3_client = create_s3_client(self)\n\n # Download file to temp location\n import tempfile\n\n # Get file extension from S3 key\n file_extension = Path(self.s3_file_key).suffix or \"\"\n\n with tempfile.NamedTemporaryFile(mode=\"wb\", suffix=file_extension, delete=False) as temp_file:\n temp_file_path = temp_file.name\n try:\n s3_client.download_fileobj(self.bucket_name, self.s3_file_key, temp_file)\n except Exception as e:\n # Clean up temp file on failure\n with contextlib.suppress(OSError):\n Path(temp_file_path).unlink()\n msg = f\"Failed to download file from S3: {e}\"\n raise RuntimeError(msg) from e\n\n # Create BaseFile object\n from lfx.schema.data import Data\n\n temp_path = Path(temp_file_path)\n data_obj = Data(data={self.SERVER_FILE_PATH_FIELDNAME: str(temp_path)})\n return [BaseFileComponent.BaseFile(data_obj, temp_path, delete_after_processing=True)]\n\n def _read_from_google_drive(self) -> list[BaseFileComponent.BaseFile]:\n \"\"\"Read file from Google Drive.\"\"\"\n import tempfile\n\n from googleapiclient.http import MediaIoBaseDownload\n\n from lfx.base.data.cloud_storage_utils import create_google_drive_service\n\n # Validate Google Drive credentials\n if not getattr(self, \"service_account_key\", None):\n msg = \"GCP Credentials Secret Key is required for Google Drive storage\"\n raise ValueError(msg)\n if not getattr(self, \"file_id\", None):\n msg = \"Google Drive File ID is required\"\n raise ValueError(msg)\n\n # Create Google Drive service with read-only scope\n drive_service = create_google_drive_service(\n self.service_account_key, scopes=[\"https://www.googleapis.com/auth/drive.readonly\"]\n )\n\n # Get file metadata to determine file name and extension\n try:\n file_metadata = drive_service.files().get(fileId=self.file_id, fields=\"name,mimeType\").execute()\n file_name = file_metadata.get(\"name\", \"download\")\n except Exception as e:\n msg = (\n f\"Unable to access file with ID '{self.file_id}'. \"\n f\"Error: {e!s}. \"\n \"Please ensure: 1) The file ID is correct, 2) The file exists, \"\n \"3) The service account has been granted access to this file.\"\n )\n raise ValueError(msg) from e\n\n # Download file to temp location\n file_extension = Path(file_name).suffix or \"\"\n with tempfile.NamedTemporaryFile(mode=\"wb\", suffix=file_extension, delete=False) as temp_file:\n temp_file_path = temp_file.name\n try:\n request = drive_service.files().get_media(fileId=self.file_id)\n downloader = MediaIoBaseDownload(temp_file, request)\n done = False\n while not done:\n _status, done = downloader.next_chunk()\n except Exception as e:\n # Clean up temp file on failure\n with contextlib.suppress(OSError):\n Path(temp_file_path).unlink()\n msg = f\"Failed to download file from Google Drive: {e}\"\n raise RuntimeError(msg) from e\n\n # Create BaseFile object\n from lfx.schema.data import Data\n\n temp_path = Path(temp_file_path)\n data_obj = Data(data={self.SERVER_FILE_PATH_FIELDNAME: str(temp_path)})\n return [BaseFileComponent.BaseFile(data_obj, temp_path, delete_after_processing=True)]\n\n def _is_docling_compatible(self, file_path: str) -> bool:\n \"\"\"Lightweight extension gate for Docling-compatible types.\"\"\"\n docling_exts = (\n \".adoc\",\n \".asciidoc\",\n \".asc\",\n \".bmp\",\n \".csv\",\n \".dotx\",\n \".dotm\",\n \".docm\",\n \".docx\",\n \".htm\",\n \".html\",\n \".jpg\",\n \".jpeg\",\n \".json\",\n \".md\",\n \".pdf\",\n \".png\",\n \".potx\",\n \".ppsx\",\n \".pptm\",\n \".potm\",\n \".ppsm\",\n \".pptx\",\n \".tiff\",\n \".txt\",\n \".xls\",\n \".xlsx\",\n \".xhtml\",\n \".xml\",\n \".webp\",\n )\n return file_path.lower().endswith(docling_exts)\n\n async def _get_local_file_for_docling(self, file_path: str) -> tuple[str, bool]:\n \"\"\"Get a local file path for Docling processing, downloading from S3 if needed.\n\n Args:\n file_path: Either a local path or S3 key (format \"flow_id/filename\")\n\n Returns:\n tuple[str, bool]: (local_path, should_delete) where should_delete indicates\n if this is a temporary file that should be cleaned up\n \"\"\"\n settings = get_settings_service().settings\n if settings.storage_type == \"local\":\n return file_path, False\n\n # S3 storage - download to temp file\n parsed = parse_storage_path(file_path)\n if not parsed:\n msg = f\"Invalid S3 path format: {file_path}. Expected 'flow_id/filename'\"\n raise ValueError(msg)\n\n storage_service = get_storage_service()\n flow_id, filename = parsed\n\n # Get file content from S3\n content = await storage_service.get_file(flow_id, filename)\n\n suffix = Path(filename).suffix\n with NamedTemporaryFile(mode=\"wb\", suffix=suffix, delete=False) as tmp_file:\n tmp_file.write(content)\n temp_path = tmp_file.name\n\n return temp_path, True\n\n def _process_docling_in_subprocess(self, file_path: str) -> Data | None:\n \"\"\"Run Docling in a separate OS process and map the result to a Data object.\n\n We avoid multiprocessing pickling by launching `python -c \"