Repository navigation
Expand file tree
/
Copy pathTestCaseRAG.json
More file actions
6396 lines (6396 loc) · 373 KB
/
Copy pathTestCaseRAG.json
File metadata and controls
6396 lines (6396 loc) · 373 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
{
"data": {
"edges": [
{
"animated": false,
"className": "",
"data": {
"sourceHandle": {
"dataType": "ChatInput",
"id": "ChatInput-zyuwr",
"name": "message",
"output_types": [
"Message"
]
},
"targetHandle": {
"fieldName": "question",
"id": "Prompt Template-mXsRg",
"inputTypes": [
"Message"
],
"type": "str"
}
},
"id": "xy-edge__ChatInput-zyuwr{œdataTypeœ:œChatInputœ,œidœ:œChatInput-zyuwrœ,œnameœ:œmessageœ,œoutput_typesœ:[œMessageœ]}-Prompt Template-mXsRg{œfieldNameœ:œquestionœ,œidœ:œPrompt Template-mXsRgœ,œinputTypesœ:[œMessageœ],œtypeœ:œstrœ}",
"selected": false,
"source": "ChatInput-zyuwr",
"sourceHandle": "{œdataTypeœ:œChatInputœ,œidœ:œChatInput-zyuwrœ,œnameœ:œmessageœ,œoutput_typesœ:[œMessageœ]}",
"target": "Prompt Template-mXsRg",
"targetHandle": "{œfieldNameœ:œquestionœ,œidœ:œPrompt Template-mXsRgœ,œinputTypesœ:[œMessageœ],œtypeœ:œstrœ}"
},
{
"animated": false,
"className": "",
"data": {
"sourceHandle": {
"dataType": "Prompt Template",
"id": "Prompt Template-mXsRg",
"name": "prompt",
"output_types": [
"Message"
]
},
"targetHandle": {
"fieldName": "input_value",
"id": "GroqModel-0PHjX",
"inputTypes": [
"Message"
],
"type": "str"
}
},
"id": "xy-edge__Prompt Template-mXsRg{œdataTypeœ:œPrompt Templateœ,œidœ:œPrompt Template-mXsRgœ,œnameœ:œpromptœ,œoutput_typesœ:[œMessageœ]}-GroqModel-0PHjX{œfieldNameœ:œinput_valueœ,œidœ:œGroqModel-0PHjXœ,œinputTypesœ:[œMessageœ],œtypeœ:œstrœ}",
"selected": false,
"source": "Prompt Template-mXsRg",
"sourceHandle": "{œdataTypeœ:œPrompt Templateœ,œidœ:œPrompt Template-mXsRgœ,œnameœ:œpromptœ,œoutput_typesœ:[œMessageœ]}",
"target": "GroqModel-0PHjX",
"targetHandle": "{œfieldNameœ:œinput_valueœ,œidœ:œGroqModel-0PHjXœ,œinputTypesœ:[œMessageœ],œtypeœ:œstrœ}"
},
{
"animated": false,
"className": "",
"data": {
"sourceHandle": {
"dataType": "GroqModel",
"id": "GroqModel-0PHjX",
"name": "text_output",
"output_types": [
"Message"
]
},
"targetHandle": {
"fieldName": "input_value",
"id": "ChatOutput-x3pnm",
"inputTypes": [
"Data",
"JSON",
"DataFrame",
"Table",
"Message"
],
"type": "other"
}
},
"id": "xy-edge__GroqModel-0PHjX{œdataTypeœ:œGroqModelœ,œidœ:œGroqModel-0PHjXœ,œnameœ:œtext_outputœ,œoutput_typesœ:[œMessageœ]}-ChatOutput-x3pnm{œfieldNameœ:œinput_valueœ,œidœ:œChatOutput-x3pnmœ,œinputTypesœ:[œDataœ,œJSONœ,œDataFrameœ,œTableœ,œMessageœ],œtypeœ:œotherœ}",
"selected": false,
"source": "GroqModel-0PHjX",
"sourceHandle": "{œdataTypeœ:œGroqModelœ,œidœ:œGroqModel-0PHjXœ,œnameœ:œtext_outputœ,œoutput_typesœ:[œMessageœ]}",
"target": "ChatOutput-x3pnm",
"targetHandle": "{œfieldNameœ:œinput_valueœ,œidœ:œChatOutput-x3pnmœ,œinputTypesœ:[œDataœ,œJSONœ,œDataFrameœ,œTableœ,œMessageœ],œtypeœ:œotherœ}"
},
{
"animated": false,
"className": "",
"data": {
"sourceHandle": {
"dataType": "MistalAIEmbeddings",
"id": "MistalAIEmbeddings-5sNGH",
"name": "embeddings",
"output_types": [
"Embeddings"
]
},
"targetHandle": {
"fieldName": "embedding",
"id": "Chroma-kIjFf",
"inputTypes": [
"Embeddings"
],
"type": "other"
}
},
"id": "xy-edge__MistalAIEmbeddings-5sNGH{œdataTypeœ:œMistalAIEmbeddingsœ,œidœ:œMistalAIEmbeddings-5sNGHœ,œnameœ:œembeddingsœ,œoutput_typesœ:[œEmbeddingsœ]}-Chroma-kIjFf{œfieldNameœ:œembeddingœ,œidœ:œChroma-kIjFfœ,œinputTypesœ:[œEmbeddingsœ],œtypeœ:œotherœ}",
"selected": false,
"source": "MistalAIEmbeddings-5sNGH",
"sourceHandle": "{œdataTypeœ:œMistalAIEmbeddingsœ,œidœ:œMistalAIEmbeddings-5sNGHœ,œnameœ:œembeddingsœ,œoutput_typesœ:[œEmbeddingsœ]}",
"target": "Chroma-kIjFf",
"targetHandle": "{œfieldNameœ:œembeddingœ,œidœ:œChroma-kIjFfœ,œinputTypesœ:[œEmbeddingsœ],œtypeœ:œotherœ}"
},
{
"animated": false,
"className": "",
"data": {
"sourceHandle": {
"dataType": "ChatInput",
"id": "ChatInput-zyuwr",
"name": "message",
"output_types": [
"Message"
]
},
"targetHandle": {
"fieldName": "question",
"id": "Prompt Template-VdP7e",
"inputTypes": [
"Message"
],
"type": "str"
}
},
"id": "xy-edge__ChatInput-zyuwr{œdataTypeœ:œChatInputœ,œidœ:œChatInput-zyuwrœ,œnameœ:œmessageœ,œoutput_typesœ:[œMessageœ]}-Prompt Template-VdP7e{œfieldNameœ:œquestionœ,œidœ:œPrompt Template-VdP7eœ,œinputTypesœ:[œMessageœ],œtypeœ:œstrœ}",
"selected": false,
"source": "ChatInput-zyuwr",
"sourceHandle": "{œdataTypeœ:œChatInputœ,œidœ:œChatInput-zyuwrœ,œnameœ:œmessageœ,œoutput_typesœ:[œMessageœ]}",
"target": "Prompt Template-VdP7e",
"targetHandle": "{œfieldNameœ:œquestionœ,œidœ:œPrompt Template-VdP7eœ,œinputTypesœ:[œMessageœ],œtypeœ:œstrœ}"
},
{
"animated": false,
"className": "",
"data": {
"sourceHandle": {
"dataType": "Prompt Template",
"id": "Prompt Template-VdP7e",
"name": "prompt",
"output_types": [
"Message"
]
},
"targetHandle": {
"fieldName": "input_value",
"id": "GroqModel-jIWNH",
"inputTypes": [
"Message"
],
"type": "str"
}
},
"id": "xy-edge__Prompt Template-VdP7e{œdataTypeœ:œPrompt Templateœ,œidœ:œPrompt Template-VdP7eœ,œnameœ:œpromptœ,œoutput_typesœ:[œMessageœ]}-GroqModel-jIWNH{œfieldNameœ:œinput_valueœ,œidœ:œGroqModel-jIWNHœ,œinputTypesœ:[œMessageœ],œtypeœ:œstrœ}",
"selected": false,
"source": "Prompt Template-VdP7e",
"sourceHandle": "{œdataTypeœ:œPrompt Templateœ,œidœ:œPrompt Template-VdP7eœ,œnameœ:œpromptœ,œoutput_typesœ:[œMessageœ]}",
"target": "GroqModel-jIWNH",
"targetHandle": "{œfieldNameœ:œinput_valueœ,œidœ:œGroqModel-jIWNHœ,œinputTypesœ:[œMessageœ],œtypeœ:œstrœ}"
},
{
"animated": false,
"className": "",
"data": {
"sourceHandle": {
"dataType": "Chroma",
"id": "Chroma-kIjFf",
"name": "search_results",
"output_types": [
"JSON"
]
},
"targetHandle": {
"fieldName": "search_results",
"id": "NvidiaRerankComponent-uqxM7",
"inputTypes": [
"Data",
"JSON"
],
"type": "other"
}
},
"id": "xy-edge__Chroma-kIjFf{œdataTypeœ:œChromaœ,œidœ:œChroma-kIjFfœ,œnameœ:œsearch_resultsœ,œoutput_typesœ:[œJSONœ]}-NvidiaRerankComponent-uqxM7{œfieldNameœ:œsearch_resultsœ,œidœ:œNvidiaRerankComponent-uqxM7œ,œinputTypesœ:[œDataœ,œJSONœ],œtypeœ:œotherœ}",
"selected": false,
"source": "Chroma-kIjFf",
"sourceHandle": "{œdataTypeœ:œChromaœ,œidœ:œChroma-kIjFfœ,œnameœ:œsearch_resultsœ,œoutput_typesœ:[œJSONœ]}",
"target": "NvidiaRerankComponent-uqxM7",
"targetHandle": "{œfieldNameœ:œsearch_resultsœ,œidœ:œNvidiaRerankComponent-uqxM7œ,œinputTypesœ:[œDataœ,œJSONœ],œtypeœ:œotherœ}"
},
{
"animated": false,
"className": "",
"data": {
"sourceHandle": {
"dataType": "GroqModel",
"id": "GroqModel-jIWNH",
"name": "text_output",
"output_types": [
"Message"
]
},
"targetHandle": {
"fieldName": "search_query",
"id": "Chroma-kIjFf",
"inputTypes": [
"Message"
],
"type": "query"
}
},
"id": "xy-edge__GroqModel-jIWNH{œdataTypeœ:œGroqModelœ,œidœ:œGroqModel-jIWNHœ,œnameœ:œtext_outputœ,œoutput_typesœ:[œMessageœ]}-Chroma-kIjFf{œfieldNameœ:œsearch_queryœ,œidœ:œChroma-kIjFfœ,œinputTypesœ:[œMessageœ],œtypeœ:œqueryœ}",
"selected": false,
"source": "GroqModel-jIWNH",
"sourceHandle": "{œdataTypeœ:œGroqModelœ,œidœ:œGroqModel-jIWNHœ,œnameœ:œtext_outputœ,œoutput_typesœ:[œMessageœ]}",
"target": "Chroma-kIjFf",
"targetHandle": "{œfieldNameœ:œsearch_queryœ,œidœ:œChroma-kIjFfœ,œinputTypesœ:[œMessageœ],œtypeœ:œqueryœ}"
},
{
"animated": false,
"className": "",
"data": {
"sourceHandle": {
"dataType": "GroqModel",
"id": "GroqModel-jIWNH",
"name": "text_output",
"output_types": [
"Message"
]
},
"targetHandle": {
"fieldName": "search_query",
"id": "NvidiaRerankComponent-uqxM7",
"inputTypes": [
"Message"
],
"type": "str"
}
},
"id": "xy-edge__GroqModel-jIWNH{œdataTypeœ:œGroqModelœ,œidœ:œGroqModel-jIWNHœ,œnameœ:œtext_outputœ,œoutput_typesœ:[œMessageœ]}-NvidiaRerankComponent-uqxM7{œfieldNameœ:œsearch_queryœ,œidœ:œNvidiaRerankComponent-uqxM7œ,œinputTypesœ:[œMessageœ],œtypeœ:œstrœ}",
"selected": false,
"source": "GroqModel-jIWNH",
"sourceHandle": "{œdataTypeœ:œGroqModelœ,œidœ:œGroqModel-jIWNHœ,œnameœ:œtext_outputœ,œoutput_typesœ:[œMessageœ]}",
"target": "NvidiaRerankComponent-uqxM7",
"targetHandle": "{œfieldNameœ:œsearch_queryœ,œidœ:œNvidiaRerankComponent-uqxM7œ,œinputTypesœ:[œMessageœ],œtypeœ:œstrœ}"
},
{
"animated": false,
"className": "",
"data": {
"sourceHandle": {
"dataType": "ChatInput",
"id": "ChatInput-zyuwr",
"name": "message",
"output_types": [
"Message"
]
},
"targetHandle": {
"fieldName": "question",
"id": "Prompt Template-Mr61R",
"inputTypes": [
"Message"
],
"type": "str"
}
},
"id": "xy-edge__ChatInput-zyuwr{œdataTypeœ:œChatInputœ,œidœ:œChatInput-zyuwrœ,œnameœ:œmessageœ,œoutput_typesœ:[œMessageœ]}-Prompt Template-Mr61R{œfieldNameœ:œquestionœ,œidœ:œPrompt Template-Mr61Rœ,œinputTypesœ:[œMessageœ],œtypeœ:œstrœ}",
"selected": false,
"source": "ChatInput-zyuwr",
"sourceHandle": "{œdataTypeœ:œChatInputœ,œidœ:œChatInput-zyuwrœ,œnameœ:œmessageœ,œoutput_typesœ:[œMessageœ]}",
"target": "Prompt Template-Mr61R",
"targetHandle": "{œfieldNameœ:œquestionœ,œidœ:œPrompt Template-Mr61Rœ,œinputTypesœ:[œMessageœ],œtypeœ:œstrœ}"
},
{
"animated": false,
"className": "",
"data": {
"sourceHandle": {
"dataType": "Prompt Template",
"id": "Prompt Template-Mr61R",
"name": "prompt",
"output_types": [
"Message"
]
},
"targetHandle": {
"fieldName": "input_value",
"id": "GroqModel-sXqJQ",
"inputTypes": [
"Message"
],
"type": "str"
}
},
"id": "xy-edge__Prompt Template-Mr61R{œdataTypeœ:œPrompt Templateœ,œidœ:œPrompt Template-Mr61Rœ,œnameœ:œpromptœ,œoutput_typesœ:[œMessageœ]}-GroqModel-sXqJQ{œfieldNameœ:œinput_valueœ,œidœ:œGroqModel-sXqJQœ,œinputTypesœ:[œMessageœ],œtypeœ:œstrœ}",
"selected": false,
"source": "Prompt Template-Mr61R",
"sourceHandle": "{œdataTypeœ:œPrompt Templateœ,œidœ:œPrompt Template-Mr61Rœ,œnameœ:œpromptœ,œoutput_typesœ:[œMessageœ]}",
"target": "GroqModel-sXqJQ",
"targetHandle": "{œfieldNameœ:œinput_valueœ,œidœ:œGroqModel-sXqJQœ,œinputTypesœ:[œMessageœ],œtypeœ:œstrœ}"
},
{
"animated": false,
"className": "",
"data": {
"sourceHandle": {
"dataType": "GroqModel",
"id": "GroqModel-sXqJQ",
"name": "text_output",
"output_types": [
"Message"
]
},
"targetHandle": {
"fieldName": "context",
"id": "Prompt Template-mXsRg",
"inputTypes": [
"Message"
],
"type": "str"
}
},
"id": "xy-edge__GroqModel-sXqJQ{œdataTypeœ:œGroqModelœ,œidœ:œGroqModel-sXqJQœ,œnameœ:œtext_outputœ,œoutput_typesœ:[œMessageœ]}-Prompt Template-mXsRg{œfieldNameœ:œcontextœ,œidœ:œPrompt Template-mXsRgœ,œinputTypesœ:[œMessageœ],œtypeœ:œstrœ}",
"selected": false,
"source": "GroqModel-sXqJQ",
"sourceHandle": "{œdataTypeœ:œGroqModelœ,œidœ:œGroqModel-sXqJQœ,œnameœ:œtext_outputœ,œoutput_typesœ:[œMessageœ]}",
"target": "Prompt Template-mXsRg",
"targetHandle": "{œfieldNameœ:œcontextœ,œidœ:œPrompt Template-mXsRgœ,œinputTypesœ:[œMessageœ],œtypeœ:œstrœ}"
},
{
"animated": false,
"className": "",
"data": {
"sourceHandle": {
"dataType": "MistalAIEmbeddings",
"id": "MistalAIEmbeddings-INuCD",
"name": "embeddings",
"output_types": [
"Embeddings"
]
},
"targetHandle": {
"fieldName": "embedding",
"id": "Chroma-pOHNs",
"inputTypes": [
"Embeddings"
],
"type": "other"
}
},
"id": "xy-edge__MistalAIEmbeddings-INuCD{œdataTypeœ:œMistalAIEmbeddingsœ,œidœ:œMistalAIEmbeddings-INuCDœ,œnameœ:œembeddingsœ,œoutput_typesœ:[œEmbeddingsœ]}-Chroma-pOHNs{œfieldNameœ:œembeddingœ,œidœ:œChroma-pOHNsœ,œinputTypesœ:[œEmbeddingsœ],œtypeœ:œotherœ}",
"selected": false,
"source": "MistalAIEmbeddings-INuCD",
"sourceHandle": "{œdataTypeœ:œMistalAIEmbeddingsœ,œidœ:œMistalAIEmbeddings-INuCDœ,œnameœ:œembeddingsœ,œoutput_typesœ:[œEmbeddingsœ]}",
"target": "Chroma-pOHNs",
"targetHandle": "{œfieldNameœ:œembeddingœ,œidœ:œChroma-pOHNsœ,œinputTypesœ:[œEmbeddingsœ],œtypeœ:œotherœ}"
},
{
"animated": false,
"className": "",
"data": {
"sourceHandle": {
"dataType": "File",
"id": "File-z7Iu4",
"name": "dataframe",
"output_types": [
"Table"
]
},
"targetHandle": {
"fieldName": "input_table",
"id": "CustomComponent-pNiiS",
"inputTypes": [
"DataFrame",
"Table"
],
"type": "other"
}
},
"id": "xy-edge__File-z7Iu4{œdataTypeœ:œFileœ,œidœ:œFile-z7Iu4œ,œnameœ:œdataframeœ,œoutput_typesœ:[œTableœ]}-CustomComponent-pNiiS{œfieldNameœ:œinput_tableœ,œidœ:œCustomComponent-pNiiSœ,œinputTypesœ:[œDataFrameœ,œTableœ],œtypeœ:œotherœ}",
"selected": false,
"source": "File-z7Iu4",
"sourceHandle": "{œdataTypeœ:œFileœ,œidœ:œFile-z7Iu4œ,œnameœ:œdataframeœ,œoutput_typesœ:[œTableœ]}",
"target": "CustomComponent-pNiiS",
"targetHandle": "{œfieldNameœ:œinput_tableœ,œidœ:œCustomComponent-pNiiSœ,œinputTypesœ:[œDataFrameœ,œTableœ],œtypeœ:œotherœ}"
},
{
"animated": false,
"className": "",
"data": {
"sourceHandle": {
"dataType": "ParentDocumentBuilder",
"id": "CustomComponent-pNiiS",
"name": "parent_documents",
"output_types": [
"Table"
]
},
"targetHandle": {
"fieldName": "ingest_data",
"id": "Chroma-pOHNs",
"inputTypes": [
"Data",
"DataFrame",
"Table"
],
"type": "other"
}
},
"id": "xy-edge__CustomComponent-pNiiS{œdataTypeœ:œParentDocumentBuilderœ,œidœ:œCustomComponent-pNiiSœ,œnameœ:œparent_documentsœ,œoutput_typesœ:[œTableœ]}-Chroma-pOHNs{œfieldNameœ:œingest_dataœ,œidœ:œChroma-pOHNsœ,œinputTypesœ:[œDataœ,œDataFrameœ,œTableœ],œtypeœ:œotherœ}",
"selected": false,
"source": "CustomComponent-pNiiS",
"sourceHandle": "{œdataTypeœ:œParentDocumentBuilderœ,œidœ:œCustomComponent-pNiiSœ,œnameœ:œparent_documentsœ,œoutput_typesœ:[œTableœ]}",
"target": "Chroma-pOHNs",
"targetHandle": "{œfieldNameœ:œingest_dataœ,œidœ:œChroma-pOHNsœ,œinputTypesœ:[œDataœ,œDataFrameœ,œTableœ],œtypeœ:œotherœ}"
},
{
"animated": false,
"className": "",
"data": {
"sourceHandle": {
"dataType": "File",
"id": "File-z7Iu4",
"name": "dataframe",
"output_types": [
"Table"
]
},
"targetHandle": {
"fieldName": "input_table",
"id": "CustomComponent-G7JdC",
"inputTypes": [
"DataFrame",
"Table"
],
"type": "other"
}
},
"id": "xy-edge__File-z7Iu4{œdataTypeœ:œFileœ,œidœ:œFile-z7Iu4œ,œnameœ:œdataframeœ,œoutput_typesœ:[œTableœ]}-CustomComponent-G7JdC{œfieldNameœ:œinput_tableœ,œidœ:œCustomComponent-G7JdCœ,œinputTypesœ:[œDataFrameœ,œTableœ],œtypeœ:œotherœ}",
"selected": false,
"source": "File-z7Iu4",
"sourceHandle": "{œdataTypeœ:œFileœ,œidœ:œFile-z7Iu4œ,œnameœ:œdataframeœ,œoutput_typesœ:[œTableœ]}",
"target": "CustomComponent-G7JdC",
"targetHandle": "{œfieldNameœ:œinput_tableœ,œidœ:œCustomComponent-G7JdCœ,œinputTypesœ:[œDataFrameœ,œTableœ],œtypeœ:œotherœ}"
},
{
"animated": false,
"className": "",
"data": {
"sourceHandle": {
"dataType": "ChildChunkBuilder",
"id": "CustomComponent-G7JdC",
"name": "child_chunks",
"output_types": [
"Table"
]
},
"targetHandle": {
"fieldName": "ingest_data",
"id": "Chroma-snVLG",
"inputTypes": [
"Data",
"DataFrame",
"Table"
],
"type": "other"
}
},
"id": "xy-edge__CustomComponent-G7JdC{œdataTypeœ:œChildChunkBuilderœ,œidœ:œCustomComponent-G7JdCœ,œnameœ:œchild_chunksœ,œoutput_typesœ:[œTableœ]}-Chroma-snVLG{œfieldNameœ:œingest_dataœ,œidœ:œChroma-snVLGœ,œinputTypesœ:[œDataœ,œDataFrameœ,œTableœ],œtypeœ:œotherœ}",
"selected": false,
"source": "CustomComponent-G7JdC",
"sourceHandle": "{œdataTypeœ:œChildChunkBuilderœ,œidœ:œCustomComponent-G7JdCœ,œnameœ:œchild_chunksœ,œoutput_typesœ:[œTableœ]}",
"target": "Chroma-snVLG",
"targetHandle": "{œfieldNameœ:œingest_dataœ,œidœ:œChroma-snVLGœ,œinputTypesœ:[œDataœ,œDataFrameœ,œTableœ],œtypeœ:œotherœ}"
},
{
"animated": false,
"className": "",
"data": {
"sourceHandle": {
"dataType": "MistalAIEmbeddings",
"id": "MistalAIEmbeddings-INuCD",
"name": "embeddings",
"output_types": [
"Embeddings"
]
},
"targetHandle": {
"fieldName": "embedding",
"id": "Chroma-snVLG",
"inputTypes": [
"Embeddings"
],
"type": "other"
}
},
"id": "xy-edge__MistalAIEmbeddings-INuCD{œdataTypeœ:œMistalAIEmbeddingsœ,œidœ:œMistalAIEmbeddings-INuCDœ,œnameœ:œembeddingsœ,œoutput_typesœ:[œEmbeddingsœ]}-Chroma-snVLG{œfieldNameœ:œembeddingœ,œidœ:œChroma-snVLGœ,œinputTypesœ:[œEmbeddingsœ],œtypeœ:œotherœ}",
"selected": false,
"source": "MistalAIEmbeddings-INuCD",
"sourceHandle": "{œdataTypeœ:œMistalAIEmbeddingsœ,œidœ:œMistalAIEmbeddings-INuCDœ,œnameœ:œembeddingsœ,œoutput_typesœ:[œEmbeddingsœ]}",
"target": "Chroma-snVLG",
"targetHandle": "{œfieldNameœ:œembeddingœ,œidœ:œChroma-snVLGœ,œinputTypesœ:[œEmbeddingsœ],œtypeœ:œotherœ}"
},
{
"animated": false,
"className": "",
"data": {
"sourceHandle": {
"dataType": "NvidiaRerankComponent",
"id": "NvidiaRerankComponent-uqxM7",
"name": "reranked_documents",
"output_types": [
"JSON"
]
},
"targetHandle": {
"fieldName": "reranked_documents",
"id": "CustomComponent-T9X3u",
"inputTypes": [
"Data",
"JSON"
],
"type": "other"
}
},
"id": "xy-edge__NvidiaRerankComponent-uqxM7{œdataTypeœ:œNvidiaRerankComponentœ,œidœ:œNvidiaRerankComponent-uqxM7œ,œnameœ:œreranked_documentsœ,œoutput_typesœ:[œJSONœ]}-CustomComponent-T9X3u{œfieldNameœ:œreranked_documentsœ,œidœ:œCustomComponent-T9X3uœ,œinputTypesœ:[œDataœ,œJSONœ],œtypeœ:œotherœ}",
"selected": false,
"source": "NvidiaRerankComponent-uqxM7",
"sourceHandle": "{œdataTypeœ:œNvidiaRerankComponentœ,œidœ:œNvidiaRerankComponent-uqxM7œ,œnameœ:œreranked_documentsœ,œoutput_typesœ:[œJSONœ]}",
"target": "CustomComponent-T9X3u",
"targetHandle": "{œfieldNameœ:œreranked_documentsœ,œidœ:œCustomComponent-T9X3uœ,œinputTypesœ:[œDataœ,œJSONœ],œtypeœ:œotherœ}"
},
{
"animated": false,
"className": "",
"data": {
"sourceHandle": {
"dataType": "ParentDocumentResolver",
"id": "CustomComponent-T9X3u",
"name": "parent_context",
"output_types": [
"Message"
]
},
"targetHandle": {
"fieldName": "context",
"id": "Prompt Template-Mr61R",
"inputTypes": [
"Message"
],
"type": "str"
}
},
"id": "xy-edge__CustomComponent-T9X3u{œdataTypeœ:œParentDocumentResolverœ,œidœ:œCustomComponent-T9X3uœ,œnameœ:œparent_contextœ,œoutput_typesœ:[œMessageœ]}-Prompt Template-Mr61R{œfieldNameœ:œcontextœ,œidœ:œPrompt Template-Mr61Rœ,œinputTypesœ:[œMessageœ],œtypeœ:œstrœ}",
"selected": false,
"source": "CustomComponent-T9X3u",
"sourceHandle": "{œdataTypeœ:œParentDocumentResolverœ,œidœ:œCustomComponent-T9X3uœ,œnameœ:œparent_contextœ,œoutput_typesœ:[œMessageœ]}",
"target": "Prompt Template-Mr61R",
"targetHandle": "{œfieldNameœ:œcontextœ,œidœ:œPrompt Template-Mr61Rœ,œinputTypesœ:[œMessageœ],œtypeœ:œstrœ}"
}
],
"nodes": [
{
"data": {
"id": "File-z7Iu4",
"node": {
"base_classes": [
"Message"
],
"beta": false,
"category": "files_and_knowledge",
"conditional_paths": [],
"custom_fields": {},
"description": "Loads and returns the content from uploaded files.",
"display_name": "Read File",
"documentation": "https://docs.langflow.org/read-file",
"edited": false,
"field_order": [
"storage_location",
"path",
"file_path",
"separator",
"silent_errors",
"delete_server_file_after_processing",
"ignore_unsupported_extensions",
"ignore_unspecified_files",
"file_path_str",
"aws_access_key_id",
"aws_secret_access_key",
"bucket_name",
"aws_region",
"s3_file_key",
"service_account_key",
"file_id",
"advanced_mode",
"pipeline",
"ocr_engine",
"md_image_placeholder",
"md_page_break_placeholder",
"doc_key",
"use_multithreading",
"concurrency_multithreading",
"markdown"
],
"frozen": false,
"icon": "file-text",
"key": "File",
"last_updated": "2026-08-03T14:38:32.622Z",
"legacy": false,
"lf_version": "1.10.2",
"metadata": {
"code_hash": "f497bdbc749b",
"dependencies": {
"dependencies": [
{
"name": "lfx",
"version": null
},
{
"name": "langchain_core",
"version": "1.4.8"
},
{
"name": "pydantic",
"version": "2.13.4"
},
{
"name": "googleapiclient",
"version": "2.198.0"
}
],
"total_dependencies": 4
},
"module": "lfx.components.files_and_knowledge.file.FileComponent"
},
"minimized": false,
"output_types": [],
"outputs": [
{
"allows_loop": false,
"cache": true,
"display_name": "Structured Content",
"group_outputs": false,
"hidden": null,
"loop_types": null,
"method": "load_files_structured",
"name": "dataframe",
"options": null,
"required_inputs": null,
"selected": "Table",
"tool_mode": true,
"types": [
"Table"
],
"value": "__UNDEFINED__"
},
{
"allows_loop": false,
"cache": true,
"display_name": "Raw Content",
"group_outputs": false,
"loop_types": null,
"method": "load_files_message",
"name": "message",
"options": null,
"required_inputs": null,
"selected": "Message",
"tool_mode": true,
"types": [
"Message"
],
"value": "__UNDEFINED__"
},
{
"allows_loop": false,
"cache": true,
"display_name": "File Path",
"group_outputs": false,
"hidden": null,
"loop_types": null,
"method": "load_files_path",
"name": "path",
"options": null,
"required_inputs": null,
"selected": "Message",
"tool_mode": true,
"types": [
"Message"
],
"value": "__UNDEFINED__"
}
],
"pinned": false,
"score": 8.569061098350962e-12,
"template": {
"_frontend_node_flow_id": {
"value": "4a55b0e2-3a24-408b-93ca-5e9bb8ddb62d"
},
"_frontend_node_folder_id": {
"value": "dd3fbcc4-db86-4fc7-8b67-96cdd77a285f"
},
"_type": "Component",
"advanced_mode": {
"_input_type": "BoolInput",
"advanced": false,
"display_name": "Advanced Parser",
"dynamic": false,
"info": "Enable advanced document processing and export with Docling for PDFs, images, and office documents. Note that advanced document processing can consume significant resources.",
"list": false,
"list_add_label": "Add More",
"name": "advanced_mode",
"override_skip": false,
"placeholder": "",
"real_time_refresh": true,
"required": false,
"show": false,
"title_case": false,
"tool_mode": false,
"trace_as_metadata": true,
"track_in_telemetry": true,
"type": "bool",
"value": false
},
"aws_access_key_id": {
"_input_type": "SecretStrInput",
"advanced": false,
"display_name": "AWS Access Key ID",
"dynamic": false,
"info": "AWS Access key ID.",
"input_types": [],
"load_from_db": false,
"name": "aws_access_key_id",
"override_skip": false,
"password": true,
"placeholder": "",
"required": true,
"show": false,
"title_case": false,
"track_in_telemetry": false,
"type": "str",
"value": ""
},
"aws_region": {
"_input_type": "StrInput",
"advanced": false,
"display_name": "AWS Region",
"dynamic": false,
"info": "AWS region (e.g., us-east-1, eu-west-1).",
"list": false,
"list_add_label": "Add More",
"load_from_db": false,
"name": "aws_region",
"override_skip": false,
"placeholder": "",
"required": false,
"show": false,
"title_case": false,
"tool_mode": false,
"trace_as_metadata": true,
"track_in_telemetry": false,
"type": "str",
"value": ""
},
"aws_secret_access_key": {
"_input_type": "SecretStrInput",
"advanced": false,
"display_name": "AWS Secret Key",
"dynamic": false,
"info": "AWS Secret Key.",
"input_types": [],
"load_from_db": false,
"name": "aws_secret_access_key",
"override_skip": false,
"password": true,
"placeholder": "",
"required": true,
"show": false,
"title_case": false,
"track_in_telemetry": false,
"type": "str",
"value": ""
},
"bucket_name": {
"_input_type": "StrInput",
"advanced": false,
"display_name": "S3 Bucket Name",
"dynamic": false,
"info": "Enter the name of the S3 bucket.",
"list": false,
"list_add_label": "Add More",
"load_from_db": false,
"name": "bucket_name",
"override_skip": false,
"placeholder": "",
"required": true,
"show": false,
"title_case": false,
"tool_mode": false,
"trace_as_metadata": true,
"track_in_telemetry": false,
"type": "str",
"value": ""
},
"code": {
"advanced": true,
"dynamic": true,
"fileTypes": [],
"file_path": "",
"info": "",
"list": false,
"load_from_db": false,
"multiline": true,
"name": "code",
"password": false,
"placeholder": "",
"required": true,
"show": true,
"title_case": false,
"type": "code",
"value": "\"\"\"Enhanced file component with Docling support and process isolation.\n\nNotes:\n-----\n- ALL Docling parsing/export runs in a separate OS process to prevent memory\n growth and native library state from impacting the main Langflow process.\n- Standard text/structured parsing continues to use existing BaseFileComponent\n utilities (and optional threading via `parallel_load_data`).\n\"\"\"\n\nfrom __future__ import annotations\n\nimport contextlib\nimport json\nimport subprocess\nimport sys\nimport textwrap\nimport time\nfrom copy import deepcopy\nfrom pathlib import Path\nfrom tempfile import NamedTemporaryFile\nfrom typing import Any\n\nfrom lfx.base.data.base_file import BaseFileComponent\nfrom lfx.base.data.storage_utils import parse_storage_path, read_file_bytes, validate_image_content_type\nfrom lfx.base.data.utils import TEXT_FILE_TYPES, parallel_load_data, parse_text_file_to_data\nfrom lfx.inputs import SortableListInput\nfrom lfx.inputs.inputs import DropdownInput, MessageTextInput, StrInput\nfrom lfx.io import BoolInput, FileInput, IntInput, Output, SecretStrInput\nfrom lfx.schema.data import Data\nfrom lfx.schema.dataframe import DataFrame # noqa: TC001\nfrom lfx.schema.message import Message\nfrom lfx.services.deps import get_settings_service, get_storage_service\nfrom lfx.utils.async_helpers import run_until_complete\nfrom lfx.utils.validate_cloud import is_astra_cloud_environment\n\n\ndef _get_storage_location_options():\n \"\"\"Get storage location options, filtering out Local if in Astra cloud environment.\"\"\"\n all_options = [{\"name\": \"AWS\", \"icon\": \"Amazon\"}, {\"name\": \"Google Drive\", \"icon\": \"google\"}]\n if is_astra_cloud_environment():\n return all_options\n return [{\"name\": \"Local\", \"icon\": \"hard-drive\"}, *all_options]\n\n\nclass FileComponent(BaseFileComponent):\n \"\"\"File component with optional Docling processing (isolated in a subprocess).\"\"\"\n\n display_name = \"Read File\"\n # description is now a dynamic property - see get_tool_description()\n _base_description = \"Loads and returns the content from uploaded files.\"\n documentation: str = \"https://docs.langflow.org/read-file\"\n icon = \"file-text\"\n name = \"File\"\n add_tool_output = True # Enable tool mode toggle without requiring tool_mode inputs\n\n # Extensions that can be processed without Docling (using standard text parsing)\n TEXT_EXTENSIONS = TEXT_FILE_TYPES\n\n # Extensions that require Docling for processing (images, advanced office formats, etc.)\n DOCLING_ONLY_EXTENSIONS = [\n \"adoc\",\n \"asciidoc\",\n \"asc\",\n \"bmp\",\n \"dotx\",\n \"dotm\",\n \"docm\",\n \"jpg\",\n \"jpeg\",\n \"png\",\n \"potx\",\n \"ppsx\",\n \"pptm\",\n \"potm\",\n \"ppsm\",\n \"pptx\",\n \"tiff\",\n \"xls\",\n \"xlsx\",\n \"xhtml\",\n \"webp\",\n ]\n\n # Docling-supported/compatible extensions; TEXT_FILE_TYPES are supported by the base loader.\n VALID_EXTENSIONS = [\n *TEXT_EXTENSIONS,\n *DOCLING_ONLY_EXTENSIONS,\n ]\n\n # Fixed export settings used when markdown export is requested.\n EXPORT_FORMAT = \"Markdown\"\n IMAGE_MODE = \"placeholder\"\n\n _base_inputs = deepcopy(BaseFileComponent.get_base_inputs())\n\n for input_item in _base_inputs:\n if isinstance(input_item, FileInput) and input_item.name == \"path\":\n input_item.real_time_refresh = True\n input_item.tool_mode = False # Disable tool mode for file upload input\n input_item.required = False # Make it optional so it doesn't error in tool mode\n break\n\n inputs = [\n SortableListInput(\n name=\"storage_location\",\n display_name=\"Storage Location\",\n placeholder=\"Select Location\",\n info=\"Choose where to read the file from.\",\n options=_get_storage_location_options(),\n real_time_refresh=True,\n limit=1,\n value=[{\"name\": \"Local\", \"icon\": \"hard-drive\"}],\n advanced=True,\n ),\n *_base_inputs,\n StrInput(\n name=\"file_path_str\",\n display_name=\"File Path\",\n info=(\n \"Path to the file to read. Used when component is called as a tool. \"\n \"If not provided, will use the uploaded file from 'path' input.\"\n ),\n show=False,\n advanced=True,\n tool_mode=True, # Required for Toolset toggle, but _get_tools() ignores this parameter\n required=False,\n ),\n # AWS S3 specific inputs\n SecretStrInput(\n name=\"aws_access_key_id\",\n display_name=\"AWS Access Key ID\",\n info=\"AWS Access key ID.\",\n show=False,\n advanced=False,\n required=True,\n ),\n SecretStrInput(\n name=\"aws_secret_access_key\",\n display_name=\"AWS Secret Key\",\n info=\"AWS Secret Key.\",\n show=False,\n advanced=False,\n required=True,\n ),\n StrInput(\n name=\"bucket_name\",\n display_name=\"S3 Bucket Name\",\n info=\"Enter the name of the S3 bucket.\",\n show=False,\n advanced=False,\n required=True,\n ),\n StrInput(\n name=\"aws_region\",\n display_name=\"AWS Region\",\n info=\"AWS region (e.g., us-east-1, eu-west-1).\",\n show=False,\n advanced=False,\n ),\n StrInput(\n name=\"s3_file_key\",\n display_name=\"S3 File Key\",\n info=\"The key (path) of the file in S3 bucket.\",\n show=False,\n advanced=False,\n required=True,\n ),\n # Google Drive specific inputs\n SecretStrInput(\n name=\"service_account_key\",\n display_name=\"GCP Credentials Secret Key\",\n info=\"Your Google Cloud Platform service account JSON key as a secret string (complete JSON content).\",\n show=False,\n advanced=False,\n required=True,\n ),\n StrInput(\n name=\"file_id\",\n display_name=\"Google Drive File ID\",\n info=(\"The Google Drive file ID to read. The file must be shared with the service account email.\"),\n show=False,\n advanced=False,\n required=True,\n ),\n BoolInput(\n name=\"advanced_mode\",\n display_name=\"Advanced Parser\",\n value=False,\n real_time_refresh=True,\n info=(\n \"Enable advanced document processing and export with Docling for PDFs, images, and office documents. \"\n \"Note that advanced document processing can consume significant resources.\"\n ),\n # Disabled in cloud\n show=not is_astra_cloud_environment(),\n ),\n DropdownInput(\n name=\"pipeline\",\n display_name=\"Pipeline\",\n info=\"Docling pipeline to use\",\n options=[\"standard\", \"vlm\"],\n value=\"standard\",\n advanced=True,\n real_time_refresh=True,\n ),\n DropdownInput(\n name=\"ocr_engine\",\n display_name=\"OCR Engine\",\n info=\"OCR engine to use. Only available when pipeline is set to 'standard'.\",\n options=[\"None\", \"easyocr\"],\n value=\"easyocr\",\n show=False,\n advanced=True,\n ),\n StrInput(\n name=\"md_image_placeholder\",\n display_name=\"Image placeholder\",\n info=\"Specify the image placeholder for markdown exports.\",\n value=\"<!-- image -->\",\n advanced=True,\n show=False,\n ),\n StrInput(\n name=\"md_page_break_placeholder\",\n display_name=\"Page break placeholder\",\n info=\"Add this placeholder between pages in the markdown output.\",\n value=\"\",\n advanced=True,\n show=False,\n ),\n MessageTextInput(\n name=\"doc_key\",\n display_name=\"Doc Key\",\n info=\"The key to use for the DoclingDocument column.\",\n value=\"doc\",\n advanced=True,\n show=False,\n ),\n # Deprecated input retained for backward-compatibility.\n BoolInput(\n name=\"use_multithreading\",\n display_name=\"[Deprecated] Use Multithreading\",\n advanced=True,\n value=True,\n info=\"Set 'Processing Concurrency' greater than 1 to enable multithreading.\",\n ),\n IntInput(\n name=\"concurrency_multithreading\",\n display_name=\"Processing Concurrency\",\n advanced=True,\n info=\"When multiple files are being processed, the number of files to process concurrently.\",\n value=1,\n ),\n BoolInput(\n name=\"markdown\",\n display_name=\"Markdown Export\",\n info=\"Export processed documents to Markdown format. Only available when advanced mode is enabled.\",\n value=False,\n show=False,\n ),\n ]\n\n outputs = [\n Output(display_name=\"Raw Content\", name=\"message\", method=\"load_files_message\", tool_mode=True),\n ]\n\n # ------------------------------ Tool description with file names --------------\n\n def get_tool_description(self) -> str:\n \"\"\"Return a dynamic description that includes the names of uploaded files.\n\n This helps the Agent understand which files are available to read.\n \"\"\"\n base_description = type(self)._base_description # noqa: SLF001\n\n # Get the list of uploaded file paths\n file_paths = getattr(self, \"path\", None)\n if not file_paths:\n return base_description\n\n # Ensure it's a list\n if not isinstance(file_paths, list):\n file_paths = [file_paths]\n\n # Extract just the file names from the paths\n file_names = []\n for fp in file_paths:\n if fp:\n name = Path(fp).name\n file_names.append(name)\n\n if file_names:\n files_str = \", \".join(file_names)\n return f\"{base_description} Available files: {files_str}. Call this tool to read these files.\"\n\n return base_description\n\n @property\n def description(self) -> str:\n \"\"\"Dynamic description property that includes uploaded file names.\"\"\"\n return self.get_tool_description()\n\n async def _get_tools(self) -> list:\n \"\"\"Override to create a tool without parameters.\n\n The Read File component should use the files already uploaded via UI,\n not accept file paths from the Agent (which wouldn't know the internal paths).\n \"\"\"\n from langchain_core.tools import StructuredTool\n from pydantic import BaseModel\n\n # Empty schema - no parameters needed\n class EmptySchema(BaseModel):\n \"\"\"No parameters required - uses pre-uploaded files.\"\"\"\n\n async def read_files_tool() -> str:\n \"\"\"Read the content of uploaded files.\"\"\"\n try:\n if getattr(self, \"advanced_mode\", False):\n # In advanced mode, use the markdown output path so that the\n # tool shares the same Docling processing as the advanced\n # outputs rather than triggering a second subprocess via\n # load_files_message.\n self.markdown = True\n result = self.load_files_markdown()\n else:\n result = self.load_files_message()\n if hasattr(result, \"get_text\"):\n return result.get_text()\n if hasattr(result, \"text\"):\n return result.text\n return str(result)\n except (FileNotFoundError, ValueError, OSError, RuntimeError) as e:\n return f\"Error reading files: {e}\"\n\n description = self.get_tool_description()\n\n tool = StructuredTool(\n name=\"load_files_message\",\n description=description,\n coroutine=read_files_tool,\n args_schema=EmptySchema,\n handle_tool_error=True,\n tags=[\"load_files_message\"],\n metadata={\n \"display_name\": \"Read File\",\n \"display_description\": description,\n },\n )\n\n return [tool]\n\n # ------------------------------ UI helpers --------------------------------------\n\n def _path_value(self, template: dict) -> list[str]:\n \"\"\"Return the list of currently selected file paths from the template.\"\"\"\n return template.get(\"path\", {}).get(\"file_path\", [])\n\n def _disable_docling_fields_in_cloud(self, build_config: dict[str, Any]) -> None:\n \"\"\"Disable all Docling-related fields in cloud environments.\"\"\"\n if \"advanced_mode\" in build_config:\n build_config[\"advanced_mode\"][\"show\"] = False\n build_config[\"advanced_mode\"][\"value\"] = False\n # Hide all Docling-related fields\n docling_fields = (\"pipeline\", \"ocr_engine\", \"doc_key\", \"md_image_placeholder\", \"md_page_break_placeholder\")\n for field in docling_fields:\n if field in build_config:\n build_config[field][\"show\"] = False\n # Also disable OCR engine specifically\n if \"ocr_engine\" in build_config:\n build_config[\"ocr_engine\"][\"value\"] = \"None\"\n\n def update_build_config(\n self,\n build_config: dict[str, Any],\n field_value: Any,\n field_name: str | None = None,\n ) -> dict[str, Any]:\n \"\"\"Show/hide Advanced Parser and related fields based on selection context.\"\"\"\n # Update storage location options dynamically based on cloud environment\n if \"storage_location\" in build_config:\n updated_options = _get_storage_location_options()\n build_config[\"storage_location\"][\"options\"] = updated_options\n\n # Handle storage location selection\n if field_name == \"storage_location\":\n # Extract selected storage location\n selected = [location[\"name\"] for location in field_value] if isinstance(field_value, list) else []\n\n # Hide all storage-specific fields first\n storage_fields = [\n \"aws_access_key_id\",\n \"aws_secret_access_key\",\n \"bucket_name\",\n \"aws_region\",\n \"s3_file_key\",\n \"service_account_key\",\n \"file_id\",\n ]\n\n for f_name in storage_fields:\n if f_name in build_config:\n build_config[f_name][\"show\"] = False\n\n # Show fields based on selected storage location\n if len(selected) == 1:\n location = selected[0]\n\n if location == \"Local\":\n # Show file upload input for local storage\n if \"path\" in build_config:\n build_config[\"path\"][\"show\"] = True\n\n elif location == \"AWS\":\n # Hide file upload input, show AWS fields\n if \"path\" in build_config:\n build_config[\"path\"][\"show\"] = False\n\n aws_fields = [\n \"aws_access_key_id\",\n \"aws_secret_access_key\",\n \"bucket_name\",\n \"aws_region\",\n \"s3_file_key\",\n ]\n for f_name in aws_fields:\n if f_name in build_config:\n build_config[f_name][\"show\"] = True\n build_config[f_name][\"advanced\"] = False\n\n elif location == \"Google Drive\":\n # Hide file upload input, show Google Drive fields\n if \"path\" in build_config:\n build_config[\"path\"][\"show\"] = False\n\n gdrive_fields = [\"service_account_key\", \"file_id\"]\n for f_name in gdrive_fields:\n if f_name in build_config:\n build_config[f_name][\"show\"] = True\n build_config[f_name][\"advanced\"] = False\n # No storage location selected - show file upload by default\n elif \"path\" in build_config:\n build_config[\"path\"][\"show\"] = True\n\n return build_config\n\n if field_name == \"path\":\n paths = self._path_value(build_config)\n\n # Disable in cloud environments\n if is_astra_cloud_environment():\n self._disable_docling_fields_in_cloud(build_config)\n else:\n # If all files can be processed by docling, do so\n allow_advanced = all(not file_path.endswith((\".csv\", \".xlsx\", \".parquet\")) for file_path in paths)\n build_config[\"advanced_mode\"][\"show\"] = allow_advanced\n if not allow_advanced:\n build_config[\"advanced_mode\"][\"value\"] = False\n docling_fields = (\n \"pipeline\",\n \"ocr_engine\",\n \"doc_key\",\n \"md_image_placeholder\",\n \"md_page_break_placeholder\",\n )\n for field in docling_fields:\n if field in build_config:\n build_config[field][\"show\"] = False\n\n # Docling Processing\n elif field_name == \"advanced_mode\":\n # Disable in cloud environments - don't show Docling fields even if advanced_mode is toggled\n if is_astra_cloud_environment():\n self._disable_docling_fields_in_cloud(build_config)\n else:\n docling_fields = (\n \"pipeline\",\n \"ocr_engine\",\n \"doc_key\",\n \"md_image_placeholder\",\n \"md_page_break_placeholder\",\n )\n for field in docling_fields:\n if field in build_config:\n build_config[field][\"show\"] = bool(field_value)\n if field == \"pipeline\":\n build_config[field][\"advanced\"] = not bool(field_value)\n\n elif field_name == \"pipeline\":\n # Disable in cloud environments - don't show OCR engine even if pipeline is changed\n if is_astra_cloud_environment():\n self._disable_docling_fields_in_cloud(build_config)\n elif field_value == \"standard\":\n build_config[\"ocr_engine\"][\"show\"] = True\n build_config[\"ocr_engine\"][\"value\"] = \"easyocr\"\n else:\n build_config[\"ocr_engine\"][\"show\"] = False\n build_config[\"ocr_engine\"][\"value\"] = \"None\"\n\n return build_config\n\n def update_outputs(self, frontend_node: dict[str, Any], field_name: str, field_value: Any) -> dict[str, Any]: # noqa: ARG002\n \"\"\"Dynamically show outputs based on file count/type and advanced mode.\"\"\"\n if field_name not in [\"path\", \"advanced_mode\", \"pipeline\"]:\n return frontend_node\n\n template = frontend_node.get(\"template\", {})\n paths = self._path_value(template)\n if not paths:\n return frontend_node\n\n frontend_node[\"outputs\"] = []\n if len(paths) == 1:\n file_path = paths[0] if field_name == \"path\" else frontend_node[\"template\"][\"path\"][\"file_path\"][0]\n if file_path.endswith((\".csv\", \".xlsx\", \".parquet\")):\n frontend_node[\"outputs\"].append(\n Output(\n display_name=\"Structured Content\",\n name=\"dataframe\",\n method=\"load_files_structured\",\n tool_mode=True,\n ),\n )\n elif file_path.endswith(\".json\"):\n frontend_node[\"outputs\"].append(\n Output(display_name=\"Structured Content\", name=\"json\", method=\"load_files_json\", tool_mode=True),\n )\n\n advanced_mode = frontend_node.get(\"template\", {}).get(\"advanced_mode\", {}).get(\"value\", False)\n if advanced_mode:\n frontend_node[\"outputs\"].append(\n Output(\n display_name=\"Structured Output\",\n name=\"advanced_dataframe\",\n method=\"load_files_dataframe\",\n tool_mode=True,\n ),\n )\n frontend_node[\"outputs\"].append(\n Output(\n display_name=\"Markdown\", name=\"advanced_markdown\", method=\"load_files_markdown\", tool_mode=True\n ),\n )\n frontend_node[\"outputs\"].append(\n Output(display_name=\"File Path\", name=\"path\", method=\"load_files_path\", tool_mode=True),\n )\n else:\n frontend_node[\"outputs\"].append(\n Output(display_name=\"Raw Content\", name=\"message\", method=\"load_files_message\", tool_mode=True),\n )\n frontend_node[\"outputs\"].append(\n Output(display_name=\"File Path\", name=\"path\", method=\"load_files_path\", tool_mode=True),\n )\n else:\n # Multiple files => DataFrame output; advanced parser disabled\n frontend_node[\"outputs\"].append(\n Output(display_name=\"Files\", name=\"dataframe\", method=\"load_files\", tool_mode=True)\n )\n\n return frontend_node\n\n # ------------------------------ Core processing ----------------------------------\n\n def _get_selected_storage_location(self) -> str:\n \"\"\"Get the selected storage location from the SortableListInput.\"\"\"\n if hasattr(self, \"storage_location\") and self.storage_location:\n if isinstance(self.storage_location, list) and len(self.storage_location) > 0:\n return self.storage_location[0].get(\"name\", \"\")\n if isinstance(self.storage_location, dict):\n return self.storage_location.get(\"name\", \"\")\n return \"Local\" # Default to Local if not specified\n\n def _validate_and_resolve_paths(self) -> list[BaseFileComponent.BaseFile]:\n \"\"\"Override to handle file_path_str input from tool mode and cloud storage.\n\n Priority:\n 1. Cloud storage (AWS/Google Drive) if selected\n 2. file_path_str (if provided by the tool call)\n 3. path (uploaded file from UI)\n \"\"\"\n storage_location = self._get_selected_storage_location()\n\n # Handle AWS S3\n if storage_location == \"AWS\":\n return self._read_from_aws_s3()\n\n # Handle Google Drive\n if storage_location == \"Google Drive\":\n return self._read_from_google_drive()\n\n # Handle Local storage\n # Check if file_path_str is provided (from tool mode)\n file_path_str = getattr(self, \"file_path_str\", None)\n if file_path_str:\n # Use the string path from tool mode\n from pathlib import Path\n\n from lfx.schema.data import Data\n\n # Use same resolution logic as BaseFileComponent (support storage paths)\n path_str = str(file_path_str)\n if parse_storage_path(path_str):\n try:\n resolved_path = Path(self.get_full_path(path_str))\n except (ValueError, AttributeError):\n resolved_path = Path(self.resolve_path(path_str))\n else:\n resolved_path = Path(self.resolve_path(path_str))\n\n if not resolved_path.exists():\n msg = f\"File or directory not found: {file_path_str}\"\n self.log(msg)\n if not self.silent_errors:\n raise ValueError(msg)\n return []\n\n data_obj = Data(data={self.SERVER_FILE_PATH_FIELDNAME: str(resolved_path)})\n return [BaseFileComponent.BaseFile(data_obj, resolved_path, delete_after_processing=False)]\n\n # Otherwise use the default implementation (uses path FileInput)\n return super()._validate_and_resolve_paths()\n\n def _read_from_aws_s3(self) -> list[BaseFileComponent.BaseFile]:\n \"\"\"Read file from AWS S3.\"\"\"\n from lfx.base.data.cloud_storage_utils import create_s3_client, validate_aws_credentials\n\n # Validate AWS credentials\n validate_aws_credentials(self)\n if not getattr(self, \"s3_file_key\", None):\n msg = \"S3 File Key is required\"\n raise ValueError(msg)\n\n # Create S3 client\n s3_client = create_s3_client(self)\n\n # Download file to temp location\n import tempfile\n\n # Get file extension from S3 key\n file_extension = Path(self.s3_file_key).suffix or \"\"\n\n with tempfile.NamedTemporaryFile(mode=\"wb\", suffix=file_extension, delete=False) as temp_file:\n temp_file_path = temp_file.name\n try:\n s3_client.download_fileobj(self.bucket_name, self.s3_file_key, temp_file)\n except Exception as e:\n # Clean up temp file on failure\n with contextlib.suppress(OSError):\n Path(temp_file_path).unlink()\n msg = f\"Failed to download file from S3: {e}\"\n raise RuntimeError(msg) from e\n\n # Create BaseFile object\n from lfx.schema.data import Data\n\n temp_path = Path(temp_file_path)\n data_obj = Data(data={self.SERVER_FILE_PATH_FIELDNAME: str(temp_path)})\n return [BaseFileComponent.BaseFile(data_obj, temp_path, delete_after_processing=True)]\n\n def _read_from_google_drive(self) -> list[BaseFileComponent.BaseFile]:\n \"\"\"Read file from Google Drive.\"\"\"\n import tempfile\n\n from googleapiclient.http import MediaIoBaseDownload\n\n from lfx.base.data.cloud_storage_utils import create_google_drive_service\n\n # Validate Google Drive credentials\n if not getattr(self, \"service_account_key\", None):\n msg = \"GCP Credentials Secret Key is required for Google Drive storage\"\n raise ValueError(msg)\n if not getattr(self, \"file_id\", None):\n msg = \"Google Drive File ID is required\"\n raise ValueError(msg)\n\n # Create Google Drive service with read-only scope\n drive_service = create_google_drive_service(\n self.service_account_key, scopes=[\"https://www.googleapis.com/auth/drive.readonly\"]\n )\n\n # Get file metadata to determine file name and extension\n try:\n file_metadata = drive_service.files().get(fileId=self.file_id, fields=\"name,mimeType\").execute()\n file_name = file_metadata.get(\"name\", \"download\")\n except Exception as e:\n msg = (\n f\"Unable to access file with ID '{self.file_id}'. \"\n f\"Error: {e!s}. \"\n \"Please ensure: 1) The file ID is correct, 2) The file exists, \"\n \"3) The service account has been granted access to this file.\"\n )\n raise ValueError(msg) from e\n\n # Download file to temp location\n file_extension = Path(file_name).suffix or \"\"\n with tempfile.NamedTemporaryFile(mode=\"wb\", suffix=file_extension, delete=False) as temp_file:\n temp_file_path = temp_file.name\n try:\n request = drive_service.files().get_media(fileId=self.file_id)\n downloader = MediaIoBaseDownload(temp_file, request)\n done = False\n while not done:\n _status, done = downloader.next_chunk()\n except Exception as e:\n # Clean up temp file on failure\n with contextlib.suppress(OSError):\n Path(temp_file_path).unlink()\n msg = f\"Failed to download file from Google Drive: {e}\"\n raise RuntimeError(msg) from e\n\n # Create BaseFile object\n from lfx.schema.data import Data\n\n temp_path = Path(temp_file_path)\n data_obj = Data(data={self.SERVER_FILE_PATH_FIELDNAME: str(temp_path)})\n return [BaseFileComponent.BaseFile(data_obj, temp_path, delete_after_processing=True)]\n\n def _is_docling_compatible(self, file_path: str) -> bool:\n \"\"\"Lightweight extension gate for Docling-compatible types.\"\"\"\n docling_exts = (\n \".adoc\",\n \".asciidoc\",\n \".asc\",\n \".bmp\",\n \".csv\",\n \".dotx\",\n \".dotm\",\n \".docm\",\n \".docx\",\n \".htm\",\n \".html\",\n \".jpg\",\n \".jpeg\",\n \".json\",\n \".md\",\n \".pdf\",\n \".png\",\n \".potx\",\n \".ppsx\",\n \".pptm\",\n \".potm\",\n \".ppsm\",\n \".pptx\",\n \".tiff\",\n \".txt\",\n \".xls\",\n \".xlsx\",\n \".xhtml\",\n \".xml\",\n \".webp\",\n )\n return file_path.lower().endswith(docling_exts)\n\n async def _get_local_file_for_docling(self, file_path: str) -> tuple[str, bool]:\n \"\"\"Get a local file path for Docling processing, downloading from S3 if needed.\n\n Args:\n file_path: Either a local path or S3 key (format \"flow_id/filename\")\n\n Returns:\n tuple[str, bool]: (local_path, should_delete) where should_delete indicates\n if this is a temporary file that should be cleaned up\n \"\"\"\n settings = get_settings_service().settings\n if settings.storage_type == \"local\":\n return file_path, False\n\n # S3 storage - download to temp file\n parsed = parse_storage_path(file_path)\n if not parsed:\n msg = f\"Invalid S3 path format: {file_path}. Expected 'flow_id/filename'\"\n raise ValueError(msg)\n\n storage_service = get_storage_service()\n flow_id, filename = parsed\n\n # Get file content from S3\n content = await storage_service.get_file(flow_id, filename)\n\n suffix = Path(filename).suffix\n with NamedTemporaryFile(mode=\"wb\", suffix=suffix, delete=False) as tmp_file:\n tmp_file.write(content)\n temp_path = tmp_file.name\n\n return temp_path, True\n\n def _process_docling_in_subprocess(self, file_path: str) -> Data | None:\n \"\"\"Run Docling in a separate OS process and map the result to a Data object.\n\n We avoid multiprocessing pickling by launching `python -c \"<script>\"` and\n passing JSON config via stdin. The child prints a JSON result to stdout.\n\n For S3 storage, the file is downloaded to a temp file first.\n \"\"\"\n if not file_path:\n return None\n\n settings = get_settings_service().settings\n if settings.storage_type == \"s3\":\n local_path, should_delete = run_until_complete(self._get_local_file_for_docling(file_path))\n else:\n local_path = file_path\n should_delete = False\n\n try:\n return self._process_docling_subprocess_impl(local_path, file_path)\n finally:\n # Clean up temp file if we created one\n if should_delete:\n with contextlib.suppress(Exception):\n Path(local_path).unlink() # Ignore cleanup errors\n\n def _process_docling_subprocess_impl(self, local_file_path: str, original_file_path: str) -> Data | None:\n \"\"\"Implementation of Docling subprocess processing.\n\n Args:\n local_file_path: Path to local file to process\n original_file_path: Original file path to include in metadata\n Returns:\n Data object with processed content\n \"\"\"\n args: dict[str, Any] = {\n \"file_path\": local_file_path,\n \"markdown\": bool(self.markdown),\n \"image_mode\": str(self.IMAGE_MODE),\n \"md_image_placeholder\": str(self.md_image_placeholder),\n \"md_page_break_placeholder\": str(self.md_page_break_placeholder),\n \"pipeline\": str(self.pipeline),\n \"ocr_engine\": (\n self.ocr_engine if self.ocr_engine and self.ocr_engine != \"None\" and self.pipeline != \"vlm\" else None\n ),\n }\n\n # Child script for isolating the docling processing\n child_script = textwrap.dedent(\n r\"\"\"\n import json, sys\n\n def try_imports():\n try:\n from docling.datamodel.base_models import ConversionStatus, InputFormat # type: ignore\n from docling.document_converter import DocumentConverter # type: ignore\n from docling_core.types.doc import ImageRefMode # type: ignore\n return ConversionStatus, InputFormat, DocumentConverter, ImageRefMode, \"latest\"\n except Exception as e:\n raise e\n\n def create_converter(strategy, input_format, DocumentConverter, pipeline, ocr_engine):\n # --- Standard PDF/IMAGE pipeline (your existing behavior), with optional OCR ---\n if pipeline == \"standard\":\n try:\n from docling.datamodel.pipeline_options import PdfPipelineOptions # type: ignore\n from docling.document_converter import PdfFormatOption # type: ignore\n\n pipe = PdfPipelineOptions()\n pipe.do_ocr = False\n\n if ocr_engine:\n try:\n from docling.models.factories import get_ocr_factory # type: ignore\n pipe.do_ocr = True\n fac = get_ocr_factory(allow_external_plugins=False)\n pipe.ocr_options = fac.create_options(kind=ocr_engine)\n except Exception:\n # If OCR setup fails, disable it\n pipe.do_ocr = False\n\n fmt = {}\n if hasattr(input_format, \"PDF\"):\n fmt[getattr(input_format, \"PDF\")] = PdfFormatOption(pipeline_options=pipe)\n if hasattr(input_format, \"IMAGE\"):\n fmt[getattr(input_format, \"IMAGE\")] = PdfFormatOption(pipeline_options=pipe)\n\n return DocumentConverter(format_options=fmt)\n except Exception:\n return DocumentConverter()\n\n # --- Vision-Language Model (VLM) pipeline ---\n if pipeline == \"vlm\":\n try:\n from docling.datamodel.pipeline_options import VlmPipelineOptions\n from docling.datamodel.vlm_model_specs import GRANITEDOCLING_MLX, GRANITEDOCLING_TRANSFORMERS\n from docling.document_converter import PdfFormatOption\n from docling.pipeline.vlm_pipeline import VlmPipeline\n\n vl_pipe = VlmPipelineOptions(\n vlm_options=GRANITEDOCLING_TRANSFORMERS,\n )\n\n if sys.platform == \"darwin\":\n try:\n import mlx_vlm\n vl_pipe.vlm_options = GRANITEDOCLING_MLX\n except ImportError as e:\n raise e\n\n # VLM paths generally don't need OCR; keep OCR off by default here.\n fmt = {}\n if hasattr(input_format, \"PDF\"):\n fmt[getattr(input_format, \"PDF\")] = PdfFormatOption(\n pipeline_cls=VlmPipeline,\n pipeline_options=vl_pipe\n )\n if hasattr(input_format, \"IMAGE\"):\n fmt[getattr(input_format, \"IMAGE\")] = PdfFormatOption(\n pipeline_cls=VlmPipeline,\n pipeline_options=vl_pipe\n )\n\n return DocumentConverter(format_options=fmt)\n except Exception as e:\n raise e\n\n # --- Fallback: default converter with no special options ---\n return DocumentConverter()\n\n def export_markdown(document, ImageRefMode, image_mode, img_ph, pg_ph):\n try:\n mode = getattr(ImageRefMode, image_mode.upper(), image_mode)\n return document.export_to_markdown(\n image_mode=mode,\n image_placeholder=img_ph,\n page_break_placeholder=pg_ph,\n )\n except Exception:\n try:\n return document.export_to_text()\n except Exception:\n return str(document)\n\n def to_rows(doc_dict):\n rows = []\n for t in doc_dict.get(\"texts\", []):\n prov = t.get(\"prov\") or []\n page_no = None\n if prov and isinstance(prov, list) and isinstance(prov[0], dict):\n page_no = prov[0].get(\"page_no\")\n rows.append({\n \"page_no\": page_no,\n \"label\": t.get(\"label\"),\n \"text\": t.get(\"text\"),\n \"level\": t.get(\"level\"),\n })\n return rows\n\n def main():\n cfg = json.loads(sys.stdin.read())\n file_path = cfg[\"file_path\"]\n markdown = cfg[\"markdown\"]\n image_mode = cfg[\"image_mode\"]\n img_ph = cfg[\"md_image_placeholder\"]\n pg_ph = cfg[\"md_page_break_placeholder\"]\n pipeline = cfg[\"pipeline\"]\n ocr_engine = cfg.get(\"ocr_engine\")\n meta = {\"file_path\": file_path}\n\n try:\n ConversionStatus, InputFormat, DocumentConverter, ImageRefMode, strategy = try_imports()\n converter = create_converter(strategy, InputFormat, DocumentConverter, pipeline, ocr_engine)\n try:\n res = converter.convert(file_path)\n except Exception as e:\n print(json.dumps({\"ok\": False, \"error\": f\"Docling conversion error: {e}\", \"meta\": meta}))\n return\n\n ok = False\n if hasattr(res, \"status\"):\n try:\n ok = (res.status == ConversionStatus.SUCCESS) or (str(res.status).lower() == \"success\")\n except Exception:\n ok = (str(res.status).lower() == \"success\")\n if not ok and hasattr(res, \"document\"):\n ok = getattr(res, \"document\", None) is not None\n if not ok:\n print(json.dumps({\"ok\": False, \"error\": \"Docling conversion failed\", \"meta\": meta}))\n return\n\n doc = getattr(res, \"document\", None)\n if doc is None:\n print(json.dumps({\"ok\": False, \"error\": \"Docling produced no document\", \"meta\": meta}))\n return\n\n # Extract DoclingDocument metadata\n if hasattr(doc, \"name\") and doc.name:\n meta[\"name\"] = doc.name\n if hasattr(doc, \"origin\") and doc.origin is not None:\n origin = doc.origin\n if hasattr(origin, \"filename\") and origin.filename:\n meta[\"filename\"] = origin.filename\n if hasattr(origin, \"binary_hash\") and origin.binary_hash:\n meta[\"document_id\"] = str(origin.binary_hash)\n if hasattr(origin, \"mimetype\") and origin.mimetype:\n meta[\"mimetype\"] = origin.mimetype\n\n if markdown:\n text = export_markdown(doc, ImageRefMode, image_mode, img_ph, pg_ph)\n print(json.dumps({\"ok\": True, \"mode\": \"markdown\", \"text\": text, \"meta\": meta}))\n return\n\n # structured\n try:\n doc_dict = doc.export_to_dict()\n except Exception as e:\n print(json.dumps({\"ok\": False, \"error\": f\"Docling export_to_dict failed: {e}\", \"meta\": meta}))\n return\n\n rows = to_rows(doc_dict)\n print(json.dumps({\"ok\": True, \"mode\": \"structured\", \"doc\": rows, \"meta\": meta}))\n except Exception as e:\n print(\n json.dumps({\n \"ok\": False,\n \"error\": f\"Docling processing error: {e}\",\n \"meta\": {\"file_path\": file_path},\n })\n )\n\n if __name__ == \"__main__\":\n main()\n \"\"\"\n )\n\n # Validate file_path to avoid command injection or unsafe input.\n # Note: $ is intentionally not blocked here because the path is passed as JSON via\n # stdin to the subprocess, not interpolated in a shell command.\n if not isinstance(args[\"file_path\"], str) or any(c in args[\"file_path\"] for c in [\";\", \"|\", \"&\", \"`\"]):\n return Data(data={\"error\": \"Unsafe file path detected.\", \"file_path\": args[\"file_path\"]})\n\n # Use Popen with a polling loop instead of blocking subprocess.run().\n # This lets us emit periodic log messages that keep the SSE event stream\n # alive in multi-worker (Gunicorn) deployments, preventing the job queue\n # from being cleaned up while Docling is still processing.\n docling_timeout = 600 # 10 minutes; large PDFs with OCR may need this\n poll_interval = 5 # seconds between progress heartbeats\n\n proc = subprocess.Popen( # noqa: S603\n [sys.executable, \"-u\", \"-c\", child_script],\n stdin=subprocess.PIPE,\n stdout=subprocess.PIPE,\n stderr=subprocess.PIPE,\n )\n # Send input and close stdin so child can proceed\n proc.stdin.write(json.dumps(args).encode(\"utf-8\"))\n proc.stdin.close()\n\n start = time.monotonic()\n while proc.poll() is None:\n elapsed = time.monotonic() - start\n if elapsed >= docling_timeout:\n proc.kill()\n proc.wait()\n return Data(\n data={\n \"error\": (\n f\"Docling processing timed out after {docling_timeout}s. \"\n \"Consider using the standalone Docling component for large documents.\"\n ),\n \"file_path\": original_file_path,\n },\n )\n # Heartbeat: emit a log so the graph event stream stays active\n self.log(f\"Docling processing in progress ({int(elapsed)}s elapsed)...\")\n time.sleep(poll_interval)\n\n stdout_bytes = proc.stdout.read()\n stderr_bytes = proc.stderr.read()\n proc.stdout.close()\n proc.stderr.close()\n\n if not stdout_bytes:\n err_msg = stderr_bytes.decode(\"utf-8\", errors=\"replace\") if stderr_bytes else \"no output from child process\"\n return Data(data={\"error\": f\"Docling subprocess error: {err_msg}\", \"file_path\": original_file_path})\n\n try:\n result = json.loads(stdout_bytes.decode(\"utf-8\"))\n except Exception as e: # noqa: BLE001\n err_msg = stderr_bytes.decode(\"utf-8\", errors=\"replace\")\n return Data(\n data={\n \"error\": f\"Invalid JSON from Docling subprocess: {e}. stderr={err_msg}\",\n \"file_path\": original_file_path,\n },\n )\n\n if not result.get(\"ok\"):\n error_msg = result.get(\"error\", \"Unknown Docling error\")\n # Override meta file_path with original_file_path to ensure correct path matching\n meta = result.get(\"meta\", {})\n meta[\"file_path\"] = original_file_path\n return Data(data={\"error\": error_msg, **meta})\n\n meta = result.get(\"meta\", {})\n # Override meta file_path with original_file_path to ensure correct path matching\n # The subprocess returns the temp file path, but we need the original S3/local path for rollup_data\n meta[\"file_path\"] = original_file_path\n if result.get(\"mode\") == \"markdown\":\n exported_content = str(result.get(\"text\", \"\"))\n return Data(\n text=exported_content,\n data={\"exported_content\": exported_content, \"export_format\": self.EXPORT_FORMAT, **meta},\n )\n\n rows = list(result.get(\"doc\", []))\n return Data(data={\"doc\": rows, \"export_format\": self.EXPORT_FORMAT, **meta})\n\n def process_files(\n self,\n file_list: list[BaseFileComponent.BaseFile],\n ) -> list[BaseFileComponent.BaseFile]:\n \"\"\"Process input files.\n\n - advanced_mode => Docling in a separate process.\n - Otherwise => standard parsing in current process (optionally threaded).\n \"\"\"\n if not file_list:\n msg = \"No files to process.\"\n raise ValueError(msg)\n\n # Validate image files to detect content/extension mismatches\n # This prevents API errors like \"Image does not match the provided media type\"\n image_extensions = {\"jpeg\", \"jpg\", \"png\", \"gif\", \"webp\", \"bmp\", \"tiff\"}\n settings = get_settings_service().settings\n for file in file_list:\n extension = file.path.suffix[1:].lower()\n if extension in image_extensions:\n # Read bytes based on storage type\n try:\n if settings.storage_type == \"s3\":\n # For S3 storage, use storage service to read file bytes\n file_path_str = str(file.path)\n content = run_until_complete(read_file_bytes(file_path_str))\n else:\n # For local storage, read bytes directly from filesystem\n content = file.path.read_bytes()\n\n is_valid, error_msg = validate_image_content_type(\n str(file.path),\n content=content,\n )\n if not is_valid:\n self.log(error_msg)\n if not self.silent_errors:\n raise ValueError(error_msg)\n except (OSError, FileNotFoundError) as e:\n self.log(f\"Could not read file for validation: {e}\")\n # Continue - let it fail later with better error\n\n # Validate that files requiring Docling are only processed when advanced mode is enabled\n if not self.advanced_mode:\n for file in file_list:\n extension = file.path.suffix[1:].lower()\n if extension in self.DOCLING_ONLY_EXTENSIONS:\n if is_astra_cloud_environment():\n msg = (\n f\"File '{file.path.name}' has extension '.{extension}' which requires \"\n f\"Advanced Parser mode. Advanced Parser is not available in cloud environments.\"\n )\n else:\n msg = (\n f\"File '{file.path.name}' has extension '.{extension}' which requires \"\n f\"Advanced Parser mode. Please enable 'Advanced Parser' to process this file.\"\n )\n self.log(msg)\n raise ValueError(msg)\n\n def process_file_standard(file_path: str, *, silent_errors: bool = False) -> Data | None:\n try:\n return parse_text_file_to_data(file_path, silent_errors=silent_errors)\n except FileNotFoundError as e:\n self.log(f\"File not found: {file_path}. Error: {e}\")\n if not silent_errors:\n raise\n return None\n except Exception as e:\n self.log(f\"Unexpected error processing {file_path}: {e}\")\n if not silent_errors:\n raise\n return None\n\n docling_compatible = all(self._is_docling_compatible(str(f.path)) for f in file_list)\n\n # Advanced path: Check if ALL files are compatible with Docling\n if self.advanced_mode and docling_compatible:\n final_return: list[BaseFileComponent.BaseFile] = []\n for file in file_list:\n file_path = str(file.path)\n advanced_data: Data | None = self._process_docling_in_subprocess(file_path)\n\n # Handle None case - Docling processing failed or returned None\n if advanced_data is None:\n error_data = Data(\n data={\n \"file_path\": file_path,\n \"error\": \"Docling processing returned no result. Check logs for details.\",\n },\n )\n final_return.extend(self.rollup_data([file], [error_data]))\n continue\n\n # --- UNNEST: expand each element in `doc` to its own Data row\n payload = getattr(advanced_data, \"data\", {}) or {}\n\n # Check for errors first\n if \"error\" in payload:\n error_msg = payload.get(\"error\", \"Unknown error\")\n error_data = Data(\n data={\n \"file_path\": file_path,\n \"error\": error_msg,\n **{k: v for k, v in payload.items() if k not in (\"error\", \"file_path\")},\n },\n )\n final_return.extend(self.rollup_data([file], [error_data]))\n continue\n\n doc_rows = payload.get(\"doc\")\n if isinstance(doc_rows, list) and doc_rows:\n # Non-empty list of structured rows\n rows: list[Data | None] = [\n Data(\n data={\n \"file_path\": file_path,\n **(item if isinstance(item, dict) else {\"value\": item}),\n },\n )\n for item in doc_rows\n ]\n final_return.extend(self.rollup_data([file], rows))\n elif isinstance(doc_rows, list) and not doc_rows:\n # Empty list - file was processed but no text content found\n # Create a Data object indicating no content was extracted\n self.log(f\"No text extracted from '{file_path}', creating placeholder data\")\n empty_data = Data(\n data={\n \"file_path\": file_path,\n \"text\": \"(No text content extracted from image)\",\n \"info\": \"Image processed successfully but contained no extractable text\",\n **{k: v for k, v in payload.items() if k != \"doc\"},\n },\n )\n final_return.extend(self.rollup_data([file], [empty_data]))\n else:\n # If not structured, keep as-is (e.g., markdown export or error dict)\n # Ensure file_path is set for proper rollup matching\n if not payload.get(\"file_path\"):\n payload[\"file_path\"] = file_path\n # Create new Data with file_path\n advanced_data = Data(\n data=payload,\n text=getattr(advanced_data, \"text\", None),\n )\n final_return.extend(self.rollup_data([file], [advanced_data]))\n return final_return\n\n # Standard multi-file (or single non-advanced) path\n concurrency = max(1, self.concurrency_multithreading)\n\n file_paths = [str(f.path) for f in file_list]\n self.log(f\"Starting parallel processing of {len(file_paths)} files with concurrency: {concurrency}.\")\n my_data = parallel_load_data(\n file_paths,\n silent_errors=self.silent_errors,\n load_function=process_file_standard,\n max_concurrency=concurrency,\n )\n return self.rollup_data(file_list, my_data)\n\n # ------------------------------ Output helpers -----------------------------------\n\n def load_files_helper(self) -> DataFrame:\n result = self.load_files()\n\n # Result is a DataFrame - check if it has any rows\n if result.empty:\n msg = \"Could not extract content from the provided file(s).\"\n raise ValueError(msg)\n\n # Check for error column with error messages\n if \"error\" in result.columns:\n errors = result[\"error\"].dropna().tolist()\n if errors and not any(col in result.columns for col in [\"text\", \"doc\", \"exported_content\"]):\n raise ValueError(errors[0])\n\n return result\n\n def load_files_dataframe(self) -> DataFrame:\n \"\"\"Load files using advanced Docling processing and export to DataFrame format.\"\"\"\n self.markdown = False\n return self.load_files_helper()\n\n def load_files_markdown(self) -> Message:\n \"\"\"Load files using advanced Docling processing and export to Markdown format.\"\"\"\n self.markdown = True\n result = self.load_files_helper()\n\n # Result is a DataFrame - check for text or exported_content columns\n if \"text\" in result.columns and not result[\"text\"].isna().all():\n text_values = result[\"text\"].dropna().tolist()\n if text_values:\n return Message(text=str(text_values[0]))\n\n if \"exported_content\" in result.columns and not result[\"exported_content\"].isna().all():\n content_values = result[\"exported_content\"].dropna().tolist()\n if content_values:\n return Message(text=str(content_values[0]))\n\n # Return empty message with info that no text was found\n return Message(text=\"(No text content extracted from file)\")\n"
},
"concurrency_multithreading": {
"_input_type": "IntInput",
"advanced": true,
"display_name": "Processing Concurrency",
"dynamic": false,
"info": "When multiple files are being processed, the number of files to process concurrently.",
"list": false,
"list_add_label": "Add More",
"name": "concurrency_multithreading",
"override_skip": false,
"placeholder": "",
"required": false,
"show": true,
"title_case": false,
"tool_mode": false,
"trace_as_metadata": true,
"track_in_telemetry": true,
"type": "int",
"value": 1
},
"delete_server_file_after_processing": {
"_input_type": "BoolInput",
"advanced": true,
"display_name": "Delete Server File After Processing",
"dynamic": false,
"info": "If true, the Server File Path will be deleted after processing.",
"list": false,
"list_add_label": "Add More",
"name": "delete_server_file_after_processing",
"override_skip": false,
"placeholder": "",
"required": false,
"show": true,
"title_case": false,
"tool_mode": false,
"trace_as_metadata": true,
"track_in_telemetry": true,
"type": "bool",
"value": true
},
"doc_key": {
"_input_type": "MessageTextInput",
"advanced": true,
"display_name": "Doc Key",
"dynamic": false,
"info": "The key to use for the DoclingDocument column.",
"input_types": [
"Message"
],
"list": false,
"list_add_label": "Add More",
"load_from_db": false,
"name": "doc_key",
"override_skip": false,
"placeholder": "",
"required": false,
"show": false,
"title_case": false,
"tool_mode": false,
"trace_as_input": true,
"trace_as_metadata": true,
"track_in_telemetry": false,
"type": "str",
"value": "doc"
},
"file_id": {
"_input_type": "StrInput",
"advanced": false,
"display_name": "Google Drive File ID",
"dynamic": false,
"info": "The Google Drive file ID to read. The file must be shared with the service account email.",
"list": false,
"list_add_label": "Add More",
"load_from_db": false,
"name": "file_id",
"override_skip": false,
"placeholder": "",
"required": true,
"show": false,
"title_case": false,
"tool_mode": false,
"trace_as_metadata": true,
"track_in_telemetry": false,
"type": "str",
"value": ""
},
"file_path": {
"_input_type": "HandleInput",
"advanced": true,
"display_name": "Server File Path",
"dynamic": false,
"info": "Data object with a 'file_path' property pointing to server file or a Message object with a path to the file. Supercedes 'Path' but supports same file types.",
"input_types": [
"Data",
"JSON",
"Message"
],
"list": true,
"list_add_label": "Add More",
"name": "file_path",
"override_skip": false,
"placeholder": "",
"required": false,
"show": true,
"title_case": false,
"trace_as_metadata": true,
"track_in_telemetry": false,
"type": "other",
"value": ""
},
"file_path_str": {
"_input_type": "StrInput",
"advanced": true,
"display_name": "File Path",
"dynamic": false,
"info": "Path to the file to read. Used when component is called as a tool. If not provided, will use the uploaded file from 'path' input.",
"list": false,
"list_add_label": "Add More",
"load_from_db": false,
"name": "file_path_str",
"override_skip": false,
"placeholder": "",
"required": false,
"show": false,
"title_case": false,
"tool_mode": true,
"trace_as_metadata": true,
"track_in_telemetry": false,
"type": "str",
"value": ""
},
"ignore_unspecified_files": {
"_input_type": "BoolInput",
"advanced": true,
"display_name": "Ignore Unspecified Files",
"dynamic": false,
"info": "If true, Data with no 'file_path' property will be ignored.",
"list": false,
"list_add_label": "Add More",
"name": "ignore_unspecified_files",
"override_skip": false,
"placeholder": "",
"required": false,
"show": true,
"title_case": false,
"tool_mode": false,
"trace_as_metadata": true,
"track_in_telemetry": true,
"type": "bool",
"value": false
},
"ignore_unsupported_extensions": {
"_input_type": "BoolInput",
"advanced": true,
"display_name": "Ignore Unsupported Extensions",
"dynamic": false,
"info": "If true, files with unsupported extensions will not be processed.",
"list": false,
"list_add_label": "Add More",
"name": "ignore_unsupported_extensions",
"override_skip": false,
"placeholder": "",
"required": false,
"show": true,