Repository navigation
Expand file tree
/
Copy pathfile_adder.py
More file actions
86 lines (63 loc) · 2.98 KB
/
Copy pathfile_adder.py
File metadata and controls
86 lines (63 loc) · 2.98 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
from logging import NullHandler
import os
from pprint import pprint
import pandas as pd
from langchain_community.document_loaders import PyPDFLoader
from langchain_community.document_loaders.csv_loader import CSVLoader
from langchain.text_splitter import RecursiveCharacterTextSplitter
from langchain.docstore.document import Document
#Given the App in question and files, it adds said files to the embeddings model
class FileAdder:
def __init__(self, chunk=800, overlap=150):
self.stored_info = []
self.text_splitter = RecursiveCharacterTextSplitter(
chunk_size=chunk,
chunk_overlap=overlap,
length_function=len,
is_separator_regex=False)
#Given the file location (Do websites have files? Is it on a file in the server the website is running on?) adds the file to the app
#csv/excel of a certain size will break this, depends on LLM, add ability to divide into chunks
#Possible Error when location doesn't exist, need to handle it
#might be a better way to remove the new csv files
def add(self, uploadedFile):
#code = ''.join(random.choices(String.ascii_uppercase + String.ascii_lowercase))
file_name = uploadedFile.name
location = "UploadedFiles\\" + file_name
with open(location, "wb") as file:
file.write(uploadedFile.getbuffer())
try:
if(".pdf" in file_name):
loader = PyPDFLoader(location)
docs = loader.load_and_split(text_splitter=self.text_splitter)
for doc in docs:
self.stored_info.append(doc)
elif(".csv" in file_name):
loader = CSVLoader(file_path=location)
docs = loader.load()
for doc in docs:
self.stored_info.append(doc)
elif(".xls" in file_name or ".xlsx" in file_name or ".xlsm" in file_name):
newLocation = location[:location.find(".")]+".csv"
df = pd.read_excel(location)
df.to_csv(newLocation, index=True)
loader = CSVLoader(file_path=newLocation)
docs = loader.load()
for doc in docs:
self.stored_info.append(doc)
#trash
df = None
os.remove(newLocation)
else:
print("Unexpected File Type")
os.remove(location)
except PermissionError:
print("We do not have the access (" + file_name + ")")
except FileNotFoundError:
print("(" + file_name + ")" + "was not found and was not added")
except:
print("An error occured while adding (" + file_name + ")")
os.remove(location)
def get_stored(self):
return self.stored_info
def reset(self):
self.stored_info = []