-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathindexer.py
More file actions
54 lines (44 loc) · 1.6 KB
/
Copy pathindexer.py
File metadata and controls
54 lines (44 loc) · 1.6 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
from langchain_text_splitters import RecursiveCharacterTextSplitter
from langchain_community.embeddings import HuggingFaceEmbeddings
from langchain_community.vectorstores import Chroma
from langchain_core.documents import Document
import os
# Initialize embeddings once
embeddings = HuggingFaceEmbeddings(model_name="all-MiniLM-L6-v2")
def index_repository(repo_name, parsed_files):
text_splitter = RecursiveCharacterTextSplitter(
chunk_size=1000,
chunk_overlap=200,
length_function=len
)
docs = []
for file in parsed_files:
path = file["path"]
content = file["content"]
if not content.strip():
continue
chunks = text_splitter.split_text(content)
current_char = 0
for i, chunk in enumerate(chunks):
# approximate line number
start_idx = content.find(chunk, current_char)
if start_idx == -1:
start_idx = current_char
line_num = content.count('\n', 0, start_idx) + 1
current_char = start_idx
docs.append(
Document(
page_content=chunk,
metadata={"source": path, "chunk_id": i, "repo": repo_name, "line": line_num}
)
)
if not docs:
return 0
persist_directory = f"./chroma_db/{repo_name}"
vectorstore = Chroma.from_documents(
documents=docs,
embedding=embeddings,
persist_directory=persist_directory,
collection_name=repo_name
)
return len(docs)