Repository navigation
Expand file tree
/
Copy pathingest.py
More file actions
86 lines (68 loc) · 2.61 KB
/
Copy pathingest.py
File metadata and controls
86 lines (68 loc) · 2.61 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
import typer
from pathlib import Path
from loguru import logger
from extract.pdf_reader import pdf_to_text
from transform.segmenter import split_by_articles
from transform.chunker import chunk_text
from vector.embeddings import embed_batch
from vector.vector_store import build_store
from storage.models import Chunk
app = typer.Typer()
def ingest_pdf(path: Path, store_dir: str) -> None:
logger.info(f"Processing {path.name}")
result = pdf_to_text(str(path))
pages = result["pages"]
raw_text = result["raw_text"]
logger.info(f" Extracted {len(pages)} pages")
sections = split_by_articles(raw_text)
logger.info(f" Found {len(sections)} sections")
all_chunks: list[Chunk] = []
chunk_index = 0
for section in sections:
text_chunks = chunk_text(section["text"])
for text in text_chunks:
if not text.strip():
continue
all_chunks.append(Chunk(
chunk_index=chunk_index,
source=path.name,
page=_estimate_page(text, pages),
livre=section.get("livre"),
titre=section.get("titre"),
chapitre=section.get("chapitre"),
section=section.get("section"),
article_ref=section.get("article_ref"),
text=text,
))
chunk_index += 1
logger.info(f" Generated {len(all_chunks)} chunks")
BATCH_SIZE = 64
all_vectors: list[list[float]] = []
for i in range(0, len(all_chunks), BATCH_SIZE):
batch = [c.text for c in all_chunks[i:i + BATCH_SIZE]]
all_vectors.extend(embed_batch(batch))
logger.info(f" Embedded {min(i + BATCH_SIZE, len(all_chunks))}/{len(all_chunks)} chunks")
# 5. Persist to FAISS
build_store(all_chunks, all_vectors, store_dir)
logger.success(f"Done: {path.name}")
def _estimate_page(chunk_text: str, pages: list[dict]) -> int:
probe = " ".join(chunk_text.split()[:15])
for page in pages:
normalized = " ".join(page["text"].split())
if probe in normalized:
return page["page"]
return 1
@app.command()
def main(
pdf_dir: str = typer.Option("data", help="Directory containing PDF files"),
store_dir: str = typer.Option("vector_store", help="Output directory for FAISS index + metadata"),
):
pdf_files = list(Path(pdf_dir).glob("*.pdf"))
if not pdf_files:
logger.error(f"No PDFs found in {pdf_dir}")
raise typer.Exit(1)
logger.info(f"Found {len(pdf_files)} PDF(s) in {pdf_dir}")
for pdf in pdf_files:
ingest_pdf(pdf, store_dir)
if __name__ == "__main__":
app()