Add fallout ingest v2 - structural chunking + systeme metadata
This commit is contained in:
@@ -0,0 +1,308 @@
|
|||||||
|
"""
|
||||||
|
Ingestion Fallout PDFs → ChromaDB v2
|
||||||
|
Améliorations v2 :
|
||||||
|
- Chunking structurel (paragraphes/sections) au lieu de token-count fixe
|
||||||
|
- Métadonnées enrichies : systeme (2d20 | tardy | none), category, priority
|
||||||
|
- Évite la collision mécanique 2D20 vs Tardy en taggant chaque chunk
|
||||||
|
- Chunks 400-600 mots, overlap 80 mots, respecte les frontières de paragraphes
|
||||||
|
"""
|
||||||
|
import json, os, re, subprocess, tempfile, time, urllib.request, urllib.parse
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
PDFFETCH_URL = "http://localhost:15001/files"
|
||||||
|
OLLAMA_EMBED = "http://localhost:11434/api/embeddings"
|
||||||
|
CHROMA_URL = "http://localhost:8800"
|
||||||
|
CHROMA_BASE = "/api/v2/tenants/default_tenant/databases/default_database"
|
||||||
|
COLLECTION = "fallout_lore"
|
||||||
|
EMBED_MODEL = "nomic-embed-text"
|
||||||
|
STATE_FILE = Path("/home/ubuntu/fallout/ingest_state.json")
|
||||||
|
FALLOUT_PREFIX = "fallout/"
|
||||||
|
|
||||||
|
# Chunking params (en mots, pas tokens)
|
||||||
|
CHUNK_WORDS = 450
|
||||||
|
OVERLAP_WORDS = 80
|
||||||
|
MIN_WORDS = 40
|
||||||
|
|
||||||
|
|
||||||
|
# ── État ──────────────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
def load_state():
|
||||||
|
if STATE_FILE.exists():
|
||||||
|
return json.loads(STATE_FILE.read_text())
|
||||||
|
return {"ingested": [], "failed": [], "version": 2}
|
||||||
|
|
||||||
|
def save_state(s):
|
||||||
|
STATE_FILE.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
STATE_FILE.write_text(json.dumps(s, indent=2))
|
||||||
|
|
||||||
|
|
||||||
|
# ── ChromaDB ──────────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
def chroma_request(method, path, data=None):
|
||||||
|
url = CHROMA_URL + path
|
||||||
|
body = json.dumps(data).encode() if data else None
|
||||||
|
req = urllib.request.Request(url, data=body, method=method,
|
||||||
|
headers={"Content-Type": "application/json"})
|
||||||
|
with urllib.request.urlopen(req, timeout=30) as r:
|
||||||
|
return json.loads(r.read())
|
||||||
|
|
||||||
|
def get_or_create_collection():
|
||||||
|
try:
|
||||||
|
r = chroma_request("POST", f"{CHROMA_BASE}/collections", {
|
||||||
|
"name": COLLECTION,
|
||||||
|
"metadata": {"hnsw:space": "cosine"}
|
||||||
|
})
|
||||||
|
print(f"Collection créée : {COLLECTION} ({r['id']})")
|
||||||
|
return r["id"]
|
||||||
|
except urllib.error.HTTPError as e:
|
||||||
|
if e.code == 409:
|
||||||
|
cols = chroma_request("GET", f"{CHROMA_BASE}/collections")
|
||||||
|
col_id = next(c["id"] for c in cols if c["name"] == COLLECTION)
|
||||||
|
print(f"Collection existante : {COLLECTION} ({col_id})")
|
||||||
|
return col_id
|
||||||
|
raise
|
||||||
|
|
||||||
|
def add_documents(col_id, ids, embeddings, documents, metadatas):
|
||||||
|
chroma_request("POST", f"{CHROMA_BASE}/collections/{col_id}/add", {
|
||||||
|
"ids": ids, "embeddings": embeddings,
|
||||||
|
"documents": documents, "metadatas": metadatas
|
||||||
|
})
|
||||||
|
|
||||||
|
|
||||||
|
# ── Embeddings ────────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
def embed(text):
|
||||||
|
payload = json.dumps({"model": EMBED_MODEL, "prompt": text}).encode()
|
||||||
|
req = urllib.request.Request(OLLAMA_EMBED, data=payload,
|
||||||
|
headers={"Content-Type": "application/json"}, method="POST")
|
||||||
|
with urllib.request.urlopen(req, timeout=60) as r:
|
||||||
|
return json.loads(r.read())["embedding"]
|
||||||
|
|
||||||
|
|
||||||
|
# ── PDF helpers ───────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
def list_fallout_pdfs():
|
||||||
|
req = urllib.request.Request(PDFFETCH_URL,
|
||||||
|
headers={"User-Agent": "fallout-ingest/2.0"})
|
||||||
|
with urllib.request.urlopen(req, timeout=15) as r:
|
||||||
|
all_files = json.loads(r.read())
|
||||||
|
return [f for f in all_files if f.startswith(FALLOUT_PREFIX)]
|
||||||
|
|
||||||
|
def download_pdf(filename):
|
||||||
|
url = f"{PDFFETCH_URL}/{urllib.parse.quote(filename)}"
|
||||||
|
req = urllib.request.Request(url, headers={"User-Agent": "fallout-ingest/2.0"})
|
||||||
|
with urllib.request.urlopen(req, timeout=300) as r:
|
||||||
|
return r.read()
|
||||||
|
|
||||||
|
def extract_text(pdf_bytes):
|
||||||
|
with tempfile.NamedTemporaryFile(suffix=".pdf", delete=False) as f:
|
||||||
|
f.write(pdf_bytes); tmp = f.name
|
||||||
|
text = ""
|
||||||
|
try:
|
||||||
|
r = subprocess.run(["pdftotext", "-layout", tmp, "-"],
|
||||||
|
capture_output=True, timeout=180)
|
||||||
|
if r.returncode == 0:
|
||||||
|
text = r.stdout.decode("utf-8", errors="replace")
|
||||||
|
finally:
|
||||||
|
os.unlink(tmp)
|
||||||
|
if len(text.strip()) < 200:
|
||||||
|
try:
|
||||||
|
import io
|
||||||
|
from pypdf import PdfReader
|
||||||
|
reader = PdfReader(io.BytesIO(pdf_bytes))
|
||||||
|
text = "\n".join(p.extract_text() or "" for p in reader.pages)
|
||||||
|
except Exception as e:
|
||||||
|
raise RuntimeError(f"Extraction impossible: {e}")
|
||||||
|
return text
|
||||||
|
|
||||||
|
|
||||||
|
# ── Chunking structurel ───────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
def is_section_header(line):
|
||||||
|
"""Détecte les titres de section : ligne courte, capitalisée, pas de ponctuation finale."""
|
||||||
|
l = line.strip()
|
||||||
|
if not l or len(l) > 80:
|
||||||
|
return False
|
||||||
|
if l.isupper() and len(l.split()) <= 8:
|
||||||
|
return True
|
||||||
|
if re.match(r'^(CHAPITRE|CHAPTER|PARTIE|PART|SECTION|\d+[\.\)])\s+\S', l, re.I):
|
||||||
|
return True
|
||||||
|
return False
|
||||||
|
|
||||||
|
def chunk_by_structure(text, filename):
|
||||||
|
"""
|
||||||
|
Chunking basé sur les paragraphes et sections, pas sur le token count fixe.
|
||||||
|
Respecte les frontières naturelles du texte.
|
||||||
|
"""
|
||||||
|
# Nettoyage
|
||||||
|
text = re.sub(r'[ \t]+', ' ', text)
|
||||||
|
text = re.sub(r'\n{3,}', '\n\n', text)
|
||||||
|
|
||||||
|
# Découpe en blocs (paragraphes ou sections)
|
||||||
|
raw_blocks = re.split(r'\n{2,}', text)
|
||||||
|
|
||||||
|
# Regroupe les headers avec le paragraphe suivant
|
||||||
|
blocks = []
|
||||||
|
i = 0
|
||||||
|
while i < len(raw_blocks):
|
||||||
|
b = raw_blocks[i].strip()
|
||||||
|
if not b:
|
||||||
|
i += 1
|
||||||
|
continue
|
||||||
|
if is_section_header(b) and i + 1 < len(raw_blocks):
|
||||||
|
merged = b + "\n\n" + raw_blocks[i+1].strip()
|
||||||
|
blocks.append(merged)
|
||||||
|
i += 2
|
||||||
|
else:
|
||||||
|
blocks.append(b)
|
||||||
|
i += 1
|
||||||
|
|
||||||
|
# Accumulation en chunks de ~CHUNK_WORDS mots avec overlap
|
||||||
|
chunks = []
|
||||||
|
current_blocks = []
|
||||||
|
current_words = 0
|
||||||
|
|
||||||
|
for block in blocks:
|
||||||
|
bwords = len(block.split())
|
||||||
|
if bwords < 5:
|
||||||
|
continue
|
||||||
|
|
||||||
|
if current_words + bwords > CHUNK_WORDS and current_blocks:
|
||||||
|
chunk = "\n\n".join(current_blocks).strip()
|
||||||
|
if len(chunk.split()) >= MIN_WORDS:
|
||||||
|
chunks.append(chunk)
|
||||||
|
|
||||||
|
# Overlap : garder les derniers blocs jusqu'à OVERLAP_WORDS mots
|
||||||
|
overlap_blocks = []
|
||||||
|
overlap_words = 0
|
||||||
|
for ob in reversed(current_blocks):
|
||||||
|
ow = len(ob.split())
|
||||||
|
if overlap_words + ow <= OVERLAP_WORDS:
|
||||||
|
overlap_blocks.insert(0, ob)
|
||||||
|
overlap_words += ow
|
||||||
|
else:
|
||||||
|
break
|
||||||
|
current_blocks = overlap_blocks + [block]
|
||||||
|
current_words = overlap_words + bwords
|
||||||
|
else:
|
||||||
|
current_blocks.append(block)
|
||||||
|
current_words += bwords
|
||||||
|
|
||||||
|
# Flush final
|
||||||
|
if current_blocks:
|
||||||
|
chunk = "\n\n".join(current_blocks).strip()
|
||||||
|
if len(chunk.split()) >= MIN_WORDS:
|
||||||
|
chunks.append(chunk)
|
||||||
|
|
||||||
|
return chunks
|
||||||
|
|
||||||
|
|
||||||
|
# ── Métadonnées ───────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
def infer_metadata(filename):
|
||||||
|
f = filename.lower()
|
||||||
|
name = Path(filename).name
|
||||||
|
meta = {"source": name, "path": filename}
|
||||||
|
|
||||||
|
# Catégorie
|
||||||
|
if "lore_post" in f: meta["category"] = "lore_canon"
|
||||||
|
elif "lore_pre" in f: meta["category"] = "lore_inspiration"
|
||||||
|
elif "regles/core" in f: meta["category"] = "regles_core"
|
||||||
|
elif "regles/suppl" in f: meta["category"] = "regles_supplement"
|
||||||
|
elif "aventures" in f: meta["category"] = "aventure"
|
||||||
|
elif "ambiance" in f: meta["category"] = "ambiance"
|
||||||
|
else: meta["category"] = "divers"
|
||||||
|
|
||||||
|
# Système de règles — CRITIQUE pour éviter la collision 2D20 vs Tardy
|
||||||
|
if "tardy" in f:
|
||||||
|
meta["systeme"] = "tardy"
|
||||||
|
elif any(k in f for k in ["manuel_officiel", "quickstart", "rules_booklet",
|
||||||
|
"gm_toolkit", "bestiaire", "armes", "special",
|
||||||
|
"competences", "pizza", "frequence"]):
|
||||||
|
meta["systeme"] = "2d20"
|
||||||
|
elif "regles" in f:
|
||||||
|
meta["systeme"] = "2d20"
|
||||||
|
else:
|
||||||
|
meta["systeme"] = "none"
|
||||||
|
|
||||||
|
# Priorité lore : la bible canon prime
|
||||||
|
if "bible_v2" in f: meta["priority"] = "haute"
|
||||||
|
elif "bible_v1" in f: meta["priority"] = "normale"
|
||||||
|
elif "lore_canon" == meta.get("category"): meta["priority"] = "normale"
|
||||||
|
else: meta["priority"] = "basse"
|
||||||
|
|
||||||
|
return meta
|
||||||
|
|
||||||
|
|
||||||
|
# ── Main ──────────────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
def main():
|
||||||
|
print("=== Ingestion Fallout v2 → ChromaDB ===")
|
||||||
|
print(f"Chunking structurel, ~{CHUNK_WORDS} mots/chunk, overlap {OVERLAP_WORDS} mots")
|
||||||
|
print(f"Modèle embedding : {EMBED_MODEL}\n")
|
||||||
|
|
||||||
|
state = load_state()
|
||||||
|
col_id = get_or_create_collection()
|
||||||
|
|
||||||
|
pdfs = list_fallout_pdfs()
|
||||||
|
print(f"{len(pdfs)} PDFs Fallout sur Guardian\n")
|
||||||
|
|
||||||
|
for filename in pdfs:
|
||||||
|
if filename in state["ingested"]:
|
||||||
|
print(f"[SKIP] {filename}")
|
||||||
|
continue
|
||||||
|
|
||||||
|
print(f"\n[{filename}]")
|
||||||
|
try:
|
||||||
|
pdf_bytes = download_pdf(filename)
|
||||||
|
print(f" {len(pdf_bytes)/1024:.0f} Ko téléchargés")
|
||||||
|
|
||||||
|
text = extract_text(pdf_bytes)
|
||||||
|
print(f" {len(text)} chars extraits")
|
||||||
|
|
||||||
|
chunks = chunk_by_structure(text, filename)
|
||||||
|
meta_base = infer_metadata(filename)
|
||||||
|
print(f" {len(chunks)} chunks | catégorie: {meta_base['category']} | système: {meta_base['systeme']}")
|
||||||
|
|
||||||
|
ids, embeddings, documents, metadatas = [], [], [], []
|
||||||
|
|
||||||
|
for i, chunk in enumerate(chunks):
|
||||||
|
chunk_id = f"{meta_base['source']}_{i:04d}"
|
||||||
|
t0 = time.time()
|
||||||
|
emb = embed(chunk)
|
||||||
|
word_count = len(chunk.split())
|
||||||
|
ids.append(chunk_id)
|
||||||
|
embeddings.append(emb)
|
||||||
|
documents.append(chunk)
|
||||||
|
metadatas.append({
|
||||||
|
**meta_base,
|
||||||
|
"chunk_index": i,
|
||||||
|
"total_chunks": len(chunks),
|
||||||
|
"word_count": word_count,
|
||||||
|
})
|
||||||
|
print(f" chunk {i+1}/{len(chunks)} ({word_count}w) OK ({time.time()-t0:.1f}s)")
|
||||||
|
|
||||||
|
if len(ids) >= 50:
|
||||||
|
add_documents(col_id, ids, embeddings, documents, metadatas)
|
||||||
|
ids, embeddings, documents, metadatas = [], [], [], []
|
||||||
|
|
||||||
|
if ids:
|
||||||
|
add_documents(col_id, ids, embeddings, documents, metadatas)
|
||||||
|
|
||||||
|
print(f"\n Indexé — {len(chunks)} chunks")
|
||||||
|
state["ingested"].append(filename)
|
||||||
|
# Retire des failed si présent
|
||||||
|
state["failed"] = [f for f in state["failed"] if f.get("file") != filename]
|
||||||
|
save_state(state)
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
print(f" ERREUR: {e}")
|
||||||
|
state["failed"].append({"file": filename, "error": str(e)})
|
||||||
|
save_state(state)
|
||||||
|
|
||||||
|
print(f"\n=== Terminé : {len(state['ingested'])} PDFs indexés, {len(state['failed'])} échecs ===")
|
||||||
|
for f in state.get("failed", []):
|
||||||
|
print(f" ECHEC: {f['file']} — {f['error']}")
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
Reference in New Issue
Block a user