Files
open-school/Programme/pdf_collecteur/index_pdfs.py

240 lines
6.9 KiB
Python

#!/usr/bin/env python3
"""
Indexe les PDF du dossier Programme en reprenant le format historique :
- 001.txt, 002.txt, ...
- 001.pdfinfo.txt, 002.pdfinfo.txt, ...
- manifest.csv avec Id, RelativePath, TextFile, InfoFile
Le script appelle Poppler (`pdftotext` et `pdfinfo`) et ignore les fichiers
temporaires du collecteur.
"""
from __future__ import annotations
import argparse
import csv
import hashlib
import os
import shutil
import subprocess
import sys
from dataclasses import dataclass
from pathlib import Path
SCRIPT_DIR = Path(__file__).resolve().parent
PROGRAMME_DIR = SCRIPT_DIR.parent
DEFAULT_OUTPUT = PROGRAMME_DIR / "extractions_pdf_indexed_global"
DEFAULT_POPPLER_BIN = (
Path.home()
/ "AppData"
/ "Local"
/ "Microsoft"
/ "WinGet"
/ "Packages"
/ "oschwartz10612.Poppler_Microsoft.Winget.Source_8wekyb3d8bbwe"
/ "poppler-25.07.0"
/ "Library"
/ "bin"
)
MANIFEST_FIELDS = [
"Id",
"RelativePath",
"TextFile",
"InfoFile",
"SHA256",
"SizeBytes",
"Status",
"Notes",
]
@dataclass
class ToolPaths:
pdftotext: str
pdfinfo: str
def find_tool(name: str, explicit: str | None = None) -> str:
if explicit:
path = Path(explicit)
if path.exists() or os.path.isfile(str(path)):
return str(path)
raise FileNotFoundError(f"Outil introuvable: {explicit}")
from_path = shutil.which(name)
if from_path:
return from_path
candidate = DEFAULT_POPPLER_BIN / f"{name}.exe"
try:
if candidate.exists():
return str(candidate)
except PermissionError:
return str(candidate)
raise FileNotFoundError(
f"{name} introuvable. Fournir --{name} ou installer Poppler."
)
def sha256_file(path: Path) -> str:
digest = hashlib.sha256()
with path.open("rb") as handle:
for chunk in iter(lambda: handle.read(1024 * 1024), b""):
digest.update(chunk)
return digest.hexdigest().upper()
def iter_pdfs(root: Path, include_root_pdfs: bool) -> list[Path]:
allowed_prefixes = [root / "Cycle 3", root / "Cycle 4"]
if include_root_pdfs:
allowed_prefixes.append(root)
pdfs: list[Path] = []
for pdf in root.rglob("*.pdf"):
if SCRIPT_DIR in pdf.parents:
continue
if any(part.startswith("extractions_pdf") for part in pdf.relative_to(root).parts):
continue
if include_root_pdfs and pdf.parent == root:
pdfs.append(pdf)
continue
if any(prefix in [pdf, *pdf.parents] for prefix in allowed_prefixes[:2]):
pdfs.append(pdf)
return sorted(pdfs, key=lambda item: str(item.relative_to(root)).casefold())
def run_tool(args: list[str], output_file: Path) -> tuple[bool, str]:
try:
completed = subprocess.run(
args,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
check=False,
)
except PermissionError:
completed = subprocess.run(
[
"powershell.exe",
"-NoProfile",
"-Command",
" ".join(ps_quote(part) for part in ["&", *args]),
],
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
check=False,
)
output = completed.stdout
error = completed.stderr.decode("utf-8", errors="replace").strip()
output_file.write_bytes(output)
if completed.returncode != 0:
return False, error
return True, error
def ps_quote(value: str) -> str:
if value == "&":
return value
return "'" + value.replace("'", "''") + "'"
def write_manifest(path: Path, rows: list[dict[str, str]]) -> None:
with path.open("w", encoding="utf-8-sig", newline="") as handle:
writer = csv.DictWriter(handle, fieldnames=MANIFEST_FIELDS, extrasaction="ignore")
writer.writeheader()
writer.writerows(rows)
def index_pdf(
pdf: Path,
pdf_id: str,
output_dir: Path,
root: Path,
tools: ToolPaths,
) -> dict[str, str]:
text_file = f"{pdf_id}.txt"
info_file = f"{pdf_id}.pdfinfo.txt"
text_path = output_dir / text_file
info_path = output_dir / info_file
row = {
"Id": pdf_id,
"RelativePath": str(pdf.relative_to(root)),
"TextFile": text_file,
"InfoFile": info_file,
"SHA256": sha256_file(pdf),
"SizeBytes": str(pdf.stat().st_size),
"Status": "ok",
"Notes": "",
}
info_ok, info_error = run_tool([tools.pdfinfo, str(pdf)], info_path)
text_ok, text_error = run_tool([tools.pdftotext, "-layout", str(pdf), "-"], text_path)
notes = []
if not info_ok:
notes.append(f"pdfinfo: {info_error}")
if not text_ok:
notes.append(f"pdftotext: {text_error}")
if notes:
row["Status"] = "error"
row["Notes"] = " | ".join(notes)
return row
def parse_args(argv: list[str]) -> argparse.Namespace:
parser = argparse.ArgumentParser(description="Indexe les PDF de Programme avec Poppler.")
parser.add_argument("--output", type=Path, default=DEFAULT_OUTPUT)
parser.add_argument("--pdftotext", default=None, help="Chemin explicite vers pdftotext.")
parser.add_argument("--pdfinfo", default=None, help="Chemin explicite vers pdfinfo.")
parser.add_argument("--fresh", action="store_true", help="Supprime le dossier de sortie avant indexation.")
parser.add_argument(
"--include-root-pdfs",
action="store_true",
help="Inclut les PDF placés directement à la racine de Programme.",
)
parser.add_argument("--limit", type=int, default=None, help="Limite de PDF pour tester.")
return parser.parse_args(argv)
def main(argv: list[str]) -> int:
args = parse_args(argv)
tools = ToolPaths(
pdftotext=find_tool("pdftotext", args.pdftotext),
pdfinfo=find_tool("pdfinfo", args.pdfinfo),
)
output_dir = args.output.resolve()
if args.fresh and output_dir.exists():
if PROGRAMME_DIR not in output_dir.parents:
raise RuntimeError(f"Refus de supprimer hors Programme: {output_dir}")
shutil.rmtree(output_dir)
output_dir.mkdir(parents=True, exist_ok=True)
pdfs = iter_pdfs(PROGRAMME_DIR, include_root_pdfs=args.include_root_pdfs)
if args.limit is not None:
pdfs = pdfs[: args.limit]
rows: list[dict[str, str]] = []
width = max(3, len(str(len(pdfs))))
for index, pdf in enumerate(pdfs, start=1):
pdf_id = str(index).zfill(width)
print(f"[{pdf_id}/{len(pdfs)}] {pdf.relative_to(PROGRAMME_DIR)}")
rows.append(index_pdf(pdf, pdf_id, output_dir, PROGRAMME_DIR, tools))
write_manifest(output_dir / "manifest.csv", rows)
errors = sum(1 for row in rows if row["Status"] != "ok")
print(f"PDF indexés: {len(rows)}")
print(f"Erreurs: {errors}")
print(f"Dossier: {output_dir}")
return 0 if errors == 0 else 1
if __name__ == "__main__":
raise SystemExit(main(sys.argv[1:]))