classify
This commit is contained in:
95
classify_pdfs.py
Normal file
95
classify_pdfs.py
Normal file
@@ -0,0 +1,95 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""
|
||||||
|
classify_pdfs.py - Analysiert einen Ordner nach deinem neuen Workflow:
|
||||||
|
|
||||||
|
1. Gruppiert Dateien nach Person (aus Dateiname 'Nachname, Vorname')
|
||||||
|
2. Trennt Notes- von Nicht-Notes-Dateien
|
||||||
|
3. Klassifiziert jede Nicht-Notes-Datei als TEXT-PDF oder IMAGE-PDF
|
||||||
|
(hat die pypdf-Textebene genug Inhalt?)
|
||||||
|
4. Prueft bei TEXT-PDFs, ob Datum + Name tatsaechlich im Text stehen
|
||||||
|
5. Findet Personen, die NUR als Image-PDF vorliegen (Sonderbehandlung spaeter)
|
||||||
|
|
||||||
|
Kein LLM, keine Netzwerk. Reine lokale Analyse zum Planen.
|
||||||
|
|
||||||
|
Aufruf:
|
||||||
|
python3 classify_pdfs.py --src ~/data/S
|
||||||
|
"""
|
||||||
|
import os, re, glob, argparse
|
||||||
|
from collections import defaultdict
|
||||||
|
from pypdf import PdfReader
|
||||||
|
|
||||||
|
TEXT_THRESHOLD = 120 # Zeichen in der Textebene -> gilt als Text-PDF
|
||||||
|
|
||||||
|
def person_from_filename(path):
|
||||||
|
base = os.path.basename(path)
|
||||||
|
base = re.sub(r"\.pdf$", "", base, flags=re.I)
|
||||||
|
base = re.sub(r"(?i)\bnotes\b.*$", "", base) # ab 'Notes' abschneiden
|
||||||
|
base = re.sub(r"\b\d{6,8}\b.*$", "", base) # ab Datum abschneiden
|
||||||
|
base = re.sub(r"\([^)]*\)", "", base) # (Talis) etc. weg
|
||||||
|
return base.strip(" -_").lower()
|
||||||
|
|
||||||
|
def get_text(path):
|
||||||
|
try:
|
||||||
|
r = PdfReader(path)
|
||||||
|
return "\n".join(p.extract_text() or "" for p in r.pages[:4])
|
||||||
|
except Exception:
|
||||||
|
return ""
|
||||||
|
|
||||||
|
def has_date(text):
|
||||||
|
# sucht MM/DD/YYYY, MM-DD-YY, etc.
|
||||||
|
return bool(re.search(r"\b\d{1,2}\s*[/.-]\s*\d{1,2}\s*[/.-]\s*\d{2,4}\b", text))
|
||||||
|
|
||||||
|
def main():
|
||||||
|
ap = argparse.ArgumentParser()
|
||||||
|
ap.add_argument("--src", required=True)
|
||||||
|
args = ap.parse_args()
|
||||||
|
|
||||||
|
pdfs = sorted(glob.glob(os.path.join(args.src, "*.pdf")))
|
||||||
|
persons = defaultdict(list)
|
||||||
|
for p in pdfs:
|
||||||
|
persons[person_from_filename(p)].append(p)
|
||||||
|
|
||||||
|
total_persons = len(persons)
|
||||||
|
n_text = n_image = n_notes = 0
|
||||||
|
text_with_date = 0
|
||||||
|
only_image_persons = []
|
||||||
|
person_has_text = {}
|
||||||
|
|
||||||
|
for person, files in persons.items():
|
||||||
|
non_notes = [f for f in files if "notes" not in os.path.basename(f).lower()]
|
||||||
|
notes = [f for f in files if "notes" in os.path.basename(f).lower()]
|
||||||
|
n_notes += len(notes)
|
||||||
|
|
||||||
|
has_any_text = False
|
||||||
|
for f in non_notes:
|
||||||
|
txt = get_text(f)
|
||||||
|
if len(txt.strip()) >= TEXT_THRESHOLD:
|
||||||
|
n_text += 1
|
||||||
|
has_any_text = True
|
||||||
|
if has_date(txt):
|
||||||
|
text_with_date += 1
|
||||||
|
else:
|
||||||
|
n_image += 1
|
||||||
|
|
||||||
|
# Person hat NUR Image-PDFs (oder nur Notes)?
|
||||||
|
if not has_any_text:
|
||||||
|
only_image_persons.append(person)
|
||||||
|
person_has_text[person] = has_any_text
|
||||||
|
|
||||||
|
print(f"=== Analyse: {args.src} ===")
|
||||||
|
print(f"PDFs gesamt: {len(pdfs)}")
|
||||||
|
print(f"Distinkte Personen: {total_persons}")
|
||||||
|
print(f"Notes-Dateien: {n_notes}")
|
||||||
|
print()
|
||||||
|
print(f"Nicht-Notes TEXT-PDFs: {n_text}")
|
||||||
|
print(f" davon mit Datum im Text: {text_with_date} ({100*text_with_date//max(n_text,1)}%)")
|
||||||
|
print(f"Nicht-Notes IMAGE-PDFs: {n_image}")
|
||||||
|
print()
|
||||||
|
print(f"Personen NUR mit Image/Notes (Sonderbehandlung spaeter): {len(only_image_persons)}")
|
||||||
|
for p in only_image_persons[:20]:
|
||||||
|
print(f" {p}")
|
||||||
|
if len(only_image_persons) > 20:
|
||||||
|
print(f" ... und {len(only_image_persons)-20} weitere")
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
Reference in New Issue
Block a user