diff --git a/notes_filter.ts b/notes_filter.ts new file mode 100644 index 0000000..2fe1e79 --- /dev/null +++ b/notes_filter.ts @@ -0,0 +1,127 @@ +#!/usr/bin/env npx tsx +/** + * notes_filter.ts — Notes-PDFs aussortieren und Seed-JSON erzeugen + * + * 1. Scannt alle PDFs unter --pdf-root rekursiv. + * 2. "Notes"-Dateien: prüft, ob eine Hauptdatei desselben Käufers existiert + * (Schlüssel: "Nachname, Vorname" — Datum wird ignoriert, da es abweicht). + * → mit Gegenstück: überspringen. Ohne Gegenstück: Report (out/notes_orphans.txt). + * 3. Schreibt Seed-JSON mit ALLEN Nicht-Notes-PDFs. Vorhandene Einträge aus + * --buyers (Gleis A / Vision) werden unverändert übernommen, neue Dateien + * bekommen einen Eintrag mit is_buyer_sheet:false (→ vision_runner nimmt sie). + * + * Aufruf: + * npx tsx notes_filter.ts \ + * --pdf-root "/mnt/bizmatch-nas/AA Buyers NDA's/Buyers NDA's A-Z" \ + * --buyers out/buyers.json \ + * --out out/buyers_seed.json + */ + +import * as fsp from "node:fs/promises"; +import * as fs from "node:fs"; +import * as path from "node:path"; +import * as os from "node:os"; + +interface BuyerRecord { file_name: string; is_buyer_sheet: boolean; [k: string]: unknown } + +function parseArgs() { + const a = process.argv.slice(2); + const get = (f: string, d: string | null = null) => { + const i = a.indexOf(f); + return i >= 0 && a[i + 1] !== undefined ? a[i + 1] : d; + }; + const pdfRoot = get("--pdf-root"); + if (!pdfRoot) { console.error("Fehler: --pdf-root fehlt."); process.exit(1); } + return { + pdfRoot: pdfRoot.replace(/^~(?=\/)/, os.homedir()), + buyers: get("--buyers", "out/buyers.json")!, + out: get("--out", "out/buyers_seed.json")!, + }; +} + +const isNotes = (base: string) => /\bnotes\b/i.test(base); + +/** "Stone, Mike via Anna Stone 031220 Notes.pdf" → "stone,mike" */ +function nameKey(base: string): string { + let s = base.replace(/\.pdf$/i, ""); + s = s.replace(/\bnotes\b/gi, " "); + s = s.replace(/\b\d{4,8}\b/g, " "); // Datums-Tokens raus + s = s.replace(/\s+/g, " ").trim(); + const comma = s.indexOf(","); + if (comma < 0) return s.toLowerCase(); // Fallback: ganzer Rest + const last = s.slice(0, comma).trim(); + const first = (s.slice(comma + 1).trim().split(" ")[0] ?? ""); + return `${last},${first}`.toLowerCase(); +} + +async function collectPdfs(root: string): Promise { + const result: string[] = []; + const stack = [root]; + while (stack.length) { + const dir = stack.pop()!; + let entries: fs.Dirent[]; + try { entries = await fsp.readdir(dir, { withFileTypes: true }); } catch { continue; } + for (const e of entries) { + const p = path.join(dir, e.name); + if (e.isDirectory()) stack.push(p); + else if (e.isFile() && e.name.toLowerCase().endsWith(".pdf")) result.push(e.name); + } + } + return result; +} + +const EMPTY_FIELDS = { + name_company: null, prospective_buyer: null, company: null, phone: null, + cell: null, email: null, address: null, state: null, how_did_you_hear: null, + interested_in_updates: null, types_of_business_raw: null, + background_experience: null, total_purchase_price: null, down_payment: null, + down_payment_raw: null, date_of_introduction: null, +}; + +async function main() { + const args = parseArgs(); + + console.error(`Scanne ${args.pdfRoot} ...`); + const all = await collectPdfs(args.pdfRoot); + const notesFiles = all.filter(isNotes); + const mainFiles = all.filter((f) => !isNotes(f)); + console.error(`${all.length} PDFs: ${mainFiles.length} Hauptdateien, ${notesFiles.length} Notes-Dateien`); + + // Notes-Abgleich + const mainKeys = new Set(mainFiles.map(nameKey)); + const skippable: string[] = []; + const orphans: string[] = []; + for (const n of notesFiles) (mainKeys.has(nameKey(n)) ? skippable : orphans).push(n); + console.error(`Notes mit Gegenstück (werden ignoriert): ${skippable.length}`); + console.error(`Notes OHNE Gegenstück (bitte prüfen): ${orphans.length}`); + if (orphans.length) { + const orphanPath = path.join(path.dirname(args.out), "notes_orphans.txt"); + await fsp.mkdir(path.dirname(orphanPath), { recursive: true }); + await fsp.writeFile(orphanPath, orphans.sort().join("\n") + "\n"); + console.error(`→ ${orphanPath}`); + } + + // Vorhandene Ergebnisse übernehmen (Notes-Einträge dabei verwerfen) + const existing = new Map(); + try { + const prev: BuyerRecord[] = JSON.parse(await fsp.readFile(args.buyers, "utf8")); + for (const r of prev) if (!isNotes(r.file_name)) existing.set(r.file_name, r); + } catch { + console.error(`Hinweis: ${args.buyers} nicht gefunden — starte mit leerem Bestand.`); + } + + // Seed: alle Hauptdateien, vorhandene Daten bleiben erhalten + let seeded = 0; + const out: BuyerRecord[] = mainFiles.sort().map((f) => { + const prev = existing.get(f); + if (prev) return prev; + seeded++; + return { file_name: f, is_buyer_sheet: false, ...EMPTY_FIELDS, _parser: "seed" }; + }); + + await fsp.mkdir(path.dirname(args.out), { recursive: true }); + await fsp.writeFile(args.out, JSON.stringify(out, null, 2)); + console.error(`Seed geschrieben: ${out.length} Einträge (${existing.size} übernommen, ${seeded} neu) → ${args.out}`); +} + +main().catch((e) => { console.error("Abbruch:", e instanceof Error ? e.message : e); process.exit(1); }); \ No newline at end of file