Compare commits

..

22 Commits

Author SHA1 Message Date
8f104cb8cc address 2026-07-19 12:49:36 -05:00
92281f33bd reprocessMissing only date 2026-07-16 18:27:18 -05:00
9a8a8c86e3 change context size 2026-07-16 15:44:03 -05:00
18f8f97a43 fixes 2026-07-16 10:47:38 -05:00
92e1f3235a sdfsd 2026-07-16 10:43:58 -05:00
e46f00c838 new 2026-07-14 18:52:19 -05:00
e367b6ac29 dfgdfg 2026-07-14 15:09:32 -05:00
7b99d18f2c dsgfds 2026-07-14 14:28:24 -05:00
db8e6cce21 dfgdfg 2026-07-14 14:17:18 -05:00
d8d62fa371 sdfsd 2026-07-14 14:01:11 -05:00
96eb7ad9b9 update 2026-07-14 12:34:44 -05:00
6c3786f200 gdfgdf 2026-07-12 22:59:50 -05:00
70f09fa9da sdfsdf 2026-07-12 22:59:21 -05:00
14699b5f26 dfgdfg 2026-07-12 18:07:58 -05:00
1ff5f429c4 asdasd 2026-07-12 17:36:58 -05:00
82cca38f29 asdsa 2026-07-12 17:34:29 -05:00
f44d998a7f sdf 2026-07-12 15:16:37 -05:00
d95b8c112d timestamp 2026-07-12 14:05:05 -05:00
6c3eadc2c3 name 2026-07-12 13:14:38 -05:00
6ead3d76ec dg 2026-07-12 13:00:07 -05:00
fe3f8f14fa new headers 2026-07-12 12:52:04 -05:00
06f7df3fe1 certs 2026-07-12 12:36:05 -05:00
9 changed files with 595 additions and 74 deletions

4
.gitignore vendored
View File

@@ -1,4 +1,6 @@
poc_out
node_modules
package-lock.json
*.jsonl
*.json
*.log
*.txt

70
add_name_from_filename.ts Normal file
View File

@@ -0,0 +1,70 @@
#!/usr/bin/env npx tsx
/**
* add_name_from_filename.ts
*
* Traegt das Feld "name_from_filename" in eine bestehende buyers_vision.json
* nach, OHNE die Vision-Ergebnisse anzufassen und OHNE GPU/Neuverarbeitung.
* Nutzt exakt dieselbe Logik wie der vision_runner, damit die Werte
* identisch zu kuenftigen Laeufen sind.
*
* Aufruf:
* npx tsx add_name_from_filename.ts out_Z/buyers_vision.json
* npx tsx add_name_from_filename.ts out_Z/buyers_vision.json --overwrite
*
* Ohne --overwrite werden nur Datensaetze ergaenzt, die das Feld noch nicht
* (oder null) haben. Ein Backup .bak wird immer angelegt.
*/
import { promises as fsp } from "node:fs";
// IDENTISCH zur Funktion im vision_runner.ts
function nameFromFilename(fileName: string): string | null {
let s = fileName.replace(/\.pdf$/i, "");
s = s.replace(/\([^)]*\)/g, " ").replace(/\s+/g, " ").trim();
const comma = s.indexOf(",");
if (comma < 0) {
const first = s.replace(/\b\d{4,8}\b.*$/, "").trim();
return first.length >= 2 ? first : null;
}
const last = s.slice(0, comma).trim();
const rest = s.slice(comma + 1).trim();
const firstName = (rest.match(/^[A-Za-zÀ-ÿ.'-]+/) || [""])[0];
if (!last || !firstName) return null;
return `${last}, ${firstName}`;
}
async function main() {
const args = process.argv.slice(2);
const file = args.find((a) => !a.startsWith("--"));
const overwrite = args.includes("--overwrite");
if (!file) {
console.error("Aufruf: npx tsx add_name_from_filename.ts <buyers_vision.json> [--overwrite]");
process.exit(1);
}
const raw = await fsp.readFile(file, "utf8");
const records: Array<Record<string, unknown>> = JSON.parse(raw);
// Backup
await fsp.writeFile(file + ".bak", raw);
let added = 0, skipped = 0, unchanged = 0;
for (const r of records) {
const fn = r["file_name"] as string | undefined;
if (!fn) { skipped++; continue; }
const has = r["name_from_filename"] != null && r["name_from_filename"] !== "";
if (has && !overwrite) { unchanged++; continue; }
const val = nameFromFilename(fn);
if (r["name_from_filename"] === val) { unchanged++; continue; }
r["name_from_filename"] = val;
added++;
}
await fsp.writeFile(file, JSON.stringify(records, null, 2));
console.log(`Datensaetze: ${records.length}`);
console.log(`Feld gesetzt: ${added}`);
console.log(`schon vorhanden: ${unchanged}`);
console.log(`ohne file_name: ${skipped}`);
console.log(`Backup: ${file}.bak`);
}
main().catch((e) => { console.error(e); process.exit(1); });

View File

@@ -10,7 +10,7 @@
FROM nvidia/cuda:12.8.1-devel-ubuntu24.04 AS build
RUN apt-get update && apt-get install -y --no-install-recommends \
git cmake build-essential libcurl4-openssl-dev \
git cmake build-essential libcurl4-openssl-dev ca-certificates \
&& rm -rf /var/lib/apt/lists/*
RUN git clone --depth 1 https://github.com/ggml-org/llama.cpp /src
@@ -19,6 +19,7 @@ RUN git clone --depth 1 https://github.com/ggml-org/llama.cpp /src
RUN cmake /src -B /build \
-DGGML_CUDA=ON \
-DCMAKE_CUDA_ARCHITECTURES=120 \
-DGGML_CUDA_FORCE_CUBLAS=OFF \
-DBUILD_SHARED_LIBS=OFF \
-DLLAMA_CURL=ON \
&& cmake --build /build --config Release -j --target llama-server
@@ -26,7 +27,7 @@ RUN cmake /src -B /build \
FROM nvidia/cuda:12.8.1-runtime-ubuntu24.04
RUN apt-get update && apt-get install -y --no-install-recommends \
libcurl4 libgomp1 curl \
libcurl4 libgomp1 curl ca-certificates \
&& rm -rf /var/lib/apt/lists/*
COPY --from=build /build/bin/llama-server /usr/local/bin/llama-server

View File

@@ -6,8 +6,8 @@
FROM ubuntu:24.04 AS build
RUN apt-get update && apt-get install -y --no-install-recommends \
git cmake build-essential libcurl4-openssl-dev \
libvulkan-dev glslc \
git cmake build-essential libcurl4-openssl-dev ca-certificates \
libvulkan-dev glslc spirv-headers \
&& rm -rf /var/lib/apt/lists/*
RUN git clone --depth 1 https://github.com/ggml-org/llama.cpp /src
@@ -23,7 +23,7 @@ FROM ubuntu:24.04
# mesa-vulkan-drivers = RADV-Treiber im Container (GPU via /dev/dri durchgereicht)
RUN apt-get update && apt-get install -y --no-install-recommends \
libvulkan1 mesa-vulkan-drivers vulkan-tools \
libcurl4 libgomp1 curl \
libcurl4 libgomp1 curl ca-certificates \
&& rm -rf /var/lib/apt/lists/*
COPY --from=build /build/bin/llama-server /usr/local/bin/llama-server

View File

@@ -43,7 +43,7 @@ services:
- -fa
- "on"
- -c
- "32768"
- "65536"
- --parallel
- "1"
- --jinja
@@ -84,14 +84,16 @@ services:
devices:
- /dev/dri:/dev/dri
- /dev/kfd:/dev/kfd
# Numerische HOST-GIDs verwenden — Namen wie "render" existieren im
# Container-Image nicht. GIDs prüfen mit: getent group video render
group_add:
- video
- render
- "44" # video (Host-GID ggf. anpassen)
- "991" # render (Host-GID ggf. anpassen)
security_opt:
- seccomp:unconfined
command:
- -m
- /models/Qwen3.6-35B-A3B-UD-Q4_K_XL.gguf
- /models/Qwen3.6-35B-A3B-UD-Q4_K_M.gguf
- --mmproj
- /models/mmproj-BF16.gguf
- --host
@@ -105,7 +107,7 @@ services:
- -fa
- "on"
- -c
- "32768"
- "40960"
- --parallel
- "1"
# KV-Cache quantisieren — 32GB VRAM, 35B-A3B + mmproj + Bildkontext:
@@ -119,7 +121,7 @@ services:
- --image-min-tokens
- "1024"
- --image-max-tokens
- "4096"
- "6144"
- --temp
- "0.1"
- --top-p

119
enrich_and_anonymize.ts Normal file
View File

@@ -0,0 +1,119 @@
#!/usr/bin/env npx tsx
/**
* enrich_and_anonymize.ts
*
* Zwei Aufgaben in einem Durchlauf auf einer buyers_vision.json:
* 1) Ergaenzt "name_from_filename" (falls fehlt) UND "_letter"
* (der A-Z-Unterordner, in dem das PDF liegt) fuer die PDF-Pfad-Aufloesung.
* 2) Optional (--anonymize): ersetzt PII-Felder (phone, cell, email, address)
* durch konsistente Faker-Dummydaten, damit das JSON weitergegeben werden
* kann. Leere Felder bleiben leer. Gleicher Originalwert -> gleicher Dummy.
*
* Der _letter wird aus dem Nachnamen (erstes Zeichen von name_from_filename)
* abgeleitet - das entspricht der Ordnerstruktur "...Buyers NDA's A-Z/<Letter>/".
*
* Aufruf:
* npx tsx enrich_and_anonymize.ts buyers_vision.json
* npx tsx enrich_and_anonymize.ts buyers_vision.json --anonymize
* npx tsx enrich_and_anonymize.ts buyers_vision.json --anonymize --overwrite
*/
import { promises as fsp } from "node:fs";
import { faker } from "@faker-js/faker";
faker.seed(1234); // reproduzierbar
// IDENTISCH zur Funktion im vision_runner.ts
function nameFromFilename(fileName: string): string | null {
let s = fileName.replace(/\.pdf$/i, "");
s = s.replace(/\([^)]*\)/g, " ").replace(/\s+/g, " ").trim();
const comma = s.indexOf(",");
if (comma < 0) {
const first = s.replace(/\b\d{4,8}\b.*$/, "").trim();
return first.length >= 2 ? first : null;
}
const last = s.slice(0, comma).trim();
const rest = s.slice(comma + 1).trim();
const firstName = (rest.match(/^[A-Za-zÀ-ÿ.'-]+/) || [""])[0];
if (!last || !firstName) return null;
return `${last}, ${firstName}`;
}
/** A-Z-Unterordner aus dem Nachnamen. Fallback: erstes Zeichen des Dateinamens. */
function letterFromName(fileName: string, nameFF: string | null): string | null {
const src = (nameFF ?? fileName).trim();
const ch = src.charAt(0).toUpperCase();
return /[A-Z]/.test(ch) ? ch : null;
}
const isEmpty = (v: unknown) => v === null || v === undefined || v === "";
// Konsistenz-Caches: gleicher Originalwert -> gleicher Dummy
const caches: Record<string, Map<string, string>> = {
phone: new Map(), cell: new Map(), email: new Map(), address: new Map(),
};
function fakePII(field: string, value: string): string {
const key = value.trim().toLowerCase();
const cache = caches[field];
if (cache.has(key)) return cache.get(key)!;
let dummy: string;
switch (field) {
case "phone":
case "cell": dummy = faker.phone.number(); break;
case "email": dummy = faker.internet.email(); break;
case "address": dummy = faker.location.streetAddress({ useFullAddress: true }); break;
default: dummy = faker.lorem.word();
}
cache.set(key, dummy);
return dummy;
}
async function main() {
const args = process.argv.slice(2);
const file = args.find((a) => !a.startsWith("--"));
const overwrite = args.includes("--overwrite");
const anonymize = args.includes("--anonymize");
if (!file) {
console.error("Aufruf: npx tsx enrich_and_anonymize.ts <buyers_vision.json> [--anonymize] [--overwrite]");
process.exit(1);
}
const raw = await fsp.readFile(file, "utf8");
const records: Array<Record<string, unknown>> = JSON.parse(raw);
await fsp.writeFile(file + ".bak", raw);
let nameSet = 0, letterSet = 0, anonFields = 0;
const PII = ["phone", "cell", "email", "address"];
for (const r of records) {
const fn = r["file_name"] as string | undefined;
if (!fn) continue;
// name_from_filename
const hasName = !isEmpty(r["name_from_filename"]);
if (!hasName || overwrite) {
const val = nameFromFilename(fn);
if (r["name_from_filename"] !== val) { r["name_from_filename"] = val; nameSet++; }
}
// _letter (A-Z-Unterordner)
const hasLetter = !isEmpty(r["_letter"]);
if (!hasLetter || overwrite) {
const lv = letterFromName(fn, (r["name_from_filename"] as string | null) ?? null);
if (r["_letter"] !== lv) { r["_letter"] = lv; letterSet++; }
}
// Anonymisierung
if (anonymize) {
for (const f of PII) {
if (!isEmpty(r[f])) { r[f] = fakePII(f, String(r[f])); anonFields++; }
}
}
}
await fsp.writeFile(file, JSON.stringify(records, null, 2));
console.log(`Datensaetze: ${records.length}`);
console.log(`name_from_filename: ${nameSet} gesetzt`);
console.log(`_letter: ${letterSet} gesetzt`);
if (anonymize) console.log(`PII anonymisiert: ${anonFields} Felder`);
console.log(`Backup: ${file}.bak`);
}
main().catch((e) => { console.error(e); process.exit(1); });

138
notes_filter.ts Normal file
View File

@@ -0,0 +1,138 @@
#!/usr/bin/env npx tsx
/**
* notes_filter.ts — Notes-PDFs aussortieren und Seed-JSON erzeugen
*
* 1. Scannt alle PDFs unter --pdf-root rekursiv.
* 2. "Notes"-Dateien: prüft, ob eine Hauptdatei desselben Käufers existiert
* (Schlüssel: "Nachname, Vorname" — Datum wird ignoriert, da es abweicht).
* → mit Gegenstück: überspringen. Ohne Gegenstück: Report (out/notes_orphans.txt).
* 3. Schreibt Seed-JSON mit ALLEN Nicht-Notes-PDFs. Vorhandene Einträge aus
* --buyers (Gleis A / Vision) werden unverändert übernommen, neue Dateien
* bekommen einen Eintrag mit is_buyer_sheet:false (→ vision_runner nimmt sie).
*
* Aufruf:
* npx tsx notes_filter.ts \
* --pdf-root "/mnt/bizmatch-nas/AA Buyers NDA's/Buyers NDA's A-Z" \
* --buyers out/buyers.json \
* --out out/buyers_seed.json
*/
import * as fsp from "node:fs/promises";
import * as fs from "node:fs";
import * as path from "node:path";
import * as os from "node:os";
interface BuyerRecord { file_name: string; is_buyer_sheet: boolean; [k: string]: unknown }
function parseArgs() {
const a = process.argv.slice(2);
const get = (f: string, d: string | null = null) => {
const i = a.indexOf(f);
return i >= 0 && a[i + 1] !== undefined ? a[i + 1] : d;
};
const pdfRoot = get("--pdf-root");
if (!pdfRoot) { console.error("Fehler: --pdf-root fehlt."); process.exit(1); }
return {
pdfRoot: pdfRoot.replace(/^~(?=\/)/, os.homedir()),
buyers: get("--buyers", "out/buyers.json")!,
out: get("--out", "out/buyers_seed.json")!,
};
}
/** Erkennt "Notes", "Note" und verklebte Varianten ("Notesb102312").
* Es wird nur der Teil NACH dem ersten Komma durchsucht, damit Nachnamen
* wie "Note, John" nicht fälschlich als Notes-Datei gelten. */
function isNotes(base: string): boolean {
const comma = base.indexOf(",");
const scope = comma >= 0 ? base.slice(comma + 1) : base;
return /\bnotes?[a-z]?\d*\b/i.test(scope);
}
/** "Stone, Mike via Anna Stone 031220 Notes.pdf" → "stone,mike" */
function nameKey(base: string): string {
let s = base.replace(/\.pdf$/i, "");
s = s.replace(/\bnotes?[a-z]?\d*\b/gi, " "); // Note/Notes/Notesb102312 raus
s = s.replace(/\b\d{4,8}\b/g, " "); // Datums-Tokens raus
s = s.replace(/\s+/g, " ").trim();
const comma = s.indexOf(",");
if (comma < 0) return s.toLowerCase(); // Fallback: ganzer Rest
const last = s.slice(0, comma).trim();
const first = (s.slice(comma + 1).trim().split(" ")[0] ?? "");
return `${last},${first}`.toLowerCase();
}
async function collectPdfs(root: string): Promise<string[]> {
const result: string[] = [];
const stack = [root];
while (stack.length) {
const dir = stack.pop()!;
let entries: fs.Dirent[];
try { entries = await fsp.readdir(dir, { withFileTypes: true }); } catch { continue; }
for (const e of entries) {
const p = path.join(dir, e.name);
if (e.isDirectory()) stack.push(p);
else if (e.isFile() && e.name.toLowerCase().endsWith(".pdf")) result.push(e.name);
}
}
return result;
}
const EMPTY_FIELDS = {
name_company: null, prospective_buyer: null, company: null, phone: null,
cell: null, email: null, address: null, state: null, how_did_you_hear: null,
interested_in_updates: null, types_of_business_raw: null,
background_experience: null, total_purchase_price: null, down_payment: null,
down_payment_raw: null, date_of_introduction: null,
};
async function main() {
const args = parseArgs();
console.error(`Scanne ${args.pdfRoot} ...`);
const all = await collectPdfs(args.pdfRoot);
const notesFiles = all.filter(isNotes);
const mainFiles = all.filter((f) => !isNotes(f));
console.error(`${all.length} PDFs: ${mainFiles.length} Hauptdateien, ${notesFiles.length} Notes-Dateien`);
// Notes-Abgleich
const mainKeys = new Set(mainFiles.map(nameKey));
const skippable: string[] = [];
const orphans: string[] = [];
for (const n of notesFiles) (mainKeys.has(nameKey(n)) ? skippable : orphans).push(n);
console.error(`Notes mit Gegenstück (werden ignoriert): ${skippable.length}`);
console.error(`Notes OHNE Gegenstück (bitte prüfen): ${orphans.length}`);
if (orphans.length) {
const orphanPath = path.join(path.dirname(args.out), "notes_orphans.txt");
await fsp.mkdir(path.dirname(orphanPath), { recursive: true });
await fsp.writeFile(orphanPath, orphans.sort().join("\n") + "\n");
console.error(`${orphanPath}`);
}
// Vorhandene Ergebnisse laden
const existing = new Map<string, BuyerRecord>();
try {
const prev: BuyerRecord[] = JSON.parse(await fsp.readFile(args.buyers, "utf8"));
for (const r of prev) existing.set(r.file_name, r);
} catch {
console.error(`Hinweis: ${args.buyers} nicht gefunden — starte mit leerem Bestand.`);
}
// In die Extraktion gehen: alle Hauptdateien + Notes OHNE Gegenstück.
// Notes MIT Gegenstück fallen weg (auch deren alte Einträge).
const includeFiles = [...mainFiles, ...orphans];
let seeded = 0, carried = 0;
const out: BuyerRecord[] = includeFiles.sort().map((f) => {
const prev = existing.get(f);
if (prev) { carried++; return prev; }
seeded++;
return { file_name: f, is_buyer_sheet: false, ...EMPTY_FIELDS, _parser: "seed" };
});
await fsp.mkdir(path.dirname(args.out), { recursive: true });
await fsp.writeFile(args.out, JSON.stringify(out, null, 2));
console.error(
`Seed geschrieben: ${out.length} Einträge (${carried} übernommen, ${seeded} neu, davon ${orphans.length} Orphan-Notes) → ${args.out}`
);
}
main().catch((e) => { console.error("Abbruch:", e instanceof Error ? e.message : e); process.exit(1); });

View File

@@ -1,5 +1,6 @@
{
"dependencies": {
"@faker-js/faker": "^10.5.0",
"canvas": "^3.2.3",
"pdf-parse": "^1.1.1",
"pdfjs-dist": "^3.11.174"

View File

@@ -42,14 +42,14 @@ const execFileP = promisify(execFile);
// ---------------------------------------------------------------------------
interface Args {
input: string;
pdfRoot: string;
api: string;
limit: number;
only: string | null;
dpi: number;
dpi: number | "auto";
maxPages: number;
force: boolean;
reprocessMissing: boolean;
outDir: string;
timeoutMs: number;
}
@@ -60,23 +60,27 @@ function parseArgs(): Args {
const i = a.indexOf(flag);
return i >= 0 && a[i + 1] !== undefined ? a[i + 1] : def;
};
const input = get("--input", "out/buyers.json")!;
const pdfRoot = get("--pdf-root");
const api = (get("--api", "http://localhost:8000/v1") || "").replace(/\/+$/, "");
if (!pdfRoot) {
console.error("Fehler: --pdf-root <Verzeichnis> ist erforderlich.");
process.exit(1);
throw new Error("unreachable"); // hilft dem TS-Narrowing
}
return {
input,
pdfRoot: pdfRoot.replace(/^~(?=\/)/, os.homedir()),
api,
limit: parseInt(get("--limit", "0")!, 10) || 0,
only: get("--only"),
dpi: parseInt(get("--dpi", "150")!, 10) || 150,
maxPages: parseInt(get("--max-pages", "8")!, 10) || 8,
dpi: ((): number | "auto" => {
const v = get("--dpi", "auto")!;
return v === "auto" ? "auto" : parseInt(v, 10) || 150;
})(),
// Nur PDFs mit HOECHSTENS so vielen Seiten werden per Vision gescannt.
maxPages: parseInt(get("--max-pages", "10")!, 10) || 10,
force: a.includes("--force"),
outDir: get("--out-dir", path.dirname(input))!,
reprocessMissing: a.includes("--reprocess-missing"),
outDir: get("--out-dir", "out")!,
timeoutMs: (parseInt(get("--timeout", "180")!, 10) || 180) * 1000,
};
}
@@ -93,9 +97,11 @@ interface BuyerRecord {
/** Rohantwort des VLM — alles verbatim, Normalisierung erfolgt in TS. */
interface VisionRaw {
is_buyer_sheet: boolean;
doc_type: "buyer_sheet" | "ca_only" | "other";
info_sheet_count: number;
info_page: number | null;
ca_page: number | null;
notes_page: number | null;
name_company: string | null;
prospective_buyer: string | null;
company: string | null;
@@ -107,6 +113,7 @@ interface VisionRaw {
how_did_you_hear: string | null;
interested_in_updates: string | null;
types_of_business_raw: string | null;
notes_business_raw: string | null;
background_experience: string | null;
total_purchase_price: string | null;
down_payment_raw: string | null;
@@ -179,8 +186,10 @@ function isoDate(y: number, mo: number, d: number): string | null {
/** US-Formate: 6/25/26, 06-25-2026, "June 25, 2026" → YYYY-MM-DD. */
function normDate(raw: string | null): string | null {
const c = cleanStr(raw);
let c = cleanStr(raw);
if (!c) return null;
// Leerzeichen um Trenner entfernen: "09 / 13 / 2021" -> "09/13/2021"
c = c.replace(/\s*([\/\-.])\s*/g, "$1").trim();
let m = c.match(/^(\d{1,2})[\/\-.](\d{1,2})[\/\-.](\d{2}|\d{4})$/);
if (m) return isoDate(parseInt(m[3], 10), parseInt(m[1], 10), parseInt(m[2], 10));
m = c.match(/^([A-Za-z]{3,9})\.?\s+(\d{1,2})(?:st|nd|rd|th)?,?\s+(\d{2}|\d{4})$/);
@@ -219,6 +228,17 @@ async function buildPdfIndex(root: string): Promise<Map<string, string>> {
return index;
}
/** Seitenzahl via pdfinfo (poppler-utils). -1 bei Fehler/beschaedigtem PDF. */
async function pdfPageCount(pdfPath: string): Promise<number> {
try {
const { stdout } = await execFileP("pdfinfo", [pdfPath], { timeout: 30_000 });
const m = stdout.match(/^Pages:\s+(\d+)/m);
return m ? parseInt(m[1], 10) : -1;
} catch {
return -1;
}
}
async function renderPdf(pdfPath: string, dpi: number, maxPages: number, tmpDir: string): Promise<string[]> {
const prefix = path.join(tmpDir, "page");
await execFileP("pdftoppm", ["-png", "-r", String(dpi), "-l", String(maxPages), pdfPath, prefix], {
@@ -233,6 +253,19 @@ async function renderPdf(pdfPath: string, dpi: number, maxPages: number, tmpDir:
return files;
}
/** Hat das PDF eine nennenswerte Textebene? (Hybrid: Werte getippt → 150 DPI
* reicht. Reiner Scan: keine Textebene, Handschrift möglich → 200 DPI.) */
async function hasTextLayer(pdfPath: string, maxPages: number): Promise<boolean> {
try {
const { stdout } = await execFileP("pdftotext", ["-l", String(maxPages), pdfPath, "-"], {
timeout: 30_000, maxBuffer: 10 * 1024 * 1024,
});
return stdout.replace(/\s+/g, "").length > 100;
} catch {
return false; // im Zweifel als Scan behandeln → hohe Auflösung
}
}
// ---------------------------------------------------------------------------
// VLM-Aufruf (llama-server, OpenAI-kompatibel, JSON-Schema)
// ---------------------------------------------------------------------------
@@ -244,15 +277,18 @@ const RESPONSE_SCHEMA = {
type: "object",
additionalProperties: false,
required: [
"is_buyer_sheet", "info_page", "ca_page", "name_company", "prospective_buyer",
"company", "phone", "cell", "email", "address", "state", "how_did_you_hear",
"interested_in_updates", "types_of_business_raw", "background_experience",
"doc_type", "info_sheet_count", "info_page", "ca_page", "notes_page", "name_company",
"prospective_buyer", "company", "phone", "cell", "email", "address",
"state", "how_did_you_hear", "interested_in_updates",
"types_of_business_raw", "notes_business_raw", "background_experience",
"total_purchase_price", "down_payment_raw", "date_of_introduction_raw",
],
properties: {
is_buyer_sheet: { type: "boolean" },
doc_type: { type: "string", enum: ["buyer_sheet", "ca_only", "other"] },
info_sheet_count: { type: "integer" },
info_page: nullableInt,
ca_page: nullableInt,
notes_page: nullableInt,
name_company: nullableString,
prospective_buyer: nullableString,
company: nullableString,
@@ -264,6 +300,7 @@ const RESPONSE_SCHEMA = {
how_did_you_hear: nullableString,
interested_in_updates: nullableString,
types_of_business_raw: nullableString,
notes_business_raw: nullableString,
background_experience: nullableString,
total_purchase_price: nullableString,
down_payment_raw: nullableString,
@@ -277,27 +314,50 @@ const SYSTEM_PROMPT =
"including typos. You never guess, infer, or invent values. If a field is blank " +
"or unreadable, you return null.";
const USER_PROMPT = `You see all pages of one PDF, in order (image 1 = page 1).
const USER_PROMPT = `You see all pages of one PDF, in order. Each image is preceded by a text marker "=== PAGE N ===" that tells you its exact page number. Use these markers to assign page numbers — never guess a page number.
Task: Decide whether this document contains a business brokerage "BUYER INFORMATION SHEET" form, and if so, transcribe its fields.
Task: Identify the page types by their headings, then transcribe form fields from a business brokerage "BUYER INFORMATION SHEET" package.
1. is_buyer_sheet: true only if a page with the heading "BUYER INFORMATION SHEET" exists. If the document is something else (notes, listing, letter, other form), return is_buyer_sheet=false and null for every field.
2. info_page: page number (1-based) of the BUYER INFORMATION SHEET page, else null.
3. ca_page: page number of the confidentiality agreement page containing text like "PROSPECTIVE BUYER AGREES TO KEEP AND HOLD CONFIDENTIAL", else null.
4. From the info page, transcribe VERBATIM (exactly as written, do not normalize, do not expand abbreviations):
- name_company: value of the "Name/Company" line
- prospective_buyer: value of the "Prospective Buyer" line
- company: value of a separate "Company" line if present
- phone, cell, email, address, state
STEP 1 — Identify each page by its UNIQUE marker text (pages can appear in ANY order):
- NOTES page: contains the text "Date NDA Scanned". It is a handwritten cover sheet with "Name:", "Date NDA Scanned:", a two-column table headed "Business Interested In:", and a free-text "Notes" area at the bottom. NOT every package has one.
- INFO SHEET page: contains the printed heading "BUYER INFORMATION SHEET" (usually with the BizMatch logo at the top).
- CA page: contains "CONFIDENTIALITY AGREEMENT" or "PROSPECTIVE BUYER AGREES TO KEEP AND HOLD CONFIDENTIAL".
Set notes_page, info_page, ca_page to the respective 1-based page numbers, or null if that page type is absent.
EXCLUSIVITY RULE — each page number may be assigned to AT MOST ONE of notes_page / info_page / ca_page. A single page is never two types at once. Decide by marker priority:
1. If the page shows "Date NDA Scanned" → it is the NOTES page. It is NEVER the info sheet, even if it is page 1 and even if it also lists businesses.
2. Else if it shows "BUYER INFORMATION SHEET" → info sheet.
3. Else if it shows the confidentiality wording → CA page.
So info_page and notes_page must be DIFFERENT numbers (or one of them null). If you were about to set them equal, you misread one — re-check which page carries "BUYER INFORMATION SHEET" versus "Date NDA Scanned".
STEP 2 — doc_type:
- "buyer_sheet": an INFO SHEET page exists.
- "ca_only": no info sheet, but a CA page exists.
- "other": none of the above — return null for every field and 0 for info_sheet_count.
2. info_sheet_count: how many separate INFO SHEET pages exist (some files contain two). 0 if none. If more than one, transcribe from the FIRST info sheet only.
STEP 3 — Transcribe VERBATIM (exactly as written, do not normalize or expand abbreviations), each field ONLY from the page type named:
5. From the INFO SHEET page (only if info_page is set):
- name_company: "Name/Company" line
- prospective_buyer: "Prospective Buyer" line
- company: separate "Company" line if present
- phone, cell, email
- address: the COMPLETE address on the "ADDRESS" line. The address field has TWO parts side by side: the left part is the street ("PO BOX / STREET"), the right part is the city/state/ZIP ("CITY / STATE / ZIP"). Transcribe BOTH parts as one full address, left to right (e.g. "2688 Grassina St. #631, San Jose, CA 95136"). Do NOT stop after the street — always include the city/state/ZIP part to the right, even if there is a wide gap between them or the right part is handwritten.
- state: the US state from the address (2-letter code if shown, e.g. "CA", "TX"), else null
- how_did_you_hear: "How did you hear about us"
- interested_in_updates: answer/checkbox for receiving updates (transcribe what is marked, e.g. "Yes" or "No"), else null
- types_of_business_raw: "Type(s) of business interested in" (may span multiple lines — join with a space)
- interested_in_updates: the marked updates answer/checkbox ("Yes"/"No"), else null
- types_of_business_raw: the "Type(s) of business interested in" list FROM THE INFO SHEET ONLY. This is usually a MULTI-LINE list with several entries stacked vertically (e.g. "RETAIL/TRADE", "FUEL STATIONS/CONVENIENCE STORES", "RESTAURANTS/BAKERY/CAFES", "TRANSPORT", "IT TELECOM"). Transcribe EVERY line of the list, not just the first one. Include entries even if they are struck through / crossed out (transcribe them as written). Join all entries with a comma in top-to-bottom order. This must come from the info sheet page, NEVER from the notes page.
- background_experience: "Background/Experience" (may span multiple lines)
- total_purchase_price: "Total Purchase Price" as written
- down_payment_raw: "Down Payment" as written (e.g. "$350,000", "1.5M", "TBD")
5. date_of_introduction_raw: the date written next to the buyer's signature on the confidentiality agreement page, verbatim (e.g. "6/25/26"), else null.
6. If doc_type is "ca_only": transcribe from the CA page — the prospective buyer's printed/signed name into prospective_buyer, plus address/phone/email if present. Everything else null.
7. date_of_introduction_raw: the "Date of Introduction" on the CA page (usually near the buyer's signature), transcribed EXACTLY as written including any spaces or separators (e.g. "09 / 13 / 2021", "6/25/26", "Sept 13 2021"). If a date is visible anywhere labeled "Date of Introduction", always return it verbatim — never leave it null just because the format looks unusual. Only null if truly no such date is present.
8. From the NOTES page (only if notes_page is set):
- notes_business_raw: the business name(s) written in the "Business Interested In" TABLE of the notes page, verbatim (join multiple with a comma). Take ONLY the table entries — do NOT include the free-text "Notes" area at the bottom of the page. If the table is empty, null.
The notes page and the info sheet are INDEPENDENT sources; never copy content from one into the other's field.
A blank field, "N/A", "n", or an empty line = transcribe it as written; if truly empty, use null. Return only the JSON object.`;
A blank field, "N/A", "n", or an empty line = transcribe it as written; if truly empty, use null.
Handwriting rules: Transcribe handwritten values letter by letter — do NOT complete them from context or from other fields. Email addresses are the most reliable spelling source on the page: read the email character by character, and if a handwritten name is ambiguous (e.g. B vs D), prefer the spelling that appears in the email address. Never alter the email itself to match your reading of the name.
Return only the JSON object.`;
async function fetchModelId(api: string): Promise<string> {
try {
@@ -309,10 +369,29 @@ async function fetchModelId(api: string): Promise<string> {
}
}
/** Wartet nach einem Server-Neustart (503/Verbindungsfehler), bis /health
* wieder OK meldet — Modell-Reload dauert 1-2 Minuten. */
async function waitForHealthy(api: string, timeoutMs: number): Promise<void> {
const healthUrl = api.replace(/\/v1\/?$/, "") + "/health";
const deadline = Date.now() + timeoutMs;
while (Date.now() < deadline) {
try {
const r = await fetch(healthUrl, { signal: AbortSignal.timeout(5000) });
if (r.ok) return;
} catch {
/* Server noch weg */
}
await new Promise((res) => setTimeout(res, 5000));
}
}
async function callVision(api: string, images: string[], timeoutMs: number): Promise<VisionRaw> {
const content: Array<Record<string, unknown>> = [{ type: "text", text: USER_PROMPT }];
for (const img of images) {
const b64 = await fsp.readFile(img, { encoding: "base64" });
// Vor jedes Bild einen Seiten-Marker setzen, damit das Modell zweifelsfrei
// weiss, welches Bild welche Seitennummer ist (loest Info/Notes-Verwechslung).
for (let i = 0; i < images.length; i++) {
const b64 = await fsp.readFile(images[i], { encoding: "base64" });
content.push({ type: "text", text: `=== PAGE ${i + 1} ===` });
content.push({ type: "image_url", image_url: { url: `data:image/png;base64,${b64}` } });
}
const body = {
@@ -353,23 +432,56 @@ async function callVision(api: string, images: string[], timeoutMs: number): Pro
// Merge: deterministischer Datensatz + Vision-Rohwerte → Zielstruktur
// ---------------------------------------------------------------------------
/**
* Zieht den Namen deterministisch aus dem Dateinamen (verlaessliche Quelle,
* unabhaengig von der Vision-Transkription). Schema: "Nachname, Vorname <datum> [Zusatz].pdf".
* Gibt "Nachname, Vorname" zurueck, oder null wenn nicht erkennbar.
* Beeinflusst die Bilderkennung NICHT — reine String-Operation.
*/
function nameFromFilename(fileName: string): string | null {
let s = fileName.replace(/\.pdf$/i, "");
// Klammer-Zusaetze wie "(Gordon Greve)" / "(JF Lehman)" entfernen
s = s.replace(/\([^)]*\)/g, " ").replace(/\s+/g, " ").trim();
// Schema ist "Nachname, Vorname ...". Nimm den Nachnamen (vor dem Komma)
// und aus dem Rest nur das erste Wort als Vorname - alles Weitere
// (Zweitnamen, Datum, Notes/Oilfield-Zusaetze) faellt weg.
const comma = s.indexOf(",");
if (comma < 0) {
// kein Komma: erstes Wort als Ganzes nehmen, ohne Datum/Zusatz
const first = s.replace(/\b\d{4,8}\b.*$/, "").trim();
return first.length >= 2 ? first : null;
}
const last = s.slice(0, comma).trim();
const rest = s.slice(comma + 1).trim();
// erstes Token des Rests = Vorname (stoppt vor Datum/Zusatzwort)
const firstName = (rest.match(/^[A-Za-zÀ-ÿ.'-]+/) || [""])[0];
if (!last || !firstName) return null;
return `${last}, ${firstName}`;
}
function mergeRecord(det: BuyerRecord, vis: VisionRaw, model: string): BuyerRecord {
const base: BuyerRecord = {
...det,
_parser: "vision",
_vision_model: model,
_vision_ts: new Date().toISOString(),
name_from_filename: nameFromFilename(det.file_name),
_vision_error: undefined,
};
delete (base as Record<string, unknown>)["_vision_error"];
if (!vis.is_buyer_sheet) {
return { ...base, is_buyer_sheet: false, _info_page: null, _ca_page: null };
if (vis.doc_type === "other") {
return { ...base, is_buyer_sheet: false, _doc_type: "other", _info_page: null, _ca_page: null, _notes_page: vis.notes_page ?? null };
}
// buyer_sheet ODER ca_only: alles übernehmen, was das Dokument hergibt
const dp = normDownPayment(vis.down_payment_raw);
return {
...base,
is_buyer_sheet: true,
is_buyer_sheet: vis.doc_type === "buyer_sheet",
_doc_type: vis.doc_type,
_info_sheet_count: vis.info_sheet_count ?? (vis.doc_type === "buyer_sheet" ? 1 : 0),
name_company: cleanStr(vis.name_company),
prospective_buyer: cleanStr(vis.prospective_buyer),
company: cleanStr(vis.company),
@@ -381,14 +493,19 @@ function mergeRecord(det: BuyerRecord, vis: VisionRaw, model: string): BuyerReco
how_did_you_hear: cleanStr(vis.how_did_you_hear),
interested_in_updates: cleanStr(vis.interested_in_updates),
types_of_business_raw: cleanStr(vis.types_of_business_raw),
notes_business_raw: cleanStr(vis.notes_business_raw),
background_experience: cleanStr(vis.background_experience),
total_purchase_price: cleanStr(vis.total_purchase_price),
down_payment: dp.value,
down_payment_raw: dp.raw,
// Vision-Datum bevorzugt; Fallback: deterministischer Wert (z.B. aus Dateinamen)
date_of_introduction: normDate(vis.date_of_introduction_raw) ?? (det.date_of_introduction as string | null) ?? null,
// Datum: ISO wenn parsebar, sonst ROHWERT behalten (Mitarbeiter korrigiert
// spaeter). Nie verwerfen, nur weil das Format ungewohnt ist.
date_of_introduction: normDate(vis.date_of_introduction_raw) ?? cleanStr(vis.date_of_introduction_raw) ?? (det.date_of_introduction as string | null) ?? null,
date_of_introduction_raw: cleanStr(vis.date_of_introduction_raw),
_info_page: vis.info_page,
_ca_page: vis.ca_page,
_notes_page: vis.notes_page,
};
}
@@ -399,6 +516,11 @@ function mergeRecord(det: BuyerRecord, vis: VisionRaw, model: string): BuyerReco
const RETRIES = 3;
function progress(line: string): void {
// Im Log-Modus (Pipe/tee): vollständige Zeilen mit Timestamp statt \r-Überschreiben
if (!process.stderr.isTTY) {
process.stderr.write(`[${new Date().toISOString()}] ${line}\n`);
return;
}
const cols = process.stderr.columns ?? 120;
process.stderr.write("\r" + line.slice(0, cols - 1).padEnd(cols - 1));
}
@@ -414,13 +536,10 @@ async function main(): Promise<void> {
process.exit(1);
}
const all: BuyerRecord[] = JSON.parse(await fsp.readFile(args.input, "utf8"));
let targets = all.filter((r) => r.is_buyer_sheet === false);
if (args.only) targets = targets.filter((r) => r.file_name === args.only);
if (args.limit > 0) targets = targets.slice(0, args.limit);
const visionPath = path.join(args.outDir, "buyers_vision.json");
const mergedPath = path.join(args.outDir, "buyers_merged.json");
await fsp.mkdir(args.outDir, { recursive: true });
// Resume-Stand laden (bereits verarbeitete Dokumente)
const done = new Map<string, BuyerRecord>();
try {
const prev: BuyerRecord[] = JSON.parse(await fsp.readFile(visionPath, "utf8"));
@@ -429,29 +548,89 @@ async function main(): Promise<void> {
/* kein Resume-Stand */
}
console.error(`Indexiere PDFs unter ${args.pdfRoot} ...`);
// ---------------------------------------------------------------------
// PHASE 1 — INDEX: alle PDFs scannen, file_name + _pages_total sofort
// ins buyers_vision.json eintragen (auch die zu grossen, dann markiert).
// ---------------------------------------------------------------------
console.error(`Phase 1: Indexiere PDFs unter ${args.pdfRoot} ...`);
const pdfIndex = await buildPdfIndex(args.pdfRoot);
console.error(`${pdfIndex.size} PDFs gefunden. ${targets.length} Einträge zu verarbeiten.`);
const names = [...pdfIndex.keys()].sort();
console.error(`${names.length} PDFs gefunden. Ermittle Seitenzahlen ...`);
let idx = 0;
for (const name of names) {
idx++;
// schon indexiert (mit gueltiger Seitenzahl)? dann nicht neu zaehlen
const existing = done.get(name);
if (existing && typeof existing["_pages_total"] === "number" && existing["_pages_total"]! >= 0 && !args.force) {
if (idx % 50 === 0) progress(`Index ${idx}/${names.length}`);
continue;
}
const pdfPath = pdfIndex.get(name)!;
const pages = await pdfPageCount(pdfPath);
const tooMany = pages < 0 ? false : pages > args.maxPages;
const prev = done.get(name) ?? { file_name: name, is_buyer_sheet: false };
done.set(name, {
...prev,
file_name: name,
_pages_total: pages,
...(pages < 0 ? { _index_error: "pdfinfo fehlgeschlagen (beschaedigt?)" } : {}),
...(tooMany ? { _skipped_too_many_pages: true } : {}),
});
if (idx % 25 === 0 || idx === names.length) progress(`Index ${idx}/${names.length}`);
}
// Index sofort persistieren, bevor die teure Phase 2 startet
await fsp.writeFile(visionPath, JSON.stringify([...done.values()], null, 2));
process.stderr.write("\n");
// ---------------------------------------------------------------------
// PHASE 2 — INHALT: nur PDFs <= maxPages, noch nicht (fehlerfrei) erledigt.
// ---------------------------------------------------------------------
let candidates = [...done.values()].filter((r) => {
const pages = r["_pages_total"] as number | undefined;
return typeof pages === "number" && pages > 0 && pages <= args.maxPages;
});
if (args.only) candidates = candidates.filter((r) => r.file_name === args.only);
// Resume-Skip VOR dem Limit: fehlerfrei mit echtem Vision-Ergebnis = fertig.
// Mit --reprocess-missing gelten Datensaetze OHNE Datum als unvollstaendig
// und werden erneut verarbeitet (fuer gezielte Nachlaeufe, ohne JSON-Editieren).
const isIncomplete = (r: Record<string, unknown>): boolean => {
if (!args.reprocessMissing) return false;
// buyer_sheet/ca_only ohne Datum gilt als unvollstaendig
const dt = r["_doc_type"];
if (dt === "other") return false;
const hasDate = r["date_of_introduction"] != null && r["date_of_introduction"] !== "";
return !hasDate;
};
const pending = candidates.filter((r) => {
const prev = done.get(r.file_name);
const hasResult = prev && prev["_parser"] === "vision" && !prev["_vision_error"];
if (hasResult && prev && isIncomplete(prev)) return true; // unvollstaendig -> neu
return !(hasResult && !args.force);
});
const skipped = candidates.length - pending.length;
const targets = args.limit > 0 ? pending.slice(0, args.limit) : pending;
const tooManyCount = [...done.values()].filter((r) => r["_skipped_too_many_pages"]).length;
console.error(
`Phase 2: ${candidates.length} Kandidaten (<=${args.maxPages} Seiten), ` +
`${tooManyCount} zu gross (uebersprungen), ${skipped} bereits verarbeitet, ` +
`${targets.length} in diesem Lauf.`
);
const model = await fetchModelId(args.api);
console.error(`Modell: ${model} @ ${args.api}`);
let ok = 0, notSheet = 0, errors = 0, skipped = 0;
let ok = 0, caOnly = 0, notSheet = 0, errors = 0;
for (let i = 0; i < targets.length; i++) {
const det = targets[i];
const tag = `[${i + 1}/${targets.length}] ${det.file_name}`;
const prev = done.get(det.file_name);
if (prev && !prev["_vision_error"] && !args.force) {
skipped++;
progress(`${tag} … übersprungen (bereits verarbeitet)`);
continue;
}
const pdfPath = pdfIndex.get(det.file_name);
if (!pdfPath) {
done.set(det.file_name, { ...det, _vision_error: "PDF nicht gefunden" });
done.set(det.file_name, { ...det, _vision_error: "PDF nicht gefunden", _vision_ts: new Date().toISOString() });
errors++;
progress(`${tag} … FEHLER: PDF nicht gefunden`);
continue;
@@ -459,8 +638,11 @@ async function main(): Promise<void> {
const tmpDir = await fsp.mkdtemp(path.join(os.tmpdir(), "bvs-"));
try {
progress(`${tag} … rendere`);
const images = await renderPdf(pdfPath, args.dpi, args.maxPages, tmpDir);
const dpi = args.dpi === "auto"
? ((await hasTextLayer(pdfPath, args.maxPages)) ? 150 : 200)
: args.dpi;
progress(`${tag} … rendere (${dpi} dpi)`);
const images = await renderPdf(pdfPath, dpi, args.maxPages, tmpDir);
let vis: VisionRaw | null = null;
let lastErr = "";
@@ -471,23 +653,32 @@ async function main(): Promise<void> {
break;
} catch (e) {
lastErr = e instanceof Error ? e.message : String(e);
if (attempt < RETRIES) await new Promise((res) => setTimeout(res, 5000 * attempt));
if (attempt < RETRIES) {
if (/HTTP 50[23]|fetch failed|aborted|ECONNREFUSED|ECONNRESET/i.test(lastErr)) {
// Server crasht/lädt neu → auf /health warten (Modell-Reload dauert)
progress(`${tag} … Server neu am Laden, warte auf /health`);
await waitForHealthy(args.api, 300_000);
} else {
await new Promise((res) => setTimeout(res, 5000 * attempt));
}
}
}
}
if (!vis) {
done.set(det.file_name, { ...det, _vision_error: lastErr });
done.set(det.file_name, { ...det, _vision_error: lastErr, _vision_ts: new Date().toISOString() });
errors++;
progress(`${tag} … FEHLER: ${lastErr}`);
} else {
const merged = mergeRecord(det, vis, model);
done.set(det.file_name, merged);
if (merged.is_buyer_sheet) { ok++; progress(`${tag}OK`); }
if (merged["_doc_type"] === "ca_only") { caOnly++; progress(`${tag}CA only`); }
else if (merged.is_buyer_sheet) { ok++; progress(`${tag} … OK`); }
else { notSheet++; progress(`${tag} … kein Buyer Sheet`); }
}
} catch (e) {
const msg = e instanceof Error ? e.message : String(e);
done.set(det.file_name, { ...det, _vision_error: msg });
done.set(det.file_name, { ...det, _vision_error: msg, _vision_ts: new Date().toISOString() });
errors++;
progress(`${tag} … FEHLER: ${msg}`);
} finally {
@@ -495,19 +686,16 @@ async function main(): Promise<void> {
}
// Inkrementell sichern (Resume-fähig)
await fsp.mkdir(args.outDir, { recursive: true });
await fsp.writeFile(visionPath, JSON.stringify([...done.values()], null, 2));
}
// Merge: Gleis A + Gleis B
const merged = all.map((r) => done.get(r.file_name) ?? r);
await fsp.writeFile(mergedPath, JSON.stringify(merged, null, 2));
await fsp.writeFile(visionPath, JSON.stringify([...done.values()], null, 2));
process.stderr.write("\n");
console.error(
`Fertig. OK: ${ok}, kein Buyer Sheet: ${notSheet}, Fehler: ${errors}, übersprungen: ${skipped}`
`Fertig. OK: ${ok}, CA only: ${caOnly}, kein Buyer Sheet: ${notSheet}, Fehler: ${errors}, übersprungen: ${skipped}`
);
console.error(`${visionPath}\n→ ${mergedPath}`);
console.error(`${visionPath}`);
}
main().catch((e) => {