Compare commits

..

3 Commits

Author SHA1 Message Date
0989c06b8f gfdfg 2026-07-12 12:30:20 -05:00
888c5c1543 dfgdfg 2026-07-11 17:08:41 -05:00
b8c862885e typescript ansatz 2026-07-11 15:56:14 -05:00
27 changed files with 3672 additions and 28 deletions

5
.gitignore vendored
View File

@@ -1 +1,4 @@
poc_out
poc_out
node_modules
package-lock.json
*.jsonl

333
BuyerSheetParser.ts Normal file
View File

@@ -0,0 +1,333 @@
import * as pdfjsLib from 'pdfjs-dist/legacy/build/pdf.js';
import * as fs from 'fs';
export interface BuyerSheetData {
file_name: string; // NEU
is_buyer_sheet: boolean;
name_company: string | null;
prospective_buyer: string | null;
company: string | null;
phone: string | null;
cell: string | null;
email: string | null;
address: string | null;
state: string | null;
how_did_you_hear: string | null;
interested_in_updates: boolean | null;
types_of_business_raw: string | null;
background_experience: string | null;
total_purchase_price: string | null;
down_payment: string | null;
down_payment_raw: string | null;
date_of_introduction: string | null;
_checkbox_pending: boolean;
_parser: string;
_info_page: number | null;
_ca_page: number | null;
}
const LABELS = [
"NAME / COMPANY", "PHONE", "ADDRESS", "EMAIL ADDRESS",
"HOW DID YOU HEAR", "ARE YOU INTERESTED", "TYPES OF BUSINESSES",
"BACKGROUND", "TOTAL PURCHASE PRICE", "DOWN PAYMENT",
"INCOME REQUIREMENTS", "ACCOUNTANT", "ATTORNEY", "BANK",
];
export class DeterministicParser {
public async parsePdf(filePath: string, fileName: string): Promise<BuyerSheetData | null> {
const dataBuffer = fs.readFileSync(filePath);
const loadingTask = pdfjsLib.getDocument({ data: new Uint8Array(dataBuffer) });
const pdfDocument = await loadingTask.promise;
// Regel: Alles über 10 Seiten wird radikal ignoriert
if (pdfDocument.numPages > 10) {
return null;
}
let infoPageNum: number | null = null;
let caPageNum: number | null = null;
let infoPageObj: any = null;
let infoPageTextContent: any = null;
// ==========================================
// 1. Dynamische Seitensuche (Visuell sortiert & kugelsicher)
// ==========================================
for (let i = 1; i <= pdfDocument.numPages; i++) {
const page = await pdfDocument.getPage(i);
const textContent = await page.getTextContent();
// Elemente mit Koordinaten versehen und wie ein Mensch lesen (von oben nach unten, links nach rechts)
const sortedItems = textContent.items
.map((item: any) => ({
text: item.str,
x: item.transform[4],
y: item.transform[5]
}))
.sort((a: any, b: any) => {
// Y-Toleranz für Buchstaben auf derselben Zeile
if (Math.abs(b.y - a.y) > 5) {
return b.y - a.y;
}
return a.x - b.x;
});
// Wir werfen ALLE Leerzeichen, Striche, Punkte und unsichtbare Artefakte weg.
// Übrig bleibt eine reine, unverwüstliche Buchstabenkette.
const textRaw = sortedItems.map((i: any) => i.text).join('').toUpperCase().replace(/[^A-Z]/g, '');
// Anker-Suche in der sauberen Zeichenkette
if (textRaw.includes('BUYERINFORMATIONSHEET')) {
infoPageNum = i;
infoPageObj = page;
infoPageTextContent = textContent;
}
if (textRaw.includes('PROSPECTIVEBUYERAGREESTOKEEPANDHOLDCONFIDENTIAL')) {
caPageNum = i;
}
}
// ==========================================
// 2. Fallback für Bild/Scan PDFs
// ==========================================
if (!infoPageNum || !infoPageObj || !infoPageTextContent) {
return this.createEmptyFallback(fileName, filePath, infoPageNum, caPageNum);
}
// ==========================================
// 3. Werte-Extraktion (nur auf der Info-Seite!)
// ==========================================
const mappedItems = infoPageTextContent.items
.filter((item: any) => item.str.trim() !== '')
.map((item: any) => ({
text: item.str,
x: item.transform[4],
y: item.transform[5],
width: item.width
}))
.sort((a: any, b: any) => b.y - a.y || a.x - b.x);
const linesGrouped: { y: number, items: any[] }[] = [];
let currentY: number | null = null;
let currentItems = [];
for (const item of mappedItems) {
if (currentY === null || Math.abs(item.y - currentY) <= 3) {
currentItems.push(item);
currentY = currentY === null ? item.y : currentY;
} else {
linesGrouped.push({ y: currentY, items: currentItems });
currentItems = [item];
currentY = item.y;
}
}
if (currentItems.length > 0 && currentY !== null) {
linesGrouped.push({ y: currentY, items: currentItems });
}
const lines: { y: number, text: string }[] = [];
for (const group of linesGrouped) {
group.items.sort((a, b) => a.x - b.x);
let lineStr = "";
let prevEnd = -1;
for (const item of group.items) {
if (prevEnd !== -1) {
const gap = item.x - prevEnd;
if (gap > 4) lineStr += " ";
}
lineStr += item.text;
prevEnd = item.x + item.width;
}
lines.push({ y: group.y, text: lineStr });
}
const rawResults: Record<string, string[]> = {};
LABELS.forEach(lbl => rawResults[lbl] = []);
let currentLabel: string | null = null;
const sortedLabels = [...LABELS].sort((a, b) => b.length - a.length);
for (const line of lines) {
const textUpper = line.text.toUpperCase().replace(/\s+/g, '');
if (textUpper.includes("SELLERMAYREQUIREVERIFICATION")) continue;
let foundLabel: string | null = null;
for (const lbl of sortedLabels) {
const lblClean = lbl.toUpperCase().replace(/\s+/g, '');
if (textUpper.includes(lblClean)) {
foundLabel = lbl;
break;
}
}
if (foundLabel) {
currentLabel = foundLabel;
if (line.text.includes(':')) {
const parts = line.text.split(':');
const val = parts.slice(1).join(':').replace(/_+/g, '').trim();
if (val.length > 0) rawResults[currentLabel].push(val);
}
} else if (currentLabel) {
const cleanVal = line.text.replace(/_+/g, '').trim();
if (cleanVal.length > 0 && !cleanVal.includes("Doc ID") && !cleanVal.includes("Bizmatch")) {
rawResults[currentLabel].push(cleanVal);
}
}
}
const visualCheckboxResult = await this.detectGraphicCheckbox(infoPageObj, mappedItems);
return this.mapToTargetStructure(rawResults, filePath, fileName, visualCheckboxResult, infoPageNum, caPageNum);
}
private createEmptyFallback(fileName: string, filePath: string, infoPage: number | null, caPage: number | null): BuyerSheetData {
let dateOfIntro = null;
const dateMatch = filePath.match(/(\d{2})(\d{2})(\d{2})\.pdf$/);
if (dateMatch) dateOfIntro = `20${dateMatch[3]}-${dateMatch[1]}-${dateMatch[2]}`;
return {
file_name: fileName,
is_buyer_sheet: false,
name_company: null,
prospective_buyer: null,
company: null,
phone: null,
cell: null,
email: null,
address: null,
state: null,
how_did_you_hear: null,
interested_in_updates: null,
types_of_business_raw: null,
background_experience: null,
total_purchase_price: null,
down_payment: null,
down_payment_raw: null,
date_of_introduction: dateOfIntro,
_checkbox_pending: true,
_parser: "deterministic", // Signal für den AI-Pass
_info_page: infoPage,
_ca_page: caPage
};
}
private mapToTargetStructure(
raw: Record<string, string[]>,
filePath: string,
fileName: string,
visualCheckboxResult: boolean | null,
infoPageNum: number | null,
caPageNum: number | null
): BuyerSheetData {
const getVal = (label: string) => raw[label] && raw[label].length > 0 ? raw[label].join(' ') : null;
const nameCompany = getVal("NAME / COMPANY");
let phoneRaw = getVal("PHONE");
let phoneClean = null;
let cellClean = null;
if (phoneRaw) {
const phoneMatch = phoneRaw.split(/FAX:|CELL:/)[0];
phoneClean = phoneMatch ? phoneMatch.trim() : null;
const cellMatch = phoneRaw.match(/CELL:\s*(.*)/);
cellClean = cellMatch && cellMatch[1] ? cellMatch[1].trim() : null;
}
let addressClean = getVal("ADDRESS");
if (addressClean) {
addressClean = addressClean.replace(/PO BOX \/ STREET\s+CITY \/ STATE \/ ZIP/g, '').trim();
if (addressClean === '') addressClean = null;
}
const interestedRaw = getVal("ARE YOU INTERESTED");
let interestedInUpdates: boolean | null = null;
let checkboxPending = true;
if (interestedRaw) {
if (/(✔|☑|X|✓)\s*YES/i.test(interestedRaw) || /YES\s*(✔|☑|X|✓)/i.test(interestedRaw)) {
interestedInUpdates = true;
checkboxPending = false;
} else if (/(✔|☑|X|✓)\s*NO/i.test(interestedRaw) || /NO\s*(✔|☑|X|✓)/i.test(interestedRaw)) {
interestedInUpdates = false;
checkboxPending = false;
}
}
if (checkboxPending && visualCheckboxResult !== null) {
interestedInUpdates = visualCheckboxResult;
checkboxPending = false;
}
const downPaymentRaw = getVal("DOWN PAYMENT")?.replace(/[^0-9,]/g, '') || null;
const downPaymentClean = downPaymentRaw ? downPaymentRaw.replace(/,/g, '') : null;
let dateOfIntro = null;
const dateMatch = filePath.match(/(\d{2})(\d{2})(\d{2})\.pdf$/);
if (dateMatch) dateOfIntro = `20${dateMatch[3]}-${dateMatch[1]}-${dateMatch[2]}`;
return {
file_name: fileName,
is_buyer_sheet: true,
name_company: nameCompany,
prospective_buyer: nameCompany ? nameCompany.split('/')[0].trim() : null,
company: null,
phone: phoneClean,
cell: cellClean,
email: getVal("EMAIL ADDRESS"),
address: addressClean,
state: null,
how_did_you_hear: getVal("HOW DID YOU HEAR"),
interested_in_updates: interestedInUpdates,
types_of_business_raw: getVal("TYPES OF BUSINESSES"),
background_experience: getVal("BACKGROUND"),
total_purchase_price: getVal("TOTAL PURCHASE PRICE"),
down_payment: downPaymentClean,
down_payment_raw: downPaymentRaw,
date_of_introduction: dateOfIntro,
_checkbox_pending: checkboxPending,
_parser: "deterministic",
_info_page: infoPageNum,
_ca_page: caPageNum
};
}
private async detectGraphicCheckbox(page: any, textItems: any[]): Promise<boolean | null> {
let yesX = null, noX = null, targetY = null;
for (const item of textItems) {
const textUpper = item.text.toUpperCase().trim();
if (textUpper.includes("YES")) { yesX = item.x; targetY = item.y; }
}
for (const item of textItems) {
const textUpper = item.text.toUpperCase().trim();
if (textUpper.includes("NO") && targetY !== null && Math.abs(item.y - targetY) < 5) noX = item.x;
}
if (yesX === null || noX === null || targetY === null) return null;
const opList = await page.getOperatorList();
let currentTransform = [1, 0, 0, 1, 0, 0];
for (let i = 0; i < opList.fnArray.length; i++) {
const fn = opList.fnArray[i];
const args = opList.argsArray[i];
if (fn === pdfjsLib.OPS.transform) currentTransform = args;
else if (
fn === pdfjsLib.OPS.paintImageXObject ||
fn === pdfjsLib.OPS.paintInlineImageXObject ||
fn === pdfjsLib.OPS.paintJpegXObject
) {
const width = Math.abs(currentTransform[0]);
const height = Math.abs(currentTransform[3]);
const imgX = currentTransform[4];
const imgY = currentTransform[5];
if (width < 40 && height < 40 && Math.abs(imgY - targetY) < 50) {
const distToYes = Math.abs(imgX - yesX);
const distToNo = Math.abs(imgX - noX);
return distToYes < distToNo;
}
}
}
return null;
}
}

101
batch_runner.ts Normal file
View File

@@ -0,0 +1,101 @@
import * as fs from 'fs';
import * as path from 'path';
import { DeterministicParser } from './BuyerSheetParser';
async function main() {
// 1. Argument-Check
const args = process.argv.slice(2);
// Verzeichnis Parameter
const dirArgIndex = args.indexOf('--dir');
if (dirArgIndex === -1 || !args[dirArgIndex + 1]) {
console.error("Fehler: Bitte ein Verzeichnis angeben!");
console.error("Nutzung: npx tsx batch_runner.ts --dir \"/Pfad/zum/Ordner\" [--limit 10]");
process.exit(1);
}
const dirPath = args[dirArgIndex + 1];
// Limit Parameter
const limitArgIndex = args.indexOf('--limit');
let limit = -1; // -1 bedeutet: Kein Limit, alle verarbeiten
if (limitArgIndex !== -1 && args[limitArgIndex + 1]) {
limit = parseInt(args[limitArgIndex + 1], 10);
}
if (!fs.existsSync(dirPath) || !fs.statSync(dirPath).isDirectory()) {
console.error(`Fehler: Der Pfad "${dirPath}" existiert nicht oder ist kein Verzeichnis.`);
process.exit(1);
}
// 2. PDFs finden und Metadaten (für die Sortierung) auslesen
const filesWithStats = fs.readdirSync(dirPath)
.filter(f => f.toLowerCase().endsWith('.pdf'))
.map(file => {
const fullPath = path.join(dirPath, file);
return {
file,
fullPath,
// Änderungsdatum der Datei auslesen (in Millisekunden)
mtime: fs.statSync(fullPath).mtimeMs
};
});
if (filesWithStats.length === 0) {
console.log("Keine PDFs in diesem Verzeichnis gefunden.");
process.exit(0);
}
// 3. Absteigend sortieren (Neueste zuerst)
filesWithStats.sort((a, b) => b.mtime - a.mtime);
// 4. Limit anwenden (falls gesetzt)
const filesToProcess = limit > 0 ? filesWithStats.slice(0, limit) : filesWithStats;
console.log(`Starte Verarbeitung: ${filesToProcess.length} von ${filesWithStats.length} PDFs ausgewählt (Sortierung: Neueste zuerst)...\n`);
const parser = new DeterministicParser();
const finalResults = [];
let skippedCounter = 0;
// 5. PDFs iterieren
for (const fileObj of filesToProcess) {
const { file, fullPath } = fileObj;
process.stdout.write(`-> Verarbeite: ${file} ... `);
try {
const data = await parser.parsePdf(fullPath, file);
if (data === null) {
console.log("ÜBERSPRUNGEN (> 10 Seiten)");
skippedCounter++;
} else if (data.is_buyer_sheet === false) {
console.log("IMAGE SCAN (Zuweisung an Vision AI)");
finalResults.push(data);
} else {
console.log("ERFOLGREICH");
finalResults.push(data);
}
} catch (error) {
console.log("FEHLER BEIM PARSEN");
console.error(error);
}
}
// 6. JSON Export unter ./out/buyers.json
const outDir = path.join(process.cwd(), 'out');
if (!fs.existsSync(outDir)) {
fs.mkdirSync(outDir);
}
const outPath = path.join(outDir, 'buyers.json');
fs.writeFileSync(outPath, JSON.stringify(finalResults, null, 2), 'utf-8');
console.log(`\n=================================================`);
console.log(`Zusammenfassung:`);
console.log(` Verarbeitet: ${finalResults.length}`);
console.log(` Verworfen (>10 Seiten): ${skippedCounter}`);
console.log(` Export gespeichert in: ${outPath}`);
console.log(`=================================================`);
}
main();

View File

@@ -0,0 +1,35 @@
# llama-server für Qwen3.6 Vision — NVIDIA RTX PRO 6000 Blackwell (sm_120)
#
# WICHTIG:
# - Muss aus aktuellem Source gebaut werden: Qwen3.6 nutzt ein neues
# Rope-Encoding (rope.dimension_sections 3 statt 4), alte Images/Builds
# brechen mit "wrong array length".
# - Blackwell braucht CUDA >= 12.8. KEIN CUDA 13.2 verwenden — erzeugt
# mit Qwen3.6 Gibberish (bekannter NVIDIA-Bug).
FROM nvidia/cuda:12.8.1-devel-ubuntu24.04 AS build
RUN apt-get update && apt-get install -y --no-install-recommends \
git cmake build-essential libcurl4-openssl-dev \
&& rm -rf /var/lib/apt/lists/*
RUN git clone --depth 1 https://github.com/ggml-org/llama.cpp /src
# 120 = Blackwell (RTX PRO 6000). Für andere Karten anpassen.
RUN cmake /src -B /build \
-DGGML_CUDA=ON \
-DCMAKE_CUDA_ARCHITECTURES=120 \
-DBUILD_SHARED_LIBS=OFF \
-DLLAMA_CURL=ON \
&& cmake --build /build --config Release -j --target llama-server
FROM nvidia/cuda:12.8.1-runtime-ubuntu24.04
RUN apt-get update && apt-get install -y --no-install-recommends \
libcurl4 libgomp1 curl \
&& rm -rf /var/lib/apt/lists/*
COPY --from=build /build/bin/llama-server /usr/local/bin/llama-server
EXPOSE 8000
ENTRYPOINT ["llama-server"]

View File

@@ -0,0 +1,32 @@
# llama-server für Qwen3.6 Vision — AMD Radeon AI Pro R9700, Vulkan-Backend
#
# Muss aus aktuellem Source gebaut werden (Qwen3.6-Rope-Änderung, s. Dockerfile.cuda).
# Vulkan statt ROCm — hat sich bei Vision als stabiler erwiesen.
FROM ubuntu:24.04 AS build
RUN apt-get update && apt-get install -y --no-install-recommends \
git cmake build-essential libcurl4-openssl-dev \
libvulkan-dev glslc \
&& rm -rf /var/lib/apt/lists/*
RUN git clone --depth 1 https://github.com/ggml-org/llama.cpp /src
RUN cmake /src -B /build \
-DGGML_VULKAN=ON \
-DBUILD_SHARED_LIBS=OFF \
-DLLAMA_CURL=ON \
&& cmake --build /build --config Release -j --target llama-server
FROM ubuntu:24.04
# mesa-vulkan-drivers = RADV-Treiber im Container (GPU via /dev/dri durchgereicht)
RUN apt-get update && apt-get install -y --no-install-recommends \
libvulkan1 mesa-vulkan-drivers vulkan-tools \
libcurl4 libgomp1 curl \
&& rm -rf /var/lib/apt/lists/*
COPY --from=build /build/bin/llama-server /usr/local/bin/llama-server
EXPOSE 8000
ENTRYPOINT ["llama-server"]

View File

@@ -1,47 +1,140 @@
# llama.cpp + Gemma-4-12B (Vision) auf CUDA (AWS G7e / RTX PRO 6000).
# llama-server für Qwen3.6 Vision — zwei Profile:
# docker compose --profile cuda up -d --build (EC2, RTX PRO 6000 Blackwell)
# docker compose --profile vulkan up -d --build (inhouse, Radeon AI Pro R9700)
#
# Bewusst SCHLANK gehalten: keine ROCm/Vulkan-Workarounds. Wir testen, ob
# CUDA die Instabilitaeten von vornherein vermeidet. Nur die inhaltlich noetigen
# Flags (--reasoning off gegen leeres content) bleiben.
# Modelle nach ./models legen (Haupt-GGUF + zugehöriger mmproj aus demselben Repo!):
# CUDA: unsloth/Qwen3.6-27B-GGUF → Qwen3.6-27B-UD-Q4_K_XL.gguf + mmproj-BF16.gguf
# Vulkan: unsloth/Qwen3.6-35B-A3B-GGUF → Qwen3.6-35B-A3B-UD-Q4_K_XL.gguf + mmproj-BF16.gguf
#
# Voraussetzung: DLAMI mit NVIDIA Container Toolkit (docker --gpus all funktioniert).
#
# Start: docker compose -f docker-compose-cuda.yml up -d
# Logs: docker compose -f docker-compose-cuda.yml logs -f
# Bewusst KEIN MTP: --mmproj + MTP ist laut Unsloth nicht unterstützt und hat
# offene OOM-/Hänger-Bugs (llama.cpp #23371, #23430). Für Batch-Extraktion
# irrelevant, da die Zeit im Image-Encoding steckt, nicht in der Generierung.
services:
llamacpp-gemma12b:
image: ghcr.io/ggml-org/llama.cpp:server-cuda
container_name: llamacpp-gemma12b
restart: unless-stopped
init: true
# GPU-Zugriff ueber das NVIDIA Container Toolkit
llama-cuda:
profiles: ["cuda"]
build:
context: .
dockerfile: Dockerfile.cuda
ports:
- "8000:8000"
volumes:
- ./models:/models
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: 1
count: all
capabilities: [gpu]
ports:
- "8000:8080"
volumes:
- ~/.cache/llama.cpp:/root/.cache/llama.cpp
command:
- -hf
- unsloth/gemma-4-12b-it-GGUF:UD-Q4_K_XL
- -m
- /models/Qwen3.6-27B-UD-Q4_K_XL.gguf
- --mmproj
- /models/mmproj-BF16.gguf
- --host
- 0.0.0.0
- --port
- "8080"
- "8000"
- --alias
- qwen3.6
- -ngl
- "99"
- --ctx-size
- "16384"
- -fa
- "on"
- -c
- "32768"
- --parallel
- "1"
- --jinja
- --reasoning
- "off"
# Erlernte Stabilitäts-Settings (Vision + Checkpoints = OOM-Bug):
- --ctx-checkpoints
- "0"
# Qwen-VL braucht min. 1024 Image-Tokens für korrektes Grounding:
- --image-min-tokens
- "1024"
- --image-max-tokens
- "4096"
# Extraktion: niedrige Temperatur, kein Repeat-Penalty
- --temp
- "0.1"
- --top-p
- "0.95"
- --top-k
- "20"
- --repeat-penalty
- "1.0"
healthcheck:
test: ["CMD", "curl", "-sf", "http://localhost:8000/health"]
interval: 15s
timeout: 5s
retries: 40
start_period: 60s
restart: unless-stopped
llama-vulkan:
profiles: ["vulkan"]
build:
context: .
dockerfile: Dockerfile.vulkan
ports:
- "8000:8000"
volumes:
- ./models:/models
devices:
- /dev/dri:/dev/dri
- /dev/kfd:/dev/kfd
group_add:
- video
- render
security_opt:
- seccomp:unconfined
command:
- -m
- /models/Qwen3.6-35B-A3B-UD-Q4_K_XL.gguf
- --mmproj
- /models/mmproj-BF16.gguf
- --host
- 0.0.0.0
- --port
- "8000"
- --alias
- gemma-4-12b
- qwen3.6
- -ngl
- "99"
- -fa
- "on"
- -c
- "32768"
- --parallel
- "1"
# KV-Cache quantisieren — 32GB VRAM, 35B-A3B + mmproj + Bildkontext:
- -ctk
- q8_0
- -ctv
- q8_0
- --jinja
- --ctx-checkpoints
- "0"
- --image-min-tokens
- "1024"
- --image-max-tokens
- "4096"
- --temp
- "0.1"
- --top-p
- "0.95"
- --top-k
- "20"
- --repeat-penalty
- "1.0"
# NOTFALL-Fallback bei Vision-Hängern/OOM unter RADV — mmproj auf CPU,
# Bildverarbeitung wird DEUTLICH langsamer. Nur aktivieren wenn nötig:
# - --no-mmproj-offload
healthcheck:
test: ["CMD", "curl", "-sf", "http://localhost:8000/health"]
interval: 15s
timeout: 5s
retries: 40
start_period: 120s
restart: unless-stopped

48
debug_text.ts Normal file
View File

@@ -0,0 +1,48 @@
import * as pdfjsLib from 'pdfjs-dist/legacy/build/pdf.js';
import * as fs from 'fs';
async function debugText(filePath: string) {
console.log(`\nLese PDF für Text-Debugging: ${filePath}`);
const dataBuffer = fs.readFileSync(filePath);
const loadingTask = pdfjsLib.getDocument({ data: new Uint8Array(dataBuffer) });
const pdfDocument = await loadingTask.promise;
console.log(`Anzahl Seiten: ${pdfDocument.numPages}`);
// Wir nehmen uns Seite 1 vor
const page = await pdfDocument.getPage(1);
const textContent = await page.getTextContent();
console.log(`\n--- RAW TEXT ITEMS (Die ersten 40 Fragmente) ---`);
for (let i = 0; i < Math.min(40, textContent.items.length); i++) {
const item = textContent.items[i] as any;
console.log(`Y: ${item.transform[5].toFixed(1).padStart(6)} | X: ${item.transform[4].toFixed(1).padStart(6)} | Text: "${item.str}"`);
}
// Unsere Sortier- und Bereinigungslogik anwenden
const sortedItems = textContent.items
.map((item: any) => ({
text: item.str,
x: item.transform[4],
y: item.transform[5]
}))
.sort((a: any, b: any) => {
if (Math.abs(b.y - a.y) > 5) return b.y - a.y;
return a.x - b.x;
});
const textRaw = sortedItems.map((i: any) => i.text).join('').toUpperCase().replace(/[^A-Z]/g, '');
console.log(`\n--- BEREINIGTER SUCH-STRING (Die ersten 200 Zeichen) ---`);
console.log(textRaw.substring(0, 200));
console.log(`\nEnthält 'BUYERINFORMATIONSHEET'? -> ${textRaw.includes('BUYERINFORMATIONSHEET')}`);
}
const args = process.argv.slice(2);
const pdfArgIndex = args.indexOf('--pdf');
if (pdfArgIndex !== -1 && args[pdfArgIndex + 1]) {
debugText(args[pdfArgIndex + 1]).catch(console.error);
} else {
console.error("Bitte --pdf Parameter angeben.");
}

88
find_images.ts Normal file
View File

@@ -0,0 +1,88 @@
import * as pdfjsLib from 'pdfjs-dist/legacy/build/pdf.js';
import * as fs from 'fs';
async function analyzeImages(filePath: string) {
console.log(`\nLese PDF für Bild-Analyse: ${filePath}`);
const dataBuffer = fs.readFileSync(filePath);
const loadingTask = pdfjsLib.getDocument({ data: new Uint8Array(dataBuffer) });
const pdfDocument = await loadingTask.promise;
// Wir analysieren nur Seite 1
const page = await pdfDocument.getPage(1);
const viewport = page.getViewport({ scale: 1.0 });
console.log(`Seitengröße: Breite = ${viewport.width.toFixed(2)}, Höhe = ${viewport.height.toFixed(2)}\n`);
const opList = await page.getOperatorList();
// Um die Koordinaten zu berechnen, müssen wir den Grafik-Status (Stack) verfolgen
let transformStack: number[][] = [[1, 0, 0, 1, 0, 0]]; // Standard-Matrix
let currentTransform = [1, 0, 0, 1, 0, 0];
let imageCount = 0;
console.log("--- GEFUNDENE BILDER AUF SEITE 1 ---");
for (let i = 0; i < opList.fnArray.length; i++) {
const fn = opList.fnArray[i];
const args = opList.argsArray[i];
// Status speichern (q)
if (fn === pdfjsLib.OPS.save) {
transformStack.push([...currentTransform]);
}
// Status wiederherstellen (Q)
else if (fn === pdfjsLib.OPS.restore) {
if (transformStack.length > 0) {
currentTransform = transformStack.pop()!;
}
}
// Transformation anwenden (cm) - In PDFs werden Matrizen multipliziert,
// für die einfache Bildanalyse reicht oft der letzte Transform-Befehl vor dem Bild,
// da dieser die Skalierung (Breite/Höhe) und Position (X/Y) des 1x1 Pixel Objekts setzt.
else if (fn === pdfjsLib.OPS.transform) {
currentTransform = args;
}
// Wenn ein Bild gezeichnet wird!
else if (
fn === pdfjsLib.OPS.paintImageXObject ||
fn === pdfjsLib.OPS.paintInlineImageXObject ||
fn === pdfjsLib.OPS.paintJpegXObject
) {
imageCount++;
// In der PDF-Matrix:
// currentTransform[0] = Skalierung X (entspricht der Bild-Breite)
// currentTransform[3] = Skalierung Y (entspricht der Bild-Höhe)
// currentTransform[4] = Position X
// currentTransform[5] = Position Y
const width = currentTransform[0];
const height = currentTransform[3];
const x = currentTransform[4];
const y = currentTransform[5];
const isPageFilling = (width >= viewport.width - 10 && height >= viewport.height - 10);
const mark = isPageFilling ? " <-- SEITENFÜLLENDER SCAN / HINTERGRUND" : "";
console.log(`Bild ${imageCount}:`);
console.log(` Position: X = ${x.toFixed(2)}, Y = ${y.toFixed(2)}`);
console.log(` Größe: Breite = ${width.toFixed(2)}, Höhe = ${height.toFixed(2)}${mark}`);
console.log(` Interner Name: ${args[0]}`);
console.log(`--------------------------------------------------`);
}
}
if (imageCount === 0) {
console.log("Keine Bilder gefunden. Das Dokument besteht rein aus Vektoren und Text.");
} else {
console.log(`\nInsgesamt ${imageCount} Bild(er) gefunden.`);
}
}
// Aufruf mit dem Pfad aus den Parametern
const args = process.argv.slice(2);
const pdfArgIndex = args.indexOf('--pdf');
if (pdfArgIndex !== -1 && args[pdfArgIndex + 1]) {
analyzeImages(args[pdfArgIndex + 1]).catch(console.error);
} else {
console.error("Bitte --pdf Parameter angeben.");
}

2354
out/buyers.json Normal file

File diff suppressed because it is too large Load Diff

13
package.json Normal file
View File

@@ -0,0 +1,13 @@
{
"dependencies": {
"canvas": "^3.2.3",
"pdf-parse": "^1.1.1",
"pdfjs-dist": "^3.11.174"
},
"devDependencies": {
"@types/node": "^26.1.1",
"@types/pdf-parse": "^1.1.5",
"ts-node": "^10.9.2",
"typescript": "^7.0.2"
}
}

28
run_parser.ts Normal file
View File

@@ -0,0 +1,28 @@
// run_parser.ts
import { DeterministicParser } from './BuyerSheetParser.ts';
async function main() {
// Einfaches Auslesen der Kommandozeilenargumente
const args = process.argv.slice(2);
const pdfArgIndex = args.indexOf('--pdf');
if (pdfArgIndex === -1 || !args[pdfArgIndex + 1]) {
console.error("Usage: npx ts-node run_parser.ts --pdf <path_to_pdf>");
process.exit(1);
}
const pdfPath = args[pdfArgIndex + 1];
const parser = new DeterministicParser();
try {
console.log(`Lese PDF: ${pdfPath} ...\n`);
const result = await parser.parsePdf(pdfPath);
// JSON formatiert und farbig (optional) ausgeben
console.log(JSON.stringify(result, null, 2));
} catch (error) {
console.error("Fehler beim Parsen des PDFs:", error);
}
}
main();

516
vision_runner.ts Normal file
View File

@@ -0,0 +1,516 @@
#!/usr/bin/env npx tsx
/**
* vision_runner.ts — Gleis B: Vision-LLM-Extraktion für Buyer Information Sheets
*
* Liest out/buyers.json (Ergebnis von Gleis A / batch_runner.ts), nimmt alle
* Einträge mit is_buyer_sheet === false, rendert die PDF-Seiten via pdftoppm
* (poppler-utils) zu PNGs und schickt sie an einen llama-server
* (OpenAI-kompatibel, Qwen3.6 + mmproj).
*
* Prinzip: Das VLM TRANSKRIBIERT nur (Rohwerte, verbatim). Sämtliche
* Normalisierung (Leer-Marker, State, down_payment, Datum) passiert
* deterministisch hier in TypeScript — identische Regeln wie Gleis A.
*
* Aufruf (AMD/Vulkan, inhouse):
* npx tsx vision_runner.ts \
* --input out/buyers.json \
* --pdf-root "/mnt/bizmatch-nas/AA Buyers NDA's/Buyers NDA's A-Z" \
* --api http://localhost:8000/v1 --limit 5
*
* Aufruf (NVIDIA/CUDA, EC2):
* npx tsx vision_runner.ts --input out/buyers.json \
* --pdf-root ~/data --api http://localhost:8000/v1 --limit 5
*
* Voraussetzungen: Node >= 18 (fetch), poppler-utils (pdftoppm) installiert.
*
* Outputs:
* out/buyers_vision.json — nur die Vision-Ergebnisse (Resume-Datei)
* out/buyers_merged.json — Gleis A + Gleis B zusammengeführt
*/
import { execFile } from "node:child_process";
import { promisify } from "node:util";
import * as fsp from "node:fs/promises";
import * as fs from "node:fs";
import * as path from "node:path";
import * as os from "node:os";
const execFileP = promisify(execFile);
// ---------------------------------------------------------------------------
// CLI
// ---------------------------------------------------------------------------
interface Args {
input: string;
pdfRoot: string;
api: string;
limit: number;
only: string | null;
dpi: number;
maxPages: number;
force: boolean;
outDir: string;
timeoutMs: number;
}
function parseArgs(): Args {
const a = process.argv.slice(2);
const get = (flag: string, def: string | null = null): string | null => {
const i = a.indexOf(flag);
return i >= 0 && a[i + 1] !== undefined ? a[i + 1] : def;
};
const input = get("--input", "out/buyers.json")!;
const pdfRoot = get("--pdf-root");
const api = (get("--api", "http://localhost:8000/v1") || "").replace(/\/+$/, "");
if (!pdfRoot) {
console.error("Fehler: --pdf-root <Verzeichnis> ist erforderlich.");
process.exit(1);
}
return {
input,
pdfRoot: pdfRoot.replace(/^~(?=\/)/, os.homedir()),
api,
limit: parseInt(get("--limit", "0")!, 10) || 0,
only: get("--only"),
dpi: parseInt(get("--dpi", "150")!, 10) || 150,
maxPages: parseInt(get("--max-pages", "8")!, 10) || 8,
force: a.includes("--force"),
outDir: get("--out-dir", path.dirname(input))!,
timeoutMs: (parseInt(get("--timeout", "180")!, 10) || 180) * 1000,
};
}
// ---------------------------------------------------------------------------
// Typen
// ---------------------------------------------------------------------------
interface BuyerRecord {
file_name: string;
is_buyer_sheet: boolean;
[k: string]: unknown;
}
/** Rohantwort des VLM — alles verbatim, Normalisierung erfolgt in TS. */
interface VisionRaw {
is_buyer_sheet: boolean;
info_page: number | null;
ca_page: number | null;
name_company: string | null;
prospective_buyer: string | null;
company: string | null;
phone: string | null;
cell: string | null;
email: string | null;
address: string | null;
state: string | null;
how_did_you_hear: string | null;
interested_in_updates: string | null;
types_of_business_raw: string | null;
background_experience: string | null;
total_purchase_price: string | null;
down_payment_raw: string | null;
date_of_introduction_raw: string | null;
}
// ---------------------------------------------------------------------------
// Normalisierung — identisch zu Gleis A halten!
// (Falls BuyerSheetParser.ts diese Funktionen exportiert, stattdessen
// importieren, damit beide Gleise garantiert dieselbe Logik nutzen.)
// ---------------------------------------------------------------------------
const NULL_MARKERS = /^(n|na|n\/a|none|nil|x|-+|\.+)$/i;
function cleanStr(v: string | null | undefined): string | null {
if (v == null) return null;
const t = String(v).replace(/\s+/g, " ").trim();
if (!t || NULL_MARKERS.test(t)) return null;
return t;
}
const STATE_MAP: Record<string, string> = {
alabama: "AL", alaska: "AK", arizona: "AZ", arkansas: "AR", california: "CA",
colorado: "CO", connecticut: "CT", delaware: "DE", florida: "FL", georgia: "GA",
hawaii: "HI", idaho: "ID", illinois: "IL", indiana: "IN", iowa: "IA",
kansas: "KS", kentucky: "KY", louisiana: "LA", maine: "ME", maryland: "MD",
massachusetts: "MA", michigan: "MI", minnesota: "MN", mississippi: "MS",
missouri: "MO", montana: "MT", nebraska: "NE", nevada: "NV",
"new hampshire": "NH", "new jersey": "NJ", "new mexico": "NM",
"new york": "NY", "north carolina": "NC", "north dakota": "ND", ohio: "OH",
oklahoma: "OK", oregon: "OR", pennsylvania: "PA", "rhode island": "RI",
"south carolina": "SC", "south dakota": "SD", tennessee: "TN", texas: "TX",
utah: "UT", vermont: "VT", virginia: "VA", washington: "WA",
"west virginia": "WV", wisconsin: "WI", wyoming: "WY",
"district of columbia": "DC", "washington dc": "DC",
};
const STATE_CODES = new Set(Object.values(STATE_MAP));
function normState(v: string | null): string | null {
const c = cleanStr(v);
if (!c) return null;
const up = c.toUpperCase().replace(/\./g, "");
if (up.length === 2 && STATE_CODES.has(up)) return up;
const full = STATE_MAP[c.toLowerCase().replace(/\./g, "")];
return full ?? c; // unbekannt: bereinigt durchreichen, nicht raten
}
/** "$350,000" → "350000"; "1.5M" → "1500000"; sonst null + raw. */
function normDownPayment(raw: string | null): { value: string | null; raw: string | null } {
const c = cleanStr(raw);
if (!c) return { value: null, raw: null };
const m = c.match(/^\$?\s*(\d[\d,]*(?:\.\d+)?)\s*([kKmM])?\s*$/);
if (!m) return { value: null, raw: c };
let num = parseFloat(m[1].replace(/,/g, ""));
if (m[2]) num *= /k/i.test(m[2]) ? 1_000 : 1_000_000;
if (!Number.isFinite(num) || num <= 0) return { value: null, raw: c };
return { value: String(Math.round(num)), raw: c };
}
const MONTHS: Record<string, number> = {
jan: 1, feb: 2, mar: 3, apr: 4, may: 5, jun: 6,
jul: 7, aug: 8, sep: 9, oct: 10, nov: 11, dec: 12,
};
function isoDate(y: number, mo: number, d: number): string | null {
if (y < 100) y += 2000;
if (y < 1990 || y > 2100 || mo < 1 || mo > 12 || d < 1 || d > 31) return null;
return `${y}-${String(mo).padStart(2, "0")}-${String(d).padStart(2, "0")}`;
}
/** US-Formate: 6/25/26, 06-25-2026, "June 25, 2026" → YYYY-MM-DD. */
function normDate(raw: string | null): string | null {
const c = cleanStr(raw);
if (!c) return null;
let m = c.match(/^(\d{1,2})[\/\-.](\d{1,2})[\/\-.](\d{2}|\d{4})$/);
if (m) return isoDate(parseInt(m[3], 10), parseInt(m[1], 10), parseInt(m[2], 10));
m = c.match(/^([A-Za-z]{3,9})\.?\s+(\d{1,2})(?:st|nd|rd|th)?,?\s+(\d{2}|\d{4})$/);
if (m) {
const mo = MONTHS[m[1].slice(0, 3).toLowerCase()];
if (mo) return isoDate(parseInt(m[3], 10), mo, parseInt(m[2], 10));
}
m = c.match(/^(\d{4})-(\d{2})-(\d{2})$/);
if (m) return isoDate(parseInt(m[1], 10), parseInt(m[2], 10), parseInt(m[3], 10));
return null;
}
// ---------------------------------------------------------------------------
// PDF-Index (rekursiv, Basename → voller Pfad) + Rendering
// ---------------------------------------------------------------------------
async function buildPdfIndex(root: string): Promise<Map<string, string>> {
const index = new Map<string, string>();
const stack = [root];
while (stack.length) {
const dir = stack.pop()!;
let entries: fs.Dirent[];
try {
entries = await fsp.readdir(dir, { withFileTypes: true });
} catch {
continue;
}
for (const e of entries) {
const p = path.join(dir, e.name);
if (e.isDirectory()) stack.push(p);
else if (e.isFile() && e.name.toLowerCase().endsWith(".pdf")) {
if (!index.has(e.name)) index.set(e.name, p);
}
}
}
return index;
}
async function renderPdf(pdfPath: string, dpi: number, maxPages: number, tmpDir: string): Promise<string[]> {
const prefix = path.join(tmpDir, "page");
await execFileP("pdftoppm", ["-png", "-r", String(dpi), "-l", String(maxPages), pdfPath, prefix], {
timeout: 120_000,
});
const pageNum = (f: string) => parseInt(f.match(/-(\d+)\.png$/)?.[1] ?? "0", 10);
const files = (await fsp.readdir(tmpDir))
.filter((f) => f.endsWith(".png"))
.sort((a, b) => pageNum(a) - pageNum(b))
.map((f) => path.join(tmpDir, f));
if (files.length === 0) throw new Error("pdftoppm hat keine Seiten erzeugt");
return files;
}
// ---------------------------------------------------------------------------
// VLM-Aufruf (llama-server, OpenAI-kompatibel, JSON-Schema)
// ---------------------------------------------------------------------------
const nullableString = { type: ["string", "null"] };
const nullableInt = { type: ["integer", "null"] };
const RESPONSE_SCHEMA = {
type: "object",
additionalProperties: false,
required: [
"is_buyer_sheet", "info_page", "ca_page", "name_company", "prospective_buyer",
"company", "phone", "cell", "email", "address", "state", "how_did_you_hear",
"interested_in_updates", "types_of_business_raw", "background_experience",
"total_purchase_price", "down_payment_raw", "date_of_introduction_raw",
],
properties: {
is_buyer_sheet: { type: "boolean" },
info_page: nullableInt,
ca_page: nullableInt,
name_company: nullableString,
prospective_buyer: nullableString,
company: nullableString,
phone: nullableString,
cell: nullableString,
email: nullableString,
address: nullableString,
state: nullableString,
how_did_you_hear: nullableString,
interested_in_updates: nullableString,
types_of_business_raw: nullableString,
background_experience: nullableString,
total_purchase_price: nullableString,
down_payment_raw: nullableString,
date_of_introduction_raw: nullableString,
},
} as const;
const SYSTEM_PROMPT =
"You are a precise document transcription engine for scanned business forms. " +
"You transcribe handwritten and typed form fields EXACTLY as written (verbatim), " +
"including typos. You never guess, infer, or invent values. If a field is blank " +
"or unreadable, you return null.";
const USER_PROMPT = `You see all pages of one PDF, in order (image 1 = page 1).
Task: Decide whether this document contains a business brokerage "BUYER INFORMATION SHEET" form, and if so, transcribe its fields.
1. is_buyer_sheet: true only if a page with the heading "BUYER INFORMATION SHEET" exists. If the document is something else (notes, listing, letter, other form), return is_buyer_sheet=false and null for every field.
2. info_page: page number (1-based) of the BUYER INFORMATION SHEET page, else null.
3. ca_page: page number of the confidentiality agreement page containing text like "PROSPECTIVE BUYER AGREES TO KEEP AND HOLD CONFIDENTIAL", else null.
4. From the info page, transcribe VERBATIM (exactly as written, do not normalize, do not expand abbreviations):
- name_company: value of the "Name/Company" line
- prospective_buyer: value of the "Prospective Buyer" line
- company: value of a separate "Company" line if present
- phone, cell, email, address, state
- how_did_you_hear: "How did you hear about us"
- interested_in_updates: answer/checkbox for receiving updates (transcribe what is marked, e.g. "Yes" or "No"), else null
- types_of_business_raw: "Type(s) of business interested in" (may span multiple lines — join with a space)
- background_experience: "Background/Experience" (may span multiple lines)
- total_purchase_price: "Total Purchase Price" as written
- down_payment_raw: "Down Payment" as written (e.g. "$350,000", "1.5M", "TBD")
5. date_of_introduction_raw: the date written next to the buyer's signature on the confidentiality agreement page, verbatim (e.g. "6/25/26"), else null.
A blank field, "N/A", "n", or an empty line = transcribe it as written; if truly empty, use null. Return only the JSON object.`;
async function fetchModelId(api: string): Promise<string> {
try {
const r = await fetch(`${api}/models`);
const j = (await r.json()) as { data?: Array<{ id?: string }> };
return j?.data?.[0]?.id ?? "unknown";
} catch {
return "unknown";
}
}
async function callVision(api: string, images: string[], timeoutMs: number): Promise<VisionRaw> {
const content: Array<Record<string, unknown>> = [{ type: "text", text: USER_PROMPT }];
for (const img of images) {
const b64 = await fsp.readFile(img, { encoding: "base64" });
content.push({ type: "image_url", image_url: { url: `data:image/png;base64,${b64}` } });
}
const body = {
model: "qwen3.6",
temperature: 0.1,
top_p: 0.95,
max_tokens: 1500,
chat_template_kwargs: { enable_thinking: false },
response_format: {
type: "json_schema",
json_schema: { name: "buyer_sheet", strict: true, schema: RESPONSE_SCHEMA },
},
messages: [
{ role: "system", content: SYSTEM_PROMPT },
{ role: "user", content },
],
};
const ctl = new AbortController();
const timer = setTimeout(() => ctl.abort(), timeoutMs);
try {
const r = await fetch(`${api}/chat/completions`, {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify(body),
signal: ctl.signal,
});
if (!r.ok) throw new Error(`HTTP ${r.status}: ${(await r.text()).slice(0, 300)}`);
const j = (await r.json()) as { choices?: Array<{ message?: { content?: string } }> };
const text = j?.choices?.[0]?.message?.content;
if (!text) throw new Error("Leere Antwort vom Server");
return JSON.parse(text) as VisionRaw;
} finally {
clearTimeout(timer);
}
}
// ---------------------------------------------------------------------------
// Merge: deterministischer Datensatz + Vision-Rohwerte → Zielstruktur
// ---------------------------------------------------------------------------
function mergeRecord(det: BuyerRecord, vis: VisionRaw, model: string): BuyerRecord {
const base: BuyerRecord = {
...det,
_parser: "vision",
_vision_model: model,
_vision_error: undefined,
};
delete (base as Record<string, unknown>)["_vision_error"];
if (!vis.is_buyer_sheet) {
return { ...base, is_buyer_sheet: false, _info_page: null, _ca_page: null };
}
const dp = normDownPayment(vis.down_payment_raw);
return {
...base,
is_buyer_sheet: true,
name_company: cleanStr(vis.name_company),
prospective_buyer: cleanStr(vis.prospective_buyer),
company: cleanStr(vis.company),
phone: cleanStr(vis.phone),
cell: cleanStr(vis.cell),
email: cleanStr(vis.email),
address: cleanStr(vis.address),
state: normState(vis.state),
how_did_you_hear: cleanStr(vis.how_did_you_hear),
interested_in_updates: cleanStr(vis.interested_in_updates),
types_of_business_raw: cleanStr(vis.types_of_business_raw),
background_experience: cleanStr(vis.background_experience),
total_purchase_price: cleanStr(vis.total_purchase_price),
down_payment: dp.value,
down_payment_raw: dp.raw,
// Vision-Datum bevorzugt; Fallback: deterministischer Wert (z.B. aus Dateinamen)
date_of_introduction: normDate(vis.date_of_introduction_raw) ?? (det.date_of_introduction as string | null) ?? null,
_info_page: vis.info_page,
_ca_page: vis.ca_page,
};
}
// ---------------------------------------------------------------------------
// Main
// ---------------------------------------------------------------------------
const RETRIES = 3;
function progress(line: string): void {
const cols = process.stderr.columns ?? 120;
process.stderr.write("\r" + line.slice(0, cols - 1).padEnd(cols - 1));
}
async function main(): Promise<void> {
const args = parseArgs();
// pdftoppm vorhanden?
try {
await execFileP("pdftoppm", ["-v"]);
} catch {
console.error("Fehler: pdftoppm nicht gefunden. Installieren: sudo apt install poppler-utils");
process.exit(1);
}
const all: BuyerRecord[] = JSON.parse(await fsp.readFile(args.input, "utf8"));
let targets = all.filter((r) => r.is_buyer_sheet === false);
if (args.only) targets = targets.filter((r) => r.file_name === args.only);
if (args.limit > 0) targets = targets.slice(0, args.limit);
const visionPath = path.join(args.outDir, "buyers_vision.json");
const mergedPath = path.join(args.outDir, "buyers_merged.json");
const done = new Map<string, BuyerRecord>();
try {
const prev: BuyerRecord[] = JSON.parse(await fsp.readFile(visionPath, "utf8"));
for (const r of prev) done.set(r.file_name, r);
} catch {
/* kein Resume-Stand */
}
console.error(`Indexiere PDFs unter ${args.pdfRoot} ...`);
const pdfIndex = await buildPdfIndex(args.pdfRoot);
console.error(`${pdfIndex.size} PDFs gefunden. ${targets.length} Einträge zu verarbeiten.`);
const model = await fetchModelId(args.api);
console.error(`Modell: ${model} @ ${args.api}`);
let ok = 0, notSheet = 0, errors = 0, skipped = 0;
for (let i = 0; i < targets.length; i++) {
const det = targets[i];
const tag = `[${i + 1}/${targets.length}] ${det.file_name}`;
const prev = done.get(det.file_name);
if (prev && !prev["_vision_error"] && !args.force) {
skipped++;
progress(`${tag} … übersprungen (bereits verarbeitet)`);
continue;
}
const pdfPath = pdfIndex.get(det.file_name);
if (!pdfPath) {
done.set(det.file_name, { ...det, _vision_error: "PDF nicht gefunden" });
errors++;
progress(`${tag} … FEHLER: PDF nicht gefunden`);
continue;
}
const tmpDir = await fsp.mkdtemp(path.join(os.tmpdir(), "bvs-"));
try {
progress(`${tag} … rendere`);
const images = await renderPdf(pdfPath, args.dpi, args.maxPages, tmpDir);
let vis: VisionRaw | null = null;
let lastErr = "";
for (let attempt = 1; attempt <= RETRIES; attempt++) {
try {
progress(`${tag} … VLM (${images.length} Seiten, Versuch ${attempt})`);
vis = await callVision(args.api, images, args.timeoutMs);
break;
} catch (e) {
lastErr = e instanceof Error ? e.message : String(e);
if (attempt < RETRIES) await new Promise((res) => setTimeout(res, 5000 * attempt));
}
}
if (!vis) {
done.set(det.file_name, { ...det, _vision_error: lastErr });
errors++;
progress(`${tag} … FEHLER: ${lastErr}`);
} else {
const merged = mergeRecord(det, vis, model);
done.set(det.file_name, merged);
if (merged.is_buyer_sheet) { ok++; progress(`${tag} … OK`); }
else { notSheet++; progress(`${tag} … kein Buyer Sheet`); }
}
} catch (e) {
const msg = e instanceof Error ? e.message : String(e);
done.set(det.file_name, { ...det, _vision_error: msg });
errors++;
progress(`${tag} … FEHLER: ${msg}`);
} finally {
await fsp.rm(tmpDir, { recursive: true, force: true });
}
// Inkrementell sichern (Resume-fähig)
await fsp.mkdir(args.outDir, { recursive: true });
await fsp.writeFile(visionPath, JSON.stringify([...done.values()], null, 2));
}
// Merge: Gleis A + Gleis B
const merged = all.map((r) => done.get(r.file_name) ?? r);
await fsp.writeFile(mergedPath, JSON.stringify(merged, null, 2));
process.stderr.write("\n");
console.error(
`Fertig. OK: ${ok}, kein Buyer Sheet: ${notSheet}, Fehler: ${errors}, übersprungen: ${skipped}`
);
console.error(`${visionPath}\n→ ${mergedPath}`);
}
main().catch((e) => {
console.error("\nAbbruch:", e instanceof Error ? e.message : e);
process.exit(1);
});