This commit is contained in:
2026-07-12 12:30:20 -05:00
parent 888c5c1543
commit 0989c06b8f
23 changed files with 3324 additions and 119 deletions

View File

@@ -2,6 +2,7 @@ import * as pdfjsLib from 'pdfjs-dist/legacy/build/pdf.js';
import * as fs from 'fs'; import * as fs from 'fs';
export interface BuyerSheetData { export interface BuyerSheetData {
file_name: string; // NEU
is_buyer_sheet: boolean; is_buyer_sheet: boolean;
name_company: string | null; name_company: string | null;
prospective_buyer: string | null; prospective_buyer: string | null;
@@ -21,8 +22,8 @@ export interface BuyerSheetData {
date_of_introduction: string | null; date_of_introduction: string | null;
_checkbox_pending: boolean; _checkbox_pending: boolean;
_parser: string; _parser: string;
_info_page: number; _info_page: number | null;
_ca_page: number; _ca_page: number | null;
} }
const LABELS = [ const LABELS = [
@@ -34,16 +35,69 @@ const LABELS = [
export class DeterministicParser { export class DeterministicParser {
public async parsePdf(filePath: string): Promise<BuyerSheetData> { public async parsePdf(filePath: string, fileName: string): Promise<BuyerSheetData | null> {
const dataBuffer = fs.readFileSync(filePath); const dataBuffer = fs.readFileSync(filePath);
const loadingTask = pdfjsLib.getDocument({ data: new Uint8Array(dataBuffer) }); const loadingTask = pdfjsLib.getDocument({ data: new Uint8Array(dataBuffer) });
const pdfDocument = await loadingTask.promise; const pdfDocument = await loadingTask.promise;
const page = await pdfDocument.getPage(1);
const textContent = await page.getTextContent(); // Regel: Alles über 10 Seiten wird radikal ignoriert
if (pdfDocument.numPages > 10) {
return null;
}
// 1. Textfragmente mit X, Y und Breite (Width) auslesen let infoPageNum: number | null = null;
const mappedItems = textContent.items let caPageNum: number | null = null;
let infoPageObj: any = null;
let infoPageTextContent: any = null;
// ==========================================
// 1. Dynamische Seitensuche (Visuell sortiert & kugelsicher)
// ==========================================
for (let i = 1; i <= pdfDocument.numPages; i++) {
const page = await pdfDocument.getPage(i);
const textContent = await page.getTextContent();
// Elemente mit Koordinaten versehen und wie ein Mensch lesen (von oben nach unten, links nach rechts)
const sortedItems = textContent.items
.map((item: any) => ({
text: item.str,
x: item.transform[4],
y: item.transform[5]
}))
.sort((a: any, b: any) => {
// Y-Toleranz für Buchstaben auf derselben Zeile
if (Math.abs(b.y - a.y) > 5) {
return b.y - a.y;
}
return a.x - b.x;
});
// Wir werfen ALLE Leerzeichen, Striche, Punkte und unsichtbare Artefakte weg.
// Übrig bleibt eine reine, unverwüstliche Buchstabenkette.
const textRaw = sortedItems.map((i: any) => i.text).join('').toUpperCase().replace(/[^A-Z]/g, '');
// Anker-Suche in der sauberen Zeichenkette
if (textRaw.includes('BUYERINFORMATIONSHEET')) {
infoPageNum = i;
infoPageObj = page;
infoPageTextContent = textContent;
}
if (textRaw.includes('PROSPECTIVEBUYERAGREESTOKEEPANDHOLDCONFIDENTIAL')) {
caPageNum = i;
}
}
// ==========================================
// 2. Fallback für Bild/Scan PDFs
// ==========================================
if (!infoPageNum || !infoPageObj || !infoPageTextContent) {
return this.createEmptyFallback(fileName, filePath, infoPageNum, caPageNum);
}
// ==========================================
// 3. Werte-Extraktion (nur auf der Info-Seite!)
// ==========================================
const mappedItems = infoPageTextContent.items
.filter((item: any) => item.str.trim() !== '') .filter((item: any) => item.str.trim() !== '')
.map((item: any) => ({ .map((item: any) => ({
text: item.str, text: item.str,
@@ -51,9 +105,8 @@ export class DeterministicParser {
y: item.transform[5], y: item.transform[5],
width: item.width width: item.width
})) }))
.sort((a, b) => b.y - a.y || a.x - b.x); // Y absteigend (oben nach unten) .sort((a: any, b: any) => b.y - a.y || a.x - b.x);
// 2. Zeilen bilden (Y-Toleranz)
const linesGrouped: { y: number, items: any[] }[] = []; const linesGrouped: { y: number, items: any[] }[] = [];
let currentY: number | null = null; let currentY: number | null = null;
let currentItems = []; let currentItems = [];
@@ -72,7 +125,6 @@ export class DeterministicParser {
linesGrouped.push({ y: currentY, items: currentItems }); linesGrouped.push({ y: currentY, items: currentItems });
} }
// 3. Zerrissene Wörter reparieren (Kerning)
const lines: { y: number, text: string }[] = []; const lines: { y: number, text: string }[] = [];
for (const group of linesGrouped) { for (const group of linesGrouped) {
group.items.sort((a, b) => a.x - b.x); group.items.sort((a, b) => a.x - b.x);
@@ -82,10 +134,7 @@ export class DeterministicParser {
for (const item of group.items) { for (const item of group.items) {
if (prevEnd !== -1) { if (prevEnd !== -1) {
const gap = item.x - prevEnd; const gap = item.x - prevEnd;
// Wenn die Lücke größer als ~4 Pixel ist, ist es ein echtes Leerzeichen if (gap > 4) lineStr += " ";
if (gap > 4) {
lineStr += " ";
}
} }
lineStr += item.text; lineStr += item.text;
prevEnd = item.x + item.width; prevEnd = item.x + item.width;
@@ -93,18 +142,13 @@ export class DeterministicParser {
lines.push({ y: group.y, text: lineStr }); lines.push({ y: group.y, text: lineStr });
} }
// 4. Y-Intervall Parsing (Die "Nutzer-Idee")
const rawResults: Record<string, string[]> = {}; const rawResults: Record<string, string[]> = {};
LABELS.forEach(lbl => rawResults[lbl] = []); LABELS.forEach(lbl => rawResults[lbl] = []);
let currentLabel: string | null = null; let currentLabel: string | null = null;
// Längste Labels zuerst suchen, damit "EMAIL ADDRESS" vor "ADDRESS" gefunden wird
const sortedLabels = [...LABELS].sort((a, b) => b.length - a.length); const sortedLabels = [...LABELS].sort((a, b) => b.length - a.length);
for (const line of lines) { for (const line of lines) {
const textUpper = line.text.toUpperCase().replace(/\s+/g, ''); const textUpper = line.text.toUpperCase().replace(/\s+/g, '');
// Boilerplate ignorieren, der Labels enthält ("VERIFICATION OF DOWN PAYMENT")
if (textUpper.includes("SELLERMAYREQUIREVERIFICATION")) continue; if (textUpper.includes("SELLERMAYREQUIREVERIFICATION")) continue;
let foundLabel: string | null = null; let foundLabel: string | null = null;
@@ -118,75 +162,87 @@ export class DeterministicParser {
if (foundLabel) { if (foundLabel) {
currentLabel = foundLabel; currentLabel = foundLabel;
// Falls der Wert direkt auf derselben Zeile steht (z.B. "NAME/COMPANY: Gunnar Schultz")
if (line.text.includes(':')) { if (line.text.includes(':')) {
const parts = line.text.split(':'); const parts = line.text.split(':');
const val = parts.slice(1).join(':').replace(/_+/g, '').trim(); // Unterstriche entfernen const val = parts.slice(1).join(':').replace(/_+/g, '').trim();
if (val.length > 0) { if (val.length > 0) rawResults[currentLabel].push(val);
rawResults[currentLabel].push(val);
}
} }
} else if (currentLabel) { } else if (currentLabel) {
// Diese Zeile ist kein Label, gehört also zum aktuellen Bereich!
const cleanVal = line.text.replace(/_+/g, '').trim(); const cleanVal = line.text.replace(/_+/g, '').trim();
// Footer und Artefakte ignorieren
if (cleanVal.length > 0 && !cleanVal.includes("Doc ID") && !cleanVal.includes("Bizmatch")) { if (cleanVal.length > 0 && !cleanVal.includes("Doc ID") && !cleanVal.includes("Bizmatch")) {
rawResults[currentLabel].push(cleanVal); rawResults[currentLabel].push(cleanVal);
} }
} }
} }
// --- NEU: Visuelle Checkbox Erkennung ---
const visualCheckboxResult = await this.detectGraphicCheckbox(page, mappedItems); const visualCheckboxResult = await this.detectGraphicCheckbox(infoPageObj, mappedItems);
return this.mapToTargetStructure(rawResults, filePath, visualCheckboxResult); return this.mapToTargetStructure(rawResults, filePath, fileName, visualCheckboxResult, infoPageNum, caPageNum);
}
private createEmptyFallback(fileName: string, filePath: string, infoPage: number | null, caPage: number | null): BuyerSheetData {
let dateOfIntro = null;
const dateMatch = filePath.match(/(\d{2})(\d{2})(\d{2})\.pdf$/);
if (dateMatch) dateOfIntro = `20${dateMatch[3]}-${dateMatch[1]}-${dateMatch[2]}`;
return {
file_name: fileName,
is_buyer_sheet: false,
name_company: null,
prospective_buyer: null,
company: null,
phone: null,
cell: null,
email: null,
address: null,
state: null,
how_did_you_hear: null,
interested_in_updates: null,
types_of_business_raw: null,
background_experience: null,
total_purchase_price: null,
down_payment: null,
down_payment_raw: null,
date_of_introduction: dateOfIntro,
_checkbox_pending: true,
_parser: "deterministic", // Signal für den AI-Pass
_info_page: infoPage,
_ca_page: caPage
};
} }
private mapToTargetStructure( private mapToTargetStructure(
raw: Record<string, string[]>, raw: Record<string, string[]>,
filePath: string, filePath: string,
visualCheckboxResult: boolean | null // <-- Neuer Parameter fileName: string,
visualCheckboxResult: boolean | null,
infoPageNum: number | null,
caPageNum: number | null
): BuyerSheetData { ): BuyerSheetData {
const getVal = (label: string) => raw[label] && raw[label].length > 0 ? raw[label].join(' ') : null; const getVal = (label: string) => raw[label] && raw[label].length > 0 ? raw[label].join(' ') : null;
const nameCompany = getVal("NAME / COMPANY"); const nameCompany = getVal("NAME / COMPANY");
// ==========================================
// 1. Telefon & Handy (Cell) separieren
// ==========================================
let phoneRaw = getVal("PHONE"); let phoneRaw = getVal("PHONE");
let phoneClean: string | null = null; let phoneClean = null;
let cellClean: string | null = null; let cellClean = null;
if (phoneRaw) { if (phoneRaw) {
// Alles vor "FAX:" oder "CELL:" ist die Telefonnummer
const phoneMatch = phoneRaw.split(/FAX:|CELL:/)[0]; const phoneMatch = phoneRaw.split(/FAX:|CELL:/)[0];
phoneClean = phoneMatch ? phoneMatch.trim() : null; phoneClean = phoneMatch ? phoneMatch.trim() : null;
// Alles nach "CELL:" ist die Handynummer
const cellMatch = phoneRaw.match(/CELL:\s*(.*)/); const cellMatch = phoneRaw.match(/CELL:\s*(.*)/);
cellClean = cellMatch && cellMatch[1] ? cellMatch[1].trim() : null; cellClean = cellMatch && cellMatch[1] ? cellMatch[1].trim() : null;
} }
// ==========================================
// 2. Adresse bereinigen
// ==========================================
let addressClean = getVal("ADDRESS"); let addressClean = getVal("ADDRESS");
if (addressClean) { if (addressClean) {
// Den statischen Formulart-Subtext entfernen
addressClean = addressClean.replace(/PO BOX \/ STREET\s+CITY \/ STATE \/ ZIP/g, '').trim(); addressClean = addressClean.replace(/PO BOX \/ STREET\s+CITY \/ STATE \/ ZIP/g, '').trim();
if (addressClean === '') addressClean = null; if (addressClean === '') addressClean = null;
} }
// ==========================================
// 3. Checkbox Auswertung (Hybrid)
// ==========================================
const interestedRaw = getVal("ARE YOU INTERESTED"); const interestedRaw = getVal("ARE YOU INTERESTED");
let interestedInUpdates: boolean | null = null; let interestedInUpdates: boolean | null = null;
let checkboxPending = true; let checkboxPending = true;
if (interestedRaw) { if (interestedRaw) {
// 1. Zuerst Text-Prüfung versuchen (wie vorher)
if (/(✔|☑|X|✓)\s*YES/i.test(interestedRaw) || /YES\s*(✔|☑|X|✓)/i.test(interestedRaw)) { if (/(✔|☑|X|✓)\s*YES/i.test(interestedRaw) || /YES\s*(✔|☑|X|✓)/i.test(interestedRaw)) {
interestedInUpdates = true; interestedInUpdates = true;
checkboxPending = false; checkboxPending = false;
@@ -195,30 +251,21 @@ export class DeterministicParser {
checkboxPending = false; checkboxPending = false;
} }
} }
// 2. Wenn Text-Prüfung versagt hat, nutzen wir unser visuelles X-Koordinaten Ergebnis!
if (checkboxPending && visualCheckboxResult !== null) { if (checkboxPending && visualCheckboxResult !== null) {
interestedInUpdates = visualCheckboxResult; interestedInUpdates = visualCheckboxResult;
checkboxPending = false; checkboxPending = false;
} }
// ==========================================
// 4. Down Payment & Datum
// ==========================================
const downPaymentRaw = getVal("DOWN PAYMENT")?.replace(/[^0-9,]/g, '') || null; const downPaymentRaw = getVal("DOWN PAYMENT")?.replace(/[^0-9,]/g, '') || null;
const downPaymentClean = downPaymentRaw ? downPaymentRaw.replace(/,/g, '') : null; const downPaymentClean = downPaymentRaw ? downPaymentRaw.replace(/,/g, '') : null;
let dateOfIntro = null; let dateOfIntro = null;
const dateMatch = filePath.match(/(\d{2})(\d{2})(\d{2})\.pdf$/); const dateMatch = filePath.match(/(\d{2})(\d{2})(\d{2})\.pdf$/);
if (dateMatch) { if (dateMatch) dateOfIntro = `20${dateMatch[3]}-${dateMatch[1]}-${dateMatch[2]}`;
dateOfIntro = `20${dateMatch[3]}-${dateMatch[1]}-${dateMatch[2]}`;
}
// ==========================================
// 5. JSON Return mit dynamischen Flags
// ==========================================
return { return {
is_buyer_sheet: true, // Wird in der Praxis dynamisch gesetzt file_name: fileName,
is_buyer_sheet: true,
name_company: nameCompany, name_company: nameCompany,
prospective_buyer: nameCompany ? nameCompany.split('/')[0].trim() : null, prospective_buyer: nameCompany ? nameCompany.split('/')[0].trim() : null,
company: null, company: null,
@@ -237,31 +284,25 @@ export class DeterministicParser {
date_of_introduction: dateOfIntro, date_of_introduction: dateOfIntro,
_checkbox_pending: checkboxPending, _checkbox_pending: checkboxPending,
_parser: "deterministic", _parser: "deterministic",
_info_page: 1, // Wird in der Praxis über eine Schleife ermittelt _info_page: infoPageNum,
_ca_page: 2 // Wird in der Praxis über eine Schleife ermittelt _ca_page: caPageNum
}; };
} }
private async detectGraphicCheckbox(page: any, textItems: any[]): Promise<boolean | null> { private async detectGraphicCheckbox(page: any, textItems: any[]): Promise<boolean | null> {
let yesX = null, noX = null, targetY = null; let yesX = null, noX = null, targetY = null;
// 1. Koordinaten von YES und NO suchen
for (const item of textItems) { for (const item of textItems) {
const textUpper = item.text.toUpperCase().trim(); const textUpper = item.text.toUpperCase().trim();
if (textUpper.includes("YES")) { yesX = item.x; targetY = item.y; } if (textUpper.includes("YES")) { yesX = item.x; targetY = item.y; }
} }
for (const item of textItems) { for (const item of textItems) {
const textUpper = item.text.toUpperCase().trim(); const textUpper = item.text.toUpperCase().trim();
// Wir suchen NO auf derselben Höhe (Toleranz 5px) if (textUpper.includes("NO") && targetY !== null && Math.abs(item.y - targetY) < 5) noX = item.x;
if (textUpper.includes("NO") && targetY !== null && Math.abs(item.y - targetY) < 5) {
noX = item.x;
}
} }
if (yesX === null || noX === null || targetY === null) return null; if (yesX === null || noX === null || targetY === null) return null;
// ==========================================
// BILD-ERKENNUNG (Der 16x16 Stempel-Trick)
// ==========================================
const opList = await page.getOperatorList(); const opList = await page.getOperatorList();
let currentTransform = [1, 0, 0, 1, 0, 0]; let currentTransform = [1, 0, 0, 1, 0, 0];
@@ -269,10 +310,7 @@ export class DeterministicParser {
const fn = opList.fnArray[i]; const fn = opList.fnArray[i];
const args = opList.argsArray[i]; const args = opList.argsArray[i];
if (fn === pdfjsLib.OPS.transform) { if (fn === pdfjsLib.OPS.transform) currentTransform = args;
currentTransform = args;
}
// Wenn ein Bild auf die Seite gezeichnet wird
else if ( else if (
fn === pdfjsLib.OPS.paintImageXObject || fn === pdfjsLib.OPS.paintImageXObject ||
fn === pdfjsLib.OPS.paintInlineImageXObject || fn === pdfjsLib.OPS.paintInlineImageXObject ||
@@ -283,25 +321,13 @@ export class DeterministicParser {
const imgX = currentTransform[4]; const imgX = currentTransform[4];
const imgY = currentTransform[5]; const imgY = currentTransform[5];
// Wir filtern nach "Stempeln" (kleine Bilder unter 40x40 Pixeln) if (width < 40 && height < 40 && Math.abs(imgY - targetY) < 50) {
if (width < 40 && height < 40) { const distToYes = Math.abs(imgX - yesX);
// Befindet sich der Stempel auf unserer Ziel-Zeile? const distToNo = Math.abs(imgX - noX);
// (Wir erlauben 50px Toleranz, da Y=378 vs Y=417) return distToYes < distToNo;
if (Math.abs(imgY - targetY) < 50) {
const distToYes = Math.abs(imgX - yesX);
const distToNo = Math.abs(imgX - noX);
// Wenn das Bildchen näher an YES ist
if (distToYes < distToNo) {
return true;
} else {
return false;
}
}
} }
} }
} }
return null; return null;
} }
} }

101
batch_runner.ts Normal file
View File

@@ -0,0 +1,101 @@
import * as fs from 'fs';
import * as path from 'path';
import { DeterministicParser } from './BuyerSheetParser';
async function main() {
// 1. Argument-Check
const args = process.argv.slice(2);
// Verzeichnis Parameter
const dirArgIndex = args.indexOf('--dir');
if (dirArgIndex === -1 || !args[dirArgIndex + 1]) {
console.error("Fehler: Bitte ein Verzeichnis angeben!");
console.error("Nutzung: npx tsx batch_runner.ts --dir \"/Pfad/zum/Ordner\" [--limit 10]");
process.exit(1);
}
const dirPath = args[dirArgIndex + 1];
// Limit Parameter
const limitArgIndex = args.indexOf('--limit');
let limit = -1; // -1 bedeutet: Kein Limit, alle verarbeiten
if (limitArgIndex !== -1 && args[limitArgIndex + 1]) {
limit = parseInt(args[limitArgIndex + 1], 10);
}
if (!fs.existsSync(dirPath) || !fs.statSync(dirPath).isDirectory()) {
console.error(`Fehler: Der Pfad "${dirPath}" existiert nicht oder ist kein Verzeichnis.`);
process.exit(1);
}
// 2. PDFs finden und Metadaten (für die Sortierung) auslesen
const filesWithStats = fs.readdirSync(dirPath)
.filter(f => f.toLowerCase().endsWith('.pdf'))
.map(file => {
const fullPath = path.join(dirPath, file);
return {
file,
fullPath,
// Änderungsdatum der Datei auslesen (in Millisekunden)
mtime: fs.statSync(fullPath).mtimeMs
};
});
if (filesWithStats.length === 0) {
console.log("Keine PDFs in diesem Verzeichnis gefunden.");
process.exit(0);
}
// 3. Absteigend sortieren (Neueste zuerst)
filesWithStats.sort((a, b) => b.mtime - a.mtime);
// 4. Limit anwenden (falls gesetzt)
const filesToProcess = limit > 0 ? filesWithStats.slice(0, limit) : filesWithStats;
console.log(`Starte Verarbeitung: ${filesToProcess.length} von ${filesWithStats.length} PDFs ausgewählt (Sortierung: Neueste zuerst)...\n`);
const parser = new DeterministicParser();
const finalResults = [];
let skippedCounter = 0;
// 5. PDFs iterieren
for (const fileObj of filesToProcess) {
const { file, fullPath } = fileObj;
process.stdout.write(`-> Verarbeite: ${file} ... `);
try {
const data = await parser.parsePdf(fullPath, file);
if (data === null) {
console.log("ÜBERSPRUNGEN (> 10 Seiten)");
skippedCounter++;
} else if (data.is_buyer_sheet === false) {
console.log("IMAGE SCAN (Zuweisung an Vision AI)");
finalResults.push(data);
} else {
console.log("ERFOLGREICH");
finalResults.push(data);
}
} catch (error) {
console.log("FEHLER BEIM PARSEN");
console.error(error);
}
}
// 6. JSON Export unter ./out/buyers.json
const outDir = path.join(process.cwd(), 'out');
if (!fs.existsSync(outDir)) {
fs.mkdirSync(outDir);
}
const outPath = path.join(outDir, 'buyers.json');
fs.writeFileSync(outPath, JSON.stringify(finalResults, null, 2), 'utf-8');
console.log(`\n=================================================`);
console.log(`Zusammenfassung:`);
console.log(` Verarbeitet: ${finalResults.length}`);
console.log(` Verworfen (>10 Seiten): ${skippedCounter}`);
console.log(` Export gespeichert in: ${outPath}`);
console.log(`=================================================`);
}
main();

View File

@@ -0,0 +1,35 @@
# llama-server für Qwen3.6 Vision — NVIDIA RTX PRO 6000 Blackwell (sm_120)
#
# WICHTIG:
# - Muss aus aktuellem Source gebaut werden: Qwen3.6 nutzt ein neues
# Rope-Encoding (rope.dimension_sections 3 statt 4), alte Images/Builds
# brechen mit "wrong array length".
# - Blackwell braucht CUDA >= 12.8. KEIN CUDA 13.2 verwenden — erzeugt
# mit Qwen3.6 Gibberish (bekannter NVIDIA-Bug).
FROM nvidia/cuda:12.8.1-devel-ubuntu24.04 AS build
RUN apt-get update && apt-get install -y --no-install-recommends \
git cmake build-essential libcurl4-openssl-dev \
&& rm -rf /var/lib/apt/lists/*
RUN git clone --depth 1 https://github.com/ggml-org/llama.cpp /src
# 120 = Blackwell (RTX PRO 6000). Für andere Karten anpassen.
RUN cmake /src -B /build \
-DGGML_CUDA=ON \
-DCMAKE_CUDA_ARCHITECTURES=120 \
-DBUILD_SHARED_LIBS=OFF \
-DLLAMA_CURL=ON \
&& cmake --build /build --config Release -j --target llama-server
FROM nvidia/cuda:12.8.1-runtime-ubuntu24.04
RUN apt-get update && apt-get install -y --no-install-recommends \
libcurl4 libgomp1 curl \
&& rm -rf /var/lib/apt/lists/*
COPY --from=build /build/bin/llama-server /usr/local/bin/llama-server
EXPOSE 8000
ENTRYPOINT ["llama-server"]

View File

@@ -0,0 +1,32 @@
# llama-server für Qwen3.6 Vision — AMD Radeon AI Pro R9700, Vulkan-Backend
#
# Muss aus aktuellem Source gebaut werden (Qwen3.6-Rope-Änderung, s. Dockerfile.cuda).
# Vulkan statt ROCm — hat sich bei Vision als stabiler erwiesen.
FROM ubuntu:24.04 AS build
RUN apt-get update && apt-get install -y --no-install-recommends \
git cmake build-essential libcurl4-openssl-dev \
libvulkan-dev glslc \
&& rm -rf /var/lib/apt/lists/*
RUN git clone --depth 1 https://github.com/ggml-org/llama.cpp /src
RUN cmake /src -B /build \
-DGGML_VULKAN=ON \
-DBUILD_SHARED_LIBS=OFF \
-DLLAMA_CURL=ON \
&& cmake --build /build --config Release -j --target llama-server
FROM ubuntu:24.04
# mesa-vulkan-drivers = RADV-Treiber im Container (GPU via /dev/dri durchgereicht)
RUN apt-get update && apt-get install -y --no-install-recommends \
libvulkan1 mesa-vulkan-drivers vulkan-tools \
libcurl4 libgomp1 curl \
&& rm -rf /var/lib/apt/lists/*
COPY --from=build /build/bin/llama-server /usr/local/bin/llama-server
EXPOSE 8000
ENTRYPOINT ["llama-server"]

View File

@@ -1,47 +1,140 @@
# llama.cpp + Gemma-4-12B (Vision) auf CUDA (AWS G7e / RTX PRO 6000). # llama-server für Qwen3.6 Vision — zwei Profile:
# docker compose --profile cuda up -d --build (EC2, RTX PRO 6000 Blackwell)
# docker compose --profile vulkan up -d --build (inhouse, Radeon AI Pro R9700)
# #
# Bewusst SCHLANK gehalten: keine ROCm/Vulkan-Workarounds. Wir testen, ob # Modelle nach ./models legen (Haupt-GGUF + zugehöriger mmproj aus demselben Repo!):
# CUDA die Instabilitaeten von vornherein vermeidet. Nur die inhaltlich noetigen # CUDA: unsloth/Qwen3.6-27B-GGUF → Qwen3.6-27B-UD-Q4_K_XL.gguf + mmproj-BF16.gguf
# Flags (--reasoning off gegen leeres content) bleiben. # Vulkan: unsloth/Qwen3.6-35B-A3B-GGUF → Qwen3.6-35B-A3B-UD-Q4_K_XL.gguf + mmproj-BF16.gguf
# #
# Voraussetzung: DLAMI mit NVIDIA Container Toolkit (docker --gpus all funktioniert). # Bewusst KEIN MTP: --mmproj + MTP ist laut Unsloth nicht unterstützt und hat
# # offene OOM-/Hänger-Bugs (llama.cpp #23371, #23430). Für Batch-Extraktion
# Start: docker compose -f docker-compose-cuda.yml up -d # irrelevant, da die Zeit im Image-Encoding steckt, nicht in der Generierung.
# Logs: docker compose -f docker-compose-cuda.yml logs -f
services: services:
llamacpp-gemma12b: llama-cuda:
image: ghcr.io/ggml-org/llama.cpp:server-cuda profiles: ["cuda"]
container_name: llamacpp-gemma12b build:
restart: unless-stopped context: .
init: true dockerfile: Dockerfile.cuda
# GPU-Zugriff ueber das NVIDIA Container Toolkit ports:
- "8000:8000"
volumes:
- ./models:/models
deploy: deploy:
resources: resources:
reservations: reservations:
devices: devices:
- driver: nvidia - driver: nvidia
count: 1 count: all
capabilities: [gpu] capabilities: [gpu]
ports:
- "8000:8080"
volumes:
- ~/.cache/llama.cpp:/root/.cache/llama.cpp
command: command:
- -hf - -m
- unsloth/gemma-4-12b-it-GGUF:UD-Q4_K_XL - /models/Qwen3.6-27B-UD-Q4_K_XL.gguf
- --mmproj
- /models/mmproj-BF16.gguf
- --host - --host
- 0.0.0.0 - 0.0.0.0
- --port - --port
- "8080" - "8000"
- --alias
- qwen3.6
- -ngl - -ngl
- "99" - "99"
- --ctx-size - -fa
- "16384" - "on"
- -c
- "32768"
- --parallel - --parallel
- "1" - "1"
- --jinja - --jinja
- --reasoning # Erlernte Stabilitäts-Settings (Vision + Checkpoints = OOM-Bug):
- "off" - --ctx-checkpoints
- "0"
# Qwen-VL braucht min. 1024 Image-Tokens für korrektes Grounding:
- --image-min-tokens
- "1024"
- --image-max-tokens
- "4096"
# Extraktion: niedrige Temperatur, kein Repeat-Penalty
- --temp
- "0.1"
- --top-p
- "0.95"
- --top-k
- "20"
- --repeat-penalty
- "1.0"
healthcheck:
test: ["CMD", "curl", "-sf", "http://localhost:8000/health"]
interval: 15s
timeout: 5s
retries: 40
start_period: 60s
restart: unless-stopped
llama-vulkan:
profiles: ["vulkan"]
build:
context: .
dockerfile: Dockerfile.vulkan
ports:
- "8000:8000"
volumes:
- ./models:/models
devices:
- /dev/dri:/dev/dri
- /dev/kfd:/dev/kfd
group_add:
- video
- render
security_opt:
- seccomp:unconfined
command:
- -m
- /models/Qwen3.6-35B-A3B-UD-Q4_K_XL.gguf
- --mmproj
- /models/mmproj-BF16.gguf
- --host
- 0.0.0.0
- --port
- "8000"
- --alias - --alias
- gemma-4-12b - qwen3.6
- -ngl
- "99"
- -fa
- "on"
- -c
- "32768"
- --parallel
- "1"
# KV-Cache quantisieren — 32GB VRAM, 35B-A3B + mmproj + Bildkontext:
- -ctk
- q8_0
- -ctv
- q8_0
- --jinja
- --ctx-checkpoints
- "0"
- --image-min-tokens
- "1024"
- --image-max-tokens
- "4096"
- --temp
- "0.1"
- --top-p
- "0.95"
- --top-k
- "20"
- --repeat-penalty
- "1.0"
# NOTFALL-Fallback bei Vision-Hängern/OOM unter RADV — mmproj auf CPU,
# Bildverarbeitung wird DEUTLICH langsamer. Nur aktivieren wenn nötig:
# - --no-mmproj-offload
healthcheck:
test: ["CMD", "curl", "-sf", "http://localhost:8000/health"]
interval: 15s
timeout: 5s
retries: 40
start_period: 120s
restart: unless-stopped

48
debug_text.ts Normal file
View File

@@ -0,0 +1,48 @@
import * as pdfjsLib from 'pdfjs-dist/legacy/build/pdf.js';
import * as fs from 'fs';
async function debugText(filePath: string) {
console.log(`\nLese PDF für Text-Debugging: ${filePath}`);
const dataBuffer = fs.readFileSync(filePath);
const loadingTask = pdfjsLib.getDocument({ data: new Uint8Array(dataBuffer) });
const pdfDocument = await loadingTask.promise;
console.log(`Anzahl Seiten: ${pdfDocument.numPages}`);
// Wir nehmen uns Seite 1 vor
const page = await pdfDocument.getPage(1);
const textContent = await page.getTextContent();
console.log(`\n--- RAW TEXT ITEMS (Die ersten 40 Fragmente) ---`);
for (let i = 0; i < Math.min(40, textContent.items.length); i++) {
const item = textContent.items[i] as any;
console.log(`Y: ${item.transform[5].toFixed(1).padStart(6)} | X: ${item.transform[4].toFixed(1).padStart(6)} | Text: "${item.str}"`);
}
// Unsere Sortier- und Bereinigungslogik anwenden
const sortedItems = textContent.items
.map((item: any) => ({
text: item.str,
x: item.transform[4],
y: item.transform[5]
}))
.sort((a: any, b: any) => {
if (Math.abs(b.y - a.y) > 5) return b.y - a.y;
return a.x - b.x;
});
const textRaw = sortedItems.map((i: any) => i.text).join('').toUpperCase().replace(/[^A-Z]/g, '');
console.log(`\n--- BEREINIGTER SUCH-STRING (Die ersten 200 Zeichen) ---`);
console.log(textRaw.substring(0, 200));
console.log(`\nEnthält 'BUYERINFORMATIONSHEET'? -> ${textRaw.includes('BUYERINFORMATIONSHEET')}`);
}
const args = process.argv.slice(2);
const pdfArgIndex = args.indexOf('--pdf');
if (pdfArgIndex !== -1 && args[pdfArgIndex + 1]) {
debugText(args[pdfArgIndex + 1]).catch(console.error);
} else {
console.error("Bitte --pdf Parameter angeben.");
}

2354
out/buyers.json Normal file

File diff suppressed because it is too large Load Diff

516
vision_runner.ts Normal file
View File

@@ -0,0 +1,516 @@
#!/usr/bin/env npx tsx
/**
* vision_runner.ts — Gleis B: Vision-LLM-Extraktion für Buyer Information Sheets
*
* Liest out/buyers.json (Ergebnis von Gleis A / batch_runner.ts), nimmt alle
* Einträge mit is_buyer_sheet === false, rendert die PDF-Seiten via pdftoppm
* (poppler-utils) zu PNGs und schickt sie an einen llama-server
* (OpenAI-kompatibel, Qwen3.6 + mmproj).
*
* Prinzip: Das VLM TRANSKRIBIERT nur (Rohwerte, verbatim). Sämtliche
* Normalisierung (Leer-Marker, State, down_payment, Datum) passiert
* deterministisch hier in TypeScript — identische Regeln wie Gleis A.
*
* Aufruf (AMD/Vulkan, inhouse):
* npx tsx vision_runner.ts \
* --input out/buyers.json \
* --pdf-root "/mnt/bizmatch-nas/AA Buyers NDA's/Buyers NDA's A-Z" \
* --api http://localhost:8000/v1 --limit 5
*
* Aufruf (NVIDIA/CUDA, EC2):
* npx tsx vision_runner.ts --input out/buyers.json \
* --pdf-root ~/data --api http://localhost:8000/v1 --limit 5
*
* Voraussetzungen: Node >= 18 (fetch), poppler-utils (pdftoppm) installiert.
*
* Outputs:
* out/buyers_vision.json — nur die Vision-Ergebnisse (Resume-Datei)
* out/buyers_merged.json — Gleis A + Gleis B zusammengeführt
*/
import { execFile } from "node:child_process";
import { promisify } from "node:util";
import * as fsp from "node:fs/promises";
import * as fs from "node:fs";
import * as path from "node:path";
import * as os from "node:os";
const execFileP = promisify(execFile);
// ---------------------------------------------------------------------------
// CLI
// ---------------------------------------------------------------------------
interface Args {
input: string;
pdfRoot: string;
api: string;
limit: number;
only: string | null;
dpi: number;
maxPages: number;
force: boolean;
outDir: string;
timeoutMs: number;
}
function parseArgs(): Args {
const a = process.argv.slice(2);
const get = (flag: string, def: string | null = null): string | null => {
const i = a.indexOf(flag);
return i >= 0 && a[i + 1] !== undefined ? a[i + 1] : def;
};
const input = get("--input", "out/buyers.json")!;
const pdfRoot = get("--pdf-root");
const api = (get("--api", "http://localhost:8000/v1") || "").replace(/\/+$/, "");
if (!pdfRoot) {
console.error("Fehler: --pdf-root <Verzeichnis> ist erforderlich.");
process.exit(1);
}
return {
input,
pdfRoot: pdfRoot.replace(/^~(?=\/)/, os.homedir()),
api,
limit: parseInt(get("--limit", "0")!, 10) || 0,
only: get("--only"),
dpi: parseInt(get("--dpi", "150")!, 10) || 150,
maxPages: parseInt(get("--max-pages", "8")!, 10) || 8,
force: a.includes("--force"),
outDir: get("--out-dir", path.dirname(input))!,
timeoutMs: (parseInt(get("--timeout", "180")!, 10) || 180) * 1000,
};
}
// ---------------------------------------------------------------------------
// Typen
// ---------------------------------------------------------------------------
interface BuyerRecord {
file_name: string;
is_buyer_sheet: boolean;
[k: string]: unknown;
}
/** Rohantwort des VLM — alles verbatim, Normalisierung erfolgt in TS. */
interface VisionRaw {
is_buyer_sheet: boolean;
info_page: number | null;
ca_page: number | null;
name_company: string | null;
prospective_buyer: string | null;
company: string | null;
phone: string | null;
cell: string | null;
email: string | null;
address: string | null;
state: string | null;
how_did_you_hear: string | null;
interested_in_updates: string | null;
types_of_business_raw: string | null;
background_experience: string | null;
total_purchase_price: string | null;
down_payment_raw: string | null;
date_of_introduction_raw: string | null;
}
// ---------------------------------------------------------------------------
// Normalisierung — identisch zu Gleis A halten!
// (Falls BuyerSheetParser.ts diese Funktionen exportiert, stattdessen
// importieren, damit beide Gleise garantiert dieselbe Logik nutzen.)
// ---------------------------------------------------------------------------
const NULL_MARKERS = /^(n|na|n\/a|none|nil|x|-+|\.+)$/i;
function cleanStr(v: string | null | undefined): string | null {
if (v == null) return null;
const t = String(v).replace(/\s+/g, " ").trim();
if (!t || NULL_MARKERS.test(t)) return null;
return t;
}
const STATE_MAP: Record<string, string> = {
alabama: "AL", alaska: "AK", arizona: "AZ", arkansas: "AR", california: "CA",
colorado: "CO", connecticut: "CT", delaware: "DE", florida: "FL", georgia: "GA",
hawaii: "HI", idaho: "ID", illinois: "IL", indiana: "IN", iowa: "IA",
kansas: "KS", kentucky: "KY", louisiana: "LA", maine: "ME", maryland: "MD",
massachusetts: "MA", michigan: "MI", minnesota: "MN", mississippi: "MS",
missouri: "MO", montana: "MT", nebraska: "NE", nevada: "NV",
"new hampshire": "NH", "new jersey": "NJ", "new mexico": "NM",
"new york": "NY", "north carolina": "NC", "north dakota": "ND", ohio: "OH",
oklahoma: "OK", oregon: "OR", pennsylvania: "PA", "rhode island": "RI",
"south carolina": "SC", "south dakota": "SD", tennessee: "TN", texas: "TX",
utah: "UT", vermont: "VT", virginia: "VA", washington: "WA",
"west virginia": "WV", wisconsin: "WI", wyoming: "WY",
"district of columbia": "DC", "washington dc": "DC",
};
const STATE_CODES = new Set(Object.values(STATE_MAP));
function normState(v: string | null): string | null {
const c = cleanStr(v);
if (!c) return null;
const up = c.toUpperCase().replace(/\./g, "");
if (up.length === 2 && STATE_CODES.has(up)) return up;
const full = STATE_MAP[c.toLowerCase().replace(/\./g, "")];
return full ?? c; // unbekannt: bereinigt durchreichen, nicht raten
}
/** "$350,000" → "350000"; "1.5M" → "1500000"; sonst null + raw. */
function normDownPayment(raw: string | null): { value: string | null; raw: string | null } {
const c = cleanStr(raw);
if (!c) return { value: null, raw: null };
const m = c.match(/^\$?\s*(\d[\d,]*(?:\.\d+)?)\s*([kKmM])?\s*$/);
if (!m) return { value: null, raw: c };
let num = parseFloat(m[1].replace(/,/g, ""));
if (m[2]) num *= /k/i.test(m[2]) ? 1_000 : 1_000_000;
if (!Number.isFinite(num) || num <= 0) return { value: null, raw: c };
return { value: String(Math.round(num)), raw: c };
}
const MONTHS: Record<string, number> = {
jan: 1, feb: 2, mar: 3, apr: 4, may: 5, jun: 6,
jul: 7, aug: 8, sep: 9, oct: 10, nov: 11, dec: 12,
};
function isoDate(y: number, mo: number, d: number): string | null {
if (y < 100) y += 2000;
if (y < 1990 || y > 2100 || mo < 1 || mo > 12 || d < 1 || d > 31) return null;
return `${y}-${String(mo).padStart(2, "0")}-${String(d).padStart(2, "0")}`;
}
/** US-Formate: 6/25/26, 06-25-2026, "June 25, 2026" → YYYY-MM-DD. */
function normDate(raw: string | null): string | null {
const c = cleanStr(raw);
if (!c) return null;
let m = c.match(/^(\d{1,2})[\/\-.](\d{1,2})[\/\-.](\d{2}|\d{4})$/);
if (m) return isoDate(parseInt(m[3], 10), parseInt(m[1], 10), parseInt(m[2], 10));
m = c.match(/^([A-Za-z]{3,9})\.?\s+(\d{1,2})(?:st|nd|rd|th)?,?\s+(\d{2}|\d{4})$/);
if (m) {
const mo = MONTHS[m[1].slice(0, 3).toLowerCase()];
if (mo) return isoDate(parseInt(m[3], 10), mo, parseInt(m[2], 10));
}
m = c.match(/^(\d{4})-(\d{2})-(\d{2})$/);
if (m) return isoDate(parseInt(m[1], 10), parseInt(m[2], 10), parseInt(m[3], 10));
return null;
}
// ---------------------------------------------------------------------------
// PDF-Index (rekursiv, Basename → voller Pfad) + Rendering
// ---------------------------------------------------------------------------
async function buildPdfIndex(root: string): Promise<Map<string, string>> {
const index = new Map<string, string>();
const stack = [root];
while (stack.length) {
const dir = stack.pop()!;
let entries: fs.Dirent[];
try {
entries = await fsp.readdir(dir, { withFileTypes: true });
} catch {
continue;
}
for (const e of entries) {
const p = path.join(dir, e.name);
if (e.isDirectory()) stack.push(p);
else if (e.isFile() && e.name.toLowerCase().endsWith(".pdf")) {
if (!index.has(e.name)) index.set(e.name, p);
}
}
}
return index;
}
async function renderPdf(pdfPath: string, dpi: number, maxPages: number, tmpDir: string): Promise<string[]> {
const prefix = path.join(tmpDir, "page");
await execFileP("pdftoppm", ["-png", "-r", String(dpi), "-l", String(maxPages), pdfPath, prefix], {
timeout: 120_000,
});
const pageNum = (f: string) => parseInt(f.match(/-(\d+)\.png$/)?.[1] ?? "0", 10);
const files = (await fsp.readdir(tmpDir))
.filter((f) => f.endsWith(".png"))
.sort((a, b) => pageNum(a) - pageNum(b))
.map((f) => path.join(tmpDir, f));
if (files.length === 0) throw new Error("pdftoppm hat keine Seiten erzeugt");
return files;
}
// ---------------------------------------------------------------------------
// VLM-Aufruf (llama-server, OpenAI-kompatibel, JSON-Schema)
// ---------------------------------------------------------------------------
const nullableString = { type: ["string", "null"] };
const nullableInt = { type: ["integer", "null"] };
const RESPONSE_SCHEMA = {
type: "object",
additionalProperties: false,
required: [
"is_buyer_sheet", "info_page", "ca_page", "name_company", "prospective_buyer",
"company", "phone", "cell", "email", "address", "state", "how_did_you_hear",
"interested_in_updates", "types_of_business_raw", "background_experience",
"total_purchase_price", "down_payment_raw", "date_of_introduction_raw",
],
properties: {
is_buyer_sheet: { type: "boolean" },
info_page: nullableInt,
ca_page: nullableInt,
name_company: nullableString,
prospective_buyer: nullableString,
company: nullableString,
phone: nullableString,
cell: nullableString,
email: nullableString,
address: nullableString,
state: nullableString,
how_did_you_hear: nullableString,
interested_in_updates: nullableString,
types_of_business_raw: nullableString,
background_experience: nullableString,
total_purchase_price: nullableString,
down_payment_raw: nullableString,
date_of_introduction_raw: nullableString,
},
} as const;
const SYSTEM_PROMPT =
"You are a precise document transcription engine for scanned business forms. " +
"You transcribe handwritten and typed form fields EXACTLY as written (verbatim), " +
"including typos. You never guess, infer, or invent values. If a field is blank " +
"or unreadable, you return null.";
const USER_PROMPT = `You see all pages of one PDF, in order (image 1 = page 1).
Task: Decide whether this document contains a business brokerage "BUYER INFORMATION SHEET" form, and if so, transcribe its fields.
1. is_buyer_sheet: true only if a page with the heading "BUYER INFORMATION SHEET" exists. If the document is something else (notes, listing, letter, other form), return is_buyer_sheet=false and null for every field.
2. info_page: page number (1-based) of the BUYER INFORMATION SHEET page, else null.
3. ca_page: page number of the confidentiality agreement page containing text like "PROSPECTIVE BUYER AGREES TO KEEP AND HOLD CONFIDENTIAL", else null.
4. From the info page, transcribe VERBATIM (exactly as written, do not normalize, do not expand abbreviations):
- name_company: value of the "Name/Company" line
- prospective_buyer: value of the "Prospective Buyer" line
- company: value of a separate "Company" line if present
- phone, cell, email, address, state
- how_did_you_hear: "How did you hear about us"
- interested_in_updates: answer/checkbox for receiving updates (transcribe what is marked, e.g. "Yes" or "No"), else null
- types_of_business_raw: "Type(s) of business interested in" (may span multiple lines — join with a space)
- background_experience: "Background/Experience" (may span multiple lines)
- total_purchase_price: "Total Purchase Price" as written
- down_payment_raw: "Down Payment" as written (e.g. "$350,000", "1.5M", "TBD")
5. date_of_introduction_raw: the date written next to the buyer's signature on the confidentiality agreement page, verbatim (e.g. "6/25/26"), else null.
A blank field, "N/A", "n", or an empty line = transcribe it as written; if truly empty, use null. Return only the JSON object.`;
async function fetchModelId(api: string): Promise<string> {
try {
const r = await fetch(`${api}/models`);
const j = (await r.json()) as { data?: Array<{ id?: string }> };
return j?.data?.[0]?.id ?? "unknown";
} catch {
return "unknown";
}
}
async function callVision(api: string, images: string[], timeoutMs: number): Promise<VisionRaw> {
const content: Array<Record<string, unknown>> = [{ type: "text", text: USER_PROMPT }];
for (const img of images) {
const b64 = await fsp.readFile(img, { encoding: "base64" });
content.push({ type: "image_url", image_url: { url: `data:image/png;base64,${b64}` } });
}
const body = {
model: "qwen3.6",
temperature: 0.1,
top_p: 0.95,
max_tokens: 1500,
chat_template_kwargs: { enable_thinking: false },
response_format: {
type: "json_schema",
json_schema: { name: "buyer_sheet", strict: true, schema: RESPONSE_SCHEMA },
},
messages: [
{ role: "system", content: SYSTEM_PROMPT },
{ role: "user", content },
],
};
const ctl = new AbortController();
const timer = setTimeout(() => ctl.abort(), timeoutMs);
try {
const r = await fetch(`${api}/chat/completions`, {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify(body),
signal: ctl.signal,
});
if (!r.ok) throw new Error(`HTTP ${r.status}: ${(await r.text()).slice(0, 300)}`);
const j = (await r.json()) as { choices?: Array<{ message?: { content?: string } }> };
const text = j?.choices?.[0]?.message?.content;
if (!text) throw new Error("Leere Antwort vom Server");
return JSON.parse(text) as VisionRaw;
} finally {
clearTimeout(timer);
}
}
// ---------------------------------------------------------------------------
// Merge: deterministischer Datensatz + Vision-Rohwerte → Zielstruktur
// ---------------------------------------------------------------------------
function mergeRecord(det: BuyerRecord, vis: VisionRaw, model: string): BuyerRecord {
const base: BuyerRecord = {
...det,
_parser: "vision",
_vision_model: model,
_vision_error: undefined,
};
delete (base as Record<string, unknown>)["_vision_error"];
if (!vis.is_buyer_sheet) {
return { ...base, is_buyer_sheet: false, _info_page: null, _ca_page: null };
}
const dp = normDownPayment(vis.down_payment_raw);
return {
...base,
is_buyer_sheet: true,
name_company: cleanStr(vis.name_company),
prospective_buyer: cleanStr(vis.prospective_buyer),
company: cleanStr(vis.company),
phone: cleanStr(vis.phone),
cell: cleanStr(vis.cell),
email: cleanStr(vis.email),
address: cleanStr(vis.address),
state: normState(vis.state),
how_did_you_hear: cleanStr(vis.how_did_you_hear),
interested_in_updates: cleanStr(vis.interested_in_updates),
types_of_business_raw: cleanStr(vis.types_of_business_raw),
background_experience: cleanStr(vis.background_experience),
total_purchase_price: cleanStr(vis.total_purchase_price),
down_payment: dp.value,
down_payment_raw: dp.raw,
// Vision-Datum bevorzugt; Fallback: deterministischer Wert (z.B. aus Dateinamen)
date_of_introduction: normDate(vis.date_of_introduction_raw) ?? (det.date_of_introduction as string | null) ?? null,
_info_page: vis.info_page,
_ca_page: vis.ca_page,
};
}
// ---------------------------------------------------------------------------
// Main
// ---------------------------------------------------------------------------
const RETRIES = 3;
function progress(line: string): void {
const cols = process.stderr.columns ?? 120;
process.stderr.write("\r" + line.slice(0, cols - 1).padEnd(cols - 1));
}
async function main(): Promise<void> {
const args = parseArgs();
// pdftoppm vorhanden?
try {
await execFileP("pdftoppm", ["-v"]);
} catch {
console.error("Fehler: pdftoppm nicht gefunden. Installieren: sudo apt install poppler-utils");
process.exit(1);
}
const all: BuyerRecord[] = JSON.parse(await fsp.readFile(args.input, "utf8"));
let targets = all.filter((r) => r.is_buyer_sheet === false);
if (args.only) targets = targets.filter((r) => r.file_name === args.only);
if (args.limit > 0) targets = targets.slice(0, args.limit);
const visionPath = path.join(args.outDir, "buyers_vision.json");
const mergedPath = path.join(args.outDir, "buyers_merged.json");
const done = new Map<string, BuyerRecord>();
try {
const prev: BuyerRecord[] = JSON.parse(await fsp.readFile(visionPath, "utf8"));
for (const r of prev) done.set(r.file_name, r);
} catch {
/* kein Resume-Stand */
}
console.error(`Indexiere PDFs unter ${args.pdfRoot} ...`);
const pdfIndex = await buildPdfIndex(args.pdfRoot);
console.error(`${pdfIndex.size} PDFs gefunden. ${targets.length} Einträge zu verarbeiten.`);
const model = await fetchModelId(args.api);
console.error(`Modell: ${model} @ ${args.api}`);
let ok = 0, notSheet = 0, errors = 0, skipped = 0;
for (let i = 0; i < targets.length; i++) {
const det = targets[i];
const tag = `[${i + 1}/${targets.length}] ${det.file_name}`;
const prev = done.get(det.file_name);
if (prev && !prev["_vision_error"] && !args.force) {
skipped++;
progress(`${tag} … übersprungen (bereits verarbeitet)`);
continue;
}
const pdfPath = pdfIndex.get(det.file_name);
if (!pdfPath) {
done.set(det.file_name, { ...det, _vision_error: "PDF nicht gefunden" });
errors++;
progress(`${tag} … FEHLER: PDF nicht gefunden`);
continue;
}
const tmpDir = await fsp.mkdtemp(path.join(os.tmpdir(), "bvs-"));
try {
progress(`${tag} … rendere`);
const images = await renderPdf(pdfPath, args.dpi, args.maxPages, tmpDir);
let vis: VisionRaw | null = null;
let lastErr = "";
for (let attempt = 1; attempt <= RETRIES; attempt++) {
try {
progress(`${tag} … VLM (${images.length} Seiten, Versuch ${attempt})`);
vis = await callVision(args.api, images, args.timeoutMs);
break;
} catch (e) {
lastErr = e instanceof Error ? e.message : String(e);
if (attempt < RETRIES) await new Promise((res) => setTimeout(res, 5000 * attempt));
}
}
if (!vis) {
done.set(det.file_name, { ...det, _vision_error: lastErr });
errors++;
progress(`${tag} … FEHLER: ${lastErr}`);
} else {
const merged = mergeRecord(det, vis, model);
done.set(det.file_name, merged);
if (merged.is_buyer_sheet) { ok++; progress(`${tag} … OK`); }
else { notSheet++; progress(`${tag} … kein Buyer Sheet`); }
}
} catch (e) {
const msg = e instanceof Error ? e.message : String(e);
done.set(det.file_name, { ...det, _vision_error: msg });
errors++;
progress(`${tag} … FEHLER: ${msg}`);
} finally {
await fsp.rm(tmpDir, { recursive: true, force: true });
}
// Inkrementell sichern (Resume-fähig)
await fsp.mkdir(args.outDir, { recursive: true });
await fsp.writeFile(visionPath, JSON.stringify([...done.values()], null, 2));
}
// Merge: Gleis A + Gleis B
const merged = all.map((r) => done.get(r.file_name) ?? r);
await fsp.writeFile(mergedPath, JSON.stringify(merged, null, 2));
process.stderr.write("\n");
console.error(
`Fertig. OK: ${ok}, kein Buyer Sheet: ${notSheet}, Fehler: ${errors}, übersprungen: ${skipped}`
);
console.error(`${visionPath}\n→ ${mergedPath}`);
}
main().catch((e) => {
console.error("\nAbbruch:", e instanceof Error ? e.message : e);
process.exit(1);
});