gfdfg
This commit is contained in:
@@ -2,6 +2,7 @@ import * as pdfjsLib from 'pdfjs-dist/legacy/build/pdf.js';
|
||||
import * as fs from 'fs';
|
||||
|
||||
export interface BuyerSheetData {
|
||||
file_name: string; // NEU
|
||||
is_buyer_sheet: boolean;
|
||||
name_company: string | null;
|
||||
prospective_buyer: string | null;
|
||||
@@ -21,8 +22,8 @@ export interface BuyerSheetData {
|
||||
date_of_introduction: string | null;
|
||||
_checkbox_pending: boolean;
|
||||
_parser: string;
|
||||
_info_page: number;
|
||||
_ca_page: number;
|
||||
_info_page: number | null;
|
||||
_ca_page: number | null;
|
||||
}
|
||||
|
||||
const LABELS = [
|
||||
@@ -34,16 +35,69 @@ const LABELS = [
|
||||
|
||||
export class DeterministicParser {
|
||||
|
||||
public async parsePdf(filePath: string): Promise<BuyerSheetData> {
|
||||
public async parsePdf(filePath: string, fileName: string): Promise<BuyerSheetData | null> {
|
||||
const dataBuffer = fs.readFileSync(filePath);
|
||||
const loadingTask = pdfjsLib.getDocument({ data: new Uint8Array(dataBuffer) });
|
||||
const pdfDocument = await loadingTask.promise;
|
||||
const page = await pdfDocument.getPage(1);
|
||||
|
||||
const textContent = await page.getTextContent();
|
||||
// Regel: Alles über 10 Seiten wird radikal ignoriert
|
||||
if (pdfDocument.numPages > 10) {
|
||||
return null;
|
||||
}
|
||||
|
||||
// 1. Textfragmente mit X, Y und Breite (Width) auslesen
|
||||
const mappedItems = textContent.items
|
||||
let infoPageNum: number | null = null;
|
||||
let caPageNum: number | null = null;
|
||||
let infoPageObj: any = null;
|
||||
let infoPageTextContent: any = null;
|
||||
|
||||
// ==========================================
|
||||
// 1. Dynamische Seitensuche (Visuell sortiert & kugelsicher)
|
||||
// ==========================================
|
||||
for (let i = 1; i <= pdfDocument.numPages; i++) {
|
||||
const page = await pdfDocument.getPage(i);
|
||||
const textContent = await page.getTextContent();
|
||||
|
||||
// Elemente mit Koordinaten versehen und wie ein Mensch lesen (von oben nach unten, links nach rechts)
|
||||
const sortedItems = textContent.items
|
||||
.map((item: any) => ({
|
||||
text: item.str,
|
||||
x: item.transform[4],
|
||||
y: item.transform[5]
|
||||
}))
|
||||
.sort((a: any, b: any) => {
|
||||
// Y-Toleranz für Buchstaben auf derselben Zeile
|
||||
if (Math.abs(b.y - a.y) > 5) {
|
||||
return b.y - a.y;
|
||||
}
|
||||
return a.x - b.x;
|
||||
});
|
||||
|
||||
// Wir werfen ALLE Leerzeichen, Striche, Punkte und unsichtbare Artefakte weg.
|
||||
// Übrig bleibt eine reine, unverwüstliche Buchstabenkette.
|
||||
const textRaw = sortedItems.map((i: any) => i.text).join('').toUpperCase().replace(/[^A-Z]/g, '');
|
||||
|
||||
// Anker-Suche in der sauberen Zeichenkette
|
||||
if (textRaw.includes('BUYERINFORMATIONSHEET')) {
|
||||
infoPageNum = i;
|
||||
infoPageObj = page;
|
||||
infoPageTextContent = textContent;
|
||||
}
|
||||
if (textRaw.includes('PROSPECTIVEBUYERAGREESTOKEEPANDHOLDCONFIDENTIAL')) {
|
||||
caPageNum = i;
|
||||
}
|
||||
}
|
||||
|
||||
// ==========================================
|
||||
// 2. Fallback für Bild/Scan PDFs
|
||||
// ==========================================
|
||||
if (!infoPageNum || !infoPageObj || !infoPageTextContent) {
|
||||
return this.createEmptyFallback(fileName, filePath, infoPageNum, caPageNum);
|
||||
}
|
||||
|
||||
// ==========================================
|
||||
// 3. Werte-Extraktion (nur auf der Info-Seite!)
|
||||
// ==========================================
|
||||
const mappedItems = infoPageTextContent.items
|
||||
.filter((item: any) => item.str.trim() !== '')
|
||||
.map((item: any) => ({
|
||||
text: item.str,
|
||||
@@ -51,9 +105,8 @@ export class DeterministicParser {
|
||||
y: item.transform[5],
|
||||
width: item.width
|
||||
}))
|
||||
.sort((a, b) => b.y - a.y || a.x - b.x); // Y absteigend (oben nach unten)
|
||||
.sort((a: any, b: any) => b.y - a.y || a.x - b.x);
|
||||
|
||||
// 2. Zeilen bilden (Y-Toleranz)
|
||||
const linesGrouped: { y: number, items: any[] }[] = [];
|
||||
let currentY: number | null = null;
|
||||
let currentItems = [];
|
||||
@@ -72,7 +125,6 @@ export class DeterministicParser {
|
||||
linesGrouped.push({ y: currentY, items: currentItems });
|
||||
}
|
||||
|
||||
// 3. Zerrissene Wörter reparieren (Kerning)
|
||||
const lines: { y: number, text: string }[] = [];
|
||||
for (const group of linesGrouped) {
|
||||
group.items.sort((a, b) => a.x - b.x);
|
||||
@@ -82,10 +134,7 @@ export class DeterministicParser {
|
||||
for (const item of group.items) {
|
||||
if (prevEnd !== -1) {
|
||||
const gap = item.x - prevEnd;
|
||||
// Wenn die Lücke größer als ~4 Pixel ist, ist es ein echtes Leerzeichen
|
||||
if (gap > 4) {
|
||||
lineStr += " ";
|
||||
}
|
||||
if (gap > 4) lineStr += " ";
|
||||
}
|
||||
lineStr += item.text;
|
||||
prevEnd = item.x + item.width;
|
||||
@@ -93,18 +142,13 @@ export class DeterministicParser {
|
||||
lines.push({ y: group.y, text: lineStr });
|
||||
}
|
||||
|
||||
// 4. Y-Intervall Parsing (Die "Nutzer-Idee")
|
||||
const rawResults: Record<string, string[]> = {};
|
||||
LABELS.forEach(lbl => rawResults[lbl] = []);
|
||||
|
||||
let currentLabel: string | null = null;
|
||||
// Längste Labels zuerst suchen, damit "EMAIL ADDRESS" vor "ADDRESS" gefunden wird
|
||||
const sortedLabels = [...LABELS].sort((a, b) => b.length - a.length);
|
||||
|
||||
for (const line of lines) {
|
||||
const textUpper = line.text.toUpperCase().replace(/\s+/g, '');
|
||||
|
||||
// Boilerplate ignorieren, der Labels enthält ("VERIFICATION OF DOWN PAYMENT")
|
||||
if (textUpper.includes("SELLERMAYREQUIREVERIFICATION")) continue;
|
||||
|
||||
let foundLabel: string | null = null;
|
||||
@@ -118,75 +162,87 @@ export class DeterministicParser {
|
||||
|
||||
if (foundLabel) {
|
||||
currentLabel = foundLabel;
|
||||
|
||||
// Falls der Wert direkt auf derselben Zeile steht (z.B. "NAME/COMPANY: Gunnar Schultz")
|
||||
if (line.text.includes(':')) {
|
||||
const parts = line.text.split(':');
|
||||
const val = parts.slice(1).join(':').replace(/_+/g, '').trim(); // Unterstriche entfernen
|
||||
if (val.length > 0) {
|
||||
rawResults[currentLabel].push(val);
|
||||
}
|
||||
const val = parts.slice(1).join(':').replace(/_+/g, '').trim();
|
||||
if (val.length > 0) rawResults[currentLabel].push(val);
|
||||
}
|
||||
} else if (currentLabel) {
|
||||
// Diese Zeile ist kein Label, gehört also zum aktuellen Bereich!
|
||||
const cleanVal = line.text.replace(/_+/g, '').trim();
|
||||
|
||||
// Footer und Artefakte ignorieren
|
||||
if (cleanVal.length > 0 && !cleanVal.includes("Doc ID") && !cleanVal.includes("Bizmatch")) {
|
||||
rawResults[currentLabel].push(cleanVal);
|
||||
}
|
||||
}
|
||||
}
|
||||
// --- NEU: Visuelle Checkbox Erkennung ---
|
||||
const visualCheckboxResult = await this.detectGraphicCheckbox(page, mappedItems);
|
||||
return this.mapToTargetStructure(rawResults, filePath, visualCheckboxResult);
|
||||
|
||||
const visualCheckboxResult = await this.detectGraphicCheckbox(infoPageObj, mappedItems);
|
||||
return this.mapToTargetStructure(rawResults, filePath, fileName, visualCheckboxResult, infoPageNum, caPageNum);
|
||||
}
|
||||
|
||||
private createEmptyFallback(fileName: string, filePath: string, infoPage: number | null, caPage: number | null): BuyerSheetData {
|
||||
let dateOfIntro = null;
|
||||
const dateMatch = filePath.match(/(\d{2})(\d{2})(\d{2})\.pdf$/);
|
||||
if (dateMatch) dateOfIntro = `20${dateMatch[3]}-${dateMatch[1]}-${dateMatch[2]}`;
|
||||
|
||||
return {
|
||||
file_name: fileName,
|
||||
is_buyer_sheet: false,
|
||||
name_company: null,
|
||||
prospective_buyer: null,
|
||||
company: null,
|
||||
phone: null,
|
||||
cell: null,
|
||||
email: null,
|
||||
address: null,
|
||||
state: null,
|
||||
how_did_you_hear: null,
|
||||
interested_in_updates: null,
|
||||
types_of_business_raw: null,
|
||||
background_experience: null,
|
||||
total_purchase_price: null,
|
||||
down_payment: null,
|
||||
down_payment_raw: null,
|
||||
date_of_introduction: dateOfIntro,
|
||||
_checkbox_pending: true,
|
||||
_parser: "deterministic", // Signal für den AI-Pass
|
||||
_info_page: infoPage,
|
||||
_ca_page: caPage
|
||||
};
|
||||
}
|
||||
|
||||
private mapToTargetStructure(
|
||||
raw: Record<string, string[]>,
|
||||
filePath: string,
|
||||
visualCheckboxResult: boolean | null // <-- Neuer Parameter
|
||||
fileName: string,
|
||||
visualCheckboxResult: boolean | null,
|
||||
infoPageNum: number | null,
|
||||
caPageNum: number | null
|
||||
): BuyerSheetData {
|
||||
const getVal = (label: string) => raw[label] && raw[label].length > 0 ? raw[label].join(' ') : null;
|
||||
|
||||
const nameCompany = getVal("NAME / COMPANY");
|
||||
|
||||
// ==========================================
|
||||
// 1. Telefon & Handy (Cell) separieren
|
||||
// ==========================================
|
||||
let phoneRaw = getVal("PHONE");
|
||||
let phoneClean: string | null = null;
|
||||
let cellClean: string | null = null;
|
||||
let phoneClean = null;
|
||||
let cellClean = null;
|
||||
|
||||
if (phoneRaw) {
|
||||
// Alles vor "FAX:" oder "CELL:" ist die Telefonnummer
|
||||
const phoneMatch = phoneRaw.split(/FAX:|CELL:/)[0];
|
||||
phoneClean = phoneMatch ? phoneMatch.trim() : null;
|
||||
|
||||
// Alles nach "CELL:" ist die Handynummer
|
||||
const cellMatch = phoneRaw.match(/CELL:\s*(.*)/);
|
||||
cellClean = cellMatch && cellMatch[1] ? cellMatch[1].trim() : null;
|
||||
}
|
||||
|
||||
// ==========================================
|
||||
// 2. Adresse bereinigen
|
||||
// ==========================================
|
||||
let addressClean = getVal("ADDRESS");
|
||||
if (addressClean) {
|
||||
// Den statischen Formulart-Subtext entfernen
|
||||
addressClean = addressClean.replace(/PO BOX \/ STREET\s+CITY \/ STATE \/ ZIP/g, '').trim();
|
||||
if (addressClean === '') addressClean = null;
|
||||
}
|
||||
|
||||
// ==========================================
|
||||
// 3. Checkbox Auswertung (Hybrid)
|
||||
// ==========================================
|
||||
const interestedRaw = getVal("ARE YOU INTERESTED");
|
||||
let interestedInUpdates: boolean | null = null;
|
||||
let checkboxPending = true;
|
||||
|
||||
if (interestedRaw) {
|
||||
// 1. Zuerst Text-Prüfung versuchen (wie vorher)
|
||||
if (/(✔|☑|X|✓)\s*YES/i.test(interestedRaw) || /YES\s*(✔|☑|X|✓)/i.test(interestedRaw)) {
|
||||
interestedInUpdates = true;
|
||||
checkboxPending = false;
|
||||
@@ -195,30 +251,21 @@ export class DeterministicParser {
|
||||
checkboxPending = false;
|
||||
}
|
||||
}
|
||||
|
||||
// 2. Wenn Text-Prüfung versagt hat, nutzen wir unser visuelles X-Koordinaten Ergebnis!
|
||||
if (checkboxPending && visualCheckboxResult !== null) {
|
||||
interestedInUpdates = visualCheckboxResult;
|
||||
checkboxPending = false;
|
||||
}
|
||||
|
||||
// ==========================================
|
||||
// 4. Down Payment & Datum
|
||||
// ==========================================
|
||||
const downPaymentRaw = getVal("DOWN PAYMENT")?.replace(/[^0-9,]/g, '') || null;
|
||||
const downPaymentClean = downPaymentRaw ? downPaymentRaw.replace(/,/g, '') : null;
|
||||
|
||||
let dateOfIntro = null;
|
||||
const dateMatch = filePath.match(/(\d{2})(\d{2})(\d{2})\.pdf$/);
|
||||
if (dateMatch) {
|
||||
dateOfIntro = `20${dateMatch[3]}-${dateMatch[1]}-${dateMatch[2]}`;
|
||||
}
|
||||
if (dateMatch) dateOfIntro = `20${dateMatch[3]}-${dateMatch[1]}-${dateMatch[2]}`;
|
||||
|
||||
// ==========================================
|
||||
// 5. JSON Return mit dynamischen Flags
|
||||
// ==========================================
|
||||
return {
|
||||
is_buyer_sheet: true, // Wird in der Praxis dynamisch gesetzt
|
||||
file_name: fileName,
|
||||
is_buyer_sheet: true,
|
||||
name_company: nameCompany,
|
||||
prospective_buyer: nameCompany ? nameCompany.split('/')[0].trim() : null,
|
||||
company: null,
|
||||
@@ -237,31 +284,25 @@ export class DeterministicParser {
|
||||
date_of_introduction: dateOfIntro,
|
||||
_checkbox_pending: checkboxPending,
|
||||
_parser: "deterministic",
|
||||
_info_page: 1, // Wird in der Praxis über eine Schleife ermittelt
|
||||
_ca_page: 2 // Wird in der Praxis über eine Schleife ermittelt
|
||||
_info_page: infoPageNum,
|
||||
_ca_page: caPageNum
|
||||
};
|
||||
}
|
||||
|
||||
private async detectGraphicCheckbox(page: any, textItems: any[]): Promise<boolean | null> {
|
||||
let yesX = null, noX = null, targetY = null;
|
||||
|
||||
// 1. Koordinaten von YES und NO suchen
|
||||
for (const item of textItems) {
|
||||
const textUpper = item.text.toUpperCase().trim();
|
||||
if (textUpper.includes("YES")) { yesX = item.x; targetY = item.y; }
|
||||
}
|
||||
for (const item of textItems) {
|
||||
const textUpper = item.text.toUpperCase().trim();
|
||||
// Wir suchen NO auf derselben Höhe (Toleranz 5px)
|
||||
if (textUpper.includes("NO") && targetY !== null && Math.abs(item.y - targetY) < 5) {
|
||||
noX = item.x;
|
||||
}
|
||||
if (textUpper.includes("NO") && targetY !== null && Math.abs(item.y - targetY) < 5) noX = item.x;
|
||||
}
|
||||
|
||||
if (yesX === null || noX === null || targetY === null) return null;
|
||||
|
||||
// ==========================================
|
||||
// BILD-ERKENNUNG (Der 16x16 Stempel-Trick)
|
||||
// ==========================================
|
||||
const opList = await page.getOperatorList();
|
||||
let currentTransform = [1, 0, 0, 1, 0, 0];
|
||||
|
||||
@@ -269,10 +310,7 @@ export class DeterministicParser {
|
||||
const fn = opList.fnArray[i];
|
||||
const args = opList.argsArray[i];
|
||||
|
||||
if (fn === pdfjsLib.OPS.transform) {
|
||||
currentTransform = args;
|
||||
}
|
||||
// Wenn ein Bild auf die Seite gezeichnet wird
|
||||
if (fn === pdfjsLib.OPS.transform) currentTransform = args;
|
||||
else if (
|
||||
fn === pdfjsLib.OPS.paintImageXObject ||
|
||||
fn === pdfjsLib.OPS.paintInlineImageXObject ||
|
||||
@@ -283,25 +321,13 @@ export class DeterministicParser {
|
||||
const imgX = currentTransform[4];
|
||||
const imgY = currentTransform[5];
|
||||
|
||||
// Wir filtern nach "Stempeln" (kleine Bilder unter 40x40 Pixeln)
|
||||
if (width < 40 && height < 40) {
|
||||
// Befindet sich der Stempel auf unserer Ziel-Zeile?
|
||||
// (Wir erlauben 50px Toleranz, da Y=378 vs Y=417)
|
||||
if (Math.abs(imgY - targetY) < 50) {
|
||||
const distToYes = Math.abs(imgX - yesX);
|
||||
const distToNo = Math.abs(imgX - noX);
|
||||
|
||||
// Wenn das Bildchen näher an YES ist
|
||||
if (distToYes < distToNo) {
|
||||
return true;
|
||||
} else {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
if (width < 40 && height < 40 && Math.abs(imgY - targetY) < 50) {
|
||||
const distToYes = Math.abs(imgX - yesX);
|
||||
const distToNo = Math.abs(imgX - noX);
|
||||
return distToYes < distToNo;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return null;
|
||||
}
|
||||
}
|
||||
101
batch_runner.ts
Normal file
101
batch_runner.ts
Normal file
@@ -0,0 +1,101 @@
|
||||
import * as fs from 'fs';
|
||||
import * as path from 'path';
|
||||
import { DeterministicParser } from './BuyerSheetParser';
|
||||
|
||||
async function main() {
|
||||
// 1. Argument-Check
|
||||
const args = process.argv.slice(2);
|
||||
|
||||
// Verzeichnis Parameter
|
||||
const dirArgIndex = args.indexOf('--dir');
|
||||
if (dirArgIndex === -1 || !args[dirArgIndex + 1]) {
|
||||
console.error("Fehler: Bitte ein Verzeichnis angeben!");
|
||||
console.error("Nutzung: npx tsx batch_runner.ts --dir \"/Pfad/zum/Ordner\" [--limit 10]");
|
||||
process.exit(1);
|
||||
}
|
||||
const dirPath = args[dirArgIndex + 1];
|
||||
|
||||
// Limit Parameter
|
||||
const limitArgIndex = args.indexOf('--limit');
|
||||
let limit = -1; // -1 bedeutet: Kein Limit, alle verarbeiten
|
||||
if (limitArgIndex !== -1 && args[limitArgIndex + 1]) {
|
||||
limit = parseInt(args[limitArgIndex + 1], 10);
|
||||
}
|
||||
|
||||
if (!fs.existsSync(dirPath) || !fs.statSync(dirPath).isDirectory()) {
|
||||
console.error(`Fehler: Der Pfad "${dirPath}" existiert nicht oder ist kein Verzeichnis.`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
// 2. PDFs finden und Metadaten (für die Sortierung) auslesen
|
||||
const filesWithStats = fs.readdirSync(dirPath)
|
||||
.filter(f => f.toLowerCase().endsWith('.pdf'))
|
||||
.map(file => {
|
||||
const fullPath = path.join(dirPath, file);
|
||||
return {
|
||||
file,
|
||||
fullPath,
|
||||
// Änderungsdatum der Datei auslesen (in Millisekunden)
|
||||
mtime: fs.statSync(fullPath).mtimeMs
|
||||
};
|
||||
});
|
||||
|
||||
if (filesWithStats.length === 0) {
|
||||
console.log("Keine PDFs in diesem Verzeichnis gefunden.");
|
||||
process.exit(0);
|
||||
}
|
||||
|
||||
// 3. Absteigend sortieren (Neueste zuerst)
|
||||
filesWithStats.sort((a, b) => b.mtime - a.mtime);
|
||||
|
||||
// 4. Limit anwenden (falls gesetzt)
|
||||
const filesToProcess = limit > 0 ? filesWithStats.slice(0, limit) : filesWithStats;
|
||||
|
||||
console.log(`Starte Verarbeitung: ${filesToProcess.length} von ${filesWithStats.length} PDFs ausgewählt (Sortierung: Neueste zuerst)...\n`);
|
||||
|
||||
const parser = new DeterministicParser();
|
||||
const finalResults = [];
|
||||
let skippedCounter = 0;
|
||||
|
||||
// 5. PDFs iterieren
|
||||
for (const fileObj of filesToProcess) {
|
||||
const { file, fullPath } = fileObj;
|
||||
process.stdout.write(`-> Verarbeite: ${file} ... `);
|
||||
|
||||
try {
|
||||
const data = await parser.parsePdf(fullPath, file);
|
||||
|
||||
if (data === null) {
|
||||
console.log("ÜBERSPRUNGEN (> 10 Seiten)");
|
||||
skippedCounter++;
|
||||
} else if (data.is_buyer_sheet === false) {
|
||||
console.log("IMAGE SCAN (Zuweisung an Vision AI)");
|
||||
finalResults.push(data);
|
||||
} else {
|
||||
console.log("ERFOLGREICH");
|
||||
finalResults.push(data);
|
||||
}
|
||||
} catch (error) {
|
||||
console.log("FEHLER BEIM PARSEN");
|
||||
console.error(error);
|
||||
}
|
||||
}
|
||||
|
||||
// 6. JSON Export unter ./out/buyers.json
|
||||
const outDir = path.join(process.cwd(), 'out');
|
||||
if (!fs.existsSync(outDir)) {
|
||||
fs.mkdirSync(outDir);
|
||||
}
|
||||
|
||||
const outPath = path.join(outDir, 'buyers.json');
|
||||
fs.writeFileSync(outPath, JSON.stringify(finalResults, null, 2), 'utf-8');
|
||||
|
||||
console.log(`\n=================================================`);
|
||||
console.log(`Zusammenfassung:`);
|
||||
console.log(` Verarbeitet: ${finalResults.length}`);
|
||||
console.log(` Verworfen (>10 Seiten): ${skippedCounter}`);
|
||||
console.log(` Export gespeichert in: ${outPath}`);
|
||||
console.log(`=================================================`);
|
||||
}
|
||||
|
||||
main();
|
||||
35
bayarea-ai-server/Dockerfile.cuda
Normal file
35
bayarea-ai-server/Dockerfile.cuda
Normal file
@@ -0,0 +1,35 @@
|
||||
# llama-server für Qwen3.6 Vision — NVIDIA RTX PRO 6000 Blackwell (sm_120)
|
||||
#
|
||||
# WICHTIG:
|
||||
# - Muss aus aktuellem Source gebaut werden: Qwen3.6 nutzt ein neues
|
||||
# Rope-Encoding (rope.dimension_sections 3 statt 4), alte Images/Builds
|
||||
# brechen mit "wrong array length".
|
||||
# - Blackwell braucht CUDA >= 12.8. KEIN CUDA 13.2 verwenden — erzeugt
|
||||
# mit Qwen3.6 Gibberish (bekannter NVIDIA-Bug).
|
||||
|
||||
FROM nvidia/cuda:12.8.1-devel-ubuntu24.04 AS build
|
||||
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
git cmake build-essential libcurl4-openssl-dev \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
RUN git clone --depth 1 https://github.com/ggml-org/llama.cpp /src
|
||||
|
||||
# 120 = Blackwell (RTX PRO 6000). Für andere Karten anpassen.
|
||||
RUN cmake /src -B /build \
|
||||
-DGGML_CUDA=ON \
|
||||
-DCMAKE_CUDA_ARCHITECTURES=120 \
|
||||
-DBUILD_SHARED_LIBS=OFF \
|
||||
-DLLAMA_CURL=ON \
|
||||
&& cmake --build /build --config Release -j --target llama-server
|
||||
|
||||
FROM nvidia/cuda:12.8.1-runtime-ubuntu24.04
|
||||
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
libcurl4 libgomp1 curl \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
COPY --from=build /build/bin/llama-server /usr/local/bin/llama-server
|
||||
|
||||
EXPOSE 8000
|
||||
ENTRYPOINT ["llama-server"]
|
||||
32
bayarea-ai-server/Dockerfile.vulkan
Normal file
32
bayarea-ai-server/Dockerfile.vulkan
Normal file
@@ -0,0 +1,32 @@
|
||||
# llama-server für Qwen3.6 Vision — AMD Radeon AI Pro R9700, Vulkan-Backend
|
||||
#
|
||||
# Muss aus aktuellem Source gebaut werden (Qwen3.6-Rope-Änderung, s. Dockerfile.cuda).
|
||||
# Vulkan statt ROCm — hat sich bei Vision als stabiler erwiesen.
|
||||
|
||||
FROM ubuntu:24.04 AS build
|
||||
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
git cmake build-essential libcurl4-openssl-dev \
|
||||
libvulkan-dev glslc \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
RUN git clone --depth 1 https://github.com/ggml-org/llama.cpp /src
|
||||
|
||||
RUN cmake /src -B /build \
|
||||
-DGGML_VULKAN=ON \
|
||||
-DBUILD_SHARED_LIBS=OFF \
|
||||
-DLLAMA_CURL=ON \
|
||||
&& cmake --build /build --config Release -j --target llama-server
|
||||
|
||||
FROM ubuntu:24.04
|
||||
|
||||
# mesa-vulkan-drivers = RADV-Treiber im Container (GPU via /dev/dri durchgereicht)
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
libvulkan1 mesa-vulkan-drivers vulkan-tools \
|
||||
libcurl4 libgomp1 curl \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
COPY --from=build /build/bin/llama-server /usr/local/bin/llama-server
|
||||
|
||||
EXPOSE 8000
|
||||
ENTRYPOINT ["llama-server"]
|
||||
@@ -1,47 +1,140 @@
|
||||
# llama.cpp + Gemma-4-12B (Vision) auf CUDA (AWS G7e / RTX PRO 6000).
|
||||
# llama-server für Qwen3.6 Vision — zwei Profile:
|
||||
# docker compose --profile cuda up -d --build (EC2, RTX PRO 6000 Blackwell)
|
||||
# docker compose --profile vulkan up -d --build (inhouse, Radeon AI Pro R9700)
|
||||
#
|
||||
# Bewusst SCHLANK gehalten: keine ROCm/Vulkan-Workarounds. Wir testen, ob
|
||||
# CUDA die Instabilitaeten von vornherein vermeidet. Nur die inhaltlich noetigen
|
||||
# Flags (--reasoning off gegen leeres content) bleiben.
|
||||
# Modelle nach ./models legen (Haupt-GGUF + zugehöriger mmproj aus demselben Repo!):
|
||||
# CUDA: unsloth/Qwen3.6-27B-GGUF → Qwen3.6-27B-UD-Q4_K_XL.gguf + mmproj-BF16.gguf
|
||||
# Vulkan: unsloth/Qwen3.6-35B-A3B-GGUF → Qwen3.6-35B-A3B-UD-Q4_K_XL.gguf + mmproj-BF16.gguf
|
||||
#
|
||||
# Voraussetzung: DLAMI mit NVIDIA Container Toolkit (docker --gpus all funktioniert).
|
||||
#
|
||||
# Start: docker compose -f docker-compose-cuda.yml up -d
|
||||
# Logs: docker compose -f docker-compose-cuda.yml logs -f
|
||||
# Bewusst KEIN MTP: --mmproj + MTP ist laut Unsloth nicht unterstützt und hat
|
||||
# offene OOM-/Hänger-Bugs (llama.cpp #23371, #23430). Für Batch-Extraktion
|
||||
# irrelevant, da die Zeit im Image-Encoding steckt, nicht in der Generierung.
|
||||
|
||||
services:
|
||||
llamacpp-gemma12b:
|
||||
image: ghcr.io/ggml-org/llama.cpp:server-cuda
|
||||
container_name: llamacpp-gemma12b
|
||||
restart: unless-stopped
|
||||
init: true
|
||||
# GPU-Zugriff ueber das NVIDIA Container Toolkit
|
||||
llama-cuda:
|
||||
profiles: ["cuda"]
|
||||
build:
|
||||
context: .
|
||||
dockerfile: Dockerfile.cuda
|
||||
ports:
|
||||
- "8000:8000"
|
||||
volumes:
|
||||
- ./models:/models
|
||||
deploy:
|
||||
resources:
|
||||
reservations:
|
||||
devices:
|
||||
- driver: nvidia
|
||||
count: 1
|
||||
count: all
|
||||
capabilities: [gpu]
|
||||
ports:
|
||||
- "8000:8080"
|
||||
volumes:
|
||||
- ~/.cache/llama.cpp:/root/.cache/llama.cpp
|
||||
command:
|
||||
- -hf
|
||||
- unsloth/gemma-4-12b-it-GGUF:UD-Q4_K_XL
|
||||
- -m
|
||||
- /models/Qwen3.6-27B-UD-Q4_K_XL.gguf
|
||||
- --mmproj
|
||||
- /models/mmproj-BF16.gguf
|
||||
- --host
|
||||
- 0.0.0.0
|
||||
- --port
|
||||
- "8080"
|
||||
- "8000"
|
||||
- --alias
|
||||
- qwen3.6
|
||||
- -ngl
|
||||
- "99"
|
||||
- --ctx-size
|
||||
- "16384"
|
||||
- -fa
|
||||
- "on"
|
||||
- -c
|
||||
- "32768"
|
||||
- --parallel
|
||||
- "1"
|
||||
- --jinja
|
||||
- --reasoning
|
||||
- "off"
|
||||
# Erlernte Stabilitäts-Settings (Vision + Checkpoints = OOM-Bug):
|
||||
- --ctx-checkpoints
|
||||
- "0"
|
||||
# Qwen-VL braucht min. 1024 Image-Tokens für korrektes Grounding:
|
||||
- --image-min-tokens
|
||||
- "1024"
|
||||
- --image-max-tokens
|
||||
- "4096"
|
||||
# Extraktion: niedrige Temperatur, kein Repeat-Penalty
|
||||
- --temp
|
||||
- "0.1"
|
||||
- --top-p
|
||||
- "0.95"
|
||||
- --top-k
|
||||
- "20"
|
||||
- --repeat-penalty
|
||||
- "1.0"
|
||||
healthcheck:
|
||||
test: ["CMD", "curl", "-sf", "http://localhost:8000/health"]
|
||||
interval: 15s
|
||||
timeout: 5s
|
||||
retries: 40
|
||||
start_period: 60s
|
||||
restart: unless-stopped
|
||||
|
||||
llama-vulkan:
|
||||
profiles: ["vulkan"]
|
||||
build:
|
||||
context: .
|
||||
dockerfile: Dockerfile.vulkan
|
||||
ports:
|
||||
- "8000:8000"
|
||||
volumes:
|
||||
- ./models:/models
|
||||
devices:
|
||||
- /dev/dri:/dev/dri
|
||||
- /dev/kfd:/dev/kfd
|
||||
group_add:
|
||||
- video
|
||||
- render
|
||||
security_opt:
|
||||
- seccomp:unconfined
|
||||
command:
|
||||
- -m
|
||||
- /models/Qwen3.6-35B-A3B-UD-Q4_K_XL.gguf
|
||||
- --mmproj
|
||||
- /models/mmproj-BF16.gguf
|
||||
- --host
|
||||
- 0.0.0.0
|
||||
- --port
|
||||
- "8000"
|
||||
- --alias
|
||||
- gemma-4-12b
|
||||
- qwen3.6
|
||||
- -ngl
|
||||
- "99"
|
||||
- -fa
|
||||
- "on"
|
||||
- -c
|
||||
- "32768"
|
||||
- --parallel
|
||||
- "1"
|
||||
# KV-Cache quantisieren — 32GB VRAM, 35B-A3B + mmproj + Bildkontext:
|
||||
- -ctk
|
||||
- q8_0
|
||||
- -ctv
|
||||
- q8_0
|
||||
- --jinja
|
||||
- --ctx-checkpoints
|
||||
- "0"
|
||||
- --image-min-tokens
|
||||
- "1024"
|
||||
- --image-max-tokens
|
||||
- "4096"
|
||||
- --temp
|
||||
- "0.1"
|
||||
- --top-p
|
||||
- "0.95"
|
||||
- --top-k
|
||||
- "20"
|
||||
- --repeat-penalty
|
||||
- "1.0"
|
||||
# NOTFALL-Fallback bei Vision-Hängern/OOM unter RADV — mmproj auf CPU,
|
||||
# Bildverarbeitung wird DEUTLICH langsamer. Nur aktivieren wenn nötig:
|
||||
# - --no-mmproj-offload
|
||||
healthcheck:
|
||||
test: ["CMD", "curl", "-sf", "http://localhost:8000/health"]
|
||||
interval: 15s
|
||||
timeout: 5s
|
||||
retries: 40
|
||||
start_period: 120s
|
||||
restart: unless-stopped
|
||||
48
debug_text.ts
Normal file
48
debug_text.ts
Normal file
@@ -0,0 +1,48 @@
|
||||
import * as pdfjsLib from 'pdfjs-dist/legacy/build/pdf.js';
|
||||
import * as fs from 'fs';
|
||||
|
||||
async function debugText(filePath: string) {
|
||||
console.log(`\nLese PDF für Text-Debugging: ${filePath}`);
|
||||
|
||||
const dataBuffer = fs.readFileSync(filePath);
|
||||
const loadingTask = pdfjsLib.getDocument({ data: new Uint8Array(dataBuffer) });
|
||||
const pdfDocument = await loadingTask.promise;
|
||||
|
||||
console.log(`Anzahl Seiten: ${pdfDocument.numPages}`);
|
||||
|
||||
// Wir nehmen uns Seite 1 vor
|
||||
const page = await pdfDocument.getPage(1);
|
||||
const textContent = await page.getTextContent();
|
||||
|
||||
console.log(`\n--- RAW TEXT ITEMS (Die ersten 40 Fragmente) ---`);
|
||||
for (let i = 0; i < Math.min(40, textContent.items.length); i++) {
|
||||
const item = textContent.items[i] as any;
|
||||
console.log(`Y: ${item.transform[5].toFixed(1).padStart(6)} | X: ${item.transform[4].toFixed(1).padStart(6)} | Text: "${item.str}"`);
|
||||
}
|
||||
|
||||
// Unsere Sortier- und Bereinigungslogik anwenden
|
||||
const sortedItems = textContent.items
|
||||
.map((item: any) => ({
|
||||
text: item.str,
|
||||
x: item.transform[4],
|
||||
y: item.transform[5]
|
||||
}))
|
||||
.sort((a: any, b: any) => {
|
||||
if (Math.abs(b.y - a.y) > 5) return b.y - a.y;
|
||||
return a.x - b.x;
|
||||
});
|
||||
|
||||
const textRaw = sortedItems.map((i: any) => i.text).join('').toUpperCase().replace(/[^A-Z]/g, '');
|
||||
|
||||
console.log(`\n--- BEREINIGTER SUCH-STRING (Die ersten 200 Zeichen) ---`);
|
||||
console.log(textRaw.substring(0, 200));
|
||||
console.log(`\nEnthält 'BUYERINFORMATIONSHEET'? -> ${textRaw.includes('BUYERINFORMATIONSHEET')}`);
|
||||
}
|
||||
|
||||
const args = process.argv.slice(2);
|
||||
const pdfArgIndex = args.indexOf('--pdf');
|
||||
if (pdfArgIndex !== -1 && args[pdfArgIndex + 1]) {
|
||||
debugText(args[pdfArgIndex + 1]).catch(console.error);
|
||||
} else {
|
||||
console.error("Bitte --pdf Parameter angeben.");
|
||||
}
|
||||
2354
out/buyers.json
Normal file
2354
out/buyers.json
Normal file
File diff suppressed because it is too large
Load Diff
516
vision_runner.ts
Normal file
516
vision_runner.ts
Normal file
@@ -0,0 +1,516 @@
|
||||
#!/usr/bin/env npx tsx
|
||||
/**
|
||||
* vision_runner.ts — Gleis B: Vision-LLM-Extraktion für Buyer Information Sheets
|
||||
*
|
||||
* Liest out/buyers.json (Ergebnis von Gleis A / batch_runner.ts), nimmt alle
|
||||
* Einträge mit is_buyer_sheet === false, rendert die PDF-Seiten via pdftoppm
|
||||
* (poppler-utils) zu PNGs und schickt sie an einen llama-server
|
||||
* (OpenAI-kompatibel, Qwen3.6 + mmproj).
|
||||
*
|
||||
* Prinzip: Das VLM TRANSKRIBIERT nur (Rohwerte, verbatim). Sämtliche
|
||||
* Normalisierung (Leer-Marker, State, down_payment, Datum) passiert
|
||||
* deterministisch hier in TypeScript — identische Regeln wie Gleis A.
|
||||
*
|
||||
* Aufruf (AMD/Vulkan, inhouse):
|
||||
* npx tsx vision_runner.ts \
|
||||
* --input out/buyers.json \
|
||||
* --pdf-root "/mnt/bizmatch-nas/AA Buyers NDA's/Buyers NDA's A-Z" \
|
||||
* --api http://localhost:8000/v1 --limit 5
|
||||
*
|
||||
* Aufruf (NVIDIA/CUDA, EC2):
|
||||
* npx tsx vision_runner.ts --input out/buyers.json \
|
||||
* --pdf-root ~/data --api http://localhost:8000/v1 --limit 5
|
||||
*
|
||||
* Voraussetzungen: Node >= 18 (fetch), poppler-utils (pdftoppm) installiert.
|
||||
*
|
||||
* Outputs:
|
||||
* out/buyers_vision.json — nur die Vision-Ergebnisse (Resume-Datei)
|
||||
* out/buyers_merged.json — Gleis A + Gleis B zusammengeführt
|
||||
*/
|
||||
|
||||
import { execFile } from "node:child_process";
|
||||
import { promisify } from "node:util";
|
||||
import * as fsp from "node:fs/promises";
|
||||
import * as fs from "node:fs";
|
||||
import * as path from "node:path";
|
||||
import * as os from "node:os";
|
||||
|
||||
const execFileP = promisify(execFile);
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// CLI
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
interface Args {
|
||||
input: string;
|
||||
pdfRoot: string;
|
||||
api: string;
|
||||
limit: number;
|
||||
only: string | null;
|
||||
dpi: number;
|
||||
maxPages: number;
|
||||
force: boolean;
|
||||
outDir: string;
|
||||
timeoutMs: number;
|
||||
}
|
||||
|
||||
function parseArgs(): Args {
|
||||
const a = process.argv.slice(2);
|
||||
const get = (flag: string, def: string | null = null): string | null => {
|
||||
const i = a.indexOf(flag);
|
||||
return i >= 0 && a[i + 1] !== undefined ? a[i + 1] : def;
|
||||
};
|
||||
const input = get("--input", "out/buyers.json")!;
|
||||
const pdfRoot = get("--pdf-root");
|
||||
const api = (get("--api", "http://localhost:8000/v1") || "").replace(/\/+$/, "");
|
||||
if (!pdfRoot) {
|
||||
console.error("Fehler: --pdf-root <Verzeichnis> ist erforderlich.");
|
||||
process.exit(1);
|
||||
}
|
||||
return {
|
||||
input,
|
||||
pdfRoot: pdfRoot.replace(/^~(?=\/)/, os.homedir()),
|
||||
api,
|
||||
limit: parseInt(get("--limit", "0")!, 10) || 0,
|
||||
only: get("--only"),
|
||||
dpi: parseInt(get("--dpi", "150")!, 10) || 150,
|
||||
maxPages: parseInt(get("--max-pages", "8")!, 10) || 8,
|
||||
force: a.includes("--force"),
|
||||
outDir: get("--out-dir", path.dirname(input))!,
|
||||
timeoutMs: (parseInt(get("--timeout", "180")!, 10) || 180) * 1000,
|
||||
};
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Typen
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
interface BuyerRecord {
|
||||
file_name: string;
|
||||
is_buyer_sheet: boolean;
|
||||
[k: string]: unknown;
|
||||
}
|
||||
|
||||
/** Rohantwort des VLM — alles verbatim, Normalisierung erfolgt in TS. */
|
||||
interface VisionRaw {
|
||||
is_buyer_sheet: boolean;
|
||||
info_page: number | null;
|
||||
ca_page: number | null;
|
||||
name_company: string | null;
|
||||
prospective_buyer: string | null;
|
||||
company: string | null;
|
||||
phone: string | null;
|
||||
cell: string | null;
|
||||
email: string | null;
|
||||
address: string | null;
|
||||
state: string | null;
|
||||
how_did_you_hear: string | null;
|
||||
interested_in_updates: string | null;
|
||||
types_of_business_raw: string | null;
|
||||
background_experience: string | null;
|
||||
total_purchase_price: string | null;
|
||||
down_payment_raw: string | null;
|
||||
date_of_introduction_raw: string | null;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Normalisierung — identisch zu Gleis A halten!
|
||||
// (Falls BuyerSheetParser.ts diese Funktionen exportiert, stattdessen
|
||||
// importieren, damit beide Gleise garantiert dieselbe Logik nutzen.)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const NULL_MARKERS = /^(n|na|n\/a|none|nil|x|-+|\.+)$/i;
|
||||
|
||||
function cleanStr(v: string | null | undefined): string | null {
|
||||
if (v == null) return null;
|
||||
const t = String(v).replace(/\s+/g, " ").trim();
|
||||
if (!t || NULL_MARKERS.test(t)) return null;
|
||||
return t;
|
||||
}
|
||||
|
||||
const STATE_MAP: Record<string, string> = {
|
||||
alabama: "AL", alaska: "AK", arizona: "AZ", arkansas: "AR", california: "CA",
|
||||
colorado: "CO", connecticut: "CT", delaware: "DE", florida: "FL", georgia: "GA",
|
||||
hawaii: "HI", idaho: "ID", illinois: "IL", indiana: "IN", iowa: "IA",
|
||||
kansas: "KS", kentucky: "KY", louisiana: "LA", maine: "ME", maryland: "MD",
|
||||
massachusetts: "MA", michigan: "MI", minnesota: "MN", mississippi: "MS",
|
||||
missouri: "MO", montana: "MT", nebraska: "NE", nevada: "NV",
|
||||
"new hampshire": "NH", "new jersey": "NJ", "new mexico": "NM",
|
||||
"new york": "NY", "north carolina": "NC", "north dakota": "ND", ohio: "OH",
|
||||
oklahoma: "OK", oregon: "OR", pennsylvania: "PA", "rhode island": "RI",
|
||||
"south carolina": "SC", "south dakota": "SD", tennessee: "TN", texas: "TX",
|
||||
utah: "UT", vermont: "VT", virginia: "VA", washington: "WA",
|
||||
"west virginia": "WV", wisconsin: "WI", wyoming: "WY",
|
||||
"district of columbia": "DC", "washington dc": "DC",
|
||||
};
|
||||
const STATE_CODES = new Set(Object.values(STATE_MAP));
|
||||
|
||||
function normState(v: string | null): string | null {
|
||||
const c = cleanStr(v);
|
||||
if (!c) return null;
|
||||
const up = c.toUpperCase().replace(/\./g, "");
|
||||
if (up.length === 2 && STATE_CODES.has(up)) return up;
|
||||
const full = STATE_MAP[c.toLowerCase().replace(/\./g, "")];
|
||||
return full ?? c; // unbekannt: bereinigt durchreichen, nicht raten
|
||||
}
|
||||
|
||||
/** "$350,000" → "350000"; "1.5M" → "1500000"; sonst null + raw. */
|
||||
function normDownPayment(raw: string | null): { value: string | null; raw: string | null } {
|
||||
const c = cleanStr(raw);
|
||||
if (!c) return { value: null, raw: null };
|
||||
const m = c.match(/^\$?\s*(\d[\d,]*(?:\.\d+)?)\s*([kKmM])?\s*$/);
|
||||
if (!m) return { value: null, raw: c };
|
||||
let num = parseFloat(m[1].replace(/,/g, ""));
|
||||
if (m[2]) num *= /k/i.test(m[2]) ? 1_000 : 1_000_000;
|
||||
if (!Number.isFinite(num) || num <= 0) return { value: null, raw: c };
|
||||
return { value: String(Math.round(num)), raw: c };
|
||||
}
|
||||
|
||||
const MONTHS: Record<string, number> = {
|
||||
jan: 1, feb: 2, mar: 3, apr: 4, may: 5, jun: 6,
|
||||
jul: 7, aug: 8, sep: 9, oct: 10, nov: 11, dec: 12,
|
||||
};
|
||||
|
||||
function isoDate(y: number, mo: number, d: number): string | null {
|
||||
if (y < 100) y += 2000;
|
||||
if (y < 1990 || y > 2100 || mo < 1 || mo > 12 || d < 1 || d > 31) return null;
|
||||
return `${y}-${String(mo).padStart(2, "0")}-${String(d).padStart(2, "0")}`;
|
||||
}
|
||||
|
||||
/** US-Formate: 6/25/26, 06-25-2026, "June 25, 2026" → YYYY-MM-DD. */
|
||||
function normDate(raw: string | null): string | null {
|
||||
const c = cleanStr(raw);
|
||||
if (!c) return null;
|
||||
let m = c.match(/^(\d{1,2})[\/\-.](\d{1,2})[\/\-.](\d{2}|\d{4})$/);
|
||||
if (m) return isoDate(parseInt(m[3], 10), parseInt(m[1], 10), parseInt(m[2], 10));
|
||||
m = c.match(/^([A-Za-z]{3,9})\.?\s+(\d{1,2})(?:st|nd|rd|th)?,?\s+(\d{2}|\d{4})$/);
|
||||
if (m) {
|
||||
const mo = MONTHS[m[1].slice(0, 3).toLowerCase()];
|
||||
if (mo) return isoDate(parseInt(m[3], 10), mo, parseInt(m[2], 10));
|
||||
}
|
||||
m = c.match(/^(\d{4})-(\d{2})-(\d{2})$/);
|
||||
if (m) return isoDate(parseInt(m[1], 10), parseInt(m[2], 10), parseInt(m[3], 10));
|
||||
return null;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// PDF-Index (rekursiv, Basename → voller Pfad) + Rendering
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
async function buildPdfIndex(root: string): Promise<Map<string, string>> {
|
||||
const index = new Map<string, string>();
|
||||
const stack = [root];
|
||||
while (stack.length) {
|
||||
const dir = stack.pop()!;
|
||||
let entries: fs.Dirent[];
|
||||
try {
|
||||
entries = await fsp.readdir(dir, { withFileTypes: true });
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
for (const e of entries) {
|
||||
const p = path.join(dir, e.name);
|
||||
if (e.isDirectory()) stack.push(p);
|
||||
else if (e.isFile() && e.name.toLowerCase().endsWith(".pdf")) {
|
||||
if (!index.has(e.name)) index.set(e.name, p);
|
||||
}
|
||||
}
|
||||
}
|
||||
return index;
|
||||
}
|
||||
|
||||
async function renderPdf(pdfPath: string, dpi: number, maxPages: number, tmpDir: string): Promise<string[]> {
|
||||
const prefix = path.join(tmpDir, "page");
|
||||
await execFileP("pdftoppm", ["-png", "-r", String(dpi), "-l", String(maxPages), pdfPath, prefix], {
|
||||
timeout: 120_000,
|
||||
});
|
||||
const pageNum = (f: string) => parseInt(f.match(/-(\d+)\.png$/)?.[1] ?? "0", 10);
|
||||
const files = (await fsp.readdir(tmpDir))
|
||||
.filter((f) => f.endsWith(".png"))
|
||||
.sort((a, b) => pageNum(a) - pageNum(b))
|
||||
.map((f) => path.join(tmpDir, f));
|
||||
if (files.length === 0) throw new Error("pdftoppm hat keine Seiten erzeugt");
|
||||
return files;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// VLM-Aufruf (llama-server, OpenAI-kompatibel, JSON-Schema)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const nullableString = { type: ["string", "null"] };
|
||||
const nullableInt = { type: ["integer", "null"] };
|
||||
|
||||
const RESPONSE_SCHEMA = {
|
||||
type: "object",
|
||||
additionalProperties: false,
|
||||
required: [
|
||||
"is_buyer_sheet", "info_page", "ca_page", "name_company", "prospective_buyer",
|
||||
"company", "phone", "cell", "email", "address", "state", "how_did_you_hear",
|
||||
"interested_in_updates", "types_of_business_raw", "background_experience",
|
||||
"total_purchase_price", "down_payment_raw", "date_of_introduction_raw",
|
||||
],
|
||||
properties: {
|
||||
is_buyer_sheet: { type: "boolean" },
|
||||
info_page: nullableInt,
|
||||
ca_page: nullableInt,
|
||||
name_company: nullableString,
|
||||
prospective_buyer: nullableString,
|
||||
company: nullableString,
|
||||
phone: nullableString,
|
||||
cell: nullableString,
|
||||
email: nullableString,
|
||||
address: nullableString,
|
||||
state: nullableString,
|
||||
how_did_you_hear: nullableString,
|
||||
interested_in_updates: nullableString,
|
||||
types_of_business_raw: nullableString,
|
||||
background_experience: nullableString,
|
||||
total_purchase_price: nullableString,
|
||||
down_payment_raw: nullableString,
|
||||
date_of_introduction_raw: nullableString,
|
||||
},
|
||||
} as const;
|
||||
|
||||
const SYSTEM_PROMPT =
|
||||
"You are a precise document transcription engine for scanned business forms. " +
|
||||
"You transcribe handwritten and typed form fields EXACTLY as written (verbatim), " +
|
||||
"including typos. You never guess, infer, or invent values. If a field is blank " +
|
||||
"or unreadable, you return null.";
|
||||
|
||||
const USER_PROMPT = `You see all pages of one PDF, in order (image 1 = page 1).
|
||||
|
||||
Task: Decide whether this document contains a business brokerage "BUYER INFORMATION SHEET" form, and if so, transcribe its fields.
|
||||
|
||||
1. is_buyer_sheet: true only if a page with the heading "BUYER INFORMATION SHEET" exists. If the document is something else (notes, listing, letter, other form), return is_buyer_sheet=false and null for every field.
|
||||
2. info_page: page number (1-based) of the BUYER INFORMATION SHEET page, else null.
|
||||
3. ca_page: page number of the confidentiality agreement page containing text like "PROSPECTIVE BUYER AGREES TO KEEP AND HOLD CONFIDENTIAL", else null.
|
||||
4. From the info page, transcribe VERBATIM (exactly as written, do not normalize, do not expand abbreviations):
|
||||
- name_company: value of the "Name/Company" line
|
||||
- prospective_buyer: value of the "Prospective Buyer" line
|
||||
- company: value of a separate "Company" line if present
|
||||
- phone, cell, email, address, state
|
||||
- how_did_you_hear: "How did you hear about us"
|
||||
- interested_in_updates: answer/checkbox for receiving updates (transcribe what is marked, e.g. "Yes" or "No"), else null
|
||||
- types_of_business_raw: "Type(s) of business interested in" (may span multiple lines — join with a space)
|
||||
- background_experience: "Background/Experience" (may span multiple lines)
|
||||
- total_purchase_price: "Total Purchase Price" as written
|
||||
- down_payment_raw: "Down Payment" as written (e.g. "$350,000", "1.5M", "TBD")
|
||||
5. date_of_introduction_raw: the date written next to the buyer's signature on the confidentiality agreement page, verbatim (e.g. "6/25/26"), else null.
|
||||
|
||||
A blank field, "N/A", "n", or an empty line = transcribe it as written; if truly empty, use null. Return only the JSON object.`;
|
||||
|
||||
async function fetchModelId(api: string): Promise<string> {
|
||||
try {
|
||||
const r = await fetch(`${api}/models`);
|
||||
const j = (await r.json()) as { data?: Array<{ id?: string }> };
|
||||
return j?.data?.[0]?.id ?? "unknown";
|
||||
} catch {
|
||||
return "unknown";
|
||||
}
|
||||
}
|
||||
|
||||
async function callVision(api: string, images: string[], timeoutMs: number): Promise<VisionRaw> {
|
||||
const content: Array<Record<string, unknown>> = [{ type: "text", text: USER_PROMPT }];
|
||||
for (const img of images) {
|
||||
const b64 = await fsp.readFile(img, { encoding: "base64" });
|
||||
content.push({ type: "image_url", image_url: { url: `data:image/png;base64,${b64}` } });
|
||||
}
|
||||
const body = {
|
||||
model: "qwen3.6",
|
||||
temperature: 0.1,
|
||||
top_p: 0.95,
|
||||
max_tokens: 1500,
|
||||
chat_template_kwargs: { enable_thinking: false },
|
||||
response_format: {
|
||||
type: "json_schema",
|
||||
json_schema: { name: "buyer_sheet", strict: true, schema: RESPONSE_SCHEMA },
|
||||
},
|
||||
messages: [
|
||||
{ role: "system", content: SYSTEM_PROMPT },
|
||||
{ role: "user", content },
|
||||
],
|
||||
};
|
||||
const ctl = new AbortController();
|
||||
const timer = setTimeout(() => ctl.abort(), timeoutMs);
|
||||
try {
|
||||
const r = await fetch(`${api}/chat/completions`, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify(body),
|
||||
signal: ctl.signal,
|
||||
});
|
||||
if (!r.ok) throw new Error(`HTTP ${r.status}: ${(await r.text()).slice(0, 300)}`);
|
||||
const j = (await r.json()) as { choices?: Array<{ message?: { content?: string } }> };
|
||||
const text = j?.choices?.[0]?.message?.content;
|
||||
if (!text) throw new Error("Leere Antwort vom Server");
|
||||
return JSON.parse(text) as VisionRaw;
|
||||
} finally {
|
||||
clearTimeout(timer);
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Merge: deterministischer Datensatz + Vision-Rohwerte → Zielstruktur
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
function mergeRecord(det: BuyerRecord, vis: VisionRaw, model: string): BuyerRecord {
|
||||
const base: BuyerRecord = {
|
||||
...det,
|
||||
_parser: "vision",
|
||||
_vision_model: model,
|
||||
_vision_error: undefined,
|
||||
};
|
||||
delete (base as Record<string, unknown>)["_vision_error"];
|
||||
|
||||
if (!vis.is_buyer_sheet) {
|
||||
return { ...base, is_buyer_sheet: false, _info_page: null, _ca_page: null };
|
||||
}
|
||||
|
||||
const dp = normDownPayment(vis.down_payment_raw);
|
||||
return {
|
||||
...base,
|
||||
is_buyer_sheet: true,
|
||||
name_company: cleanStr(vis.name_company),
|
||||
prospective_buyer: cleanStr(vis.prospective_buyer),
|
||||
company: cleanStr(vis.company),
|
||||
phone: cleanStr(vis.phone),
|
||||
cell: cleanStr(vis.cell),
|
||||
email: cleanStr(vis.email),
|
||||
address: cleanStr(vis.address),
|
||||
state: normState(vis.state),
|
||||
how_did_you_hear: cleanStr(vis.how_did_you_hear),
|
||||
interested_in_updates: cleanStr(vis.interested_in_updates),
|
||||
types_of_business_raw: cleanStr(vis.types_of_business_raw),
|
||||
background_experience: cleanStr(vis.background_experience),
|
||||
total_purchase_price: cleanStr(vis.total_purchase_price),
|
||||
down_payment: dp.value,
|
||||
down_payment_raw: dp.raw,
|
||||
// Vision-Datum bevorzugt; Fallback: deterministischer Wert (z.B. aus Dateinamen)
|
||||
date_of_introduction: normDate(vis.date_of_introduction_raw) ?? (det.date_of_introduction as string | null) ?? null,
|
||||
_info_page: vis.info_page,
|
||||
_ca_page: vis.ca_page,
|
||||
};
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Main
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const RETRIES = 3;
|
||||
|
||||
function progress(line: string): void {
|
||||
const cols = process.stderr.columns ?? 120;
|
||||
process.stderr.write("\r" + line.slice(0, cols - 1).padEnd(cols - 1));
|
||||
}
|
||||
|
||||
async function main(): Promise<void> {
|
||||
const args = parseArgs();
|
||||
|
||||
// pdftoppm vorhanden?
|
||||
try {
|
||||
await execFileP("pdftoppm", ["-v"]);
|
||||
} catch {
|
||||
console.error("Fehler: pdftoppm nicht gefunden. Installieren: sudo apt install poppler-utils");
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const all: BuyerRecord[] = JSON.parse(await fsp.readFile(args.input, "utf8"));
|
||||
let targets = all.filter((r) => r.is_buyer_sheet === false);
|
||||
if (args.only) targets = targets.filter((r) => r.file_name === args.only);
|
||||
if (args.limit > 0) targets = targets.slice(0, args.limit);
|
||||
|
||||
const visionPath = path.join(args.outDir, "buyers_vision.json");
|
||||
const mergedPath = path.join(args.outDir, "buyers_merged.json");
|
||||
const done = new Map<string, BuyerRecord>();
|
||||
try {
|
||||
const prev: BuyerRecord[] = JSON.parse(await fsp.readFile(visionPath, "utf8"));
|
||||
for (const r of prev) done.set(r.file_name, r);
|
||||
} catch {
|
||||
/* kein Resume-Stand */
|
||||
}
|
||||
|
||||
console.error(`Indexiere PDFs unter ${args.pdfRoot} ...`);
|
||||
const pdfIndex = await buildPdfIndex(args.pdfRoot);
|
||||
console.error(`${pdfIndex.size} PDFs gefunden. ${targets.length} Einträge zu verarbeiten.`);
|
||||
|
||||
const model = await fetchModelId(args.api);
|
||||
console.error(`Modell: ${model} @ ${args.api}`);
|
||||
|
||||
let ok = 0, notSheet = 0, errors = 0, skipped = 0;
|
||||
|
||||
for (let i = 0; i < targets.length; i++) {
|
||||
const det = targets[i];
|
||||
const tag = `[${i + 1}/${targets.length}] ${det.file_name}`;
|
||||
|
||||
const prev = done.get(det.file_name);
|
||||
if (prev && !prev["_vision_error"] && !args.force) {
|
||||
skipped++;
|
||||
progress(`${tag} … übersprungen (bereits verarbeitet)`);
|
||||
continue;
|
||||
}
|
||||
|
||||
const pdfPath = pdfIndex.get(det.file_name);
|
||||
if (!pdfPath) {
|
||||
done.set(det.file_name, { ...det, _vision_error: "PDF nicht gefunden" });
|
||||
errors++;
|
||||
progress(`${tag} … FEHLER: PDF nicht gefunden`);
|
||||
continue;
|
||||
}
|
||||
|
||||
const tmpDir = await fsp.mkdtemp(path.join(os.tmpdir(), "bvs-"));
|
||||
try {
|
||||
progress(`${tag} … rendere`);
|
||||
const images = await renderPdf(pdfPath, args.dpi, args.maxPages, tmpDir);
|
||||
|
||||
let vis: VisionRaw | null = null;
|
||||
let lastErr = "";
|
||||
for (let attempt = 1; attempt <= RETRIES; attempt++) {
|
||||
try {
|
||||
progress(`${tag} … VLM (${images.length} Seiten, Versuch ${attempt})`);
|
||||
vis = await callVision(args.api, images, args.timeoutMs);
|
||||
break;
|
||||
} catch (e) {
|
||||
lastErr = e instanceof Error ? e.message : String(e);
|
||||
if (attempt < RETRIES) await new Promise((res) => setTimeout(res, 5000 * attempt));
|
||||
}
|
||||
}
|
||||
|
||||
if (!vis) {
|
||||
done.set(det.file_name, { ...det, _vision_error: lastErr });
|
||||
errors++;
|
||||
progress(`${tag} … FEHLER: ${lastErr}`);
|
||||
} else {
|
||||
const merged = mergeRecord(det, vis, model);
|
||||
done.set(det.file_name, merged);
|
||||
if (merged.is_buyer_sheet) { ok++; progress(`${tag} … OK`); }
|
||||
else { notSheet++; progress(`${tag} … kein Buyer Sheet`); }
|
||||
}
|
||||
} catch (e) {
|
||||
const msg = e instanceof Error ? e.message : String(e);
|
||||
done.set(det.file_name, { ...det, _vision_error: msg });
|
||||
errors++;
|
||||
progress(`${tag} … FEHLER: ${msg}`);
|
||||
} finally {
|
||||
await fsp.rm(tmpDir, { recursive: true, force: true });
|
||||
}
|
||||
|
||||
// Inkrementell sichern (Resume-fähig)
|
||||
await fsp.mkdir(args.outDir, { recursive: true });
|
||||
await fsp.writeFile(visionPath, JSON.stringify([...done.values()], null, 2));
|
||||
}
|
||||
|
||||
// Merge: Gleis A + Gleis B
|
||||
const merged = all.map((r) => done.get(r.file_name) ?? r);
|
||||
await fsp.writeFile(mergedPath, JSON.stringify(merged, null, 2));
|
||||
|
||||
process.stderr.write("\n");
|
||||
console.error(
|
||||
`Fertig. OK: ${ok}, kein Buyer Sheet: ${notSheet}, Fehler: ${errors}, übersprungen: ${skipped}`
|
||||
);
|
||||
console.error(`→ ${visionPath}\n→ ${mergedPath}`);
|
||||
}
|
||||
|
||||
main().catch((e) => {
|
||||
console.error("\nAbbruch:", e instanceof Error ? e.message : e);
|
||||
process.exit(1);
|
||||
});
|
||||
Reference in New Issue
Block a user