import { join } from "node:path"; import { dataPath } from "../../deployPaths"; import { jsonError, normalizeStem, securityHeaders, withAuth } from "../../security"; const UPLOAD_PAGE_PATH = join(import.meta.dir, "upload.html"); import { mkdir } from "node:fs/promises"; import { flattenTemplate, extractFields, buildOutput, getNested } from "./extractor"; const TEMPLATE_PATH = process.env.BMP_TEMPLATE_PATH ?? dataPath(join(import.meta.dir, "../templatebuilder/template.json"), "templatebuilder", "template.json"); const OUTPUTS_DIR = process.env.BMP_EXTRACTOR_OUTPUTS_DIR ?? dataPath(join(import.meta.dir, "outputs"), "extractor", "outputs"); const MAX_UPLOAD_BYTES = Number(process.env.BMP_MAX_UPLOAD_BYTES ?? 15 * 1024 * 1024); const MAX_PDF_PAGES = Number(process.env.BMP_MAX_PDF_PAGES ?? 35); await mkdir(OUTPUTS_DIR, { recursive: true }); async function loadTemplate(): Promise> { const file = Bun.file(TEMPLATE_PATH); if (!(await file.exists())) { throw new Error( `template.json not found at ${TEMPLATE_PATH}. Run the templatebuilder first.`, ); } return file.json(); } function normalizeFilenamePart(value: string): string { return normalizeStem(value); } function toSafeFilename(...parts: unknown[]): string { const normalizedParts: string[] = []; for (const part of parts) { if (typeof part !== "string") continue; const trimmed = part.trim(); if (!trimmed) continue; const normalized = normalizeFilenamePart(trimmed); const previous = normalizedParts[normalizedParts.length - 1]; if (normalized && normalized !== previous) { normalizedParts.push(normalized); } } return normalizedParts.join("_"); } function extractScalarText(value: unknown): string | undefined { if (typeof value === "string") return value.trim() || undefined; if (typeof value === "object" && value !== null) { const antwort = (value as Record).antwort; if (typeof antwort === "string") return antwort.trim() || undefined; } return undefined; } export const routes = { "/extractor": { GET: withAuth(async () => new Response(Bun.file(UPLOAD_PAGE_PATH), { headers: { ...securityHeaders, "Content-Type": "text/html" }, }), ), }, "/extract": { POST: withAuth(async (req: Request) => { const contentLength = Number(req.headers.get("content-length") ?? 0); if (contentLength > MAX_UPLOAD_BYTES + 4096) { return jsonError(`PDF is too large. Maximum size is ${Math.floor(MAX_UPLOAD_BYTES / 1024 / 1024)} MB.`, 413); } let form; try { form = await req.formData(); } catch { return Response.json({ error: "Expected multipart/form-data" }, { status: 400 }); } const entry = form.get("pdf"); if (!(entry instanceof File)) { return Response.json({ error: 'Missing "pdf" file field' }, { status: 400 }); } if (!entry.name.toLowerCase().endsWith(".pdf") || entry.size > MAX_UPLOAD_BYTES) { return jsonError(`Only PDF uploads up to ${Math.floor(MAX_UPLOAD_BYTES / 1024 / 1024)} MB are allowed.`, 400); } let template: Record; try { template = await loadTemplate(); } catch (err) { return Response.json({ error: String(err) }, { status: 500 }); } const index = flattenTemplate(template); const data = await entry.arrayBuffer(); if (!isPdfMagic(data)) { return jsonError("Uploaded file is not a valid PDF.", 400); } // pdfjs transfers the ArrayBuffer to its Worker, detaching it (byteLength → 0). // Slice a copy first so we still have the original bytes to write to disk. const pdfBytes = data.slice(0); let values; try { values = await extractFields(data, { maxPages: MAX_PDF_PAGES }); } catch (error) { return jsonError(`PDF could not be processed: ${String(error).replace(/^Error:\s*/, "")}`, 400); } const output = buildOutput(values, index); const name = extractScalarText(getNested(output, "kontakt.unternehmen.name")); const rechtsform = extractScalarText(getNested(output, "kontakt.unternehmen.rechtsform")); const stem = name && rechtsform && normalizeFilenamePart(name).endsWith(normalizeFilenamePart(rechtsform)) ? toSafeFilename(name) : toSafeFilename(name, rechtsform) || "output"; const outDir = join(OUTPUTS_DIR, stem); await mkdir(outDir, { recursive: true }); const jsonPath = join(outDir, `${stem}.json`); const pdfPath = join(outDir, `${stem}.pdf`); await Promise.all([ Bun.write(jsonPath, JSON.stringify(output, null, 2)), Bun.write(pdfPath, pdfBytes), ]); return Response.json({ ok: true, stem, files: { json: `${stem}.json`, pdf: `${stem}.pdf` } }); }, { csrf: true, limit: { key: "extract", max: 12, windowMs: 60_000 }, }), }, "/template": { GET: withAuth(async () => { try { return Response.json(await loadTemplate()); } catch (err) { return Response.json({ error: String(err) }, { status: 500 }); } }), }, } as const; function isPdfMagic(data: ArrayBuffer): boolean { const header = new TextDecoder().decode(new Uint8Array(data.slice(0, 5))); return header === "%PDF-"; }