﻿import { execFile } from "node:child_process";
import { readFile } from "node:fs/promises";
import { promisify } from "node:util";

const execFileAsync = promisify(execFile);

export interface ReceiptAnalysisResult {
  purchaseDate?: string;
  vendor?: string;
  purchasePrice?: string;
  name?: string;
  brand?: string;
  model?: string;
  serialNumber?: string;
  rawText?: string;
  method: "pdf-text" | "pdf-ocr";
}

function normalizeSpaces(value: string): string {
  return value.replace(/\s+/g, " ").trim();
}

function extractDateFromText(text: string): string | undefined {
  const normalized = text.replace(/\r/g, "\n");

  const numericMatch = normalized.match(/\b(\d{1,2})[\/\-\.](\d{1,2})[\/\-\.](\d{4})\b/);
  if (numericMatch) {
    const [, day, month, year] = numericMatch;
    const parsed = new Date(`${year}-${month!.padStart(2, "0")}-${day!.padStart(2, "0")}`);
    if (!Number.isNaN(parsed.getTime())) return `${year}-${month!.padStart(2, "0")}-${day!.padStart(2, "0")}`;
  }

  const isoMatch = normalized.match(/\b(\d{4})[\/\-\.](\d{1,2})[\/\-\.](\d{1,2})\b/);
  if (isoMatch) {
    const [, year, month, day] = isoMatch;
    return `${year}-${month!.padStart(2, "0")}-${day!.padStart(2, "0")}`;
  }

  const months: Record<string, string> = {
    enero: "01",
    febrero: "02",
    marzo: "03",
    abril: "04",
    mayo: "05",
    junio: "06",
    julio: "07",
    agosto: "08",
    septiembre: "09",
    setiembre: "09",
    octubre: "10",
    noviembre: "11",
    diciembre: "12",
  };

  const namedMatch = normalized.match(/\b(\d{1,2})\s+(?:de\s+)?(enero|febrero|marzo|abril|mayo|junio|julio|agosto|septiembre|setiembre|octubre|noviembre|diciembre)\s+(?:de\s+)?(\d{4})\b/i);
  if (namedMatch) {
    const [, day, monthName, year] = namedMatch;
    const month = months[monthName!.toLowerCase()];
    if (month) return `${year}-${month}-${day!.padStart(2, "0")}`;
  }

  return undefined;
}

function extractAmountFromText(text: string): string | undefined {
  const normalized = text.replace(/\r/g, "\n");
  const candidates: Array<{ raw: string; priority: number }> = [];
  const pushMatches = (pattern: RegExp, priority: number) => {
    for (const match of normalized.matchAll(pattern)) {
      const raw = match[1];
      if (raw) candidates.push({ raw, priority });
    }
  };

  pushMatches(/(?:total pendiente|importe total|total pagado|total a pagar|total|importe|a pagar|subtotal|suma)\s*:?\s*[€$]?\s*([\d][\d\s.,]*)/gi, 3);
  pushMatches(/[€$]\s*([\d][\d\s.,]*)/g, 2);
  pushMatches(/([\d][\d\s.,]*)\s*[€$]/g, 2);

  let best: { value: number; priority: number; text: string } | null = null;
  for (const candidate of candidates) {
    const parsed = parseAmountCandidate(candidate.raw);
    if (!parsed) continue;
    if (!best || candidate.priority > best.priority || (candidate.priority === best.priority && parsed.value > best.value)) {
      best = { ...parsed, priority: candidate.priority };
    }
  }

  return best ? best.text : undefined;
}

function parseAmountCandidate(raw: string): { value: number; text: string } | null {
  const cleaned = raw.replace(/\s+/g, "").replace(/[^\d.,]/g, "");
  if (!cleaned) return null;

  const hasComma = cleaned.includes(",");
  const hasDot = cleaned.includes(".");
  const digitsOnly = cleaned.replace(/[.,]/g, "");
  if (!digitsOnly) return null;

  const toResult = (integerPart: string, decimals: string) => {
    const intValue = Number.parseInt(integerPart.replace(/^0+(?=\d)/, "") || "0", 10);
    if (!Number.isFinite(intValue)) return null;
    const cents = decimals.padEnd(2, "0").slice(0, 2);
    const value = Number(`${intValue}.${cents}`);
    return Number.isFinite(value) ? { value, text: `${intValue}.${cents}` } : null;
  };

  if (hasComma && hasDot) {
    const decimalSep = cleaned.lastIndexOf(",") > cleaned.lastIndexOf(".") ? "," : ".";
    const parts = cleaned.split(decimalSep);
    const decimals = parts.pop() ?? "";
    const integerPart = parts.join("").replace(/[.,]/g, "");
    return toResult(integerPart, decimals);
  }

  if (hasComma || hasDot) {
    const sep = hasComma ? "," : ".";
    const parts = cleaned.split(sep);
    const decimals = parts[parts.length - 1] ?? "";
    const integerPart = parts.slice(0, -1).join("").replace(/[.,]/g, "");

    if (decimals.length === 2) return toResult(integerPart || parts[0] || "0", decimals);
    if (decimals.length === 3) return toResult((integerPart || parts[0] || "0") + decimals, "00");
    if (decimals.length === 1) return toResult(integerPart || parts[0] || "0", `${decimals}0`);
    return toResult(cleaned.replace(/[.,]/g, ""), "00");
  }

  return toResult(digitsOnly, "00");
}
function extractVendorFromText(text: string): string | undefined {
  const explicit = text.match(/(?:vendido por|enviado por)\s+([^\n]+)/i);
  if (explicit?.[1]) {
    return normalizeSpaces(explicit[1].replace(/\s+(?:iva|nif|cif|total)\b.*$/i, ""));
  }

  const amazon = text.match(/\bAmazon EU S\.à r\.l\., Sucursal en España\b/i);
  if (amazon?.[0]) {
    return "Amazon EU S.à r.l., Sucursal en España";
  }

  const lines = text
    .replace(/\r/g, "\n")
    .split("\n")
    .map((line) => normalizeSpaces(line))
    .filter((line) => line.length > 2 && line.length < 120);

  const skip = /^(?:\d{1,2}[\/\-\.]\d{1,2}[\/\-\.]\d{2,4}|[\d\s€$\.,\+\-]+|(?:cif|nif|iva|total|tpv|factura|pedido|p[aá]gina|pagado))/i;
  for (const line of lines.slice(0, 10)) {
    if (!skip.test(line)) return line;
  }

  return undefined;
}
function extractProductDetailsFromText(text: string): Pick<ReceiptAnalysisResult, "name" | "brand" | "model" | "serialNumber"> {
  const normalizedText = text.replace(/\r/g, "\n");
  const lines = normalizedText
    .split("\n")
    .map((line) => normalizeSpaces(line))
    .filter((line) => line.length > 0);

  const detailsIndex = lines.findIndex((line) => /detalles de la factura/i.test(line));
  const detailsSection = detailsIndex >= 0 ? lines.slice(detailsIndex + 1) : lines;

  const modelMatch = normalizedText.match(/\b([A-Z]{1,5}\d{2,5}(?:\.[A-Z0-9]{1,8})?)\b/);
  const asinMatch = normalizedText.match(/\bASIN\s*[:#]?\s*([A-Z0-9]{8,})\b/i);
  const serialNumber = asinMatch?.[1] ?? undefined;
  const model = modelMatch?.[1] ?? undefined;

  const productKeywords = /(?:cafe|cafetera|televisor|portatil|aspiradora|lavadora|movil|telefono|frigorifico|tablet|ordenador|monitor|impresora|horno|microondas|batidora|licuadora)/i;
  const productStartIndex = detailsSection.findIndex((line) => productKeywords.test(line) || /^de'longhi/i.test(line) || /\bASIN\b/i.test(line));
  const productSourceLines = productStartIndex >= 0 ? detailsSection.slice(productStartIndex) : detailsSection;
  const productLines = productSourceLines.filter(
    (line) =>
      !/^(?:descripci[oó]n|cant\.|p\.\s*unitario|iva\s*%|precio total|precio\s+total|\(iva excluido\)|\(iva incluido\)|env[ií]o|total\b|iva\b)/i.test(line),
  );
  const stopWords = /^(?:asin|envio|envío|total|pagado|fecha|numero|número|vendido por|enviado por|iva|precio total|cant\.|descripci[oó]n)/i;
  const numericOnly = /^[\d\s.,\-+?]+$/;

  const directProductLine = productLines.find(
    (line) => productKeywords.test(line) && !/^(?:descripci[oó]n|cant\.|p\.\s*unitario|iva\b|precio\b)/i.test(line),
  );

  let name = directProductLine
    ? normalizeSpaces(directProductLine).replace(/\s+\d+\s+[\d.,]+\s*€.*$/u, "").trim()
    : "";
  if (!name) {
    const nameLines: string[] = [];
    for (const line of productLines) {
      const cleanedLine = normalizeSpaces(line).replace(/\s+\d+\s+[\d.,]+\s*€.*$/u, "").trim();
      if (stopWords.test(cleanedLine) || numericOnly.test(cleanedLine) || /[%€]/.test(cleanedLine)) break;
      if (nameLines.length === 0 && !productKeywords.test(cleanedLine) && cleanedLine.length < 18) continue;
      nameLines.push(cleanedLine);
      if (nameLines.join(" ").length >= 240 || nameLines.length >= 4) break;
    }
    name = normalizeSpaces(nameLines.join(" "));
  }

  if (!name) {
    const candidate = productLines.find((line) => productKeywords.test(line) && line.length > 12);
    name = candidate ? normalizeSpaces(candidate) : "";
  }
  if (!name) {
    const fallback = productLines.find((line) => line.length > 20 && /[\u00C0-\u017F]/.test(line));
    name = fallback ? normalizeSpaces(fallback) : "";
  }

  let brand: string | undefined;
  if (name) {
    const firstSegment = name.split(/[-–—|]/)[0].trim();
    const firstToken = firstSegment.split(" ").filter(Boolean)[0];
    if (firstToken) brand = firstToken.replace(/[,:;]+$/g, "");
    if (!brand && /^De'Longhi/i.test(name)) brand = "De'Longhi";
  }

  return {
    name: name || undefined,
    brand: brand || undefined,
    model,
    serialNumber,
  };
}
function countExtractedFields(result: Pick<ReceiptAnalysisResult, "purchaseDate" | "vendor" | "purchasePrice" | "name" | "brand" | "model" | "serialNumber">): number {
  return [
    result.purchaseDate,
    result.vendor,
    result.purchasePrice,
    result.name,
    result.brand,
    result.model,
    result.serialNumber,
  ].filter(Boolean).length;
}

async function extractPdfText(buffer: Uint8Array): Promise<string> {
  const pdfjsLib = await import("pdfjs-dist/legacy/build/pdf.mjs");
  const document = await pdfjsLib.getDocument({ data: buffer, disableWorker: true } as any).promise;
  const maxPages = Math.min(document.numPages, 2);
  const chunks: string[] = [];

  for (let pageNumber = 1; pageNumber <= maxPages; pageNumber += 1) {
    const page = await document.getPage(pageNumber);
    const content = await page.getTextContent();
    const pageText = content.items
      .map((item) => (typeof item === "object" && item !== null && "str" in item ? String((item as { str?: string }).str ?? "") : ""))
      .join("\n")
      .trim();
    if (pageText) chunks.push(pageText);
  }

  await (document as any).destroy?.();
  return chunks.join("\n");
}
async function extractPdfTextWithPoppler(filePath: string): Promise<string> {
  try {
    const { stdout } = await execFileAsync("pdftotext", ["-layout", "-nopgbrk", "-q", filePath, "-"], {
      timeout: 30000,
      maxBuffer: 1024 * 1024,
    });
    return String(stdout ?? "");
  } catch {
    return "";
  }
}

function mergeAnalysis(primary: Partial<ReceiptAnalysisResult>, secondary: Partial<ReceiptAnalysisResult>): ReceiptAnalysisResult {
  return {
    method: primary.method ?? secondary.method ?? "pdf-text",
    rawText: primary.rawText || secondary.rawText,
    purchaseDate: primary.purchaseDate || secondary.purchaseDate,
    vendor: primary.vendor || secondary.vendor,
    purchasePrice: primary.purchasePrice || secondary.purchasePrice,
    name: primary.name || secondary.name,
    brand: primary.brand || secondary.brand,
    model: primary.model || secondary.model,
    serialNumber: primary.serialNumber || secondary.serialNumber,
  };
}

function analyzeText(text: string, method: ReceiptAnalysisResult["method"]): ReceiptAnalysisResult {
  const base: ReceiptAnalysisResult = {
    method,
    rawText: text,
    purchaseDate: extractDateFromText(text),
    purchasePrice: extractAmountFromText(text),
    vendor: extractVendorFromText(text),
    ...extractProductDetailsFromText(text),
  };
  return base;
}

export async function analyzeReceiptPdfFile(filePath: string): Promise<ReceiptAnalysisResult> {
  const text = (await extractPdfTextWithPoppler(filePath)).trim() || await extractPdfText(new Uint8Array(await readFile(filePath)));
  return analyzeText(text, "pdf-text");
}





