import { Readability } from "@mozilla/readability";
import { JSDOM } from "jsdom";
import type { ToolHandler, ToolResult } from "./types";
import { isPrivateHost } from "./security";

const MAX_CHARS = 8000;
const USER_AGENT = "Mozilla/5.0 (compatible; OllamaAgent/1.0)";

function fallbackExtract(html: string): string {
  return html
    .replace(/<script[\s\S]*?<\/script>/gi, " ")
    .replace(/<style[\s\S]*?<\/style>/gi, " ")
    .replace(/<noscript[\s\S]*?<\/noscript>/gi, " ")
    .replace(/<(nav|footer|header|aside|form)[\s\S]*?<\/\1>/gi, " ")
    .replace(/<br\s*\/?>/gi, "\n")
    .replace(/<\/(p|div|li|h[1-6])>/gi, "\n")
    .replace(/<[^>]+>/g, " ")
    .replace(/&nbsp;/g, " ")
    .replace(/&amp;/g, "&")
    .replace(/&lt;/g, "<")
    .replace(/&gt;/g, ">")
    .replace(/&quot;/g, '"')
    .replace(/[ \t]+/g, " ")
    .replace(/\n{3,}/g, "\n\n")
    .trim();
}

function extractText(html: string, url: string): { title: string | null; text: string } {
  try {
    const dom = new JSDOM(html, { url });
    const reader = new Readability(dom.window.document);
    const article = reader.parse();
    if (article?.textContent && article.textContent.trim().length > 100) {
      return {
        title: article.title ?? null,
        text: article.textContent.trim().replace(/\n{3,}/g, "\n\n"),
      };
    }
  } catch {
    /* fall through */
  }
  return { title: null, text: fallbackExtract(html) };
}

export const readPage: ToolHandler = async (args): Promise<ToolResult> => {
  const raw = typeof args.url === "string" ? args.url.trim() : "";
  if (!raw) return { content: "URL vide.", error: "empty_url" };

  let parsed: URL;
  try {
    parsed = new URL(raw);
  } catch {
    return { content: `URL invalide : ${raw}`, error: "invalid_url" };
  }
  if (parsed.protocol !== "http:" && parsed.protocol !== "https:") {
    return { content: "Seuls les protocoles http/https sont supportés.", error: "invalid_protocol" };
  }
  if (await isPrivateHost(parsed.hostname)) {
    return {
      content: `Accès refusé à ${parsed.hostname} (adresse privée / loopback).`,
      error: "private_address",
    };
  }

  try {
    const res = await fetch(parsed.toString(), {
      headers: {
        "User-Agent": USER_AGENT,
        Accept: "text/html,application/xhtml+xml,*/*;q=0.8",
      },
      redirect: "follow",
      signal: AbortSignal.timeout(10_000),
    });
    if (!res.ok) {
      return {
        content: `${parsed.hostname} a renvoyé ${res.status} ${res.statusText}.`,
        error: `http_${res.status}`,
      };
    }
    const ct = res.headers.get("content-type") ?? "";
    if (!ct.includes("html") && !ct.includes("xml") && !ct.includes("text")) {
      return {
        content: `Type de contenu non supporté : ${ct || "inconnu"}.`,
        error: "unsupported_content_type",
      };
    }
    const html = await res.text();
    const { title, text } = extractText(html, parsed.toString());
    const truncated = text.length > MAX_CHARS;
    const body = truncated ? text.slice(0, MAX_CHARS) + "\n\n... (tronqué)" : text;

    const header = title ? `# ${title}\nSource : ${parsed.toString()}\n\n` : `Source : ${parsed.toString()}\n\n`;
    return {
      content: header + body,
      meta: { url: parsed.toString(), title, truncated, length: text.length },
    };
  } catch (e) {
    return {
      content: `Erreur lors de la lecture : ${e instanceof Error ? e.message : "inconnue"}`,
      error: "fetch_error",
    };
  }
};
