#!/usr/bin/env python3
"""Offline adapters for semantic HTML extraction protocol v1.2."""

from __future__ import annotations

import importlib.metadata
import json
import pathlib
import re
import sys

import html2text
import justext
import markdownify
from newspaper import Article
from readability import Document


def version(name: str) -> str:
    try:
        return importlib.metadata.version(name)
    except importlib.metadata.PackageNotFoundError:
        return "unknown"


def main() -> int:
    if len(sys.argv) != 2:
        raise SystemExit("usage: extract_multitool.py INPUT.html")

    source = pathlib.Path(sys.argv[1])
    html = source.read_text(encoding="utf-8")
    language_match = re.search(r'<html\b[^>]*\blang=["\']([^"\']+)', html, flags=re.IGNORECASE)
    language = (language_match.group(1).split("-")[0].lower() if language_match else "en")
    stoplist = "French" if language == "fr" else "English"

    readability_document = Document(html)
    readability_html = readability_document.summary(html_partial=True)

    newspaper_article = Article(url=f"https://fixtures.invalid/{source.name}", language=language)
    newspaper_article.html = html
    newspaper_article.parse()

    justext_paragraphs = justext.justext(html, justext.get_stoplist(stoplist))
    justext_text = "\n".join(paragraph.text for paragraph in justext_paragraphs if not paragraph.is_boilerplate)

    html2text_converter = html2text.HTML2Text()
    html2text_converter.body_width = 0
    html2text_converter.ignore_links = False

    payload = {
        "source": source.name,
        "readability_lxml": {
            "tool": "readability-lxml",
            "version": version("readability-lxml"),
            "title": readability_document.short_title(),
            "html": readability_html,
            "status": "ok" if readability_html else "no_output",
        },
        "newspaper4k": {
            "tool": "newspaper4k",
            "version": version("newspaper4k"),
            "title": newspaper_article.title or None,
            "text": newspaper_article.text or None,
            "html": newspaper_article.article_html or None,
            "status": "ok" if newspaper_article.text or newspaper_article.article_html else "no_output",
        },
        "justext": {
            "tool": "jusText",
            "version": version("jusText"),
            "text": justext_text or None,
            "paragraphs": [
                {"text": paragraph.text, "is_boilerplate": paragraph.is_boilerplate}
                for paragraph in justext_paragraphs
            ],
            "status": "ok" if justext_text else "no_output",
        },
        "html2text": {
            "tool": "html2text",
            "version": version("html2text"),
            "markdown": html2text_converter.handle(html),
            "status": "ok",
        },
        "markdownify": {
            "tool": "markdownify",
            "version": version("markdownify"),
            "markdown": markdownify.markdownify(html, heading_style="ATX"),
            "status": "ok",
        },
    }
    print(json.dumps(payload, ensure_ascii=False, indent=2))
    return 0


if __name__ == "__main__":
    raise SystemExit(main())
