"""aeowatch: the AEO Watch checker as a command line tool.

One file, standard library only, no dependencies and no account. It reads the
public API of https://bikoosh.com/aeo, which answers from stored facts, so
running it never makes anyone's site get fetched.

    python -m aeo.cli check example.com
    python -m aeo.cli check example.com --json
    python -m aeo.cli crawlers
    python -m aeo.cli dataset --out top-1000.csv
    python -m aeo.cli badge example.com

Keys. The free tier wants a key of your own choosing (any string of 16
characters or more) and the URL of the page that will carry the credit line:

    export AEOWATCH_KEY=my-own-key-1234567890
    export AEOWATCH_ATTRIBUTION=https://mysite.example/colophon

Members send their Gumroad licence key in AEOWATCH_LICENCE, or the address
they paid with in AEOWATCH_EMAIL, and need no attribution.

Attribution, always: data from AEO Watch by Bikoosh, https://bikoosh.com/aeo,
CC BY 4.0. The ranking behind the dataset is Tranco's, cited in the file.

This module is published as a standalone package from packages/aeo-cli (see
that directory's PUBLISHING.md); nothing here imports the rest of this repo,
which is what makes that possible.
"""
from __future__ import annotations

import argparse
import json
import os
import sys
import urllib.error
import urllib.parse
import urllib.request

DEFAULT_BASE = "https://bikoosh.com"
USER_AGENT = "aeowatch-cli/1.0 (+https://bikoosh.com/aeo)"
CREDIT = ("Data: AEO Watch by Bikoosh, https://bikoosh.com/aeo, CC BY 4.0. "
          "Ranking source for the dataset: Tranco, https://tranco-list.eu/.")
TIMEOUT_S = 30


def http_get(url: str, headers: dict | None = None) -> tuple[int, str]:
    """(status, body). The one network call in this file, injectable so the
    tests can drive every command without touching the network."""
    req = urllib.request.Request(url, headers={"User-Agent": USER_AGENT,
                                               "Accept": "*/*",
                                               **(headers or {})})
    try:
        with urllib.request.urlopen(req, timeout=TIMEOUT_S) as resp:  # noqa: S310
            return resp.status, resp.read().decode("utf-8", "replace")
    except urllib.error.HTTPError as e:
        return e.code, e.read().decode("utf-8", "replace")
    except (urllib.error.URLError, TimeoutError) as e:
        return 0, json.dumps({"error": f"{type(e).__name__}: {e}"})


def auth_headers(args) -> dict:
    out = {}
    licence = args.licence or os.environ.get("AEOWATCH_LICENCE", "")
    email = args.email or os.environ.get("AEOWATCH_EMAIL", "")
    key = args.key or os.environ.get("AEOWATCH_KEY", "")
    attribution = args.attribution or os.environ.get("AEOWATCH_ATTRIBUTION", "")
    if licence:
        out["X-License-Key"] = licence
    elif email:
        out["X-Receipt-Email"] = email
    else:
        if key:
            out["X-Api-Key"] = key
        if attribution:
            out["X-Attribution"] = attribution
    return out


def _lines_for_check(data: dict) -> list[str]:
    if data.get("error"):
        return [f"{data['error']}: {data.get('note', '')}".strip(": ")]
    if data.get("opted_out"):
        return [f"{data['domain']}: {data.get('note', '')}"]
    counts = data.get("counts") or {}
    out = [data.get("summary", ""), ""]
    for row in data.get("crawlers") or []:
        rule = row.get("rule") or ""
        when = (row.get("observed_at") or "")[:10]
        out.append(f"  {row.get('crawler_token', ''):<22} {rule:<14} {when}")
    out += ["",
            f"tokens resolved {counts.get('tokens_resolved', 0)}, "
            f"allowed for / {counts.get('allowed_for_root', 0)}, "
            f"disallowed for / {counts.get('disallowed_for_root', 0)}",
            f"report: {data.get('report_url', '')}",
            CREDIT]
    return out


def cmd_check(args, fetch) -> int:
    query = urllib.parse.urlencode({"domain": args.domain})
    status, body = fetch(f"{args.base}/api/v1/aeo/check?{query}",
                         auth_headers(args))
    try:
        data = json.loads(body)
    except ValueError:
        print(f"unreadable answer (HTTP {status})", file=sys.stderr)
        return 2
    if args.json:
        print(json.dumps(data, indent=1, ensure_ascii=False))
    else:
        print("\n".join(_lines_for_check(data)))
    return 0 if status == 200 else 1


def cmd_crawlers(args, fetch) -> int:
    status, body = fetch(f"{args.base}/api/v1/aeo/crawlers", {})
    try:
        data = json.loads(body)
    except ValueError:
        return 2
    if args.json:
        print(json.dumps(data, indent=1, ensure_ascii=False))
        return 0 if status == 200 else 1
    for e in data.get("crawlers") or []:
        print(f"{e.get('token', ''):<24} {e.get('vendor', ''):<18} "
              f"{e.get('documentation_url', '')}")
    print(CREDIT)
    return 0 if status == 200 else 1


def cmd_dataset(args, fetch) -> int:
    status, body = fetch(f"{args.base}/aeo/top-1000.csv", {})
    if status != 200:
        print(f"HTTP {status}", file=sys.stderr)
        return 1
    if args.out:
        with open(args.out, "w", encoding="utf-8") as fh:
            fh.write(body)
        print(f"wrote {args.out} ({len(body.encode())} bytes). {CREDIT}")
    else:
        sys.stdout.write(body)
    return 0


def cmd_badge(args, _fetch) -> int:
    d = args.domain.strip().lower()
    print(f'<a href="{args.base}/aeo/site/{d}" rel="noopener">\n'
          f'  <img src="{args.base}/aeo/badge/{d}.svg"\n'
          f'       alt="AEO Watch: what {d} robots.txt says to AI crawlers, '
          f'with the date"\n'
          '       width="420" height="28" loading="lazy">\n'
          "</a>")
    return 0


def build_parser() -> argparse.ArgumentParser:
    ap = argparse.ArgumentParser(
        prog="aeowatch",
        description="Can AI answer engines read and cite a site? Reads the "
                    "public AEO Watch API, which answers from stored facts "
                    "and never fetches the site you ask about.")
    ap.add_argument("--base", default=os.environ.get("AEOWATCH_BASE",
                                                     DEFAULT_BASE))
    ap.add_argument("--key", default="", help="free tier key, 16+ characters")
    ap.add_argument("--attribution", default="",
                    help="URL of the page that will carry the credit line")
    ap.add_argument("--licence", default="", help="Gumroad licence key")
    ap.add_argument("--email", default="", help="the address you paid with")
    ap.add_argument("--json", action="store_true", help="raw JSON output")
    sub = ap.add_subparsers(dest="command", required=True)
    c = sub.add_parser("check", help="what one domain's robots.txt says")
    c.add_argument("domain")
    sub.add_parser("crawlers", help="the AI crawler registry")
    d = sub.add_parser("dataset", help="download the free CSV")
    d.add_argument("--out", default="")
    b = sub.add_parser("badge", help="print the embed snippet for a domain")
    b.add_argument("domain")
    return ap


COMMANDS = {"check": cmd_check, "crawlers": cmd_crawlers,
            "dataset": cmd_dataset, "badge": cmd_badge}


def main(argv=None, fetch=None) -> int:
    args = build_parser().parse_args(argv if argv is not None else sys.argv[1:])
    args.base = args.base.rstrip("/")
    return COMMANDS[args.command](args, fetch or http_get)


if __name__ == "__main__":                # pragma: no cover
    raise SystemExit(main())
