#!/usr/bin/env python3
"""ai-visibility-check — is your site visible to AI answers and usable by AI agents?

Free, standalone, no dependencies (Python 3.9+ stdlib only). Nothing leaves your
machine except the requests to YOUR OWN site.

    python3 ai_visibility_check.py yourdomain.com
    python3 ai_visibility_check.py yourdomain.com --json

What it checks (the ~90% that is automatable):
  1. robots.txt        — which AI crawlers you block (14 known agents)
  2. llms.txt          — the optional agent guide surface
  3. Shopify stores only (auto-detected):
     .well-known/ucp   — agent-commerce profile
     products.json     — can agents read your catalog (prices > 0)?
     PDP JSON-LD       — Product schema, offers, aggregateRating, GTIN, FAQPage
                         on every product page

What it can NOT check (the honest 10%): whether AI answers actually CITE you for
your category queries. That layer lives in editorial placements, Reddit, reviews,
PR — it needs a human running real queries monthly and judging the results.
This script tells you if the machine-readable layer is in order; it cannot tell
you if you are worth citing.

License: MIT. From the fx-labs project — tools that get you to ~90% of what a
professional would set up. The last 10% (judgment, accountability) is the part
you hire for — or skip, knowingly.
"""
import argparse
import json
import re
import sys
import urllib.error
import urllib.request

VERSION = "0.1 (verified 2026-07-03)"
UA = "Mozilla/5.0 (compatible; ai-visibility-check/0.1; +standalone)"
TIMEOUT = 25
PDP_CAP = 30  # politeness cap on product pages fetched

AI_BOTS = [
    "GPTBot", "OAI-SearchBot", "ChatGPT-User", "ClaudeBot", "Claude-Web", "anthropic-ai",
    "PerplexityBot", "Perplexity-User", "Google-Extended", "CCBot", "Applebot-Extended",
    "Amazonbot", "meta-externalagent", "Bytespider",
]

LDJSON_RE = re.compile(
    r'<script[^>]*type=["\']application/ld\+json["\'][^>]*>(.*?)</script>', re.S | re.I)


def fetch(url):
    req = urllib.request.Request(url, headers={"User-Agent": UA})
    with urllib.request.urlopen(req, timeout=TIMEOUT) as r:
        return r.status, r.read().decode("utf-8", errors="replace")


def ld_objects(html):
    objs = []
    for m in LDJSON_RE.finditer(html):
        try:
            data = json.loads(m.group(1).strip())
        except json.JSONDecodeError:
            continue
        stack = [data]
        while stack:
            o = stack.pop()
            if isinstance(o, list):
                stack.extend(o)
            elif isinstance(o, dict):
                if "@graph" in o:
                    g = o["@graph"]
                    stack.extend(g if isinstance(g, list) else [g])
                objs.append(o)
    return objs


def has_type(obj, t):
    ty = obj.get("@type")
    return ty == t or (isinstance(ty, list) and t in ty)


def blocked_ai_bots(robots_txt):
    """AI agents blocked with `Disallow: /` (simple group parser)."""
    blocked, agents, in_group = set(), [], False
    for raw in robots_txt.splitlines():
        line = raw.split("#", 1)[0].strip()
        if not line:
            continue
        key, _, val = line.partition(":")
        key, val = key.strip().lower(), val.strip()
        if key == "user-agent":
            if in_group:
                agents, in_group = [], False
            agents.append(val)
        else:
            in_group = True
            if key == "disallow" and val == "/":
                for a in agents:
                    for bot in AI_BOTS:
                        if a.lower() == bot.lower():
                            blocked.add(bot)
    return sorted(blocked)


def pdp_summary(html):
    objs = ld_objects(html)
    product = next((o for o in objs if has_type(o, "Product")), None)
    gtin = None
    if product:
        for k in ("gtin", "gtin8", "gtin12", "gtin13", "gtin14"):
            if product.get(k):
                gtin = product[k]
                break
    return {
        "product_ld": product is not None,
        "offers": bool(product and product.get("offers")),
        "aggregate_rating": bool(product and product.get("aggregateRating")),
        "gtin": bool(gtin),
        "faq_page_ld": any(has_type(o, "FAQPage") for o in objs),
    }


# ── report helpers ──────────────────────────────────────────────────────────

RESULTS = []          # (level, title, detail)  level: ok | action | info | warn
OK, ACTION, INFO, WARN = "ok", "action", "info", "warn"
GLYPH = {"ok": "✓", "action": "✗", "info": "–", "warn": "⚠"}


def add(level, title, detail=""):
    RESULTS.append({"level": level, "title": title, "detail": detail})


def run_checks(domain):
    base = f"https://{domain}"

    # 1. robots.txt
    try:
        _, body = fetch(f"{base}/robots.txt")
        blocked = blocked_ai_bots(body)
        if blocked:
            add(ACTION, f"robots.txt blocks AI crawlers: {', '.join(blocked)}",
                "Blocked agents cannot read the site — it disappears from AI-answer "
                "retrieval. If that is not a deliberate content-licensing stance, "
                "remove those Disallow groups.")
        else:
            add(OK, f"robots.txt blocks none of {len(AI_BOTS)} known AI agents")
    except urllib.error.HTTPError as e:
        add(OK, f"no robots.txt (HTTP {e.code}) — all crawlers allowed by default")
    except Exception as e:
        add(WARN, "robots.txt unreachable — no data, NOT 'all clear'", str(e))

    # 2. llms.txt
    try:
        status, body = fetch(f"{base}/llms.txt")
        if status == 200 and body.strip():
            add(OK, f"llms.txt live ({len(body)} bytes)")
        else:
            add(INFO, "no llms.txt (optional)",
                "A curated llms.txt hands AI agents a guide to your site. "
                "Optional, cheap, worth having.")
    except urllib.error.HTTPError:
        add(INFO, "no llms.txt (optional)",
            "A curated llms.txt hands AI agents a guide to your site. "
            "Optional, cheap, worth having.")
    except Exception as e:
        add(WARN, "llms.txt unreachable — no data, NOT 'all clear'", str(e))

    # Shopify auto-detect via products.json
    handles, shopify = [], False
    try:
        _, body = fetch(f"{base}/products.json?limit=250")
        products = json.loads(body).get("products", [])
        shopify = True
        handles = [p.get("handle") for p in products if p.get("handle")]
        no_price = [p.get("handle") for p in products
                    if not any(float(v.get("price") or 0) > 0 for v in p.get("variants", []))]
        if products and not no_price:
            add(OK, f"products.json readable: {len(products)} products, all priced > 0")
        elif not products:
            add(ACTION, "products.json returns 0 products",
                "The agent-facing catalog is empty — agents cannot see what you sell.")
        else:
            add(WARN, f"products.json: {len(no_price)}/{len(products)} products without "
                f"a positive price", f"Handles: {', '.join(no_price)}")
    except Exception:
        add(INFO, "not a Shopify storefront (no products.json) — store checks skipped",
            "UCP / catalog / PDP-schema checks apply to Shopify stores.")

    if shopify:
        # 3. UCP
        try:
            status, body = fetch(f"{base}/.well-known/ucp")
            valid = False
            try:
                valid = isinstance(json.loads(body), dict)
            except json.JSONDecodeError:
                pass
            if status == 200 and valid:
                add(OK, "UCP endpoint live and valid JSON (/.well-known/ucp)")
            else:
                add(ACTION, "UCP endpoint present but invalid",
                    "The agent-commerce profile does not parse as JSON.")
        except urllib.error.HTTPError as e:
            add(INFO, f"no UCP endpoint (HTTP {e.code})",
                "Shopify serves /.well-known/ucp on current plans — if you expect it, "
                "check plan/settings.")
        except Exception as e:
            add(WARN, "UCP unreachable — no data, NOT 'all clear'", str(e))

        # 4. PDP JSON-LD on every product page (capped)
        capped = handles[:PDP_CAP]
        if len(handles) > PDP_CAP:
            add(INFO, f"checking first {PDP_CAP} of {len(handles)} product pages "
                "(politeness cap)")
        stats = {"product_ld": [], "aggregate_rating": [], "gtin": [], "faq_page_ld": []}
        fetched = 0
        for h in capped:
            try:
                _, html = fetch(f"{base}/products/{h}")
                s = pdp_summary(html)
                fetched += 1
                for k in stats:
                    if not s[k]:
                        stats[k].append(h)
                print(f"    … {h}", file=sys.stderr, flush=True)
            except Exception as e:
                add(WARN, f"PDP fetch failed: {h}", str(e))
        if fetched:
            def report(key, label, why, optional=False):
                missing = stats[key]
                if not missing:
                    add(OK, f"{label} on {fetched}/{fetched} product pages")
                else:
                    add(INFO if optional else ACTION,
                        f"{label} missing on {len(missing)}/{fetched} product pages — {why}",
                        f"Handles: {', '.join(missing)}")
            report("product_ld", "Product JSON-LD",
                   "agents/search cannot parse the product at all")
            report("aggregate_rating", "aggregateRating",
                   "your review trust signal is invisible (no SERP stars, nothing "
                   "for agents to weigh). Most review apps can inject this — it is "
                   "usually a checkbox, not a build")
            report("gtin", "GTIN/barcode in Product schema",
                   "GTIN is the join key that lets agents recognize the same product "
                   "across your site, marketplaces and feeds")
            report("faq_page_ld", "FAQPage JSON-LD",
                   "nice-to-have: feeds Q&A surfaces in search/AI answers", optional=True)

    # 5. The honest boundary
    add(INFO, "NOT checked: whether AI answers actually cite you (citation layer)",
        "That requires running your real category queries against AI surfaces "
        "monthly and judging the results — editorial placements, Reddit presence, "
        "reviews and PR are what move it. No script can automate the judgment.")


def main():
    ap = argparse.ArgumentParser(description="AI visibility check (standalone, stdlib-only)")
    ap.add_argument("domain", help="yourdomain.com (no scheme)")
    ap.add_argument("--json", action="store_true", help="machine-readable output")
    args = ap.parse_args()
    domain = re.sub(r"^https?://", "", args.domain).strip("/")

    print(f"ai-visibility-check {VERSION} — {domain}\n", file=sys.stderr, flush=True)
    run_checks(domain)

    if args.json:
        print(json.dumps({"domain": domain, "version": VERSION, "results": RESULTS},
                         indent=2))
    else:
        order = {ACTION: 0, WARN: 1, OK: 2, INFO: 3}
        for r in sorted(RESULTS, key=lambda r: order[r["level"]]):
            print(f" {GLYPH[r['level']]} {r['title']}")
            if r["detail"]:
                print(f"   {r['detail']}")
        n_action = sum(1 for r in RESULTS if r["level"] == ACTION)
        n_warn = sum(1 for r in RESULTS if r["level"] == WARN)
        print(f"\n {n_action} to fix · {n_warn} unclear · "
              f"{sum(1 for r in RESULTS if r['level'] == OK)} ok")
        if n_action == 0 and n_warn == 0:
            print(" Machine-readable layer is in order. The citation layer is yours to earn.")
    sys.exit(1 if any(r["level"] == ACTION for r in RESULTS) else 0)


if __name__ == "__main__":
    main()
