#!/usr/bin/env python3
"""Suede AI Crawler Access Index v2. Python 3 standard library only.

Usage: python3 ai-crawler-access-index-method-v2.py --list-id ID --output results.json
Measures declared policy for the root path only, not actual crawler access.
The pinned Tranco list, timestamps, response hashes and relevant rules make
the observations auditable. Re-running later measures a new point in time.
"""
import argparse
import concurrent.futures
import csv
import hashlib
import io
import json
import re
import urllib.error
import urllib.request
import zipfile
from datetime import datetime, timezone
from pathlib import Path

AGENTS = ['GPTBot', 'OAI-SearchBot', 'ChatGPT-User', 'ClaudeBot', 'anthropic-ai',
          'Claude-Web', 'PerplexityBot', 'CCBot', 'Google-Extended', 'Bytespider',
          'Applebot-Extended', 'meta-externalagent']
CORE = ['CCBot', 'ClaudeBot', 'GPTBot', 'Google-Extended', 'PerplexityBot']
UA = 'SuedeCrawlerIndex/2.0 (+https://suedeai.ai/blog/ai-crawler-access-index)'
LIMIT = 512000

def now():
    return datetime.now(timezone.utc).isoformat()

def sha(data):
    return hashlib.sha256(data).hexdigest()

def fetch(url, limit=LIMIT):
    out = {'requested_url': url, 'observed_at': now()}
    try:
        try:
            response = urllib.request.urlopen(urllib.request.Request(url, headers={'User-Agent': UA}), timeout=10)
        except urllib.error.HTTPError as error:
            response = error
        with response:
            raw = response.read(limit + 1)
            out.update(status=response.status, final_url=response.url,
                       content_type=response.headers.get('Content-Type', ''),
                       truncated=len(raw) > limit, body_sha256=sha(raw), bytes_read=len(raw))
            return out, raw.decode('utf-8-sig', errors='replace')
    except (OSError, ValueError) as error:
        out.update(status=None, error=type(error).__name__)
        return out, ''

def groups(text):
    result, agents, rules = [], [], []
    for line in text.splitlines():
        line = line.split('#', 1)[0].strip()
        if ':' not in line:
            continue
        key, value = (part.strip() for part in line.split(':', 1))
        key = key.lower()
        if key == 'user-agent':
            if rules:
                result.append((agents, rules))
                agents, rules = [], []
            if value:
                agents.append(value.lower())
        elif key in ('allow', 'disallow') and agents:
            rules.append((key, value))
    if agents:
        result.append((agents, rules))
    return result

def root_policy(parsed, agent):
    """Exact product-token groups merged; wildcard used only without a match.

    Only evaluates URI path '/'. Returns the matching rules used in the
    decision, so readers can independently recompute this limited metric.
    """
    specific = [rules for names, rules in parsed if agent.lower() in names]
    selected = specific or [rules for names, rules in parsed if '*' in names]
    matches = []
    for rules in selected:
        for kind, path in rules:
            if not path:
                continue
            anchored = path.endswith('$')
            pattern = path[:-1] if anchored else path
            expression = '^' + re.escape(pattern).replace(r'\*', '.*') + ('$' if anchored else '')
            if re.search(expression, '/'):
                matches.append((kind, path))
    # A trailing wildcard adds no specificity; a terminal anchor does.
    specificity = lambda rule: len(rule[1].rstrip('*').encode('utf-8'))
    longest = max((specificity(rule) for rule in matches), default=-1)
    winners = [rule for rule in matches if specificity(rule) == longest]
    denied = bool(winners) and not any(kind == 'allow' for kind, _ in winners)
    return {'root_disallowed': denied, 'named_group': bool(specific),
            'matching_rules': sorted(set(matches))}

def text_response(meta, text):
    return (meta.get('status') == 200 and not meta.get('truncated') and
            'html' not in meta.get('content_type', '').lower() and
            not re.search(r'<\s*(?:!doctype|html|head|body)\b', text[:1000], re.I))

def llms_candidate(meta, text):
    # Conservative observable shape of the proposal; not proof of adoption.
    return (text_response(meta, text) and bool(re.search(r'^#\s+\S', text, re.M)) and
            bool(re.search(r'\[[^\]\n]+\]\(https?://[^\s)]+\)', text)))

def probe(item):
    rank, domain = item
    base = 'https://' + domain
    robots, body = fetch(base + '/robots.txt')
    # No alternate-host retry: www and apex may publish different policies.
    readable = text_response(robots, body) and bool(re.search(r'^\s*user-agent\s*:', body, re.I | re.M))
    policies = {agent: root_policy(groups(body), agent) for agent in AGENTS} if readable else None
    llms, llms_body = fetch(base + '/llms.txt')
    state = 'not_detected' if llms.get('status') in (200, 404, 410) and not llms.get('truncated') else 'unavailable'
    control = None
    if llms_candidate(llms, llms_body):
        control, control_body = fetch(base + '/suede-index-control-9c52d073-no-file.txt')
        if control.get('status') not in (200, 404, 410) or control.get('truncated'):
            state = 'unknown_control'
        elif control.get('body_sha256') == llms.get('body_sha256') or control.get('final_url') == llms.get('final_url'):
            state = 'catch_all'
        else:
            state = 'candidate'
    return {'rank': rank, 'domain': domain, 'robots': robots,
            'robots_readable': readable, 'policies': policies,
            'llms': llms, 'llms_state': state, 'control': control}

def summarize(results):
    readable = [row for row in results if row['robots_readable']]
    return {'sample_size': len(results), 'robots_readable': len(readable),
            'robots_unclassified': len(results) - len(readable),
            'root_disallowed': {agent: sum(row['policies'][agent]['root_disallowed'] for row in readable) for agent in AGENTS},
            'any_core_root_disallowed': sum(any(row['policies'][a]['root_disallowed'] for a in CORE) for row in readable),
            'all_core_root_disallowed': sum(all(row['policies'][a]['root_disallowed'] for a in CORE) for row in readable),
            'llms_states': {state: sum(row['llms_state'] == state for row in results) for state in sorted({row['llms_state'] for row in results})}}

def main():
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument('--list-id', required=True)
    parser.add_argument('--size', type=int, default=1000)
    parser.add_argument('--workers', type=int, default=16)
    parser.add_argument('--output', required=True)
    args = parser.parse_args()
    if not re.fullmatch(r'[A-Za-z0-9]+', args.list_id) or not 1 <= args.size <= 1000 or not 1 <= args.workers <= 32:
        parser.error('Use an alphanumeric list ID, 1–1000 domains and 1–32 workers.')
    started = now()
    url = 'https://tranco-list.eu/download/' + args.list_id + '/1000000'
    with urllib.request.urlopen(url, timeout=45) as response:
        archive = response.read(50000000)
    if zipfile.is_zipfile(io.BytesIO(archive)):
        with zipfile.ZipFile(io.BytesIO(archive)) as zipped:
            frame = zipped.read(zipped.namelist()[0]).decode('utf-8')
    else:
        frame = archive.decode('utf-8')
    rows = csv.reader(io.StringIO(frame))
    sample = [(int(rank), domain) for rank, domain in rows if int(rank) <= args.size]
    if len(sample) != args.size or any(not re.fullmatch(r'[a-zA-Z0-9.-]+', domain) for _, domain in sample):
        raise ValueError('Unexpected sample frame')
    results = []
    with concurrent.futures.ThreadPoolExecutor(max_workers=args.workers) as pool:
        for row in pool.map(probe, sample):
            results.append(row)
            if len(results) % 100 == 0:
                print(f'{len(results)}/{args.size}', flush=True)
    output = {'method_version': 2, 'started_at': started, 'finished_at': now(),
              'method_sha256': sha(Path(__file__).read_bytes()),
              'sample': {'list_id': args.list_id, 'permalink': 'https://tranco-list.eu/list/' + args.list_id,
                         'download_url': url, 'archive_sha256': sha(archive), 'size': args.size},
              'user_agent': UA, 'agents': AGENTS, 'core_agents': CORE,
              'scope': "Declared robots.txt policy for URI path '/' only. llms.txt candidates pass a Markdown-shape and differing-control-response check. Neither metric proves indexing, citation, actual crawler access or vendor use.",
              'summary': summarize(results), 'results': results}
    Path(args.output).write_text(json.dumps(output, indent=2) + '\n')
    print(json.dumps(output['summary']), flush=True)

if __name__ == '__main__':
    main()
