#!/usr/bin/env python3 """Crawler watch for a Traefik-fronted Forgejo (or any Traefik site). Separates real AI crawlers from traffic that only claims to be, by checking the source IP against each vendor's published range list. Modes: (default) daily threshold alert: mail only if yesterday exceeded CRAWLER_THRESHOLD requests --weekly summary: always mail stats for the past 7 days --dry-run print instead of mailing (combinable with --weekly) Configuration is by environment variable; see README.md. The only value with no sensible default is CRAWLER_RECIPIENT. """ import configparser, glob, gzip, ipaddress, json, os, re, smtplib, subprocess, sys, urllib.request from collections import Counter from datetime import date, timedelta from email.mime.text import MIMEText # --- configuration ------------------------------------------------------ THRESHOLD = int(os.environ.get("CRAWLER_THRESHOLD", "100000")) RECIPIENT = os.environ.get("CRAWLER_RECIPIENT", "") SITE = os.environ.get("CRAWLER_SITE", "this site") LOG_GLOB = os.environ.get("CRAWLER_LOG", "/var/log/traefik/access.log*") # Traefik router name to filter on, e.g. "forgejo@docker". Empty = all # traffic reaching Traefik, which is what you want for a single-site host. ROUTER = os.environ.get("CRAWLER_ROUTER", "") # Container to read SMTP settings from (Forgejo/Gitea app.ini [mailer]). # Set CRAWLER_SMTP_* instead to configure SMTP directly. MAIL_CONTAINER = os.environ.get("CRAWLER_MAIL_CONTAINER", "forgejo") MAIL_INI = os.environ.get("CRAWLER_MAIL_INI", "/data/gitea/conf/app.ini") WEEKLY = "--weekly" in sys.argv DRY = "--dry-run" in sys.argv if not RECIPIENT and not DRY: sys.exit("CRAWLER_RECIPIENT is not set (no --dry-run either); refusing " "to run. See README.md.") # --- verified-crawler support ------------------------------------------- # Vendors publishing machine-readable IP ranges. A UA token alone proves # nothing: on the instance this was written for, ~83% of AI-bot-labelled # traffic came from hosts outside these ranges. Only IP-verified hits count. RANGE_SOURCES = { "openai": ["https://openai.com/chatgpt-user.json", "https://openai.com/gptbot.json", "https://openai.com/searchbot.json"], "anthropic": ["https://claude.com/crawling/bots.json"], "perplexity":["https://www.perplexity.ai/perplexitybot.json"], "apple": ["https://search.developer.apple.com/applebot.json"], "google": ["https://developers.google.com/static/crawling/ipranges/special-crawlers.json", "https://developers.google.com/static/crawling/ipranges/common-crawlers.json", "https://developers.google.com/static/crawling/ipranges/user-triggered-fetchers.json", "https://developers.google.com/static/crawling/ipranges/user-triggered-fetchers-google.json"], } # UA token -> vendor whose ranges must contain the source IP BOT_VENDOR = { "ChatGPT-User": "openai", "GPTBot": "openai", "OAI-SearchBot": "openai", "ClaudeBot": "anthropic", "Claude-User": "anthropic", "Claude-SearchBot": "anthropic", "PerplexityBot": "perplexity", "Perplexity-User": "perplexity", "Applebot-Extended": "apple", "Applebot": "apple", "Googlebot": "google", "Google-Extended": "google", "GoogleOther": "google", } CACHE = os.environ.get("CRAWLER_CACHE", "/var/cache/crawler-alert/ranges.json") CF_URL = "https://www.cloudflare.com/ips-v4" # claude.com and cloudflare.com sit behind Cloudflare, which 403s the # default Python-urllib UA. Send a real one. UA = "Mozilla/5.0 (compatible; crawler-alert/1.0; +https://codeberg.org/)" def fetch(url, timeout=20): req = urllib.request.Request(url, headers={"User-Agent": UA}) with urllib.request.urlopen(req, timeout=timeout) as r: return r.read() def load_ranges(): """Fetch vendor prefixes; fall back to cache so a vendor outage degrades the report instead of breaking the mail.""" out, stale = {}, [] for vendor, urls in RANGE_SOURCES.items(): nets = [] for u in urls: try: for p in json.loads(fetch(u)).get("prefixes", []): for k, v in p.items(): if k.startswith("ipv"): nets.append(v) except Exception: stale.append(vendor) if nets: out[vendor] = nets cf = [] try: cf = [l.strip() for l in fetch(CF_URL).decode().splitlines() if l.strip()] except Exception: stale.append("cloudflare") try: cached = json.load(open(CACHE)) except Exception: cached = {} for vendor in RANGE_SOURCES: if vendor not in out and vendor in cached.get("vendors", {}): out[vendor] = cached["vendors"][vendor] if not cf: cf = cached.get("cloudflare", []) if out: try: os.makedirs(os.path.dirname(CACHE), exist_ok=True) json.dump({"vendors": out, "cloudflare": cf}, open(CACHE, "w")) except Exception: pass # bucket by first octet so lookups stay cheap over ~1425 prefixes idx = {} for vendor, nets in out.items(): b = {} for s in nets: n = ipaddress.ip_network(s) b.setdefault(int(n.network_address) >> 24 if n.version == 4 else -1, []).append(n) idx[vendor] = b cfnets = [ipaddress.ip_network(s) for s in cf] return idx, cfnets, sorted(set(stale)) def in_index(idx_v, addr): if idx_v is None: return False key = int(addr) >> 24 if addr.version == 4 else -1 return any(addr in n for n in idx_v.get(key, [])) days = [date.today() - timedelta(days=i) for i in range(7 if WEEKLY else 1, 0, -1)] stamps = [d.strftime("%d/%b/%Y") for d in days] RANGES, CFNETS, STALE = (load_ranges() if WEEKLY else ({}, [], [])) per_day = {s: [0, set()] for s in stamps} repos = Counter() uas = Counter() verified = Counter() # bot -> IP-verified hits spoofed = Counter() # bot -> UA claimed, IP outside vendor ranges proxied = Counter() # bot -> arrived via Cloudflare, unverifiable ver_paths = Counter() # paths fetched by verified bots ver_ips = {} # bot -> set of verified source IPs for path in sorted(glob.glob(LOG_GLOB)): opener = gzip.open if path.endswith(".gz") else open with opener(path, "rt", errors="replace") as f: for line in f: if ROUTER and ROUTER not in line: continue for s in stamps: if s in line: per_day[s][0] += 1 per_day[s][1].add(line.split(" ", 1)[0]) parts = line.split() if len(parts) > 6 and parts[6].count("/") >= 2: repos["/".join(parts[6].split("/")[1:3])] += 1 # Traefik CLF: ... "" "" "" ... # Anchor on the request counter so this works whatever # the router is called. ua = re.search(r"\"[^\"]*\" \"([^\"]*)\" \d+ \"", line) if ua: u = ua.group(1) m = re.search(r"(bot|crawler|spider|scrapy|externalagent|gpt|claude|perplexity)[\w./-]*", u, re.I) uas[m.group(0) if m else ("(none)" if u == "-" else "(browser-like)")] += 1 if WEEKLY: # longest token first so Applebot-Extended wins over Applebot bot = next((b for b in sorted(BOT_VENDOR, key=len, reverse=True) if b in u), None) if bot: try: addr = ipaddress.ip_address(parts[0]) except ValueError: addr = None if addr is None: pass elif any(addr in n for n in CFNETS): proxied[bot] += 1 elif in_index(RANGES.get(BOT_VENDOR[bot]), addr): verified[bot] += 1 ver_ips.setdefault(bot, set()).add(parts[0]) if len(parts) > 6: ver_paths[parts[6][:60]] += 1 else: spoofed[bot] += 1 break total = sum(v[0] for v in per_day.values()) all_ips = set().union(*(v[1] for v in per_day.values())) if not WEEKLY: s0 = stamps[0] count, ips = per_day[s0][0], len(per_day[s0][1]) print(f"{s0}: {count} requests, {ips} unique IPs (threshold {THRESHOLD})") if DRY or count <= THRESHOLD: sys.exit(0) subject = f"{SITE}: {count} requests on {s0} (over {THRESHOLD}/day threshold)" body = (f"Forgejo served {count} requests from {ips} unique IPs on {s0} - " f"above the {THRESHOLD}/day watch threshold.\n\n" "Time to consider Anubis (PoW challenge) in front of Forgejo.\n") else: lines = [f"{SITE} weekly crawler report ({stamps[0]} - {stamps[-1]})", ""] lines.append(f"Total: {total} requests, {len(all_ips)} unique IPs") lines.append(f"Daily threshold alert fires above {THRESHOLD} req/day (none = quiet week)") lines.append("") lines.append("Per day:") for s in stamps: lines.append(f" {s}: {per_day[s][0]:>7} requests, {len(per_day[s][1]):>6} unique IPs") lines.append("") lines.append("Top repos:") for r, c in repos.most_common(5): lines.append(f" {c:>7} {r}") lines.append("") lines.append("User-agent classes (browser-like = mostly the spoofing swarm):") for u, c in uas.most_common(8): lines.append(f" {c:>7} {u}") lines.append("") # --- legitimate (IP-verified) crawlers ------------------------------- lines.append("=" * 62) lines.append("LEGITIMATE CRAWLERS (source IP inside vendor's published ranges)") lines.append("=" * 62) if STALE: lines.append(f"NOTE: range fetch failed for {', '.join(STALE)} - used cache") tv, ts, tp = sum(verified.values()), sum(spoofed.values()), sum(proxied.values()) claimed = tv + ts + tp if not claimed: lines.append(" No AI-bot user-agents seen this week.") else: lines.append(f" {'bot':<20}{'verified':>9}{'spoofed':>9}{'via CF':>8} {'IPs':>4}") for bot in sorted(set(verified) | set(spoofed) | set(proxied), key=lambda b: -(verified[b] + spoofed[b] + proxied[b])): lines.append(f" {bot:<20}{verified[bot]:>9}{spoofed[bot]:>9}" f"{proxied[bot]:>8} {len(ver_ips.get(bot, ())):>4}") pct = tv / claimed * 100 lines.append("") lines.append(f" {claimed} requests claimed an AI-bot identity; {tv} verified ({pct:.1f}%).") lines.append(f" {ts} failed IP verification (spoofed). {tp} arrived via Cloudflare") lines.append(" and cannot be verified by IP - not counted either way.") if ver_paths: lines.append("") lines.append(" Top paths fetched by verified crawlers:") for p, c in ver_paths.most_common(8): lines.append(f" {c:>6} {p}") body = "\n".join(lines) + "\n" subject = f"{SITE} weekly crawler report: {total} requests, {len(all_ips)} IPs" if DRY: print(subject); print(); print(body) sys.exit(0) # SMTP settings: use CRAWLER_SMTP_* if given, otherwise borrow them from # a Forgejo/Gitea app.ini [mailer] section so there is no second copy of # the credentials to keep in sync. if os.environ.get("CRAWLER_SMTP_ADDR"): addr = os.environ["CRAWLER_SMTP_ADDR"] port = int(os.environ.get("CRAWLER_SMTP_PORT", "587")) user = os.environ.get("CRAWLER_SMTP_USER", "") passwd = os.environ.get("CRAWLER_SMTP_PASSWORD", "") sender = os.environ.get("CRAWLER_SMTP_FROM", user or RECIPIENT) proto = os.environ.get("CRAWLER_SMTP_PROTOCOL", "smtp+starttls") else: ini = subprocess.run(["docker", "exec", MAIL_CONTAINER, "cat", MAIL_INI], capture_output=True, text=True).stdout if not ini.strip(): sys.exit(f"could not read {MAIL_INI} from container " f"'{MAIL_CONTAINER}' and CRAWLER_SMTP_ADDR is unset") cp = configparser.ConfigParser(interpolation=None, strict=False) cp.read_string("[DEFAULT]\n" + ini) m = cp["mailer"] proto = m.get("PROTOCOL", "smtps").strip() addr, port = m.get("SMTP_ADDR").strip(), int(m.get("SMTP_PORT", "465").strip()) user, passwd = m.get("USER", "").strip(), m.get("PASSWD", "").strip().strip("`\"") sender = m.get("FROM", user).strip() msg = MIMEText(body) msg["Subject"], msg["From"], msg["To"] = subject, sender, RECIPIENT if proto == "smtps": s = smtplib.SMTP_SSL(addr, port, timeout=30) else: s = smtplib.SMTP(addr, port, timeout=30) if proto != "smtp": # plain smtp = no TLS (local relay) s.starttls() if user: # unauthenticated local relays exist s.login(user, passwd) s.sendmail(sender, [RECIPIENT], msg.as_string()) s.quit() print(f"mailed to {RECIPIENT}")