#!/usr/bin/env python3 """Crawler watch for a Traefik-fronted Forgejo (or any Traefik site). Separates real AI crawlers from traffic that only claims to be, by checking the source IP against each vendor's published range list. Modes: (default) daily threshold alert: mail only if yesterday exceeded CRAWLER_THRESHOLD requests --weekly summary: always mail stats for the past 7 days --dry-run print instead of mailing (combinable with --weekly) Configuration is by environment variable; see README.md. The only value with no sensible default is CRAWLER_RECIPIENT. """ import configparser, glob, gzip, ipaddress, json, os, re, smtplib, socket, subprocess, sys, urllib.request from collections import Counter from datetime import date, timedelta from email.mime.text import MIMEText # --- configuration ------------------------------------------------------ THRESHOLD = int(os.environ.get("CRAWLER_THRESHOLD", "100000")) RECIPIENT = os.environ.get("CRAWLER_RECIPIENT", "") SITE = os.environ.get("CRAWLER_SITE", "this site") LOG_GLOB = os.environ.get("CRAWLER_LOG", "/var/log/traefik/access.log*") # Traefik router name to filter on, e.g. "forgejo@docker". Empty = all # traffic reaching Traefik, which is what you want for a single-site host. ROUTER = os.environ.get("CRAWLER_ROUTER", "") # Container to read SMTP settings from (Forgejo/Gitea app.ini [mailer]). # Set CRAWLER_SMTP_* instead to configure SMTP directly. MAIL_CONTAINER = os.environ.get("CRAWLER_MAIL_CONTAINER", "forgejo") MAIL_INI = os.environ.get("CRAWLER_MAIL_INI", "/data/gitea/conf/app.ini") WEEKLY = "--weekly" in sys.argv DRY = "--dry-run" in sys.argv if not RECIPIENT and not DRY: sys.exit("CRAWLER_RECIPIENT is not set (no --dry-run either); refusing " "to run. See README.md.") # --- verified-crawler support ------------------------------------------- # Vendors publishing machine-readable IP ranges. A UA token alone proves # nothing: on the instance this was written for, ~83% of AI-bot-labelled # traffic came from hosts outside these ranges. Only IP-verified hits count. RANGE_SOURCES = { "openai": ["https://openai.com/chatgpt-user.json", "https://openai.com/gptbot.json", "https://openai.com/searchbot.json"], "anthropic": ["https://claude.com/crawling/bots.json"], "perplexity":["https://www.perplexity.ai/perplexitybot.json"], "apple": ["https://search.developer.apple.com/applebot.json"], "google": ["https://developers.google.com/static/crawling/ipranges/special-crawlers.json", "https://developers.google.com/static/crawling/ipranges/common-crawlers.json", "https://developers.google.com/static/crawling/ipranges/user-triggered-fetchers.json", "https://developers.google.com/static/crawling/ipranges/user-triggered-fetchers-google.json"], } # UA token -> vendor whose ranges must contain the source IP BOT_VENDOR = { "ChatGPT-User": "openai", "GPTBot": "openai", "OAI-SearchBot": "openai", "ClaudeBot": "anthropic", "Claude-User": "anthropic", "Claude-SearchBot": "anthropic", "PerplexityBot": "perplexity", "Perplexity-User": "perplexity", "Applebot-Extended": "apple", "Applebot": "apple", "Googlebot": "google", "Google-Extended": "google", "GoogleOther": "google", } # --- reverse-DNS verification ------------------------------------------- # For vendors that publish no IP ranges. The check is forward-confirmed # (FCrDNS): PTR of the address must end in one of the vendor's domains, # AND that hostname must resolve back to the same address. The forward # step is what makes it meaningful - a PTR record alone is set by # whoever controls the address block, so it proves nothing on its own. BOT_RDNS = { "Amazonbot": (".crawl.amazonbot.amazon",), "meta-externalagent": (".facebook.com", ".fbsv.net"), "meta-externalfetcher": (".facebook.com", ".fbsv.net"), "FacebookBot": (".facebook.com", ".fbsv.net"), "YouBot": (".search.you.com",), "Bytespider": (".bytedance.com", ".byteoversea.com"), "PetalBot": (".petalsearch.com", ".aspiegel.com"), "DuckAssistBot": (".duckduckgo.com",), # Apple and Google publish ranges, but also answer rDNS; listed so a # range miss can still be resolved when the published file is stale. "Applebot": (".applebot.apple.com",), "Googlebot": (".googlebot.com", ".google.com"), } # rDNS is slow (two lookups per address), so only unique IPs are probed # and results are memoised for the run. RDNS_MAX = int(os.environ.get("CRAWLER_RDNS_MAX", "400")) RDNS_ENABLE = os.environ.get("CRAWLER_RDNS", "1") != "0" # --- ASN verification, for single-tenant networks only ------------------- # An ASN match usually proves nothing: genuine YouBot and genuine # Amazonbot both live in AS14618, which is also every EC2 instance a # spoofer could rent. It is only evidence when the ASN belongs to the # vendor exclusively and sells no compute in it. Meta qualifies - # AS32934 is Meta's own network and cannot be rented - and Meta # publishes no IP range file (their docs say to email webmasters@meta.com), # so without this there is nothing to check them against at all. BOT_ASN = { "meta-externalagent": {"32934"}, "meta-externalfetcher": {"32934"}, "FacebookBot": {"32934"}, "facebookexternalhit": {"32934"}, } ASN_ENABLE = os.environ.get("CRAWLER_ASN", "1") != "0" # Team Cymru's DNS interface: .origin[6].asn.cymru.com TXT CYMRU4 = "origin.asn.cymru.com" CYMRU6 = "origin6.asn.cymru.com" CACHE = os.environ.get("CRAWLER_CACHE", "/var/cache/crawler-alert/ranges.json") CF_URL = "https://www.cloudflare.com/ips-v4" # claude.com and cloudflare.com sit behind Cloudflare, which 403s the # default Python-urllib UA. Send a real one. UA = "Mozilla/5.0 (compatible; crawler-alert/1.0; +https://codeberg.org/)" def fetch(url, timeout=20): req = urllib.request.Request(url, headers={"User-Agent": UA}) with urllib.request.urlopen(req, timeout=timeout) as r: return r.read() def load_ranges(): """Fetch vendor prefixes; fall back to cache so a vendor outage degrades the report instead of breaking the mail.""" out, stale = {}, [] for vendor, urls in RANGE_SOURCES.items(): nets = [] for u in urls: try: for p in json.loads(fetch(u)).get("prefixes", []): for k, v in p.items(): if k.startswith("ipv"): nets.append(v) except Exception: stale.append(vendor) if nets: out[vendor] = nets cf = [] try: cf = [l.strip() for l in fetch(CF_URL).decode().splitlines() if l.strip()] except Exception: stale.append("cloudflare") try: cached = json.load(open(CACHE)) except Exception: cached = {} for vendor in RANGE_SOURCES: if vendor not in out and vendor in cached.get("vendors", {}): out[vendor] = cached["vendors"][vendor] if not cf: cf = cached.get("cloudflare", []) if out: try: os.makedirs(os.path.dirname(CACHE), exist_ok=True) json.dump({"vendors": out, "cloudflare": cf}, open(CACHE, "w")) except Exception: pass # bucket by first octet so lookups stay cheap over ~1425 prefixes idx = {} for vendor, nets in out.items(): b = {} for s in nets: n = ipaddress.ip_network(s) b.setdefault(int(n.network_address) >> 24 if n.version == 4 else -1, []).append(n) idx[vendor] = b cfnets = [ipaddress.ip_network(s) for s in cf] return idx, cfnets, sorted(set(stale)) def in_index(idx_v, addr): if idx_v is None: return False key = int(addr) >> 24 if addr.version == 4 else -1 return any(addr in n for n in idx_v.get(key, [])) _asn_cache = {} def asn_of(ip): """Origin ASN for an address via Team Cymru's DNS interface. Returns the ASN as a string, or "" if it cannot be determined. Needs `dig`; absence degrades to no ASN evidence, never an error. """ if ip in _asn_cache: return _asn_cache[ip] asn = "" try: a = ipaddress.ip_address(ip) if a.version == 4: q = ".".join(reversed(ip.split("."))) + "." + CYMRU4 else: nibbles = a.exploded.replace(":", "") q = ".".join(reversed(nibbles)) + "." + CYMRU6 out = subprocess.run(["dig", "+short", "+time=3", "+tries=1", q, "TXT"], capture_output=True, text=True, timeout=12).stdout # "32934 | 2a03:2880:f814::/48 | IE | ripencc | 2011-08-22" first = out.strip().strip('"').split("|")[0].strip() if first.split(" ")[0].isdigit(): asn = first.split(" ")[0] except (OSError, ValueError, subprocess.SubprocessError): pass _asn_cache[ip] = asn return asn _rdns_cache = {} def fcrdns(ip, suffixes): """Forward-confirmed reverse DNS. PTR must end in one of `suffixes`, and the name it gives must resolve back to `ip`. Returns (ok, hostname). Without the forward step this would only prove the address owner can write their own PTR record. """ key = (ip, suffixes) if key in _rdns_cache: return _rdns_cache[key] result = (False, "", "mismatch") try: host = socket.gethostbyaddr(ip)[0].rstrip(".").lower() if any(host.endswith(s) or host == s.lstrip(".") for s in suffixes): _, _, addrs = socket.gethostbyname_ex(host) try: v6 = socket.getaddrinfo(host, None, socket.AF_INET6) addrs = addrs + [a[4][0] for a in v6] except OSError: pass if ip in addrs: result = (True, host, "ok") else: result = (False, host, "mismatch") else: result = (False, host, "mismatch") except OSError: # No PTR at all. This is absence of evidence, not evidence of # forgery: Meta's IPv6 crawler space, for one, publishes no PTR # records. Kept distinct so a real mismatch stays visible. result = (False, "", "no-ptr") _rdns_cache[key] = result return result days = [date.today() - timedelta(days=i) for i in range(7 if WEEKLY else 1, 0, -1)] stamps = [d.strftime("%d/%b/%Y") for d in days] RANGES, CFNETS, STALE = (load_ranges() if WEEKLY else ({}, [], [])) per_day = {s: [0, set()] for s in stamps} repos = Counter() uas = Counter() verified = Counter() # bot -> IP-verified hits spoofed = Counter() # bot -> UA claimed, IP outside vendor ranges proxied = Counter() # bot -> arrived via Cloudflare, unverifiable ver_paths = Counter() # paths fetched by verified bots ver_ips = {} # bot -> set of verified source IPs # rDNS-verified vendors: {bot: {ip: hits}}, resolved after the log pass rdns_hits = {b: Counter() for b in BOT_RDNS} rdns_proxied = Counter() # {bot: {ip: Counter(path)}} - kept so that once an address is confirmed # by DNS/ASN its paths can join the verified-path breakdown, which # otherwise only ever showed IP-range-verified vendors. rdns_paths = {b: {} for b in BOT_RDNS} for path in sorted(glob.glob(LOG_GLOB)): opener = gzip.open if path.endswith(".gz") else open with opener(path, "rt", errors="replace") as f: for line in f: if ROUTER and ROUTER not in line: continue for s in stamps: if s in line: per_day[s][0] += 1 per_day[s][1].add(line.split(" ", 1)[0]) parts = line.split() if len(parts) > 6 and parts[6].count("/") >= 2: repos["/".join(parts[6].split("/")[1:3])] += 1 # Traefik CLF: ... "" "" "" ... # Anchor on the request counter so this works whatever # the router is called. ua = re.search(r"\"[^\"]*\" \"([^\"]*)\" \d+ \"", line) if ua: u = ua.group(1) m = re.search(r"(bot|crawler|spider|scrapy|externalagent|gpt|claude|perplexity)[\w./-]*", u, re.I) uas[m.group(0) if m else ("(none)" if u == "-" else "(browser-like)")] += 1 if WEEKLY: # longest token first so Applebot-Extended wins over Applebot bot = next((b for b in sorted(BOT_VENDOR, key=len, reverse=True) if b in u), None) if bot: try: addr = ipaddress.ip_address(parts[0]) except ValueError: addr = None if addr is None: pass elif any(addr in n for n in CFNETS): proxied[bot] += 1 elif in_index(RANGES.get(BOT_VENDOR[bot]), addr): verified[bot] += 1 ver_ips.setdefault(bot, set()).add(parts[0]) if len(parts) > 6: ver_paths[parts[6][:60]] += 1 else: spoofed[bot] += 1 if WEEKLY and RDNS_ENABLE: rb = next((b for b in sorted(BOT_RDNS, key=len, reverse=True) if b in u), None) # skip if already counted by IP-range check if rb and not (rb in BOT_VENDOR and rb == bot): try: a2 = ipaddress.ip_address(parts[0]) except ValueError: a2 = None if a2 is None: pass elif any(a2 in n for n in CFNETS): rdns_proxied[rb] += 1 else: rdns_hits[rb][parts[0]] += 1 if len(parts) > 6 and not a2.is_private: rdns_paths[rb].setdefault( parts[0], Counter())[parts[6][:60]] += 1 break total = sum(v[0] for v in per_day.values()) all_ips = set().union(*(v[1] for v in per_day.values())) # --- resolve rDNS candidates --------------------------------------------- # One pass over unique addresses, busiest first, capped by RDNS_MAX so a # flood of distinct forgeries cannot stall the report on DNS timeouts. rdns_ok = Counter() # bot -> forward-confirmed by PTR rdns_mismatch = Counter() # bot -> PTR exists but points elsewhere: forgery rdns_noptr = Counter() # bot -> no PTR at all: cannot be judged by DNS rdns_asn = Counter() # bot -> no PTR, but origin ASN is the vendor's rdns_skipped = Counter() # bot -> hits left unresolved by the cap rdns_private = Counter() # bot -> hits from RFC1918/loopback (see below) rdns_ok_ips = {} if WEEKLY and RDNS_ENABLE: socket.setdefaulttimeout(3) for bot, counter in rdns_hits.items(): budget = RDNS_MAX for ip, hits in counter.most_common(): # A private source address means the real client IP never # reached the log: something in front (Docker's bridge # gateway, a local proxy) is rewriting it. Neither verified # nor spoofed - the evidence simply is not there. if ipaddress.ip_address(ip).is_private: rdns_private[bot] += hits continue if budget <= 0: rdns_skipped[bot] += hits continue budget -= 1 ok, _host, why = fcrdns(ip, BOT_RDNS[bot]) confirmed = False if ok: rdns_ok[bot] += hits rdns_ok_ips.setdefault(bot, set()).add(ip) confirmed = True elif why == "no-ptr": # Fall back to ASN, but only for vendors whose network is # single-tenant (see BOT_ASN). Otherwise leave it unknown. if ASN_ENABLE and bot in BOT_ASN and asn_of(ip) in BOT_ASN[bot]: rdns_asn[bot] += hits rdns_ok_ips.setdefault(bot, set()).add(ip) confirmed = True else: rdns_noptr[bot] += hits else: rdns_mismatch[bot] += hits if confirmed: ver_paths.update(rdns_paths.get(bot, {}).get(ip, ())) socket.setdefaulttimeout(None) if not WEEKLY: s0 = stamps[0] count, ips = per_day[s0][0], len(per_day[s0][1]) print(f"{s0}: {count} requests, {ips} unique IPs (threshold {THRESHOLD})") if DRY or count <= THRESHOLD: sys.exit(0) subject = f"{SITE}: {count} requests on {s0} (over {THRESHOLD}/day threshold)" body = (f"Forgejo served {count} requests from {ips} unique IPs on {s0} - " f"above the {THRESHOLD}/day watch threshold.\n\n" "Time to consider Anubis (PoW challenge) in front of Forgejo.\n") else: lines = [f"{SITE} weekly crawler report ({stamps[0]} - {stamps[-1]})", ""] lines.append(f"Total: {total} requests, {len(all_ips)} unique IPs") lines.append(f"Daily threshold alert fires above {THRESHOLD} req/day (none = quiet week)") lines.append("") lines.append("Per day:") for s in stamps: lines.append(f" {s}: {per_day[s][0]:>7} requests, {len(per_day[s][1]):>6} unique IPs") lines.append("") lines.append("Top repos:") for r, c in repos.most_common(5): lines.append(f" {c:>7} {r}") lines.append("") lines.append("User-agent classes (browser-like = mostly the spoofing swarm):") for u, c in uas.most_common(8): lines.append(f" {c:>7} {u}") lines.append("") # --- legitimate (IP-verified) crawlers ------------------------------- lines.append("=" * 62) lines.append("LEGITIMATE CRAWLERS (source IP inside vendor's published ranges)") lines.append("=" * 62) if STALE: lines.append(f"NOTE: range fetch failed for {', '.join(STALE)} - used cache") tv, ts, tp = sum(verified.values()), sum(spoofed.values()), sum(proxied.values()) claimed = tv + ts + tp if not claimed: lines.append(" No AI-bot user-agents seen this week.") else: lines.append(f" {'bot':<20}{'verified':>9}{'spoofed':>9}{'via CF':>8} {'IPs':>4}") for bot in sorted(set(verified) | set(spoofed) | set(proxied), key=lambda b: -(verified[b] + spoofed[b] + proxied[b])): lines.append(f" {bot:<20}{verified[bot]:>9}{spoofed[bot]:>9}" f"{proxied[bot]:>8} {len(ver_ips.get(bot, ())):>4}") pct = tv / claimed * 100 lines.append("") lines.append(f" {claimed} requests claimed an AI-bot identity; {tv} verified ({pct:.1f}%).") lines.append(f" {ts} failed IP verification (spoofed). {tp} arrived via Cloudflare") lines.append(" and cannot be verified by IP - not counted either way.") # --- vendors verified by reverse DNS --------------------------------- if RDNS_ENABLE and (rdns_ok or rdns_mismatch or rdns_noptr or rdns_asn or rdns_proxied or rdns_private): lines.append("") lines.append("-" * 62) lines.append("VERIFIED BY DNS / ASN (vendors publishing no IP ranges)") lines.append("-" * 62) lines.append(f" {'bot':<20}{'PTR ok':>7}{'ASN ok':>7}{'FORGED':>8}" f"{'no-PTR':>7}{'CF':>6}{'priv':>7} {'IPs':>4}") allbots = (set(rdns_ok) | set(rdns_mismatch) | set(rdns_noptr) | set(rdns_asn) | set(rdns_proxied) | set(rdns_private)) for bot in sorted(allbots, key=lambda b: -(rdns_ok[b] + rdns_mismatch[b] + rdns_noptr[b] + rdns_asn[b] + rdns_proxied[b] + rdns_private[b])): lines.append(f" {bot:<20}{rdns_ok[bot]:>7}{rdns_asn[bot]:>7}" f"{rdns_mismatch[bot]:>8}{rdns_noptr[bot]:>7}" f"{rdns_proxied[bot]:>6}{rdns_private[bot]:>7}" f" {len(rdns_ok_ips.get(bot, ())):>4}") if rdns_skipped: sk = ", ".join(f"{b} {c}" for b, c in rdns_skipped.most_common(4)) lines.append(f" (unresolved, over CRAWLER_RDNS_MAX={RDNS_MAX}: {sk})") lines.append("") lines.append(" PTR ok forward-confirmed: PTR ends in a vendor domain AND") lines.append(" that hostname resolves back to the same address.") lines.append(" ASN ok no PTR, but the address sits in an ASN belonging") lines.append(" solely to the vendor (Meta's AS32934). Only used") lines.append(" where the ASN sells no compute - never for clouds.") lines.append(" FORGED a PTR exists and points somewhere else. This is") lines.append(" the only column that is evidence of forgery.") lines.append(" no-PTR no PTR record and no usable ASN: unknown, not bad.") if sum(rdns_private.values()): lines.append(" priv logged source was RFC1918: a proxy or Docker") lines.append(" bridge masked the client. Unjudgeable. See README.") # --- what the verified crawlers actually fetched --------------------- # Last, because it spans both tables above: an address only counts here # once it has been confirmed, whether by IP range, PTR or ASN. if ver_paths: lines.append("") lines.append("=" * 62) lines.append("TOP PATHS FETCHED BY VERIFIED CRAWLERS (all methods)") lines.append("=" * 62) for p, c in ver_paths.most_common(12): lines.append(f" {c:>6} {p}") body = "\n".join(lines) + "\n" subject = f"{SITE} weekly crawler report: {total} requests, {len(all_ips)} IPs" if DRY: print(subject); print(); print(body) sys.exit(0) # SMTP settings: use CRAWLER_SMTP_* if given, otherwise borrow them from # a Forgejo/Gitea app.ini [mailer] section so there is no second copy of # the credentials to keep in sync. if os.environ.get("CRAWLER_SMTP_ADDR"): addr = os.environ["CRAWLER_SMTP_ADDR"] port = int(os.environ.get("CRAWLER_SMTP_PORT", "587")) user = os.environ.get("CRAWLER_SMTP_USER", "") passwd = os.environ.get("CRAWLER_SMTP_PASSWORD", "") sender = os.environ.get("CRAWLER_SMTP_FROM", user or RECIPIENT) proto = os.environ.get("CRAWLER_SMTP_PROTOCOL", "smtp+starttls") else: ini = subprocess.run(["docker", "exec", MAIL_CONTAINER, "cat", MAIL_INI], capture_output=True, text=True).stdout if not ini.strip(): sys.exit(f"could not read {MAIL_INI} from container " f"'{MAIL_CONTAINER}' and CRAWLER_SMTP_ADDR is unset") cp = configparser.ConfigParser(interpolation=None, strict=False) cp.read_string("[DEFAULT]\n" + ini) m = cp["mailer"] proto = m.get("PROTOCOL", "smtps").strip() addr, port = m.get("SMTP_ADDR").strip(), int(m.get("SMTP_PORT", "465").strip()) user, passwd = m.get("USER", "").strip(), m.get("PASSWD", "").strip().strip("`\"") sender = m.get("FROM", user).strip() msg = MIMEText(body) msg["Subject"], msg["From"], msg["To"] = subject, sender, RECIPIENT if proto == "smtps": s = smtplib.SMTP_SSL(addr, port, timeout=30) else: s = smtplib.SMTP(addr, port, timeout=30) if proto != "smtp": # plain smtp = no TLS (local relay) s.starttls() if user: # unauthenticated local relays exist s.login(user, passwd) s.sendmail(sender, [RECIPIENT], msg.as_string()) s.quit() print(f"mailed to {RECIPIENT}")