traefik-crawlers-statistics/crawler-alert.py
Sergei Poljanski 81716c2fc9
Verify range-less vendors by forward-confirmed reverse DNS
Amazonbot, meta-externalagent, YouBot, Bytespider and PetalBot publish
no IP range file, so they were counted in the user-agent tally and
nowhere else - the largest single source of traffic on the sample
instance was also the least examined.

Check them by FCrDNS: the PTR record must end in a vendor domain and
that hostname must resolve back to the same address. The forward step
is the part that matters. A PTR record alone is written by whoever
controls the address block, so without confirming it forward the check
would accept anything its owner chose to claim.

ASN verification was considered and deliberately left out. Genuine
YouBot and genuine Amazonbot both live in AS14618, which is also every
EC2 instance a spoofer could rent. An ASN match shows the traffic came
from a cloud, not from the vendor; presenting that as verification
would be worse than presenting nothing.

Report RFC1918 sources separately rather than as failures. On the
sample instance 21624 of meta-externalagent's 21651 requests came from
172.18.0.1, the Docker bridge gateway: the real client address was
replaced before it reached the log. Those are not spoofed, they are
unjudgeable, and calling them spoofed would be a false accusation
caused by the reader's own proxy configuration.

Resolution is capped at CRAWLER_RDNS_MAX unique addresses per bot
(busiest first) so a flood of distinct forgeries cannot stall the
weekly mail on DNS timeouts, and can be disabled with CRAWLER_RDNS=0.

Assisted-by: Claude:opus-5
2026-08-11 02:55:57 +04:00

423 lines
20 KiB
Python
Executable file

#!/usr/bin/env python3
"""Crawler watch for a Traefik-fronted Forgejo (or any Traefik site).
Separates real AI crawlers from traffic that only claims to be, by
checking the source IP against each vendor's published range list.
Modes:
(default) daily threshold alert: mail only if yesterday exceeded
CRAWLER_THRESHOLD requests
--weekly summary: always mail stats for the past 7 days
--dry-run print instead of mailing (combinable with --weekly)
Configuration is by environment variable; see README.md. The only
value with no sensible default is CRAWLER_RECIPIENT.
"""
import configparser, glob, gzip, ipaddress, json, os, re, smtplib, socket, subprocess, sys, urllib.request
from collections import Counter
from datetime import date, timedelta
from email.mime.text import MIMEText
# --- configuration ------------------------------------------------------
THRESHOLD = int(os.environ.get("CRAWLER_THRESHOLD", "100000"))
RECIPIENT = os.environ.get("CRAWLER_RECIPIENT", "")
SITE = os.environ.get("CRAWLER_SITE", "this site")
LOG_GLOB = os.environ.get("CRAWLER_LOG", "/var/log/traefik/access.log*")
# Traefik router name to filter on, e.g. "forgejo@docker". Empty = all
# traffic reaching Traefik, which is what you want for a single-site host.
ROUTER = os.environ.get("CRAWLER_ROUTER", "")
# Container to read SMTP settings from (Forgejo/Gitea app.ini [mailer]).
# Set CRAWLER_SMTP_* instead to configure SMTP directly.
MAIL_CONTAINER = os.environ.get("CRAWLER_MAIL_CONTAINER", "forgejo")
MAIL_INI = os.environ.get("CRAWLER_MAIL_INI", "/data/gitea/conf/app.ini")
WEEKLY = "--weekly" in sys.argv
DRY = "--dry-run" in sys.argv
if not RECIPIENT and not DRY:
sys.exit("CRAWLER_RECIPIENT is not set (no --dry-run either); refusing "
"to run. See README.md.")
# --- verified-crawler support -------------------------------------------
# Vendors publishing machine-readable IP ranges. A UA token alone proves
# nothing: on the instance this was written for, ~83% of AI-bot-labelled
# traffic came from hosts outside these ranges. Only IP-verified hits count.
RANGE_SOURCES = {
"openai": ["https://openai.com/chatgpt-user.json",
"https://openai.com/gptbot.json",
"https://openai.com/searchbot.json"],
"anthropic": ["https://claude.com/crawling/bots.json"],
"perplexity":["https://www.perplexity.ai/perplexitybot.json"],
"apple": ["https://search.developer.apple.com/applebot.json"],
"google": ["https://developers.google.com/static/crawling/ipranges/special-crawlers.json",
"https://developers.google.com/static/crawling/ipranges/common-crawlers.json",
"https://developers.google.com/static/crawling/ipranges/user-triggered-fetchers.json",
"https://developers.google.com/static/crawling/ipranges/user-triggered-fetchers-google.json"],
}
# UA token -> vendor whose ranges must contain the source IP
BOT_VENDOR = {
"ChatGPT-User": "openai", "GPTBot": "openai", "OAI-SearchBot": "openai",
"ClaudeBot": "anthropic", "Claude-User": "anthropic", "Claude-SearchBot": "anthropic",
"PerplexityBot": "perplexity", "Perplexity-User": "perplexity",
"Applebot-Extended": "apple", "Applebot": "apple",
"Googlebot": "google", "Google-Extended": "google", "GoogleOther": "google",
}
# --- reverse-DNS verification -------------------------------------------
# For vendors that publish no IP ranges. The check is forward-confirmed
# (FCrDNS): PTR of the address must end in one of the vendor's domains,
# AND that hostname must resolve back to the same address. The forward
# step is what makes it meaningful - a PTR record alone is set by
# whoever controls the address block, so it proves nothing on its own.
BOT_RDNS = {
"Amazonbot": (".crawl.amazonbot.amazon",),
"meta-externalagent": (".facebook.com", ".fbsv.net"),
"meta-externalfetcher": (".facebook.com", ".fbsv.net"),
"FacebookBot": (".facebook.com", ".fbsv.net"),
"YouBot": (".search.you.com",),
"Bytespider": (".bytedance.com", ".byteoversea.com"),
"PetalBot": (".petalsearch.com", ".aspiegel.com"),
"DuckAssistBot": (".duckduckgo.com",),
# Apple and Google publish ranges, but also answer rDNS; listed so a
# range miss can still be resolved when the published file is stale.
"Applebot": (".applebot.apple.com",),
"Googlebot": (".googlebot.com", ".google.com"),
}
# rDNS is slow (two lookups per address), so only unique IPs are probed
# and results are memoised for the run.
RDNS_MAX = int(os.environ.get("CRAWLER_RDNS_MAX", "400"))
RDNS_ENABLE = os.environ.get("CRAWLER_RDNS", "1") != "0"
CACHE = os.environ.get("CRAWLER_CACHE", "/var/cache/crawler-alert/ranges.json")
CF_URL = "https://www.cloudflare.com/ips-v4"
# claude.com and cloudflare.com sit behind Cloudflare, which 403s the
# default Python-urllib UA. Send a real one.
UA = "Mozilla/5.0 (compatible; crawler-alert/1.0; +https://codeberg.org/)"
def fetch(url, timeout=20):
req = urllib.request.Request(url, headers={"User-Agent": UA})
with urllib.request.urlopen(req, timeout=timeout) as r:
return r.read()
def load_ranges():
"""Fetch vendor prefixes; fall back to cache so a vendor outage
degrades the report instead of breaking the mail."""
out, stale = {}, []
for vendor, urls in RANGE_SOURCES.items():
nets = []
for u in urls:
try:
for p in json.loads(fetch(u)).get("prefixes", []):
for k, v in p.items():
if k.startswith("ipv"):
nets.append(v)
except Exception:
stale.append(vendor)
if nets:
out[vendor] = nets
cf = []
try:
cf = [l.strip() for l in fetch(CF_URL).decode().splitlines() if l.strip()]
except Exception:
stale.append("cloudflare")
try:
cached = json.load(open(CACHE))
except Exception:
cached = {}
for vendor in RANGE_SOURCES:
if vendor not in out and vendor in cached.get("vendors", {}):
out[vendor] = cached["vendors"][vendor]
if not cf:
cf = cached.get("cloudflare", [])
if out:
try:
os.makedirs(os.path.dirname(CACHE), exist_ok=True)
json.dump({"vendors": out, "cloudflare": cf}, open(CACHE, "w"))
except Exception:
pass
# bucket by first octet so lookups stay cheap over ~1425 prefixes
idx = {}
for vendor, nets in out.items():
b = {}
for s in nets:
n = ipaddress.ip_network(s)
b.setdefault(int(n.network_address) >> 24 if n.version == 4 else -1, []).append(n)
idx[vendor] = b
cfnets = [ipaddress.ip_network(s) for s in cf]
return idx, cfnets, sorted(set(stale))
def in_index(idx_v, addr):
if idx_v is None:
return False
key = int(addr) >> 24 if addr.version == 4 else -1
return any(addr in n for n in idx_v.get(key, []))
_rdns_cache = {}
def fcrdns(ip, suffixes):
"""Forward-confirmed reverse DNS.
PTR must end in one of `suffixes`, and the name it gives must resolve
back to `ip`. Returns (ok, hostname). Without the forward step this
would only prove the address owner can write their own PTR record.
"""
key = (ip, suffixes)
if key in _rdns_cache:
return _rdns_cache[key]
result = (False, "")
try:
host = socket.gethostbyaddr(ip)[0].rstrip(".").lower()
if any(host.endswith(s) or host == s.lstrip(".") for s in suffixes):
_, _, addrs = socket.gethostbyname_ex(host)
try:
v6 = socket.getaddrinfo(host, None, socket.AF_INET6)
addrs = addrs + [a[4][0] for a in v6]
except OSError:
pass
if ip in addrs:
result = (True, host)
else:
result = (False, host + " (no forward match)")
else:
result = (False, host)
except OSError:
result = (False, "")
_rdns_cache[key] = result
return result
days = [date.today() - timedelta(days=i) for i in range(7 if WEEKLY else 1, 0, -1)]
stamps = [d.strftime("%d/%b/%Y") for d in days]
RANGES, CFNETS, STALE = (load_ranges() if WEEKLY else ({}, [], []))
per_day = {s: [0, set()] for s in stamps}
repos = Counter()
uas = Counter()
verified = Counter() # bot -> IP-verified hits
spoofed = Counter() # bot -> UA claimed, IP outside vendor ranges
proxied = Counter() # bot -> arrived via Cloudflare, unverifiable
ver_paths = Counter() # paths fetched by verified bots
ver_ips = {} # bot -> set of verified source IPs
# rDNS-verified vendors: {bot: {ip: hits}}, resolved after the log pass
rdns_hits = {b: Counter() for b in BOT_RDNS}
rdns_proxied = Counter()
for path in sorted(glob.glob(LOG_GLOB)):
opener = gzip.open if path.endswith(".gz") else open
with opener(path, "rt", errors="replace") as f:
for line in f:
if ROUTER and ROUTER not in line:
continue
for s in stamps:
if s in line:
per_day[s][0] += 1
per_day[s][1].add(line.split(" ", 1)[0])
parts = line.split()
if len(parts) > 6 and parts[6].count("/") >= 2:
repos["/".join(parts[6].split("/")[1:3])] += 1
# Traefik CLF: ... "<referer>" "<user-agent>" <n> "<router>" ...
# Anchor on the request counter so this works whatever
# the router is called.
ua = re.search(r"\"[^\"]*\" \"([^\"]*)\" \d+ \"", line)
if ua:
u = ua.group(1)
m = re.search(r"(bot|crawler|spider|scrapy|externalagent|gpt|claude|perplexity)[\w./-]*", u, re.I)
uas[m.group(0) if m else ("(none)" if u == "-" else "(browser-like)")] += 1
if WEEKLY:
# longest token first so Applebot-Extended wins over Applebot
bot = next((b for b in sorted(BOT_VENDOR, key=len, reverse=True) if b in u), None)
if bot:
try:
addr = ipaddress.ip_address(parts[0])
except ValueError:
addr = None
if addr is None:
pass
elif any(addr in n for n in CFNETS):
proxied[bot] += 1
elif in_index(RANGES.get(BOT_VENDOR[bot]), addr):
verified[bot] += 1
ver_ips.setdefault(bot, set()).add(parts[0])
if len(parts) > 6:
ver_paths[parts[6][:60]] += 1
else:
spoofed[bot] += 1
if WEEKLY and RDNS_ENABLE:
rb = next((b for b in sorted(BOT_RDNS, key=len, reverse=True)
if b in u), None)
# skip if already counted by IP-range check
if rb and not (rb in BOT_VENDOR and rb == bot):
try:
a2 = ipaddress.ip_address(parts[0])
except ValueError:
a2 = None
if a2 is None:
pass
elif any(a2 in n for n in CFNETS):
rdns_proxied[rb] += 1
else:
rdns_hits[rb][parts[0]] += 1
break
total = sum(v[0] for v in per_day.values())
all_ips = set().union(*(v[1] for v in per_day.values()))
# --- resolve rDNS candidates ---------------------------------------------
# One pass over unique addresses, busiest first, capped by RDNS_MAX so a
# flood of distinct forgeries cannot stall the report on DNS timeouts.
rdns_ok = Counter() # bot -> hits from forward-confirmed addresses
rdns_bad = Counter() # bot -> hits that failed confirmation
rdns_skipped = Counter() # bot -> hits left unresolved by the cap
rdns_private = Counter() # bot -> hits from RFC1918/loopback (see below)
rdns_ok_ips = {}
if WEEKLY and RDNS_ENABLE:
socket.setdefaulttimeout(3)
for bot, counter in rdns_hits.items():
budget = RDNS_MAX
for ip, hits in counter.most_common():
# A private source address means the real client IP never
# reached the log: something in front (Docker's bridge
# gateway, a local proxy) is rewriting it. Neither verified
# nor spoofed - the evidence simply is not there.
if ipaddress.ip_address(ip).is_private:
rdns_private[bot] += hits
continue
if budget <= 0:
rdns_skipped[bot] += hits
continue
budget -= 1
ok, _host = fcrdns(ip, BOT_RDNS[bot])
if ok:
rdns_ok[bot] += hits
rdns_ok_ips.setdefault(bot, set()).add(ip)
else:
rdns_bad[bot] += hits
socket.setdefaulttimeout(None)
if not WEEKLY:
s0 = stamps[0]
count, ips = per_day[s0][0], len(per_day[s0][1])
print(f"{s0}: {count} requests, {ips} unique IPs (threshold {THRESHOLD})")
if DRY or count <= THRESHOLD:
sys.exit(0)
subject = f"{SITE}: {count} requests on {s0} (over {THRESHOLD}/day threshold)"
body = (f"Forgejo served {count} requests from {ips} unique IPs on {s0} - "
f"above the {THRESHOLD}/day watch threshold.\n\n"
"Time to consider Anubis (PoW challenge) in front of Forgejo.\n")
else:
lines = [f"{SITE} weekly crawler report ({stamps[0]} - {stamps[-1]})", ""]
lines.append(f"Total: {total} requests, {len(all_ips)} unique IPs")
lines.append(f"Daily threshold alert fires above {THRESHOLD} req/day (none = quiet week)")
lines.append("")
lines.append("Per day:")
for s in stamps:
lines.append(f" {s}: {per_day[s][0]:>7} requests, {len(per_day[s][1]):>6} unique IPs")
lines.append("")
lines.append("Top repos:")
for r, c in repos.most_common(5):
lines.append(f" {c:>7} {r}")
lines.append("")
lines.append("User-agent classes (browser-like = mostly the spoofing swarm):")
for u, c in uas.most_common(8):
lines.append(f" {c:>7} {u}")
lines.append("")
# --- legitimate (IP-verified) crawlers -------------------------------
lines.append("=" * 62)
lines.append("LEGITIMATE CRAWLERS (source IP inside vendor's published ranges)")
lines.append("=" * 62)
if STALE:
lines.append(f"NOTE: range fetch failed for {', '.join(STALE)} - used cache")
tv, ts, tp = sum(verified.values()), sum(spoofed.values()), sum(proxied.values())
claimed = tv + ts + tp
if not claimed:
lines.append(" No AI-bot user-agents seen this week.")
else:
lines.append(f" {'bot':<20}{'verified':>9}{'spoofed':>9}{'via CF':>8} {'IPs':>4}")
for bot in sorted(set(verified) | set(spoofed) | set(proxied),
key=lambda b: -(verified[b] + spoofed[b] + proxied[b])):
lines.append(f" {bot:<20}{verified[bot]:>9}{spoofed[bot]:>9}"
f"{proxied[bot]:>8} {len(ver_ips.get(bot, ())):>4}")
pct = tv / claimed * 100
lines.append("")
lines.append(f" {claimed} requests claimed an AI-bot identity; {tv} verified ({pct:.1f}%).")
lines.append(f" {ts} failed IP verification (spoofed). {tp} arrived via Cloudflare")
lines.append(" and cannot be verified by IP - not counted either way.")
if ver_paths:
lines.append("")
lines.append(" Top paths fetched by verified crawlers:")
for p, c in ver_paths.most_common(8):
lines.append(f" {c:>6} {p}")
# --- vendors verified by reverse DNS ---------------------------------
if RDNS_ENABLE and (rdns_ok or rdns_bad or rdns_proxied):
lines.append("")
lines.append("-" * 62)
lines.append("VERIFIED BY REVERSE DNS (vendors publishing no IP ranges)")
lines.append("-" * 62)
lines.append(f" {'bot':<22}{'confirmed':>10}{'failed':>8}{'via CF':>8}"
f"{'private':>8} {'IPs':>4}")
for bot in sorted(set(rdns_ok) | set(rdns_bad) | set(rdns_proxied) | set(rdns_private),
key=lambda b: -(rdns_ok[b] + rdns_bad[b] + rdns_proxied[b]
+ rdns_private[b])):
lines.append(f" {bot:<22}{rdns_ok[bot]:>10}{rdns_bad[bot]:>8}"
f"{rdns_proxied[bot]:>8}{rdns_private[bot]:>8}"
f" {len(rdns_ok_ips.get(bot, ())):>4}")
if rdns_skipped:
sk = ", ".join(f"{b} {c}" for b, c in rdns_skipped.most_common(4))
lines.append(f" (unresolved, over CRAWLER_RDNS_MAX={RDNS_MAX}: {sk})")
lines.append("")
lines.append(" Forward-confirmed: the address's PTR record ends in a vendor")
lines.append(" domain AND that hostname resolves back to the same address.")
lines.append(" 'failed' means the PTR was absent, pointed elsewhere, or did")
lines.append(" not confirm - i.e. the user-agent is unsupported by DNS.")
if sum(rdns_private.values()):
lines.append(" 'private' means the logged source was an RFC1918 address:")
lines.append(" a proxy or Docker bridge is masking the real client, so")
lines.append(" these cannot be judged either way. See README.")
body = "\n".join(lines) + "\n"
subject = f"{SITE} weekly crawler report: {total} requests, {len(all_ips)} IPs"
if DRY:
print(subject); print(); print(body)
sys.exit(0)
# SMTP settings: use CRAWLER_SMTP_* if given, otherwise borrow them from
# a Forgejo/Gitea app.ini [mailer] section so there is no second copy of
# the credentials to keep in sync.
if os.environ.get("CRAWLER_SMTP_ADDR"):
addr = os.environ["CRAWLER_SMTP_ADDR"]
port = int(os.environ.get("CRAWLER_SMTP_PORT", "587"))
user = os.environ.get("CRAWLER_SMTP_USER", "")
passwd = os.environ.get("CRAWLER_SMTP_PASSWORD", "")
sender = os.environ.get("CRAWLER_SMTP_FROM", user or RECIPIENT)
proto = os.environ.get("CRAWLER_SMTP_PROTOCOL", "smtp+starttls")
else:
ini = subprocess.run(["docker", "exec", MAIL_CONTAINER, "cat", MAIL_INI],
capture_output=True, text=True).stdout
if not ini.strip():
sys.exit(f"could not read {MAIL_INI} from container "
f"'{MAIL_CONTAINER}' and CRAWLER_SMTP_ADDR is unset")
cp = configparser.ConfigParser(interpolation=None, strict=False)
cp.read_string("[DEFAULT]\n" + ini)
m = cp["mailer"]
proto = m.get("PROTOCOL", "smtps").strip()
addr, port = m.get("SMTP_ADDR").strip(), int(m.get("SMTP_PORT", "465").strip())
user, passwd = m.get("USER", "").strip(), m.get("PASSWD", "").strip().strip("`\"")
sender = m.get("FROM", user).strip()
msg = MIMEText(body)
msg["Subject"], msg["From"], msg["To"] = subject, sender, RECIPIENT
if proto == "smtps":
s = smtplib.SMTP_SSL(addr, port, timeout=30)
else:
s = smtplib.SMTP(addr, port, timeout=30)
if proto != "smtp": # plain smtp = no TLS (local relay)
s.starttls()
if user: # unauthenticated local relays exist
s.login(user, passwd)
s.sendmail(sender, [RECIPIENT], msg.as_string())
s.quit()
print(f"mailed to {RECIPIENT}")