Amazonbot, meta-externalagent, YouBot, Bytespider and PetalBot publish no IP range file, so they were counted in the user-agent tally and nowhere else - the largest single source of traffic on the sample instance was also the least examined. Check them by FCrDNS: the PTR record must end in a vendor domain and that hostname must resolve back to the same address. The forward step is the part that matters. A PTR record alone is written by whoever controls the address block, so without confirming it forward the check would accept anything its owner chose to claim. ASN verification was considered and deliberately left out. Genuine YouBot and genuine Amazonbot both live in AS14618, which is also every EC2 instance a spoofer could rent. An ASN match shows the traffic came from a cloud, not from the vendor; presenting that as verification would be worse than presenting nothing. Report RFC1918 sources separately rather than as failures. On the sample instance 21624 of meta-externalagent's 21651 requests came from 172.18.0.1, the Docker bridge gateway: the real client address was replaced before it reached the log. Those are not spoofed, they are unjudgeable, and calling them spoofed would be a false accusation caused by the reader's own proxy configuration. Resolution is capped at CRAWLER_RDNS_MAX unique addresses per bot (busiest first) so a flood of distinct forgeries cannot stall the weekly mail on DNS timeouts, and can be disabled with CRAWLER_RDNS=0. Assisted-by: Claude:opus-5
423 lines
20 KiB
Python
Executable file
423 lines
20 KiB
Python
Executable file
#!/usr/bin/env python3
|
|
"""Crawler watch for a Traefik-fronted Forgejo (or any Traefik site).
|
|
|
|
Separates real AI crawlers from traffic that only claims to be, by
|
|
checking the source IP against each vendor's published range list.
|
|
|
|
Modes:
|
|
(default) daily threshold alert: mail only if yesterday exceeded
|
|
CRAWLER_THRESHOLD requests
|
|
--weekly summary: always mail stats for the past 7 days
|
|
--dry-run print instead of mailing (combinable with --weekly)
|
|
|
|
Configuration is by environment variable; see README.md. The only
|
|
value with no sensible default is CRAWLER_RECIPIENT.
|
|
"""
|
|
import configparser, glob, gzip, ipaddress, json, os, re, smtplib, socket, subprocess, sys, urllib.request
|
|
from collections import Counter
|
|
from datetime import date, timedelta
|
|
from email.mime.text import MIMEText
|
|
|
|
# --- configuration ------------------------------------------------------
|
|
THRESHOLD = int(os.environ.get("CRAWLER_THRESHOLD", "100000"))
|
|
RECIPIENT = os.environ.get("CRAWLER_RECIPIENT", "")
|
|
SITE = os.environ.get("CRAWLER_SITE", "this site")
|
|
LOG_GLOB = os.environ.get("CRAWLER_LOG", "/var/log/traefik/access.log*")
|
|
# Traefik router name to filter on, e.g. "forgejo@docker". Empty = all
|
|
# traffic reaching Traefik, which is what you want for a single-site host.
|
|
ROUTER = os.environ.get("CRAWLER_ROUTER", "")
|
|
# Container to read SMTP settings from (Forgejo/Gitea app.ini [mailer]).
|
|
# Set CRAWLER_SMTP_* instead to configure SMTP directly.
|
|
MAIL_CONTAINER = os.environ.get("CRAWLER_MAIL_CONTAINER", "forgejo")
|
|
MAIL_INI = os.environ.get("CRAWLER_MAIL_INI", "/data/gitea/conf/app.ini")
|
|
|
|
WEEKLY = "--weekly" in sys.argv
|
|
DRY = "--dry-run" in sys.argv
|
|
|
|
if not RECIPIENT and not DRY:
|
|
sys.exit("CRAWLER_RECIPIENT is not set (no --dry-run either); refusing "
|
|
"to run. See README.md.")
|
|
|
|
# --- verified-crawler support -------------------------------------------
|
|
# Vendors publishing machine-readable IP ranges. A UA token alone proves
|
|
# nothing: on the instance this was written for, ~83% of AI-bot-labelled
|
|
# traffic came from hosts outside these ranges. Only IP-verified hits count.
|
|
RANGE_SOURCES = {
|
|
"openai": ["https://openai.com/chatgpt-user.json",
|
|
"https://openai.com/gptbot.json",
|
|
"https://openai.com/searchbot.json"],
|
|
"anthropic": ["https://claude.com/crawling/bots.json"],
|
|
"perplexity":["https://www.perplexity.ai/perplexitybot.json"],
|
|
"apple": ["https://search.developer.apple.com/applebot.json"],
|
|
"google": ["https://developers.google.com/static/crawling/ipranges/special-crawlers.json",
|
|
"https://developers.google.com/static/crawling/ipranges/common-crawlers.json",
|
|
"https://developers.google.com/static/crawling/ipranges/user-triggered-fetchers.json",
|
|
"https://developers.google.com/static/crawling/ipranges/user-triggered-fetchers-google.json"],
|
|
}
|
|
# UA token -> vendor whose ranges must contain the source IP
|
|
BOT_VENDOR = {
|
|
"ChatGPT-User": "openai", "GPTBot": "openai", "OAI-SearchBot": "openai",
|
|
"ClaudeBot": "anthropic", "Claude-User": "anthropic", "Claude-SearchBot": "anthropic",
|
|
"PerplexityBot": "perplexity", "Perplexity-User": "perplexity",
|
|
"Applebot-Extended": "apple", "Applebot": "apple",
|
|
"Googlebot": "google", "Google-Extended": "google", "GoogleOther": "google",
|
|
}
|
|
|
|
# --- reverse-DNS verification -------------------------------------------
|
|
# For vendors that publish no IP ranges. The check is forward-confirmed
|
|
# (FCrDNS): PTR of the address must end in one of the vendor's domains,
|
|
# AND that hostname must resolve back to the same address. The forward
|
|
# step is what makes it meaningful - a PTR record alone is set by
|
|
# whoever controls the address block, so it proves nothing on its own.
|
|
BOT_RDNS = {
|
|
"Amazonbot": (".crawl.amazonbot.amazon",),
|
|
"meta-externalagent": (".facebook.com", ".fbsv.net"),
|
|
"meta-externalfetcher": (".facebook.com", ".fbsv.net"),
|
|
"FacebookBot": (".facebook.com", ".fbsv.net"),
|
|
"YouBot": (".search.you.com",),
|
|
"Bytespider": (".bytedance.com", ".byteoversea.com"),
|
|
"PetalBot": (".petalsearch.com", ".aspiegel.com"),
|
|
"DuckAssistBot": (".duckduckgo.com",),
|
|
# Apple and Google publish ranges, but also answer rDNS; listed so a
|
|
# range miss can still be resolved when the published file is stale.
|
|
"Applebot": (".applebot.apple.com",),
|
|
"Googlebot": (".googlebot.com", ".google.com"),
|
|
}
|
|
# rDNS is slow (two lookups per address), so only unique IPs are probed
|
|
# and results are memoised for the run.
|
|
RDNS_MAX = int(os.environ.get("CRAWLER_RDNS_MAX", "400"))
|
|
RDNS_ENABLE = os.environ.get("CRAWLER_RDNS", "1") != "0"
|
|
CACHE = os.environ.get("CRAWLER_CACHE", "/var/cache/crawler-alert/ranges.json")
|
|
CF_URL = "https://www.cloudflare.com/ips-v4"
|
|
# claude.com and cloudflare.com sit behind Cloudflare, which 403s the
|
|
# default Python-urllib UA. Send a real one.
|
|
UA = "Mozilla/5.0 (compatible; crawler-alert/1.0; +https://codeberg.org/)"
|
|
|
|
|
|
def fetch(url, timeout=20):
|
|
req = urllib.request.Request(url, headers={"User-Agent": UA})
|
|
with urllib.request.urlopen(req, timeout=timeout) as r:
|
|
return r.read()
|
|
|
|
|
|
def load_ranges():
|
|
"""Fetch vendor prefixes; fall back to cache so a vendor outage
|
|
degrades the report instead of breaking the mail."""
|
|
out, stale = {}, []
|
|
for vendor, urls in RANGE_SOURCES.items():
|
|
nets = []
|
|
for u in urls:
|
|
try:
|
|
for p in json.loads(fetch(u)).get("prefixes", []):
|
|
for k, v in p.items():
|
|
if k.startswith("ipv"):
|
|
nets.append(v)
|
|
except Exception:
|
|
stale.append(vendor)
|
|
if nets:
|
|
out[vendor] = nets
|
|
cf = []
|
|
try:
|
|
cf = [l.strip() for l in fetch(CF_URL).decode().splitlines() if l.strip()]
|
|
except Exception:
|
|
stale.append("cloudflare")
|
|
try:
|
|
cached = json.load(open(CACHE))
|
|
except Exception:
|
|
cached = {}
|
|
for vendor in RANGE_SOURCES:
|
|
if vendor not in out and vendor in cached.get("vendors", {}):
|
|
out[vendor] = cached["vendors"][vendor]
|
|
if not cf:
|
|
cf = cached.get("cloudflare", [])
|
|
if out:
|
|
try:
|
|
os.makedirs(os.path.dirname(CACHE), exist_ok=True)
|
|
json.dump({"vendors": out, "cloudflare": cf}, open(CACHE, "w"))
|
|
except Exception:
|
|
pass
|
|
# bucket by first octet so lookups stay cheap over ~1425 prefixes
|
|
idx = {}
|
|
for vendor, nets in out.items():
|
|
b = {}
|
|
for s in nets:
|
|
n = ipaddress.ip_network(s)
|
|
b.setdefault(int(n.network_address) >> 24 if n.version == 4 else -1, []).append(n)
|
|
idx[vendor] = b
|
|
cfnets = [ipaddress.ip_network(s) for s in cf]
|
|
return idx, cfnets, sorted(set(stale))
|
|
|
|
|
|
def in_index(idx_v, addr):
|
|
if idx_v is None:
|
|
return False
|
|
key = int(addr) >> 24 if addr.version == 4 else -1
|
|
return any(addr in n for n in idx_v.get(key, []))
|
|
|
|
|
|
_rdns_cache = {}
|
|
|
|
|
|
def fcrdns(ip, suffixes):
|
|
"""Forward-confirmed reverse DNS.
|
|
|
|
PTR must end in one of `suffixes`, and the name it gives must resolve
|
|
back to `ip`. Returns (ok, hostname). Without the forward step this
|
|
would only prove the address owner can write their own PTR record.
|
|
"""
|
|
key = (ip, suffixes)
|
|
if key in _rdns_cache:
|
|
return _rdns_cache[key]
|
|
result = (False, "")
|
|
try:
|
|
host = socket.gethostbyaddr(ip)[0].rstrip(".").lower()
|
|
if any(host.endswith(s) or host == s.lstrip(".") for s in suffixes):
|
|
_, _, addrs = socket.gethostbyname_ex(host)
|
|
try:
|
|
v6 = socket.getaddrinfo(host, None, socket.AF_INET6)
|
|
addrs = addrs + [a[4][0] for a in v6]
|
|
except OSError:
|
|
pass
|
|
if ip in addrs:
|
|
result = (True, host)
|
|
else:
|
|
result = (False, host + " (no forward match)")
|
|
else:
|
|
result = (False, host)
|
|
except OSError:
|
|
result = (False, "")
|
|
_rdns_cache[key] = result
|
|
return result
|
|
|
|
days = [date.today() - timedelta(days=i) for i in range(7 if WEEKLY else 1, 0, -1)]
|
|
stamps = [d.strftime("%d/%b/%Y") for d in days]
|
|
|
|
RANGES, CFNETS, STALE = (load_ranges() if WEEKLY else ({}, [], []))
|
|
|
|
per_day = {s: [0, set()] for s in stamps}
|
|
repos = Counter()
|
|
uas = Counter()
|
|
verified = Counter() # bot -> IP-verified hits
|
|
spoofed = Counter() # bot -> UA claimed, IP outside vendor ranges
|
|
proxied = Counter() # bot -> arrived via Cloudflare, unverifiable
|
|
ver_paths = Counter() # paths fetched by verified bots
|
|
ver_ips = {} # bot -> set of verified source IPs
|
|
# rDNS-verified vendors: {bot: {ip: hits}}, resolved after the log pass
|
|
rdns_hits = {b: Counter() for b in BOT_RDNS}
|
|
rdns_proxied = Counter()
|
|
for path in sorted(glob.glob(LOG_GLOB)):
|
|
opener = gzip.open if path.endswith(".gz") else open
|
|
with opener(path, "rt", errors="replace") as f:
|
|
for line in f:
|
|
if ROUTER and ROUTER not in line:
|
|
continue
|
|
for s in stamps:
|
|
if s in line:
|
|
per_day[s][0] += 1
|
|
per_day[s][1].add(line.split(" ", 1)[0])
|
|
parts = line.split()
|
|
if len(parts) > 6 and parts[6].count("/") >= 2:
|
|
repos["/".join(parts[6].split("/")[1:3])] += 1
|
|
# Traefik CLF: ... "<referer>" "<user-agent>" <n> "<router>" ...
|
|
# Anchor on the request counter so this works whatever
|
|
# the router is called.
|
|
ua = re.search(r"\"[^\"]*\" \"([^\"]*)\" \d+ \"", line)
|
|
if ua:
|
|
u = ua.group(1)
|
|
m = re.search(r"(bot|crawler|spider|scrapy|externalagent|gpt|claude|perplexity)[\w./-]*", u, re.I)
|
|
uas[m.group(0) if m else ("(none)" if u == "-" else "(browser-like)")] += 1
|
|
if WEEKLY:
|
|
# longest token first so Applebot-Extended wins over Applebot
|
|
bot = next((b for b in sorted(BOT_VENDOR, key=len, reverse=True) if b in u), None)
|
|
if bot:
|
|
try:
|
|
addr = ipaddress.ip_address(parts[0])
|
|
except ValueError:
|
|
addr = None
|
|
if addr is None:
|
|
pass
|
|
elif any(addr in n for n in CFNETS):
|
|
proxied[bot] += 1
|
|
elif in_index(RANGES.get(BOT_VENDOR[bot]), addr):
|
|
verified[bot] += 1
|
|
ver_ips.setdefault(bot, set()).add(parts[0])
|
|
if len(parts) > 6:
|
|
ver_paths[parts[6][:60]] += 1
|
|
else:
|
|
spoofed[bot] += 1
|
|
if WEEKLY and RDNS_ENABLE:
|
|
rb = next((b for b in sorted(BOT_RDNS, key=len, reverse=True)
|
|
if b in u), None)
|
|
# skip if already counted by IP-range check
|
|
if rb and not (rb in BOT_VENDOR and rb == bot):
|
|
try:
|
|
a2 = ipaddress.ip_address(parts[0])
|
|
except ValueError:
|
|
a2 = None
|
|
if a2 is None:
|
|
pass
|
|
elif any(a2 in n for n in CFNETS):
|
|
rdns_proxied[rb] += 1
|
|
else:
|
|
rdns_hits[rb][parts[0]] += 1
|
|
break
|
|
|
|
total = sum(v[0] for v in per_day.values())
|
|
all_ips = set().union(*(v[1] for v in per_day.values()))
|
|
|
|
# --- resolve rDNS candidates ---------------------------------------------
|
|
# One pass over unique addresses, busiest first, capped by RDNS_MAX so a
|
|
# flood of distinct forgeries cannot stall the report on DNS timeouts.
|
|
rdns_ok = Counter() # bot -> hits from forward-confirmed addresses
|
|
rdns_bad = Counter() # bot -> hits that failed confirmation
|
|
rdns_skipped = Counter() # bot -> hits left unresolved by the cap
|
|
rdns_private = Counter() # bot -> hits from RFC1918/loopback (see below)
|
|
rdns_ok_ips = {}
|
|
if WEEKLY and RDNS_ENABLE:
|
|
socket.setdefaulttimeout(3)
|
|
for bot, counter in rdns_hits.items():
|
|
budget = RDNS_MAX
|
|
for ip, hits in counter.most_common():
|
|
# A private source address means the real client IP never
|
|
# reached the log: something in front (Docker's bridge
|
|
# gateway, a local proxy) is rewriting it. Neither verified
|
|
# nor spoofed - the evidence simply is not there.
|
|
if ipaddress.ip_address(ip).is_private:
|
|
rdns_private[bot] += hits
|
|
continue
|
|
if budget <= 0:
|
|
rdns_skipped[bot] += hits
|
|
continue
|
|
budget -= 1
|
|
ok, _host = fcrdns(ip, BOT_RDNS[bot])
|
|
if ok:
|
|
rdns_ok[bot] += hits
|
|
rdns_ok_ips.setdefault(bot, set()).add(ip)
|
|
else:
|
|
rdns_bad[bot] += hits
|
|
socket.setdefaulttimeout(None)
|
|
|
|
if not WEEKLY:
|
|
s0 = stamps[0]
|
|
count, ips = per_day[s0][0], len(per_day[s0][1])
|
|
print(f"{s0}: {count} requests, {ips} unique IPs (threshold {THRESHOLD})")
|
|
if DRY or count <= THRESHOLD:
|
|
sys.exit(0)
|
|
subject = f"{SITE}: {count} requests on {s0} (over {THRESHOLD}/day threshold)"
|
|
body = (f"Forgejo served {count} requests from {ips} unique IPs on {s0} - "
|
|
f"above the {THRESHOLD}/day watch threshold.\n\n"
|
|
"Time to consider Anubis (PoW challenge) in front of Forgejo.\n")
|
|
else:
|
|
lines = [f"{SITE} weekly crawler report ({stamps[0]} - {stamps[-1]})", ""]
|
|
lines.append(f"Total: {total} requests, {len(all_ips)} unique IPs")
|
|
lines.append(f"Daily threshold alert fires above {THRESHOLD} req/day (none = quiet week)")
|
|
lines.append("")
|
|
lines.append("Per day:")
|
|
for s in stamps:
|
|
lines.append(f" {s}: {per_day[s][0]:>7} requests, {len(per_day[s][1]):>6} unique IPs")
|
|
lines.append("")
|
|
lines.append("Top repos:")
|
|
for r, c in repos.most_common(5):
|
|
lines.append(f" {c:>7} {r}")
|
|
lines.append("")
|
|
lines.append("User-agent classes (browser-like = mostly the spoofing swarm):")
|
|
for u, c in uas.most_common(8):
|
|
lines.append(f" {c:>7} {u}")
|
|
lines.append("")
|
|
|
|
# --- legitimate (IP-verified) crawlers -------------------------------
|
|
lines.append("=" * 62)
|
|
lines.append("LEGITIMATE CRAWLERS (source IP inside vendor's published ranges)")
|
|
lines.append("=" * 62)
|
|
if STALE:
|
|
lines.append(f"NOTE: range fetch failed for {', '.join(STALE)} - used cache")
|
|
tv, ts, tp = sum(verified.values()), sum(spoofed.values()), sum(proxied.values())
|
|
claimed = tv + ts + tp
|
|
if not claimed:
|
|
lines.append(" No AI-bot user-agents seen this week.")
|
|
else:
|
|
lines.append(f" {'bot':<20}{'verified':>9}{'spoofed':>9}{'via CF':>8} {'IPs':>4}")
|
|
for bot in sorted(set(verified) | set(spoofed) | set(proxied),
|
|
key=lambda b: -(verified[b] + spoofed[b] + proxied[b])):
|
|
lines.append(f" {bot:<20}{verified[bot]:>9}{spoofed[bot]:>9}"
|
|
f"{proxied[bot]:>8} {len(ver_ips.get(bot, ())):>4}")
|
|
pct = tv / claimed * 100
|
|
lines.append("")
|
|
lines.append(f" {claimed} requests claimed an AI-bot identity; {tv} verified ({pct:.1f}%).")
|
|
lines.append(f" {ts} failed IP verification (spoofed). {tp} arrived via Cloudflare")
|
|
lines.append(" and cannot be verified by IP - not counted either way.")
|
|
if ver_paths:
|
|
lines.append("")
|
|
lines.append(" Top paths fetched by verified crawlers:")
|
|
for p, c in ver_paths.most_common(8):
|
|
lines.append(f" {c:>6} {p}")
|
|
|
|
# --- vendors verified by reverse DNS ---------------------------------
|
|
if RDNS_ENABLE and (rdns_ok or rdns_bad or rdns_proxied):
|
|
lines.append("")
|
|
lines.append("-" * 62)
|
|
lines.append("VERIFIED BY REVERSE DNS (vendors publishing no IP ranges)")
|
|
lines.append("-" * 62)
|
|
lines.append(f" {'bot':<22}{'confirmed':>10}{'failed':>8}{'via CF':>8}"
|
|
f"{'private':>8} {'IPs':>4}")
|
|
for bot in sorted(set(rdns_ok) | set(rdns_bad) | set(rdns_proxied) | set(rdns_private),
|
|
key=lambda b: -(rdns_ok[b] + rdns_bad[b] + rdns_proxied[b]
|
|
+ rdns_private[b])):
|
|
lines.append(f" {bot:<22}{rdns_ok[bot]:>10}{rdns_bad[bot]:>8}"
|
|
f"{rdns_proxied[bot]:>8}{rdns_private[bot]:>8}"
|
|
f" {len(rdns_ok_ips.get(bot, ())):>4}")
|
|
if rdns_skipped:
|
|
sk = ", ".join(f"{b} {c}" for b, c in rdns_skipped.most_common(4))
|
|
lines.append(f" (unresolved, over CRAWLER_RDNS_MAX={RDNS_MAX}: {sk})")
|
|
lines.append("")
|
|
lines.append(" Forward-confirmed: the address's PTR record ends in a vendor")
|
|
lines.append(" domain AND that hostname resolves back to the same address.")
|
|
lines.append(" 'failed' means the PTR was absent, pointed elsewhere, or did")
|
|
lines.append(" not confirm - i.e. the user-agent is unsupported by DNS.")
|
|
if sum(rdns_private.values()):
|
|
lines.append(" 'private' means the logged source was an RFC1918 address:")
|
|
lines.append(" a proxy or Docker bridge is masking the real client, so")
|
|
lines.append(" these cannot be judged either way. See README.")
|
|
body = "\n".join(lines) + "\n"
|
|
subject = f"{SITE} weekly crawler report: {total} requests, {len(all_ips)} IPs"
|
|
if DRY:
|
|
print(subject); print(); print(body)
|
|
sys.exit(0)
|
|
|
|
# SMTP settings: use CRAWLER_SMTP_* if given, otherwise borrow them from
|
|
# a Forgejo/Gitea app.ini [mailer] section so there is no second copy of
|
|
# the credentials to keep in sync.
|
|
if os.environ.get("CRAWLER_SMTP_ADDR"):
|
|
addr = os.environ["CRAWLER_SMTP_ADDR"]
|
|
port = int(os.environ.get("CRAWLER_SMTP_PORT", "587"))
|
|
user = os.environ.get("CRAWLER_SMTP_USER", "")
|
|
passwd = os.environ.get("CRAWLER_SMTP_PASSWORD", "")
|
|
sender = os.environ.get("CRAWLER_SMTP_FROM", user or RECIPIENT)
|
|
proto = os.environ.get("CRAWLER_SMTP_PROTOCOL", "smtp+starttls")
|
|
else:
|
|
ini = subprocess.run(["docker", "exec", MAIL_CONTAINER, "cat", MAIL_INI],
|
|
capture_output=True, text=True).stdout
|
|
if not ini.strip():
|
|
sys.exit(f"could not read {MAIL_INI} from container "
|
|
f"'{MAIL_CONTAINER}' and CRAWLER_SMTP_ADDR is unset")
|
|
cp = configparser.ConfigParser(interpolation=None, strict=False)
|
|
cp.read_string("[DEFAULT]\n" + ini)
|
|
m = cp["mailer"]
|
|
proto = m.get("PROTOCOL", "smtps").strip()
|
|
addr, port = m.get("SMTP_ADDR").strip(), int(m.get("SMTP_PORT", "465").strip())
|
|
user, passwd = m.get("USER", "").strip(), m.get("PASSWD", "").strip().strip("`\"")
|
|
sender = m.get("FROM", user).strip()
|
|
|
|
msg = MIMEText(body)
|
|
msg["Subject"], msg["From"], msg["To"] = subject, sender, RECIPIENT
|
|
if proto == "smtps":
|
|
s = smtplib.SMTP_SSL(addr, port, timeout=30)
|
|
else:
|
|
s = smtplib.SMTP(addr, port, timeout=30)
|
|
if proto != "smtp": # plain smtp = no TLS (local relay)
|
|
s.starttls()
|
|
if user: # unauthenticated local relays exist
|
|
s.login(user, passwd)
|
|
s.sendmail(sender, [RECIPIENT], msg.as_string())
|
|
s.quit()
|
|
print(f"mailed to {RECIPIENT}")
|