traefik-crawlers-statistics/crawler-alert.py
Sergei Poljanski 5860cbee9a
Make the tool reusable on other instances
Everything was hardcoded for one deployment: hostname in mail subjects,
recipient address, log path, and a router name of "forgejo@docker" that
also appeared inside the user-agent regex. That last one fails silently
rather than loudly - on any instance whose Traefik router is named
something else, the regex matches nothing, every user-agent is dropped,
and the report claims zero bot traffic instead of erroring.

Move configuration to environment variables read from
/etc/crawler-alert.env, with defaults that suit a plain Traefik host.
Only CRAWLER_RECIPIENT is required, and the script now refuses to run
without it rather than mailing into the void. Anchor the user-agent
regex on the request counter instead of the router name.

SMTP can now be configured directly via CRAWLER_SMTP_*, so the tool no
longer requires Forgejo in Docker; borrowing credentials from app.ini
stays the default since it avoids a second copy of the password.
Unauthenticated and non-TLS local relays are handled.

Document the Traefik access log configuration in
traefik-accesslog.yml. This is the one real prerequisite: Traefik drops
all headers by default, so without an explicit User-Agent: keep there
is nothing to analyse. Includes the field layout the parser expects,
logrotate config, and the forwardedHeaders setup needed to recover real
client IPs from behind Cloudflare.

Rewrite README for someone arriving without context, and add a 0BSD
LICENSE so the code can actually be reused.

Assisted-by: Claude:opus-5
2026-08-11 02:49:51 +04:00

286 lines
13 KiB
Python
Executable file

#!/usr/bin/env python3
"""Crawler watch for a Traefik-fronted Forgejo (or any Traefik site).
Separates real AI crawlers from traffic that only claims to be, by
checking the source IP against each vendor's published range list.
Modes:
(default) daily threshold alert: mail only if yesterday exceeded
CRAWLER_THRESHOLD requests
--weekly summary: always mail stats for the past 7 days
--dry-run print instead of mailing (combinable with --weekly)
Configuration is by environment variable; see README.md. The only
value with no sensible default is CRAWLER_RECIPIENT.
"""
import configparser, glob, gzip, ipaddress, json, os, re, smtplib, subprocess, sys, urllib.request
from collections import Counter
from datetime import date, timedelta
from email.mime.text import MIMEText
# --- configuration ------------------------------------------------------
THRESHOLD = int(os.environ.get("CRAWLER_THRESHOLD", "100000"))
RECIPIENT = os.environ.get("CRAWLER_RECIPIENT", "")
SITE = os.environ.get("CRAWLER_SITE", "this site")
LOG_GLOB = os.environ.get("CRAWLER_LOG", "/var/log/traefik/access.log*")
# Traefik router name to filter on, e.g. "forgejo@docker". Empty = all
# traffic reaching Traefik, which is what you want for a single-site host.
ROUTER = os.environ.get("CRAWLER_ROUTER", "")
# Container to read SMTP settings from (Forgejo/Gitea app.ini [mailer]).
# Set CRAWLER_SMTP_* instead to configure SMTP directly.
MAIL_CONTAINER = os.environ.get("CRAWLER_MAIL_CONTAINER", "forgejo")
MAIL_INI = os.environ.get("CRAWLER_MAIL_INI", "/data/gitea/conf/app.ini")
WEEKLY = "--weekly" in sys.argv
DRY = "--dry-run" in sys.argv
if not RECIPIENT and not DRY:
sys.exit("CRAWLER_RECIPIENT is not set (no --dry-run either); refusing "
"to run. See README.md.")
# --- verified-crawler support -------------------------------------------
# Vendors publishing machine-readable IP ranges. A UA token alone proves
# nothing: on the instance this was written for, ~83% of AI-bot-labelled
# traffic came from hosts outside these ranges. Only IP-verified hits count.
RANGE_SOURCES = {
"openai": ["https://openai.com/chatgpt-user.json",
"https://openai.com/gptbot.json",
"https://openai.com/searchbot.json"],
"anthropic": ["https://claude.com/crawling/bots.json"],
"perplexity":["https://www.perplexity.ai/perplexitybot.json"],
"apple": ["https://search.developer.apple.com/applebot.json"],
"google": ["https://developers.google.com/static/crawling/ipranges/special-crawlers.json",
"https://developers.google.com/static/crawling/ipranges/common-crawlers.json",
"https://developers.google.com/static/crawling/ipranges/user-triggered-fetchers.json",
"https://developers.google.com/static/crawling/ipranges/user-triggered-fetchers-google.json"],
}
# UA token -> vendor whose ranges must contain the source IP
BOT_VENDOR = {
"ChatGPT-User": "openai", "GPTBot": "openai", "OAI-SearchBot": "openai",
"ClaudeBot": "anthropic", "Claude-User": "anthropic", "Claude-SearchBot": "anthropic",
"PerplexityBot": "perplexity", "Perplexity-User": "perplexity",
"Applebot-Extended": "apple", "Applebot": "apple",
"Googlebot": "google", "Google-Extended": "google", "GoogleOther": "google",
}
CACHE = os.environ.get("CRAWLER_CACHE", "/var/cache/crawler-alert/ranges.json")
CF_URL = "https://www.cloudflare.com/ips-v4"
# claude.com and cloudflare.com sit behind Cloudflare, which 403s the
# default Python-urllib UA. Send a real one.
UA = "Mozilla/5.0 (compatible; crawler-alert/1.0; +https://codeberg.org/)"
def fetch(url, timeout=20):
req = urllib.request.Request(url, headers={"User-Agent": UA})
with urllib.request.urlopen(req, timeout=timeout) as r:
return r.read()
def load_ranges():
"""Fetch vendor prefixes; fall back to cache so a vendor outage
degrades the report instead of breaking the mail."""
out, stale = {}, []
for vendor, urls in RANGE_SOURCES.items():
nets = []
for u in urls:
try:
for p in json.loads(fetch(u)).get("prefixes", []):
for k, v in p.items():
if k.startswith("ipv"):
nets.append(v)
except Exception:
stale.append(vendor)
if nets:
out[vendor] = nets
cf = []
try:
cf = [l.strip() for l in fetch(CF_URL).decode().splitlines() if l.strip()]
except Exception:
stale.append("cloudflare")
try:
cached = json.load(open(CACHE))
except Exception:
cached = {}
for vendor in RANGE_SOURCES:
if vendor not in out and vendor in cached.get("vendors", {}):
out[vendor] = cached["vendors"][vendor]
if not cf:
cf = cached.get("cloudflare", [])
if out:
try:
os.makedirs(os.path.dirname(CACHE), exist_ok=True)
json.dump({"vendors": out, "cloudflare": cf}, open(CACHE, "w"))
except Exception:
pass
# bucket by first octet so lookups stay cheap over ~1425 prefixes
idx = {}
for vendor, nets in out.items():
b = {}
for s in nets:
n = ipaddress.ip_network(s)
b.setdefault(int(n.network_address) >> 24 if n.version == 4 else -1, []).append(n)
idx[vendor] = b
cfnets = [ipaddress.ip_network(s) for s in cf]
return idx, cfnets, sorted(set(stale))
def in_index(idx_v, addr):
if idx_v is None:
return False
key = int(addr) >> 24 if addr.version == 4 else -1
return any(addr in n for n in idx_v.get(key, []))
days = [date.today() - timedelta(days=i) for i in range(7 if WEEKLY else 1, 0, -1)]
stamps = [d.strftime("%d/%b/%Y") for d in days]
RANGES, CFNETS, STALE = (load_ranges() if WEEKLY else ({}, [], []))
per_day = {s: [0, set()] for s in stamps}
repos = Counter()
uas = Counter()
verified = Counter() # bot -> IP-verified hits
spoofed = Counter() # bot -> UA claimed, IP outside vendor ranges
proxied = Counter() # bot -> arrived via Cloudflare, unverifiable
ver_paths = Counter() # paths fetched by verified bots
ver_ips = {} # bot -> set of verified source IPs
for path in sorted(glob.glob(LOG_GLOB)):
opener = gzip.open if path.endswith(".gz") else open
with opener(path, "rt", errors="replace") as f:
for line in f:
if ROUTER and ROUTER not in line:
continue
for s in stamps:
if s in line:
per_day[s][0] += 1
per_day[s][1].add(line.split(" ", 1)[0])
parts = line.split()
if len(parts) > 6 and parts[6].count("/") >= 2:
repos["/".join(parts[6].split("/")[1:3])] += 1
# Traefik CLF: ... "<referer>" "<user-agent>" <n> "<router>" ...
# Anchor on the request counter so this works whatever
# the router is called.
ua = re.search(r"\"[^\"]*\" \"([^\"]*)\" \d+ \"", line)
if ua:
u = ua.group(1)
m = re.search(r"(bot|crawler|spider|scrapy|externalagent|gpt|claude|perplexity)[\w./-]*", u, re.I)
uas[m.group(0) if m else ("(none)" if u == "-" else "(browser-like)")] += 1
if WEEKLY:
# longest token first so Applebot-Extended wins over Applebot
bot = next((b for b in sorted(BOT_VENDOR, key=len, reverse=True) if b in u), None)
if bot:
try:
addr = ipaddress.ip_address(parts[0])
except ValueError:
addr = None
if addr is None:
pass
elif any(addr in n for n in CFNETS):
proxied[bot] += 1
elif in_index(RANGES.get(BOT_VENDOR[bot]), addr):
verified[bot] += 1
ver_ips.setdefault(bot, set()).add(parts[0])
if len(parts) > 6:
ver_paths[parts[6][:60]] += 1
else:
spoofed[bot] += 1
break
total = sum(v[0] for v in per_day.values())
all_ips = set().union(*(v[1] for v in per_day.values()))
if not WEEKLY:
s0 = stamps[0]
count, ips = per_day[s0][0], len(per_day[s0][1])
print(f"{s0}: {count} requests, {ips} unique IPs (threshold {THRESHOLD})")
if DRY or count <= THRESHOLD:
sys.exit(0)
subject = f"{SITE}: {count} requests on {s0} (over {THRESHOLD}/day threshold)"
body = (f"Forgejo served {count} requests from {ips} unique IPs on {s0} - "
f"above the {THRESHOLD}/day watch threshold.\n\n"
"Time to consider Anubis (PoW challenge) in front of Forgejo.\n")
else:
lines = [f"{SITE} weekly crawler report ({stamps[0]} - {stamps[-1]})", ""]
lines.append(f"Total: {total} requests, {len(all_ips)} unique IPs")
lines.append(f"Daily threshold alert fires above {THRESHOLD} req/day (none = quiet week)")
lines.append("")
lines.append("Per day:")
for s in stamps:
lines.append(f" {s}: {per_day[s][0]:>7} requests, {len(per_day[s][1]):>6} unique IPs")
lines.append("")
lines.append("Top repos:")
for r, c in repos.most_common(5):
lines.append(f" {c:>7} {r}")
lines.append("")
lines.append("User-agent classes (browser-like = mostly the spoofing swarm):")
for u, c in uas.most_common(8):
lines.append(f" {c:>7} {u}")
lines.append("")
# --- legitimate (IP-verified) crawlers -------------------------------
lines.append("=" * 62)
lines.append("LEGITIMATE CRAWLERS (source IP inside vendor's published ranges)")
lines.append("=" * 62)
if STALE:
lines.append(f"NOTE: range fetch failed for {', '.join(STALE)} - used cache")
tv, ts, tp = sum(verified.values()), sum(spoofed.values()), sum(proxied.values())
claimed = tv + ts + tp
if not claimed:
lines.append(" No AI-bot user-agents seen this week.")
else:
lines.append(f" {'bot':<20}{'verified':>9}{'spoofed':>9}{'via CF':>8} {'IPs':>4}")
for bot in sorted(set(verified) | set(spoofed) | set(proxied),
key=lambda b: -(verified[b] + spoofed[b] + proxied[b])):
lines.append(f" {bot:<20}{verified[bot]:>9}{spoofed[bot]:>9}"
f"{proxied[bot]:>8} {len(ver_ips.get(bot, ())):>4}")
pct = tv / claimed * 100
lines.append("")
lines.append(f" {claimed} requests claimed an AI-bot identity; {tv} verified ({pct:.1f}%).")
lines.append(f" {ts} failed IP verification (spoofed). {tp} arrived via Cloudflare")
lines.append(" and cannot be verified by IP - not counted either way.")
if ver_paths:
lines.append("")
lines.append(" Top paths fetched by verified crawlers:")
for p, c in ver_paths.most_common(8):
lines.append(f" {c:>6} {p}")
body = "\n".join(lines) + "\n"
subject = f"{SITE} weekly crawler report: {total} requests, {len(all_ips)} IPs"
if DRY:
print(subject); print(); print(body)
sys.exit(0)
# SMTP settings: use CRAWLER_SMTP_* if given, otherwise borrow them from
# a Forgejo/Gitea app.ini [mailer] section so there is no second copy of
# the credentials to keep in sync.
if os.environ.get("CRAWLER_SMTP_ADDR"):
addr = os.environ["CRAWLER_SMTP_ADDR"]
port = int(os.environ.get("CRAWLER_SMTP_PORT", "587"))
user = os.environ.get("CRAWLER_SMTP_USER", "")
passwd = os.environ.get("CRAWLER_SMTP_PASSWORD", "")
sender = os.environ.get("CRAWLER_SMTP_FROM", user or RECIPIENT)
proto = os.environ.get("CRAWLER_SMTP_PROTOCOL", "smtp+starttls")
else:
ini = subprocess.run(["docker", "exec", MAIL_CONTAINER, "cat", MAIL_INI],
capture_output=True, text=True).stdout
if not ini.strip():
sys.exit(f"could not read {MAIL_INI} from container "
f"'{MAIL_CONTAINER}' and CRAWLER_SMTP_ADDR is unset")
cp = configparser.ConfigParser(interpolation=None, strict=False)
cp.read_string("[DEFAULT]\n" + ini)
m = cp["mailer"]
proto = m.get("PROTOCOL", "smtps").strip()
addr, port = m.get("SMTP_ADDR").strip(), int(m.get("SMTP_PORT", "465").strip())
user, passwd = m.get("USER", "").strip(), m.get("PASSWD", "").strip().strip("`\"")
sender = m.get("FROM", user).strip()
msg = MIMEText(body)
msg["Subject"], msg["From"], msg["To"] = subject, sender, RECIPIENT
if proto == "smtps":
s = smtplib.SMTP_SSL(addr, port, timeout=30)
else:
s = smtplib.SMTP(addr, port, timeout=30)
if proto != "smtp": # plain smtp = no TLS (local relay)
s.starttls()
if user: # unauthenticated local relays exist
s.login(user, passwd)
s.sendmail(sender, [RECIPIENT], msg.as_string())
s.quit()
print(f"mailed to {RECIPIENT}")