Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
130 changes: 130 additions & 0 deletions .github/legit-bots-sources.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,130 @@
[
{
"name": "anthropic",
"user_agent": "claude-searchbot|claude-user",
"comment": "Anthropic Claude-SearchBot / Claude-User - https://claude.com/crawling/bots.json (auto-synced). ClaudeBot (training crawler) shares these IPs but is deliberately excluded from the user_agent match.",
"urls": ["https://claude.com/crawling/bots.json"],
"jq_filter": "[.prefixes[] | .ipv4Prefix // .ipv6Prefix] | .[]"
},
{
"name": "duckduckbot",
"user_agent": "duckduckbot",
"comment": "DuckDuckBot - https://duckduckgo.com/duckduckbot.json (auto-synced)",
"urls": ["https://duckduckgo.com/duckduckbot.json"],
"jq_filter": "[.prefixes[] | .ipv4Prefix // .ipv6Prefix] | .[]"
},
{
"name": "duckassistbot",
"user_agent": "duckassistbot",
"comment": "DuckAssistBot (DuckDuckGo AI assistant) - https://duckduckgo.com/duckassistbot.json (auto-synced). DuckDuckGo serves one combined list for DuckDuckBot and DuckAssistBot; kept separate so search crawling can be allowed without AI assistant fetches.",
"urls": ["https://duckduckgo.com/duckassistbot.json"],
"jq_filter": "[.prefixes[] | .ipv4Prefix // .ipv6Prefix] | .[]"
},
{
"name": "gptbot",
"user_agent": "gptbot",
"comment": "GPTBot (OpenAI) - https://openai.com/gptbot.json (auto-synced)",
"urls": ["https://openai.com/gptbot.json"],
"jq_filter": "[.prefixes[] | .ipv4Prefix // .ipv6Prefix] | .[]"
},
{
"name": "openai-searchbot",
"user_agent": "oai-searchbot",
"comment": "OpenAI SearchBot - https://openai.com/searchbot.json (auto-synced)",
"urls": ["https://openai.com/searchbot.json"],
"jq_filter": "[.prefixes[] | .ipv4Prefix // .ipv6Prefix] | .[]"
},
{
"name": "openai-chatgpt-user",
"user_agent": "chatgpt-user",
"comment": "OpenAI ChatGPT-User - https://openai.com/chatgpt-user.json (auto-synced)",
"urls": ["https://openai.com/chatgpt-user.json"],
"jq_filter": "[.prefixes[] | .ipv4Prefix // .ipv6Prefix] | .[]"
},
{
"name": "openai-adsbot",
"user_agent": "oai-adsbot",
"comment": "OpenAI OAI-AdsBot - https://openai.com/adsbot.json (auto-synced)",
"urls": ["https://openai.com/adsbot.json"],
"jq_filter": "[.prefixes[] | .ipv4Prefix // .ipv6Prefix] | .[]"
},
{
"name": "perplexitybot",
"user_agent": "perplexitybot",
"comment": "PerplexityBot - https://www.perplexity.ai/perplexitybot.json (auto-synced)",
"urls": ["https://www.perplexity.ai/perplexitybot.json"],
"jq_filter": "[.prefixes[] | .ipv4Prefix // .ipv6Prefix] | .[]"
},
{
"name": "perplexity-user",
"user_agent": "perplexity-user",
"comment": "Perplexity-User (user-triggered fetches) - https://www.perplexity.ai/perplexity-user.json (auto-synced)",
"urls": ["https://www.perplexity.ai/perplexity-user.json"],
"jq_filter": "[.prefixes[] | .ipv4Prefix // .ipv6Prefix] | .[]"
},
{
"name": "mistralai-index",
"user_agent": "mistralai-index",
"comment": "MistralAI-Index - https://mistral.ai/mistralai-index-ips.json (auto-synced)",
"urls": ["https://mistral.ai/mistralai-index-ips.json"],
"jq_filter": "[.prefixes[] | .ipv4Prefix // .ipv6Prefix] | .[]"
},
{
"name": "mistralai-user",
"user_agent": "mistralai-user",
"comment": "MistralAI-User (user-triggered fetches) - https://mistral.ai/mistralai-user-ips.json (auto-synced)",
"urls": ["https://mistral.ai/mistralai-user-ips.json"],
"jq_filter": "[.prefixes[] | .ipv4Prefix // .ipv6Prefix] | .[]"
},
{
"name": "ahrefs",
"user_agent": "ahrefsbot|ahrefssiteaudit",
"comment": "AhrefsBot / AhrefsSiteAudit - https://api.ahrefs.com/v3/public/crawler-ip-ranges (auto-synced). rDNS ahrefs.net does not forward-confirm, so this is IP based.",
"urls": ["https://api.ahrefs.com/v3/public/crawler-ip-ranges"],
"jq_filter": "[.prefixes[] | .ipv4Prefix // .ipv6Prefix] | .[]"
},
{
"name": "flipboard",
"user_agent": "flipboardproxy|flipboardbot|flipboardrss",
"comment": "Flipboard proxy / link preview - https://cdn.flipboard.com/flipboard_ip.txt (auto-synced). rDNS proxy.flipboard.com does not forward-confirm, so this is IP based.",
"urls": ["https://cdn.flipboard.com/flipboard_ip.txt"]
},
{
"name": "telegram",
"user_agent": "telegrambot",
"comment": "Telegram - https://core.telegram.org/resources/cidr.txt (auto-synced)",
"urls": ["https://core.telegram.org/resources/cidr.txt"]
},
{
"name": "uptimerobot",
"user_agent": "uptimerobot",
"comment": "UptimeRobot monitors - https://cdn.uptimerobot.com/api/IPv4andIPv6.txt (auto-synced)",
"urls": ["https://cdn.uptimerobot.com/api/IPv4andIPv6.txt"]
},
{
"name": "datadog",
"user_agent": "datadog",
"comment": "Datadog Synthetics monitors - https://ip-ranges.datadoghq.com/ (.synthetics; auto-synced)",
"urls": ["https://ip-ranges.datadoghq.com/"],
"jq_filter": "[.synthetics.prefixes_ipv4[], .synthetics.prefixes_ipv6[]] | .[]"
},
{
"name": "pagerduty",
"user_agent": "pagerduty-webhook",
"comment": "PagerDuty outbound webhooks (UA PagerDuty-Webhook/Vx.0) - US + EU service regions, https://developer.pagerduty.com/docs/webhook-ips (auto-synced)",
"urls": [
"https://developer.pagerduty.com/ip-safelists/webhooks-us-service-region-json",
"https://developer.pagerduty.com/ip-safelists/webhooks-eu-service-region-json"
],
"jq_filter": ".[]"
},
{
"name": "lumar",
"user_agent": "lumar|deepcrawl",
"comment": "Lumar (ex-Deepcrawl) - see https://www.lumar.io/ for the current IP list",
"urls": ["https://www.lumar.io/wp-content/uploads/2026/02/lumar_ip_list_feb_2026.json"],
"jq_filter": "[.prefixes[] | .ipv4Prefix // .ipv6Prefix] | .[]",
"disabled": true,
"note": "Lumar publishes its IP list under a date-stamped wp-content URL that rotates (…/2026/02/lumar_ip_list_feb_2026.json), so it cannot be polled on a fixed URL. Refresh whitelists/benign_bots/legit_bots/lumar.json by hand and bump the URL here when Lumar publishes a new one."
}
]
200 changes: 200 additions & 0 deletions .github/scripts/sync-legit-bots.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,200 @@
#!/usr/bin/env python3
"""Refresh the IP-based entries in whitelists/benign_bots/legit_bots/ from their
upstream vendor sources.

Driven by .github/legit-bots-sources.json. Each entry names the target bot, the
user_agent regex and file comment to keep, and one or more URLs to pull ranges
from. JSON sources are reduced with `jq_filter`; sources without one are read as
plain text, one address or CIDR per line.

Entries verified by reverse DNS (googlebot, seznam, linkedin, ...) have nothing
to poll and are deliberately absent from the config - this script never touches
them.

Writes GitHub Actions outputs (changed, names, files) when GITHUB_OUTPUT is set.
Exits non-zero only if every source failed, so one broken vendor endpoint cannot
block the others from updating.
"""
import ipaddress
import json
import os
import subprocess
import sys

ROOT = os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
CONFIG = os.environ.get("LEGIT_BOTS_CONFIG") or os.path.join(
ROOT, ".github", "legit-bots-sources.json")
TARGET_DIR = os.environ.get("LEGIT_BOTS_TARGET_DIR") or os.path.join(
ROOT, "whitelists", "benign_bots", "legit_bots")

# A vendor serving a truncated or half-broken list would otherwise silently gut a
# whitelist. Below this ratio the change is still applied but called out loudly in
# the PR body for a human to confirm - a real shrink does happen (Perplexity went
# from 254 stale entries to the 8 they actually publish), so this must not hard-fail.
SHRINK_RATIO = 0.5
SHRINK_FLOOR = 10

UA = "Mozilla/5.0 (compatible; crowdsec-sec-lists-sync)"


def fetch(url):
r = subprocess.run(
["curl", "-fsSL", "--max-time", "60", "--retry", "3", "--retry-delay", "5",
"-A", UA, url],
capture_output=True, text=True,
)
if r.returncode != 0:
raise RuntimeError(f"fetch failed ({r.returncode}): {r.stderr.strip()[:200]}")
return r.stdout


def apply_jq(payload, expr):
r = subprocess.run(["jq", "-er", expr], input=payload,
capture_output=True, text=True)
if r.returncode != 0:
raise RuntimeError(f"jq filter failed: {r.stderr.strip()[:200]}")
return r.stdout


def parse_ranges(text):
"""Normalise a newline-separated list of addresses/CIDRs to canonical CIDRs."""
out = []
for raw in text.splitlines():
line = raw.strip().strip('"').strip(",").strip('"')
if not line or line.startswith("#"):
continue
try:
if "/" in line:
net = ipaddress.ip_network(line, strict=True)
else:
addr = ipaddress.ip_address(line)
net = ipaddress.ip_network(f"{addr}/{addr.max_prefixlen}")
except ValueError as ex:
raise RuntimeError(f"invalid address {line!r}: {ex}")
if not net.is_global:
raise RuntimeError(f"refusing non-public range {line!r}")
out.append(net)
# deterministic order so a vendor reshuffling its list produces no diff
uniq = sorted(set(out), key=lambda n: (n.version, n.network_address, n.prefixlen))
return [str(n) for n in uniq]


def render(entry, ranges):
body = json.dumps(
{"name": entry["name"], "user_agent": entry["user_agent"], "ranges": ranges},
separators=(",", ":"), ensure_ascii=False,
)
return f"# {entry['comment']}\n{body}\n"


def current_range_count(path):
try:
with open(path) as fh:
line = [l for l in fh.read().splitlines() if not l.startswith("#")][0]
return len(json.loads(line).get("ranges", []))
except (OSError, IndexError, ValueError):
return 0


def main():
with open(CONFIG) as fh:
config = json.load(fh)

changed, warnings, errors, skipped = [], [], [], []

for entry in config:
name = entry["name"]
if entry.get("disabled"):
skipped.append(name)
print(f"-- {name}: disabled ({entry.get('note', 'no reason given')})")
continue

path = os.path.join(TARGET_DIR, f"{name}.json")
try:
ranges = []
for url in entry["urls"]:
payload = fetch(url)
if "jq_filter" in entry:
payload = apply_jq(payload, entry["jq_filter"])
ranges.extend(parse_ranges(payload))
ranges = parse_ranges("\n".join(ranges)) # merge + re-sort across URLs
except RuntimeError as ex:
errors.append(f"{name}: {ex}")
print(f"!! {name}: {ex}", file=sys.stderr)
continue

if not ranges:
errors.append(f"{name}: source returned no ranges, keeping existing file")
print(f"!! {name}: empty result, refusing to write", file=sys.stderr)
continue

before = current_range_count(path)
if before >= SHRINK_FLOOR and len(ranges) < before * SHRINK_RATIO:
warnings.append(
f"`{name}` shrank sharply: {before} -> {len(ranges)} ranges. "
f"Confirm the vendor really retired these before merging."
)

new = render(entry, ranges)
old = open(path).read() if os.path.exists(path) else None
if old == new:
print(f" {name}: unchanged ({len(ranges)} ranges)")
continue

with open(path, "w") as fh:
fh.write(new)
changed.append((name, before, len(ranges)))
print(f"++ {name}: {before} -> {len(ranges)} ranges")

if errors and not changed and len(errors) == len([e for e in config if not e.get("disabled")]):
print("\nevery source failed", file=sys.stderr)
return 1

lines = []
if changed:
lines.append("Automated refresh of `whitelists/benign_bots/legit_bots/` "
"from upstream vendor sources.\n")
lines.append("| bot | ranges before | ranges after |")
lines.append("| --- | ---: | ---: |")
for name, before, after in changed:
lines.append(f"| `{name}` | {before} | {after} |")
if warnings:
lines.append("\n**Review carefully:**\n")
lines.extend(f"- {w}" for w in warnings)
if errors:
lines.append("\n**Sources that failed this run** (their files were left "
"untouched):\n")
lines.extend(f"- {e}" for e in errors)
if skipped:
lines.append(f"\nNot polled (no stable URL): {', '.join(f'`{s}`' for s in skipped)}.")
body = "\n".join(lines)

out = os.environ.get("GITHUB_OUTPUT")
# only materialise the PR body under CI, and outside the checkout, so running
# this locally never leaves a stray file in the working tree
body_file = os.path.join(os.environ.get("RUNNER_TEMP") or ROOT, "pr-body.md")
if body and out:
with open(body_file, "w") as fh:
fh.write(body + "\n")
elif body:
print("\n--- PR body ---\n" + body)

if out:
with open(out, "a") as fh:
fh.write(f"changed={'true' if changed else 'false'}\n")
fh.write(f"names={' '.join(n for n, _, _ in changed)}\n")
fh.write(f"files={' '.join(f'whitelists/benign_bots/legit_bots/{n}.json' for n, _, _ in changed)}\n")
fh.write(f"body_file={body_file}\n")

summary = os.environ.get("GITHUB_STEP_SUMMARY")
if summary and body:
with open(summary, "a") as fh:
fh.write(body + "\n")

print(f"\n{len(changed)} changed, {len(warnings)} warning(s), "
f"{len(errors)} error(s), {len(skipped)} skipped")
return 0


if __name__ == "__main__":
sys.exit(main())
Loading