Improved fake traffic generation with abuse bots, utm tags etc.
This commit is contained in:
+594
-146
@@ -6,23 +6,18 @@
|
||||
# "playwright>=1.45.0",
|
||||
# ]
|
||||
# ///
|
||||
"""Generate fake browser visits and crawler hits for a Pagerite site.
|
||||
"""Generate fake browser visits, crawler hits, and abuse scans for a Pagerite site.
|
||||
|
||||
The script drives a real Chromium browser with Playwright, clicking visible
|
||||
internal links so the site's own analytics JavaScript records normal visits
|
||||
(POST /_a). Most browser sessions enter the site with a cross-origin
|
||||
``Referer: https://somedomain.com/`` header, and outbound links found on the
|
||||
page are followed to real external sites (ending the session). Browser
|
||||
sessions and crawler GETs send a small rotating pool of real public IPs in
|
||||
X-Forwarded-For, so the backend can reverse-DNS and GeoIP them instead of seeing
|
||||
every hit as 127.0.0.1.
|
||||
|
||||
Sessions start with a Poisson inter-arrival delay (``--arrival-rate``) to
|
||||
spread traffic out a little, while still keeping the overall run fast.
|
||||
Browser sessions (ordinary users) come from realistic residential IPv4 and IPv6
|
||||
addresses and stay mostly stable; an IPv6 host part may rotate once mid-session,
|
||||
and an IPv4 session may switch to another residential address. Crawler hits come
|
||||
from datacenter IPs, with each crawler profile paired to a matching provider IP
|
||||
when possible. Abuse scanners fire bursts of vulnerability probes from pinned
|
||||
datacenter IPs.
|
||||
|
||||
Run against a local dev server, e.g.:
|
||||
|
||||
uv run scripts/fake_traffic.py http://localhost:3200 -b 8 -c 20
|
||||
uv run scripts/fake_traffic.py http://localhost:3200
|
||||
|
||||
Repeat whenever you want more traffic; each run appends new events to the
|
||||
site's analytics file.
|
||||
@@ -39,7 +34,7 @@ from collections.abc import Sequence
|
||||
from dataclasses import dataclass
|
||||
from datetime import UTC, datetime
|
||||
from typing import Any
|
||||
from urllib.parse import urljoin, urlparse
|
||||
from urllib.parse import urlencode, urljoin, urlparse
|
||||
|
||||
import httpx
|
||||
|
||||
@@ -59,6 +54,7 @@ class BrowserProfile:
|
||||
class CrawlerProfile:
|
||||
name: str
|
||||
user_agent: str
|
||||
ip: str
|
||||
|
||||
|
||||
BROWSER_PROFILES: list[BrowserProfile] = [
|
||||
@@ -95,39 +91,397 @@ CRAWLER_PROFILES: list[CrawlerProfile] = [
|
||||
CrawlerProfile(
|
||||
"googlebot",
|
||||
"Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; Googlebot/2.1; +http://www.google.com/bot.html) Chrome/128.0.0.0 Safari/537.36",
|
||||
"66.249.64.66", # US, Google
|
||||
),
|
||||
CrawlerProfile(
|
||||
"bingbot",
|
||||
"Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; bingbot/2.0; +http://www.bing.com/bingbot.htm) Chrome/128.0.0.0 Safari/537.36",
|
||||
"40.77.167.0", # US, Microsoft
|
||||
),
|
||||
CrawlerProfile(
|
||||
"duckduckbot", "DuckDuckBot/1.1; (+http://duckduckgo.com/duckduckbot.html)"
|
||||
"duckduckbot",
|
||||
"DuckDuckBot/1.1; (+http://duckduckgo.com/duckduckbot.html)",
|
||||
"95.217.0.1", # Germany, Hetzner VPS
|
||||
),
|
||||
CrawlerProfile(
|
||||
"curl",
|
||||
"curl/8.5.0",
|
||||
"139.162.0.1", # Singapore, Linode VPS
|
||||
),
|
||||
CrawlerProfile("curl", "curl/8.5.0"),
|
||||
]
|
||||
|
||||
# Small pool of real public resolver IPs. They have real reverse-DNS and GeoIP
|
||||
# entries, and cycling through a handful avoids hammering DNS during traffic
|
||||
# generation.
|
||||
SOURCE_IPS: list[str] = [
|
||||
"8.8.8.8",
|
||||
"1.1.1.1",
|
||||
"9.9.9.9",
|
||||
"208.67.222.222",
|
||||
"185.228.168.9",
|
||||
"94.140.14.14",
|
||||
# Residential IPv4 addresses and IPv6 /64 prefixes used for ordinary browser
|
||||
# sessions. IPv6 entries keep the network part stable and randomise only the
|
||||
# host part; the host may rotate once mid-session.
|
||||
RESIDENTIAL_SOURCE_IPS: list[str] = [
|
||||
# Residential IPv4
|
||||
"91.154.140.209", # Finland, Elisa
|
||||
"84.143.145.207", # Germany, Deutsche Telekom
|
||||
"220.165.255.254", # China, Chinanet / China Telecom
|
||||
"84.235.83.162", # Saudi Arabia, SaudiNet / STC
|
||||
# Residential IPv6 /64 prefixes
|
||||
"2a02:8109:ac82:6f0c::/64", # Germany, Deutsche Telekom
|
||||
"240e:45d:1e60:5b0::/64", # China, China Telecom
|
||||
"2409:8904:6720:4123::/64", # China, China Unicom
|
||||
]
|
||||
|
||||
# Concrete datacenter IPs used for abuse scanner bursts. They stay pinned for
|
||||
# the whole scan burst.
|
||||
# Index 0 randomises its UA per request, index 1 uses a fixed browser UA,
|
||||
# and index 2 uses a fixed crawler UA.
|
||||
ABUSE_SOURCE_IPS: list[str] = [
|
||||
"45.63.0.12", # US, Vultr VPS
|
||||
"138.197.0.89", # US, DigitalOcean / Cloudways
|
||||
"2a01:4f8:0:2::1234", # Germany, Hetzner VPS
|
||||
]
|
||||
|
||||
# Paths commonly probed by attackers looking for exposed config, admin panels,
|
||||
# version control, credentials, backups, or debug endpoints.
|
||||
SUSPICIOUS_PATHS: list[str] = [
|
||||
"/.env",
|
||||
"/env",
|
||||
"/.env.local",
|
||||
"/env.development",
|
||||
"/config",
|
||||
"/config.json",
|
||||
"/config.yaml",
|
||||
"/config.yml",
|
||||
"/configuration.json",
|
||||
"/configuration.yaml",
|
||||
"/configuration.yml",
|
||||
"/settings.json",
|
||||
"/settings.yaml",
|
||||
"/settings.yml",
|
||||
"/app.config",
|
||||
"/appsettings.json",
|
||||
"/appsettings.Development.json",
|
||||
"/credentials",
|
||||
"/credentials.json",
|
||||
"/secrets",
|
||||
"/secrets.json",
|
||||
"/.aws/credentials",
|
||||
"/.ssh/id_rsa",
|
||||
"/id_rsa",
|
||||
"/id_rsa.pub",
|
||||
"/known_hosts",
|
||||
"/sftp-config.json",
|
||||
"/admin",
|
||||
"/administrator",
|
||||
"/adminer.php",
|
||||
"/login",
|
||||
"/signin",
|
||||
"/auth/login",
|
||||
"/api/login",
|
||||
"/api/.env",
|
||||
"/api/config",
|
||||
"/api/v1/config",
|
||||
"/api/v2/config",
|
||||
"/webhook",
|
||||
"/webhooks",
|
||||
"/callback",
|
||||
"/proxy",
|
||||
"/image",
|
||||
"/images",
|
||||
"/preview",
|
||||
"/download",
|
||||
"/downloads",
|
||||
"/log",
|
||||
"/logs",
|
||||
"/debug",
|
||||
"/trace",
|
||||
"/phpinfo.php",
|
||||
"/info.php",
|
||||
"/phpmyadmin",
|
||||
"/pma",
|
||||
"/myadmin",
|
||||
"/phpMyAdmin",
|
||||
"/wp-admin",
|
||||
"/wp-login.php",
|
||||
"/wp-config.php",
|
||||
"/xmlrpc.php",
|
||||
"/wp-json/wp/v2/users",
|
||||
"/.git/config",
|
||||
"/.git/HEAD",
|
||||
"/git/config",
|
||||
"/swagger-ui.html",
|
||||
"/v2/api-docs",
|
||||
"/actuator/env",
|
||||
"/actuator/health",
|
||||
"/actuator/configprops",
|
||||
"/server-status",
|
||||
"/.htaccess",
|
||||
"/web.config",
|
||||
"/package.json",
|
||||
"/composer.json",
|
||||
"/vendor/autoload.php",
|
||||
"/docker-compose.yml",
|
||||
"/Dockerfile",
|
||||
"/manage",
|
||||
"/console",
|
||||
"/manager",
|
||||
"/manager/html",
|
||||
"/metrics",
|
||||
"/prometheus",
|
||||
"/healthz",
|
||||
"/_api",
|
||||
"/api",
|
||||
"/api/v1/",
|
||||
"/api/v2/",
|
||||
"/graphql",
|
||||
"/query",
|
||||
"/feed",
|
||||
"/rss",
|
||||
"/_debug",
|
||||
"/test",
|
||||
"/testing",
|
||||
"/tmp",
|
||||
"/temp",
|
||||
"/backup",
|
||||
"/backups",
|
||||
"/dump",
|
||||
"/dumps",
|
||||
"/sql",
|
||||
"/db",
|
||||
"/database",
|
||||
"/dump.sql",
|
||||
"/backup.sql",
|
||||
"/db.sql",
|
||||
"/backup.zip",
|
||||
"/backup.tar.gz",
|
||||
"/site.zip",
|
||||
"/site.tar.gz",
|
||||
"/source.zip",
|
||||
"/src.zip",
|
||||
"/upload",
|
||||
"/uploads",
|
||||
"/import",
|
||||
"/export",
|
||||
"/token",
|
||||
"/tokens",
|
||||
"/oauth",
|
||||
"/oauth2",
|
||||
"/openid",
|
||||
"/jwks",
|
||||
"/keys",
|
||||
"/key",
|
||||
"/private",
|
||||
"/public",
|
||||
]
|
||||
|
||||
# Realistic external referers. Most sessions arrive with a generic referer;
|
||||
# a subset carries matching UTM tags on the landing URL.
|
||||
PLAIN_REFERRERS: list[str] = [
|
||||
"https://example.com/",
|
||||
"https://somedomain.com/",
|
||||
"https://another-site.org/",
|
||||
"https://friend-site.net/",
|
||||
]
|
||||
|
||||
# (referer origin, utm parameter dict) pairs used for tagged traffic.
|
||||
TAGGED_REFERRERS: list[tuple[str, dict[str, str]]] = [
|
||||
("https://chatgpt.com/", {"utm_source": "chatgpt.com"}),
|
||||
("https://www.google.com/", {"utm_source": "google", "utm_medium": "organic"}),
|
||||
("https://twitter.com/", {"utm_source": "twitter", "utm_medium": "social"}),
|
||||
("https://www.linkedin.com/", {"utm_source": "linkedin", "utm_medium": "social"}),
|
||||
("https://github.com/", {"utm_source": "github", "utm_medium": "referral"}),
|
||||
("https://news.ycombinator.com/", {"utm_source": "hackernews", "utm_medium": "referral"}),
|
||||
("https://www.reddit.com/", {"utm_source": "reddit", "utm_medium": "social"}),
|
||||
("https://medium.com/", {"utm_source": "medium", "utm_medium": "referral"}),
|
||||
("https://www.producthunt.com/", {"utm_source": "producthunt", "utm_medium": "referral"}),
|
||||
]
|
||||
|
||||
# Fraction of referered sessions that also carry UTM tags.
|
||||
UTM_RATE = 0.25
|
||||
|
||||
# Innocent-looking paths that do not exist on a Pagerite site. Hitting many of
|
||||
# these from a single IP is itself a telltale of a spray-and-pray scanner.
|
||||
NORMAL_404_PATHS: list[str] = [
|
||||
"/about",
|
||||
"/about-us",
|
||||
"/services",
|
||||
"/products",
|
||||
"/contact",
|
||||
"/contact-us",
|
||||
"/team",
|
||||
"/careers",
|
||||
"/jobs",
|
||||
"/pricing",
|
||||
"/features",
|
||||
"/demo",
|
||||
"/trial",
|
||||
"/docs",
|
||||
"/documentation",
|
||||
"/api-docs",
|
||||
"/support",
|
||||
"/help",
|
||||
"/faq",
|
||||
"/knowledge-base",
|
||||
"/terms",
|
||||
"/terms-of-service",
|
||||
"/privacy",
|
||||
"/privacy-policy",
|
||||
"/legal",
|
||||
"/blog",
|
||||
"/news",
|
||||
"/articles",
|
||||
"/press",
|
||||
"/events",
|
||||
"/webinars",
|
||||
"/podcast",
|
||||
"/videos",
|
||||
"/resources",
|
||||
"/whitepapers",
|
||||
"/case-studies",
|
||||
"/customers",
|
||||
"/clients",
|
||||
"/testimonials",
|
||||
"/reviews",
|
||||
"/partners",
|
||||
"/integrations",
|
||||
"/api-reference",
|
||||
"/developers",
|
||||
"/status",
|
||||
"/security",
|
||||
"/trust",
|
||||
"/compliance",
|
||||
"/gdpr",
|
||||
"/ccpa",
|
||||
"/sitemap",
|
||||
"/archive",
|
||||
"/tags",
|
||||
"/categories",
|
||||
"/search",
|
||||
"/users",
|
||||
"/accounts",
|
||||
"/dashboard",
|
||||
"/profile",
|
||||
"/settings",
|
||||
"/preferences",
|
||||
"/notifications",
|
||||
"/messages",
|
||||
"/inbox",
|
||||
"/calendar",
|
||||
"/reports",
|
||||
"/analytics",
|
||||
"/billing",
|
||||
"/invoice",
|
||||
"/orders",
|
||||
"/cart",
|
||||
"/checkout",
|
||||
"/store",
|
||||
"/shop",
|
||||
"/home",
|
||||
"/main",
|
||||
"/start",
|
||||
"/welcome",
|
||||
"/intro",
|
||||
"/overview",
|
||||
"/summary",
|
||||
"/portfolio",
|
||||
"/projects",
|
||||
"/work",
|
||||
"/solutions",
|
||||
]
|
||||
|
||||
ABUSE_USER_AGENTS: list[str] = [
|
||||
# Desktop browsers
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
|
||||
"(KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36",
|
||||
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/605.1.15 "
|
||||
"(KHTML, like Gecko) Version/17.5 Safari/605.1.15",
|
||||
"Mozilla/5.0 (X11; Linux x86_64; rv:130.0) Gecko/20100101 Firefox/130.0",
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:130.0) Gecko/20100101 Firefox/130.0",
|
||||
"Mozilla/5.0 (Linux; Android 14; SM-S918B) AppleWebKit/537.36 "
|
||||
"(KHTML, like Gecko) Chrome/128.0.0.0 Mobile Safari/537.36",
|
||||
"Mozilla/5.0 (iPhone; CPU iPhone OS 17_5 like Mac OS X) AppleWebKit/605.1.15 "
|
||||
"(KHTML, like Gecko) Version/17.5 Mobile/15E148 Safari/604.1",
|
||||
# Well-known crawlers / bots
|
||||
"Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; Googlebot/2.1; "
|
||||
"+http://www.google.com/bot.html) Chrome/128.0.0.0 Safari/537.36",
|
||||
"Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; bingbot/2.0; "
|
||||
"+http://www.bing.com/bingbot.htm) Chrome/128.0.0.0 Safari/537.36",
|
||||
"Mozilla/5.0 (compatible; DuckDuckBot/1.1; +http://duckduckgo.com/duckduckbot.html)",
|
||||
"Mozilla/5.0 (compatible; Baiduspider/2.0; +http://www.baidu.com/search/spider.html)",
|
||||
"Mozilla/5.0 (Linux; Android 6.0.1; Nexus 5X Build/MMB29P) AppleWebKit/537.36 "
|
||||
"(KHTML, like Gecko) Chrome/128.0.0.0 Mobile Safari/537.36 "
|
||||
"(compatible; Googlebot/2.1; +http://www.google.com/bot.html)",
|
||||
"Mozilla/5.0 (compatible; YandexBot/3.0; +http://yandex.com/bots)",
|
||||
"Mozilla/5.0 (compatible; DotBot/1.2; +https://opensiteexplorer.org/dotbot; help@moz.com)",
|
||||
"Mozilla/5.0 (compatible; SemrushBot/7~bl; +http://www.semrush.com/bot.html)",
|
||||
"Mozilla/5.0 (compatible; AhrefsBot/7.0; +http://ahrefs.com/robot/)",
|
||||
# Social / service fetchers
|
||||
"facebookexternalhit/1.1 (+http://www.facebook.com/externalhit_uatext.php)",
|
||||
"Twitterbot/1.0",
|
||||
"LinkedInBot/1.0 (compatible; Mozilla/5.0; Apache-HttpClient +http://www.linkedin.com)",
|
||||
"Slackbot-LinkExpanding 1.0 (+https://api.slack.com/robots)",
|
||||
"WhatsApp/2.23.20.0",
|
||||
# Command-line / library clients
|
||||
"curl/8.5.0",
|
||||
"Wget/1.21.4 (linux-gnu)",
|
||||
"python-requests/2.32.3",
|
||||
"Go-http-client/1.1",
|
||||
"Node.js/20.5.1",
|
||||
]
|
||||
|
||||
|
||||
def _random_ipv6_host(prefix: str) -> str:
|
||||
"""Return a concrete address within an IPv6 /64 prefix.
|
||||
|
||||
The host part is generated randomly, mimicking a fresh OS privacy address.
|
||||
The input prefix must end in ``::/64`` (e.g. ``2a02:8109:ac82:6f0c::/64``).
|
||||
"""
|
||||
if "/" not in prefix:
|
||||
return prefix
|
||||
base, mask = prefix.split("/")
|
||||
if mask != "64":
|
||||
raise ValueError(f"only /64 IPv6 prefixes are supported, got {prefix!r}")
|
||||
if base.endswith("::"):
|
||||
base = base[:-2]
|
||||
host = ":".join(f"{random.randint(0, 0xffff):04x}" for _ in range(4))
|
||||
return f"{base}:{host}"
|
||||
|
||||
|
||||
def _concretize_ip(entry: str) -> str:
|
||||
"""Return a concrete IP address; randomise the host part for IPv6 /64 prefixes."""
|
||||
if ":" in entry and "/" in entry:
|
||||
return _random_ipv6_host(entry)
|
||||
return entry
|
||||
|
||||
|
||||
class _SessionIP:
|
||||
"""Stable IP for a browser session, with one optional mid-session rotation.
|
||||
|
||||
IPv6 prefixes get a fresh random host part; IPv4 addresses are swapped for
|
||||
another address from the residential pool.
|
||||
"""
|
||||
|
||||
def __init__(self, entry: str, pool: Sequence[str]):
|
||||
self.entry = entry
|
||||
self.pool = pool
|
||||
self._value = _concretize_ip(entry)
|
||||
|
||||
def current(self) -> str:
|
||||
return self._value
|
||||
|
||||
def rotate(self) -> None:
|
||||
if ":" in self.entry and "/" in self.entry:
|
||||
self._value = _random_ipv6_host(self.entry)
|
||||
return
|
||||
# IPv4: switch to another IPv4 address from the residential pool.
|
||||
for _ in range(20):
|
||||
candidate_entry = random.choice(self.pool)
|
||||
if ":" in candidate_entry and "/" in candidate_entry:
|
||||
continue
|
||||
candidate = _concretize_ip(candidate_entry)
|
||||
if candidate != self._value:
|
||||
self._value = candidate
|
||||
return
|
||||
|
||||
|
||||
def _sleep(base: float, jitter: float) -> None:
|
||||
time.sleep(max(0.0, base + random.uniform(-jitter, jitter)))
|
||||
|
||||
|
||||
def _source_ip(index: int) -> str:
|
||||
"""Pick one of the small pool of real public IPs."""
|
||||
return SOURCE_IPS[index % len(SOURCE_IPS)]
|
||||
|
||||
|
||||
def _normalize_url(url: str) -> str:
|
||||
"""Return a usable base URL, adding missing scheme/host/port parts.
|
||||
|
||||
@@ -232,43 +586,75 @@ def _run_browser_session(
|
||||
paths: Sequence[str],
|
||||
profile: BrowserProfile,
|
||||
session_index: int,
|
||||
max_clicks: int,
|
||||
stay: tuple[float, float],
|
||||
headless: bool,
|
||||
fake_ip: str,
|
||||
referer_rate: float,
|
||||
include_external: bool = True,
|
||||
ip_entry: str,
|
||||
) -> dict[str, Any]:
|
||||
from playwright.sync_api import sync_playwright
|
||||
|
||||
MAX_CLICKS = 6
|
||||
STAY = (2.0, 6.0)
|
||||
HEADLESS = True
|
||||
REFERER_RATE = 0.75
|
||||
INCLUDE_EXTERNAL = True
|
||||
|
||||
ip_provider = _SessionIP(ip_entry, RESIDENTIAL_SOURCE_IPS)
|
||||
ips_used: list[str] = [ip_provider.current()]
|
||||
trail: list[str] = []
|
||||
start_time = datetime.now(UTC)
|
||||
try:
|
||||
with sync_playwright() as p:
|
||||
browser = p.chromium.launch(
|
||||
headless=headless,
|
||||
headless=HEADLESS,
|
||||
args=["--no-sandbox", "--disable-dev-shm-usage"],
|
||||
)
|
||||
extra_headers = {
|
||||
"X-Forwarded-For": fake_ip,
|
||||
"X-Forwarded-For": ip_provider.current(),
|
||||
"Accept-Language": profile.accept_language,
|
||||
}
|
||||
# Most sessions arrive from an external origin; some are direct.
|
||||
if random.random() < referer_rate:
|
||||
extra_headers["Referer"] = "https://somedomain.com/"
|
||||
# A subset of referered sessions carries realistic UTM tags on the
|
||||
# landing URL; the referer origin is paired with the UTM source.
|
||||
tagged: dict[str, str] = {}
|
||||
if random.random() < REFERER_RATE:
|
||||
if random.random() < UTM_RATE:
|
||||
referer, tagged = random.choice(TAGGED_REFERRERS)
|
||||
else:
|
||||
referer = random.choice(PLAIN_REFERRERS)
|
||||
extra_headers["Referer"] = referer
|
||||
context = browser.new_context(
|
||||
user_agent=profile.user_agent,
|
||||
viewport={"width": profile.viewport[0], "height": profile.viewport[1]},
|
||||
extra_http_headers=extra_headers,
|
||||
)
|
||||
page = context.new_page()
|
||||
|
||||
# Update X-Forwarded-For per request; the value stays stable unless we
|
||||
# explicitly rotate it once mid-session.
|
||||
def _route_handler(route, request):
|
||||
headers = dict(request.headers)
|
||||
headers["X-Forwarded-For"] = ip_provider.current()
|
||||
ips_used.append(headers["X-Forwarded-For"])
|
||||
route.continue_(headers=headers)
|
||||
|
||||
page.route("**/*", _route_handler)
|
||||
|
||||
# Pick one point during the session to emulate an IP rotation.
|
||||
rotate_at = random.randint(0, MAX_CLICKS - 1) if MAX_CLICKS > 0 else -1
|
||||
|
||||
entry = random.choice(paths) if paths else "/"
|
||||
page.goto(urljoin(base, entry), wait_until="networkidle")
|
||||
landing = urljoin(base, entry)
|
||||
if tagged:
|
||||
sep = "&" if "?" in landing else "?"
|
||||
landing += sep + urlencode(tagged)
|
||||
page.goto(landing, wait_until="networkidle")
|
||||
trail.append(page.url)
|
||||
|
||||
for _ in range(max_clicks):
|
||||
_sleep(random.uniform(*stay) / 2, 0.3)
|
||||
links = _collect_links(page, include_external)
|
||||
for click_idx in range(MAX_CLICKS):
|
||||
_sleep(random.uniform(*STAY) / 2, 0.3)
|
||||
if click_idx == rotate_at:
|
||||
ip_provider.rotate()
|
||||
ips_used.append(ip_provider.current())
|
||||
logger.debug("rotated session IP to %s", ip_provider.current())
|
||||
links = _collect_links(page, INCLUDE_EXTERNAL)
|
||||
visible = [item for item in links if item.get("visible")]
|
||||
if not visible:
|
||||
visible = links
|
||||
@@ -291,14 +677,15 @@ def _run_browser_session(
|
||||
break
|
||||
page.wait_for_load_state("networkidle")
|
||||
trail.append(page.url)
|
||||
_sleep(random.uniform(*stay), 0.5)
|
||||
_sleep(random.uniform(*STAY), 0.5)
|
||||
|
||||
browser.close()
|
||||
|
||||
return {
|
||||
"profile": profile.name,
|
||||
"entry": entry,
|
||||
"ip": fake_ip,
|
||||
"ip": ips_used[0],
|
||||
"ips_seen": len(set(ips_used)),
|
||||
"pages": len(trail),
|
||||
"trail": [urlparse(u).path or "/" for u in trail],
|
||||
"duration": (datetime.now(UTC) - start_time).total_seconds(),
|
||||
@@ -312,12 +699,10 @@ def _run_crawler_hit(
|
||||
base: str,
|
||||
paths: Sequence[str],
|
||||
profile: CrawlerProfile,
|
||||
profile_index: int,
|
||||
session_index: int,
|
||||
) -> dict[str, Any]:
|
||||
path = random.choice(paths) if paths else "/"
|
||||
url = urljoin(base, path)
|
||||
fake_ip = _source_ip(session_index)
|
||||
fake_ip = profile.ip
|
||||
headers = {
|
||||
"User-Agent": profile.user_agent,
|
||||
"X-Forwarded-For": fake_ip,
|
||||
@@ -337,6 +722,78 @@ def _run_crawler_hit(
|
||||
return {"profile": profile.name, "path": path, "error": str(exc)}
|
||||
|
||||
|
||||
def _abuse_ua() -> str:
|
||||
"""Return a randomized, syntactically valid user agent for an abuse scan."""
|
||||
return random.choice(ABUSE_USER_AGENTS)
|
||||
|
||||
|
||||
def _run_abuse_scanner(base: str, ip_index: int) -> dict[str, Any]:
|
||||
"""Fire a burst of vulnerability probes from a single fake IP.
|
||||
|
||||
Scanner 0 randomises its user agent every request, scanner 1 uses a fixed
|
||||
browser UA, and scanner 2 uses a fixed crawler UA.
|
||||
"""
|
||||
ip_entry = ABUSE_SOURCE_IPS[ip_index % len(ABUSE_SOURCE_IPS)]
|
||||
if ":" in ip_entry and "/" in ip_entry:
|
||||
fake_ip = _random_ipv6_host(ip_entry)
|
||||
else:
|
||||
fake_ip = ip_entry
|
||||
|
||||
MIN_HITS = 15
|
||||
MAX_HITS = 25
|
||||
total_hits = random.randint(MIN_HITS, MAX_HITS)
|
||||
|
||||
# Ensure the burst contains both telltales: suspicious paths and more
|
||||
# than ten normal-looking 404 paths.
|
||||
suspicious_count = max(5, total_hits // 3)
|
||||
normal_count = total_hits - suspicious_count
|
||||
if normal_count < 11:
|
||||
normal_count = 11
|
||||
suspicious_count = max(3, total_hits - normal_count)
|
||||
|
||||
paths = random.choices(SUSPICIOUS_PATHS, k=suspicious_count) + random.choices(
|
||||
NORMAL_404_PATHS, k=normal_count
|
||||
)
|
||||
random.shuffle(paths)
|
||||
|
||||
ua_mode = ip_index % 3
|
||||
if ua_mode == 0:
|
||||
get_ua = _abuse_ua
|
||||
elif ua_mode == 1:
|
||||
def get_ua() -> str:
|
||||
return BROWSER_PROFILES[0].user_agent
|
||||
else:
|
||||
def get_ua() -> str:
|
||||
return CRAWLER_PROFILES[0].user_agent
|
||||
|
||||
scan_results: list[dict[str, Any]] = []
|
||||
with httpx.Client(follow_redirects=True, timeout=15.0) as client:
|
||||
for path in paths:
|
||||
headers = {
|
||||
"User-Agent": get_ua(),
|
||||
"X-Forwarded-For": fake_ip,
|
||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
|
||||
"Accept-Language": random.choice(
|
||||
["en-US,en;q=0.9", "en-GB,en;q=0.8", "en;q=0.7"]
|
||||
),
|
||||
}
|
||||
try:
|
||||
r = client.get(urljoin(base, path), headers=headers)
|
||||
scan_results.append(
|
||||
{"path": path, "status": r.status_code, "ua": headers["User-Agent"]}
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
scan_results.append({"path": path, "error": str(exc)})
|
||||
_sleep(0.15, 0.1)
|
||||
|
||||
return {
|
||||
"scanner": ip_index + 1,
|
||||
"ip": fake_ip,
|
||||
"hits": len(scan_results),
|
||||
"results": scan_results,
|
||||
}
|
||||
|
||||
|
||||
def _parse_args(argv: Sequence[str] | None) -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Generate fake traffic for a Pagerite site.",
|
||||
@@ -351,54 +808,13 @@ def _parse_args(argv: Sequence[str] | None) -> argparse.Namespace:
|
||||
"missing scheme defaults to http://.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"-b",
|
||||
"--browsers",
|
||||
type=int,
|
||||
default=5,
|
||||
help="Number of simulated browser sessions",
|
||||
)
|
||||
parser.add_argument(
|
||||
"-c", "--crawlers", type=int, default=10, help="Number of crawler HTTP GETs"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--max-clicks",
|
||||
type=int,
|
||||
default=6,
|
||||
help="Max internal link clicks per browser session",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--stay",
|
||||
"-t",
|
||||
"--duration",
|
||||
type=float,
|
||||
nargs=2,
|
||||
default=[2.0, 6.0],
|
||||
metavar=("MIN", "MAX"),
|
||||
help="Seconds to stay on a page before clicking again",
|
||||
default=60.0,
|
||||
metavar="SECONDS",
|
||||
help="Rough maximum time to generate traffic (0 runs one preset batch)",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--headless",
|
||||
action=argparse.BooleanOptionalAction,
|
||||
default=True,
|
||||
help="Run browsers headlessly",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--arrival-rate",
|
||||
type=float,
|
||||
default=1.0,
|
||||
help="Average arrivals per second (Poisson). 0 disables inter-arrival waits",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--referer-rate",
|
||||
type=float,
|
||||
default=0.75,
|
||||
help="Share of browser sessions that arrive with a cross-origin Referer",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--external-links",
|
||||
action=argparse.BooleanOptionalAction,
|
||||
default=True,
|
||||
help="Include real outbound links in random navigation",
|
||||
)
|
||||
parser.add_argument("--seed", type=int, default=None, help="Random seed")
|
||||
parser.add_argument("-v", "--verbose", action="store_true", help="Debug logging")
|
||||
return parser.parse_args(argv)
|
||||
|
||||
@@ -414,8 +830,6 @@ def main(argv: Sequence[str] | None = None) -> int:
|
||||
logger.error("%s", exc)
|
||||
return 2
|
||||
|
||||
random.seed(args.seed)
|
||||
|
||||
# Discover content paths from the public page tree if we can.
|
||||
paths: list[str] = []
|
||||
try:
|
||||
@@ -428,61 +842,95 @@ def main(argv: Sequence[str] | None = None) -> int:
|
||||
paths = ["/"]
|
||||
|
||||
logger.info(
|
||||
"Generating fake traffic against %s (%d content paths, %d browsers, %d crawlers)",
|
||||
"Generating fake traffic against %s (%d content paths, duration=%ss)",
|
||||
base,
|
||||
len(paths),
|
||||
args.browsers,
|
||||
args.crawlers,
|
||||
args.duration,
|
||||
)
|
||||
|
||||
results: list[dict[str, Any]] = []
|
||||
arrival_rate = 1.0
|
||||
|
||||
for i in range(args.browsers):
|
||||
if i > 0:
|
||||
wait = _poisson_wait(args.arrival_rate)
|
||||
logger.debug("waiting %.2fs before next browser session", wait)
|
||||
time.sleep(wait)
|
||||
profile = random.choice(BROWSER_PROFILES)
|
||||
fake_ip = _source_ip(i)
|
||||
logger.info(
|
||||
"[%d/%d] browser session: %s (ip=%s)",
|
||||
i + 1,
|
||||
args.browsers,
|
||||
profile.name,
|
||||
fake_ip,
|
||||
)
|
||||
result = _run_browser_session(
|
||||
base,
|
||||
paths,
|
||||
profile,
|
||||
i,
|
||||
args.max_clicks,
|
||||
(args.stay[0], args.stay[1]),
|
||||
args.headless,
|
||||
fake_ip,
|
||||
args.referer_rate,
|
||||
args.external_links,
|
||||
)
|
||||
results.append(result)
|
||||
logger.debug(" trail: %s", result.get("trail", []))
|
||||
def _wait() -> None:
|
||||
wait = _poisson_wait(arrival_rate)
|
||||
logger.debug("waiting %.2fs before next session", wait)
|
||||
time.sleep(wait)
|
||||
|
||||
for i in range(args.crawlers):
|
||||
if i > 0:
|
||||
wait = _poisson_wait(args.arrival_rate)
|
||||
logger.debug("waiting %.2fs before next crawler hit", wait)
|
||||
time.sleep(wait)
|
||||
profile_index = i % len(CRAWLER_PROFILES)
|
||||
profile = CRAWLER_PROFILES[profile_index]
|
||||
fake_ip = _source_ip(i)
|
||||
logger.info(
|
||||
"[%d/%d] crawler hit: %s (ip=%s)",
|
||||
i + 1,
|
||||
args.crawlers,
|
||||
profile.name,
|
||||
fake_ip,
|
||||
)
|
||||
result = _run_crawler_hit(base, paths, profile, profile_index, i)
|
||||
results.append(result)
|
||||
if args.duration <= 0:
|
||||
# One preset batch.
|
||||
for i in range(5):
|
||||
if i > 0:
|
||||
_wait()
|
||||
profile = random.choice(BROWSER_PROFILES)
|
||||
ip_entry = random.choice(RESIDENTIAL_SOURCE_IPS)
|
||||
logger.info(
|
||||
"browser session: %s (ip=%s)",
|
||||
profile.name,
|
||||
_concretize_ip(ip_entry),
|
||||
)
|
||||
result = _run_browser_session(base, paths, profile, i, ip_entry)
|
||||
results.append(result)
|
||||
logger.debug(" trail: %s", result.get("trail", []))
|
||||
|
||||
for i in range(10):
|
||||
if i > 0:
|
||||
_wait()
|
||||
profile = random.choice(CRAWLER_PROFILES)
|
||||
logger.info(
|
||||
"crawler hit: %s (ip=%s)",
|
||||
profile.name,
|
||||
profile.ip,
|
||||
)
|
||||
result = _run_crawler_hit(base, paths, profile)
|
||||
results.append(result)
|
||||
|
||||
for i in range(3):
|
||||
if i > 0:
|
||||
_wait()
|
||||
ip_entry = ABUSE_SOURCE_IPS[i % len(ABUSE_SOURCE_IPS)]
|
||||
logger.info("abuse scanner: %s", ip_entry)
|
||||
result = _run_abuse_scanner(base, i)
|
||||
results.append(result)
|
||||
logger.debug(
|
||||
" hits: %s", [r.get("path") for r in result.get("results", [])]
|
||||
)
|
||||
else:
|
||||
deadline = time.time() + args.duration
|
||||
session_index = 0
|
||||
while time.time() < deadline:
|
||||
if session_index > 0:
|
||||
_wait()
|
||||
phase = session_index % 3
|
||||
if phase == 0:
|
||||
profile = random.choice(BROWSER_PROFILES)
|
||||
ip_entry = random.choice(RESIDENTIAL_SOURCE_IPS)
|
||||
logger.info(
|
||||
"browser session: %s (ip=%s)",
|
||||
profile.name,
|
||||
_concretize_ip(ip_entry),
|
||||
)
|
||||
result = _run_browser_session(
|
||||
base, paths, profile, session_index, ip_entry
|
||||
)
|
||||
logger.debug(" trail: %s", result.get("trail", []))
|
||||
elif phase == 1:
|
||||
profile = random.choice(CRAWLER_PROFILES)
|
||||
logger.info(
|
||||
"crawler hit: %s (ip=%s)",
|
||||
profile.name,
|
||||
profile.ip,
|
||||
)
|
||||
result = _run_crawler_hit(base, paths, profile)
|
||||
else:
|
||||
ip_entry = ABUSE_SOURCE_IPS[session_index % len(ABUSE_SOURCE_IPS)]
|
||||
logger.info("abuse scanner: %s", ip_entry)
|
||||
result = _run_abuse_scanner(base, session_index // 3)
|
||||
logger.debug(
|
||||
" hits: %s",
|
||||
[r.get("path") for r in result.get("results", [])],
|
||||
)
|
||||
results.append(result)
|
||||
session_index += 1
|
||||
|
||||
ok = sum(1 for r in results if "error" not in r)
|
||||
logger.info("Done: %d/%d requests succeeded.", ok, len(results))
|
||||
|
||||
Reference in New Issue
Block a user