From f341d22aa01a3e8fe036ac5f1e3a0e9bd8faf223 Mon Sep 17 00:00:00 2001 From: Leo Vasanko Date: Fri, 21 Aug 2026 20:10:41 +0000 Subject: [PATCH] Improved fake traffic generation with abuse bots, utm tags etc. --- scripts/fake_traffic.py | 740 ++++++++++++++++++++++++++++++++-------- 1 file changed, 594 insertions(+), 146 deletions(-) diff --git a/scripts/fake_traffic.py b/scripts/fake_traffic.py index 331ebe0..6dc1652 100755 --- a/scripts/fake_traffic.py +++ b/scripts/fake_traffic.py @@ -6,23 +6,18 @@ # "playwright>=1.45.0", # ] # /// -"""Generate fake browser visits and crawler hits for a Pagerite site. +"""Generate fake browser visits, crawler hits, and abuse scans for a Pagerite site. -The script drives a real Chromium browser with Playwright, clicking visible -internal links so the site's own analytics JavaScript records normal visits -(POST /_a). Most browser sessions enter the site with a cross-origin -``Referer: https://somedomain.com/`` header, and outbound links found on the -page are followed to real external sites (ending the session). Browser -sessions and crawler GETs send a small rotating pool of real public IPs in -X-Forwarded-For, so the backend can reverse-DNS and GeoIP them instead of seeing -every hit as 127.0.0.1. - -Sessions start with a Poisson inter-arrival delay (``--arrival-rate``) to -spread traffic out a little, while still keeping the overall run fast. +Browser sessions (ordinary users) come from realistic residential IPv4 and IPv6 +addresses and stay mostly stable; an IPv6 host part may rotate once mid-session, +and an IPv4 session may switch to another residential address. Crawler hits come +from datacenter IPs, with each crawler profile paired to a matching provider IP +when possible. Abuse scanners fire bursts of vulnerability probes from pinned +datacenter IPs. Run against a local dev server, e.g.: - uv run scripts/fake_traffic.py http://localhost:3200 -b 8 -c 20 + uv run scripts/fake_traffic.py http://localhost:3200 Repeat whenever you want more traffic; each run appends new events to the site's analytics file. @@ -39,7 +34,7 @@ from collections.abc import Sequence from dataclasses import dataclass from datetime import UTC, datetime from typing import Any -from urllib.parse import urljoin, urlparse +from urllib.parse import urlencode, urljoin, urlparse import httpx @@ -59,6 +54,7 @@ class BrowserProfile: class CrawlerProfile: name: str user_agent: str + ip: str BROWSER_PROFILES: list[BrowserProfile] = [ @@ -95,39 +91,397 @@ CRAWLER_PROFILES: list[CrawlerProfile] = [ CrawlerProfile( "googlebot", "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; Googlebot/2.1; +http://www.google.com/bot.html) Chrome/128.0.0.0 Safari/537.36", + "66.249.64.66", # US, Google ), CrawlerProfile( "bingbot", "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; bingbot/2.0; +http://www.bing.com/bingbot.htm) Chrome/128.0.0.0 Safari/537.36", + "40.77.167.0", # US, Microsoft ), CrawlerProfile( - "duckduckbot", "DuckDuckBot/1.1; (+http://duckduckgo.com/duckduckbot.html)" + "duckduckbot", + "DuckDuckBot/1.1; (+http://duckduckgo.com/duckduckbot.html)", + "95.217.0.1", # Germany, Hetzner VPS + ), + CrawlerProfile( + "curl", + "curl/8.5.0", + "139.162.0.1", # Singapore, Linode VPS ), - CrawlerProfile("curl", "curl/8.5.0"), ] -# Small pool of real public resolver IPs. They have real reverse-DNS and GeoIP -# entries, and cycling through a handful avoids hammering DNS during traffic -# generation. -SOURCE_IPS: list[str] = [ - "8.8.8.8", - "1.1.1.1", - "9.9.9.9", - "208.67.222.222", - "185.228.168.9", - "94.140.14.14", +# Residential IPv4 addresses and IPv6 /64 prefixes used for ordinary browser +# sessions. IPv6 entries keep the network part stable and randomise only the +# host part; the host may rotate once mid-session. +RESIDENTIAL_SOURCE_IPS: list[str] = [ + # Residential IPv4 + "91.154.140.209", # Finland, Elisa + "84.143.145.207", # Germany, Deutsche Telekom + "220.165.255.254", # China, Chinanet / China Telecom + "84.235.83.162", # Saudi Arabia, SaudiNet / STC + # Residential IPv6 /64 prefixes + "2a02:8109:ac82:6f0c::/64", # Germany, Deutsche Telekom + "240e:45d:1e60:5b0::/64", # China, China Telecom + "2409:8904:6720:4123::/64", # China, China Unicom ] +# Concrete datacenter IPs used for abuse scanner bursts. They stay pinned for +# the whole scan burst. +# Index 0 randomises its UA per request, index 1 uses a fixed browser UA, +# and index 2 uses a fixed crawler UA. +ABUSE_SOURCE_IPS: list[str] = [ + "45.63.0.12", # US, Vultr VPS + "138.197.0.89", # US, DigitalOcean / Cloudways + "2a01:4f8:0:2::1234", # Germany, Hetzner VPS +] + +# Paths commonly probed by attackers looking for exposed config, admin panels, +# version control, credentials, backups, or debug endpoints. +SUSPICIOUS_PATHS: list[str] = [ + "/.env", + "/env", + "/.env.local", + "/env.development", + "/config", + "/config.json", + "/config.yaml", + "/config.yml", + "/configuration.json", + "/configuration.yaml", + "/configuration.yml", + "/settings.json", + "/settings.yaml", + "/settings.yml", + "/app.config", + "/appsettings.json", + "/appsettings.Development.json", + "/credentials", + "/credentials.json", + "/secrets", + "/secrets.json", + "/.aws/credentials", + "/.ssh/id_rsa", + "/id_rsa", + "/id_rsa.pub", + "/known_hosts", + "/sftp-config.json", + "/admin", + "/administrator", + "/adminer.php", + "/login", + "/signin", + "/auth/login", + "/api/login", + "/api/.env", + "/api/config", + "/api/v1/config", + "/api/v2/config", + "/webhook", + "/webhooks", + "/callback", + "/proxy", + "/image", + "/images", + "/preview", + "/download", + "/downloads", + "/log", + "/logs", + "/debug", + "/trace", + "/phpinfo.php", + "/info.php", + "/phpmyadmin", + "/pma", + "/myadmin", + "/phpMyAdmin", + "/wp-admin", + "/wp-login.php", + "/wp-config.php", + "/xmlrpc.php", + "/wp-json/wp/v2/users", + "/.git/config", + "/.git/HEAD", + "/git/config", + "/swagger-ui.html", + "/v2/api-docs", + "/actuator/env", + "/actuator/health", + "/actuator/configprops", + "/server-status", + "/.htaccess", + "/web.config", + "/package.json", + "/composer.json", + "/vendor/autoload.php", + "/docker-compose.yml", + "/Dockerfile", + "/manage", + "/console", + "/manager", + "/manager/html", + "/metrics", + "/prometheus", + "/healthz", + "/_api", + "/api", + "/api/v1/", + "/api/v2/", + "/graphql", + "/query", + "/feed", + "/rss", + "/_debug", + "/test", + "/testing", + "/tmp", + "/temp", + "/backup", + "/backups", + "/dump", + "/dumps", + "/sql", + "/db", + "/database", + "/dump.sql", + "/backup.sql", + "/db.sql", + "/backup.zip", + "/backup.tar.gz", + "/site.zip", + "/site.tar.gz", + "/source.zip", + "/src.zip", + "/upload", + "/uploads", + "/import", + "/export", + "/token", + "/tokens", + "/oauth", + "/oauth2", + "/openid", + "/jwks", + "/keys", + "/key", + "/private", + "/public", +] + +# Realistic external referers. Most sessions arrive with a generic referer; +# a subset carries matching UTM tags on the landing URL. +PLAIN_REFERRERS: list[str] = [ + "https://example.com/", + "https://somedomain.com/", + "https://another-site.org/", + "https://friend-site.net/", +] + +# (referer origin, utm parameter dict) pairs used for tagged traffic. +TAGGED_REFERRERS: list[tuple[str, dict[str, str]]] = [ + ("https://chatgpt.com/", {"utm_source": "chatgpt.com"}), + ("https://www.google.com/", {"utm_source": "google", "utm_medium": "organic"}), + ("https://twitter.com/", {"utm_source": "twitter", "utm_medium": "social"}), + ("https://www.linkedin.com/", {"utm_source": "linkedin", "utm_medium": "social"}), + ("https://github.com/", {"utm_source": "github", "utm_medium": "referral"}), + ("https://news.ycombinator.com/", {"utm_source": "hackernews", "utm_medium": "referral"}), + ("https://www.reddit.com/", {"utm_source": "reddit", "utm_medium": "social"}), + ("https://medium.com/", {"utm_source": "medium", "utm_medium": "referral"}), + ("https://www.producthunt.com/", {"utm_source": "producthunt", "utm_medium": "referral"}), +] + +# Fraction of referered sessions that also carry UTM tags. +UTM_RATE = 0.25 + +# Innocent-looking paths that do not exist on a Pagerite site. Hitting many of +# these from a single IP is itself a telltale of a spray-and-pray scanner. +NORMAL_404_PATHS: list[str] = [ + "/about", + "/about-us", + "/services", + "/products", + "/contact", + "/contact-us", + "/team", + "/careers", + "/jobs", + "/pricing", + "/features", + "/demo", + "/trial", + "/docs", + "/documentation", + "/api-docs", + "/support", + "/help", + "/faq", + "/knowledge-base", + "/terms", + "/terms-of-service", + "/privacy", + "/privacy-policy", + "/legal", + "/blog", + "/news", + "/articles", + "/press", + "/events", + "/webinars", + "/podcast", + "/videos", + "/resources", + "/whitepapers", + "/case-studies", + "/customers", + "/clients", + "/testimonials", + "/reviews", + "/partners", + "/integrations", + "/api-reference", + "/developers", + "/status", + "/security", + "/trust", + "/compliance", + "/gdpr", + "/ccpa", + "/sitemap", + "/archive", + "/tags", + "/categories", + "/search", + "/users", + "/accounts", + "/dashboard", + "/profile", + "/settings", + "/preferences", + "/notifications", + "/messages", + "/inbox", + "/calendar", + "/reports", + "/analytics", + "/billing", + "/invoice", + "/orders", + "/cart", + "/checkout", + "/store", + "/shop", + "/home", + "/main", + "/start", + "/welcome", + "/intro", + "/overview", + "/summary", + "/portfolio", + "/projects", + "/work", + "/solutions", +] + +ABUSE_USER_AGENTS: list[str] = [ + # Desktop browsers + "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 " + "(KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36", + "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/605.1.15 " + "(KHTML, like Gecko) Version/17.5 Safari/605.1.15", + "Mozilla/5.0 (X11; Linux x86_64; rv:130.0) Gecko/20100101 Firefox/130.0", + "Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:130.0) Gecko/20100101 Firefox/130.0", + "Mozilla/5.0 (Linux; Android 14; SM-S918B) AppleWebKit/537.36 " + "(KHTML, like Gecko) Chrome/128.0.0.0 Mobile Safari/537.36", + "Mozilla/5.0 (iPhone; CPU iPhone OS 17_5 like Mac OS X) AppleWebKit/605.1.15 " + "(KHTML, like Gecko) Version/17.5 Mobile/15E148 Safari/604.1", + # Well-known crawlers / bots + "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; Googlebot/2.1; " + "+http://www.google.com/bot.html) Chrome/128.0.0.0 Safari/537.36", + "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; bingbot/2.0; " + "+http://www.bing.com/bingbot.htm) Chrome/128.0.0.0 Safari/537.36", + "Mozilla/5.0 (compatible; DuckDuckBot/1.1; +http://duckduckgo.com/duckduckbot.html)", + "Mozilla/5.0 (compatible; Baiduspider/2.0; +http://www.baidu.com/search/spider.html)", + "Mozilla/5.0 (Linux; Android 6.0.1; Nexus 5X Build/MMB29P) AppleWebKit/537.36 " + "(KHTML, like Gecko) Chrome/128.0.0.0 Mobile Safari/537.36 " + "(compatible; Googlebot/2.1; +http://www.google.com/bot.html)", + "Mozilla/5.0 (compatible; YandexBot/3.0; +http://yandex.com/bots)", + "Mozilla/5.0 (compatible; DotBot/1.2; +https://opensiteexplorer.org/dotbot; help@moz.com)", + "Mozilla/5.0 (compatible; SemrushBot/7~bl; +http://www.semrush.com/bot.html)", + "Mozilla/5.0 (compatible; AhrefsBot/7.0; +http://ahrefs.com/robot/)", + # Social / service fetchers + "facebookexternalhit/1.1 (+http://www.facebook.com/externalhit_uatext.php)", + "Twitterbot/1.0", + "LinkedInBot/1.0 (compatible; Mozilla/5.0; Apache-HttpClient +http://www.linkedin.com)", + "Slackbot-LinkExpanding 1.0 (+https://api.slack.com/robots)", + "WhatsApp/2.23.20.0", + # Command-line / library clients + "curl/8.5.0", + "Wget/1.21.4 (linux-gnu)", + "python-requests/2.32.3", + "Go-http-client/1.1", + "Node.js/20.5.1", +] + + +def _random_ipv6_host(prefix: str) -> str: + """Return a concrete address within an IPv6 /64 prefix. + + The host part is generated randomly, mimicking a fresh OS privacy address. + The input prefix must end in ``::/64`` (e.g. ``2a02:8109:ac82:6f0c::/64``). + """ + if "/" not in prefix: + return prefix + base, mask = prefix.split("/") + if mask != "64": + raise ValueError(f"only /64 IPv6 prefixes are supported, got {prefix!r}") + if base.endswith("::"): + base = base[:-2] + host = ":".join(f"{random.randint(0, 0xffff):04x}" for _ in range(4)) + return f"{base}:{host}" + + +def _concretize_ip(entry: str) -> str: + """Return a concrete IP address; randomise the host part for IPv6 /64 prefixes.""" + if ":" in entry and "/" in entry: + return _random_ipv6_host(entry) + return entry + + +class _SessionIP: + """Stable IP for a browser session, with one optional mid-session rotation. + + IPv6 prefixes get a fresh random host part; IPv4 addresses are swapped for + another address from the residential pool. + """ + + def __init__(self, entry: str, pool: Sequence[str]): + self.entry = entry + self.pool = pool + self._value = _concretize_ip(entry) + + def current(self) -> str: + return self._value + + def rotate(self) -> None: + if ":" in self.entry and "/" in self.entry: + self._value = _random_ipv6_host(self.entry) + return + # IPv4: switch to another IPv4 address from the residential pool. + for _ in range(20): + candidate_entry = random.choice(self.pool) + if ":" in candidate_entry and "/" in candidate_entry: + continue + candidate = _concretize_ip(candidate_entry) + if candidate != self._value: + self._value = candidate + return + def _sleep(base: float, jitter: float) -> None: time.sleep(max(0.0, base + random.uniform(-jitter, jitter))) -def _source_ip(index: int) -> str: - """Pick one of the small pool of real public IPs.""" - return SOURCE_IPS[index % len(SOURCE_IPS)] - - def _normalize_url(url: str) -> str: """Return a usable base URL, adding missing scheme/host/port parts. @@ -232,43 +586,75 @@ def _run_browser_session( paths: Sequence[str], profile: BrowserProfile, session_index: int, - max_clicks: int, - stay: tuple[float, float], - headless: bool, - fake_ip: str, - referer_rate: float, - include_external: bool = True, + ip_entry: str, ) -> dict[str, Any]: from playwright.sync_api import sync_playwright + MAX_CLICKS = 6 + STAY = (2.0, 6.0) + HEADLESS = True + REFERER_RATE = 0.75 + INCLUDE_EXTERNAL = True + + ip_provider = _SessionIP(ip_entry, RESIDENTIAL_SOURCE_IPS) + ips_used: list[str] = [ip_provider.current()] trail: list[str] = [] start_time = datetime.now(UTC) try: with sync_playwright() as p: browser = p.chromium.launch( - headless=headless, + headless=HEADLESS, args=["--no-sandbox", "--disable-dev-shm-usage"], ) extra_headers = { - "X-Forwarded-For": fake_ip, + "X-Forwarded-For": ip_provider.current(), "Accept-Language": profile.accept_language, } # Most sessions arrive from an external origin; some are direct. - if random.random() < referer_rate: - extra_headers["Referer"] = "https://somedomain.com/" + # A subset of referered sessions carries realistic UTM tags on the + # landing URL; the referer origin is paired with the UTM source. + tagged: dict[str, str] = {} + if random.random() < REFERER_RATE: + if random.random() < UTM_RATE: + referer, tagged = random.choice(TAGGED_REFERRERS) + else: + referer = random.choice(PLAIN_REFERRERS) + extra_headers["Referer"] = referer context = browser.new_context( user_agent=profile.user_agent, viewport={"width": profile.viewport[0], "height": profile.viewport[1]}, extra_http_headers=extra_headers, ) page = context.new_page() + + # Update X-Forwarded-For per request; the value stays stable unless we + # explicitly rotate it once mid-session. + def _route_handler(route, request): + headers = dict(request.headers) + headers["X-Forwarded-For"] = ip_provider.current() + ips_used.append(headers["X-Forwarded-For"]) + route.continue_(headers=headers) + + page.route("**/*", _route_handler) + + # Pick one point during the session to emulate an IP rotation. + rotate_at = random.randint(0, MAX_CLICKS - 1) if MAX_CLICKS > 0 else -1 + entry = random.choice(paths) if paths else "/" - page.goto(urljoin(base, entry), wait_until="networkidle") + landing = urljoin(base, entry) + if tagged: + sep = "&" if "?" in landing else "?" + landing += sep + urlencode(tagged) + page.goto(landing, wait_until="networkidle") trail.append(page.url) - for _ in range(max_clicks): - _sleep(random.uniform(*stay) / 2, 0.3) - links = _collect_links(page, include_external) + for click_idx in range(MAX_CLICKS): + _sleep(random.uniform(*STAY) / 2, 0.3) + if click_idx == rotate_at: + ip_provider.rotate() + ips_used.append(ip_provider.current()) + logger.debug("rotated session IP to %s", ip_provider.current()) + links = _collect_links(page, INCLUDE_EXTERNAL) visible = [item for item in links if item.get("visible")] if not visible: visible = links @@ -291,14 +677,15 @@ def _run_browser_session( break page.wait_for_load_state("networkidle") trail.append(page.url) - _sleep(random.uniform(*stay), 0.5) + _sleep(random.uniform(*STAY), 0.5) browser.close() return { "profile": profile.name, "entry": entry, - "ip": fake_ip, + "ip": ips_used[0], + "ips_seen": len(set(ips_used)), "pages": len(trail), "trail": [urlparse(u).path or "/" for u in trail], "duration": (datetime.now(UTC) - start_time).total_seconds(), @@ -312,12 +699,10 @@ def _run_crawler_hit( base: str, paths: Sequence[str], profile: CrawlerProfile, - profile_index: int, - session_index: int, ) -> dict[str, Any]: path = random.choice(paths) if paths else "/" url = urljoin(base, path) - fake_ip = _source_ip(session_index) + fake_ip = profile.ip headers = { "User-Agent": profile.user_agent, "X-Forwarded-For": fake_ip, @@ -337,6 +722,78 @@ def _run_crawler_hit( return {"profile": profile.name, "path": path, "error": str(exc)} +def _abuse_ua() -> str: + """Return a randomized, syntactically valid user agent for an abuse scan.""" + return random.choice(ABUSE_USER_AGENTS) + + +def _run_abuse_scanner(base: str, ip_index: int) -> dict[str, Any]: + """Fire a burst of vulnerability probes from a single fake IP. + + Scanner 0 randomises its user agent every request, scanner 1 uses a fixed + browser UA, and scanner 2 uses a fixed crawler UA. + """ + ip_entry = ABUSE_SOURCE_IPS[ip_index % len(ABUSE_SOURCE_IPS)] + if ":" in ip_entry and "/" in ip_entry: + fake_ip = _random_ipv6_host(ip_entry) + else: + fake_ip = ip_entry + + MIN_HITS = 15 + MAX_HITS = 25 + total_hits = random.randint(MIN_HITS, MAX_HITS) + + # Ensure the burst contains both telltales: suspicious paths and more + # than ten normal-looking 404 paths. + suspicious_count = max(5, total_hits // 3) + normal_count = total_hits - suspicious_count + if normal_count < 11: + normal_count = 11 + suspicious_count = max(3, total_hits - normal_count) + + paths = random.choices(SUSPICIOUS_PATHS, k=suspicious_count) + random.choices( + NORMAL_404_PATHS, k=normal_count + ) + random.shuffle(paths) + + ua_mode = ip_index % 3 + if ua_mode == 0: + get_ua = _abuse_ua + elif ua_mode == 1: + def get_ua() -> str: + return BROWSER_PROFILES[0].user_agent + else: + def get_ua() -> str: + return CRAWLER_PROFILES[0].user_agent + + scan_results: list[dict[str, Any]] = [] + with httpx.Client(follow_redirects=True, timeout=15.0) as client: + for path in paths: + headers = { + "User-Agent": get_ua(), + "X-Forwarded-For": fake_ip, + "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8", + "Accept-Language": random.choice( + ["en-US,en;q=0.9", "en-GB,en;q=0.8", "en;q=0.7"] + ), + } + try: + r = client.get(urljoin(base, path), headers=headers) + scan_results.append( + {"path": path, "status": r.status_code, "ua": headers["User-Agent"]} + ) + except Exception as exc: # noqa: BLE001 + scan_results.append({"path": path, "error": str(exc)}) + _sleep(0.15, 0.1) + + return { + "scanner": ip_index + 1, + "ip": fake_ip, + "hits": len(scan_results), + "results": scan_results, + } + + def _parse_args(argv: Sequence[str] | None) -> argparse.Namespace: parser = argparse.ArgumentParser( description="Generate fake traffic for a Pagerite site.", @@ -351,54 +808,13 @@ def _parse_args(argv: Sequence[str] | None) -> argparse.Namespace: "missing scheme defaults to http://.", ) parser.add_argument( - "-b", - "--browsers", - type=int, - default=5, - help="Number of simulated browser sessions", - ) - parser.add_argument( - "-c", "--crawlers", type=int, default=10, help="Number of crawler HTTP GETs" - ) - parser.add_argument( - "--max-clicks", - type=int, - default=6, - help="Max internal link clicks per browser session", - ) - parser.add_argument( - "--stay", + "-t", + "--duration", type=float, - nargs=2, - default=[2.0, 6.0], - metavar=("MIN", "MAX"), - help="Seconds to stay on a page before clicking again", + default=60.0, + metavar="SECONDS", + help="Rough maximum time to generate traffic (0 runs one preset batch)", ) - parser.add_argument( - "--headless", - action=argparse.BooleanOptionalAction, - default=True, - help="Run browsers headlessly", - ) - parser.add_argument( - "--arrival-rate", - type=float, - default=1.0, - help="Average arrivals per second (Poisson). 0 disables inter-arrival waits", - ) - parser.add_argument( - "--referer-rate", - type=float, - default=0.75, - help="Share of browser sessions that arrive with a cross-origin Referer", - ) - parser.add_argument( - "--external-links", - action=argparse.BooleanOptionalAction, - default=True, - help="Include real outbound links in random navigation", - ) - parser.add_argument("--seed", type=int, default=None, help="Random seed") parser.add_argument("-v", "--verbose", action="store_true", help="Debug logging") return parser.parse_args(argv) @@ -414,8 +830,6 @@ def main(argv: Sequence[str] | None = None) -> int: logger.error("%s", exc) return 2 - random.seed(args.seed) - # Discover content paths from the public page tree if we can. paths: list[str] = [] try: @@ -428,61 +842,95 @@ def main(argv: Sequence[str] | None = None) -> int: paths = ["/"] logger.info( - "Generating fake traffic against %s (%d content paths, %d browsers, %d crawlers)", + "Generating fake traffic against %s (%d content paths, duration=%ss)", base, len(paths), - args.browsers, - args.crawlers, + args.duration, ) results: list[dict[str, Any]] = [] + arrival_rate = 1.0 - for i in range(args.browsers): - if i > 0: - wait = _poisson_wait(args.arrival_rate) - logger.debug("waiting %.2fs before next browser session", wait) - time.sleep(wait) - profile = random.choice(BROWSER_PROFILES) - fake_ip = _source_ip(i) - logger.info( - "[%d/%d] browser session: %s (ip=%s)", - i + 1, - args.browsers, - profile.name, - fake_ip, - ) - result = _run_browser_session( - base, - paths, - profile, - i, - args.max_clicks, - (args.stay[0], args.stay[1]), - args.headless, - fake_ip, - args.referer_rate, - args.external_links, - ) - results.append(result) - logger.debug(" trail: %s", result.get("trail", [])) + def _wait() -> None: + wait = _poisson_wait(arrival_rate) + logger.debug("waiting %.2fs before next session", wait) + time.sleep(wait) - for i in range(args.crawlers): - if i > 0: - wait = _poisson_wait(args.arrival_rate) - logger.debug("waiting %.2fs before next crawler hit", wait) - time.sleep(wait) - profile_index = i % len(CRAWLER_PROFILES) - profile = CRAWLER_PROFILES[profile_index] - fake_ip = _source_ip(i) - logger.info( - "[%d/%d] crawler hit: %s (ip=%s)", - i + 1, - args.crawlers, - profile.name, - fake_ip, - ) - result = _run_crawler_hit(base, paths, profile, profile_index, i) - results.append(result) + if args.duration <= 0: + # One preset batch. + for i in range(5): + if i > 0: + _wait() + profile = random.choice(BROWSER_PROFILES) + ip_entry = random.choice(RESIDENTIAL_SOURCE_IPS) + logger.info( + "browser session: %s (ip=%s)", + profile.name, + _concretize_ip(ip_entry), + ) + result = _run_browser_session(base, paths, profile, i, ip_entry) + results.append(result) + logger.debug(" trail: %s", result.get("trail", [])) + + for i in range(10): + if i > 0: + _wait() + profile = random.choice(CRAWLER_PROFILES) + logger.info( + "crawler hit: %s (ip=%s)", + profile.name, + profile.ip, + ) + result = _run_crawler_hit(base, paths, profile) + results.append(result) + + for i in range(3): + if i > 0: + _wait() + ip_entry = ABUSE_SOURCE_IPS[i % len(ABUSE_SOURCE_IPS)] + logger.info("abuse scanner: %s", ip_entry) + result = _run_abuse_scanner(base, i) + results.append(result) + logger.debug( + " hits: %s", [r.get("path") for r in result.get("results", [])] + ) + else: + deadline = time.time() + args.duration + session_index = 0 + while time.time() < deadline: + if session_index > 0: + _wait() + phase = session_index % 3 + if phase == 0: + profile = random.choice(BROWSER_PROFILES) + ip_entry = random.choice(RESIDENTIAL_SOURCE_IPS) + logger.info( + "browser session: %s (ip=%s)", + profile.name, + _concretize_ip(ip_entry), + ) + result = _run_browser_session( + base, paths, profile, session_index, ip_entry + ) + logger.debug(" trail: %s", result.get("trail", [])) + elif phase == 1: + profile = random.choice(CRAWLER_PROFILES) + logger.info( + "crawler hit: %s (ip=%s)", + profile.name, + profile.ip, + ) + result = _run_crawler_hit(base, paths, profile) + else: + ip_entry = ABUSE_SOURCE_IPS[session_index % len(ABUSE_SOURCE_IPS)] + logger.info("abuse scanner: %s", ip_entry) + result = _run_abuse_scanner(base, session_index // 3) + logger.debug( + " hits: %s", + [r.get("path") for r in result.get("results", [])], + ) + results.append(result) + session_index += 1 ok = sum(1 for r in results if "error" not in r) logger.info("Done: %d/%d requests succeeded.", ok, len(results))