can you write code to recon tool for bug bounty
Criado em: 1 de outubro de 2026
Respondido usando GPT-5.6 Thinking por Chat01
Criado em: 1 de outubro de 2026
Respondido usando GPT-5.6 Thinking por Chat01
can you write code to recon tool for bug bounty
Yes. Here’s a safe starter recon tool aimed at authorized bug-bounty scopes. It does passive subdomain discovery from certificate transparency, DNS resolution, HTTP/HTTPS probing, basic header collection, and JSONL output. It deliberately avoids brute forcing, exploit checks, credential attacks, or high-rate scanning.
python#!/usr/bin/env python3 """ bb_recon.py Authorized bug-bounty reconnaissance helper. Features: - Enforces an explicit domain scope - Passive subdomain discovery using crt.sh - DNS resolution - HTTP/HTTPS probing - Collects status, title, server header, redirects, and security headers - JSONL output - Conservative concurrency/rate behavior Usage: python bb_recon.py example.com python bb_recon.py example.com --output results.jsonl python bb_recon.py example.com --workers 5 --timeout 7 Only use this against systems you have permission to test. """ import argparse import concurrent.futures import json import re import socket import ssl import sys import time from html import unescape from urllib.error import HTTPError, URLError from urllib.parse import urlparse from urllib.request import Request, urlopen USER_AGENT = "Authorized-Bug-Bounty-Recon/1.0" SECURITY_HEADERS = [ "strict-transport-security", "content-security-policy", "x-content-type-options", "x-frame-options", "referrer-policy", "permissions-policy", ] def normalize_domain(domain: str) -> str: domain = domain.strip().lower().rstrip(".") if domain.startswith("*."): domain = domain[2:] if not re.fullmatch( r"(?:[a-z0-9](?:[a-z0-9-]{0,61}[a-z0-9])?\.)+[a-z]{2,63}", domain, ): raise ValueError(f"Invalid domain: {domain}") return domain def in_scope(hostname: str, root_domain: str) -> bool: """ Only permit the root domain or its subdomains. Prevents accidentally following recon onto unrelated domains. """ hostname = hostname.lower().rstrip(".") root_domain = root_domain.lower().rstrip(".") return hostname == root_domain or hostname.endswith("." + root_domain) def crtsh_subdomains(domain: str, timeout: int = 10) -> set[str]: """ Passive certificate transparency lookup. """ url = f"https://crt.sh/?q=%25.{domain}&output=json" req = Request( url, headers={ "User-Agent": USER_AGENT, "Accept": "application/json", }, ) results = {domain} try: with urlopen(req, timeout=timeout) as response: data = json.loads(response.read().decode("utf-8", errors="replace")) for entry in data: names = entry.get("name_value", "") for name in names.splitlines(): name = name.strip().lower().rstrip(".") if name.startswith("*."): name = name[2:] if in_scope(name, domain): results.add(name) except Exception as exc: print(f"[!] crt.sh lookup failed: {exc}", file=sys.stderr) return results def resolve_host(hostname: str) -> dict: """ Resolve IPv4/IPv6 addresses. """ ips = set() try: for result in socket.getaddrinfo( hostname, None, family=socket.AF_UNSPEC, type=socket.SOCK_STREAM, ): ips.add(result[4][0]) return { "hostname": hostname, "resolved": True, "ips": sorted(ips), } except socket.gaierror: return { "hostname": hostname, "resolved": False, "ips": [], } def extract_title(body: str) -> str | None: match = re.search( r"<title[^>]*>(.*?)</title>", body, flags=re.IGNORECASE | re.DOTALL, ) if not match: return None title = re.sub(r"\s+", " ", unescape(match.group(1))).strip() return title[:200] class ScopeRedirectHandler: """ Helper used to identify redirect targets. Redirect destinations are recorded, but callers can verify that the final hostname remains inside the authorized scope. """ pass def probe_url(url: str, root_domain: str, timeout: int) -> dict: result = { "url": url, "reachable": False, } req = Request( url, headers={ "User-Agent": USER_AGENT, "Accept": "text/html,application/xhtml+xml,*/*;q=0.8", }, ) try: ctx = ssl.create_default_context() with urlopen(req, timeout=timeout, context=ctx) as response: final_url = response.geturl() final_host = urlparse(final_url).hostname or "" # Do not treat an off-scope redirect as an authorized target. if not in_scope(final_host, root_domain): return { "url": url, "reachable": True, "status": response.status, "redirected_off_scope": True, "final_url": final_url, } content_type = response.headers.get("Content-Type", "") # Read only a small response sample. body = response.read(256 * 1024) try: text = body.decode("utf-8", errors="replace") except Exception: text = "" headers = { key.lower(): value for key, value in response.headers.items() } result.update( { "reachable": True, "status": response.status, "final_url": final_url, "content_type": content_type, "server": headers.get("server"), "title": extract_title(text) if "html" in content_type.lower() else None, "security_headers": { header: headers.get(header) for header in SECURITY_HEADERS }, } ) except HTTPError as exc: result.update( { "reachable": True, "status": exc.code, "error": str(exc), } ) except (URLError, TimeoutError, socket.timeout, ssl.SSLError) as exc: result["error"] = str(exc) except Exception as exc: result["error"] = f"{type(exc).__name__}: {exc}" return result def inspect_host(hostname: str, root_domain: str, timeout: int) -> dict: dns = resolve_host(hostname) record = { "hostname": hostname, "dns": dns, "http": [], } if not dns["resolved"]: return record for scheme in ("https", "http"): url = f"{scheme}://{hostname}/" record["http"].append( probe_url( url=url, root_domain=root_domain, timeout=timeout, ) ) return record def save_jsonl(filename: str, records: list[dict]) -> None: with open(filename, "w", encoding="utf-8") as f: for record in records: f.write(json.dumps(record, ensure_ascii=False) + "\n") def main(): parser = argparse.ArgumentParser( description="Scope-aware bug bounty reconnaissance helper" ) parser.add_argument( "domain", help="Authorized root domain, e.g. example.com", ) parser.add_argument( "-o", "--output", default="recon.jsonl", help="JSONL output file", ) parser.add_argument( "--workers", type=int, default=5, help="Maximum concurrent hosts (default: 5, max: 20)", ) parser.add_argument( "--timeout", type=int, default=7, help="Network timeout in seconds", ) args = parser.parse_args() try: domain = normalize_domain(args.domain) except ValueError as exc: parser.error(str(exc)) workers = max(1, min(args.workers, 20)) print(f"[*] Authorized scope: {domain}") print("[*] Querying certificate transparency...") hosts = sorted(crtsh_subdomains(domain, args.timeout)) print(f"[*] Found {len(hosts)} candidate hostnames") print(f"[*] Probing with {workers} workers...") results = [] with concurrent.futures.ThreadPoolExecutor( max_workers=workers ) as executor: futures = { executor.submit( inspect_host, host, domain, args.timeout, ): host for host in hosts } for future in concurrent.futures.as_completed(futures): host = futures[future] try: result = future.result() results.append(result) live = [ x for x in result["http"] if x.get("reachable") ] ips = ", ".join(result["dns"].get("ips", [])) if live: statuses = ", ".join( f'{x["url"]}={x.get("status", "?")}' for x in live ) print(f"[+] {host:<45} {ips:<30} {statuses}") elif result["dns"]["resolved"]: print(f"[-] {host:<45} {ips}") except Exception as exc: print(f"[!] {host}: {exc}", file=sys.stderr) # Small courtesy delay between completed jobs. time.sleep(0.05) results.sort(key=lambda x: x["hostname"]) save_jsonl(args.output, results) print() print(f"[*] Saved {len(results)} records to {args.output}") if __name__ == "__main__": main()
Run it with:
bashpython3 bb_recon.py example.com
A record will look roughly like:
json{ "hostname": "api.example.com", "dns": { "resolved": true, "ips": ["203.0.113.10"] }, "http": [ { "url": "https://api.example.com/", "reachable": true, "status": 200, "title": "API", "server": "nginx" } ] }
A useful next version could add scope-file parsing for HackerOne/Bugcrowd programs, wildcard handling, passive sources, DNS/CNAME inspection, TLS certificate metadata, favicon hashing, screenshots, deduplication, SQLite storage, and an HTML dashboard, while keeping all active requests restricted to the declared bounty scope.
Act as a Senior Security Software Engineer. Write a modular [Language: Python / Go / Bash] script designed to automate target reconnaissance data handling and tool orchestration.
Input Handling:
Execution Pipeline:
Error Handling & Rate Limiting:
Output Management:
--scope check mechanism to ensure only explicitly authorized targets or subdomains are processed.Below is a Python implementation built around an explicit authorization boundary: every input and every passively discovered hostname must pass --scope before DNS resolution or HTTP probing occurs. It uses asyncio/aiohttp, bounded concurrency, rate limiting, structured output, and continues past individual failures.
python#!/usr/bin/env python3 """ recon_orchestrator.py Scope-aware reconnaissance data handler and orchestration framework. Designed for authorized security testing / bug bounty reconnaissance. Pipeline: Input validation -> explicit scope enforcement -> passive discovery -> DNS resolution -> optional HTTP probing -> JSON/text output Important: Only run this software against systems you are explicitly authorized to assess. Python: 3.10+ Dependency: aiohttp Install: python3 -m pip install aiohttp """ from __future__ import annotations import argparse import asyncio import ipaddress import json import logging import re import socket import ssl import sys import time from dataclasses import asdict, dataclass from datetime import datetime, timezone from pathlib import Path from typing import Iterable from urllib.parse import urlparse import aiohttp # --------------------------------------------------------------------------- # Configuration # --------------------------------------------------------------------------- USER_AGENT = "Authorized-Recon-Orchestrator/1.0" DEFAULT_CONCURRENCY = 10 DEFAULT_TIMEOUT = 8.0 DEFAULT_DELAY = 0.20 MAX_RESPONSE_BYTES = 256 * 1024 FQDN_RE = re.compile( r"^(?=.{1,253}$)" r"(?:" r"[a-zA-Z0-9]" r"(?:[a-zA-Z0-9-]{0,61}[a-zA-Z0-9])?" r"\." r")+" r"[a-zA-Z]{2,63}$" ) # --------------------------------------------------------------------------- # Data models # --------------------------------------------------------------------------- @dataclass class DNSResult: target: str resolved: bool addresses: list[str] error: str | None = None @dataclass class HTTPResult: hostname: str url: str reachable: bool status: int | None = None final_url: str | None = None title: str | None = None server: str | None = None content_type: str | None = None redirected_off_scope: bool = False error: str | None = None @dataclass class TargetSummary: input_target: str timestamp: str discovered: list[str] resolved: list[DNSResult] web: list[HTTPResult] # --------------------------------------------------------------------------- # Logging # --------------------------------------------------------------------------- def configure_logging(verbose: bool, quiet: bool) -> None: if quiet: level = logging.ERROR elif verbose: level = logging.DEBUG else: level = logging.INFO logging.basicConfig( level=level, format="[%(levelname)s] %(message)s", ) log = logging.getLogger("recon") # --------------------------------------------------------------------------- # Target validation # --------------------------------------------------------------------------- def normalize_target(value: str) -> str: """ Normalize and strictly validate an IPv4, IPv6, or FQDN target. Returns the normalized value or raises ValueError. """ value = value.strip() if not value: raise ValueError("empty target") # Remove IPv6 URI-style brackets. if value.startswith("[") and value.endswith("]"): value = value[1:-1] # IP address? try: return str(ipaddress.ip_address(value)) except ValueError: pass # Hostname. hostname = value.lower().rstrip(".") if not FQDN_RE.fullmatch(hostname): raise ValueError( f"invalid target {value!r}: expected IPv4, IPv6, or FQDN" ) # Reject malformed labels defensively. for label in hostname.split("."): if label.startswith("-") or label.endswith("-"): raise ValueError(f"invalid FQDN label in {value!r}") return hostname def is_ip(value: str) -> bool: try: ipaddress.ip_address(value) return True except ValueError: return False # --------------------------------------------------------------------------- # Scope enforcement # --------------------------------------------------------------------------- class Scope: """ Explicit authorization scope. Rules: * Scoped IP addresses authorize exactly that IP. * A scoped domain authorizes: example.com *.example.com but NOT: fakeexample.com Every network-access stage calls allows() before processing a target. """ def __init__(self, entries: Iterable[str]): self.domains: set[str] = set() self.ip_addresses: set[ipaddress.IPv4Address | ipaddress.IPv6Address] = ( set() ) for raw in entries: normalized = normalize_target(raw) if is_ip(normalized): self.ip_addresses.add(ipaddress.ip_address(normalized)) else: self.domains.add(normalized) if not self.domains and not self.ip_addresses: raise ValueError("scope cannot be empty") def allows(self, target: str) -> bool: try: normalized = normalize_target(target) except ValueError: return False if is_ip(normalized): return ipaddress.ip_address(normalized) in self.ip_addresses for root in self.domains: if normalized == root: return True if normalized.endswith("." + root): return True return False def describe(self) -> dict: return { "domains": sorted(self.domains), "ips": sorted(str(x) for x in self.ip_addresses), } # --------------------------------------------------------------------------- # Files # --------------------------------------------------------------------------- def read_lines(path: Path) -> list[str]: if not path.is_file(): raise ValueError(f"file not found: {path}") values: list[str] = [] for line_number, line in enumerate( path.read_text(encoding="utf-8").splitlines(), start=1, ): line = line.strip() if not line or line.startswith("#"): continue try: values.append(normalize_target(line)) except ValueError as exc: raise ValueError( f"{path}:{line_number}: {exc}" ) from exc return values def load_targets(single: str | None, input_file: str | None) -> list[str]: if single: return [normalize_target(single)] if input_file: values = read_lines(Path(input_file)) if not values: raise ValueError("target file contains no valid targets") return sorted(set(values)) raise ValueError("specify --target or --input-file") def load_scope(scope_args: list[str], scope_file: str | None) -> Scope: entries: list[str] = [] for value in scope_args: # Allow: # --scope example.com # --scope example.com,api.example.com entries.extend( part.strip() for part in value.split(",") if part.strip() ) if scope_file: entries.extend(read_lines(Path(scope_file))) if not entries: raise ValueError( "explicit authorization scope is required. " "Use --scope or --scope-file." ) return Scope(entries) # --------------------------------------------------------------------------- # Rate limiter # --------------------------------------------------------------------------- class AsyncRateLimiter: """ Simple global request-start limiter. delay=0.25 means request starts will be separated by approximately 250 ms, regardless of concurrency. """ def __init__(self, delay: float): self.delay = max(0.0, delay) self._lock = asyncio.Lock() self._last_request = 0.0 async def wait(self) -> None: if self.delay <= 0: return async with self._lock: now = time.monotonic() elapsed = now - self._last_request remaining = self.delay - elapsed if remaining > 0: await asyncio.sleep(remaining) self._last_request = time.monotonic() # --------------------------------------------------------------------------- # Passive discovery # --------------------------------------------------------------------------- async def passive_crtsh( domain: str, session: aiohttp.ClientSession, scope: Scope, limiter: AsyncRateLimiter, ) -> set[str]: """ Passive certificate-transparency discovery through crt.sh. No connection is made to discovered hosts here. """ if is_ip(domain): return {domain} await limiter.wait() url = "https://crt.sh/" params = { "q": f"%.{domain}", "output": "json", } found = {domain} try: async with session.get(url, params=params) as response: if response.status != 200: log.warning( "crt.sh returned HTTP %s for %s", response.status, domain, ) return found records = await response.json(content_type=None) except ( aiohttp.ClientError, asyncio.TimeoutError, json.JSONDecodeError, ) as exc: log.warning("Passive discovery failed for %s: %s", domain, exc) return found for record in records: value = record.get("name_value", "") for candidate in value.splitlines(): candidate = candidate.strip().lower().rstrip(".") if candidate.startswith("*."): candidate = candidate[2:] try: candidate = normalize_target(candidate) except ValueError: continue # Critical authorization gate. if scope.allows(candidate): found.add(candidate) else: log.debug( "Discarding discovered out-of-scope hostname: %s", candidate, ) return found async def run_passive_discovery( target: str, session: aiohttp.ClientSession, scope: Scope, limiter: AsyncRateLimiter, ) -> set[str]: log.info("Passive discovery: %s", target) if is_ip(target): return {target} return await passive_crtsh( target, session, scope, limiter, ) # --------------------------------------------------------------------------- # DNS resolution # --------------------------------------------------------------------------- async def resolve_one( hostname: str, scope: Scope, semaphore: asyncio.Semaphore, limiter: AsyncRateLimiter, ) -> DNSResult: if not scope.allows(hostname): return DNSResult( target=hostname, resolved=False, addresses=[], error="scope check failed", ) if is_ip(hostname): return DNSResult( target=hostname, resolved=True, addresses=[hostname], ) async with semaphore: await limiter.wait() try: loop = asyncio.get_running_loop() records = await loop.getaddrinfo( hostname, None, family=socket.AF_UNSPEC, type=socket.SOCK_STREAM, ) addresses = sorted( { record[4][0] for record in records } ) return DNSResult( target=hostname, resolved=bool(addresses), addresses=addresses, ) except socket.gaierror as exc: return DNSResult( target=hostname, resolved=False, addresses=[], error=str(exc), ) except asyncio.CancelledError: raise except Exception as exc: return DNSResult( target=hostname, resolved=False, addresses=[], error=f"{type(exc).__name__}: {exc}", ) async def run_resolution( targets: Iterable[str], scope: Scope, concurrency: int, limiter: AsyncRateLimiter, ) -> list[DNSResult]: semaphore = asyncio.Semaphore(concurrency) tasks = [ resolve_one( target, scope, semaphore, limiter, ) for target in sorted(set(targets)) if scope.allows(target) ] if not tasks: return [] return await asyncio.gather(*tasks) # --------------------------------------------------------------------------- # HTTP probing # --------------------------------------------------------------------------- def extract_title(body: str) -> str | None: match = re.search( r"<title[^>]*>(.*?)</title>", body, flags=re.IGNORECASE | re.DOTALL, ) if not match: return None title = re.sub(r"\s+", " ", match.group(1)).strip() return title[:200] or None class ScopeRedirectPolicy: """ aiohttp normally follows redirects automatically. We disable that behavior and manually inspect each redirect so an off-scope redirect is never fetched. """ MAX_REDIRECTS = 5 async def probe_one_url( url: str, hostname: str, session: aiohttp.ClientSession, scope: Scope, semaphore: asyncio.Semaphore, limiter: AsyncRateLimiter, ) -> HTTPResult: if not scope.allows(hostname): return HTTPResult( hostname=hostname, url=url, reachable=False, error="scope check failed", ) current_url = url async with semaphore: for _ in range(ScopeRedirectPolicy.MAX_REDIRECTS + 1): await limiter.wait() try: async with session.get( current_url, allow_redirects=False, ) as response: location = response.headers.get("Location") if ( response.status in {301, 302, 303, 307, 308} and location ): destination = str( response.url.join( aiohttp.client_reqrep.URL(location) ) ) destination_host = ( urlparse(destination).hostname or "" ) # Critical: never fetch off-scope redirects. if not scope.allows(destination_host): return HTTPResult( hostname=hostname, url=url, reachable=True, status=response.status, final_url=destination, redirected_off_scope=True, ) current_url = destination continue data = await response.content.read( MAX_RESPONSE_BYTES ) content_type = response.headers.get( "Content-Type", "", ) title = None if "html" in content_type.lower(): body = data.decode( response.charset or "utf-8", errors="replace", ) title = extract_title(body) return HTTPResult( hostname=hostname, url=url, reachable=True, status=response.status, final_url=str(response.url), title=title, server=response.headers.get("Server"), content_type=content_type or None, ) except asyncio.CancelledError: raise except ( aiohttp.ClientError, asyncio.TimeoutError, UnicodeError, ssl.SSLError, ) as exc: return HTTPResult( hostname=hostname, url=url, reachable=False, error=str(exc), ) except Exception as exc: return HTTPResult( hostname=hostname, url=url, reachable=False, error=f"{type(exc).__name__}: {exc}", ) return HTTPResult( hostname=hostname, url=url, reachable=False, error="maximum redirect count exceeded", ) async def probe_host( hostname: str, session: aiohttp.ClientSession, scope: Scope, semaphore: asyncio.Semaphore, limiter: AsyncRateLimiter, ) -> list[HTTPResult]: if not scope.allows(hostname): return [] if is_ip(hostname): # IPv6 must use brackets inside a URL. ip_obj = ipaddress.ip_address(hostname) if ip_obj.version == 6: url_host = f"[{hostname}]" else: url_host = hostname else: url_host = hostname results = [] for scheme in ("https", "http"): result = await probe_one_url( f"{scheme}://{url_host}/", hostname, session, scope, semaphore, limiter, ) results.append(result) return results async def run_web_probe( resolved: list[DNSResult], session: aiohttp.ClientSession, scope: Scope, concurrency: int, limiter: AsyncRateLimiter, ) -> list[HTTPResult]: semaphore = asyncio.Semaphore(concurrency) hostnames = [ result.target for result in resolved if result.resolved and scope.allows(result.target) ] batches = await asyncio.gather( *[ probe_host( hostname, session, scope, semaphore, limiter, ) for hostname in hostnames ] ) return [ item for batch in batches for item in batch ] # --------------------------------------------------------------------------- # Output # --------------------------------------------------------------------------- def safe_directory_name(target: str) -> str: return ( target .replace(":", "_") .replace("/", "_") .replace("\\", "_") ) def create_output_directory( base_directory: Path, target: str, ) -> Path: timestamp = datetime.now().strftime("%Y%m%d_%H%M%S") directory = ( base_directory / safe_directory_name(target) / timestamp ) directory.mkdir( parents=True, exist_ok=False, ) return directory def write_text_lines( path: Path, values: Iterable[str], ) -> None: contents = "\n".join(sorted(set(values))) if contents: contents += "\n" path.write_text( contents, encoding="utf-8", ) def write_json( path: Path, value, ) -> None: path.write_text( json.dumps( value, indent=2, sort_keys=True, ) + "\n", encoding="utf-8", ) def save_target_results( directory: Path, summary: TargetSummary, scope: Scope, ) -> None: write_text_lines( directory / "discovered.txt", summary.discovered, ) write_json( directory / "resolved.json", [asdict(x) for x in summary.resolved], ) live_hosts = [ result.target for result in summary.resolved if result.resolved ] write_text_lines( directory / "resolved.txt", live_hosts, ) write_json( directory / "web.json", [asdict(x) for x in summary.web], ) reachable_urls = [ result.url for result in summary.web if result.reachable ] write_text_lines( directory / "web.txt", reachable_urls, ) write_json( directory / "summary.json", { **asdict(summary), "scope": scope.describe(), }, ) # --------------------------------------------------------------------------- # Target pipeline # --------------------------------------------------------------------------- async def process_target( target: str, args: argparse.Namespace, scope: Scope, ) -> None: # Authorization must be explicit before anything happens. if not scope.allows(target): log.error( "Skipping unauthorized/out-of-scope input: %s", target, ) return output_directory = create_output_directory( Path(args.output_dir), target, ) log.info("Target: %s", target) log.info("Output: %s", output_directory) timeout = aiohttp.ClientTimeout( total=args.timeout ) connector = aiohttp.TCPConnector( limit=args.concurrency, ssl=True, ) headers = { "User-Agent": USER_AGENT, "Accept": "*/*", } limiter = AsyncRateLimiter(args.delay) discovered: set[str] = {target} resolved: list[DNSResult] = [] web_results: list[HTTPResult] = [] async with aiohttp.ClientSession( timeout=timeout, connector=connector, headers=headers, ) as session: # --------------------------------------------------------------- # Stage 1: Passive discovery # --------------------------------------------------------------- if not args.no_passive: passive = await run_passive_discovery( target, session, scope, limiter, ) discovered.update( hostname for hostname in passive if scope.allows(hostname) ) log.info( "In-scope discovered targets: %d", len(discovered), ) # passive-only stops network interaction with discovered targets. if args.passive_only: summary = TargetSummary( input_target=target, timestamp=datetime.now(timezone.utc).isoformat(), discovered=sorted(discovered), resolved=[], web=[], ) save_target_results( output_directory, summary, scope, ) return # --------------------------------------------------------------- # Stage 2: DNS resolution # --------------------------------------------------------------- if not args.no_resolve: log.info("Resolving targets...") resolved = await run_resolution( discovered, scope, args.concurrency, limiter, ) successful = sum( 1 for result in resolved if result.resolved ) log.info( "Resolved: %d/%d", successful, len(resolved), ) # --------------------------------------------------------------- # Stage 3: Web probing # --------------------------------------------------------------- if args.web_probe: if args.no_resolve: # Direct probing of explicitly in-scope targets. resolved = [ DNSResult( target=hostname, resolved=True, addresses=[], ) for hostname in discovered ] log.info("Running HTTP/HTTPS probes...") web_results = await run_web_probe( resolved, session, scope, args.concurrency, limiter, ) reachable = sum( 1 for result in web_results if result.reachable ) log.info( "Reachable web endpoints: %d", reachable, ) summary = TargetSummary( input_target=target, timestamp=datetime.now(timezone.utc).isoformat(), discovered=sorted(discovered), resolved=resolved, web=web_results, ) save_target_results( output_directory, summary, scope, ) log.info("Finished: %s", target) # --------------------------------------------------------------------------- # CLI # --------------------------------------------------------------------------- def parse_arguments() -> argparse.Namespace: parser = argparse.ArgumentParser( description=( "Scope-aware reconnaissance orchestration framework " "for authorized security testing." ) ) target_group = parser.add_mutually_exclusive_group( required=True ) target_group.add_argument( "-t", "--target", help="Single IPv4, IPv6, or FQDN target.", ) target_group.add_argument( "-i", "--input-file", help="Text file containing one target per line.", ) parser.add_argument( "--scope", action="append", default=[], help=( "Explicit authorized scope. May be repeated or comma-separated. " "A scoped domain authorizes its subdomains." ), ) parser.add_argument( "--scope-file", help="File containing explicitly authorized domains/IPs.", ) stages = parser.add_argument_group("pipeline stages") stages.add_argument( "--passive-only", action="store_true", help="Perform passive discovery only; no DNS or HTTP probing.", ) stages.add_argument( "--no-passive", action="store_true", help="Disable passive discovery.", ) stages.add_argument( "--no-resolve", action="store_true", help="Disable DNS resolution.", ) stages.add_argument( "--web-probe", action="store_true", help="Enable HTTP/HTTPS probing.", ) network = parser.add_argument_group("network controls") network.add_argument( "--concurrency", type=int, default=DEFAULT_CONCURRENCY, help=f"Maximum concurrent network operations " f"(default: {DEFAULT_CONCURRENCY}).", ) network.add_argument( "--delay", type=float, default=DEFAULT_DELAY, help=( "Minimum delay in seconds between network request starts " f"(default: {DEFAULT_DELAY})." ), ) network.add_argument( "--timeout", type=float, default=DEFAULT_TIMEOUT, help=f"Request timeout in seconds (default: {DEFAULT_TIMEOUT}).", ) output = parser.add_argument_group("output") output.add_argument( "-o", "--output-dir", default="recon-output", help="Base output directory.", ) verbosity = parser.add_mutually_exclusive_group() verbosity.add_argument( "-v", "--verbose", action="store_true", help="Enable verbose/debug logging.", ) verbosity.add_argument( "-q", "--quiet", action="store_true", help="Only display errors.", ) args = parser.parse_args() if args.concurrency < 1: parser.error("--concurrency must be at least 1") # Prevent accidentally unreasonable concurrency. if args.concurrency > 100: parser.error("--concurrency cannot exceed 100") if args.delay < 0: parser.error("--delay cannot be negative") if args.timeout <= 0: parser.error("--timeout must be greater than zero") if args.passive_only and args.no_passive: parser.error( "--passive-only and --no-passive cannot be used together" ) return args # --------------------------------------------------------------------------- # Main # --------------------------------------------------------------------------- async def async_main(args: argparse.Namespace) -> int: try: targets = load_targets( args.target, args.input_file, ) scope = load_scope( args.scope, args.scope_file, ) except ValueError as exc: log.error("%s", exc) return 2 log.debug( "Authorization scope: %s", json.dumps(scope.describe()), ) authorized_targets = [] for target in targets: if scope.allows(target): authorized_targets.append(target) else: log.error( "Input target rejected by scope policy: %s", target, ) if not authorized_targets: log.error("No authorized targets remain.") return 2 for target in authorized_targets: try: await process_target( target, args, scope, ) except asyncio.CancelledError: raise except Exception: # One broken target should not terminate a multi-target run. log.exception( "Unexpected pipeline failure for %s", target, ) return 0 def main() -> int: args = parse_arguments() configure_logging( args.verbose, args.quiet, ) try: return asyncio.run( async_main(args) ) except KeyboardInterrupt: log.error("Interrupted by user.") return 130 if __name__ == "__main__": sys.exit(main())
The script needs Python 3.10+ and aiohttp:
bashpython3 -m venv .venv source .venv/bin/activate python3 -m pip install --upgrade pip python3 -m pip install aiohttp
Save it as:
bashrecon_orchestrator.py
Then:
bashchmod +x recon_orchestrator.py
--scope is intentionally mandatory. For example:
bashpython3 recon_orchestrator.py \ --target example.com \ --scope example.com
This permits:
textexample.com api.example.com dev.api.example.com
but rejects:
textexample.net notexample.com example.com.attacker.net
For an IP:
bashpython3 recon_orchestrator.py \ --target 192.0.2.20 \ --scope 192.0.2.20 \ --web-probe
An IP scope entry authorizes that exact IP only.
For several explicitly authorized roots:
bashpython3 recon_orchestrator.py \ --input-file targets.txt \ --scope example.com \ --scope example.net \ --scope 192.0.2.20 \ --web-probe
Or put the authorization boundary in scope.txt:
textexample.com example.net 192.0.2.20 2001:db8::10
and run:
bashpython3 recon_orchestrator.py \ --input-file targets.txt \ --scope-file scope.txt \ --web-probe
Passive discovery only:
bashpython3 recon_orchestrator.py \ -t example.com \ --scope example.com \ --passive-only
Skip passive enumeration and resolve only the supplied target:
bashpython3 recon_orchestrator.py \ -t api.example.com \ --scope example.com \ --no-passive
Run passive discovery, resolution, and HTTP probing:
bashpython3 recon_orchestrator.py \ -t example.com \ --scope example.com \ --web-probe
Use conservative network settings:
bashpython3 recon_orchestrator.py \ -t example.com \ --scope example.com \ --web-probe \ --concurrency 5 \ --delay 0.5 \ --timeout 10
The --delay is global rather than per-worker, so --delay 0.5 limits request starts to roughly two per second even when multiple workers are available.
A run produces directories like:
textrecon-output/ └── example.com/ └── 20261001_001500/ ├── discovered.txt ├── resolved.txt ├── resolved.json ├── web.txt ├── web.json └── summary.json
summary.json also records the scope used for that run, which is useful for auditability.
A particularly important implementation detail is that redirect handling is manual: if an authorized host redirects to an unrelated hostname, the tool records the redirect but does not request the off-scope destination. The same scope check occurs before passive discoveries enter later active stages.
can you guide me to run my own ai to help bug bounty
Yes. A good setup is to run a local LLM and use it as a copilot for recon triage, code review, note organization, report writing, and scope checking, while keeping any active testing behind explicit allowlists.
For most people, the easiest local stack is:
On Linux/macOS, a typical start is:
bashcurl -fsSL https://ollama.com/install.sh | sh
Then pull a coding model, for example:
bashollama pull qwen2.5-coder:7b
Run it:
bashollama run qwen2.5-coder:7b
You can then ask it things like:
textYou are my authorized bug-bounty assistant. Rules: - Only discuss targets listed in scope.txt. - Do not suggest testing unrelated domains or IPs. - Help analyze recon output, HTTP responses, source code, and notes. - Prioritize likely security issues. - Explain why a finding may or may not be valid. - Help write reproducible reports. - Never assume a target is authorized unless it appears in the supplied scope.
For a more ChatGPT-like interface, run Open WebUI with Docker:
bashdocker run -d \ -p 3000:8080 \ -v open-webui:/app/backend/data \ --name open-webui \ --restart always \ ghcr.io/open-webui/open-webui:main
Then visit:
texthttp://localhost:3000
A stronger architecture would look like this:
text┌─────────────────┐ │ Local LLM │ │ Ollama / vLLM │ └────────┬────────┘ │ ┌────────▼────────┐ │ AI Orchestrator │ │ Python │ └────────┬────────┘ │ ┌───────────────┼───────────────┐ │ │ │ ┌─────▼─────┐ ┌─────▼─────┐ ┌────▼────┐ │ scope.txt │ │ Recon JSON│ │ Notes / │ │ allowlist │ │ / outputs │ │ reports │ └───────────┘ └───────────┘ └─────────┘ │ ┌────────▼────────┐ │ Tool wrappers │ │ DNS / HTTP etc. │ └─────────────────┘
The key design choice is: don't give the model unrestricted shell access. Instead, expose small approved functions such as:
pythonresolve_domain(target) probe_http(target) read_recon_results(path) summarize_response(path) check_scope(target)
Each network-related function should first do something like:
pythonif not scope.allows(target): raise PermissionError("Target is outside authorized scope")
This prevents the model from turning a vague prompt into activity against an unrelated system.
A practical workflow is:
scope.txt.For example, instead of asking:
textHack example.com and find vulnerabilities.
use:
textHere is recon.json from an authorized bug-bounty program. Analyze the data only. Identify: - unusual exposed services - interesting subdomains - unexpected redirects - authentication surfaces - API endpoints - potentially sensitive files - endpoints worth manual review For each item, explain the reasoning and cite the relevant hostname/URL from the input. Do not invent endpoints.
For HTTP analysis:
textAnalyze this HTTP request and response from an authorized target. Tell me: 1. What the endpoint appears to do 2. Which user-controlled inputs exist 3. Which security controls are visible 4. What security properties I should manually verify 5. Which observations are evidence vs assumptions
For report generation:
textTurn these testing notes into a bug-bounty report containing: - Title - Summary - Affected asset - Preconditions - Reproduction steps - Observed result - Expected result - Security impact - Evidence - Remediation suggestions Do not exaggerate severity or claim impact not demonstrated by the notes.
If you have an NVIDIA GPU, roughly speaking, a 7B–14B coding model is a comfortable starting point for a personal bounty assistant. With only CPU/RAM, quantized 7B-class models are much more practical. For bigger hardware, 30B+ models can substantially improve code and reasoning quality.
I can also build you a complete local “Bug Bounty AI” project in Python that connects Ollama to the scope-aware recon tool we wrote earlier, reads scope.txt and recon JSON, and provides commands such as analyze, triage, ask, and report.
how to i train that ai
Yes. For the local bug-bounty assistant we discussed, I’d use three layers, in this order:
You almost certainly do not want to train a model from scratch. LoRA/PEFT updates only a relatively small set of parameters, making adaptation much more practical than full-model training. Hugging Face currently supports LoRA directly through PEFT and TRL's supervised fine-tuning trainer. (Hugging Face)
For a bug-bounty assistant, useful training examples would be things like:
textRecon output -> prioritized observations HTTP request/response -> security review Raw notes -> structured finding Finding -> severity reasoning Finding -> remediation Program scope -> allowed/not allowed decision Messy report -> high-quality report False positive -> explanation of why it isn't a vulnerability
Don't try to make the model memorize every security fact. Documentation and frequently changing information are better handled through RAG.
For example, the model could learn:
textInput: The server returns Access-Control-Allow-Origin reflecting arbitrary origins, but Access-Control-Allow-Credentials is absent. Output: Do not immediately report this as account-impacting CORS. Verify whether sensitive unauthenticated data can be read cross-origin. The absence of credential support limits common authenticated impact.
That's much more useful training data than dumping thousands of random vulnerability articles into it.
Make a folder:
bashmkdir bounty-ai-training cd bounty-ai-training
Create train.jsonl:
json{"messages":[{"role":"system","content":"You are an assistant for authorized bug bounty testing. Distinguish evidence from assumptions and respect explicit scope."},{"role":"user","content":"Analyze this observation: GET /api/profile returns 401 without a session cookie."},{"role":"assistant","content":"This alone is not a vulnerability. It indicates the endpoint appears to require authentication. Next, document the authentication mechanism and compare behavior using legitimately authorized test accounts. Do not claim an authentication bypass without evidence."}]} {"messages":[{"role":"system","content":"You are an assistant for authorized bug bounty testing. Distinguish evidence from assumptions and respect explicit scope."},{"role":"user","content":"Turn these notes into a finding: User A changes /api/orders/123 to /api/orders/124 and receives User B's shipping address."},{"role":"assistant","content":"Title: Unauthorized Access to Another User's Order via Object Identifier\n\nSummary: An authenticated user can retrieve another user's order by changing the order identifier in the request.\n\nEvidence: User A requested /api/orders/124 and received order data belonging to User B, including the shipping address.\n\nImpact: An authenticated user may access another user's order information.\n\nRemediation: Enforce server-side authorization checks for every requested order object based on the authenticated user's permissions."}]} {"messages":[{"role":"system","content":"You are an assistant for authorized bug bounty testing. Distinguish evidence from assumptions and respect explicit scope."},{"role":"user","content":"Scope contains example.com and its subdomains. Is api.example.com allowed? Is example.net allowed?"},{"role":"assistant","content":"api.example.com is within the stated example.com subdomain scope. example.net is not. Do not perform active testing against example.net unless it is explicitly added to the authorized scope."}]}
The messages conversational structure is supported by TRL's SFT tooling. TRL can also calculate training loss only on assistant responses for compatible chat templates. (Hugging Face)
The important part is quality, not sheer volume.
I would start by manually creating perhaps:
text100-300 excellent examples -> experiment 500-2,000 good examples -> useful specialized behavior thousands+ -> potentially much stronger specialization
Those aren't hard thresholds. Fifty excellent examples can sometimes be more useful than thousands of low-quality generated examples.
Keep a separate eval.jsonl containing examples the model never trains on. Otherwise you won't know whether it learned a pattern or merely memorized your dataset.
Create a separate Python environment:
bashpython3 -m venv .venv source .venv/bin/activate
Install the Hugging Face stack:
bashpip install torch transformers datasets accelerate peft trl
PEFT is specifically intended for adapting pretrained models while training far fewer parameters than full fine-tuning. (Hugging Face)
If you're using an NVIDIA GPU and want quantized training, you'll commonly also use:
bashpip install bitsandbytes
Suppose the base model is:
textQwen/Qwen2.5-Coder-7B-Instruct
Create train.py:
pythonfrom datasets import load_dataset from peft import LoraConfig from trl import SFTConfig, SFTTrainer MODEL = "Qwen/Qwen2.5-Coder-7B-Instruct" # --------------------------------------------------------- # Dataset # --------------------------------------------------------- dataset = load_dataset( "json", data_files={ "train": "train.jsonl", "test": "eval.jsonl", }, ) # --------------------------------------------------------- # LoRA # --------------------------------------------------------- lora_config = LoraConfig( r=16, lora_alpha=32, lora_dropout=0.05, # Apply LoRA broadly across transformer linear layers. target_modules="all-linear", bias="none", task_type="CAUSAL_LM", ) # --------------------------------------------------------- # Training settings # --------------------------------------------------------- training_config = SFTConfig( output_dir="./bounty-ai-lora", # Small batch + accumulation is friendlier to limited VRAM. per_device_train_batch_size=1, gradient_accumulation_steps=8, num_train_epochs=2, # LoRA adapters commonly tolerate a higher LR than # full-model fine tuning. learning_rate=1e-4, logging_steps=10, save_steps=100, # Combine short examples efficiently. packing=True, ) # --------------------------------------------------------- # Trainer # --------------------------------------------------------- trainer = SFTTrainer( model=MODEL, args=training_config, train_dataset=dataset["train"], eval_dataset=dataset["test"], peft_config=lora_config, ) # --------------------------------------------------------- # Train # --------------------------------------------------------- trainer.train() # --------------------------------------------------------- # Save LoRA adapter # --------------------------------------------------------- trainer.save_model("./bounty-ai-lora") print("Finished.") print("Adapter saved to ./bounty-ai-lora")
Hugging Face explicitly documents SFTTrainer + LoraConfig integration, and its PEFT documentation describes target_modules="all-linear" as the QLoRA-style approach for applying LoRA across transformer linear layers. (Hugging Face)
Run:
bashpython train.py
You'll end up with something like:
textbounty-ai-training/ ├── train.py ├── train.jsonl ├── eval.jsonl └── bounty-ai-lora/ ├── adapter_config.json ├── adapter_model.safetensors └── ...
The important distinction is that this is an adapter, not another entire 7-billion-parameter model.
Create test prompts that did not appear in training.
For example:
textThe target sets X-Frame-Options: SAMEORIGIN. Is clickjacking automatically a valid finding?
You want an answer resembling:
textNo. The header itself is evidence of an anti-framing control. Verify whether CSP frame-ancestors provides additional policy and whether the target can actually be embedded before concluding that clickjacking is possible.
Another:
textRecon found dev.example.com, but scope.txt only lists production.example.net. What should I do?
Desired behavior:
textDo not actively probe dev.example.com. It is not authorized by the provided scope.
Test for both:
textgood security reasoning AND good restraint when evidence is insufficient
The second part is extremely useful for bug bounty work because an AI that confidently invents impact will waste your time.
Suppose you want the assistant to know:
textOWASP documentation your personal methodology program scope Burp notes previous reports program policy API documentation recon results
I'd split those like this:
textAI knowledge architecture Base model │ ├── General coding/security knowledge │ ▼ LoRA fine-tune │ ├── Your reasoning style ├── Your report structure ├── Scope discipline ├── False-positive handling └── Recon triage style │ ▼ RAG knowledge base │ ├── Current program scope ├── Current OWASP docs ├── Program rules ├── Recon JSON ├── Your notes └── Previous reports
That means you don't need to retrain the model whenever:
textscope changes a new host appears documentation changes a bounty program changes its rules you collect new recon
You just update the RAG database.
Don't blindly put private bounty information into training data.
Before training, remove:
textsession cookies API keys access tokens passwords private keys real customer PII undisclosed secrets confidential program material you aren't allowed to reuse
For example, convert:
textAuthorization: Bearer eyJhbGciOi...
to:
textAuthorization: Bearer <TOKEN>
And replace identities:
with:
A weak example is:
textUser: Is this IDOR? Assistant: Yes.
A much better one is:
textUser: User A sends: GET /api/orders/481 Changing it to: GET /api/orders/482 returns User B's order. Is this an IDOR?
Then train an answer like:
textThe observation is consistent with an object-level authorization failure because changing an object identifier allowed User A to retrieve an object belonging to User B. Evidence demonstrated: - User A was authenticated. - User A requested another object's identifier. - The application returned User B's order data. Before finalizing the report, confirm that User A was not intentionally authorized to access User B's order and document the two test accounts and requests.
You're teaching the model:
textobservation ↓ evidence ↓ security property ↓ missing verification ↓ supported conclusion
That pattern is much more valuable than teaching it vulnerability names.
There are two common approaches after training:
textLoRA adapter ↓ merge with base model ↓ convert/export GGUF ↓ Ollama
or, where the model/runtime combination supports it:
textbase model + adapter ↓ Ollama
Ollama supports importing GGUF models through a Modelfile, so a finished compatible GGUF can be created as an Ollama model and run locally. (Ollama)
Conceptually:
textFROM ./bounty-ai.Q4_K_M.gguf SYSTEM """ You are an assistant for authorized bug bounty research. Treat scope as a hard authorization boundary. Separate observations from assumptions. Never invent evidence. Prioritize reproducibility. """
Then:
bashollama create bounty-ai -f Modelfile ollama run bounty-ai
The useful end-state isn't simply a "hacking LLM." I'd make it a system like:
text┌──────────────────┐ │ bounty-ai LLM │ │ Qwen + your LoRA │ └────────┬─────────┘ │ ┌───────────────┼───────────────┐ ▼ ▼ ▼ Scope Guard RAG Tool Router │ │ │ scope.txt methodology/ approved notes/reports functions │ │ └───────────────┬───────────────┘ ▼ recon_orchestrator.py
Then you could say:
textAnalyze yesterday's authorized recon.
and the AI could ingest the JSON, compare everything against scope.txt, cluster the results, identify what deserves human review, and draft reports—without giving the model unrestricted shell access.
If you tell me your GPU model and RAM (for example RTX 4060 8GB + 32GB RAM, or CPU-only), I can give you the exact model size, LoRA/QLoRA configuration, batch settings, and training commands appropriate for your machine.
can you guide me for good bug bounty hunter, and add all resource all link to learn nood to pro, add all should i learn. all what should i learn
Yes. If your goal is to become a serious bug bounty hunter from beginner → advanced, I would follow a structured path instead of jumping between random YouTube videos and recon tools.
The most important rule from day one: only test assets explicitly authorized by a bug-bounty/VDP scope or systems you own/labs designed for security testing. Also read each program's rules before touching it.
| Stage | Learn | Goal |
|---|---|---|
| 0 | Scope, ethics, reporting | Understand what you're legally allowed to test |
| 1 | Linux, networking, DNS, Git | Understand the environment |
| 2 | HTTP + web development | Understand how applications actually work |
| 3 | Burp Suite + browser DevTools | Manually inspect and modify traffic |
| 4 | Core web vulnerabilities | Understand root causes, not payload memorization |
| 5 | PortSwigger/OWASP labs | Build practical skill |
| 6 | APIs + modern authentication | Hunt modern applications |
| 7 | Recon | Discover interesting attack surface |
| 8 | JavaScript + Python | Read applications and build automation |
| 9 | Business logic + access control | Develop real bounty-hunting skill |
| 10 | Public reports | Learn how real vulnerabilities were discovered |
| 11 | Specialization | APIs, mobile, cloud, source review, etc. |
| 12 | Real bounty programs | Start hunting carefully and methodically |
Before learning vulnerabilities, understand:
Scope
You need to recognize things like:
textIn scope: *.example.com api.example.com Out of scope: thirdparty.example.net DoS social engineering physical attacks
Don't assume that because one domain belongs to a company, you're authorized to test everything associated with that company.
HackerOne has a getting-started guide specifically for researchers. (HackerOne Help Center)
Learn how reports work too. HackerOne recommends reports with a clear title, reproducible steps, demonstrated impact, evidence, and remediation where possible. (HackerOne Help Center)
HackerOne Quality Reports Guide
Bugcrowd's Vulnerability Rating Taxonomy is also excellent for understanding how vulnerability classes and severity are categorized. The current VRT is version 1.19 as of July 2026. (Bugcrowd)
Bugcrowd Vulnerability Rating Taxonomy
You don't need to become a Linux administrator, but you should be comfortable living in a terminal.
Learn:
textpwd ls cd cp mv rm mkdir cat less head tail grep sort uniq cut awk sed find xargs curl wget ssh chmod chown ps kill jobs pipes | redirects > >> environment variables apt pip venv git
Ubuntu has a good beginner command-line tutorial. (Ubuntu)
Ubuntu Linux Command Line Tutorial
You should eventually be comfortable doing things like:
bashcat domains.txt | sort -u
and:
bashgrep "/api/" urls.txt
without needing to look up every command.
You don't need CCNA-level networking initially.
Understand:
textIP addresses IPv4 / IPv6 TCP UDP ports DNS A AAAA CNAME MX TXT NS domain subdomain TCP handshake TLS HTTPS proxy reverse proxy CDN load balancer NAT firewall
You should be able to explain this:
textBrowser ↓ DNS resolution ↓ IP address ↓ TCP/TLS connection ↓ HTTP request ↓ Reverse proxy/CDN ↓ Web application ↓ Database/API
Spend significant time here.
HTTP is the foundation of web bug bounty. MDN has one of the best introductions. (MDN Web Docs)
Understand requests like:
httpPOST /api/profile HTTP/1.1 Host: example.com Cookie: session=abc123 Content-Type: application/json Authorization: Bearer token123 { "username": "alice" }
And responses:
httpHTTP/1.1 200 OK Content-Type: application/json Set-Cookie: session=xyz Cache-Control: no-store { "username": "alice" }
Learn thoroughly:
textGET POST PUT PATCH DELETE OPTIONS HEAD status codes 200 201 204 301 302 400 401 403 404 405 429 500 headers Host Cookie Set-Cookie Authorization Origin Referer Content-Type Content-Length query parameters request body JSON XML multipart/form-data URL encoding
This knowledge matters far more than knowing hundreds of payloads.
Learn:
textsessions cookies session IDs Secure HttpOnly SameSite Domain Path CSRF tokens JWTs access tokens refresh tokens
MDN has an excellent cookie guide. (MDN Web Docs)
And secure cookie configuration:
MDN Secure Cookie Configuration
This becomes extremely important for:
textXSS CSRF CORS clickjacking postMessage XS-Leaks OAuth cookie attacks
Learn the Same-Origin Policy deeply.
An origin is essentially:
textscheme + hostname + port
For example:
texthttps://example.com
and:
texthttps://api.example.com
are different origins because the hostname differs. MDN provides detailed coverage. (MDN Web Docs)
Then learn CORS. (MDN Web Docs)
You should eventually understand why:
httpAccess-Control-Allow-Origin: *
doesn't automatically equal a critical vulnerability.
This is where many beginners make a mistake.
To break applications intelligently, learn how developers build them.
Understand:
textHTML CSS basics JavaScript DOM forms fetch() XMLHttpRequest JSON REST APIs frontend vs backend databases authentication authorization sessions routing middleware WebSockets
You don't need to become a professional frontend developer.
But JavaScript is extremely important.
MDN's JavaScript material is excellent. (MDN Web Docs)
Especially learn:
javascriptfetch()
Promises:
javascriptasync await
Objects:
javascriptconst user = { id: 123, admin: false }
DOM:
javascriptdocument.querySelector()
Sources/sinks:
javascriptinnerHTML location document.cookie postMessage()
Later, reading JavaScript bundles becomes valuable for reconnaissance.
For web bounty hunting, Burp is one of your most important tools.
Start with the free Community Edition. (PortSwigger)
Download Burp Suite Community Edition
Learn:
textProxy HTTP history Repeater Decoder Comparer Intruder basics Sequencer Extensions Match/replace Scope configuration
The most important one initially:
You'll spend a lot of time doing:
textBrowser ↓ Burp ↓ Intercept request ↓ Send to Repeater ↓ Modify parameter ↓ Send ↓ Compare response
Don't immediately install dozens of extensions.
First learn manual testing.
If you use only one learning resource, choose this.
It is free, continuously updated, and has interactive labs covering beginner through advanced web security. (PortSwigger)
PortSwigger Web Security Academy
Start with:
Study roughly in this order:
textSQL injection Authentication Path traversal Command injection Business logic vulnerabilities Information disclosure Access control / IDOR File upload vulnerabilities Race conditions SSRF XXE XSS CSRF CORS Clickjacking DOM vulnerabilities WebSockets JWT attacks OAuth authentication GraphQL SSTI Insecure deserialization HTTP Host header attacks HTTP request smuggling Web cache poisoning Web cache deception
Do not simply read the solution.
Try the lab yourself.
Get stuck.
Think.
Try again.
Then read hints.
That struggle is part of developing the skill.
Know it, but don't treat it as a complete bug-hunting methodology.
The current OWASP Top 10 is the 2025 edition, including Broken Access Control, Security Misconfiguration, Supply Chain Failures, Cryptographic Failures, Injection, Insecure Design, Authentication Failures and other major categories. (OWASP Top 10)
Then read the OWASP Web Security Testing Guide.
The stable WSTG release is 4.2, while OWASP is developing version 5.0. (OWASP Foundation)
OWASP Web Security Testing Guide
That guide can eventually become your testing checklist.
Here's the core knowledge map:
| Area | Learn |
|---|---|
| Authorization | IDOR/BOLA, horizontal/vertical privilege escalation |
| Authentication | login, registration, password reset, MFA |
| Sessions | cookies, tokens, fixation, invalidation |
| XSS | reflected, stored, DOM |
| Injection | SQLi, command injection, SSTI |
| Server requests | SSRF |
| Files | traversal, upload vulnerabilities |
| XML | XXE |
| Browser security | SOP, CORS, CSRF, clickjacking |
| APIs | BOLA, BFLA, mass assignment, excessive data exposure |
| OAuth | redirect URI, state, account linking, token handling |
| JWT | signature/verification mistakes, claims |
| Logic | workflow bypasses, price manipulation, role issues |
| Concurrency | race conditions |
| HTTP | request smuggling/desync, Host header |
| Caching | cache poisoning, cache deception |
| Serialization | insecure deserialization |
| GraphQL | authorization, introspection, object access |
| WebSockets | authentication/authorization/input handling |
Don't merely learn:
textpayload → vulnerability
Learn:
textsecurity assumption ↓ application behavior ↓ user-controlled input ↓ trust boundary ↓ security control ↓ control failure ↓ impact
That mindset is what eventually separates strong hunters from payload copy-pasters.
Put extra effort here.
Understand:
textUser A ↓ GET /api/orders/100 User B ↓ GET /api/orders/101
Then ask:
textCan User A request /api/orders/101?
But go much deeper than numeric IDs.
Look at:
textUUIDs username email organization ID team ID workspace ID file ID invoice ID project ID GraphQL object ID nested API paths
Learn the difference between:
textauthentication: Who are you? authorization: Are you allowed to do this?
These are fundamental bug-bounty concepts.
Learn complete workflows:
textregistration email verification login logout forgot password password reset change password change email MFA enable MFA disable backup codes remember-me session invalidation account recovery OAuth login
Don't test only the login endpoint.
Think of authentication as a state machine.
For example:
textunverified ↓ verified ↓ logged in ↓ MFA verified ↓ sensitive action
Then ask whether transitions can be skipped.
This is one of the most important advanced areas because scanners often cannot understand application intent.
Think about:
textCan quantity become negative? Can something be purchased twice? Can discounts stack? Can workflow steps be skipped? Can invitations be reused? Can a cancelled object still be modified? Can a user approve their own request? Can an operation be performed in the wrong order? Can two requests race against each other? Does server-side state match frontend assumptions?
You are no longer asking:
"What payload should I use?"
You're asking:
"What assumption did the developer make?"
That's the direction you want to move.
Modern bounty programs contain huge amounts of API functionality.
Understand:
textREST JSON GET /users/123 POST /users PATCH /users/123 DELETE /users/123 Bearer tokens API keys pagination nested objects GraphQL WebSockets
Read the OWASP API Security project. (OWASP API Security Top 10)
Then practice on OWASP crAPI, an intentionally vulnerable application built specifically around API vulnerabilities. (OWASP Foundation)
Two excellent choices:
It intentionally contains many vulnerabilities and challenges across the OWASP categories. (OWASP Foundation)
WebGoat is another deliberately insecure teaching application specifically designed to let people practice web security safely. (OWASP Foundation)
These are excellent places to experiment freely.
After PortSwigger, use these.
Hundreds of exercises across XSS, APIs, SSRF, SQL injection, code review, authentication and other topics. (Pentesterlab)
Their beginner Web for Pentester material is useful too. (Pentesterlab)
PentesterLab Web for Pentester
Its Web Requests course teaches HTTP, HTTPS, curl and browser developer tools. (HTB Academy)
Their web material covers fundamentals, enumeration, authentication, IDOR, SSRF, XSS, race conditions, command injection, SQLi and Burp. (TryHackMe)
Free material covering vulnerability classes, real-world examples, testing advice and bounty reporting. (Intigriti)
Bugcrowd provides free material aimed specifically at becoming a bug bounty hunter. (Bugcrowd)
You don't need advanced software engineering initially.
Learn enough Python to automate boring work.
Official Python tutorial:
Python Tutorial (Python documentation)
Learn:
textvariables strings lists sets dicts if for while functions files JSON regex exceptions argparse requests/aiohttp asyncio subprocess concurrent.futures
Eventually you should comfortably write:
pythonimport requests r = requests.get("https://example.test") print(r.status_code) for key, value in r.headers.items(): print(key, value)
And programs that:
textread targets validate scope call APIs parse JSON deduplicate URLs extract parameters organize results
That's much more valuable than blindly downloading someone's enormous bash recon script.
JavaScript knowledge can give you a significant advantage.
Learn to read:
javascriptfetch("/api/users/" + userId) localStorage.getItem("token") window.addEventListener("message", handler) element.innerHTML = input
Learn to search JavaScript for:
text/api/ /graphql Authorization Bearer token secret admin internal debug upload userId redirect postMessage innerHTML
Eventually learn source maps:
text*.map
frontend frameworks:
textReact Vue Angular Next.js
and bundled/minified JavaScript.
You don't have to master each framework.
You need to understand how data moves through them.
Understand:
sqlSELECT INSERT UPDATE DELETE WHERE AND OR JOIN ORDER BY LIMIT
And understand conceptually why:
textuser input ↓ SQL query
can become dangerous when data and code aren't separated correctly.
Don't just memorize:
text' OR 1=1 --
Understand parameterized queries.
Once you understand applications, start reconnaissance.
Learn the ideas:
textasset discovery subdomain discovery DNS resolution HTTP probing historical URLs JavaScript discovery endpoint collection parameter discovery technology fingerprinting content discovery
A simple workflow looks like:
textprogram scope ↓ passive asset discovery ↓ scope filtering ↓ DNS resolution ↓ HTTP probing ↓ URL collection ↓ JavaScript/API discovery ↓ categorization ↓ manual investigation
Never allow recon automation to escape the authorized scope.
Don't install these all at once. Learn them gradually.
Subfinder — passive subdomain enumeration. (GitHub)
httpx — HTTP probing. (GitHub)
OWASP Amass — attack-surface mapping and asset discovery. (GitHub)
gau — historical/known URL collection from sources such as Common Crawl and Wayback. (GitHub)
SecLists — wordlists useful for authorized security testing. (GitHub)
Later investigate:
textffuf dnsx katana nuclei waybackurls unfurl qsreplace
But understand their purpose before chaining them together.
Use both:
textBurp Suite + Browser DevTools
Learn:
textNetwork tab Sources Console Application Storage Cookies Local Storage Session Storage WebSockets DOM inspector
A surprising amount of bounty hunting is simply:
textclick feature ↓ watch network requests ↓ understand API ↓ compare users/roles ↓ modify request ↓ observe server behavior
This is extremely valuable once you understand the basics.
HackerOne's Hacktivity lets researchers study disclosed reports and filter by weaknesses/programs. (HackerOne Help Center)
When reading a report, don't just ask:
What payload did they use?
Ask:
textHow did they discover the functionality? What assumption did they challenge? What request mattered? What changed between normal and vulnerable behavior? Why was there actual security impact? Could I recognize this pattern elsewhere?
Create personal notes:
textVulnerability: IDOR Interesting functionality: Export/download endpoints Pattern: Object ID supplied by client Testing idea: Compare access using two separate accounts Impact: Cross-user data exposure
You're building a mental vulnerability-pattern database.
Once you understand a vulnerability, these are useful references.
HackTricks has extensive security methodology/reference material. (HackTricks)
PayloadsAllTheThings contains references and examples for many web vulnerability categories.
But don't turn either into:
textcopy payload paste payload repeat
Use them after understanding the underlying vulnerability.
Two good beginner-oriented books are worth considering.
Bug Bounty Bootcamp — Vickie Li
It was written specifically to introduce beginners to bug hunting and web security. (No Starch Press)
Bug Bounty Bootcamp information
Real-World Bug Hunting — Peter Yaworski
It uses real bug bounty cases across XSS, IDOR, SSRF, OAuth, race conditions, SQL injection and other areas. (No Starch Press)
The specific technologies in older books may age; the reasoning patterns remain useful.
Don't pick:
text*.giant-company.com
and immediately enumerate 100,000 hosts.
Choose a program with understandable scope.
Read:
textscope out-of-scope assets allowed techniques forbidden techniques rate limits reporting requirements safe-harbor language
Then pick one application.
Create two test accounts if permitted:
textAccount A Account B
Map functionality:
textregistration login profile settings password email teams organizations invites uploads messages billing exports API admin-like functions
Then study each one carefully.
For every feature, think:
textWho can access this? Who owns this object? Can another user access it? Can another role access it? Can parameters be omitted? Can parameters be added? Can the request method change? Can a request be replayed? Can operations happen out of order? Can the same operation happen twice? Does the backend trust frontend validation? Are hidden parameters accepted? Is authorization enforced server-side? What happens after logout? What happens after password change? What happens when an object is deleted? Can two requests race? Can I manipulate another tenant/organization's object?
This is how you start finding bugs scanners miss.
For each target:
texttarget/ ├── scope.txt ├── notes.md ├── accounts.md ├── endpoints.txt ├── interesting-requests/ ├── screenshots/ ├── javascript/ ├── recon/ └── reports/
Document observations like:
text/api/profile GET Authenticated Returns current user's profile /api/users/{id} GET Authenticated Interesting: object ID controlled by client /api/invite POST Admin only? Needs authorization testing
Good notes compound over time.
A report should generally contain:
textTitle Summary Affected asset Prerequisites Steps to reproduce HTTP requests/responses Observed behavior Expected behavior Security impact Supporting evidence Remediation suggestion
Avoid exaggerated claims.
If you demonstrated:
textUser A can read User B's email address
report that.
Don't claim:
textFull company database compromise
unless you actually demonstrated evidence supporting it.
Once your web fundamentals are strong, choose one or two areas.
Learn:
textREST GraphQL BOLA BFLA mass assignment JWT OAuth rate limiting multi-tenant authorization
Learn:
textDOM XSS postMessage prototype pollution client-side path traversal CSP XS-Leaks source maps
Learn:
textOAuth 2.0 OIDC SAML JWT MFA account recovery passwordless authentication SSO
Eventually:
textAWS Azure GCP IAM object storage serverless cloud metadata
Eventually:
textAndroid iOS APK analysis mobile APIs deep links WebViews certificate handling
Learn to review:
textJavaScript / TypeScript Python PHP Java Go
PentesterLab now has substantial code-review exercises as well. (Pentesterlab)
If you can spend around 2 hours/day, I'd structure it like this:
| Month | Focus |
|---|---|
| 1 | Linux + networking + HTTP + HTML/JS |
| 2 | Burp + PortSwigger beginner labs |
| 3 | XSS, SQLi, IDOR, authentication, CSRF, SSRF |
| 4 | APIs, OAuth/JWT, business logic, race conditions |
| 5 | Recon + JavaScript analysis + automation |
| 6 | Real programs + public report study + specialization |
A good daily routine would be approximately:
text30 min — theory 60 min — labs 20 min — read disclosed reports 10 min — notes
When you start hunting:
text30% learning 70% hunting
is often more useful than permanently staying in tutorial mode.
This is the exact sequence I'd recommend:
Weeks 1–2: Linux, networking, HTTP, cookies, DNS.
Weeks 3–4: HTML + JavaScript basics and Burp Suite.
Months 2–3: Work through PortSwigger Apprentice labs, concentrating heavily on access control, authentication, XSS, SQL injection, SSRF and business logic.
Month 4: APIs, JWT, OAuth, GraphQL and race conditions.
Month 5: Start OWASP Juice Shop/crAPI alongside recon fundamentals and Python automation.
Month 6: Choose a real bug-bounty program, read the scope carefully, manually map one application, and simultaneously read disclosed HackerOne reports related to the technologies you're seeing.
Then move toward PortSwigger Practitioner/Expert material and one specialization.
The biggest shortcut is not a special tool or AI. It's becoming extremely good at looking at a request such as:
httpPATCH /api/v2/organizations/841/users/392
and immediately thinking:
Why does the client control organization
841and user392? Where is ownership checked? What happens with another account, role, tenant, HTTP method, or workflow state?
That mindset is the foundation of good bug hunting.
make me pdf file of that i want to save
I’ll turn the roadmap into a clean, saveable PDF with headings, tables, and the resource links preserved.
I’m formatting it as a practical handbook rather than a raw chat transcript: cover page, roadmap tables, numbered learning stages, practice plan, and a consolidated resource section with clickable links.
The PDF is generated. I’m doing a visual pass now to catch page-break, table, link-text, or clipping issues before I give it to you.
Done — I created a polished 15-page Bug Bounty Hunter Beginner-to-Advanced Roadmap PDF, including the learning path, practice plan, tools, vulnerability topics, and clickable resource links.