← All cheat sheets

KATANA

Authorized use only. Offensive reference for systems you own or are explicitly permitted to test. You are responsible for staying within the law.

Fast web crawler by ProjectDiscovery. Discovers endpoints, URLs,
JavaScript files, and API routes. Designed for security testing
and recon pipelines.

INSTALLATION#

go install github.com/projectdiscovery/katana/cmd/katana@latest

# Homebrew
brew install katana

# Docker
docker pull projectdiscovery/katana

BASIC USAGE#

# Single URL
katana -u https://example.com

# Multiple URLs
katana -u https://example.com -u https://test.com

# From file
katana -list urls.txt

# From stdin
echo "https://example.com" | katana
cat urls.txt | katana

CRAWLING MODES#

# Standard crawling (default)
katana -u URL

# JavaScript parsing (extract endpoints from JS)
katana -u URL -jc                           # JS crawling enabled
katana -u URL -jsluice                      # Enhanced JS parsing

# Headless browser crawling (for SPAs)
katana -u URL -headless
katana -u URL -headless -no-sandbox

# Passive crawling (no direct requests)
katana -u URL -passive
katana -u URL -passive -ps waybackarchive
katana -u URL -passive -ps commoncrawl
katana -u URL -passive -ps alienvault

DEPTH & SCOPE#

# Crawl depth
katana -u URL -d 3                          # Max depth 3 (default 3)
katana -u URL -d 5                          # Deeper crawl

# Scope control
katana -u URL -fs example.com              # Field scope (domain)
katana -u URL -cs ".*\.example\.com"       # Crawl scope regex
katana -u URL -do                           # Disable redirect outside scope

# Include/exclude patterns
katana -u URL -ef "png,jpg,gif,css,svg"    # Exclude extensions
katana -u URL -ef "woff,woff2,ttf,eot"     # Exclude fonts
katana -u URL -em "logout|signout"         # Exclude match
katana -u URL -f "api"                      # Filter (include) match

# Subdomains
katana -u URL -subs                         # Include subdomains

OUTPUT OPTIONS#

# File output
katana -u URL -o output.txt

# JSON output
katana -u URL -jsonl -o output.json

# Store responses
katana -u URL -store-response -store-response-dir ./responses/

# Store specific fields
katana -u URL -f url,path,fqdn              # Output specific fields

# Silent mode
katana -u URL -silent                       # URLs only

# Display fields
katana -u URL -display-out-scope            # Show out-of-scope too

FILTERING OUTPUT#

# By extension
katana -u URL -em ".js"                     # Only JS files
katana -u URL -ef "png,jpg,gif,css"         # Exclude static

# By content type
katana -u URL -ct "application/json"        # Only JSON responses

# By response code
katana -u URL -mdc "status_code == 200"     # Only 200s
katana -u URL -mdc "status_code != 404"     # Exclude 404s

# By response size
katana -u URL -mdc "response_size > 1000"   # Larger responses

# Extract specific patterns
katana -u URL -fx regex                     # Custom regex extraction

PERFORMANCE TUNING#

# Concurrency
katana -u URL -c 20                         # Concurrent requests
katana -u URL -p 10                         # Parallelism (hosts)

# Rate limiting
katana -u URL -rl 50                        # Requests per second
katana -u URL -rd 500ms                     # Request delay

# Timeout
katana -u URL -timeout 15                   # Request timeout

# Retry
katana -u URL -retry 2                      # Retry count

AUTHENTICATION#

# Custom headers
katana -u URL -H "Authorization: Bearer TOKEN"
katana -u URL -H "Cookie: session=abc123"

# Form-based auth with headless
katana -u URL -headless -ffa                # Automatic form fill

PROXY#

katana -u URL -proxy http://127.0.0.1:8080
katana -u URL -proxy socks5://127.0.0.1:9050

JAVASCRIPT ANALYSIS#

# Extract endpoints from JavaScript files
katana -u URL -jc                           # JS crawl mode
katana -u URL -jsluice                      # Enhanced JS parsing

# What JS crawling finds:
  - API endpoints hardcoded in JS
  - Hidden admin routes
  - Internal hostnames and IPs
  - S3 bucket names
  - API keys and tokens
  - WebSocket URLs

HEADLESS CRAWLING#

# For JavaScript-heavy SPAs (React, Angular, Vue)
katana -u URL -headless

# Chrome options
katana -u URL -headless -no-sandbox
katana -u URL -headless -show-browser       # Show browser window
katana -u URL -headless -system-chrome      # Use system Chrome

# Wait for JS to load
katana -u URL -headless -headless-options "waitForTimeout=5000"

PIPELINE INTEGRATION#

# Full recon pipeline
subfinder -d example.com | httpx | katana | nuclei

# Parameter discovery for fuzzing
katana -u URL -jc | grep "=" | dalfox pipe

# Find JavaScript files
katana -u URL | grep "\.js$" | sort -u > js_files.txt

# API endpoint discovery
katana -u URL -jc -silent | grep "/api/" | sort -u

# Feed to Arjun for param discovery
katana -u URL -silent | arjun --urls /dev/stdin

# With wayback data
echo example.com | waybackurls | katana -silent

COMMON WORKFLOWS#

# 1. Discover all endpoints
katana -u https://target.com -d 5 -jc -silent | sort -u

# 2. Find forms and parameters
katana -u URL -jc -silent | grep "=" | sort -u

# 3. JavaScript analysis
katana -u URL -jc -silent | grep "\.js$" | sort -u

# 4. Full authenticated crawl
katana -u URL -d 5 -jc -H "Cookie: session=TOKEN" -o crawl.txt

# 5. Passive + active recon
katana -u URL -passive -ps waybackarchive,commoncrawl -o passive.txt
katana -u URL -d 3 -jc -o active.txt
cat passive.txt active.txt | sort -u > all_urls.txt

# 6. SPA crawling
katana -u URL -headless -d 3 -jc -o spa_urls.txt

TIPS#

  - -jc (JS crawl) is essential for modern web apps
  - Use -ef to exclude static assets and reduce noise
  - Headless mode required for SPAs (React, Angular, Vue)
  - Combine with httpx for live host verification
  - Pipeline: subfinder → httpx → katana → nuclei
  - -passive mode avoids touching the target directly
  - Look for endpoints with parameters (containing "=")
  - JS files often contain hardcoded API keys and internal URLs
  - Set -d 3 to -d 5 for thorough crawling
  - Use proxy to review traffic in Burp/Caido simultaneously