REGULAR EXPRESSIONS
Authorized use only. Offensive reference for systems you own or are explicitly permitted to test. You are responsible for staying within the law.
BASIC CHARACTERS#
. # Any single character \d # Digit [0-9] \D # Non-digit [^0-9] \w # Word character [a-zA-Z0-9_] \W # Non-word character \s # Whitespace [\t\n\r\f\v ] \S # Non-whitespace \b # Word boundary \B # Non-word boundary \\ # Literal backslash
ANCHORS#
^ # Start of string/line $ # End of string/line \A # Start of string (absolute) \Z # End of string (absolute) \z # End of string (absolute, no newline)
QUANTIFIERS#
* # Zero or more
+ # One or more
? # Zero or one (optional)
{n} # Exactly n times
{n,} # n or more times
{n,m} # Between n and m times
# Greedy vs Lazy
*? # Zero or more (lazy)
+? # One or more (lazy)
?? # Zero or one (lazy)
{n,m}? # Between n and m (lazy)
CHARACTER CLASSES#
[abc] # Any of a, b, or c [^abc] # Not a, b, or c [a-z] # Lowercase letters [A-Z] # Uppercase letters [a-zA-Z] # All letters [0-9] # Digits [a-zA-Z0-9] # Alphanumeric [^a-zA-Z0-9] # Non-alphanumeric # Special characters in classes [\-] # Literal hyphen [\]] # Literal bracket [\\] # Literal backslash
POSIX CLASSES (in brackets)#
[:alnum:] # Alphanumeric [:alpha:] # Alphabetic [:ascii:] # ASCII characters [:blank:] # Space and tab [:cntrl:] # Control characters [:digit:] # Digits [:graph:] # Visible characters [:lower:] # Lowercase letters [:print:] # Printable characters [:punct:] # Punctuation [:space:] # Whitespace [:upper:] # Uppercase letters [:word:] # Word characters [:xdigit:] # Hexadecimal digits # Usage: [[:alpha:]] not [:alpha:]
GROUPS AND CAPTURING#
(abc) # Capturing group (?:abc) # Non-capturing group (?<name>abc) # Named group (?P<name>abc) # Named group (Python) \1, \2, ... # Backreference \k<name> # Named backreference (?P=name) # Named backreference (Python)
ALTERNATION#
a|b # a or b (cat|dog) # cat or dog (?:cat|dog) # Non-capturing alternation
LOOKAROUND#
# Lookahead (?=abc) # Positive lookahead (?!abc) # Negative lookahead # Lookbehind (?<=abc) # Positive lookbehind (?<!abc) # Negative lookbehind # Examples \d+(?=\s*USD) # Digits followed by USD (?<=\$)\d+ # Digits preceded by $ \b\w+(?!\d) # Words not followed by digit (?<!\d)\w+ # Words not preceded by digit
FLAGS/MODIFIERS#
i # Case insensitive g # Global (all matches) m # Multiline (^ and $ match lines) s # Dotall (. matches newlines) x # Verbose (ignore whitespace) u # Unicode # Usage varies by language /pattern/i # JavaScript, Perl (?i)pattern # Inline modifier re.IGNORECASE # Python
SPECIAL SEQUENCES#
\n # Newline \r # Carriage return \t # Tab \v # Vertical tab \f # Form feed \0 # Null character \xHH # Hex character \uHHHH # Unicode character
COMMON PATTERNS#
# Email
[\w.+-]+@[\w.-]+\.[a-zA-Z]{2,}
# URL
https?://[\w.-]+(?:/[\w./-]*)?
# IP Address (IPv4)
\b(?:\d{1,3}\.){3}\d{1,3}\b
# More strict IPv4
\b(?:(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?)\.){3}(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?)\b
# Phone Number (US)
\(?\d{3}\)?[-.\s]?\d{3}[-.\s]?\d{4}
# Date (YYYY-MM-DD)
\d{4}-(?:0[1-9]|1[0-2])-(?:0[1-9]|[12][0-9]|3[01])
# Time (HH:MM:SS)
(?:[01]\d|2[0-3]):[0-5]\d:[0-5]\d
# Hex Color
#(?:[0-9a-fA-F]{3}){1,2}\b
# Credit Card (basic)
\b\d{4}[- ]?\d{4}[- ]?\d{4}[- ]?\d{4}\b
# Username (alphanumeric, 3-16 chars)
^[a-zA-Z0-9_]{3,16}$
# Password (8+ chars, upper, lower, digit)
^(?=.*[a-z])(?=.*[A-Z])(?=.*\d).{8,}$
# HTML Tag
<([a-z]+)([^<]+)*(?:>(.*)<\/\1>|\s+\/>)
# HTML Comment
<!--[\s\S]*?-->
# Blank Lines
^\s*$
# Trailing Whitespace
\s+$
# Leading Whitespace
^\s+
# Duplicate Words
\b(\w+)\s+\1\b
# Quoted String
"[^"]*"|'[^']*'
# File Extension
\.[a-zA-Z0-9]+$
# Windows Path
[a-zA-Z]:\\(?:[^\\/:*?"<>|\r\n]+\\)*[^\\/:*?"<>|\r\n]*
# Unix Path
(?:/[^/\0]+)+/?
LANGUAGE-SPECIFIC#
# Python
import re
re.search(pattern, string)
re.match(pattern, string) # Match at start
re.findall(pattern, string)
re.sub(pattern, replacement, string)
re.compile(pattern)
# JavaScript
/pattern/.test(string)
string.match(/pattern/)
string.replace(/pattern/g, replacement)
string.split(/pattern/)
# Bash (grep, sed)
grep -E 'pattern' file # Extended regex
grep -P 'pattern' file # Perl regex
sed 's/pattern/replacement/g' file
# PHP
preg_match('/pattern/', $string)
preg_match_all('/pattern/', $string, $matches)
preg_replace('/pattern/', $replacement, $string)
ESCAPING SPECIAL CHARACTERS#
# Characters that need escaping in most engines
. \ + * ? [ ] ^ $ ( ) { } | /
# Escape with backslash
\. # Literal dot
\\ # Literal backslash
\+ # Literal plus
\* # Literal asterisk
\? # Literal question mark
\[ # Literal bracket
REGEX FLAVORS#
BRE # Basic Regular Expressions (grep)
ERE # Extended Regular Expressions (egrep, grep -E)
PCRE # Perl Compatible Regular Expressions
POSIX # POSIX standard
JavaScript # ECMAScript regex
Python # Python re module
.NET # .NET regex engine
# Key differences
BRE: \( \) \{ \} # Escape for special meaning
ERE: ( ) { } # Special by default
PCRE: Full features, lookaround, etc.
TIPS AND BEST PRACTICES#
# Be specific, avoid .* when possible \d+ better than .+ for numbers # Use non-capturing groups when you don't need capture (?:pattern) instead of (pattern) # Use lazy quantifiers to match shortest .*? instead of .* to avoid over-matching # Anchor when possible ^pattern$ for full string match # Use character classes for readability [aeiou] instead of (a|e|i|o|u) # Test regex with online tools regex101.com regexr.com