96 lines
3.5 KiB
Python
96 lines
3.5 KiB
Python
"""LogFormat directive -> compiled regex + named groups (Chapter 07).
|
|
|
|
One code path for Common, Combined, and arbitrary custom formats — presets
|
|
below are just specific format strings run through this same compiler.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import re
|
|
from dataclasses import dataclass
|
|
from datetime import datetime, timezone
|
|
from ipaddress import ip_address
|
|
|
|
# Directive -> (group name, regex pattern). NOTE: %r and %{...}i patterns
|
|
# deliberately do NOT include quote characters — the quotes around them
|
|
# are literal text already present in the format string (e.g. `"%r"`),
|
|
# and get regex-escaped by the literal-text path below. %t is different:
|
|
# its brackets are part of what %t itself produces, not literal format-
|
|
# string text, so they belong in this directive's own pattern.
|
|
_SIMPLE_DIRECTIVES: dict[str, tuple[str, str]] = {
|
|
"%h": ("ip", r"(?P<ip>\S+)"),
|
|
"%l": ("ident", r"(?P<ident>\S+)"),
|
|
"%u": ("user", r"(?P<user>\S+)"),
|
|
"%t": ("timestamp", r"\[(?P<timestamp>[^\]]+)\]"),
|
|
"%r": ("request", r'(?P<request>[^"]*)'),
|
|
"%>s": ("status", r"(?P<status>\d{3})"),
|
|
"%s": ("status", r"(?P<status>\d{3})"),
|
|
"%b": ("bytes", r"(?P<bytes>\d+|-)"),
|
|
}
|
|
|
|
_HEADER_DIRECTIVE_RE = re.compile(r'%\{([^}]+)\}i')
|
|
_KNOWN_HEADER_GROUPS = {"referer": "referrer", "user-agent": "user_agent"}
|
|
_DIRECTIVE_TOKEN_RE = re.compile(r"%>?\{[^}]+\}i|%>?[a-zA-Z]")
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class CompiledFormat:
|
|
pattern: re.Pattern
|
|
group_names: frozenset[str]
|
|
|
|
|
|
def compile_format(format_string: str) -> CompiledFormat:
|
|
"""Compile a LogFormat directive (Common/Combined preset or a pasted
|
|
custom vhost format) into one regex."""
|
|
regex_parts: list[str] = []
|
|
group_names: set[str] = set()
|
|
pos = 0
|
|
for match in _DIRECTIVE_TOKEN_RE.finditer(format_string):
|
|
literal = format_string[pos:match.start()]
|
|
if literal:
|
|
regex_parts.append(re.escape(literal))
|
|
regex_parts.append(_directive_to_pattern(match.group(0), group_names))
|
|
pos = match.end()
|
|
regex_parts.append(re.escape(format_string[pos:]))
|
|
|
|
pattern = re.compile("^" + "".join(regex_parts) + r"\s*$")
|
|
return CompiledFormat(pattern=pattern, group_names=frozenset(group_names))
|
|
|
|
|
|
def _directive_to_pattern(token: str, group_names: set[str]) -> str:
|
|
header_match = _HEADER_DIRECTIVE_RE.fullmatch(token)
|
|
if header_match:
|
|
header_name = header_match.group(1).lower()
|
|
group = _KNOWN_HEADER_GROUPS.get(header_name, re.sub(r"[^a-z0-9]+", "_", header_name))
|
|
group_names.add(group)
|
|
return f'(?P<{group}>[^"]*)'
|
|
|
|
if token not in _SIMPLE_DIRECTIVES:
|
|
raise ValueError(f"Unsupported LogFormat directive: {token!r}")
|
|
group, pattern = _SIMPLE_DIRECTIVES[token]
|
|
group_names.add(group)
|
|
return pattern
|
|
|
|
|
|
def parse_apache_timestamp(raw: str) -> datetime:
|
|
"""'10/Oct/2026:13:55:36 -0700' -> naive UTC datetime (Ch07: store normalized to UTC)."""
|
|
dt = datetime.strptime(raw, "%d/%b/%Y:%H:%M:%S %z")
|
|
return dt.astimezone(timezone.utc).replace(tzinfo=None)
|
|
|
|
|
|
def parse_request_line(raw: str) -> tuple[str, str] | None:
|
|
"""Split '%r' ("GET /path HTTP/1.1") into (method, path); None if malformed."""
|
|
parts = raw.split()
|
|
if len(parts) != 3:
|
|
return None
|
|
method, path, _protocol = parts
|
|
return method, path
|
|
|
|
|
|
def validate_ip(raw: str) -> str | None:
|
|
"""Return `raw` if a valid IPv4/IPv6 address, else None (Ch07)."""
|
|
try:
|
|
ip_address(raw)
|
|
except ValueError:
|
|
return None
|
|
return raw
|