Files
Kavosh/app/services/log_parser/format_compiler.py
T
2026-08-07 21:17:17 +03:30

96 lines
3.5 KiB
Python

"""LogFormat directive -> compiled regex + named groups (Chapter 07).
One code path for Common, Combined, and arbitrary custom formats — presets
below are just specific format strings run through this same compiler.
"""
from __future__ import annotations
import re
from dataclasses import dataclass
from datetime import datetime, timezone
from ipaddress import ip_address
# Directive -> (group name, regex pattern). NOTE: %r and %{...}i patterns
# deliberately do NOT include quote characters — the quotes around them
# are literal text already present in the format string (e.g. `"%r"`),
# and get regex-escaped by the literal-text path below. %t is different:
# its brackets are part of what %t itself produces, not literal format-
# string text, so they belong in this directive's own pattern.
_SIMPLE_DIRECTIVES: dict[str, tuple[str, str]] = {
"%h": ("ip", r"(?P<ip>\S+)"),
"%l": ("ident", r"(?P<ident>\S+)"),
"%u": ("user", r"(?P<user>\S+)"),
"%t": ("timestamp", r"\[(?P<timestamp>[^\]]+)\]"),
"%r": ("request", r'(?P<request>[^"]*)'),
"%>s": ("status", r"(?P<status>\d{3})"),
"%s": ("status", r"(?P<status>\d{3})"),
"%b": ("bytes", r"(?P<bytes>\d+|-)"),
}
_HEADER_DIRECTIVE_RE = re.compile(r'%\{([^}]+)\}i')
_KNOWN_HEADER_GROUPS = {"referer": "referrer", "user-agent": "user_agent"}
_DIRECTIVE_TOKEN_RE = re.compile(r"%>?\{[^}]+\}i|%>?[a-zA-Z]")
@dataclass(frozen=True)
class CompiledFormat:
pattern: re.Pattern
group_names: frozenset[str]
def compile_format(format_string: str) -> CompiledFormat:
"""Compile a LogFormat directive (Common/Combined preset or a pasted
custom vhost format) into one regex."""
regex_parts: list[str] = []
group_names: set[str] = set()
pos = 0
for match in _DIRECTIVE_TOKEN_RE.finditer(format_string):
literal = format_string[pos:match.start()]
if literal:
regex_parts.append(re.escape(literal))
regex_parts.append(_directive_to_pattern(match.group(0), group_names))
pos = match.end()
regex_parts.append(re.escape(format_string[pos:]))
pattern = re.compile("^" + "".join(regex_parts) + r"\s*$")
return CompiledFormat(pattern=pattern, group_names=frozenset(group_names))
def _directive_to_pattern(token: str, group_names: set[str]) -> str:
header_match = _HEADER_DIRECTIVE_RE.fullmatch(token)
if header_match:
header_name = header_match.group(1).lower()
group = _KNOWN_HEADER_GROUPS.get(header_name, re.sub(r"[^a-z0-9]+", "_", header_name))
group_names.add(group)
return f'(?P<{group}>[^"]*)'
if token not in _SIMPLE_DIRECTIVES:
raise ValueError(f"Unsupported LogFormat directive: {token!r}")
group, pattern = _SIMPLE_DIRECTIVES[token]
group_names.add(group)
return pattern
def parse_apache_timestamp(raw: str) -> datetime:
"""'10/Oct/2026:13:55:36 -0700' -> naive UTC datetime (Ch07: store normalized to UTC)."""
dt = datetime.strptime(raw, "%d/%b/%Y:%H:%M:%S %z")
return dt.astimezone(timezone.utc).replace(tzinfo=None)
def parse_request_line(raw: str) -> tuple[str, str] | None:
"""Split '%r' ("GET /path HTTP/1.1") into (method, path); None if malformed."""
parts = raw.split()
if len(parts) != 3:
return None
method, path, _protocol = parts
return method, path
def validate_ip(raw: str) -> str | None:
"""Return `raw` if a valid IPv4/IPv6 address, else None (Ch07)."""
try:
ip_address(raw)
except ValueError:
return None
return raw