"""Suspicious-pattern matching (Chapter 07), reused as-is by Chapter 10. Pattern dictionary is data (JSON), same reasoning as the bot signature table — editable without a code change. Severity here is a provisional per-rule-type default; Chapter 10 owns the full weighted/escalating score. """ from __future__ import annotations import json from dataclasses import dataclass from pathlib import Path from typing import TYPE_CHECKING if TYPE_CHECKING: from app.services.log_parser import ParsedEntry PATTERNS_PATH = Path(__file__).parent / "data" / "threat_patterns.json" @dataclass(frozen=True) class ThreatMatch: rule_matched: str severity: str # low|medium|high — provisional; Ch10 may escalate def _load_patterns() -> dict: return json.loads(PATTERNS_PATH.read_text()) _PATTERNS = _load_patterns() def scan(entry: "ParsedEntry") -> ThreatMatch | None: """Check one parsed entry against the pattern dictionary. Cheap substring checks only — no regex backtracking risk, no network calls (Ch03: CPU-cheap, explainable rule matching). """ path_lower = entry.path.lower() ua_lower = (entry.user_agent or "").lower() for scanner_ua in _PATTERNS["scanner_user_agents"]: if scanner_ua in ua_lower: return ThreatMatch(rule_matched=f"scanner_ua:{scanner_ua}", severity="high") for sensitive in _PATTERNS["sensitive_paths"]: if sensitive.lower() in path_lower: return ThreatMatch(rule_matched=f"sensitive_path:{sensitive}", severity="medium") for marker in _PATTERNS["injection_markers"]: if marker.lower() in path_lower: return ThreatMatch(rule_matched=f"injection:{marker}", severity="high") return None