55 lines
1.7 KiB
Python
55 lines
1.7 KiB
Python
"""Suspicious-pattern matching (Chapter 07), reused as-is by Chapter 10.
|
|
|
|
Pattern dictionary is data (JSON), same reasoning as the bot signature
|
|
table — editable without a code change. Severity here is a provisional
|
|
per-rule-type default; Chapter 10 owns the full weighted/escalating score.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
from dataclasses import dataclass
|
|
from pathlib import Path
|
|
from typing import TYPE_CHECKING
|
|
|
|
if TYPE_CHECKING:
|
|
from app.services.log_parser import ParsedEntry
|
|
|
|
PATTERNS_PATH = Path(__file__).parent / "data" / "threat_patterns.json"
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class ThreatMatch:
|
|
rule_matched: str
|
|
severity: str # low|medium|high — provisional; Ch10 may escalate
|
|
|
|
|
|
def _load_patterns() -> dict:
|
|
return json.loads(PATTERNS_PATH.read_text())
|
|
|
|
|
|
_PATTERNS = _load_patterns()
|
|
|
|
|
|
def scan(entry: "ParsedEntry") -> ThreatMatch | None:
|
|
"""Check one parsed entry against the pattern dictionary.
|
|
|
|
Cheap substring checks only — no regex backtracking risk, no network
|
|
calls (Ch03: CPU-cheap, explainable rule matching).
|
|
"""
|
|
path_lower = entry.path.lower()
|
|
ua_lower = (entry.user_agent or "").lower()
|
|
|
|
for scanner_ua in _PATTERNS["scanner_user_agents"]:
|
|
if scanner_ua in ua_lower:
|
|
return ThreatMatch(rule_matched=f"scanner_ua:{scanner_ua}", severity="high")
|
|
|
|
for sensitive in _PATTERNS["sensitive_paths"]:
|
|
if sensitive.lower() in path_lower:
|
|
return ThreatMatch(rule_matched=f"sensitive_path:{sensitive}", severity="medium")
|
|
|
|
for marker in _PATTERNS["injection_markers"]:
|
|
if marker.lower() in path_lower:
|
|
return ThreatMatch(rule_matched=f"injection:{marker}", severity="high")
|
|
|
|
return None
|