start project
This commit is contained in:
@@ -0,0 +1,199 @@
|
||||
"""Rollup computation (Chapter 06 + Method A extensions from Ch08/09/10).
|
||||
|
||||
Called once per parsed file (app/cli.py::process_logs) for the dates it
|
||||
touched, ad hoc via `flask rollup` for a manual recompute, and now also
|
||||
after a file deletion (app/services/file_deletion.py) for whatever dates
|
||||
the deleted file touched. Every _upsert_* function scans log_entries
|
||||
inside this background batch job, never at request time — that's what
|
||||
makes Ch03 rule 6 compliance possible.
|
||||
|
||||
CORRECTNESS FIX: every rollup writer below now deletes a day's existing
|
||||
rows before writing whatever the fresh scan finds (including writing
|
||||
nothing, if a day now has zero data). Three of the five writers
|
||||
previously only ever upserted-when-present and silently left stale rows
|
||||
behind when a day's data disappeared — unreachable before file deletion
|
||||
existed (rollups only ever grew), but a real correctness bug once
|
||||
deletion makes "this day now has less data than before" possible. Only
|
||||
the two per-IP writers already had this right (Ch10 follow-up); the
|
||||
other three are fixed here to match.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from collections import defaultdict
|
||||
from datetime import date, datetime, time, timedelta
|
||||
|
||||
from sqlalchemy import case, func
|
||||
|
||||
from app.extensions import db
|
||||
from app.models.browser_stats import BrowserStatsDaily
|
||||
from app.models.human_path_stats import HumanPathStatsDaily
|
||||
from app.models.ip_traffic_stats import IpPathStatsDaily, IpStatusStatsDaily
|
||||
from app.models.log_entry import LogEntry
|
||||
from app.models.referrer_stats import ReferrerStatsDaily
|
||||
from app.models.request_stats import RequestStatsDaily, RequestStatsHourly
|
||||
from app.services.blocklist import refresh_blocklist_suggestions
|
||||
from app.services.referrer import referrer_domain
|
||||
from app.services.ua_classifier import classify_browser, classify_os
|
||||
from app.utils.http_status import status_bucket
|
||||
|
||||
TOP_PATHS_PER_IP_PER_DAY = 15 # bounds ip_path_stats_daily row growth (Ch10 follow-up)
|
||||
|
||||
|
||||
def compute_rollups_for_range(start: date, end: date) -> None:
|
||||
"""Recompute every rollup for each day in [start, end]. Site-wide,
|
||||
not per-file (Ch01: single site) — recomputing from scratch per day
|
||||
avoids double-counting when two uploads cover the same period, and
|
||||
correctly shrinks a day's numbers back down when a file covering
|
||||
that day is deleted.
|
||||
"""
|
||||
current = start
|
||||
while current <= end:
|
||||
_upsert_hourly(current)
|
||||
_upsert_daily(current)
|
||||
_upsert_per_line_derived_stats(current)
|
||||
refresh_blocklist_suggestions(current)
|
||||
current += timedelta(days=1)
|
||||
|
||||
|
||||
def _day_bounds(day: date) -> tuple[datetime, datetime]:
|
||||
start = datetime.combine(day, time.min)
|
||||
return start, start + timedelta(days=1)
|
||||
|
||||
|
||||
def _upsert_hourly(day: date) -> None:
|
||||
start, end = _day_bounds(day)
|
||||
rows = (
|
||||
db.session.query(
|
||||
func.strftime("%Y-%m-%d %H:00:00", LogEntry.timestamp).label("date_hour"),
|
||||
LogEntry.path,
|
||||
LogEntry.status_code,
|
||||
func.count().label("count"),
|
||||
func.coalesce(func.sum(LogEntry.bytes_sent), 0).label("bytes_sent_sum"),
|
||||
)
|
||||
.filter(LogEntry.timestamp >= start, LogEntry.timestamp < end)
|
||||
.group_by("date_hour", LogEntry.path, LogEntry.status_code)
|
||||
.all()
|
||||
)
|
||||
|
||||
# Delete-then-insert: replaces the day's hourly rows entirely,
|
||||
# including leaving none behind if `rows` is now empty (e.g. the
|
||||
# only file covering this day was just deleted).
|
||||
db.session.query(RequestStatsHourly).filter(
|
||||
RequestStatsHourly.date_hour >= start, RequestStatsHourly.date_hour < end
|
||||
).delete()
|
||||
|
||||
if rows:
|
||||
payload = [
|
||||
{
|
||||
"date_hour": datetime.strptime(r.date_hour, "%Y-%m-%d %H:%M:%S"),
|
||||
"path": r.path,
|
||||
"status_code": r.status_code,
|
||||
"count": r.count,
|
||||
"bytes_sent_sum": r.bytes_sent_sum,
|
||||
}
|
||||
for r in rows
|
||||
]
|
||||
db.session.execute(RequestStatsHourly.__table__.insert(), payload)
|
||||
|
||||
db.session.commit()
|
||||
|
||||
|
||||
def _upsert_daily(day: date) -> None:
|
||||
start, end = _day_bounds(day)
|
||||
result = (
|
||||
db.session.query(
|
||||
func.count().label("count"),
|
||||
func.count(func.distinct(LogEntry.ip)).label("unique_ips"),
|
||||
func.coalesce(func.sum(LogEntry.bytes_sent), 0).label("bytes_sum"),
|
||||
func.coalesce(func.sum(case((LogEntry.status_code >= 400, 1), else_=0)), 0).label("error_count"),
|
||||
)
|
||||
.filter(LogEntry.timestamp >= start, LogEntry.timestamp < end)
|
||||
.one()
|
||||
)
|
||||
|
||||
# Delete-then-insert: if this day now has zero entries (its only
|
||||
# contributing file was deleted), the stale row is removed rather
|
||||
# than left behind — no rollup row is better than a wrong one.
|
||||
db.session.query(RequestStatsDaily).filter(RequestStatsDaily.date == day).delete()
|
||||
|
||||
if result.count > 0:
|
||||
db.session.execute(
|
||||
RequestStatsDaily.__table__.insert(),
|
||||
{
|
||||
"date": day,
|
||||
"count": result.count,
|
||||
"unique_ips": result.unique_ips,
|
||||
"bytes_sum": result.bytes_sum,
|
||||
"error_count": result.error_count,
|
||||
},
|
||||
)
|
||||
|
||||
db.session.commit()
|
||||
|
||||
|
||||
def _upsert_per_line_derived_stats(day: date) -> None:
|
||||
"""Referrer domain, browser/OS, human-only path counts (Ch08/09), and
|
||||
per-IP path/status counts (Ch10 follow-up) — one streamed pass over
|
||||
log_entries (Ch03 rule 1: bounded per-chunk memory via yield_per,
|
||||
never the whole day loaded at once). Every table here uses the same
|
||||
delete-then-insert pattern so a day's rows are fully replaced by
|
||||
whatever the fresh scan finds, including nothing.
|
||||
"""
|
||||
start, end = _day_bounds(day)
|
||||
referrer_counts: dict[str, int] = defaultdict(int)
|
||||
browser_counts: dict[tuple[str, str], int] = defaultdict(int)
|
||||
human_path_counts: dict[str, int] = defaultdict(int)
|
||||
ip_path_counts: dict[str, dict[str, int]] = defaultdict(lambda: defaultdict(int))
|
||||
ip_status_counts: dict[str, dict[str, int]] = defaultdict(lambda: defaultdict(int))
|
||||
|
||||
query = (
|
||||
db.session.query(
|
||||
LogEntry.referrer, LogEntry.user_agent, LogEntry.is_bot,
|
||||
LogEntry.path, LogEntry.ip, LogEntry.status_code,
|
||||
)
|
||||
.filter(LogEntry.timestamp >= start, LogEntry.timestamp < end)
|
||||
)
|
||||
for referrer, user_agent, is_bot, path, ip, status_code in query.yield_per(1000):
|
||||
domain = referrer_domain(referrer)
|
||||
if domain:
|
||||
referrer_counts[domain] += 1
|
||||
if not is_bot: # Ch08: bot traffic excluded from human browser/OS breakdown
|
||||
browser_counts[(classify_browser(user_agent), classify_os(user_agent))] += 1
|
||||
human_path_counts[path] += 1
|
||||
ip_path_counts[ip][path] += 1
|
||||
ip_status_counts[ip][status_bucket(status_code)] += 1
|
||||
|
||||
db.session.query(ReferrerStatsDaily).filter(ReferrerStatsDaily.date == day).delete()
|
||||
if referrer_counts:
|
||||
payload = [{"date": day, "referrer_domain": d, "count": c} for d, c in referrer_counts.items()]
|
||||
db.session.execute(ReferrerStatsDaily.__table__.insert(), payload)
|
||||
|
||||
db.session.query(BrowserStatsDaily).filter(BrowserStatsDaily.date == day).delete()
|
||||
if browser_counts:
|
||||
payload = [{"date": day, "browser": b, "os": o, "count": c} for (b, o), c in browser_counts.items()]
|
||||
db.session.execute(BrowserStatsDaily.__table__.insert(), payload)
|
||||
|
||||
db.session.query(HumanPathStatsDaily).filter(HumanPathStatsDaily.date == day).delete()
|
||||
if human_path_counts:
|
||||
payload = [{"date": day, "path": p, "count": c} for p, c in human_path_counts.items()]
|
||||
db.session.execute(HumanPathStatsDaily.__table__.insert(), payload)
|
||||
|
||||
db.session.query(IpPathStatsDaily).filter(IpPathStatsDaily.date == day).delete()
|
||||
if ip_path_counts:
|
||||
payload = []
|
||||
for ip, paths in ip_path_counts.items():
|
||||
top_paths = sorted(paths.items(), key=lambda kv: kv[1], reverse=True)[:TOP_PATHS_PER_IP_PER_DAY]
|
||||
payload.extend({"date": day, "ip": ip, "path": p, "count": c} for p, c in top_paths)
|
||||
if payload:
|
||||
db.session.execute(IpPathStatsDaily.__table__.insert(), payload)
|
||||
|
||||
db.session.query(IpStatusStatsDaily).filter(IpStatusStatsDaily.date == day).delete()
|
||||
if ip_status_counts:
|
||||
payload = [
|
||||
{"date": day, "ip": ip, "status_bucket": bucket, "count": c}
|
||||
for ip, buckets in ip_status_counts.items()
|
||||
for bucket, c in buckets.items()
|
||||
]
|
||||
db.session.execute(IpStatusStatsDaily.__table__.insert(), payload)
|
||||
|
||||
db.session.commit()
|
||||
@@ -0,0 +1,97 @@
|
||||
"""No-cron automatic processing (Chapter 12 simplification, per project
|
||||
owner request): triggers file parsing immediately in a background thread
|
||||
right after upload, and opportunistically resumes any incomplete files
|
||||
when the Overview page loads — replacing the cron-triggered model.
|
||||
`flask process-logs` still exists in app/cli.py for anyone who'd rather
|
||||
use cron, but nothing requires it anymore.
|
||||
|
||||
TRADEOFF (flagged, deviating from Chapter 02/03's "no persistent
|
||||
background workers, cron-triggered CLI only" stance): a background thread
|
||||
lives inside the same worker process that handled the upload request. If
|
||||
Passenger recycles that process mid-parse, the thread dies with it —
|
||||
progress up to the last commit is still safely checkpointed (same bounded-
|
||||
batch model as before), but nothing will automatically resume it without
|
||||
either cron or a page visit. The "resume on page load" hook below is the
|
||||
deliberate replacement for that guarantee: visiting the Overview tab
|
||||
re-triggers processing for anything left incomplete, so in the worst case
|
||||
a stuck file resumes the next time the admin looks at the dashboard,
|
||||
rather than never.
|
||||
|
||||
This is not a long-lived daemon: each thread terminates once its file
|
||||
reaches "done"/"error" (or the process is killed), and no thread survives
|
||||
a process restart — it just gets re-triggered fresh next time.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import threading
|
||||
|
||||
from flask import Flask
|
||||
|
||||
from app.extensions import db
|
||||
from app.models.log_file import LogFile
|
||||
from app.services.log_processor import process_one_batch
|
||||
|
||||
# In-process guard against launching two threads for the same file at
|
||||
# once (e.g. the upload trigger and a page-load resume firing close
|
||||
# together). Per-worker-process only — a different entry process picking
|
||||
# up the same file concurrently is a low-probability edge case accepted
|
||||
# for this simplification; each write is still a small checkpointed
|
||||
# commit, not a giant one, which limits how bad a collision could be.
|
||||
_active_file_ids: set[int] = set()
|
||||
_lock = threading.Lock()
|
||||
|
||||
|
||||
def _claim(log_file_id: int) -> bool:
|
||||
with _lock:
|
||||
if log_file_id in _active_file_ids:
|
||||
return False
|
||||
_active_file_ids.add(log_file_id)
|
||||
return True
|
||||
|
||||
|
||||
def _release(log_file_id: int) -> None:
|
||||
with _lock:
|
||||
_active_file_ids.discard(log_file_id)
|
||||
|
||||
|
||||
def _run_to_completion(app: Flask, log_file_id: int, batch_size: int) -> None:
|
||||
with app.app_context():
|
||||
try:
|
||||
log_file = db.session.get(LogFile, log_file_id)
|
||||
if log_file is None:
|
||||
return
|
||||
while log_file.status in ("queued", "processing"):
|
||||
process_one_batch(log_file, batch_size)
|
||||
db.session.refresh(log_file)
|
||||
except Exception:
|
||||
app.logger.exception("Background processing failed for log_file_id=%s", log_file_id)
|
||||
log_file = db.session.get(LogFile, log_file_id)
|
||||
if log_file is not None and log_file.status != "done":
|
||||
log_file.status = "error"
|
||||
log_file.error_message = "Processing failed unexpectedly; see server logs."
|
||||
db.session.commit()
|
||||
finally:
|
||||
_release(log_file_id)
|
||||
|
||||
|
||||
def trigger_processing(app: Flask, log_file_id: int) -> None:
|
||||
"""Start background processing for one file; no-ops if already running."""
|
||||
if not _claim(log_file_id):
|
||||
return
|
||||
batch_size = app.config["PARSE_BATCH_SIZE"]
|
||||
thread = threading.Thread(
|
||||
target=_run_to_completion, args=(app, log_file_id, batch_size), daemon=True
|
||||
)
|
||||
thread.start()
|
||||
|
||||
|
||||
def resume_incomplete_files(app: Flask) -> None:
|
||||
"""Opportunistic resume hook, called from the Overview page load —
|
||||
the deliberate replacement for cron's "there's always a next tick"
|
||||
guarantee. Cheap: one indexed status-filtered query.
|
||||
"""
|
||||
incomplete_ids = [
|
||||
lf.id for lf in LogFile.query.filter(LogFile.status.in_(["queued", "processing"])).all()
|
||||
]
|
||||
for log_file_id in incomplete_ids:
|
||||
trigger_processing(app, log_file_id)
|
||||
@@ -0,0 +1,67 @@
|
||||
"""Blocklist-suggestion generation (Chapter 10 follow-up): no earlier
|
||||
chapter assigned ownership of populating blocklist_suggestions or setting
|
||||
ip_registry.is_flagged. Runs in the background aggregator pass (batch, not
|
||||
request-time, per Ch02/03), reusing severity_scoring.py so there's exactly
|
||||
one scoring model between the dashboard display and the flagging decision.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from collections import defaultdict
|
||||
from datetime import date, datetime, timedelta
|
||||
|
||||
from app.extensions import db
|
||||
from app.models.blocklist_suggestion import BlocklistSuggestion
|
||||
from app.models.ip_registry import IPRegistry
|
||||
from app.models.suspicious_event import SuspiciousEvent
|
||||
from app.services.severity_scoring import SeverityInputs, compute_effective_severity
|
||||
|
||||
_RANK = {"low": 0, "medium": 1, "high": 2}
|
||||
|
||||
|
||||
def refresh_blocklist_suggestions(day: date) -> None:
|
||||
"""Flag an IP (is_flagged + a suggestion row) if its escalated severity
|
||||
for `day` reaches 'high'. Idempotent — skips IPs already suggested.
|
||||
"""
|
||||
start = datetime.combine(day, datetime.min.time())
|
||||
end = start + timedelta(days=1)
|
||||
|
||||
events = (
|
||||
db.session.query(SuspiciousEvent.ip, SuspiciousEvent.timestamp, SuspiciousEvent.severity)
|
||||
.filter(SuspiciousEvent.timestamp >= start, SuspiciousEvent.timestamp < end)
|
||||
.all()
|
||||
)
|
||||
if not events:
|
||||
return
|
||||
|
||||
by_ip: dict[str, list] = defaultdict(list)
|
||||
for ip, ts, sev in events:
|
||||
by_ip[ip].append((ts, sev))
|
||||
|
||||
already_suggested = {ip for (ip,) in db.session.query(BlocklistSuggestion.ip).distinct().all()}
|
||||
|
||||
for ip, ip_events in by_ip.items():
|
||||
if ip in already_suggested:
|
||||
continue
|
||||
timestamps = sorted(ts for ts, _ in ip_events)
|
||||
avg_interval = (
|
||||
(timestamps[-1] - timestamps[0]).total_seconds() / (len(timestamps) - 1)
|
||||
if len(timestamps) > 1 else None
|
||||
)
|
||||
worst_base = max((sev for _, sev in ip_events), key=lambda s: _RANK.get(s, 0))
|
||||
effective = compute_effective_severity(
|
||||
SeverityInputs(base_severity=worst_base, ip_event_count=len(ip_events), avg_interval_seconds=avg_interval)
|
||||
)
|
||||
if effective != "high":
|
||||
continue
|
||||
|
||||
db.session.add(BlocklistSuggestion(
|
||||
ip=ip,
|
||||
reason=f"{len(ip_events)} suspicious event(s) on {day.isoformat()}, escalated to high severity",
|
||||
created_at=datetime.utcnow(),
|
||||
exported=False,
|
||||
))
|
||||
ip_row = db.session.get(IPRegistry, ip)
|
||||
if ip_row is not None:
|
||||
ip_row.is_flagged = True
|
||||
|
||||
db.session.commit()
|
||||
@@ -0,0 +1,78 @@
|
||||
"""Bot signature matching + reverse-DNS verification (Chapter 07).
|
||||
|
||||
Signatures are data (JSON), not hardcoded logic, so the list grows without
|
||||
a code change. Verification does real DNS I/O — only ever called from the
|
||||
background process-logs batch job (app/services/classification.py), never
|
||||
synchronously inside a dashboard request, per Chapter 07's explicit rule.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import socket
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime, timedelta
|
||||
from pathlib import Path
|
||||
|
||||
SIGNATURES_PATH = Path(__file__).parent / "data" / "bot_signatures.json"
|
||||
|
||||
# Skip re-verifying the same IP more often than this (Ch07: DNS latency is
|
||||
# a real cost on constrained hosting).
|
||||
VERIFICATION_TTL = timedelta(days=7)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class BotSignature:
|
||||
name: str
|
||||
ua_substrings: tuple[str, ...]
|
||||
verify_suffixes: tuple[str, ...] # PTR hostname must end in one of these
|
||||
|
||||
|
||||
def _load_signatures() -> list[BotSignature]:
|
||||
raw = json.loads(SIGNATURES_PATH.read_text())
|
||||
return [
|
||||
BotSignature(name=e["name"], ua_substrings=tuple(e["ua_substrings"]), verify_suffixes=tuple(e["verify_suffixes"]))
|
||||
for e in raw
|
||||
]
|
||||
|
||||
|
||||
_SIGNATURES = _load_signatures()
|
||||
|
||||
|
||||
def classify_bot(user_agent: str | None) -> str | None:
|
||||
"""Return the claimed bot name via UA substring match, or None."""
|
||||
if not user_agent:
|
||||
return None
|
||||
ua_lower = user_agent.lower()
|
||||
for sig in _SIGNATURES:
|
||||
if any(sub.lower() in ua_lower for sub in sig.ua_substrings):
|
||||
return sig.name
|
||||
return None
|
||||
|
||||
|
||||
def _signature_for(bot_name: str) -> BotSignature | None:
|
||||
return next((s for s in _SIGNATURES if s.name == bot_name), None)
|
||||
|
||||
|
||||
def verify_bot_ip(ip: str, bot_name: str) -> bool:
|
||||
"""Reverse-DNS + forward-confirm that `ip` really belongs to `bot_name`."""
|
||||
sig = _signature_for(bot_name)
|
||||
if sig is None:
|
||||
return False
|
||||
try:
|
||||
hostname, _, _ = socket.gethostbyaddr(ip)
|
||||
except (socket.herror, socket.gaierror, OSError):
|
||||
return False
|
||||
if not any(hostname.lower().endswith(suffix) for suffix in sig.verify_suffixes):
|
||||
return False
|
||||
try:
|
||||
forward_ips = socket.gethostbyname_ex(hostname)[2]
|
||||
except (socket.herror, socket.gaierror, OSError):
|
||||
return False
|
||||
return ip in forward_ips
|
||||
|
||||
|
||||
def is_verification_stale(last_verified_at: datetime | None) -> bool:
|
||||
"""True if this IP needs a fresh DNS check (Ch07 caching rule)."""
|
||||
if last_verified_at is None:
|
||||
return True
|
||||
return datetime.utcnow() - last_verified_at > VERIFICATION_TTL
|
||||
@@ -0,0 +1,122 @@
|
||||
"""Per-batch classification pipeline (Chapter 07): ties bot_identifier and
|
||||
threat_scanner into the bot_hits/suspicious_events/ip_registry write path.
|
||||
Called once per bulk-insert chunk from app/cli.py — DNS-based verification
|
||||
belongs here (background batch job), never in a dashboard request.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime
|
||||
|
||||
from sqlalchemy import func, insert
|
||||
from sqlalchemy.dialects.sqlite import insert as sqlite_insert
|
||||
|
||||
from app.extensions import db
|
||||
from app.models.bot_hit import BotHit
|
||||
from app.models.ip_registry import IPRegistry
|
||||
from app.models.suspicious_event import SuspiciousEvent
|
||||
from app.services import bot_identifier, threat_scanner
|
||||
from app.services.log_parser import ParsedEntry
|
||||
|
||||
|
||||
@dataclass
|
||||
class EntryClassification:
|
||||
"""Cheap, no-I/O flags for the log_entries row itself."""
|
||||
is_bot: bool
|
||||
flagged: bool
|
||||
|
||||
|
||||
def classify_entry(entry: ParsedEntry) -> EntryClassification:
|
||||
"""No DNS I/O here — bot *verification* is batched separately below,
|
||||
since it's stateful (cached per IP) and only worth doing once per IP
|
||||
per batch, not once per line.
|
||||
"""
|
||||
return EntryClassification(
|
||||
is_bot=bot_identifier.classify_bot(entry.user_agent) is not None,
|
||||
flagged=threat_scanner.scan(entry) is not None,
|
||||
)
|
||||
|
||||
|
||||
def write_batch_side_effects(log_file_id: int, entries: list[ParsedEntry]) -> None:
|
||||
"""Derive and bulk-write bot_hits, suspicious_events, ip_registry upserts."""
|
||||
if not entries:
|
||||
return
|
||||
|
||||
bot_hit_rows: list[dict] = []
|
||||
suspicious_rows: list[dict] = []
|
||||
ip_agg: dict[str, dict] = {}
|
||||
verified_this_batch: dict[str, bool] = {} # avoid repeat DNS for the same IP in one batch
|
||||
|
||||
for entry in entries:
|
||||
agg = ip_agg.setdefault(entry.ip, {"first": entry.timestamp, "last": entry.timestamp, "count": 0})
|
||||
agg["count"] += 1
|
||||
agg["first"] = min(agg["first"], entry.timestamp)
|
||||
agg["last"] = max(agg["last"], entry.timestamp)
|
||||
|
||||
bot_name = bot_identifier.classify_bot(entry.user_agent)
|
||||
if bot_name is not None:
|
||||
verified = verified_this_batch.get(entry.ip)
|
||||
if verified is None:
|
||||
verified = _verify_with_cache(entry.ip, bot_name)
|
||||
verified_this_batch[entry.ip] = verified
|
||||
bot_hit_rows.append({
|
||||
"log_file_id": log_file_id, "timestamp": entry.timestamp, "ip": entry.ip,
|
||||
"bot_name": bot_name, "verified": verified, "path": entry.path,
|
||||
"status_code": entry.status_code,
|
||||
})
|
||||
if not verified:
|
||||
# Single detection, two dashboard consumers (Ch07): spoofed
|
||||
# bot also surfaces as a suspicious_events row.
|
||||
suspicious_rows.append({
|
||||
"log_file_id": log_file_id, "ip": entry.ip, "timestamp": entry.timestamp,
|
||||
"path": entry.path, "rule_matched": f"spoofed_bot:{bot_name}", "severity": "medium",
|
||||
})
|
||||
|
||||
threat = threat_scanner.scan(entry)
|
||||
if threat is not None:
|
||||
suspicious_rows.append({
|
||||
"log_file_id": log_file_id, "ip": entry.ip, "timestamp": entry.timestamp,
|
||||
"path": entry.path, "rule_matched": threat.rule_matched, "severity": threat.severity,
|
||||
})
|
||||
|
||||
if bot_hit_rows:
|
||||
db.session.execute(insert(BotHit.__table__), bot_hit_rows)
|
||||
if suspicious_rows:
|
||||
db.session.execute(insert(SuspiciousEvent.__table__), suspicious_rows)
|
||||
|
||||
_upsert_ip_registry(ip_agg, verified_this_batch)
|
||||
db.session.commit()
|
||||
|
||||
|
||||
def _verify_with_cache(ip: str, bot_name: str) -> bool:
|
||||
"""TTL-gated reverse/forward DNS check, cached via
|
||||
ip_registry.last_verified_bot_result (Ch07 schema addition).
|
||||
"""
|
||||
row = db.session.get(IPRegistry, ip)
|
||||
if row is not None and not bot_identifier.is_verification_stale(row.last_verified_at):
|
||||
return bool(row.last_verified_bot_result)
|
||||
return bot_identifier.verify_bot_ip(ip, bot_name)
|
||||
|
||||
|
||||
def _upsert_ip_registry(ip_agg: dict[str, dict], verified_this_batch: dict[str, bool]) -> None:
|
||||
now = datetime.utcnow()
|
||||
for ip, agg in ip_agg.items():
|
||||
values = {
|
||||
"ip": ip, "first_seen": agg["first"], "last_seen": agg["last"],
|
||||
"total_requests": agg["count"], "reputation_score": 0, "is_flagged": False,
|
||||
}
|
||||
if ip in verified_this_batch:
|
||||
values["last_verified_at"] = now
|
||||
values["last_verified_bot_result"] = verified_this_batch[ip]
|
||||
|
||||
stmt = sqlite_insert(IPRegistry.__table__).values(**values)
|
||||
update_set = {
|
||||
"last_seen": func.max(IPRegistry.last_seen, stmt.excluded.last_seen),
|
||||
"first_seen": func.min(IPRegistry.first_seen, stmt.excluded.first_seen),
|
||||
"total_requests": IPRegistry.total_requests + stmt.excluded.total_requests,
|
||||
}
|
||||
if ip in verified_this_batch:
|
||||
update_set["last_verified_at"] = stmt.excluded.last_verified_at
|
||||
update_set["last_verified_bot_result"] = stmt.excluded.last_verified_bot_result
|
||||
stmt = stmt.on_conflict_do_update(index_elements=["ip"], set_=update_set)
|
||||
db.session.execute(stmt)
|
||||
@@ -0,0 +1,9 @@
|
||||
[
|
||||
{"name": "Googlebot", "ua_substrings": ["Googlebot"], "verify_suffixes": [".googlebot.com", ".google.com"]},
|
||||
{"name": "Bingbot", "ua_substrings": ["bingbot"], "verify_suffixes": [".search.msn.com"]},
|
||||
{"name": "Yandex", "ua_substrings": ["YandexBot"], "verify_suffixes": [".yandex.ru", ".yandex.com", ".yandex.net"]},
|
||||
{"name": "Baidu", "ua_substrings": ["Baiduspider"], "verify_suffixes": [".baidu.com", ".baidu.jp"]},
|
||||
{"name": "DuckDuckBot", "ua_substrings": ["DuckDuckBot"], "verify_suffixes": [".duckduckgo.com"]},
|
||||
{"name": "AhrefsBot", "ua_substrings": ["AhrefsBot"], "verify_suffixes": [".ahrefs.com"]},
|
||||
{"name": "SemrushBot", "ua_substrings": ["SemrushBot"], "verify_suffixes": [".semrush.com"]}
|
||||
]
|
||||
@@ -0,0 +1,8 @@
|
||||
[
|
||||
{"name": "Edge", "ua_substrings": ["Edg/", "EdgA/", "EdgiOS/"]},
|
||||
{"name": "Opera", "ua_substrings": ["OPR/", "Opera"]},
|
||||
{"name": "Chrome", "ua_substrings": ["Chrome/", "CriOS/"]},
|
||||
{"name": "Firefox", "ua_substrings": ["Firefox/", "FxiOS/"]},
|
||||
{"name": "Safari", "ua_substrings": ["Safari/"]},
|
||||
{"name": "Internet Explorer", "ua_substrings": ["MSIE ", "Trident/"]}
|
||||
]
|
||||
@@ -0,0 +1,7 @@
|
||||
[
|
||||
{"name": "Windows", "ua_substrings": ["Windows NT"]},
|
||||
{"name": "iOS", "ua_substrings": ["iPhone", "iPad", "iPod"]},
|
||||
{"name": "macOS", "ua_substrings": ["Mac OS X", "Macintosh"]},
|
||||
{"name": "Android", "ua_substrings": ["Android"]},
|
||||
{"name": "Linux", "ua_substrings": ["Linux"]}
|
||||
]
|
||||
@@ -0,0 +1,11 @@
|
||||
{
|
||||
"sensitive_paths": [
|
||||
"/.env", "/.git/config", "wp-config.php", "/.htpasswd", "/phpmyadmin",
|
||||
"/xmlrpc.php", ".sql.gz", ".sql.bak", ".zip", ".bak", ".old"
|
||||
],
|
||||
"injection_markers": [
|
||||
"union select", "' or '1'='1", "<script>", "javascript:",
|
||||
"../", "..%2f", "%2e%2e%2f", "%252e%252e%252f"
|
||||
],
|
||||
"scanner_user_agents": ["sqlmap", "nikto", "nmap", "acunetix", "nessus"]
|
||||
}
|
||||
@@ -0,0 +1,93 @@
|
||||
"""Uploaded-file deletion (project-owner follow-up request).
|
||||
|
||||
Deleting a LogFile is more than one DELETE statement: log_entries,
|
||||
bot_hits, and suspicious_events all reference log_file_id and must be
|
||||
removed explicitly — this project's SQLite connections don't have
|
||||
`PRAGMA foreign_keys=ON`, so the `ondelete="CASCADE"` in the migrations
|
||||
(Ch06) is declarative documentation only, not an enforced behavior.
|
||||
Relying on it would silently leave orphaned rows behind.
|
||||
|
||||
The harder part is the rollup tables (request_stats_hourly/daily,
|
||||
referrer/browser/human-path/ip-path/ip-status): none of them have a
|
||||
log_file_id column (Ch06: they're site-wide, since two files can share a
|
||||
date). So instead of deleting rollup rows for the deleted file's dates
|
||||
directly (which could wipe out another file's contribution to the same
|
||||
date), this captures the affected dates BEFORE deleting, then calls
|
||||
aggregator.compute_rollups_for_range() AFTER deleting — which recomputes
|
||||
each affected day from whatever log_entries remain, correctly handling
|
||||
both "another file still covers this day" and "no file covers this day
|
||||
anymore" (the latter now handled correctly by the aggregator.py fix that
|
||||
shipped alongside this feature).
|
||||
|
||||
KNOWN LIMITATION (flagged, not fixed): ip_registry (total_requests,
|
||||
first_seen, last_seen, verification cache) and blocklist_suggestions are
|
||||
NOT per-file and are NOT recomputed on deletion — doing so correctly
|
||||
would mean re-scanning all remaining log_entries for every affected IP,
|
||||
which is unbounded work for a single delete action and would violate the
|
||||
same "no raw-row scan on a hot path" reasoning Ch03 applies elsewhere.
|
||||
After deleting a file, IP History numbers may include a deleted file's
|
||||
historical contribution until a full site-wide recompute is added as a
|
||||
separate feature. Doesn't affect correctness of Overview/SEO's date-
|
||||
scoped numbers, which is what this feature was actually asked to fix.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
|
||||
from app.extensions import db
|
||||
from app.models.bot_hit import BotHit
|
||||
from app.models.log_entry import LogEntry
|
||||
from app.models.log_file import LogFile
|
||||
from app.models.suspicious_event import SuspiciousEvent
|
||||
from app.services import aggregator
|
||||
from app.utils.upload_paths import upload_path_for
|
||||
|
||||
|
||||
@dataclass
|
||||
class BulkDeleteResult:
|
||||
deleted: list[int] = field(default_factory=list)
|
||||
skipped: list[dict] = field(default_factory=list) # [{"id": ..., "reason": ...}]
|
||||
|
||||
|
||||
def delete_log_files(log_file_ids: list[int]) -> BulkDeleteResult:
|
||||
"""Delete each id in `log_file_ids`; recompute rollups once at the end
|
||||
for the full span of dates any deleted file touched (cheaper than
|
||||
recomputing per-file when multiple files share dates).
|
||||
"""
|
||||
result = BulkDeleteResult()
|
||||
all_touched_dates: set = set()
|
||||
|
||||
for log_file_id in log_file_ids:
|
||||
log_file = db.session.get(LogFile, log_file_id)
|
||||
if log_file is None:
|
||||
result.skipped.append({"id": log_file_id, "reason": "not found"})
|
||||
continue
|
||||
if log_file.status == "processing":
|
||||
result.skipped.append({"id": log_file_id, "reason": "currently being analyzed"})
|
||||
continue
|
||||
|
||||
touched_dates = {
|
||||
row.date() for row, in db.session.query(LogEntry.timestamp)
|
||||
.filter(LogEntry.log_file_id == log_file_id).distinct()
|
||||
}
|
||||
# distinct() on a full timestamp rarely collapses much; reduce to
|
||||
# calendar dates in Python since SQLite's DATE() in a DISTINCT
|
||||
# clause is a bit more awkward to express portably here.
|
||||
all_touched_dates |= touched_dates
|
||||
|
||||
db.session.query(SuspiciousEvent).filter(SuspiciousEvent.log_file_id == log_file_id).delete()
|
||||
db.session.query(BotHit).filter(BotHit.log_file_id == log_file_id).delete()
|
||||
db.session.query(LogEntry).filter(LogEntry.log_file_id == log_file_id).delete()
|
||||
|
||||
raw_path = upload_path_for(log_file)
|
||||
if raw_path.exists():
|
||||
raw_path.unlink()
|
||||
|
||||
db.session.delete(log_file)
|
||||
db.session.commit()
|
||||
result.deleted.append(log_file_id)
|
||||
|
||||
if all_touched_dates:
|
||||
aggregator.compute_rollups_for_range(min(all_touched_dates), max(all_touched_dates))
|
||||
|
||||
return result
|
||||
@@ -0,0 +1,47 @@
|
||||
"""Log parser abstraction (Chapter 07)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime
|
||||
from typing import Protocol, TYPE_CHECKING
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from app.models.log_file import LogFile
|
||||
|
||||
|
||||
@dataclass
|
||||
class ParsedEntry:
|
||||
timestamp: datetime
|
||||
ip: str
|
||||
method: str
|
||||
path: str
|
||||
status_code: int
|
||||
bytes_sent: int
|
||||
referrer: str | None
|
||||
user_agent: str | None
|
||||
|
||||
|
||||
class LogParser(Protocol):
|
||||
def parse_line(self, line: str) -> ParsedEntry | None: ...
|
||||
|
||||
|
||||
def get_parser_for(log_file: "LogFile") -> LogParser:
|
||||
"""Build the right LogParser for a given upload's format_string.
|
||||
|
||||
ASSUMPTION (flagged): a blank format_string — which the upload form
|
||||
allows — falls back to Combined, since Chapter 07 states no default
|
||||
for that case. Anything else is compiled as-is; a real custom
|
||||
LiteSpeed format is never silently replaced by a preset.
|
||||
"""
|
||||
# Imported lazily to avoid a circular import (apache.py imports
|
||||
# ParsedEntry from this module).
|
||||
from app.services.log_parser.apache import FormatCompiledParser, combined_parser, common_parser
|
||||
from app.services.log_parser.format_compiler import compile_format
|
||||
|
||||
fmt = (log_file.format_string or "").strip()
|
||||
key = fmt.lower()
|
||||
if key in ("", "combined"):
|
||||
return combined_parser()
|
||||
if key == "common":
|
||||
return common_parser()
|
||||
return FormatCompiledParser(compile_format(fmt))
|
||||
@@ -0,0 +1,57 @@
|
||||
"""Apache Common/Combined presets + the generic parser built from
|
||||
format_compiler (Chapter 07)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from app.services.log_parser import ParsedEntry
|
||||
from app.services.log_parser.format_compiler import (
|
||||
CompiledFormat, compile_format, parse_apache_timestamp, parse_request_line, validate_ip,
|
||||
)
|
||||
|
||||
COMMON_LOG_FORMAT = '%h %l %u %t "%r" %>s %b'
|
||||
COMBINED_LOG_FORMAT = '%h %l %u %t "%r" %>s %b "%{Referer}i" "%{User-agent}i"'
|
||||
|
||||
|
||||
class FormatCompiledParser:
|
||||
"""LogParser implementation driven by any CompiledFormat (Ch04/07 protocol)."""
|
||||
|
||||
def __init__(self, compiled: CompiledFormat):
|
||||
self._compiled = compiled
|
||||
|
||||
def parse_line(self, line: str) -> ParsedEntry | None:
|
||||
match = self._compiled.pattern.match(line.rstrip("\n"))
|
||||
if match is None:
|
||||
return None # malformed line — skip, never raise (Ch07)
|
||||
fields = match.groupdict()
|
||||
|
||||
ip = validate_ip(fields["ip"])
|
||||
if ip is None:
|
||||
return None
|
||||
|
||||
req = parse_request_line(fields["request"])
|
||||
if req is None:
|
||||
return None
|
||||
method, path = req
|
||||
|
||||
try:
|
||||
timestamp = parse_apache_timestamp(fields["timestamp"])
|
||||
status_code = int(fields["status"])
|
||||
bytes_sent = 0 if fields["bytes"] == "-" else int(fields["bytes"])
|
||||
except (ValueError, KeyError):
|
||||
return None
|
||||
|
||||
referrer = fields.get("referrer")
|
||||
user_agent = fields.get("user_agent")
|
||||
return ParsedEntry(
|
||||
timestamp=timestamp, ip=ip, method=method, path=path,
|
||||
status_code=status_code, bytes_sent=bytes_sent,
|
||||
referrer=None if referrer in (None, "-") else referrer,
|
||||
user_agent=None if user_agent in (None, "-") else user_agent,
|
||||
)
|
||||
|
||||
|
||||
def common_parser() -> FormatCompiledParser:
|
||||
return FormatCompiledParser(compile_format(COMMON_LOG_FORMAT))
|
||||
|
||||
|
||||
def combined_parser() -> FormatCompiledParser:
|
||||
return FormatCompiledParser(compile_format(COMBINED_LOG_FORMAT))
|
||||
@@ -0,0 +1,95 @@
|
||||
"""LogFormat directive -> compiled regex + named groups (Chapter 07).
|
||||
|
||||
One code path for Common, Combined, and arbitrary custom formats — presets
|
||||
below are just specific format strings run through this same compiler.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime, timezone
|
||||
from ipaddress import ip_address
|
||||
|
||||
# Directive -> (group name, regex pattern). NOTE: %r and %{...}i patterns
|
||||
# deliberately do NOT include quote characters — the quotes around them
|
||||
# are literal text already present in the format string (e.g. `"%r"`),
|
||||
# and get regex-escaped by the literal-text path below. %t is different:
|
||||
# its brackets are part of what %t itself produces, not literal format-
|
||||
# string text, so they belong in this directive's own pattern.
|
||||
_SIMPLE_DIRECTIVES: dict[str, tuple[str, str]] = {
|
||||
"%h": ("ip", r"(?P<ip>\S+)"),
|
||||
"%l": ("ident", r"(?P<ident>\S+)"),
|
||||
"%u": ("user", r"(?P<user>\S+)"),
|
||||
"%t": ("timestamp", r"\[(?P<timestamp>[^\]]+)\]"),
|
||||
"%r": ("request", r'(?P<request>[^"]*)'),
|
||||
"%>s": ("status", r"(?P<status>\d{3})"),
|
||||
"%s": ("status", r"(?P<status>\d{3})"),
|
||||
"%b": ("bytes", r"(?P<bytes>\d+|-)"),
|
||||
}
|
||||
|
||||
_HEADER_DIRECTIVE_RE = re.compile(r'%\{([^}]+)\}i')
|
||||
_KNOWN_HEADER_GROUPS = {"referer": "referrer", "user-agent": "user_agent"}
|
||||
_DIRECTIVE_TOKEN_RE = re.compile(r"%>?\{[^}]+\}i|%>?[a-zA-Z]")
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class CompiledFormat:
|
||||
pattern: re.Pattern
|
||||
group_names: frozenset[str]
|
||||
|
||||
|
||||
def compile_format(format_string: str) -> CompiledFormat:
|
||||
"""Compile a LogFormat directive (Common/Combined preset or a pasted
|
||||
custom vhost format) into one regex."""
|
||||
regex_parts: list[str] = []
|
||||
group_names: set[str] = set()
|
||||
pos = 0
|
||||
for match in _DIRECTIVE_TOKEN_RE.finditer(format_string):
|
||||
literal = format_string[pos:match.start()]
|
||||
if literal:
|
||||
regex_parts.append(re.escape(literal))
|
||||
regex_parts.append(_directive_to_pattern(match.group(0), group_names))
|
||||
pos = match.end()
|
||||
regex_parts.append(re.escape(format_string[pos:]))
|
||||
|
||||
pattern = re.compile("^" + "".join(regex_parts) + r"\s*$")
|
||||
return CompiledFormat(pattern=pattern, group_names=frozenset(group_names))
|
||||
|
||||
|
||||
def _directive_to_pattern(token: str, group_names: set[str]) -> str:
|
||||
header_match = _HEADER_DIRECTIVE_RE.fullmatch(token)
|
||||
if header_match:
|
||||
header_name = header_match.group(1).lower()
|
||||
group = _KNOWN_HEADER_GROUPS.get(header_name, re.sub(r"[^a-z0-9]+", "_", header_name))
|
||||
group_names.add(group)
|
||||
return f'(?P<{group}>[^"]*)'
|
||||
|
||||
if token not in _SIMPLE_DIRECTIVES:
|
||||
raise ValueError(f"Unsupported LogFormat directive: {token!r}")
|
||||
group, pattern = _SIMPLE_DIRECTIVES[token]
|
||||
group_names.add(group)
|
||||
return pattern
|
||||
|
||||
|
||||
def parse_apache_timestamp(raw: str) -> datetime:
|
||||
"""'10/Oct/2026:13:55:36 -0700' -> naive UTC datetime (Ch07: store normalized to UTC)."""
|
||||
dt = datetime.strptime(raw, "%d/%b/%Y:%H:%M:%S %z")
|
||||
return dt.astimezone(timezone.utc).replace(tzinfo=None)
|
||||
|
||||
|
||||
def parse_request_line(raw: str) -> tuple[str, str] | None:
|
||||
"""Split '%r' ("GET /path HTTP/1.1") into (method, path); None if malformed."""
|
||||
parts = raw.split()
|
||||
if len(parts) != 3:
|
||||
return None
|
||||
method, path, _protocol = parts
|
||||
return method, path
|
||||
|
||||
|
||||
def validate_ip(raw: str) -> str | None:
|
||||
"""Return `raw` if a valid IPv4/IPv6 address, else None (Ch07)."""
|
||||
try:
|
||||
ip_address(raw)
|
||||
except ValueError:
|
||||
return None
|
||||
return raw
|
||||
@@ -0,0 +1,10 @@
|
||||
"""LiteSpeed uses the same LogFormat directives as Apache (Chapter 07):
|
||||
Combined generally works unmodified. get_parser_for() still always
|
||||
compiles the vhost's real format_string though — this module is just the
|
||||
named default, never a silent override of a custom format.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from app.services.log_parser.apache import combined_parser as default_litespeed_parser
|
||||
|
||||
__all__ = ["default_litespeed_parser"]
|
||||
@@ -0,0 +1,148 @@
|
||||
"""Core log-file processing pipeline (Chapter 07), factored out of
|
||||
app/cli.py so it has exactly one implementation shared by:
|
||||
- the optional `flask process-logs` CLI command (for anyone who still
|
||||
wants cron), and
|
||||
- the automatic background-thread trigger fired right after upload and
|
||||
opportunistically on page load (app/services/background.py) — the
|
||||
no-cron-required simplification.
|
||||
|
||||
process_one_batch() keeps the same bounded-batch, checkpointed-commit
|
||||
contract as the original cron design (Ch03 rule 1/4/9): it still never
|
||||
loads a whole file into memory, still commits progress incrementally, and
|
||||
still stops after `batch_size` lines so a very large file doesn't hold one
|
||||
enormous open transaction — the only thing that changed is *who* calls it
|
||||
again for the next batch.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import gzip
|
||||
import itertools
|
||||
from datetime import date
|
||||
from pathlib import Path
|
||||
|
||||
from sqlalchemy import insert
|
||||
|
||||
from app.extensions import db
|
||||
from app.models.log_entry import LogEntry
|
||||
from app.models.log_file import LogFile
|
||||
from app.services import aggregator
|
||||
from app.services.classification import classify_entry, write_batch_side_effects
|
||||
from app.services.log_parser import ParsedEntry, get_parser_for
|
||||
from app.utils.upload_paths import upload_path_for
|
||||
|
||||
|
||||
def open_log_stream(path: Path):
|
||||
"""Open a log file, transparently decompressing .gz, one line at a time."""
|
||||
if path.suffix == ".gz":
|
||||
return gzip.open(path, mode="rt", encoding="utf-8", errors="replace")
|
||||
return path.open("r", encoding="utf-8", errors="replace")
|
||||
|
||||
|
||||
def count_total_lines(path: Path) -> int:
|
||||
"""One streaming pass to count lines (bounded memory — never the whole
|
||||
file at once, Ch03 rule 1), so the UI can show a real
|
||||
processed/total percentage instead of an indeterminate spinner.
|
||||
|
||||
This is an extra sequential read of the file beyond the parse pass
|
||||
itself. That cost is now worth paying: processing used to be silently
|
||||
triggered by cron with no live audience watching, but now it runs
|
||||
automatically right after upload while the admin is looking at a
|
||||
progress bar — the UX value of a real percentage justifies the extra
|
||||
I/O pass.
|
||||
"""
|
||||
with open_log_stream(path) as fh:
|
||||
return sum(1 for _ in fh)
|
||||
|
||||
|
||||
def process_one_batch(log_file: LogFile, batch_size: int) -> None:
|
||||
"""Parse up to `batch_size` lines from `log_file`'s current checkpoint.
|
||||
|
||||
Leaves status as "processing" (with progress already committed) if
|
||||
the file isn't finished yet — the caller decides whether to invoke
|
||||
this again: the CLI calls it once per pending file per invocation
|
||||
(unchanged cron-tick semantics); the background thread loops it until
|
||||
the file reaches "done"/"error".
|
||||
"""
|
||||
path = upload_path_for(log_file)
|
||||
if not path.exists():
|
||||
log_file.status = "error"
|
||||
log_file.error_message = f"Upload file missing on disk: {path}"
|
||||
db.session.commit()
|
||||
return
|
||||
|
||||
if log_file.total_lines is None:
|
||||
# First pickup of this file — count once, not on every batch.
|
||||
log_file.total_lines = count_total_lines(path)
|
||||
|
||||
log_file.status = "processing"
|
||||
db.session.commit()
|
||||
parser = get_parser_for(log_file)
|
||||
|
||||
lines_seen_this_run = 0
|
||||
skipped_this_run = 0
|
||||
dates_touched: set[date] = set()
|
||||
|
||||
with open_log_stream(path) as fh:
|
||||
remainder = itertools.islice(fh, log_file.processed_lines, None)
|
||||
batch_entries: list[ParsedEntry] = []
|
||||
|
||||
for line in remainder:
|
||||
entry = parser.parse_line(line)
|
||||
if entry is not None:
|
||||
batch_entries.append(entry)
|
||||
dates_touched.add(entry.timestamp.date())
|
||||
else:
|
||||
skipped_this_run += 1 # malformed line — skip, don't abort the batch (Ch07)
|
||||
|
||||
log_file.processed_lines += 1
|
||||
lines_seen_this_run += 1
|
||||
|
||||
if len(batch_entries) >= 500:
|
||||
_bulk_insert_entries(log_file.id, batch_entries)
|
||||
batch_entries = []
|
||||
|
||||
if lines_seen_this_run >= batch_size:
|
||||
_record_skip_count(log_file, skipped_this_run)
|
||||
db.session.commit() # persist checkpoint; resumable if interrupted
|
||||
return
|
||||
|
||||
if batch_entries:
|
||||
_bulk_insert_entries(log_file.id, batch_entries)
|
||||
|
||||
if dates_touched:
|
||||
aggregator.compute_rollups_for_range(min(dates_touched), max(dates_touched))
|
||||
|
||||
_record_skip_count(log_file, skipped_this_run)
|
||||
log_file.status = "done"
|
||||
db.session.commit()
|
||||
|
||||
|
||||
def _record_skip_count(log_file: LogFile, skipped_this_run: int) -> None:
|
||||
"""Informational only — doesn't touch `status`.
|
||||
|
||||
STOPGAP (flagged in Ch07): log_files has no dedicated skip-counter
|
||||
column, so this overwrites error_message with the latest run's count
|
||||
rather than accumulating across runs.
|
||||
"""
|
||||
if skipped_this_run:
|
||||
log_file.error_message = f"{skipped_this_run} unparsable line(s) skipped in the most recent parse run."
|
||||
|
||||
|
||||
def _bulk_insert_entries(log_file_id: int, entries: list[ParsedEntry]) -> None:
|
||||
"""Bulk-insert log_entries (Ch03 rule 4), classifying is_bot/flagged
|
||||
per line, then derive bot_hits/suspicious_events/ip_registry (Ch07).
|
||||
"""
|
||||
if not entries:
|
||||
return
|
||||
payload = []
|
||||
for e in entries:
|
||||
classification = classify_entry(e)
|
||||
payload.append({
|
||||
"log_file_id": log_file_id, "timestamp": e.timestamp, "ip": e.ip,
|
||||
"method": e.method, "path": e.path, "status_code": e.status_code,
|
||||
"bytes_sent": e.bytes_sent, "referrer": e.referrer, "user_agent": e.user_agent,
|
||||
"is_bot": classification.is_bot, "flagged": classification.flagged,
|
||||
})
|
||||
db.session.execute(insert(LogEntry.__table__), payload)
|
||||
db.session.commit()
|
||||
write_batch_side_effects(log_file_id, entries)
|
||||
@@ -0,0 +1,19 @@
|
||||
"""Referrer -> domain bucketing (Chapter 08 rollup, Method A)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from urllib.parse import urlparse
|
||||
|
||||
|
||||
def referrer_domain(referrer: str | None) -> str | None:
|
||||
"""Bucket a raw referrer URL to its host, stripping a leading 'www.'.
|
||||
|
||||
Full-URL cardinality would make the daily rollup unbounded in row
|
||||
count; domain bucketing keeps it small (Ch03's bounded-resource bias).
|
||||
Empty/unparsable referrers return None and are excluded from the rollup.
|
||||
"""
|
||||
if not referrer:
|
||||
return None
|
||||
host = urlparse(referrer).netloc.lower()
|
||||
if not host:
|
||||
return None
|
||||
return host[4:] if host.startswith("www.") else host
|
||||
@@ -0,0 +1,49 @@
|
||||
"""Severity scoring (Chapter 10): a simple, transparent rank-escalation
|
||||
model — no ML, easy to explain to a non-expert user (Ch10's own requirement).
|
||||
|
||||
Chapter 07 assigns a PROVISIONAL severity per rule type at parse time.
|
||||
This module computes an EFFECTIVE severity by escalating that base rank
|
||||
for repeated/rapid hits from the same IP — the exact rule Ch10 names.
|
||||
|
||||
Escalation is rank-based and strictly non-decreasing: it starts at the
|
||||
base severity's rank and only ever moves up (repeat-count / hit-rate
|
||||
bonuses), clamped at "high". An earlier weighted-score-vs-fixed-threshold
|
||||
version could silently *downgrade* an isolated high-severity event (e.g.
|
||||
a single sqlmap hit) to "medium" purely because the thresholds weren't
|
||||
calibrated to the base weights — caught by running the pipeline against
|
||||
real sample data. A single dangerous event must never end up rated below
|
||||
its own base severity; only repetition/rate should push it higher.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
|
||||
_SEVERITY_RANKS = ["low", "medium", "high"]
|
||||
_RANK_BY_SEVERITY = {name: rank for rank, name in enumerate(_SEVERITY_RANKS)}
|
||||
|
||||
_REPEAT_COUNT_HIGH = 20
|
||||
_REPEAT_COUNT_MEDIUM = 5
|
||||
_RAPID_AVG_INTERVAL_SECONDS = 60 # >1 event/minute from one IP suggests automation
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class SeverityInputs:
|
||||
base_severity: str
|
||||
ip_event_count: int
|
||||
avg_interval_seconds: float | None # None if this IP has < 2 events in the window
|
||||
|
||||
|
||||
def compute_effective_severity(inputs: SeverityInputs) -> str:
|
||||
"""Two named, inspectable escalation rules: repeat-count and hit-rate."""
|
||||
rank = _RANK_BY_SEVERITY.get(inputs.base_severity, 0)
|
||||
|
||||
if inputs.ip_event_count >= _REPEAT_COUNT_HIGH:
|
||||
rank += 2
|
||||
elif inputs.ip_event_count >= _REPEAT_COUNT_MEDIUM:
|
||||
rank += 1
|
||||
|
||||
if inputs.avg_interval_seconds is not None and inputs.avg_interval_seconds < _RAPID_AVG_INTERVAL_SECONDS:
|
||||
rank += 1
|
||||
|
||||
rank = min(rank, len(_SEVERITY_RANKS) - 1)
|
||||
return _SEVERITY_RANKS[rank]
|
||||
@@ -0,0 +1,54 @@
|
||||
"""Suspicious-pattern matching (Chapter 07), reused as-is by Chapter 10.
|
||||
|
||||
Pattern dictionary is data (JSON), same reasoning as the bot signature
|
||||
table — editable without a code change. Severity here is a provisional
|
||||
per-rule-type default; Chapter 10 owns the full weighted/escalating score.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from app.services.log_parser import ParsedEntry
|
||||
|
||||
PATTERNS_PATH = Path(__file__).parent / "data" / "threat_patterns.json"
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ThreatMatch:
|
||||
rule_matched: str
|
||||
severity: str # low|medium|high — provisional; Ch10 may escalate
|
||||
|
||||
|
||||
def _load_patterns() -> dict:
|
||||
return json.loads(PATTERNS_PATH.read_text())
|
||||
|
||||
|
||||
_PATTERNS = _load_patterns()
|
||||
|
||||
|
||||
def scan(entry: "ParsedEntry") -> ThreatMatch | None:
|
||||
"""Check one parsed entry against the pattern dictionary.
|
||||
|
||||
Cheap substring checks only — no regex backtracking risk, no network
|
||||
calls (Ch03: CPU-cheap, explainable rule matching).
|
||||
"""
|
||||
path_lower = entry.path.lower()
|
||||
ua_lower = (entry.user_agent or "").lower()
|
||||
|
||||
for scanner_ua in _PATTERNS["scanner_user_agents"]:
|
||||
if scanner_ua in ua_lower:
|
||||
return ThreatMatch(rule_matched=f"scanner_ua:{scanner_ua}", severity="high")
|
||||
|
||||
for sensitive in _PATTERNS["sensitive_paths"]:
|
||||
if sensitive.lower() in path_lower:
|
||||
return ThreatMatch(rule_matched=f"sensitive_path:{sensitive}", severity="medium")
|
||||
|
||||
for marker in _PATTERNS["injection_markers"]:
|
||||
if marker.lower() in path_lower:
|
||||
return ThreatMatch(rule_matched=f"injection:{marker}", severity="high")
|
||||
|
||||
return None
|
||||
@@ -0,0 +1,54 @@
|
||||
"""Lightweight browser/OS classification for human traffic (Chapter 08).
|
||||
|
||||
Separate from Chapter 07's bot_identifier — that identifies automated
|
||||
crawlers via UA substring + DNS verification. This classifies ordinary
|
||||
browsers/OSes, same ordered-substring-match pattern, signatures as JSON
|
||||
data (not hardcoded), no third-party UA-parsing dependency — consistent
|
||||
with the low-footprint bias in Chapter 03.
|
||||
|
||||
Order matters within each signature list: e.g. Edge/Opera must be checked
|
||||
before Chrome (their UAs also contain "Chrome/"); iOS before macOS (iPhone
|
||||
UAs also contain "like Mac OS X"); Android before Linux (Android UAs also
|
||||
contain "Linux").
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
|
||||
_DATA_DIR = Path(__file__).parent / "data"
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class UASignature:
|
||||
name: str
|
||||
ua_substrings: tuple[str, ...]
|
||||
|
||||
|
||||
def _load(filename: str) -> list[UASignature]:
|
||||
raw = json.loads((_DATA_DIR / filename).read_text())
|
||||
return [UASignature(name=e["name"], ua_substrings=tuple(e["ua_substrings"])) for e in raw]
|
||||
|
||||
|
||||
_BROWSER_SIGNATURES = _load("browser_signatures.json")
|
||||
_OS_SIGNATURES = _load("os_signatures.json")
|
||||
|
||||
|
||||
def _match(user_agent: str, signatures: list[UASignature], default: str) -> str:
|
||||
for sig in signatures:
|
||||
if any(sub in user_agent for sub in sig.ua_substrings):
|
||||
return sig.name
|
||||
return default
|
||||
|
||||
|
||||
def classify_browser(user_agent: str | None) -> str:
|
||||
if not user_agent:
|
||||
return "Unknown"
|
||||
return _match(user_agent, _BROWSER_SIGNATURES, "Other")
|
||||
|
||||
|
||||
def classify_os(user_agent: str | None) -> str:
|
||||
if not user_agent:
|
||||
return "Unknown"
|
||||
return _match(user_agent, _OS_SIGNATURES, "Other")
|
||||
Reference in New Issue
Block a user