start project

This commit is contained in:
Hemmat
2026-08-07 21:17:17 +03:30
commit ea1e1eead6
121 changed files with 8108 additions and 0 deletions
View File
+199
View File
@@ -0,0 +1,199 @@
"""Rollup computation (Chapter 06 + Method A extensions from Ch08/09/10).
Called once per parsed file (app/cli.py::process_logs) for the dates it
touched, ad hoc via `flask rollup` for a manual recompute, and now also
after a file deletion (app/services/file_deletion.py) for whatever dates
the deleted file touched. Every _upsert_* function scans log_entries
inside this background batch job, never at request time — that's what
makes Ch03 rule 6 compliance possible.
CORRECTNESS FIX: every rollup writer below now deletes a day's existing
rows before writing whatever the fresh scan finds (including writing
nothing, if a day now has zero data). Three of the five writers
previously only ever upserted-when-present and silently left stale rows
behind when a day's data disappeared — unreachable before file deletion
existed (rollups only ever grew), but a real correctness bug once
deletion makes "this day now has less data than before" possible. Only
the two per-IP writers already had this right (Ch10 follow-up); the
other three are fixed here to match.
"""
from __future__ import annotations
from collections import defaultdict
from datetime import date, datetime, time, timedelta
from sqlalchemy import case, func
from app.extensions import db
from app.models.browser_stats import BrowserStatsDaily
from app.models.human_path_stats import HumanPathStatsDaily
from app.models.ip_traffic_stats import IpPathStatsDaily, IpStatusStatsDaily
from app.models.log_entry import LogEntry
from app.models.referrer_stats import ReferrerStatsDaily
from app.models.request_stats import RequestStatsDaily, RequestStatsHourly
from app.services.blocklist import refresh_blocklist_suggestions
from app.services.referrer import referrer_domain
from app.services.ua_classifier import classify_browser, classify_os
from app.utils.http_status import status_bucket
TOP_PATHS_PER_IP_PER_DAY = 15 # bounds ip_path_stats_daily row growth (Ch10 follow-up)
def compute_rollups_for_range(start: date, end: date) -> None:
"""Recompute every rollup for each day in [start, end]. Site-wide,
not per-file (Ch01: single site) — recomputing from scratch per day
avoids double-counting when two uploads cover the same period, and
correctly shrinks a day's numbers back down when a file covering
that day is deleted.
"""
current = start
while current <= end:
_upsert_hourly(current)
_upsert_daily(current)
_upsert_per_line_derived_stats(current)
refresh_blocklist_suggestions(current)
current += timedelta(days=1)
def _day_bounds(day: date) -> tuple[datetime, datetime]:
start = datetime.combine(day, time.min)
return start, start + timedelta(days=1)
def _upsert_hourly(day: date) -> None:
start, end = _day_bounds(day)
rows = (
db.session.query(
func.strftime("%Y-%m-%d %H:00:00", LogEntry.timestamp).label("date_hour"),
LogEntry.path,
LogEntry.status_code,
func.count().label("count"),
func.coalesce(func.sum(LogEntry.bytes_sent), 0).label("bytes_sent_sum"),
)
.filter(LogEntry.timestamp >= start, LogEntry.timestamp < end)
.group_by("date_hour", LogEntry.path, LogEntry.status_code)
.all()
)
# Delete-then-insert: replaces the day's hourly rows entirely,
# including leaving none behind if `rows` is now empty (e.g. the
# only file covering this day was just deleted).
db.session.query(RequestStatsHourly).filter(
RequestStatsHourly.date_hour >= start, RequestStatsHourly.date_hour < end
).delete()
if rows:
payload = [
{
"date_hour": datetime.strptime(r.date_hour, "%Y-%m-%d %H:%M:%S"),
"path": r.path,
"status_code": r.status_code,
"count": r.count,
"bytes_sent_sum": r.bytes_sent_sum,
}
for r in rows
]
db.session.execute(RequestStatsHourly.__table__.insert(), payload)
db.session.commit()
def _upsert_daily(day: date) -> None:
start, end = _day_bounds(day)
result = (
db.session.query(
func.count().label("count"),
func.count(func.distinct(LogEntry.ip)).label("unique_ips"),
func.coalesce(func.sum(LogEntry.bytes_sent), 0).label("bytes_sum"),
func.coalesce(func.sum(case((LogEntry.status_code >= 400, 1), else_=0)), 0).label("error_count"),
)
.filter(LogEntry.timestamp >= start, LogEntry.timestamp < end)
.one()
)
# Delete-then-insert: if this day now has zero entries (its only
# contributing file was deleted), the stale row is removed rather
# than left behind — no rollup row is better than a wrong one.
db.session.query(RequestStatsDaily).filter(RequestStatsDaily.date == day).delete()
if result.count > 0:
db.session.execute(
RequestStatsDaily.__table__.insert(),
{
"date": day,
"count": result.count,
"unique_ips": result.unique_ips,
"bytes_sum": result.bytes_sum,
"error_count": result.error_count,
},
)
db.session.commit()
def _upsert_per_line_derived_stats(day: date) -> None:
"""Referrer domain, browser/OS, human-only path counts (Ch08/09), and
per-IP path/status counts (Ch10 follow-up) — one streamed pass over
log_entries (Ch03 rule 1: bounded per-chunk memory via yield_per,
never the whole day loaded at once). Every table here uses the same
delete-then-insert pattern so a day's rows are fully replaced by
whatever the fresh scan finds, including nothing.
"""
start, end = _day_bounds(day)
referrer_counts: dict[str, int] = defaultdict(int)
browser_counts: dict[tuple[str, str], int] = defaultdict(int)
human_path_counts: dict[str, int] = defaultdict(int)
ip_path_counts: dict[str, dict[str, int]] = defaultdict(lambda: defaultdict(int))
ip_status_counts: dict[str, dict[str, int]] = defaultdict(lambda: defaultdict(int))
query = (
db.session.query(
LogEntry.referrer, LogEntry.user_agent, LogEntry.is_bot,
LogEntry.path, LogEntry.ip, LogEntry.status_code,
)
.filter(LogEntry.timestamp >= start, LogEntry.timestamp < end)
)
for referrer, user_agent, is_bot, path, ip, status_code in query.yield_per(1000):
domain = referrer_domain(referrer)
if domain:
referrer_counts[domain] += 1
if not is_bot: # Ch08: bot traffic excluded from human browser/OS breakdown
browser_counts[(classify_browser(user_agent), classify_os(user_agent))] += 1
human_path_counts[path] += 1
ip_path_counts[ip][path] += 1
ip_status_counts[ip][status_bucket(status_code)] += 1
db.session.query(ReferrerStatsDaily).filter(ReferrerStatsDaily.date == day).delete()
if referrer_counts:
payload = [{"date": day, "referrer_domain": d, "count": c} for d, c in referrer_counts.items()]
db.session.execute(ReferrerStatsDaily.__table__.insert(), payload)
db.session.query(BrowserStatsDaily).filter(BrowserStatsDaily.date == day).delete()
if browser_counts:
payload = [{"date": day, "browser": b, "os": o, "count": c} for (b, o), c in browser_counts.items()]
db.session.execute(BrowserStatsDaily.__table__.insert(), payload)
db.session.query(HumanPathStatsDaily).filter(HumanPathStatsDaily.date == day).delete()
if human_path_counts:
payload = [{"date": day, "path": p, "count": c} for p, c in human_path_counts.items()]
db.session.execute(HumanPathStatsDaily.__table__.insert(), payload)
db.session.query(IpPathStatsDaily).filter(IpPathStatsDaily.date == day).delete()
if ip_path_counts:
payload = []
for ip, paths in ip_path_counts.items():
top_paths = sorted(paths.items(), key=lambda kv: kv[1], reverse=True)[:TOP_PATHS_PER_IP_PER_DAY]
payload.extend({"date": day, "ip": ip, "path": p, "count": c} for p, c in top_paths)
if payload:
db.session.execute(IpPathStatsDaily.__table__.insert(), payload)
db.session.query(IpStatusStatsDaily).filter(IpStatusStatsDaily.date == day).delete()
if ip_status_counts:
payload = [
{"date": day, "ip": ip, "status_bucket": bucket, "count": c}
for ip, buckets in ip_status_counts.items()
for bucket, c in buckets.items()
]
db.session.execute(IpStatusStatsDaily.__table__.insert(), payload)
db.session.commit()
+97
View File
@@ -0,0 +1,97 @@
"""No-cron automatic processing (Chapter 12 simplification, per project
owner request): triggers file parsing immediately in a background thread
right after upload, and opportunistically resumes any incomplete files
when the Overview page loads — replacing the cron-triggered model.
`flask process-logs` still exists in app/cli.py for anyone who'd rather
use cron, but nothing requires it anymore.
TRADEOFF (flagged, deviating from Chapter 02/03's "no persistent
background workers, cron-triggered CLI only" stance): a background thread
lives inside the same worker process that handled the upload request. If
Passenger recycles that process mid-parse, the thread dies with it —
progress up to the last commit is still safely checkpointed (same bounded-
batch model as before), but nothing will automatically resume it without
either cron or a page visit. The "resume on page load" hook below is the
deliberate replacement for that guarantee: visiting the Overview tab
re-triggers processing for anything left incomplete, so in the worst case
a stuck file resumes the next time the admin looks at the dashboard,
rather than never.
This is not a long-lived daemon: each thread terminates once its file
reaches "done"/"error" (or the process is killed), and no thread survives
a process restart — it just gets re-triggered fresh next time.
"""
from __future__ import annotations
import threading
from flask import Flask
from app.extensions import db
from app.models.log_file import LogFile
from app.services.log_processor import process_one_batch
# In-process guard against launching two threads for the same file at
# once (e.g. the upload trigger and a page-load resume firing close
# together). Per-worker-process only — a different entry process picking
# up the same file concurrently is a low-probability edge case accepted
# for this simplification; each write is still a small checkpointed
# commit, not a giant one, which limits how bad a collision could be.
_active_file_ids: set[int] = set()
_lock = threading.Lock()
def _claim(log_file_id: int) -> bool:
with _lock:
if log_file_id in _active_file_ids:
return False
_active_file_ids.add(log_file_id)
return True
def _release(log_file_id: int) -> None:
with _lock:
_active_file_ids.discard(log_file_id)
def _run_to_completion(app: Flask, log_file_id: int, batch_size: int) -> None:
with app.app_context():
try:
log_file = db.session.get(LogFile, log_file_id)
if log_file is None:
return
while log_file.status in ("queued", "processing"):
process_one_batch(log_file, batch_size)
db.session.refresh(log_file)
except Exception:
app.logger.exception("Background processing failed for log_file_id=%s", log_file_id)
log_file = db.session.get(LogFile, log_file_id)
if log_file is not None and log_file.status != "done":
log_file.status = "error"
log_file.error_message = "Processing failed unexpectedly; see server logs."
db.session.commit()
finally:
_release(log_file_id)
def trigger_processing(app: Flask, log_file_id: int) -> None:
"""Start background processing for one file; no-ops if already running."""
if not _claim(log_file_id):
return
batch_size = app.config["PARSE_BATCH_SIZE"]
thread = threading.Thread(
target=_run_to_completion, args=(app, log_file_id, batch_size), daemon=True
)
thread.start()
def resume_incomplete_files(app: Flask) -> None:
"""Opportunistic resume hook, called from the Overview page load —
the deliberate replacement for cron's "there's always a next tick"
guarantee. Cheap: one indexed status-filtered query.
"""
incomplete_ids = [
lf.id for lf in LogFile.query.filter(LogFile.status.in_(["queued", "processing"])).all()
]
for log_file_id in incomplete_ids:
trigger_processing(app, log_file_id)
+67
View File
@@ -0,0 +1,67 @@
"""Blocklist-suggestion generation (Chapter 10 follow-up): no earlier
chapter assigned ownership of populating blocklist_suggestions or setting
ip_registry.is_flagged. Runs in the background aggregator pass (batch, not
request-time, per Ch02/03), reusing severity_scoring.py so there's exactly
one scoring model between the dashboard display and the flagging decision.
"""
from __future__ import annotations
from collections import defaultdict
from datetime import date, datetime, timedelta
from app.extensions import db
from app.models.blocklist_suggestion import BlocklistSuggestion
from app.models.ip_registry import IPRegistry
from app.models.suspicious_event import SuspiciousEvent
from app.services.severity_scoring import SeverityInputs, compute_effective_severity
_RANK = {"low": 0, "medium": 1, "high": 2}
def refresh_blocklist_suggestions(day: date) -> None:
"""Flag an IP (is_flagged + a suggestion row) if its escalated severity
for `day` reaches 'high'. Idempotent — skips IPs already suggested.
"""
start = datetime.combine(day, datetime.min.time())
end = start + timedelta(days=1)
events = (
db.session.query(SuspiciousEvent.ip, SuspiciousEvent.timestamp, SuspiciousEvent.severity)
.filter(SuspiciousEvent.timestamp >= start, SuspiciousEvent.timestamp < end)
.all()
)
if not events:
return
by_ip: dict[str, list] = defaultdict(list)
for ip, ts, sev in events:
by_ip[ip].append((ts, sev))
already_suggested = {ip for (ip,) in db.session.query(BlocklistSuggestion.ip).distinct().all()}
for ip, ip_events in by_ip.items():
if ip in already_suggested:
continue
timestamps = sorted(ts for ts, _ in ip_events)
avg_interval = (
(timestamps[-1] - timestamps[0]).total_seconds() / (len(timestamps) - 1)
if len(timestamps) > 1 else None
)
worst_base = max((sev for _, sev in ip_events), key=lambda s: _RANK.get(s, 0))
effective = compute_effective_severity(
SeverityInputs(base_severity=worst_base, ip_event_count=len(ip_events), avg_interval_seconds=avg_interval)
)
if effective != "high":
continue
db.session.add(BlocklistSuggestion(
ip=ip,
reason=f"{len(ip_events)} suspicious event(s) on {day.isoformat()}, escalated to high severity",
created_at=datetime.utcnow(),
exported=False,
))
ip_row = db.session.get(IPRegistry, ip)
if ip_row is not None:
ip_row.is_flagged = True
db.session.commit()
+78
View File
@@ -0,0 +1,78 @@
"""Bot signature matching + reverse-DNS verification (Chapter 07).
Signatures are data (JSON), not hardcoded logic, so the list grows without
a code change. Verification does real DNS I/O — only ever called from the
background process-logs batch job (app/services/classification.py), never
synchronously inside a dashboard request, per Chapter 07's explicit rule.
"""
from __future__ import annotations
import json
import socket
from dataclasses import dataclass
from datetime import datetime, timedelta
from pathlib import Path
SIGNATURES_PATH = Path(__file__).parent / "data" / "bot_signatures.json"
# Skip re-verifying the same IP more often than this (Ch07: DNS latency is
# a real cost on constrained hosting).
VERIFICATION_TTL = timedelta(days=7)
@dataclass(frozen=True)
class BotSignature:
name: str
ua_substrings: tuple[str, ...]
verify_suffixes: tuple[str, ...] # PTR hostname must end in one of these
def _load_signatures() -> list[BotSignature]:
raw = json.loads(SIGNATURES_PATH.read_text())
return [
BotSignature(name=e["name"], ua_substrings=tuple(e["ua_substrings"]), verify_suffixes=tuple(e["verify_suffixes"]))
for e in raw
]
_SIGNATURES = _load_signatures()
def classify_bot(user_agent: str | None) -> str | None:
"""Return the claimed bot name via UA substring match, or None."""
if not user_agent:
return None
ua_lower = user_agent.lower()
for sig in _SIGNATURES:
if any(sub.lower() in ua_lower for sub in sig.ua_substrings):
return sig.name
return None
def _signature_for(bot_name: str) -> BotSignature | None:
return next((s for s in _SIGNATURES if s.name == bot_name), None)
def verify_bot_ip(ip: str, bot_name: str) -> bool:
"""Reverse-DNS + forward-confirm that `ip` really belongs to `bot_name`."""
sig = _signature_for(bot_name)
if sig is None:
return False
try:
hostname, _, _ = socket.gethostbyaddr(ip)
except (socket.herror, socket.gaierror, OSError):
return False
if not any(hostname.lower().endswith(suffix) for suffix in sig.verify_suffixes):
return False
try:
forward_ips = socket.gethostbyname_ex(hostname)[2]
except (socket.herror, socket.gaierror, OSError):
return False
return ip in forward_ips
def is_verification_stale(last_verified_at: datetime | None) -> bool:
"""True if this IP needs a fresh DNS check (Ch07 caching rule)."""
if last_verified_at is None:
return True
return datetime.utcnow() - last_verified_at > VERIFICATION_TTL
+122
View File
@@ -0,0 +1,122 @@
"""Per-batch classification pipeline (Chapter 07): ties bot_identifier and
threat_scanner into the bot_hits/suspicious_events/ip_registry write path.
Called once per bulk-insert chunk from app/cli.py — DNS-based verification
belongs here (background batch job), never in a dashboard request.
"""
from __future__ import annotations
from dataclasses import dataclass
from datetime import datetime
from sqlalchemy import func, insert
from sqlalchemy.dialects.sqlite import insert as sqlite_insert
from app.extensions import db
from app.models.bot_hit import BotHit
from app.models.ip_registry import IPRegistry
from app.models.suspicious_event import SuspiciousEvent
from app.services import bot_identifier, threat_scanner
from app.services.log_parser import ParsedEntry
@dataclass
class EntryClassification:
"""Cheap, no-I/O flags for the log_entries row itself."""
is_bot: bool
flagged: bool
def classify_entry(entry: ParsedEntry) -> EntryClassification:
"""No DNS I/O here — bot *verification* is batched separately below,
since it's stateful (cached per IP) and only worth doing once per IP
per batch, not once per line.
"""
return EntryClassification(
is_bot=bot_identifier.classify_bot(entry.user_agent) is not None,
flagged=threat_scanner.scan(entry) is not None,
)
def write_batch_side_effects(log_file_id: int, entries: list[ParsedEntry]) -> None:
"""Derive and bulk-write bot_hits, suspicious_events, ip_registry upserts."""
if not entries:
return
bot_hit_rows: list[dict] = []
suspicious_rows: list[dict] = []
ip_agg: dict[str, dict] = {}
verified_this_batch: dict[str, bool] = {} # avoid repeat DNS for the same IP in one batch
for entry in entries:
agg = ip_agg.setdefault(entry.ip, {"first": entry.timestamp, "last": entry.timestamp, "count": 0})
agg["count"] += 1
agg["first"] = min(agg["first"], entry.timestamp)
agg["last"] = max(agg["last"], entry.timestamp)
bot_name = bot_identifier.classify_bot(entry.user_agent)
if bot_name is not None:
verified = verified_this_batch.get(entry.ip)
if verified is None:
verified = _verify_with_cache(entry.ip, bot_name)
verified_this_batch[entry.ip] = verified
bot_hit_rows.append({
"log_file_id": log_file_id, "timestamp": entry.timestamp, "ip": entry.ip,
"bot_name": bot_name, "verified": verified, "path": entry.path,
"status_code": entry.status_code,
})
if not verified:
# Single detection, two dashboard consumers (Ch07): spoofed
# bot also surfaces as a suspicious_events row.
suspicious_rows.append({
"log_file_id": log_file_id, "ip": entry.ip, "timestamp": entry.timestamp,
"path": entry.path, "rule_matched": f"spoofed_bot:{bot_name}", "severity": "medium",
})
threat = threat_scanner.scan(entry)
if threat is not None:
suspicious_rows.append({
"log_file_id": log_file_id, "ip": entry.ip, "timestamp": entry.timestamp,
"path": entry.path, "rule_matched": threat.rule_matched, "severity": threat.severity,
})
if bot_hit_rows:
db.session.execute(insert(BotHit.__table__), bot_hit_rows)
if suspicious_rows:
db.session.execute(insert(SuspiciousEvent.__table__), suspicious_rows)
_upsert_ip_registry(ip_agg, verified_this_batch)
db.session.commit()
def _verify_with_cache(ip: str, bot_name: str) -> bool:
"""TTL-gated reverse/forward DNS check, cached via
ip_registry.last_verified_bot_result (Ch07 schema addition).
"""
row = db.session.get(IPRegistry, ip)
if row is not None and not bot_identifier.is_verification_stale(row.last_verified_at):
return bool(row.last_verified_bot_result)
return bot_identifier.verify_bot_ip(ip, bot_name)
def _upsert_ip_registry(ip_agg: dict[str, dict], verified_this_batch: dict[str, bool]) -> None:
now = datetime.utcnow()
for ip, agg in ip_agg.items():
values = {
"ip": ip, "first_seen": agg["first"], "last_seen": agg["last"],
"total_requests": agg["count"], "reputation_score": 0, "is_flagged": False,
}
if ip in verified_this_batch:
values["last_verified_at"] = now
values["last_verified_bot_result"] = verified_this_batch[ip]
stmt = sqlite_insert(IPRegistry.__table__).values(**values)
update_set = {
"last_seen": func.max(IPRegistry.last_seen, stmt.excluded.last_seen),
"first_seen": func.min(IPRegistry.first_seen, stmt.excluded.first_seen),
"total_requests": IPRegistry.total_requests + stmt.excluded.total_requests,
}
if ip in verified_this_batch:
update_set["last_verified_at"] = stmt.excluded.last_verified_at
update_set["last_verified_bot_result"] = stmt.excluded.last_verified_bot_result
stmt = stmt.on_conflict_do_update(index_elements=["ip"], set_=update_set)
db.session.execute(stmt)
+9
View File
@@ -0,0 +1,9 @@
[
{"name": "Googlebot", "ua_substrings": ["Googlebot"], "verify_suffixes": [".googlebot.com", ".google.com"]},
{"name": "Bingbot", "ua_substrings": ["bingbot"], "verify_suffixes": [".search.msn.com"]},
{"name": "Yandex", "ua_substrings": ["YandexBot"], "verify_suffixes": [".yandex.ru", ".yandex.com", ".yandex.net"]},
{"name": "Baidu", "ua_substrings": ["Baiduspider"], "verify_suffixes": [".baidu.com", ".baidu.jp"]},
{"name": "DuckDuckBot", "ua_substrings": ["DuckDuckBot"], "verify_suffixes": [".duckduckgo.com"]},
{"name": "AhrefsBot", "ua_substrings": ["AhrefsBot"], "verify_suffixes": [".ahrefs.com"]},
{"name": "SemrushBot", "ua_substrings": ["SemrushBot"], "verify_suffixes": [".semrush.com"]}
]
@@ -0,0 +1,8 @@
[
{"name": "Edge", "ua_substrings": ["Edg/", "EdgA/", "EdgiOS/"]},
{"name": "Opera", "ua_substrings": ["OPR/", "Opera"]},
{"name": "Chrome", "ua_substrings": ["Chrome/", "CriOS/"]},
{"name": "Firefox", "ua_substrings": ["Firefox/", "FxiOS/"]},
{"name": "Safari", "ua_substrings": ["Safari/"]},
{"name": "Internet Explorer", "ua_substrings": ["MSIE ", "Trident/"]}
]
+7
View File
@@ -0,0 +1,7 @@
[
{"name": "Windows", "ua_substrings": ["Windows NT"]},
{"name": "iOS", "ua_substrings": ["iPhone", "iPad", "iPod"]},
{"name": "macOS", "ua_substrings": ["Mac OS X", "Macintosh"]},
{"name": "Android", "ua_substrings": ["Android"]},
{"name": "Linux", "ua_substrings": ["Linux"]}
]
+11
View File
@@ -0,0 +1,11 @@
{
"sensitive_paths": [
"/.env", "/.git/config", "wp-config.php", "/.htpasswd", "/phpmyadmin",
"/xmlrpc.php", ".sql.gz", ".sql.bak", ".zip", ".bak", ".old"
],
"injection_markers": [
"union select", "' or '1'='1", "<script>", "javascript:",
"../", "..%2f", "%2e%2e%2f", "%252e%252e%252f"
],
"scanner_user_agents": ["sqlmap", "nikto", "nmap", "acunetix", "nessus"]
}
+93
View File
@@ -0,0 +1,93 @@
"""Uploaded-file deletion (project-owner follow-up request).
Deleting a LogFile is more than one DELETE statement: log_entries,
bot_hits, and suspicious_events all reference log_file_id and must be
removed explicitly — this project's SQLite connections don't have
`PRAGMA foreign_keys=ON`, so the `ondelete="CASCADE"` in the migrations
(Ch06) is declarative documentation only, not an enforced behavior.
Relying on it would silently leave orphaned rows behind.
The harder part is the rollup tables (request_stats_hourly/daily,
referrer/browser/human-path/ip-path/ip-status): none of them have a
log_file_id column (Ch06: they're site-wide, since two files can share a
date). So instead of deleting rollup rows for the deleted file's dates
directly (which could wipe out another file's contribution to the same
date), this captures the affected dates BEFORE deleting, then calls
aggregator.compute_rollups_for_range() AFTER deleting — which recomputes
each affected day from whatever log_entries remain, correctly handling
both "another file still covers this day" and "no file covers this day
anymore" (the latter now handled correctly by the aggregator.py fix that
shipped alongside this feature).
KNOWN LIMITATION (flagged, not fixed): ip_registry (total_requests,
first_seen, last_seen, verification cache) and blocklist_suggestions are
NOT per-file and are NOT recomputed on deletion — doing so correctly
would mean re-scanning all remaining log_entries for every affected IP,
which is unbounded work for a single delete action and would violate the
same "no raw-row scan on a hot path" reasoning Ch03 applies elsewhere.
After deleting a file, IP History numbers may include a deleted file's
historical contribution until a full site-wide recompute is added as a
separate feature. Doesn't affect correctness of Overview/SEO's date-
scoped numbers, which is what this feature was actually asked to fix.
"""
from __future__ import annotations
from dataclasses import dataclass, field
from app.extensions import db
from app.models.bot_hit import BotHit
from app.models.log_entry import LogEntry
from app.models.log_file import LogFile
from app.models.suspicious_event import SuspiciousEvent
from app.services import aggregator
from app.utils.upload_paths import upload_path_for
@dataclass
class BulkDeleteResult:
deleted: list[int] = field(default_factory=list)
skipped: list[dict] = field(default_factory=list) # [{"id": ..., "reason": ...}]
def delete_log_files(log_file_ids: list[int]) -> BulkDeleteResult:
"""Delete each id in `log_file_ids`; recompute rollups once at the end
for the full span of dates any deleted file touched (cheaper than
recomputing per-file when multiple files share dates).
"""
result = BulkDeleteResult()
all_touched_dates: set = set()
for log_file_id in log_file_ids:
log_file = db.session.get(LogFile, log_file_id)
if log_file is None:
result.skipped.append({"id": log_file_id, "reason": "not found"})
continue
if log_file.status == "processing":
result.skipped.append({"id": log_file_id, "reason": "currently being analyzed"})
continue
touched_dates = {
row.date() for row, in db.session.query(LogEntry.timestamp)
.filter(LogEntry.log_file_id == log_file_id).distinct()
}
# distinct() on a full timestamp rarely collapses much; reduce to
# calendar dates in Python since SQLite's DATE() in a DISTINCT
# clause is a bit more awkward to express portably here.
all_touched_dates |= touched_dates
db.session.query(SuspiciousEvent).filter(SuspiciousEvent.log_file_id == log_file_id).delete()
db.session.query(BotHit).filter(BotHit.log_file_id == log_file_id).delete()
db.session.query(LogEntry).filter(LogEntry.log_file_id == log_file_id).delete()
raw_path = upload_path_for(log_file)
if raw_path.exists():
raw_path.unlink()
db.session.delete(log_file)
db.session.commit()
result.deleted.append(log_file_id)
if all_touched_dates:
aggregator.compute_rollups_for_range(min(all_touched_dates), max(all_touched_dates))
return result
+47
View File
@@ -0,0 +1,47 @@
"""Log parser abstraction (Chapter 07)."""
from __future__ import annotations
from dataclasses import dataclass
from datetime import datetime
from typing import Protocol, TYPE_CHECKING
if TYPE_CHECKING:
from app.models.log_file import LogFile
@dataclass
class ParsedEntry:
timestamp: datetime
ip: str
method: str
path: str
status_code: int
bytes_sent: int
referrer: str | None
user_agent: str | None
class LogParser(Protocol):
def parse_line(self, line: str) -> ParsedEntry | None: ...
def get_parser_for(log_file: "LogFile") -> LogParser:
"""Build the right LogParser for a given upload's format_string.
ASSUMPTION (flagged): a blank format_string — which the upload form
allows — falls back to Combined, since Chapter 07 states no default
for that case. Anything else is compiled as-is; a real custom
LiteSpeed format is never silently replaced by a preset.
"""
# Imported lazily to avoid a circular import (apache.py imports
# ParsedEntry from this module).
from app.services.log_parser.apache import FormatCompiledParser, combined_parser, common_parser
from app.services.log_parser.format_compiler import compile_format
fmt = (log_file.format_string or "").strip()
key = fmt.lower()
if key in ("", "combined"):
return combined_parser()
if key == "common":
return common_parser()
return FormatCompiledParser(compile_format(fmt))
+57
View File
@@ -0,0 +1,57 @@
"""Apache Common/Combined presets + the generic parser built from
format_compiler (Chapter 07)."""
from __future__ import annotations
from app.services.log_parser import ParsedEntry
from app.services.log_parser.format_compiler import (
CompiledFormat, compile_format, parse_apache_timestamp, parse_request_line, validate_ip,
)
COMMON_LOG_FORMAT = '%h %l %u %t "%r" %>s %b'
COMBINED_LOG_FORMAT = '%h %l %u %t "%r" %>s %b "%{Referer}i" "%{User-agent}i"'
class FormatCompiledParser:
"""LogParser implementation driven by any CompiledFormat (Ch04/07 protocol)."""
def __init__(self, compiled: CompiledFormat):
self._compiled = compiled
def parse_line(self, line: str) -> ParsedEntry | None:
match = self._compiled.pattern.match(line.rstrip("\n"))
if match is None:
return None # malformed line — skip, never raise (Ch07)
fields = match.groupdict()
ip = validate_ip(fields["ip"])
if ip is None:
return None
req = parse_request_line(fields["request"])
if req is None:
return None
method, path = req
try:
timestamp = parse_apache_timestamp(fields["timestamp"])
status_code = int(fields["status"])
bytes_sent = 0 if fields["bytes"] == "-" else int(fields["bytes"])
except (ValueError, KeyError):
return None
referrer = fields.get("referrer")
user_agent = fields.get("user_agent")
return ParsedEntry(
timestamp=timestamp, ip=ip, method=method, path=path,
status_code=status_code, bytes_sent=bytes_sent,
referrer=None if referrer in (None, "-") else referrer,
user_agent=None if user_agent in (None, "-") else user_agent,
)
def common_parser() -> FormatCompiledParser:
return FormatCompiledParser(compile_format(COMMON_LOG_FORMAT))
def combined_parser() -> FormatCompiledParser:
return FormatCompiledParser(compile_format(COMBINED_LOG_FORMAT))
@@ -0,0 +1,95 @@
"""LogFormat directive -> compiled regex + named groups (Chapter 07).
One code path for Common, Combined, and arbitrary custom formats — presets
below are just specific format strings run through this same compiler.
"""
from __future__ import annotations
import re
from dataclasses import dataclass
from datetime import datetime, timezone
from ipaddress import ip_address
# Directive -> (group name, regex pattern). NOTE: %r and %{...}i patterns
# deliberately do NOT include quote characters — the quotes around them
# are literal text already present in the format string (e.g. `"%r"`),
# and get regex-escaped by the literal-text path below. %t is different:
# its brackets are part of what %t itself produces, not literal format-
# string text, so they belong in this directive's own pattern.
_SIMPLE_DIRECTIVES: dict[str, tuple[str, str]] = {
"%h": ("ip", r"(?P<ip>\S+)"),
"%l": ("ident", r"(?P<ident>\S+)"),
"%u": ("user", r"(?P<user>\S+)"),
"%t": ("timestamp", r"\[(?P<timestamp>[^\]]+)\]"),
"%r": ("request", r'(?P<request>[^"]*)'),
"%>s": ("status", r"(?P<status>\d{3})"),
"%s": ("status", r"(?P<status>\d{3})"),
"%b": ("bytes", r"(?P<bytes>\d+|-)"),
}
_HEADER_DIRECTIVE_RE = re.compile(r'%\{([^}]+)\}i')
_KNOWN_HEADER_GROUPS = {"referer": "referrer", "user-agent": "user_agent"}
_DIRECTIVE_TOKEN_RE = re.compile(r"%>?\{[^}]+\}i|%>?[a-zA-Z]")
@dataclass(frozen=True)
class CompiledFormat:
pattern: re.Pattern
group_names: frozenset[str]
def compile_format(format_string: str) -> CompiledFormat:
"""Compile a LogFormat directive (Common/Combined preset or a pasted
custom vhost format) into one regex."""
regex_parts: list[str] = []
group_names: set[str] = set()
pos = 0
for match in _DIRECTIVE_TOKEN_RE.finditer(format_string):
literal = format_string[pos:match.start()]
if literal:
regex_parts.append(re.escape(literal))
regex_parts.append(_directive_to_pattern(match.group(0), group_names))
pos = match.end()
regex_parts.append(re.escape(format_string[pos:]))
pattern = re.compile("^" + "".join(regex_parts) + r"\s*$")
return CompiledFormat(pattern=pattern, group_names=frozenset(group_names))
def _directive_to_pattern(token: str, group_names: set[str]) -> str:
header_match = _HEADER_DIRECTIVE_RE.fullmatch(token)
if header_match:
header_name = header_match.group(1).lower()
group = _KNOWN_HEADER_GROUPS.get(header_name, re.sub(r"[^a-z0-9]+", "_", header_name))
group_names.add(group)
return f'(?P<{group}>[^"]*)'
if token not in _SIMPLE_DIRECTIVES:
raise ValueError(f"Unsupported LogFormat directive: {token!r}")
group, pattern = _SIMPLE_DIRECTIVES[token]
group_names.add(group)
return pattern
def parse_apache_timestamp(raw: str) -> datetime:
"""'10/Oct/2026:13:55:36 -0700' -> naive UTC datetime (Ch07: store normalized to UTC)."""
dt = datetime.strptime(raw, "%d/%b/%Y:%H:%M:%S %z")
return dt.astimezone(timezone.utc).replace(tzinfo=None)
def parse_request_line(raw: str) -> tuple[str, str] | None:
"""Split '%r' ("GET /path HTTP/1.1") into (method, path); None if malformed."""
parts = raw.split()
if len(parts) != 3:
return None
method, path, _protocol = parts
return method, path
def validate_ip(raw: str) -> str | None:
"""Return `raw` if a valid IPv4/IPv6 address, else None (Ch07)."""
try:
ip_address(raw)
except ValueError:
return None
return raw
+10
View File
@@ -0,0 +1,10 @@
"""LiteSpeed uses the same LogFormat directives as Apache (Chapter 07):
Combined generally works unmodified. get_parser_for() still always
compiles the vhost's real format_string though — this module is just the
named default, never a silent override of a custom format.
"""
from __future__ import annotations
from app.services.log_parser.apache import combined_parser as default_litespeed_parser
__all__ = ["default_litespeed_parser"]
+148
View File
@@ -0,0 +1,148 @@
"""Core log-file processing pipeline (Chapter 07), factored out of
app/cli.py so it has exactly one implementation shared by:
- the optional `flask process-logs` CLI command (for anyone who still
wants cron), and
- the automatic background-thread trigger fired right after upload and
opportunistically on page load (app/services/background.py) — the
no-cron-required simplification.
process_one_batch() keeps the same bounded-batch, checkpointed-commit
contract as the original cron design (Ch03 rule 1/4/9): it still never
loads a whole file into memory, still commits progress incrementally, and
still stops after `batch_size` lines so a very large file doesn't hold one
enormous open transaction — the only thing that changed is *who* calls it
again for the next batch.
"""
from __future__ import annotations
import gzip
import itertools
from datetime import date
from pathlib import Path
from sqlalchemy import insert
from app.extensions import db
from app.models.log_entry import LogEntry
from app.models.log_file import LogFile
from app.services import aggregator
from app.services.classification import classify_entry, write_batch_side_effects
from app.services.log_parser import ParsedEntry, get_parser_for
from app.utils.upload_paths import upload_path_for
def open_log_stream(path: Path):
"""Open a log file, transparently decompressing .gz, one line at a time."""
if path.suffix == ".gz":
return gzip.open(path, mode="rt", encoding="utf-8", errors="replace")
return path.open("r", encoding="utf-8", errors="replace")
def count_total_lines(path: Path) -> int:
"""One streaming pass to count lines (bounded memory — never the whole
file at once, Ch03 rule 1), so the UI can show a real
processed/total percentage instead of an indeterminate spinner.
This is an extra sequential read of the file beyond the parse pass
itself. That cost is now worth paying: processing used to be silently
triggered by cron with no live audience watching, but now it runs
automatically right after upload while the admin is looking at a
progress bar — the UX value of a real percentage justifies the extra
I/O pass.
"""
with open_log_stream(path) as fh:
return sum(1 for _ in fh)
def process_one_batch(log_file: LogFile, batch_size: int) -> None:
"""Parse up to `batch_size` lines from `log_file`'s current checkpoint.
Leaves status as "processing" (with progress already committed) if
the file isn't finished yet — the caller decides whether to invoke
this again: the CLI calls it once per pending file per invocation
(unchanged cron-tick semantics); the background thread loops it until
the file reaches "done"/"error".
"""
path = upload_path_for(log_file)
if not path.exists():
log_file.status = "error"
log_file.error_message = f"Upload file missing on disk: {path}"
db.session.commit()
return
if log_file.total_lines is None:
# First pickup of this file — count once, not on every batch.
log_file.total_lines = count_total_lines(path)
log_file.status = "processing"
db.session.commit()
parser = get_parser_for(log_file)
lines_seen_this_run = 0
skipped_this_run = 0
dates_touched: set[date] = set()
with open_log_stream(path) as fh:
remainder = itertools.islice(fh, log_file.processed_lines, None)
batch_entries: list[ParsedEntry] = []
for line in remainder:
entry = parser.parse_line(line)
if entry is not None:
batch_entries.append(entry)
dates_touched.add(entry.timestamp.date())
else:
skipped_this_run += 1 # malformed line — skip, don't abort the batch (Ch07)
log_file.processed_lines += 1
lines_seen_this_run += 1
if len(batch_entries) >= 500:
_bulk_insert_entries(log_file.id, batch_entries)
batch_entries = []
if lines_seen_this_run >= batch_size:
_record_skip_count(log_file, skipped_this_run)
db.session.commit() # persist checkpoint; resumable if interrupted
return
if batch_entries:
_bulk_insert_entries(log_file.id, batch_entries)
if dates_touched:
aggregator.compute_rollups_for_range(min(dates_touched), max(dates_touched))
_record_skip_count(log_file, skipped_this_run)
log_file.status = "done"
db.session.commit()
def _record_skip_count(log_file: LogFile, skipped_this_run: int) -> None:
"""Informational only — doesn't touch `status`.
STOPGAP (flagged in Ch07): log_files has no dedicated skip-counter
column, so this overwrites error_message with the latest run's count
rather than accumulating across runs.
"""
if skipped_this_run:
log_file.error_message = f"{skipped_this_run} unparsable line(s) skipped in the most recent parse run."
def _bulk_insert_entries(log_file_id: int, entries: list[ParsedEntry]) -> None:
"""Bulk-insert log_entries (Ch03 rule 4), classifying is_bot/flagged
per line, then derive bot_hits/suspicious_events/ip_registry (Ch07).
"""
if not entries:
return
payload = []
for e in entries:
classification = classify_entry(e)
payload.append({
"log_file_id": log_file_id, "timestamp": e.timestamp, "ip": e.ip,
"method": e.method, "path": e.path, "status_code": e.status_code,
"bytes_sent": e.bytes_sent, "referrer": e.referrer, "user_agent": e.user_agent,
"is_bot": classification.is_bot, "flagged": classification.flagged,
})
db.session.execute(insert(LogEntry.__table__), payload)
db.session.commit()
write_batch_side_effects(log_file_id, entries)
+19
View File
@@ -0,0 +1,19 @@
"""Referrer -> domain bucketing (Chapter 08 rollup, Method A)."""
from __future__ import annotations
from urllib.parse import urlparse
def referrer_domain(referrer: str | None) -> str | None:
"""Bucket a raw referrer URL to its host, stripping a leading 'www.'.
Full-URL cardinality would make the daily rollup unbounded in row
count; domain bucketing keeps it small (Ch03's bounded-resource bias).
Empty/unparsable referrers return None and are excluded from the rollup.
"""
if not referrer:
return None
host = urlparse(referrer).netloc.lower()
if not host:
return None
return host[4:] if host.startswith("www.") else host
+49
View File
@@ -0,0 +1,49 @@
"""Severity scoring (Chapter 10): a simple, transparent rank-escalation
model — no ML, easy to explain to a non-expert user (Ch10's own requirement).
Chapter 07 assigns a PROVISIONAL severity per rule type at parse time.
This module computes an EFFECTIVE severity by escalating that base rank
for repeated/rapid hits from the same IP — the exact rule Ch10 names.
Escalation is rank-based and strictly non-decreasing: it starts at the
base severity's rank and only ever moves up (repeat-count / hit-rate
bonuses), clamped at "high". An earlier weighted-score-vs-fixed-threshold
version could silently *downgrade* an isolated high-severity event (e.g.
a single sqlmap hit) to "medium" purely because the thresholds weren't
calibrated to the base weights — caught by running the pipeline against
real sample data. A single dangerous event must never end up rated below
its own base severity; only repetition/rate should push it higher.
"""
from __future__ import annotations
from dataclasses import dataclass
_SEVERITY_RANKS = ["low", "medium", "high"]
_RANK_BY_SEVERITY = {name: rank for rank, name in enumerate(_SEVERITY_RANKS)}
_REPEAT_COUNT_HIGH = 20
_REPEAT_COUNT_MEDIUM = 5
_RAPID_AVG_INTERVAL_SECONDS = 60 # >1 event/minute from one IP suggests automation
@dataclass(frozen=True)
class SeverityInputs:
base_severity: str
ip_event_count: int
avg_interval_seconds: float | None # None if this IP has < 2 events in the window
def compute_effective_severity(inputs: SeverityInputs) -> str:
"""Two named, inspectable escalation rules: repeat-count and hit-rate."""
rank = _RANK_BY_SEVERITY.get(inputs.base_severity, 0)
if inputs.ip_event_count >= _REPEAT_COUNT_HIGH:
rank += 2
elif inputs.ip_event_count >= _REPEAT_COUNT_MEDIUM:
rank += 1
if inputs.avg_interval_seconds is not None and inputs.avg_interval_seconds < _RAPID_AVG_INTERVAL_SECONDS:
rank += 1
rank = min(rank, len(_SEVERITY_RANKS) - 1)
return _SEVERITY_RANKS[rank]
+54
View File
@@ -0,0 +1,54 @@
"""Suspicious-pattern matching (Chapter 07), reused as-is by Chapter 10.
Pattern dictionary is data (JSON), same reasoning as the bot signature
table — editable without a code change. Severity here is a provisional
per-rule-type default; Chapter 10 owns the full weighted/escalating score.
"""
from __future__ import annotations
import json
from dataclasses import dataclass
from pathlib import Path
from typing import TYPE_CHECKING
if TYPE_CHECKING:
from app.services.log_parser import ParsedEntry
PATTERNS_PATH = Path(__file__).parent / "data" / "threat_patterns.json"
@dataclass(frozen=True)
class ThreatMatch:
rule_matched: str
severity: str # low|medium|high — provisional; Ch10 may escalate
def _load_patterns() -> dict:
return json.loads(PATTERNS_PATH.read_text())
_PATTERNS = _load_patterns()
def scan(entry: "ParsedEntry") -> ThreatMatch | None:
"""Check one parsed entry against the pattern dictionary.
Cheap substring checks only — no regex backtracking risk, no network
calls (Ch03: CPU-cheap, explainable rule matching).
"""
path_lower = entry.path.lower()
ua_lower = (entry.user_agent or "").lower()
for scanner_ua in _PATTERNS["scanner_user_agents"]:
if scanner_ua in ua_lower:
return ThreatMatch(rule_matched=f"scanner_ua:{scanner_ua}", severity="high")
for sensitive in _PATTERNS["sensitive_paths"]:
if sensitive.lower() in path_lower:
return ThreatMatch(rule_matched=f"sensitive_path:{sensitive}", severity="medium")
for marker in _PATTERNS["injection_markers"]:
if marker.lower() in path_lower:
return ThreatMatch(rule_matched=f"injection:{marker}", severity="high")
return None
+54
View File
@@ -0,0 +1,54 @@
"""Lightweight browser/OS classification for human traffic (Chapter 08).
Separate from Chapter 07's bot_identifier — that identifies automated
crawlers via UA substring + DNS verification. This classifies ordinary
browsers/OSes, same ordered-substring-match pattern, signatures as JSON
data (not hardcoded), no third-party UA-parsing dependency — consistent
with the low-footprint bias in Chapter 03.
Order matters within each signature list: e.g. Edge/Opera must be checked
before Chrome (their UAs also contain "Chrome/"); iOS before macOS (iPhone
UAs also contain "like Mac OS X"); Android before Linux (Android UAs also
contain "Linux").
"""
from __future__ import annotations
import json
from dataclasses import dataclass
from pathlib import Path
_DATA_DIR = Path(__file__).parent / "data"
@dataclass(frozen=True)
class UASignature:
name: str
ua_substrings: tuple[str, ...]
def _load(filename: str) -> list[UASignature]:
raw = json.loads((_DATA_DIR / filename).read_text())
return [UASignature(name=e["name"], ua_substrings=tuple(e["ua_substrings"])) for e in raw]
_BROWSER_SIGNATURES = _load("browser_signatures.json")
_OS_SIGNATURES = _load("os_signatures.json")
def _match(user_agent: str, signatures: list[UASignature], default: str) -> str:
for sig in signatures:
if any(sub in user_agent for sub in sig.ua_substrings):
return sig.name
return default
def classify_browser(user_agent: str | None) -> str:
if not user_agent:
return "Unknown"
return _match(user_agent, _BROWSER_SIGNATURES, "Other")
def classify_os(user_agent: str | None) -> str:
if not user_agent:
return "Unknown"
return _match(user_agent, _OS_SIGNATURES, "Other")