164 lines
6.8 KiB
Python
164 lines
6.8 KiB
Python
"""Flask CLI admin commands (factor 12).
|
|
|
|
process-logs is now OPTIONAL (Chapter 12 simplification, per project
|
|
owner request): file processing is triggered automatically in-app right
|
|
after upload (app/services/background.py), so cron is no longer required.
|
|
This command still exists for anyone who'd rather run it manually or via
|
|
cron — it shares the exact same processing code (app/services/
|
|
log_processor.py) as the automatic background trigger, so both paths
|
|
behave identically. cleanup (Chapter 06/12) enforces the retention
|
|
policy. create-admin (Chapter 12) bootstraps the single admin account.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
from datetime import date, datetime, timedelta
|
|
|
|
import click
|
|
from flask import Flask, current_app
|
|
|
|
from app.extensions import db
|
|
from app.models.ip_traffic_stats import IpPathStatsDaily, IpStatusStatsDaily
|
|
from app.models.log_entry import LogEntry
|
|
from app.models.log_file import LogFile
|
|
from app.models.request_stats import RequestStatsHourly
|
|
from app.models.user import User
|
|
from app.services import aggregator
|
|
from app.services.log_processor import process_one_batch
|
|
from app.utils.upload_paths import upload_path_for
|
|
|
|
# Chapter 06 retention policy — the constants `flask cleanup` enforces.
|
|
LOG_ENTRIES_RETENTION_DAYS = 30
|
|
HOURLY_STATS_RETENTION_DAYS = 90
|
|
IP_TRAFFIC_STATS_RETENTION_DAYS = 30
|
|
|
|
|
|
def register_commands(app: Flask) -> None:
|
|
app.cli.add_command(process_logs)
|
|
app.cli.add_command(rollup)
|
|
app.cli.add_command(cleanup)
|
|
app.cli.add_command(create_admin)
|
|
|
|
|
|
@click.command("process-logs")
|
|
@click.option("--batch-size", default=None, type=int, help="Override PARSE_BATCH_SIZE.")
|
|
def process_logs(batch_size: int | None) -> None:
|
|
"""Optional manual/cron fallback — processing now also runs
|
|
automatically in-app after upload. Parses queued/processing
|
|
log_files in bounded, checkpointed batches; one batch per file per
|
|
invocation, same as before.
|
|
"""
|
|
batch = batch_size or current_app.config["PARSE_BATCH_SIZE"]
|
|
pending = LogFile.query.filter(LogFile.status.in_(["queued", "processing"])).all()
|
|
for log_file in pending:
|
|
process_one_batch(log_file, batch)
|
|
|
|
|
|
@click.command("rollup")
|
|
@click.option("--from", "from_date", required=True, help="ISO date, e.g. 2026-07-01")
|
|
@click.option("--to", "to_date", required=True, help="ISO date, e.g. 2026-07-26")
|
|
def rollup(from_date: str, to_date: str) -> None:
|
|
"""Manual rollup recompute for an explicit range (e.g. after a backfill).
|
|
|
|
Requires an explicit range — no "all time" default, mirroring Ch03 rule 7
|
|
even for an admin command, to avoid an unbounded scan on constrained hosting.
|
|
"""
|
|
start = date.fromisoformat(from_date)
|
|
end = date.fromisoformat(to_date)
|
|
aggregator.compute_rollups_for_range(start, end)
|
|
click.echo(f"Rolled up {start} .. {end}")
|
|
|
|
|
|
@click.command("cleanup")
|
|
def cleanup() -> None:
|
|
"""Enforce the Chapter 06 retention policy (Chapter 12).
|
|
|
|
Previously a stub through every earlier chapter — implemented here as
|
|
Chapter 12's non-functional/deployment concern. Cron-invoked (e.g.
|
|
daily, off-peak), never a long-running daemon.
|
|
"""
|
|
now = datetime.utcnow()
|
|
deleted_entries = _delete_old_log_entries(now)
|
|
deleted_hourly = _collapse_old_hourly_stats(now)
|
|
deleted_ip_stats = _delete_old_ip_traffic_stats(now)
|
|
archived_files = _delete_parsed_upload_files()
|
|
click.echo(
|
|
f"Cleanup complete: {deleted_entries} log_entries, {deleted_hourly} "
|
|
f"request_stats_hourly, {deleted_ip_stats} ip traffic-rollup rows "
|
|
f"deleted; {archived_files} parsed upload file(s) removed from disk."
|
|
)
|
|
|
|
|
|
def _delete_old_log_entries(now: datetime) -> int:
|
|
"""Ch06: raw log_entries retained ~30 days, then deleted."""
|
|
cutoff = now - timedelta(days=LOG_ENTRIES_RETENTION_DAYS)
|
|
count = db.session.query(LogEntry).filter(LogEntry.timestamp < cutoff).delete(synchronize_session=False)
|
|
db.session.commit()
|
|
return count
|
|
|
|
|
|
def _collapse_old_hourly_stats(now: datetime) -> int:
|
|
"""Ch06: request_stats_hourly retained ~90 days, then collapsed into
|
|
request_stats_daily only. request_stats_daily is already computed
|
|
independently by the aggregator straight from raw log_entries, so
|
|
"collapsing" here just means deleting the now-redundant hourly rows
|
|
once the retention window passes — no data is lost, since the daily
|
|
rollup for that period was already written when the file was parsed.
|
|
"""
|
|
cutoff = now - timedelta(days=HOURLY_STATS_RETENTION_DAYS)
|
|
count = db.session.query(RequestStatsHourly).filter(
|
|
RequestStatsHourly.date_hour < cutoff
|
|
).delete(synchronize_session=False)
|
|
db.session.commit()
|
|
return count
|
|
|
|
|
|
def _delete_old_ip_traffic_stats(now: datetime) -> int:
|
|
"""The bounded per-IP rollup (added as a Ch10 follow-up) was designed
|
|
for ~30-day retention, matching log_entries — see aggregator.py.
|
|
"""
|
|
cutoff_date = (now - timedelta(days=IP_TRAFFIC_STATS_RETENTION_DAYS)).date()
|
|
count = db.session.query(IpPathStatsDaily).filter(IpPathStatsDaily.date < cutoff_date).delete(synchronize_session=False)
|
|
count += db.session.query(IpStatusStatsDaily).filter(IpStatusStatsDaily.date < cutoff_date).delete(synchronize_session=False)
|
|
db.session.commit()
|
|
return count
|
|
|
|
|
|
def _delete_parsed_upload_files() -> int:
|
|
"""Ch06: 'compress or delete after successful parse + rollup, rather
|
|
than keeping both the raw file and a full raw-row copy.' Deletes
|
|
(rather than compresses) — simpler, and avoids spending extra CPU/IOPS
|
|
gzip-ing data that's already been fully parsed into the database.
|
|
"""
|
|
done_files = LogFile.query.filter_by(status="done").all()
|
|
removed = 0
|
|
for log_file in done_files:
|
|
path = upload_path_for(log_file)
|
|
if path.exists():
|
|
path.unlink()
|
|
removed += 1
|
|
return removed
|
|
|
|
|
|
@click.command("create-admin")
|
|
@click.option("--email", default=None, help="Defaults to the ADMIN_EMAIL env var.")
|
|
@click.password_option()
|
|
def create_admin(email: str | None, password: str) -> None:
|
|
"""Bootstrap the single admin account (Chapter 12).
|
|
|
|
Interactive password prompt (via --password-option's confirmation
|
|
prompt) keeps the raw credential out of process env/config, unlike an
|
|
ADMIN_PASSWORD env var would — Chapter 12 doesn't specify a bootstrap
|
|
mechanism beyond documenting ADMIN_EMAIL, so this is a flagged addition.
|
|
"""
|
|
resolved_email = (email or current_app.config.get("ADMIN_EMAIL") or "").strip().lower()
|
|
if not resolved_email:
|
|
raise click.UsageError("No --email given and ADMIN_EMAIL is not set.")
|
|
if User.query.filter_by(email=resolved_email).first():
|
|
raise click.UsageError(f"User {resolved_email} already exists.")
|
|
|
|
user = User(email=resolved_email)
|
|
user.set_password(password)
|
|
db.session.add(user)
|
|
db.session.commit()
|
|
click.echo(f"Created admin user {resolved_email}")
|