94 lines
4.2 KiB
Python
94 lines
4.2 KiB
Python
"""Uploaded-file deletion (project-owner follow-up request).
|
|
|
|
Deleting a LogFile is more than one DELETE statement: log_entries,
|
|
bot_hits, and suspicious_events all reference log_file_id and must be
|
|
removed explicitly — this project's SQLite connections don't have
|
|
`PRAGMA foreign_keys=ON`, so the `ondelete="CASCADE"` in the migrations
|
|
(Ch06) is declarative documentation only, not an enforced behavior.
|
|
Relying on it would silently leave orphaned rows behind.
|
|
|
|
The harder part is the rollup tables (request_stats_hourly/daily,
|
|
referrer/browser/human-path/ip-path/ip-status): none of them have a
|
|
log_file_id column (Ch06: they're site-wide, since two files can share a
|
|
date). So instead of deleting rollup rows for the deleted file's dates
|
|
directly (which could wipe out another file's contribution to the same
|
|
date), this captures the affected dates BEFORE deleting, then calls
|
|
aggregator.compute_rollups_for_range() AFTER deleting — which recomputes
|
|
each affected day from whatever log_entries remain, correctly handling
|
|
both "another file still covers this day" and "no file covers this day
|
|
anymore" (the latter now handled correctly by the aggregator.py fix that
|
|
shipped alongside this feature).
|
|
|
|
KNOWN LIMITATION (flagged, not fixed): ip_registry (total_requests,
|
|
first_seen, last_seen, verification cache) and blocklist_suggestions are
|
|
NOT per-file and are NOT recomputed on deletion — doing so correctly
|
|
would mean re-scanning all remaining log_entries for every affected IP,
|
|
which is unbounded work for a single delete action and would violate the
|
|
same "no raw-row scan on a hot path" reasoning Ch03 applies elsewhere.
|
|
After deleting a file, IP History numbers may include a deleted file's
|
|
historical contribution until a full site-wide recompute is added as a
|
|
separate feature. Doesn't affect correctness of Overview/SEO's date-
|
|
scoped numbers, which is what this feature was actually asked to fix.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
from dataclasses import dataclass, field
|
|
|
|
from app.extensions import db
|
|
from app.models.bot_hit import BotHit
|
|
from app.models.log_entry import LogEntry
|
|
from app.models.log_file import LogFile
|
|
from app.models.suspicious_event import SuspiciousEvent
|
|
from app.services import aggregator
|
|
from app.utils.upload_paths import upload_path_for
|
|
|
|
|
|
@dataclass
|
|
class BulkDeleteResult:
|
|
deleted: list[int] = field(default_factory=list)
|
|
skipped: list[dict] = field(default_factory=list) # [{"id": ..., "reason": ...}]
|
|
|
|
|
|
def delete_log_files(log_file_ids: list[int]) -> BulkDeleteResult:
|
|
"""Delete each id in `log_file_ids`; recompute rollups once at the end
|
|
for the full span of dates any deleted file touched (cheaper than
|
|
recomputing per-file when multiple files share dates).
|
|
"""
|
|
result = BulkDeleteResult()
|
|
all_touched_dates: set = set()
|
|
|
|
for log_file_id in log_file_ids:
|
|
log_file = db.session.get(LogFile, log_file_id)
|
|
if log_file is None:
|
|
result.skipped.append({"id": log_file_id, "reason": "not found"})
|
|
continue
|
|
if log_file.status == "processing":
|
|
result.skipped.append({"id": log_file_id, "reason": "currently being analyzed"})
|
|
continue
|
|
|
|
touched_dates = {
|
|
row.date() for row, in db.session.query(LogEntry.timestamp)
|
|
.filter(LogEntry.log_file_id == log_file_id).distinct()
|
|
}
|
|
# distinct() on a full timestamp rarely collapses much; reduce to
|
|
# calendar dates in Python since SQLite's DATE() in a DISTINCT
|
|
# clause is a bit more awkward to express portably here.
|
|
all_touched_dates |= touched_dates
|
|
|
|
db.session.query(SuspiciousEvent).filter(SuspiciousEvent.log_file_id == log_file_id).delete()
|
|
db.session.query(BotHit).filter(BotHit.log_file_id == log_file_id).delete()
|
|
db.session.query(LogEntry).filter(LogEntry.log_file_id == log_file_id).delete()
|
|
|
|
raw_path = upload_path_for(log_file)
|
|
if raw_path.exists():
|
|
raw_path.unlink()
|
|
|
|
db.session.delete(log_file)
|
|
db.session.commit()
|
|
result.deleted.append(log_file_id)
|
|
|
|
if all_touched_dates:
|
|
aggregator.compute_rollups_for_range(min(all_touched_dates), max(all_touched_dates))
|
|
|
|
return result
|