Files
Kavosh/app/services/file_deletion.py
2026-08-07 21:17:17 +03:30

94 lines
4.2 KiB
Python

"""Uploaded-file deletion (project-owner follow-up request).
Deleting a LogFile is more than one DELETE statement: log_entries,
bot_hits, and suspicious_events all reference log_file_id and must be
removed explicitly — this project's SQLite connections don't have
`PRAGMA foreign_keys=ON`, so the `ondelete="CASCADE"` in the migrations
(Ch06) is declarative documentation only, not an enforced behavior.
Relying on it would silently leave orphaned rows behind.
The harder part is the rollup tables (request_stats_hourly/daily,
referrer/browser/human-path/ip-path/ip-status): none of them have a
log_file_id column (Ch06: they're site-wide, since two files can share a
date). So instead of deleting rollup rows for the deleted file's dates
directly (which could wipe out another file's contribution to the same
date), this captures the affected dates BEFORE deleting, then calls
aggregator.compute_rollups_for_range() AFTER deleting — which recomputes
each affected day from whatever log_entries remain, correctly handling
both "another file still covers this day" and "no file covers this day
anymore" (the latter now handled correctly by the aggregator.py fix that
shipped alongside this feature).
KNOWN LIMITATION (flagged, not fixed): ip_registry (total_requests,
first_seen, last_seen, verification cache) and blocklist_suggestions are
NOT per-file and are NOT recomputed on deletion — doing so correctly
would mean re-scanning all remaining log_entries for every affected IP,
which is unbounded work for a single delete action and would violate the
same "no raw-row scan on a hot path" reasoning Ch03 applies elsewhere.
After deleting a file, IP History numbers may include a deleted file's
historical contribution until a full site-wide recompute is added as a
separate feature. Doesn't affect correctness of Overview/SEO's date-
scoped numbers, which is what this feature was actually asked to fix.
"""
from __future__ import annotations
from dataclasses import dataclass, field
from app.extensions import db
from app.models.bot_hit import BotHit
from app.models.log_entry import LogEntry
from app.models.log_file import LogFile
from app.models.suspicious_event import SuspiciousEvent
from app.services import aggregator
from app.utils.upload_paths import upload_path_for
@dataclass
class BulkDeleteResult:
deleted: list[int] = field(default_factory=list)
skipped: list[dict] = field(default_factory=list) # [{"id": ..., "reason": ...}]
def delete_log_files(log_file_ids: list[int]) -> BulkDeleteResult:
"""Delete each id in `log_file_ids`; recompute rollups once at the end
for the full span of dates any deleted file touched (cheaper than
recomputing per-file when multiple files share dates).
"""
result = BulkDeleteResult()
all_touched_dates: set = set()
for log_file_id in log_file_ids:
log_file = db.session.get(LogFile, log_file_id)
if log_file is None:
result.skipped.append({"id": log_file_id, "reason": "not found"})
continue
if log_file.status == "processing":
result.skipped.append({"id": log_file_id, "reason": "currently being analyzed"})
continue
touched_dates = {
row.date() for row, in db.session.query(LogEntry.timestamp)
.filter(LogEntry.log_file_id == log_file_id).distinct()
}
# distinct() on a full timestamp rarely collapses much; reduce to
# calendar dates in Python since SQLite's DATE() in a DISTINCT
# clause is a bit more awkward to express portably here.
all_touched_dates |= touched_dates
db.session.query(SuspiciousEvent).filter(SuspiciousEvent.log_file_id == log_file_id).delete()
db.session.query(BotHit).filter(BotHit.log_file_id == log_file_id).delete()
db.session.query(LogEntry).filter(LogEntry.log_file_id == log_file_id).delete()
raw_path = upload_path_for(log_file)
if raw_path.exists():
raw_path.unlink()
db.session.delete(log_file)
db.session.commit()
result.deleted.append(log_file_id)
if all_touched_dates:
aggregator.compute_rollups_for_range(min(all_touched_dates), max(all_touched_dates))
return result