189 lines
7.4 KiB
Python
189 lines
7.4 KiB
Python
"""Upload endpoint (Chapter 04): validated, streamed-to-disk save.
|
|
|
|
Parsing happens in a background thread triggered right after this request
|
|
completes (app/services/background.py) — never synchronously inside this
|
|
request (Chapter 03, rule 5 still holds: the response returns immediately
|
|
regardless of file size).
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import hashlib
|
|
import uuid
|
|
from pathlib import Path
|
|
|
|
from flask import current_app, jsonify, render_template, request
|
|
from werkzeug.utils import secure_filename
|
|
|
|
from app.blueprints.uploads import bp
|
|
from app.blueprints.uploads.queries import get_uploaded_files
|
|
from app.extensions import db, limiter
|
|
from app.models.log_file import LogFile
|
|
from app.services.background import trigger_processing
|
|
from app.services.file_deletion import delete_log_files
|
|
from app.utils.pagination import parse_pagination
|
|
from app.utils.upload_paths import upload_path_for
|
|
|
|
_ALLOWED_EXTENSIONS = {".log", ".txt", ".gz"}
|
|
_CHUNK_SIZE = 64 * 1024 # 64 KB per read — never buffer the whole upload
|
|
_GZIP_MAGIC = b"\x1f\x8b"
|
|
|
|
|
|
class UploadRejected(Exception):
|
|
"""Raised when an upload fails extension/content validation."""
|
|
|
|
|
|
def _validate_extension(filename: str) -> str:
|
|
ext = Path(filename).suffix.lower()
|
|
if ext not in _ALLOWED_EXTENSIONS:
|
|
raise UploadRejected(f"Unsupported extension {ext!r}; allowed: {_ALLOWED_EXTENSIONS}")
|
|
return ext
|
|
|
|
|
|
def _sniff_content(first_chunk: bytes, ext: str) -> None:
|
|
"""Light content sniff — don't just trust the client-supplied MIME type."""
|
|
if ext == ".gz":
|
|
if not first_chunk.startswith(_GZIP_MAGIC):
|
|
raise UploadRejected("File has a .gz extension but isn't gzip-magic-prefixed.")
|
|
return
|
|
if b"\x00" in first_chunk:
|
|
raise UploadRejected("File extension claims text but content looks binary.")
|
|
|
|
|
|
def _stream_to_temp(file_storage, tmp_path: Path) -> tuple[int, str, bytes]:
|
|
"""Stream the upload to disk in bounded chunks; return (size, sha256_hex, first_chunk).
|
|
|
|
Never calls file.read() on the whole stream (Chapter 03, rule 1).
|
|
"""
|
|
sha256 = hashlib.sha256()
|
|
size = 0
|
|
first_chunk: bytes | None = None
|
|
with tmp_path.open("wb") as out:
|
|
while True:
|
|
chunk = file_storage.stream.read(_CHUNK_SIZE)
|
|
if not chunk:
|
|
break
|
|
if first_chunk is None:
|
|
first_chunk = chunk
|
|
sha256.update(chunk)
|
|
size += len(chunk)
|
|
out.write(chunk)
|
|
if first_chunk is None:
|
|
raise UploadRejected("Uploaded file is empty.")
|
|
return size, sha256.hexdigest(), first_chunk
|
|
|
|
|
|
@bp.post("/uploads")
|
|
@limiter.limit("20 per minute") # Chapter 12: rate limiting on /uploads at minimum
|
|
def upload_log_file():
|
|
"""Validate, stream, and register an uploaded access log (Ch04/Ch11)."""
|
|
file_storage = request.files.get("logfile")
|
|
if file_storage is None or not file_storage.filename:
|
|
return render_template("uploads/_error.html", message="No file provided."), 400
|
|
|
|
upload_dir = Path(current_app.config["UPLOAD_DIR"])
|
|
(upload_dir / "tmp").mkdir(parents=True, exist_ok=True)
|
|
tmp_path = upload_dir / "tmp" / f"{uuid.uuid4().hex}.part"
|
|
|
|
try:
|
|
ext = _validate_extension(file_storage.filename)
|
|
size_bytes, checksum, first_chunk = _stream_to_temp(file_storage, tmp_path)
|
|
_sniff_content(first_chunk, ext)
|
|
|
|
max_bytes = current_app.config["UPLOAD_MAX_SIZE_MB"] * 1024 * 1024
|
|
if size_bytes > max_bytes:
|
|
raise UploadRejected(f"File exceeds {current_app.config['UPLOAD_MAX_SIZE_MB']}MB limit.")
|
|
except UploadRejected as exc:
|
|
tmp_path.unlink(missing_ok=True)
|
|
return render_template("uploads/_error.html", message=str(exc)), 400
|
|
|
|
existing = LogFile.query.filter_by(checksum=checksum).first()
|
|
if existing is not None:
|
|
tmp_path.unlink(missing_ok=True)
|
|
return render_template("uploads/_duplicate.html", log_file=existing)
|
|
|
|
log_file = LogFile(
|
|
filename=secure_filename(file_storage.filename),
|
|
server_type=request.form.get("server_type", "apache"),
|
|
format_string=request.form.get("format_string", ""),
|
|
status="queued",
|
|
size_bytes=size_bytes,
|
|
checksum=checksum,
|
|
)
|
|
db.session.add(log_file)
|
|
db.session.commit() # need the assigned id before the final rename
|
|
|
|
tmp_path.rename(upload_path_for(log_file))
|
|
|
|
# Chapter 12 simplification: no cron required — kick off processing
|
|
# immediately in a background thread. The response below returns as
|
|
# soon as the file is queued (Ch03 rule 5 still holds: this request
|
|
# never blocks on parsing), while the thread runs independently.
|
|
trigger_processing(current_app._get_current_object(), log_file.id)
|
|
|
|
return render_template("uploads/_queued.html", log_file=log_file)
|
|
|
|
|
|
@bp.get("/api/uploads/<int:log_file_id>/status")
|
|
def upload_status(log_file_id: int):
|
|
"""Polled every 3s by the browser (hx-trigger) until done/error (Ch04)."""
|
|
log_file = db.get_or_404(LogFile, log_file_id)
|
|
|
|
if request.headers.get("Accept") == "application/json":
|
|
return jsonify(
|
|
data={
|
|
"id": log_file.id,
|
|
"status": log_file.status,
|
|
"processed_lines": log_file.processed_lines,
|
|
"total_lines": log_file.total_lines,
|
|
},
|
|
meta={},
|
|
)
|
|
|
|
template = {
|
|
"done": "uploads/_status_done.html",
|
|
"error": "uploads/_status_error.html",
|
|
# BUG FIX: this key was missing, so "queued" fell through to the
|
|
# "processing" fallback below — a file that hadn't been picked up
|
|
# by `flask process-logs` yet displayed as "Processing X: 0 lines"
|
|
# instead of "Queued — waiting for the next parse cycle", making a
|
|
# cron job that simply hasn't run yet indistinguishable from one
|
|
# that's actually hung mid-parse.
|
|
"queued": "uploads/_queued.html",
|
|
}.get(log_file.status, "uploads/_status_processing.html")
|
|
return render_template(template, log_file=log_file)
|
|
|
|
|
|
@bp.get("/api/uploads")
|
|
def list_uploads():
|
|
"""Uploaded-files list (project-owner follow-up request) — Grid.js-
|
|
backed, same page/per_page convention as every other table (Ch11).
|
|
Sorted by upload recency, independent of the dashboard date-range
|
|
picker (a single file can span many log dates).
|
|
"""
|
|
page, per_page = parse_pagination(request)
|
|
rows, total = get_uploaded_files(page, per_page)
|
|
return jsonify(
|
|
data={"rows": rows, "total": total},
|
|
meta={"page": page, "per_page": per_page},
|
|
)
|
|
|
|
|
|
@bp.delete("/api/uploads")
|
|
def bulk_delete_uploads():
|
|
"""Delete one or more uploaded files and everything derived from them
|
|
(log_entries/bot_hits/suspicious_events, the raw file on disk, and a
|
|
correct rollup recompute for the affected dates — see
|
|
app/services/file_deletion.py for why a rollup recompute is needed
|
|
rather than a simple per-file delete).
|
|
|
|
Body: {"ids": [1, 2, 3]}. Files currently "processing" are skipped,
|
|
not force-deleted, to avoid racing the background parse thread.
|
|
"""
|
|
body = request.get_json(silent=True) or {}
|
|
ids = body.get("ids")
|
|
if not isinstance(ids, list) or not ids or not all(isinstance(i, int) for i in ids):
|
|
return jsonify(error={"code": "invalid_request", "message": "Expected {\"ids\": [int, ...]}."}), 400
|
|
|
|
result = delete_log_files(ids)
|
|
return jsonify(data={"deleted": result.deleted, "skipped": result.skipped}, meta={})
|