"""Upload endpoint (Chapter 04): validated, streamed-to-disk save. Parsing happens in a background thread triggered right after this request completes (app/services/background.py) — never synchronously inside this request (Chapter 03, rule 5 still holds: the response returns immediately regardless of file size). """ from __future__ import annotations import hashlib import uuid from pathlib import Path from flask import current_app, jsonify, render_template, request from werkzeug.utils import secure_filename from app.blueprints.uploads import bp from app.blueprints.uploads.queries import get_uploaded_files from app.extensions import db, limiter from app.models.log_file import LogFile from app.services.background import trigger_processing from app.services.file_deletion import delete_log_files from app.utils.pagination import parse_pagination from app.utils.upload_paths import upload_path_for _ALLOWED_EXTENSIONS = {".log", ".txt", ".gz"} _CHUNK_SIZE = 64 * 1024 # 64 KB per read — never buffer the whole upload _GZIP_MAGIC = b"\x1f\x8b" class UploadRejected(Exception): """Raised when an upload fails extension/content validation.""" def _validate_extension(filename: str) -> str: ext = Path(filename).suffix.lower() if ext not in _ALLOWED_EXTENSIONS: raise UploadRejected(f"Unsupported extension {ext!r}; allowed: {_ALLOWED_EXTENSIONS}") return ext def _sniff_content(first_chunk: bytes, ext: str) -> None: """Light content sniff — don't just trust the client-supplied MIME type.""" if ext == ".gz": if not first_chunk.startswith(_GZIP_MAGIC): raise UploadRejected("File has a .gz extension but isn't gzip-magic-prefixed.") return if b"\x00" in first_chunk: raise UploadRejected("File extension claims text but content looks binary.") def _stream_to_temp(file_storage, tmp_path: Path) -> tuple[int, str, bytes]: """Stream the upload to disk in bounded chunks; return (size, sha256_hex, first_chunk). Never calls file.read() on the whole stream (Chapter 03, rule 1). """ sha256 = hashlib.sha256() size = 0 first_chunk: bytes | None = None with tmp_path.open("wb") as out: while True: chunk = file_storage.stream.read(_CHUNK_SIZE) if not chunk: break if first_chunk is None: first_chunk = chunk sha256.update(chunk) size += len(chunk) out.write(chunk) if first_chunk is None: raise UploadRejected("Uploaded file is empty.") return size, sha256.hexdigest(), first_chunk @bp.post("/uploads") @limiter.limit("20 per minute") # Chapter 12: rate limiting on /uploads at minimum def upload_log_file(): """Validate, stream, and register an uploaded access log (Ch04/Ch11).""" file_storage = request.files.get("logfile") if file_storage is None or not file_storage.filename: return render_template("uploads/_error.html", message="No file provided."), 400 upload_dir = Path(current_app.config["UPLOAD_DIR"]) (upload_dir / "tmp").mkdir(parents=True, exist_ok=True) tmp_path = upload_dir / "tmp" / f"{uuid.uuid4().hex}.part" try: ext = _validate_extension(file_storage.filename) size_bytes, checksum, first_chunk = _stream_to_temp(file_storage, tmp_path) _sniff_content(first_chunk, ext) max_bytes = current_app.config["UPLOAD_MAX_SIZE_MB"] * 1024 * 1024 if size_bytes > max_bytes: raise UploadRejected(f"File exceeds {current_app.config['UPLOAD_MAX_SIZE_MB']}MB limit.") except UploadRejected as exc: tmp_path.unlink(missing_ok=True) return render_template("uploads/_error.html", message=str(exc)), 400 existing = LogFile.query.filter_by(checksum=checksum).first() if existing is not None: tmp_path.unlink(missing_ok=True) return render_template("uploads/_duplicate.html", log_file=existing) log_file = LogFile( filename=secure_filename(file_storage.filename), server_type=request.form.get("server_type", "apache"), format_string=request.form.get("format_string", ""), status="queued", size_bytes=size_bytes, checksum=checksum, ) db.session.add(log_file) db.session.commit() # need the assigned id before the final rename tmp_path.rename(upload_path_for(log_file)) # Chapter 12 simplification: no cron required — kick off processing # immediately in a background thread. The response below returns as # soon as the file is queued (Ch03 rule 5 still holds: this request # never blocks on parsing), while the thread runs independently. trigger_processing(current_app._get_current_object(), log_file.id) return render_template("uploads/_queued.html", log_file=log_file) @bp.get("/api/uploads//status") def upload_status(log_file_id: int): """Polled every 3s by the browser (hx-trigger) until done/error (Ch04).""" log_file = db.get_or_404(LogFile, log_file_id) if request.headers.get("Accept") == "application/json": return jsonify( data={ "id": log_file.id, "status": log_file.status, "processed_lines": log_file.processed_lines, "total_lines": log_file.total_lines, }, meta={}, ) template = { "done": "uploads/_status_done.html", "error": "uploads/_status_error.html", # BUG FIX: this key was missing, so "queued" fell through to the # "processing" fallback below — a file that hadn't been picked up # by `flask process-logs` yet displayed as "Processing X: 0 lines" # instead of "Queued — waiting for the next parse cycle", making a # cron job that simply hasn't run yet indistinguishable from one # that's actually hung mid-parse. "queued": "uploads/_queued.html", }.get(log_file.status, "uploads/_status_processing.html") return render_template(template, log_file=log_file) @bp.get("/api/uploads") def list_uploads(): """Uploaded-files list (project-owner follow-up request) — Grid.js- backed, same page/per_page convention as every other table (Ch11). Sorted by upload recency, independent of the dashboard date-range picker (a single file can span many log dates). """ page, per_page = parse_pagination(request) rows, total = get_uploaded_files(page, per_page) return jsonify( data={"rows": rows, "total": total}, meta={"page": page, "per_page": per_page}, ) @bp.delete("/api/uploads") def bulk_delete_uploads(): """Delete one or more uploaded files and everything derived from them (log_entries/bot_hits/suspicious_events, the raw file on disk, and a correct rollup recompute for the affected dates — see app/services/file_deletion.py for why a rollup recompute is needed rather than a simple per-file delete). Body: {"ids": [1, 2, 3]}. Files currently "processing" are skipped, not force-deleted, to avoid racing the background parse thread. """ body = request.get_json(silent=True) or {} ids = body.get("ids") if not isinstance(ids, list) or not ids or not all(isinstance(i, int) for i in ids): return jsonify(error={"code": "invalid_request", "message": "Expected {\"ids\": [int, ...]}."}), 400 result = delete_log_files(ids) return jsonify(data={"deleted": result.deleted, "skipped": result.skipped}, meta={})