start project

This commit is contained in:
Hemmat
2026-08-07 21:17:17 +03:30
commit ea1e1eead6
121 changed files with 8108 additions and 0 deletions
+188
View File
@@ -0,0 +1,188 @@
"""Upload endpoint (Chapter 04): validated, streamed-to-disk save.
Parsing happens in a background thread triggered right after this request
completes (app/services/background.py) — never synchronously inside this
request (Chapter 03, rule 5 still holds: the response returns immediately
regardless of file size).
"""
from __future__ import annotations
import hashlib
import uuid
from pathlib import Path
from flask import current_app, jsonify, render_template, request
from werkzeug.utils import secure_filename
from app.blueprints.uploads import bp
from app.blueprints.uploads.queries import get_uploaded_files
from app.extensions import db, limiter
from app.models.log_file import LogFile
from app.services.background import trigger_processing
from app.services.file_deletion import delete_log_files
from app.utils.pagination import parse_pagination
from app.utils.upload_paths import upload_path_for
_ALLOWED_EXTENSIONS = {".log", ".txt", ".gz"}
_CHUNK_SIZE = 64 * 1024 # 64 KB per read — never buffer the whole upload
_GZIP_MAGIC = b"\x1f\x8b"
class UploadRejected(Exception):
"""Raised when an upload fails extension/content validation."""
def _validate_extension(filename: str) -> str:
ext = Path(filename).suffix.lower()
if ext not in _ALLOWED_EXTENSIONS:
raise UploadRejected(f"Unsupported extension {ext!r}; allowed: {_ALLOWED_EXTENSIONS}")
return ext
def _sniff_content(first_chunk: bytes, ext: str) -> None:
"""Light content sniff — don't just trust the client-supplied MIME type."""
if ext == ".gz":
if not first_chunk.startswith(_GZIP_MAGIC):
raise UploadRejected("File has a .gz extension but isn't gzip-magic-prefixed.")
return
if b"\x00" in first_chunk:
raise UploadRejected("File extension claims text but content looks binary.")
def _stream_to_temp(file_storage, tmp_path: Path) -> tuple[int, str, bytes]:
"""Stream the upload to disk in bounded chunks; return (size, sha256_hex, first_chunk).
Never calls file.read() on the whole stream (Chapter 03, rule 1).
"""
sha256 = hashlib.sha256()
size = 0
first_chunk: bytes | None = None
with tmp_path.open("wb") as out:
while True:
chunk = file_storage.stream.read(_CHUNK_SIZE)
if not chunk:
break
if first_chunk is None:
first_chunk = chunk
sha256.update(chunk)
size += len(chunk)
out.write(chunk)
if first_chunk is None:
raise UploadRejected("Uploaded file is empty.")
return size, sha256.hexdigest(), first_chunk
@bp.post("/uploads")
@limiter.limit("20 per minute") # Chapter 12: rate limiting on /uploads at minimum
def upload_log_file():
"""Validate, stream, and register an uploaded access log (Ch04/Ch11)."""
file_storage = request.files.get("logfile")
if file_storage is None or not file_storage.filename:
return render_template("uploads/_error.html", message="No file provided."), 400
upload_dir = Path(current_app.config["UPLOAD_DIR"])
(upload_dir / "tmp").mkdir(parents=True, exist_ok=True)
tmp_path = upload_dir / "tmp" / f"{uuid.uuid4().hex}.part"
try:
ext = _validate_extension(file_storage.filename)
size_bytes, checksum, first_chunk = _stream_to_temp(file_storage, tmp_path)
_sniff_content(first_chunk, ext)
max_bytes = current_app.config["UPLOAD_MAX_SIZE_MB"] * 1024 * 1024
if size_bytes > max_bytes:
raise UploadRejected(f"File exceeds {current_app.config['UPLOAD_MAX_SIZE_MB']}MB limit.")
except UploadRejected as exc:
tmp_path.unlink(missing_ok=True)
return render_template("uploads/_error.html", message=str(exc)), 400
existing = LogFile.query.filter_by(checksum=checksum).first()
if existing is not None:
tmp_path.unlink(missing_ok=True)
return render_template("uploads/_duplicate.html", log_file=existing)
log_file = LogFile(
filename=secure_filename(file_storage.filename),
server_type=request.form.get("server_type", "apache"),
format_string=request.form.get("format_string", ""),
status="queued",
size_bytes=size_bytes,
checksum=checksum,
)
db.session.add(log_file)
db.session.commit() # need the assigned id before the final rename
tmp_path.rename(upload_path_for(log_file))
# Chapter 12 simplification: no cron required — kick off processing
# immediately in a background thread. The response below returns as
# soon as the file is queued (Ch03 rule 5 still holds: this request
# never blocks on parsing), while the thread runs independently.
trigger_processing(current_app._get_current_object(), log_file.id)
return render_template("uploads/_queued.html", log_file=log_file)
@bp.get("/api/uploads/<int:log_file_id>/status")
def upload_status(log_file_id: int):
"""Polled every 3s by the browser (hx-trigger) until done/error (Ch04)."""
log_file = db.get_or_404(LogFile, log_file_id)
if request.headers.get("Accept") == "application/json":
return jsonify(
data={
"id": log_file.id,
"status": log_file.status,
"processed_lines": log_file.processed_lines,
"total_lines": log_file.total_lines,
},
meta={},
)
template = {
"done": "uploads/_status_done.html",
"error": "uploads/_status_error.html",
# BUG FIX: this key was missing, so "queued" fell through to the
# "processing" fallback below — a file that hadn't been picked up
# by `flask process-logs` yet displayed as "Processing X: 0 lines"
# instead of "Queued — waiting for the next parse cycle", making a
# cron job that simply hasn't run yet indistinguishable from one
# that's actually hung mid-parse.
"queued": "uploads/_queued.html",
}.get(log_file.status, "uploads/_status_processing.html")
return render_template(template, log_file=log_file)
@bp.get("/api/uploads")
def list_uploads():
"""Uploaded-files list (project-owner follow-up request) — Grid.js-
backed, same page/per_page convention as every other table (Ch11).
Sorted by upload recency, independent of the dashboard date-range
picker (a single file can span many log dates).
"""
page, per_page = parse_pagination(request)
rows, total = get_uploaded_files(page, per_page)
return jsonify(
data={"rows": rows, "total": total},
meta={"page": page, "per_page": per_page},
)
@bp.delete("/api/uploads")
def bulk_delete_uploads():
"""Delete one or more uploaded files and everything derived from them
(log_entries/bot_hits/suspicious_events, the raw file on disk, and a
correct rollup recompute for the affected dates — see
app/services/file_deletion.py for why a rollup recompute is needed
rather than a simple per-file delete).
Body: {"ids": [1, 2, 3]}. Files currently "processing" are skipped,
not force-deleted, to avoid racing the background parse thread.
"""
body = request.get_json(silent=True) or {}
ids = body.get("ids")
if not isinstance(ids, list) or not ids or not all(isinstance(i, int) for i in ids):
return jsonify(error={"code": "invalid_request", "message": "Expected {\"ids\": [int, ...]}."}), 400
result = delete_log_files(ids)
return jsonify(data={"deleted": result.deleted, "skipped": result.skipped}, meta={})