"""No-cron automatic processing (Chapter 12 simplification, per project owner request): triggers file parsing immediately in a background thread right after upload, and opportunistically resumes any incomplete files when the Overview page loads — replacing the cron-triggered model. `flask process-logs` still exists in app/cli.py for anyone who'd rather use cron, but nothing requires it anymore. TRADEOFF (flagged, deviating from Chapter 02/03's "no persistent background workers, cron-triggered CLI only" stance): a background thread lives inside the same worker process that handled the upload request. If Passenger recycles that process mid-parse, the thread dies with it — progress up to the last commit is still safely checkpointed (same bounded- batch model as before), but nothing will automatically resume it without either cron or a page visit. The "resume on page load" hook below is the deliberate replacement for that guarantee: visiting the Overview tab re-triggers processing for anything left incomplete, so in the worst case a stuck file resumes the next time the admin looks at the dashboard, rather than never. This is not a long-lived daemon: each thread terminates once its file reaches "done"/"error" (or the process is killed), and no thread survives a process restart — it just gets re-triggered fresh next time. """ from __future__ import annotations import threading from flask import Flask from app.extensions import db from app.models.log_file import LogFile from app.services.log_processor import process_one_batch # In-process guard against launching two threads for the same file at # once (e.g. the upload trigger and a page-load resume firing close # together). Per-worker-process only — a different entry process picking # up the same file concurrently is a low-probability edge case accepted # for this simplification; each write is still a small checkpointed # commit, not a giant one, which limits how bad a collision could be. _active_file_ids: set[int] = set() _lock = threading.Lock() def _claim(log_file_id: int) -> bool: with _lock: if log_file_id in _active_file_ids: return False _active_file_ids.add(log_file_id) return True def _release(log_file_id: int) -> None: with _lock: _active_file_ids.discard(log_file_id) def _run_to_completion(app: Flask, log_file_id: int, batch_size: int) -> None: with app.app_context(): try: log_file = db.session.get(LogFile, log_file_id) if log_file is None: return while log_file.status in ("queued", "processing"): process_one_batch(log_file, batch_size) db.session.refresh(log_file) except Exception: app.logger.exception("Background processing failed for log_file_id=%s", log_file_id) log_file = db.session.get(LogFile, log_file_id) if log_file is not None and log_file.status != "done": log_file.status = "error" log_file.error_message = "Processing failed unexpectedly; see server logs." db.session.commit() finally: _release(log_file_id) def trigger_processing(app: Flask, log_file_id: int) -> None: """Start background processing for one file; no-ops if already running.""" if not _claim(log_file_id): return batch_size = app.config["PARSE_BATCH_SIZE"] thread = threading.Thread( target=_run_to_completion, args=(app, log_file_id, batch_size), daemon=True ) thread.start() def resume_incomplete_files(app: Flask) -> None: """Opportunistic resume hook, called from the Overview page load — the deliberate replacement for cron's "there's always a next tick" guarantee. Cheap: one indexed status-filtered query. """ incomplete_ids = [ lf.id for lf in LogFile.query.filter(LogFile.status.in_(["queued", "processing"])).all() ] for log_file_id in incomplete_ids: trigger_processing(app, log_file_id)