start project
This commit is contained in:
@@ -0,0 +1,97 @@
|
||||
"""No-cron automatic processing (Chapter 12 simplification, per project
|
||||
owner request): triggers file parsing immediately in a background thread
|
||||
right after upload, and opportunistically resumes any incomplete files
|
||||
when the Overview page loads — replacing the cron-triggered model.
|
||||
`flask process-logs` still exists in app/cli.py for anyone who'd rather
|
||||
use cron, but nothing requires it anymore.
|
||||
|
||||
TRADEOFF (flagged, deviating from Chapter 02/03's "no persistent
|
||||
background workers, cron-triggered CLI only" stance): a background thread
|
||||
lives inside the same worker process that handled the upload request. If
|
||||
Passenger recycles that process mid-parse, the thread dies with it —
|
||||
progress up to the last commit is still safely checkpointed (same bounded-
|
||||
batch model as before), but nothing will automatically resume it without
|
||||
either cron or a page visit. The "resume on page load" hook below is the
|
||||
deliberate replacement for that guarantee: visiting the Overview tab
|
||||
re-triggers processing for anything left incomplete, so in the worst case
|
||||
a stuck file resumes the next time the admin looks at the dashboard,
|
||||
rather than never.
|
||||
|
||||
This is not a long-lived daemon: each thread terminates once its file
|
||||
reaches "done"/"error" (or the process is killed), and no thread survives
|
||||
a process restart — it just gets re-triggered fresh next time.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import threading
|
||||
|
||||
from flask import Flask
|
||||
|
||||
from app.extensions import db
|
||||
from app.models.log_file import LogFile
|
||||
from app.services.log_processor import process_one_batch
|
||||
|
||||
# In-process guard against launching two threads for the same file at
|
||||
# once (e.g. the upload trigger and a page-load resume firing close
|
||||
# together). Per-worker-process only — a different entry process picking
|
||||
# up the same file concurrently is a low-probability edge case accepted
|
||||
# for this simplification; each write is still a small checkpointed
|
||||
# commit, not a giant one, which limits how bad a collision could be.
|
||||
_active_file_ids: set[int] = set()
|
||||
_lock = threading.Lock()
|
||||
|
||||
|
||||
def _claim(log_file_id: int) -> bool:
|
||||
with _lock:
|
||||
if log_file_id in _active_file_ids:
|
||||
return False
|
||||
_active_file_ids.add(log_file_id)
|
||||
return True
|
||||
|
||||
|
||||
def _release(log_file_id: int) -> None:
|
||||
with _lock:
|
||||
_active_file_ids.discard(log_file_id)
|
||||
|
||||
|
||||
def _run_to_completion(app: Flask, log_file_id: int, batch_size: int) -> None:
|
||||
with app.app_context():
|
||||
try:
|
||||
log_file = db.session.get(LogFile, log_file_id)
|
||||
if log_file is None:
|
||||
return
|
||||
while log_file.status in ("queued", "processing"):
|
||||
process_one_batch(log_file, batch_size)
|
||||
db.session.refresh(log_file)
|
||||
except Exception:
|
||||
app.logger.exception("Background processing failed for log_file_id=%s", log_file_id)
|
||||
log_file = db.session.get(LogFile, log_file_id)
|
||||
if log_file is not None and log_file.status != "done":
|
||||
log_file.status = "error"
|
||||
log_file.error_message = "Processing failed unexpectedly; see server logs."
|
||||
db.session.commit()
|
||||
finally:
|
||||
_release(log_file_id)
|
||||
|
||||
|
||||
def trigger_processing(app: Flask, log_file_id: int) -> None:
|
||||
"""Start background processing for one file; no-ops if already running."""
|
||||
if not _claim(log_file_id):
|
||||
return
|
||||
batch_size = app.config["PARSE_BATCH_SIZE"]
|
||||
thread = threading.Thread(
|
||||
target=_run_to_completion, args=(app, log_file_id, batch_size), daemon=True
|
||||
)
|
||||
thread.start()
|
||||
|
||||
|
||||
def resume_incomplete_files(app: Flask) -> None:
|
||||
"""Opportunistic resume hook, called from the Overview page load —
|
||||
the deliberate replacement for cron's "there's always a next tick"
|
||||
guarantee. Cheap: one indexed status-filtered query.
|
||||
"""
|
||||
incomplete_ids = [
|
||||
lf.id for lf in LogFile.query.filter(LogFile.status.in_(["queued", "processing"])).all()
|
||||
]
|
||||
for log_file_id in incomplete_ids:
|
||||
trigger_processing(app, log_file_id)
|
||||
Reference in New Issue
Block a user