Files
Kavosh/app/services/background.py
T
2026-08-07 21:17:17 +03:30

98 lines
4.0 KiB
Python

"""No-cron automatic processing (Chapter 12 simplification, per project
owner request): triggers file parsing immediately in a background thread
right after upload, and opportunistically resumes any incomplete files
when the Overview page loads — replacing the cron-triggered model.
`flask process-logs` still exists in app/cli.py for anyone who'd rather
use cron, but nothing requires it anymore.
TRADEOFF (flagged, deviating from Chapter 02/03's "no persistent
background workers, cron-triggered CLI only" stance): a background thread
lives inside the same worker process that handled the upload request. If
Passenger recycles that process mid-parse, the thread dies with it —
progress up to the last commit is still safely checkpointed (same bounded-
batch model as before), but nothing will automatically resume it without
either cron or a page visit. The "resume on page load" hook below is the
deliberate replacement for that guarantee: visiting the Overview tab
re-triggers processing for anything left incomplete, so in the worst case
a stuck file resumes the next time the admin looks at the dashboard,
rather than never.
This is not a long-lived daemon: each thread terminates once its file
reaches "done"/"error" (or the process is killed), and no thread survives
a process restart — it just gets re-triggered fresh next time.
"""
from __future__ import annotations
import threading
from flask import Flask
from app.extensions import db
from app.models.log_file import LogFile
from app.services.log_processor import process_one_batch
# In-process guard against launching two threads for the same file at
# once (e.g. the upload trigger and a page-load resume firing close
# together). Per-worker-process only — a different entry process picking
# up the same file concurrently is a low-probability edge case accepted
# for this simplification; each write is still a small checkpointed
# commit, not a giant one, which limits how bad a collision could be.
_active_file_ids: set[int] = set()
_lock = threading.Lock()
def _claim(log_file_id: int) -> bool:
with _lock:
if log_file_id in _active_file_ids:
return False
_active_file_ids.add(log_file_id)
return True
def _release(log_file_id: int) -> None:
with _lock:
_active_file_ids.discard(log_file_id)
def _run_to_completion(app: Flask, log_file_id: int, batch_size: int) -> None:
with app.app_context():
try:
log_file = db.session.get(LogFile, log_file_id)
if log_file is None:
return
while log_file.status in ("queued", "processing"):
process_one_batch(log_file, batch_size)
db.session.refresh(log_file)
except Exception:
app.logger.exception("Background processing failed for log_file_id=%s", log_file_id)
log_file = db.session.get(LogFile, log_file_id)
if log_file is not None and log_file.status != "done":
log_file.status = "error"
log_file.error_message = "Processing failed unexpectedly; see server logs."
db.session.commit()
finally:
_release(log_file_id)
def trigger_processing(app: Flask, log_file_id: int) -> None:
"""Start background processing for one file; no-ops if already running."""
if not _claim(log_file_id):
return
batch_size = app.config["PARSE_BATCH_SIZE"]
thread = threading.Thread(
target=_run_to_completion, args=(app, log_file_id, batch_size), daemon=True
)
thread.start()
def resume_incomplete_files(app: Flask) -> None:
"""Opportunistic resume hook, called from the Overview page load —
the deliberate replacement for cron's "there's always a next tick"
guarantee. Cheap: one indexed status-filtered query.
"""
incomplete_ids = [
lf.id for lf in LogFile.query.filter(LogFile.status.in_(["queued", "processing"])).all()
]
for log_file_id in incomplete_ids:
trigger_processing(app, log_file_id)