start project
This commit is contained in:
@@ -0,0 +1,84 @@
|
||||
"""Application factory (Chapter 02 / Chapter 04)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from flask import Flask, jsonify, redirect, request, url_for
|
||||
from flask_login import current_user
|
||||
from sqlalchemy import text
|
||||
|
||||
from app.budget import validate_budget_config
|
||||
from app.config import get_config
|
||||
from app.extensions import cache, csrf, db, limiter, login_manager, migrate
|
||||
from app.logging_setup import configure_logging
|
||||
from app.utils.assets import register_asset_helper
|
||||
|
||||
|
||||
def create_app(config_name: str | None = None) -> Flask:
|
||||
"""Build and configure the Flask application instance."""
|
||||
app = Flask(__name__)
|
||||
app.config.from_object(get_config(config_name))
|
||||
validate_budget_config(app) # Chapter 03: fail loudly on out-of-budget config
|
||||
configure_logging(app) # Chapter 02 factor 11 / Chapter 12
|
||||
|
||||
if not app.config.get("SECRET_KEY") and not app.testing:
|
||||
raise RuntimeError("SECRET_KEY must be set via environment variable")
|
||||
|
||||
db.init_app(app)
|
||||
cache.init_app(app)
|
||||
csrf.init_app(app)
|
||||
login_manager.init_app(app)
|
||||
migrate.init_app(app, db)
|
||||
limiter.init_app(app) # Chapter 12: rate limiting
|
||||
|
||||
register_blueprints(app)
|
||||
register_cli(app)
|
||||
register_asset_helper(app)
|
||||
|
||||
@login_manager.unauthorized_handler
|
||||
def handle_unauthorized():
|
||||
"""Chapter 11 envelope compliance: JSON 401 for /api/... paths
|
||||
instead of Flask-Login's default redirect (which would hand back
|
||||
login HTML to a fetch()/HTMX JSON caller). Full-page routes still
|
||||
redirect to the login form as normal.
|
||||
"""
|
||||
if request.path.startswith("/api/"):
|
||||
return jsonify(error={"code": "unauthorized", "message": "Authentication required."}), 401
|
||||
return redirect(url_for("auth.login", next=request.path))
|
||||
|
||||
@app.get("/healthz")
|
||||
def healthz():
|
||||
"""Liveness/readiness check (Chapter 12) — DB reachable, no long-running work."""
|
||||
db.session.execute(text("SELECT 1"))
|
||||
return {"data": {"status": "ok"}, "meta": {}}
|
||||
|
||||
@app.get("/")
|
||||
def index():
|
||||
"""Root URL: authenticated -> Overview, otherwise -> the login page."""
|
||||
if current_user.is_authenticated:
|
||||
return redirect(url_for("overview.overview"))
|
||||
return redirect(url_for("auth.login"))
|
||||
|
||||
return app
|
||||
|
||||
|
||||
def register_blueprints(app: Flask) -> None:
|
||||
"""Register one blueprint per dashboard section + support area (Ch. 04)."""
|
||||
from app.blueprints.api import bp as api_bp
|
||||
from app.blueprints.auth import bp as auth_bp
|
||||
from app.blueprints.overview import bp as overview_bp
|
||||
from app.blueprints.security import bp as security_bp
|
||||
from app.blueprints.seo import bp as seo_bp
|
||||
from app.blueprints.uploads import bp as uploads_bp
|
||||
|
||||
app.register_blueprint(overview_bp)
|
||||
app.register_blueprint(seo_bp)
|
||||
app.register_blueprint(security_bp)
|
||||
app.register_blueprint(uploads_bp)
|
||||
app.register_blueprint(api_bp)
|
||||
app.register_blueprint(auth_bp)
|
||||
|
||||
|
||||
def register_cli(app: Flask) -> None:
|
||||
"""Attach `flask <command>` admin processes (factor 12)."""
|
||||
from app.cli import register_commands
|
||||
|
||||
register_commands(app)
|
||||
@@ -0,0 +1,7 @@
|
||||
"""Deliberately minimal — Chapter 02's tree lists a standalone `api`
|
||||
blueprint, but Chapter 04/11 route each section's JSON endpoints through
|
||||
its own blueprint instead. Left unpopulated per the project owner's
|
||||
instruction to defer this branch."""
|
||||
from flask import Blueprint
|
||||
|
||||
bp = Blueprint("api", __name__)
|
||||
@@ -0,0 +1,5 @@
|
||||
from flask import Blueprint
|
||||
|
||||
bp = Blueprint("auth", __name__, template_folder="templates")
|
||||
|
||||
from app.blueprints.auth import routes # noqa: E402,F401 registers routes
|
||||
@@ -0,0 +1,18 @@
|
||||
"""Login form (Chapter 12). CSRF handled automatically via form.hidden_tag()
|
||||
(Flask-WTF, Ch12's CSRF requirement)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from flask_wtf import FlaskForm
|
||||
from wtforms import PasswordField, StringField, SubmitField
|
||||
from wtforms.validators import DataRequired
|
||||
|
||||
|
||||
class LoginForm(FlaskForm):
|
||||
# NOTE: no Email() validator — that validator requires the extra
|
||||
# `email-validator` package. Login checks the value against a stored
|
||||
# user row anyway, so a malformed email simply fails to match rather
|
||||
# than needing format validation up front; kept simple to avoid a new
|
||||
# dependency.
|
||||
email = StringField("Email", validators=[DataRequired()])
|
||||
password = PasswordField("Password", validators=[DataRequired()])
|
||||
submit = SubmitField("Log in")
|
||||
@@ -0,0 +1,45 @@
|
||||
"""Single-admin auth routes (Chapter 12 / Chapter 11's route table)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from flask import flash, redirect, render_template, request, url_for
|
||||
from flask_login import current_user, login_required, login_user, logout_user
|
||||
|
||||
from app.blueprints.auth import bp
|
||||
from app.blueprints.auth.forms import LoginForm
|
||||
from app.extensions import db, limiter, login_manager
|
||||
from app.models.user import User
|
||||
|
||||
|
||||
@login_manager.user_loader
|
||||
def load_user(user_id: str) -> User | None:
|
||||
return db.session.get(User, int(user_id))
|
||||
|
||||
|
||||
@bp.route("/login", methods=["GET", "POST"])
|
||||
@limiter.limit("10 per minute") # Ch12: rate limiting on /login at minimum
|
||||
def login():
|
||||
if current_user.is_authenticated:
|
||||
return redirect(url_for("overview.overview"))
|
||||
|
||||
form = LoginForm()
|
||||
if form.validate_on_submit():
|
||||
user = User.query.filter_by(email=form.email.data.strip().lower()).first()
|
||||
if user is not None and user.check_password(form.password.data):
|
||||
login_user(user)
|
||||
next_url = request.args.get("next") or url_for("overview.overview")
|
||||
return redirect(next_url)
|
||||
flash("Invalid email or password.")
|
||||
|
||||
return render_template("auth/login.html", form=form)
|
||||
|
||||
|
||||
@bp.post("/logout")
|
||||
@login_required
|
||||
def logout():
|
||||
# NOTE (flagged): Ch11's route table lists "/login, /logout" under a
|
||||
# shared "GET/POST" column. Logout is POST-only here — a state-changing
|
||||
# action behind a plain GET is a CSRF-adjacent anti-pattern Ch12's own
|
||||
# CSRF requirement argues against; login stays GET (show form) + POST
|
||||
# (submit), matching the table as-is.
|
||||
logout_user()
|
||||
return redirect(url_for("auth.login"))
|
||||
@@ -0,0 +1,27 @@
|
||||
{% extends "base.html" %}
|
||||
{% block title %}Log in — Kavosh{% endblock %}
|
||||
{% block content %}
|
||||
<div class="max-w-sm mx-auto mt-16 sm:mt-24">
|
||||
<div class="text-center mb-6">
|
||||
<h1 class="font-display font-bold text-2xl tracking-tight">Kavosh</h1>
|
||||
<p class="text-sm text-muted dark:text-muted-dark mt-1">Sign in to view your site's traffic</p>
|
||||
</div>
|
||||
<div class="bg-surface dark:bg-surface-dark border border-line dark:border-line-dark rounded-xl p-6 shadow-sm">
|
||||
{% for message in get_flashed_messages() %}
|
||||
<p class="text-danger dark:text-danger-dark text-sm mb-3 bg-danger/10 dark:bg-danger-dark/10 rounded-md px-3 py-2">{{ message }}</p>
|
||||
{% endfor %}
|
||||
<form method="post">
|
||||
{{ form.hidden_tag() }}
|
||||
<div class="mb-4">
|
||||
{{ form.email.label(class="block text-sm font-medium text-muted dark:text-muted-dark mb-1") }}
|
||||
{{ form.email(class="w-full border border-line dark:border-line-dark rounded-md px-3 py-2 bg-paper dark:bg-paper-dark text-ink dark:text-ink-dark focus:outline-none focus:ring-2 focus:ring-accent dark:focus:ring-accent-dark", autofocus=true) }}
|
||||
</div>
|
||||
<div class="mb-5">
|
||||
{{ form.password.label(class="block text-sm font-medium text-muted dark:text-muted-dark mb-1") }}
|
||||
{{ form.password(class="w-full border border-line dark:border-line-dark rounded-md px-3 py-2 bg-paper dark:bg-paper-dark text-ink dark:text-ink-dark focus:outline-none focus:ring-2 focus:ring-accent dark:focus:ring-accent-dark") }}
|
||||
</div>
|
||||
{{ form.submit(class="w-full bg-accent dark:bg-accent-dark text-white dark:text-paper-dark font-medium rounded-md px-4 py-2 hover:opacity-90 transition-opacity cursor-pointer") }}
|
||||
</form>
|
||||
</div>
|
||||
</div>
|
||||
{% endblock %}
|
||||
@@ -0,0 +1,13 @@
|
||||
from flask import Blueprint
|
||||
from flask_login import login_required
|
||||
|
||||
bp = Blueprint("overview", __name__, template_folder="templates")
|
||||
|
||||
|
||||
@bp.before_request
|
||||
@login_required
|
||||
def require_login():
|
||||
pass
|
||||
|
||||
|
||||
from app.blueprints.overview import routes # noqa: E402,F401 registers routes
|
||||
@@ -0,0 +1,167 @@
|
||||
"""Context-builder query functions for the Overview tab (Chapter 08).
|
||||
|
||||
Every function here reads request_stats_hourly / request_stats_daily /
|
||||
referrer_stats_daily / browser_stats_daily — never log_entries — per
|
||||
Ch03 rule 6 / Ch08's own data-source rule.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import date
|
||||
|
||||
from sqlalchemy import case, func
|
||||
|
||||
from app.extensions import db
|
||||
from app.models.browser_stats import BrowserStatsDaily
|
||||
from app.models.referrer_stats import ReferrerStatsDaily
|
||||
from app.models.request_stats import RequestStatsDaily, RequestStatsHourly
|
||||
from app.utils.dates import day_bounds
|
||||
|
||||
# ASSUMPTION (flagged in Ch08): Ch08 doesn't give a numeric threshold for
|
||||
# "hourly vs daily depending on range width" — picked 3 days.
|
||||
HOURLY_GRANULARITY_THRESHOLD_DAYS = 3
|
||||
|
||||
|
||||
def get_kpis(from_date: date, to_date: date) -> dict:
|
||||
"""All six Ch08 KPI-card fields, plus peak-day (folded in — Ch11 has
|
||||
no dedicated route for it and Ch08 calls for only a 'simple max-lookup').
|
||||
"""
|
||||
row = (
|
||||
db.session.query(
|
||||
func.coalesce(func.sum(RequestStatsDaily.count), 0).label("total_requests"),
|
||||
func.coalesce(func.sum(RequestStatsDaily.unique_ips), 0).label("unique_ips_sum"),
|
||||
func.coalesce(func.sum(RequestStatsDaily.bytes_sum), 0).label("total_bandwidth"),
|
||||
func.coalesce(func.sum(RequestStatsDaily.error_count), 0).label("total_errors"),
|
||||
)
|
||||
.filter(RequestStatsDaily.date >= from_date, RequestStatsDaily.date <= to_date)
|
||||
.one()
|
||||
)
|
||||
num_days = (to_date - from_date).days + 1
|
||||
avg_response_size = (row.total_bandwidth / row.total_requests) if row.total_requests else 0.0
|
||||
error_rate_pct = (row.total_errors / row.total_requests * 100) if row.total_requests else 0.0
|
||||
avg_requests_per_day = row.total_requests / num_days if num_days else 0.0
|
||||
|
||||
peak_row = (
|
||||
db.session.query(RequestStatsDaily.date, RequestStatsDaily.count)
|
||||
.filter(RequestStatsDaily.date >= from_date, RequestStatsDaily.date <= to_date)
|
||||
.order_by(RequestStatsDaily.count.desc())
|
||||
.first()
|
||||
)
|
||||
|
||||
return {
|
||||
"total_requests": row.total_requests,
|
||||
# APPROXIMATION (flagged in Ch08): sum of daily unique_ips over-counts
|
||||
# repeat visitors across days.
|
||||
"unique_ips": row.unique_ips_sum,
|
||||
"total_bandwidth_bytes": row.total_bandwidth,
|
||||
"avg_response_size_bytes": round(avg_response_size, 1),
|
||||
"error_rate_pct": round(error_rate_pct, 2),
|
||||
"avg_requests_per_day": round(avg_requests_per_day, 1),
|
||||
"peak_day": {"date": peak_row.date.isoformat(), "count": peak_row.count} if peak_row else None,
|
||||
}
|
||||
|
||||
|
||||
def get_traffic_chart_series(from_date: date, to_date: date) -> dict:
|
||||
span_days = (to_date - from_date).days + 1
|
||||
if span_days <= HOURLY_GRANULARITY_THRESHOLD_DAYS:
|
||||
start, end = day_bounds(from_date, to_date)
|
||||
rows = (
|
||||
db.session.query(RequestStatsHourly.date_hour, func.sum(RequestStatsHourly.count).label("count"))
|
||||
.filter(RequestStatsHourly.date_hour >= start, RequestStatsHourly.date_hour < end)
|
||||
.group_by(RequestStatsHourly.date_hour)
|
||||
.order_by(RequestStatsHourly.date_hour)
|
||||
.all()
|
||||
)
|
||||
return {"granularity": "hourly", "series": [{"t": r.date_hour.isoformat(), "count": r.count} for r in rows]}
|
||||
|
||||
rows = (
|
||||
db.session.query(RequestStatsDaily.date, RequestStatsDaily.count)
|
||||
.filter(RequestStatsDaily.date >= from_date, RequestStatsDaily.date <= to_date)
|
||||
.order_by(RequestStatsDaily.date)
|
||||
.all()
|
||||
)
|
||||
return {"granularity": "daily", "series": [{"t": r.date.isoformat(), "count": r.count} for r in rows]}
|
||||
|
||||
|
||||
def get_status_code_breakdown(from_date: date, to_date: date) -> dict:
|
||||
start, end = day_bounds(from_date, to_date)
|
||||
bucket = case(
|
||||
(RequestStatsHourly.status_code < 300, "2xx"),
|
||||
(RequestStatsHourly.status_code < 400, "3xx"),
|
||||
(RequestStatsHourly.status_code < 500, "4xx"),
|
||||
else_="5xx",
|
||||
)
|
||||
rows = (
|
||||
db.session.query(bucket.label("bucket"), func.sum(RequestStatsHourly.count).label("count"))
|
||||
.filter(RequestStatsHourly.date_hour >= start, RequestStatsHourly.date_hour < end)
|
||||
.group_by("bucket")
|
||||
.all()
|
||||
)
|
||||
breakdown = {"2xx": 0, "3xx": 0, "4xx": 0, "5xx": 0}
|
||||
for r in rows:
|
||||
breakdown[r.bucket] = r.count
|
||||
return breakdown
|
||||
|
||||
|
||||
def get_top_urls(from_date: date, to_date: date, page: int, per_page: int) -> tuple[list[list], int]:
|
||||
"""Top URLs by hits. NOTE (flagged in Ch08): only available within the
|
||||
~90-day hourly retention window (Ch06) — request_stats_daily has no
|
||||
path column, so a wider range returns nothing here.
|
||||
"""
|
||||
start, end = day_bounds(from_date, to_date)
|
||||
hits = func.sum(RequestStatsHourly.count)
|
||||
errors = func.sum(case((RequestStatsHourly.status_code >= 400, RequestStatsHourly.count), else_=0))
|
||||
bytes_sum = func.sum(RequestStatsHourly.bytes_sent_sum)
|
||||
|
||||
base_query = (
|
||||
db.session.query(RequestStatsHourly.path, hits.label("hits"), bytes_sum.label("bytes_sum"), errors.label("errors"))
|
||||
.filter(RequestStatsHourly.date_hour >= start, RequestStatsHourly.date_hour < end)
|
||||
.group_by(RequestStatsHourly.path)
|
||||
)
|
||||
total = base_query.count()
|
||||
rows = base_query.order_by(hits.desc()).offset((page - 1) * per_page).limit(per_page).all()
|
||||
|
||||
results = [
|
||||
[
|
||||
r.path,
|
||||
r.hits,
|
||||
round(r.bytes_sum / r.hits, 1) if r.hits else 0.0,
|
||||
round(r.errors / r.hits * 100, 2) if r.hits else 0.0,
|
||||
]
|
||||
for r in rows
|
||||
]
|
||||
return results, total
|
||||
|
||||
|
||||
def get_top_referrers(from_date: date, to_date: date, page: int, per_page: int) -> tuple[list[list], int]:
|
||||
"""Domain-bucketed referrers (Method A, Ch08 follow-up)."""
|
||||
hits = func.sum(ReferrerStatsDaily.count)
|
||||
base_query = (
|
||||
db.session.query(ReferrerStatsDaily.referrer_domain, hits.label("hits"))
|
||||
.filter(ReferrerStatsDaily.date >= from_date, ReferrerStatsDaily.date <= to_date)
|
||||
.group_by(ReferrerStatsDaily.referrer_domain)
|
||||
)
|
||||
total = base_query.count()
|
||||
rows = base_query.order_by(hits.desc()).offset((page - 1) * per_page).limit(per_page).all()
|
||||
return [[r.referrer_domain, r.hits] for r in rows], total
|
||||
|
||||
|
||||
def get_browser_breakdown(from_date: date, to_date: date) -> dict:
|
||||
"""Human-only (bots excluded at rollup-write time, Ch08)."""
|
||||
browser_rows = (
|
||||
db.session.query(BrowserStatsDaily.browser, func.sum(BrowserStatsDaily.count).label("count"))
|
||||
.filter(BrowserStatsDaily.date >= from_date, BrowserStatsDaily.date <= to_date)
|
||||
.group_by(BrowserStatsDaily.browser)
|
||||
.order_by(func.sum(BrowserStatsDaily.count).desc())
|
||||
.all()
|
||||
)
|
||||
os_rows = (
|
||||
db.session.query(BrowserStatsDaily.os, func.sum(BrowserStatsDaily.count).label("count"))
|
||||
.filter(BrowserStatsDaily.date >= from_date, BrowserStatsDaily.date <= to_date)
|
||||
.group_by(BrowserStatsDaily.os)
|
||||
.order_by(func.sum(BrowserStatsDaily.count).desc())
|
||||
.all()
|
||||
)
|
||||
return {
|
||||
"by_browser": [{"name": r.browser, "count": r.count} for r in browser_rows],
|
||||
"by_os": [{"name": r.os, "count": r.count} for r in os_rows],
|
||||
}
|
||||
@@ -0,0 +1,94 @@
|
||||
"""Overview blueprint routes (Chapter 08 / Chapter 11's route table)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from flask import current_app, jsonify, request
|
||||
|
||||
from app.blueprints.overview import bp
|
||||
from app.blueprints.overview.queries import (
|
||||
get_browser_breakdown,
|
||||
get_kpis,
|
||||
get_status_code_breakdown,
|
||||
get_top_referrers,
|
||||
get_top_urls,
|
||||
get_traffic_chart_series,
|
||||
)
|
||||
from app.services.background import resume_incomplete_files
|
||||
from app.utils.dates import parse_date_range
|
||||
from app.utils.htmx import render_htmx_aware
|
||||
from app.utils.pagination import parse_pagination
|
||||
|
||||
|
||||
@bp.route("/overview")
|
||||
def overview():
|
||||
"""Full page on first load, HTMX partial on tab switch / range change (Ch04).
|
||||
|
||||
Also opportunistically resumes any log_files stuck in queued/processing
|
||||
(Ch12 simplification: the replacement for cron's "there's always a
|
||||
next tick" guarantee — see app/services/background.py).
|
||||
"""
|
||||
resume_incomplete_files(current_app._get_current_object())
|
||||
from_date, to_date = parse_date_range(request)
|
||||
return render_htmx_aware(
|
||||
request,
|
||||
full_template="overview/index.html",
|
||||
partial_template="overview/_content.html",
|
||||
from_date=from_date,
|
||||
to_date=to_date,
|
||||
)
|
||||
|
||||
|
||||
@bp.get("/api/overview/kpis")
|
||||
def api_kpis():
|
||||
from_date, to_date = parse_date_range(request)
|
||||
return jsonify(data=get_kpis(from_date, to_date), meta={"from": from_date.isoformat(), "to": to_date.isoformat()})
|
||||
|
||||
|
||||
@bp.get("/api/overview/traffic-chart")
|
||||
def api_traffic_chart():
|
||||
from_date, to_date = parse_date_range(request)
|
||||
return jsonify(
|
||||
data=get_traffic_chart_series(from_date, to_date),
|
||||
meta={"from": from_date.isoformat(), "to": to_date.isoformat()},
|
||||
)
|
||||
|
||||
|
||||
@bp.get("/api/overview/status-codes")
|
||||
def api_status_codes():
|
||||
from_date, to_date = parse_date_range(request)
|
||||
return jsonify(
|
||||
data=get_status_code_breakdown(from_date, to_date),
|
||||
meta={"from": from_date.isoformat(), "to": to_date.isoformat()},
|
||||
)
|
||||
|
||||
|
||||
@bp.get("/api/overview/top-urls")
|
||||
def api_top_urls():
|
||||
from_date, to_date = parse_date_range(request)
|
||||
page, per_page = parse_pagination(request)
|
||||
rows, total = get_top_urls(from_date, to_date, page, per_page)
|
||||
return jsonify(
|
||||
data={"rows": rows, "total": total},
|
||||
meta={"from": from_date.isoformat(), "to": to_date.isoformat(), "page": page, "per_page": per_page},
|
||||
)
|
||||
|
||||
|
||||
@bp.get("/api/overview/top-referrers")
|
||||
def api_top_referrers():
|
||||
from_date, to_date = parse_date_range(request)
|
||||
page, per_page = parse_pagination(request)
|
||||
rows, total = get_top_referrers(from_date, to_date, page, per_page)
|
||||
return jsonify(
|
||||
data={"rows": rows, "total": total},
|
||||
meta={"from": from_date.isoformat(), "to": to_date.isoformat(), "page": page, "per_page": per_page},
|
||||
)
|
||||
|
||||
|
||||
@bp.get("/api/overview/browser-breakdown")
|
||||
def api_browser_breakdown():
|
||||
"""Chapter 11 addition (Method A, Ch08 follow-up) — not in the
|
||||
original route table; see docs/api-contract-final.md."""
|
||||
from_date, to_date = parse_date_range(request)
|
||||
return jsonify(
|
||||
data=get_browser_breakdown(from_date, to_date),
|
||||
meta={"from": from_date.isoformat(), "to": to_date.isoformat()},
|
||||
)
|
||||
@@ -0,0 +1,286 @@
|
||||
<div id="overview-content"
|
||||
hx-get="{{ url_for('overview.overview') }}"
|
||||
hx-trigger="change from:#date-range-form"
|
||||
hx-include="#date-range-form"
|
||||
hx-target="#overview-content"
|
||||
hx-swap="outerHTML"
|
||||
data-from="{{ from_date.isoformat() }}"
|
||||
data-to="{{ to_date.isoformat() }}">
|
||||
|
||||
<div class="flex flex-wrap items-end justify-between gap-4 mb-6">
|
||||
<div>
|
||||
<h1 class="font-display font-bold text-xl">Overview</h1>
|
||||
<p class="text-sm text-muted dark:text-muted-dark">What happened on your site recently</p>
|
||||
</div>
|
||||
<form id="date-range-form" class="flex gap-3 items-end">
|
||||
<label class="text-sm text-muted dark:text-muted-dark">From
|
||||
<input type="date" name="from" value="{{ from_date.isoformat() }}"
|
||||
class="block border border-line dark:border-line-dark rounded-md px-2 py-1 mt-1 bg-surface dark:bg-surface-dark text-ink dark:text-ink-dark font-data text-sm">
|
||||
</label>
|
||||
<label class="text-sm text-muted dark:text-muted-dark">To
|
||||
<input type="date" name="to" value="{{ to_date.isoformat() }}"
|
||||
class="block border border-line dark:border-line-dark rounded-md px-2 py-1 mt-1 bg-surface dark:bg-surface-dark text-ink dark:text-ink-dark font-data text-sm">
|
||||
</label>
|
||||
</form>
|
||||
</div>
|
||||
|
||||
<!-- Upload panel — real byte-transfer progress via XHR (uploads.js), then
|
||||
a real parse-progress bar (Ch12) once the file is queued. No cron
|
||||
required: processing starts automatically in the background. -->
|
||||
<div class="mb-6 border border-line dark:border-line-dark rounded-xl p-4 bg-surface dark:bg-surface-dark">
|
||||
<h2 class="font-display font-semibold text-sm mb-3">Upload a log file</h2>
|
||||
<form id="upload-form" action="{{ url_for('uploads.upload_log_file') }}"
|
||||
class="flex flex-wrap gap-3 items-center">
|
||||
<input type="file" name="logfile" accept=".log,.txt,.gz" required
|
||||
class="text-sm text-muted dark:text-muted-dark file:mr-3 file:py-1.5 file:px-3 file:rounded-md file:border-0 file:bg-accent/10 file:text-accent dark:file:bg-accent-dark/15 dark:file:text-accent-dark file:text-sm file:font-medium hover:file:bg-accent/20 dark:hover:file:bg-accent-dark/25 file:cursor-pointer cursor-pointer">
|
||||
<select name="server_type" class="border border-line dark:border-line-dark rounded-md px-2 py-1.5 text-sm bg-surface dark:bg-surface-dark text-ink dark:text-ink-dark">
|
||||
<option value="apache">Apache</option>
|
||||
<option value="litespeed">LiteSpeed</option>
|
||||
</select>
|
||||
<input type="text" name="format_string" placeholder="LogFormat (optional — defaults to Combined)"
|
||||
class="border border-line dark:border-line-dark rounded-md px-2 py-1.5 text-sm flex-1 min-w-[220px] bg-surface dark:bg-surface-dark text-ink dark:text-ink-dark placeholder:text-muted dark:placeholder:text-muted-dark">
|
||||
<button type="submit" class="bg-accent dark:bg-accent-dark text-white dark:text-paper-dark text-sm font-medium rounded-md px-4 py-1.5 hover:opacity-90 transition-opacity">
|
||||
Upload
|
||||
</button>
|
||||
</form>
|
||||
|
||||
<div id="upload-progress-wrap" class="hidden mt-3">
|
||||
<div class="flex items-center justify-between text-sm mb-1">
|
||||
<span id="upload-progress-label" class="text-muted dark:text-muted-dark">Uploading…</span>
|
||||
</div>
|
||||
<div class="w-full h-2 rounded-full bg-line dark:bg-line-dark overflow-hidden">
|
||||
<div id="upload-progress-bar" class="h-full rounded-full bg-accent dark:bg-accent-dark transition-all duration-150" style="width: 0%"></div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div id="upload-result" class="mt-3"></div>
|
||||
</div>
|
||||
|
||||
<!-- Uploaded files — select and delete (project-owner follow-up
|
||||
request). Independent of the date-range picker above: this lists
|
||||
uploads by when they arrived, not by which log dates they cover
|
||||
(a single file can span many dates). Selection is intentionally
|
||||
scoped to the currently-rendered page/sort/search view — it
|
||||
doesn't try to persist across a Grid.js re-render, which keeps
|
||||
"what's selected" always visually honest at the cost of losing
|
||||
selection if you page/sort/search mid-selection. -->
|
||||
<div class="mb-6 border border-line dark:border-line-dark rounded-xl bg-surface dark:bg-surface-dark overflow-hidden">
|
||||
<div class="flex items-center justify-between p-4 pb-3">
|
||||
<h2 class="font-display font-semibold text-sm">Uploaded files</h2>
|
||||
<button type="button" id="delete-selected-btn" disabled
|
||||
class="text-sm font-medium rounded-md px-3 py-1.5 border border-danger/40 dark:border-danger-dark/40 text-danger dark:text-danger-dark opacity-40 cursor-not-allowed disabled:opacity-40 enabled:opacity-100 enabled:hover:bg-danger/10 dark:enabled:hover:bg-danger-dark/10 transition-opacity">
|
||||
Delete selected
|
||||
</button>
|
||||
</div>
|
||||
<div id="uploaded-files-grid" class="px-4 pb-4" data-endpoint="{{ url_for('uploads.list_uploads') }}"></div>
|
||||
</div>
|
||||
|
||||
<!-- KPI readout — the one deliberate signature treatment: values set in
|
||||
tabular mono, like numbers straight off the log line, with a thin
|
||||
top rule that switches color when a metric needs attention
|
||||
(structure carries information, not just decoration). -->
|
||||
<div id="kpi-cards" class="grid grid-cols-2 sm:grid-cols-3 gap-3 mb-6" data-endpoint="{{ url_for('overview.api_kpis') }}"></div>
|
||||
|
||||
<div class="grid grid-cols-1 lg:grid-cols-2 gap-4 mb-6">
|
||||
<div class="border border-line dark:border-line-dark rounded-xl p-4 bg-surface dark:bg-surface-dark h-72">
|
||||
<h3 class="text-xs font-medium uppercase tracking-wide text-muted dark:text-muted-dark mb-2">Traffic over time</h3>
|
||||
<div class="h-56"><canvas id="traffic-chart" data-endpoint="{{ url_for('overview.api_traffic_chart') }}"></canvas></div>
|
||||
</div>
|
||||
<div class="border border-line dark:border-line-dark rounded-xl p-4 bg-surface dark:bg-surface-dark h-72">
|
||||
<h3 class="text-xs font-medium uppercase tracking-wide text-muted dark:text-muted-dark mb-2">Status codes</h3>
|
||||
<div class="h-56"><canvas id="status-code-chart" data-endpoint="{{ url_for('overview.api_status_codes') }}"></canvas></div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="grid grid-cols-1 lg:grid-cols-2 gap-4">
|
||||
<div>
|
||||
<h3 class="text-xs font-medium uppercase tracking-wide text-muted dark:text-muted-dark mb-2">Top URLs</h3>
|
||||
<div id="top-urls-grid" data-endpoint="{{ url_for('overview.api_top_urls') }}"></div>
|
||||
</div>
|
||||
<div>
|
||||
<h3 class="text-xs font-medium uppercase tracking-wide text-muted dark:text-muted-dark mb-2">Top referrers</h3>
|
||||
<div id="top-referrers-grid" data-endpoint="{{ url_for('overview.api_top_referrers') }}"></div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="grid grid-cols-1 lg:grid-cols-2 gap-4 mt-6">
|
||||
<div class="border border-line dark:border-line-dark rounded-xl p-4 bg-surface dark:bg-surface-dark h-64">
|
||||
<h3 class="text-xs font-medium uppercase tracking-wide text-muted dark:text-muted-dark mb-2">Browsers</h3>
|
||||
<div class="h-48"><canvas id="browser-breakdown-chart" data-endpoint="{{ url_for('overview.api_browser_breakdown') }}"></canvas></div>
|
||||
</div>
|
||||
<div class="border border-line dark:border-line-dark rounded-xl p-4 bg-surface dark:bg-surface-dark h-64">
|
||||
<h3 class="text-xs font-medium uppercase tracking-wide text-muted dark:text-muted-dark mb-2">Operating systems</h3>
|
||||
<div class="h-48"><canvas id="os-breakdown-chart"></canvas></div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<script>
|
||||
(function initOverviewWidgets() {
|
||||
const root = document.getElementById('overview-content');
|
||||
const from = root.dataset.from, to = root.dataset.to;
|
||||
const withRange = (url) => `${url}?from=${from}&to=${to}`;
|
||||
|
||||
window.initUploadForm('upload-form');
|
||||
initUploadedFilesGrid();
|
||||
|
||||
fetch(withRange(document.getElementById('kpi-cards').dataset.endpoint))
|
||||
.then((r) => r.json())
|
||||
.then(({ data }) => renderKpiCards(data));
|
||||
|
||||
const trafficEl = document.getElementById('traffic-chart');
|
||||
fetch(withRange(trafficEl.dataset.endpoint))
|
||||
.then((r) => r.json())
|
||||
.then(({ data }) => window.initChart('traffic-chart', {
|
||||
type: 'line',
|
||||
data: {
|
||||
labels: data.series.map((p) => p.t),
|
||||
datasets: [{
|
||||
label: 'Requests', data: data.series.map((p) => p.count), tension: 0.3,
|
||||
borderColor: '#0E7C86', backgroundColor: '#0E7C8622', fill: true,
|
||||
}],
|
||||
},
|
||||
}));
|
||||
|
||||
const statusEl = document.getElementById('status-code-chart');
|
||||
fetch(withRange(statusEl.dataset.endpoint))
|
||||
.then((r) => r.json())
|
||||
.then(({ data }) => window.initChart('status-code-chart', {
|
||||
type: 'doughnut',
|
||||
data: { labels: Object.keys(data), datasets: [{ data: Object.values(data), backgroundColor: window.KAVOSH_CHART_PALETTE }] },
|
||||
}));
|
||||
|
||||
window.initGrid(
|
||||
'top-urls-grid',
|
||||
document.getElementById('top-urls-grid').dataset.endpoint,
|
||||
[{ name: 'Path' }, { name: 'Hits' }, { name: 'Avg Size (B)' }, { name: 'Error Rate %' }],
|
||||
{ from, to },
|
||||
);
|
||||
|
||||
window.initGrid(
|
||||
'top-referrers-grid',
|
||||
document.getElementById('top-referrers-grid').dataset.endpoint,
|
||||
[{ name: 'Referrer Domain' }, { name: 'Hits' }],
|
||||
{ from, to },
|
||||
);
|
||||
|
||||
const browserEl = document.getElementById('browser-breakdown-chart');
|
||||
fetch(withRange(browserEl.dataset.endpoint))
|
||||
.then((r) => r.json())
|
||||
.then(({ data }) => {
|
||||
window.initChart('browser-breakdown-chart', {
|
||||
type: 'bar',
|
||||
data: { labels: data.by_browser.map((r) => r.name), datasets: [{ label: 'Browser', data: data.by_browser.map((r) => r.count), backgroundColor: '#0E7C86' }] },
|
||||
});
|
||||
window.initChart('os-breakdown-chart', {
|
||||
type: 'bar',
|
||||
data: { labels: data.by_os.map((r) => r.name), datasets: [{ label: 'OS', data: data.by_os.map((r) => r.count), backgroundColor: '#B8860B' }] },
|
||||
});
|
||||
});
|
||||
|
||||
function renderKpiCards(kpi) {
|
||||
const errorTone = kpi.error_rate_pct >= 5 ? 'danger' : kpi.error_rate_pct >= 1 ? 'warn' : 'ok';
|
||||
const cards = [
|
||||
['Total requests', kpi.total_requests.toLocaleString(), 'accent'],
|
||||
['Unique IPs (approx.)', kpi.unique_ips.toLocaleString(), 'accent'],
|
||||
['Bandwidth', `${(kpi.total_bandwidth_bytes / 1e6).toFixed(1)} MB`, 'accent'],
|
||||
['Avg response size', `${kpi.avg_response_size_bytes} B`, 'accent'],
|
||||
['Error rate', `${kpi.error_rate_pct}%`, errorTone],
|
||||
['Avg requests / day', kpi.avg_requests_per_day.toLocaleString(), 'accent'],
|
||||
];
|
||||
const toneBorder = {
|
||||
accent: 'border-t-accent dark:border-t-accent-dark',
|
||||
ok: 'border-t-ok dark:border-t-ok-dark',
|
||||
warn: 'border-t-warn dark:border-t-warn-dark',
|
||||
danger: 'border-t-danger dark:border-t-danger-dark',
|
||||
};
|
||||
const el = document.getElementById('kpi-cards');
|
||||
el.innerHTML = cards.map(([label, value, tone]) => `
|
||||
<div class="bg-surface dark:bg-surface-dark rounded-lg border border-line dark:border-line-dark border-t-2 ${toneBorder[tone]} p-3">
|
||||
<p class="text-xs text-muted dark:text-muted-dark uppercase tracking-wide">${label}</p>
|
||||
<p class="font-data text-2xl font-medium mt-0.5">${value}</p>
|
||||
</div>`).join('');
|
||||
if (kpi.peak_day) {
|
||||
el.innerHTML += `
|
||||
<div class="bg-surface dark:bg-surface-dark rounded-lg border border-line dark:border-line-dark border-t-2 border-t-accent dark:border-t-accent-dark p-3 col-span-2 sm:col-span-3">
|
||||
<p class="text-xs text-muted dark:text-muted-dark uppercase tracking-wide">Peak day</p>
|
||||
<p class="font-data text-xl font-medium mt-0.5">${kpi.peak_day.date} <span class="text-muted dark:text-muted-dark text-sm">— ${kpi.peak_day.count.toLocaleString()} requests</span></p>
|
||||
</div>`;
|
||||
}
|
||||
}
|
||||
|
||||
function initUploadedFilesGrid() {
|
||||
const container = document.getElementById('uploaded-files-grid');
|
||||
const deleteBtn = document.getElementById('delete-selected-btn');
|
||||
const selectedIds = new Set();
|
||||
|
||||
function updateDeleteButton() {
|
||||
deleteBtn.disabled = selectedIds.size === 0;
|
||||
deleteBtn.textContent = selectedIds.size > 0 ? `Delete selected (${selectedIds.size})` : 'Delete selected';
|
||||
}
|
||||
|
||||
// Selection is intentionally NOT restored across a Grid.js
|
||||
// re-render (page/sort/search) — simpler and always visually
|
||||
// honest, at the cost of losing selection if you page away
|
||||
// mid-selection. See the comment above the HTML for this panel.
|
||||
container.addEventListener('change', (e) => {
|
||||
if (!e.target.matches('.file-select-checkbox')) return;
|
||||
const id = parseInt(e.target.value, 10);
|
||||
if (e.target.checked) selectedIds.add(id); else selectedIds.delete(id);
|
||||
updateDeleteButton();
|
||||
});
|
||||
|
||||
const columns = [
|
||||
{
|
||||
name: '',
|
||||
formatter: (cell, row) => {
|
||||
const rawStatus = row.cells[6].data;
|
||||
const disabled = rawStatus === 'processing';
|
||||
return window.gridHtml(
|
||||
`<input type="checkbox" class="file-select-checkbox w-4 h-4 rounded border-line dark:border-line-dark text-accent focus:ring-accent cursor-pointer disabled:cursor-not-allowed disabled:opacity-40"
|
||||
value="${cell}" ${disabled ? 'disabled title="Still analyzing — wait for it to finish"' : ''}>`
|
||||
);
|
||||
},
|
||||
},
|
||||
{ name: 'Filename' },
|
||||
{ name: 'Server' },
|
||||
{ name: 'Status' },
|
||||
{ name: 'Uploaded' },
|
||||
{ name: 'Size' },
|
||||
{ name: 'Raw Status', hidden: true },
|
||||
];
|
||||
|
||||
const filesGrid = window.initGrid(
|
||||
'uploaded-files-grid',
|
||||
container.dataset.endpoint,
|
||||
columns,
|
||||
{},
|
||||
);
|
||||
|
||||
deleteBtn.addEventListener('click', () => {
|
||||
if (selectedIds.size === 0) return;
|
||||
if (!confirm(`Delete ${selectedIds.size} file(s) and all data derived from them? This can't be undone.`)) return;
|
||||
|
||||
const token = document.querySelector('meta[name="csrf-token"]')?.content;
|
||||
fetch('{{ url_for("uploads.bulk_delete_uploads") }}', {
|
||||
method: 'DELETE',
|
||||
headers: { 'Content-Type': 'application/json', 'X-CSRFToken': token || '' },
|
||||
body: JSON.stringify({ ids: Array.from(selectedIds) }),
|
||||
})
|
||||
.then((r) => r.json())
|
||||
.then(({ data }) => {
|
||||
selectedIds.clear();
|
||||
updateDeleteButton();
|
||||
filesGrid.forceRender();
|
||||
if (data.skipped && data.skipped.length) {
|
||||
alert('Some files were skipped:\n' + data.skipped.map((s) => `#${s.id} — ${s.reason}`).join('\n'));
|
||||
}
|
||||
// Deleting a file can change rollups for any date it
|
||||
// touched — refresh the whole page's KPIs/charts via the
|
||||
// same mechanism the date-range picker itself uses.
|
||||
window.htmx.trigger(document.getElementById('date-range-form'), 'change');
|
||||
});
|
||||
});
|
||||
}
|
||||
})();
|
||||
</script>
|
||||
</div>
|
||||
@@ -0,0 +1,5 @@
|
||||
{% extends "base.html" %}
|
||||
{% block title %}Overview — Kavosh{% endblock %}
|
||||
{% block content %}
|
||||
{% include "overview/_content.html" %}
|
||||
{% endblock %}
|
||||
@@ -0,0 +1,13 @@
|
||||
from flask import Blueprint
|
||||
from flask_login import login_required
|
||||
|
||||
bp = Blueprint("security", __name__, template_folder="templates")
|
||||
|
||||
|
||||
@bp.before_request
|
||||
@login_required
|
||||
def require_login():
|
||||
pass
|
||||
|
||||
|
||||
from app.blueprints.security import routes # noqa: E402,F401 registers routes
|
||||
@@ -0,0 +1,154 @@
|
||||
"""Context-builder query functions for the Security tab (Chapter 10),
|
||||
with the IP investigation panel's traffic breakdown upgraded to full-
|
||||
traffic data (bounded per-IP rollup, added as an explicit follow-up to
|
||||
Chapter 10's scope gap).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from collections import defaultdict
|
||||
from datetime import date
|
||||
|
||||
from sqlalchemy import func
|
||||
|
||||
from app.extensions import db
|
||||
from app.models.blocklist_suggestion import BlocklistSuggestion
|
||||
from app.models.bot_hit import BotHit
|
||||
from app.models.ip_registry import IPRegistry
|
||||
from app.models.ip_traffic_stats import IpPathStatsDaily, IpStatusStatsDaily
|
||||
from app.models.suspicious_event import SuspiciousEvent
|
||||
from app.services.severity_scoring import SeverityInputs, compute_effective_severity
|
||||
from app.utils.dates import day_bounds
|
||||
|
||||
IP_HISTORY_EVENT_LIMIT = 50
|
||||
IP_HISTORY_PATH_LIMIT = 20
|
||||
|
||||
|
||||
def get_suspicious_events(
|
||||
from_date: date, to_date: date, severity: str | None, rule_type: str | None, page: int, per_page: int,
|
||||
) -> tuple[list[list], int]:
|
||||
"""suspicious_events is small/indexed/indefinitely-retained (Ch06) —
|
||||
same precedent as Ch09's bot_hits queries, so loading + escalating in
|
||||
Python doesn't violate Ch03 rule 6 (that targets raw per-request rows).
|
||||
"""
|
||||
start, end = day_bounds(from_date, to_date)
|
||||
query = db.session.query(SuspiciousEvent).filter(
|
||||
SuspiciousEvent.timestamp >= start, SuspiciousEvent.timestamp < end
|
||||
)
|
||||
if rule_type:
|
||||
query = query.filter(SuspiciousEvent.rule_matched.like(f"{rule_type}:%"))
|
||||
events = query.order_by(SuspiciousEvent.timestamp.desc()).all()
|
||||
|
||||
by_ip: dict[str, list[SuspiciousEvent]] = defaultdict(list)
|
||||
for e in events:
|
||||
by_ip[e.ip].append(e)
|
||||
|
||||
enriched = []
|
||||
for e in events:
|
||||
ip_events = by_ip[e.ip]
|
||||
timestamps = sorted(ev.timestamp for ev in ip_events)
|
||||
avg_interval = (
|
||||
(timestamps[-1] - timestamps[0]).total_seconds() / (len(timestamps) - 1)
|
||||
if len(timestamps) > 1 else None
|
||||
)
|
||||
effective = compute_effective_severity(
|
||||
SeverityInputs(base_severity=e.severity, ip_event_count=len(ip_events), avg_interval_seconds=avg_interval)
|
||||
)
|
||||
if severity and effective != severity:
|
||||
continue
|
||||
enriched.append([e.timestamp.isoformat(), e.ip, e.path, e.rule_matched, effective])
|
||||
|
||||
total = len(enriched)
|
||||
offset = (page - 1) * per_page
|
||||
return enriched[offset : offset + per_page], total
|
||||
|
||||
|
||||
def get_sensitive_path_summary(from_date: date, to_date: date) -> list[dict]:
|
||||
"""Grouped by request path; filtered to Ch07's sensitive_path rule
|
||||
category. One ranked list, not sub-grouped into config/admin/VCS —
|
||||
Ch07's dictionary has no such taxonomy to reuse.
|
||||
"""
|
||||
start, end = day_bounds(from_date, to_date)
|
||||
rows = (
|
||||
db.session.query(
|
||||
SuspiciousEvent.path,
|
||||
func.count().label("hit_count"),
|
||||
func.count(func.distinct(SuspiciousEvent.ip)).label("distinct_ip_count"),
|
||||
)
|
||||
.filter(
|
||||
SuspiciousEvent.timestamp >= start, SuspiciousEvent.timestamp < end,
|
||||
SuspiciousEvent.rule_matched.like("sensitive_path:%"),
|
||||
)
|
||||
.group_by(SuspiciousEvent.path)
|
||||
.order_by(func.count().desc())
|
||||
.all()
|
||||
)
|
||||
return [{"path": r.path, "hit_count": r.hit_count, "distinct_ip_count": r.distinct_ip_count} for r in rows]
|
||||
|
||||
|
||||
def get_ip_history(ip: str) -> dict | None:
|
||||
"""Pulled from ip_registry (identity + true total_requests), plus
|
||||
bot_hits/suspicious_events (flagged activity), plus the bounded
|
||||
per-IP traffic rollup (top_paths / status_code_distribution — true
|
||||
full-traffic breakdown, added as a follow-up to Ch10's original scope
|
||||
gap). Retention caveat: the per-IP rollup covers roughly the last 30
|
||||
days (see aggregator.py / flask cleanup).
|
||||
"""
|
||||
registry = db.session.get(IPRegistry, ip)
|
||||
if registry is None:
|
||||
return None
|
||||
|
||||
path_rows = (
|
||||
db.session.query(IpPathStatsDaily.path, func.sum(IpPathStatsDaily.count).label("count"))
|
||||
.filter(IpPathStatsDaily.ip == ip)
|
||||
.group_by(IpPathStatsDaily.path)
|
||||
.order_by(func.sum(IpPathStatsDaily.count).desc())
|
||||
.limit(IP_HISTORY_PATH_LIMIT)
|
||||
.all()
|
||||
)
|
||||
status_rows = (
|
||||
db.session.query(IpStatusStatsDaily.status_bucket, func.sum(IpStatusStatsDaily.count).label("count"))
|
||||
.filter(IpStatusStatsDaily.ip == ip)
|
||||
.group_by(IpStatusStatsDaily.status_bucket)
|
||||
.all()
|
||||
)
|
||||
|
||||
bot_rows = (
|
||||
db.session.query(BotHit).filter(BotHit.ip == ip)
|
||||
.order_by(BotHit.timestamp.desc()).limit(IP_HISTORY_EVENT_LIMIT).all()
|
||||
)
|
||||
suspicious_rows = (
|
||||
db.session.query(SuspiciousEvent).filter(SuspiciousEvent.ip == ip)
|
||||
.order_by(SuspiciousEvent.timestamp.desc()).limit(IP_HISTORY_EVENT_LIMIT).all()
|
||||
)
|
||||
spoofed_bot_names = sorted({b.bot_name for b in bot_rows if not b.verified})
|
||||
|
||||
return {
|
||||
"ip": ip,
|
||||
"first_seen": registry.first_seen.isoformat(),
|
||||
"last_seen": registry.last_seen.isoformat(),
|
||||
"total_requests": registry.total_requests,
|
||||
"reputation_score": registry.reputation_score,
|
||||
"is_flagged": registry.is_flagged,
|
||||
"spoofed_bot_names": spoofed_bot_names,
|
||||
"top_paths": [[r.path, r.count] for r in path_rows],
|
||||
"status_code_distribution": {r.status_bucket: r.count for r in status_rows},
|
||||
"traffic_window_note": "Path/status breakdown reflects roughly the last 30 days (bounded retention).",
|
||||
"recent_suspicious_events": [
|
||||
{"timestamp": s.timestamp.isoformat(), "path": s.path, "rule_matched": s.rule_matched, "severity": s.severity}
|
||||
for s in suspicious_rows
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
def format_blocklist(suggestions: list[BlocklistSuggestion], fmt: str) -> str:
|
||||
"""Ch10: '.htaccess Deny/iptables/fail2ban-style'. 'plain' (a bare IP
|
||||
list) is the most portable interpretation of "fail2ban-style input"
|
||||
without assuming a specific fail2ban jail configuration Ch10 doesn't
|
||||
specify.
|
||||
"""
|
||||
ips = [s.ip for s in suggestions]
|
||||
if fmt == "htaccess":
|
||||
return "".join(f"Deny from {ip}\n" for ip in ips)
|
||||
if fmt == "iptables":
|
||||
return "".join(f"iptables -A INPUT -s {ip} -j DROP\n" for ip in ips)
|
||||
return "".join(f"{ip}\n" for ip in ips)
|
||||
@@ -0,0 +1,71 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from flask import Response, jsonify, render_template, request
|
||||
|
||||
from app.blueprints.security import bp
|
||||
from app.blueprints.security.queries import (
|
||||
format_blocklist, get_ip_history, get_sensitive_path_summary, get_suspicious_events,
|
||||
)
|
||||
from app.extensions import db
|
||||
from app.models.blocklist_suggestion import BlocklistSuggestion
|
||||
from app.utils.dates import parse_date_range
|
||||
from app.utils.htmx import render_htmx_aware
|
||||
from app.utils.pagination import parse_pagination
|
||||
|
||||
|
||||
@bp.route("/security")
|
||||
def security():
|
||||
from_date, to_date = parse_date_range(request)
|
||||
severity = request.args.get("severity") or ""
|
||||
rule_type = request.args.get("rule_type") or ""
|
||||
return render_htmx_aware(
|
||||
request, full_template="security/index.html", partial_template="security/_content.html",
|
||||
from_date=from_date, to_date=to_date, severity=severity, rule_type=rule_type,
|
||||
)
|
||||
|
||||
|
||||
@bp.get("/api/security/events")
|
||||
def api_security_events():
|
||||
from_date, to_date = parse_date_range(request)
|
||||
page, per_page = parse_pagination(request)
|
||||
severity = request.args.get("severity") or None
|
||||
rule_type = request.args.get("rule_type") or None
|
||||
rows, total = get_suspicious_events(from_date, to_date, severity, rule_type, page, per_page)
|
||||
return jsonify(
|
||||
data={"rows": rows, "total": total},
|
||||
meta={"from": from_date.isoformat(), "to": to_date.isoformat(), "page": page, "per_page": per_page},
|
||||
)
|
||||
|
||||
|
||||
@bp.get("/api/security/sensitive-paths")
|
||||
def api_sensitive_paths():
|
||||
from_date, to_date = parse_date_range(request)
|
||||
return jsonify(data=get_sensitive_path_summary(from_date, to_date), meta={"from": from_date.isoformat(), "to": to_date.isoformat()})
|
||||
|
||||
|
||||
@bp.get("/api/security/ip/<ip>")
|
||||
def api_ip_history(ip: str):
|
||||
history = get_ip_history(ip)
|
||||
if history is None:
|
||||
return render_template("security/_ip_not_found.html", ip=ip), 404
|
||||
return render_template("security/_ip_history.html", ip_data=history)
|
||||
|
||||
|
||||
@bp.get("/api/security/export-blocklist")
|
||||
def api_export_blocklist():
|
||||
fmt = request.args.get("format", "plain")
|
||||
include_all = request.args.get("all", "false").lower() == "true"
|
||||
query = BlocklistSuggestion.query
|
||||
if not include_all:
|
||||
query = query.filter_by(exported=False)
|
||||
suggestions = query.order_by(BlocklistSuggestion.created_at).all()
|
||||
|
||||
body = format_blocklist(suggestions, fmt)
|
||||
for s in suggestions:
|
||||
s.exported = True
|
||||
db.session.commit()
|
||||
|
||||
return Response(
|
||||
body, mimetype="text/plain",
|
||||
headers={"Content-Disposition": "attachment; filename=kavosh-blocklist.txt"},
|
||||
)
|
||||
@@ -0,0 +1,125 @@
|
||||
<div id="security-content"
|
||||
hx-get="{{ url_for('security.security') }}"
|
||||
hx-trigger="change from:#security-filter-form"
|
||||
hx-include="#security-filter-form"
|
||||
hx-target="#security-content"
|
||||
hx-swap="outerHTML"
|
||||
data-from="{{ from_date.isoformat() }}"
|
||||
data-to="{{ to_date.isoformat() }}"
|
||||
data-severity="{{ severity }}"
|
||||
data-rule-type="{{ rule_type }}">
|
||||
|
||||
<div class="flex flex-wrap items-end justify-between gap-4 mb-6">
|
||||
<div>
|
||||
<h1 class="font-display font-bold text-xl">Suspicious Requests & IP History</h1>
|
||||
<p class="text-sm text-muted dark:text-muted-dark">Who's poking at your site, and how hard</p>
|
||||
</div>
|
||||
<form id="security-filter-form" class="flex flex-wrap gap-3 items-end">
|
||||
<label class="text-sm text-muted dark:text-muted-dark">From
|
||||
<input type="date" name="from" value="{{ from_date.isoformat() }}" class="block border border-line dark:border-line-dark rounded-md px-2 py-1 mt-1 bg-surface dark:bg-surface-dark text-ink dark:text-ink-dark font-data text-sm">
|
||||
</label>
|
||||
<label class="text-sm text-muted dark:text-muted-dark">To
|
||||
<input type="date" name="to" value="{{ to_date.isoformat() }}" class="block border border-line dark:border-line-dark rounded-md px-2 py-1 mt-1 bg-surface dark:bg-surface-dark text-ink dark:text-ink-dark font-data text-sm">
|
||||
</label>
|
||||
<label class="text-sm text-muted dark:text-muted-dark">Severity
|
||||
<select name="severity" class="block border border-line dark:border-line-dark rounded-md px-2 py-1 mt-1 bg-surface dark:bg-surface-dark text-ink dark:text-ink-dark text-sm">
|
||||
<option value="" {{ 'selected' if not severity }}>All</option>
|
||||
<option value="low" {{ 'selected' if severity == 'low' }}>Low</option>
|
||||
<option value="medium" {{ 'selected' if severity == 'medium' }}>Medium</option>
|
||||
<option value="high" {{ 'selected' if severity == 'high' }}>High</option>
|
||||
</select>
|
||||
</label>
|
||||
<label class="text-sm text-muted dark:text-muted-dark">Rule Type
|
||||
<select name="rule_type" class="block border border-line dark:border-line-dark rounded-md px-2 py-1 mt-1 bg-surface dark:bg-surface-dark text-ink dark:text-ink-dark text-sm">
|
||||
<option value="" {{ 'selected' if not rule_type }}>All</option>
|
||||
<option value="sensitive_path" {{ 'selected' if rule_type == 'sensitive_path' }}>Sensitive Path</option>
|
||||
<option value="injection" {{ 'selected' if rule_type == 'injection' }}>Injection</option>
|
||||
<option value="scanner_ua" {{ 'selected' if rule_type == 'scanner_ua' }}>Scanner UA</option>
|
||||
<option value="spoofed_bot" {{ 'selected' if rule_type == 'spoofed_bot' }}>Spoofed Bot</option>
|
||||
</select>
|
||||
</label>
|
||||
</form>
|
||||
</div>
|
||||
|
||||
<div class="mb-6">
|
||||
<h3 class="text-xs font-medium uppercase tracking-wide text-muted dark:text-muted-dark mb-2">Suspicious events</h3>
|
||||
<div id="suspicious-events-grid" data-endpoint="{{ url_for('security.api_security_events') }}"></div>
|
||||
</div>
|
||||
|
||||
<div class="mb-6">
|
||||
<h3 class="text-xs font-medium uppercase tracking-wide text-muted dark:text-muted-dark mb-2">Sensitive-path probes</h3>
|
||||
<div id="sensitive-paths-panel" class="border border-line dark:border-line-dark rounded-xl bg-surface dark:bg-surface-dark overflow-hidden"
|
||||
data-endpoint="{{ url_for('security.api_sensitive_paths') }}"></div>
|
||||
</div>
|
||||
|
||||
<div class="mb-6 border border-line dark:border-line-dark rounded-xl p-4 bg-surface dark:bg-surface-dark">
|
||||
<h3 class="font-display font-semibold text-sm mb-3">Export blocklist</h3>
|
||||
<div class="flex flex-wrap gap-2 items-center">
|
||||
<a href="{{ url_for('security.api_export_blocklist') }}"
|
||||
class="bg-accent dark:bg-accent-dark text-white dark:text-paper-dark rounded-md px-3 py-1.5 text-sm font-medium hover:opacity-90 transition-opacity">Download new (.txt)</a>
|
||||
<a href="{{ url_for('security.api_export_blocklist', all='true') }}"
|
||||
class="border border-line dark:border-line-dark rounded-md px-3 py-1.5 text-sm hover:bg-paper dark:hover:bg-paper-dark transition-colors">Re-export all</a>
|
||||
<select id="blocklist-format" class="border border-line dark:border-line-dark rounded-md px-2 py-1.5 text-sm bg-surface dark:bg-surface-dark text-ink dark:text-ink-dark" onchange="updateBlocklistLinks(this.value)">
|
||||
<option value="plain">Plain IP list</option>
|
||||
<option value="htaccess">.htaccess Deny</option>
|
||||
<option value="iptables">iptables</option>
|
||||
</select>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div id="ip-history-modal" class="fixed inset-0 bg-ink/40 dark:bg-ink-dark/60 items-center justify-center empty:hidden flex z-50"></div>
|
||||
|
||||
<script>
|
||||
(function initSecurityWidgets() {
|
||||
const root = document.getElementById('security-content');
|
||||
const from = root.dataset.from, to = root.dataset.to;
|
||||
const severity = root.dataset.severity, ruleType = root.dataset.ruleType;
|
||||
|
||||
window.initGrid(
|
||||
'suspicious-events-grid',
|
||||
document.getElementById('suspicious-events-grid').dataset.endpoint,
|
||||
[
|
||||
{ name: 'Timestamp' },
|
||||
{
|
||||
name: 'IP',
|
||||
formatter: (cell) => window.gridHtml(
|
||||
`<button class="text-accent dark:text-accent-dark underline" hx-get="/api/security/ip/${cell}" hx-target="#ip-history-modal" hx-swap="innerHTML">${cell}</button>`
|
||||
),
|
||||
},
|
||||
{ name: 'Path' }, { name: 'Rule Matched' },
|
||||
{
|
||||
name: 'Severity',
|
||||
formatter: (cell) => {
|
||||
const tone = { low: 'text-muted dark:text-muted-dark', medium: 'text-warn dark:text-warn-dark', high: 'text-danger dark:text-danger-dark' }[cell] || '';
|
||||
return window.gridHtml(`<span class="font-medium ${tone}">${cell}</span>`);
|
||||
},
|
||||
},
|
||||
],
|
||||
{ from, to, severity, rule_type: ruleType },
|
||||
);
|
||||
|
||||
const pathsEl = document.getElementById('sensitive-paths-panel');
|
||||
fetch(`${pathsEl.dataset.endpoint}?from=${from}&to=${to}`)
|
||||
.then((r) => r.json())
|
||||
.then(({ data }) => {
|
||||
pathsEl.innerHTML = data.length
|
||||
? `<table class="w-full text-sm font-data">
|
||||
<thead><tr class="text-left text-muted dark:text-muted-dark text-xs uppercase tracking-wide bg-surface-raised dark:bg-surface-raised-dark font-sans">
|
||||
<th class="px-3 py-2">Path</th><th class="px-3 py-2">Hits</th><th class="px-3 py-2">Distinct IPs</th>
|
||||
</tr></thead>
|
||||
<tbody class="divide-y divide-line dark:divide-line-dark">${
|
||||
data.map((r) => `<tr><td class="px-3 py-2">${r.path}</td><td class="px-3 py-2">${r.hit_count}</td><td class="px-3 py-2">${r.distinct_ip_count}</td></tr>`).join('')
|
||||
}</tbody></table>`
|
||||
: `<p class="text-sm text-muted dark:text-muted-dark p-4">No sensitive-path probes in range.</p>`;
|
||||
});
|
||||
|
||||
window.updateBlocklistLinks = (fmt) => {
|
||||
document.querySelectorAll('a[href*="export-blocklist"]').forEach((a) => {
|
||||
const url = new URL(a.href, window.location.origin);
|
||||
url.searchParams.set('format', fmt);
|
||||
a.href = url.toString();
|
||||
});
|
||||
};
|
||||
})();
|
||||
</script>
|
||||
</div>
|
||||
@@ -0,0 +1,26 @@
|
||||
<div class="bg-surface dark:bg-surface-dark rounded-xl p-6 max-w-lg w-full relative border border-line dark:border-line-dark">
|
||||
<button class="absolute top-3 right-3 text-muted dark:text-muted-dark hover:text-ink dark:hover:text-ink-dark" onclick="document.getElementById('ip-history-modal').innerHTML=''">
|
||||
<svg class="w-4 h-4"><use href="/static/dist/icons.svg#x"/></svg>
|
||||
</button>
|
||||
<h3 class="font-display font-semibold text-lg mb-3 font-data">{{ ip_data.ip }}</h3>
|
||||
<dl class="text-sm grid grid-cols-2 gap-y-1.5 mb-4 font-data">
|
||||
<dt class="text-muted dark:text-muted-dark font-sans">First seen</dt><dd>{{ ip_data.first_seen }}</dd>
|
||||
<dt class="text-muted dark:text-muted-dark font-sans">Last seen</dt><dd>{{ ip_data.last_seen }}</dd>
|
||||
<dt class="text-muted dark:text-muted-dark font-sans">Total requests</dt><dd>{{ ip_data.total_requests }}</dd>
|
||||
<dt class="text-muted dark:text-muted-dark font-sans">Reputation score</dt><dd>{{ ip_data.reputation_score }}</dd>
|
||||
<dt class="text-muted dark:text-muted-dark font-sans">Flagged</dt>
|
||||
<dd class="{{ 'text-danger dark:text-danger-dark font-medium' if ip_data.is_flagged else '' }}">{{ 'Yes' if ip_data.is_flagged else 'No' }}</dd>
|
||||
{% if ip_data.spoofed_bot_names %}
|
||||
<dt class="text-muted dark:text-muted-dark font-sans">Spoofed bot claims</dt><dd class="text-danger dark:text-danger-dark">{{ ip_data.spoofed_bot_names | join(', ') }}</dd>
|
||||
{% endif %}
|
||||
</dl>
|
||||
<p class="text-xs text-muted dark:text-muted-dark mb-3">{{ ip_data.traffic_window_note }}</p>
|
||||
<h4 class="font-medium text-sm mb-1">Top paths</h4>
|
||||
<ul class="text-sm mb-3 font-data text-ink dark:text-ink-dark space-y-0.5">{% for path, count in ip_data.top_paths %}<li>{{ path }} <span class="text-muted dark:text-muted-dark">— {{ count }}</span></li>{% endfor %}</ul>
|
||||
<h4 class="font-medium text-sm mb-1">Status codes</h4>
|
||||
<ul class="text-sm mb-3 font-data text-ink dark:text-ink-dark space-y-0.5">{% for code, count in ip_data.status_code_distribution.items() %}<li>{{ code }} <span class="text-muted dark:text-muted-dark">— {{ count }}</span></li>{% endfor %}</ul>
|
||||
{% if ip_data.recent_suspicious_events %}
|
||||
<h4 class="font-medium text-sm mb-1">Recent flagged events</h4>
|
||||
<ul class="text-sm font-data text-ink dark:text-ink-dark space-y-0.5">{% for e in ip_data.recent_suspicious_events %}<li>{{ e.timestamp }} — {{ e.path }} <span class="text-muted dark:text-muted-dark">({{ e.rule_matched }}, {{ e.severity }})</span></li>{% endfor %}</ul>
|
||||
{% endif %}
|
||||
</div>
|
||||
@@ -0,0 +1,4 @@
|
||||
<div class="bg-surface dark:bg-surface-dark rounded-xl p-6 max-w-sm w-full border border-line dark:border-line-dark">
|
||||
<p class="text-sm text-ink dark:text-ink-dark">No history found for <span class="font-data">{{ ip }}</span> — it hasn't been seen yet.</p>
|
||||
<button onclick="document.getElementById('ip-history-modal').innerHTML=''" class="mt-3 text-sm text-accent dark:text-accent-dark hover:underline">Close</button>
|
||||
</div>
|
||||
@@ -0,0 +1,5 @@
|
||||
{% extends "base.html" %}
|
||||
{% block title %}Security — Kavosh{% endblock %}
|
||||
{% block content %}
|
||||
{% include "security/_content.html" %}
|
||||
{% endblock %}
|
||||
@@ -0,0 +1,13 @@
|
||||
from flask import Blueprint
|
||||
from flask_login import login_required
|
||||
|
||||
bp = Blueprint("seo", __name__, template_folder="templates")
|
||||
|
||||
|
||||
@bp.before_request
|
||||
@login_required
|
||||
def require_login():
|
||||
pass
|
||||
|
||||
|
||||
from app.blueprints.seo import routes # noqa: E402,F401 registers routes
|
||||
@@ -0,0 +1,140 @@
|
||||
"""Context-builder query functions for the SEO tab (Chapter 09).
|
||||
|
||||
Per Ch09's own Output instruction, bot-related widgets query bot_hits
|
||||
directly (small, indefinitely-retained, indexed — Ch06) rather than a new
|
||||
rollup; only the human-side of the crawled-vs-visited comparison needs
|
||||
the new human_path_stats_daily table.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from collections import defaultdict
|
||||
from datetime import date
|
||||
|
||||
from sqlalchemy import case, func
|
||||
|
||||
from app.extensions import db
|
||||
from app.models.bot_hit import BotHit
|
||||
from app.models.human_path_stats import HumanPathStatsDaily
|
||||
from app.utils.dates import day_bounds
|
||||
|
||||
# ASSUMPTION (flagged): Ch09 doesn't define "major crawler" for the
|
||||
# crawl-frequency chart's series cap.
|
||||
TOP_N_BOTS_FOR_CHART = 6
|
||||
|
||||
|
||||
def get_bot_summary(from_date: date, to_date: date) -> list[dict]:
|
||||
start, end = day_bounds(from_date, to_date)
|
||||
rows = (
|
||||
db.session.query(
|
||||
BotHit.bot_name,
|
||||
func.count().label("hits"),
|
||||
func.sum(case((BotHit.verified.is_(True), 1), else_=0)).label("verified_hits"),
|
||||
func.max(BotHit.timestamp).label("last_seen"),
|
||||
)
|
||||
.filter(BotHit.timestamp >= start, BotHit.timestamp < end)
|
||||
.group_by(BotHit.bot_name)
|
||||
.order_by(func.count().desc())
|
||||
.all()
|
||||
)
|
||||
return [
|
||||
{
|
||||
"bot_name": r.bot_name,
|
||||
"hits": r.hits,
|
||||
"verified_pct": round(r.verified_hits / r.hits * 100, 1) if r.hits else 0.0,
|
||||
"last_seen": r.last_seen.isoformat() if r.last_seen else None,
|
||||
}
|
||||
for r in rows
|
||||
]
|
||||
|
||||
|
||||
def get_crawl_chart_data(from_date: date, to_date: date, bot_name: str | None) -> dict:
|
||||
start, end = day_bounds(from_date, to_date)
|
||||
base_filters = [BotHit.timestamp >= start, BotHit.timestamp < end]
|
||||
|
||||
if bot_name:
|
||||
allowed_bots = [bot_name]
|
||||
else:
|
||||
top_rows = (
|
||||
db.session.query(BotHit.bot_name, func.count().label("hits"))
|
||||
.filter(*base_filters)
|
||||
.group_by(BotHit.bot_name)
|
||||
.order_by(func.count().desc())
|
||||
.limit(TOP_N_BOTS_FOR_CHART)
|
||||
.all()
|
||||
)
|
||||
allowed_bots = [r.bot_name for r in top_rows]
|
||||
|
||||
if not allowed_bots:
|
||||
return {"series": []}
|
||||
|
||||
rows = (
|
||||
db.session.query(func.date(BotHit.timestamp).label("day"), BotHit.bot_name, func.count().label("count"))
|
||||
.filter(*base_filters, BotHit.bot_name.in_(allowed_bots))
|
||||
.group_by("day", BotHit.bot_name)
|
||||
.order_by("day")
|
||||
.all()
|
||||
)
|
||||
points_by_bot: dict[str, list[dict]] = defaultdict(list)
|
||||
for r in rows:
|
||||
points_by_bot[r.bot_name].append({"t": r.day, "count": r.count})
|
||||
|
||||
return {"series": [{"bot_name": b, "points": points_by_bot.get(b, [])} for b in allowed_bots]}
|
||||
|
||||
|
||||
def get_bot_status_codes(from_date: date, to_date: date) -> dict:
|
||||
start, end = day_bounds(from_date, to_date)
|
||||
bucket = case(
|
||||
(BotHit.status_code < 300, "2xx"),
|
||||
(BotHit.status_code < 400, "3xx"),
|
||||
(BotHit.status_code < 500, "4xx"),
|
||||
else_="5xx",
|
||||
)
|
||||
rows = (
|
||||
db.session.query(bucket.label("bucket"), func.count().label("count"))
|
||||
.filter(BotHit.timestamp >= start, BotHit.timestamp < end)
|
||||
.group_by("bucket")
|
||||
.all()
|
||||
)
|
||||
breakdown = {"2xx": 0, "3xx": 0, "4xx": 0, "5xx": 0}
|
||||
for r in rows:
|
||||
breakdown[r.bucket] = r.count
|
||||
|
||||
# Ch09 calls out 404 by name specifically, not just the 4xx bucket.
|
||||
not_found_404 = (
|
||||
db.session.query(func.count())
|
||||
.filter(BotHit.timestamp >= start, BotHit.timestamp < end, BotHit.status_code == 404)
|
||||
.scalar()
|
||||
)
|
||||
return {"breakdown": breakdown, "not_found_404": not_found_404 or 0}
|
||||
|
||||
|
||||
def get_crawled_vs_visited(from_date: date, to_date: date, page: int, per_page: int) -> tuple[list[list], int]:
|
||||
"""Diffed table: one row per path, bot hits vs. human hits."""
|
||||
start, end = day_bounds(from_date, to_date)
|
||||
bot_rows = (
|
||||
db.session.query(BotHit.path, func.count().label("hits"))
|
||||
.filter(BotHit.timestamp >= start, BotHit.timestamp < end)
|
||||
.group_by(BotHit.path)
|
||||
.all()
|
||||
)
|
||||
human_rows = (
|
||||
db.session.query(HumanPathStatsDaily.path, func.sum(HumanPathStatsDaily.count).label("hits"))
|
||||
.filter(HumanPathStatsDaily.date >= from_date, HumanPathStatsDaily.date <= to_date)
|
||||
.group_by(HumanPathStatsDaily.path)
|
||||
.all()
|
||||
)
|
||||
bot_counts = {r.path: r.hits for r in bot_rows}
|
||||
human_counts = {r.path: r.hits for r in human_rows}
|
||||
|
||||
combined = []
|
||||
for path in set(bot_counts) | set(human_counts):
|
||||
b, h = bot_counts.get(path, 0), human_counts.get(path, 0)
|
||||
total = b + h
|
||||
combined.append([path, b, h, round(b / total * 100, 1) if total else 0.0])
|
||||
|
||||
# ASSUMPTION (flagged): sorted by bot hits desc — Ch09 doesn't specify.
|
||||
combined.sort(key=lambda row: row[1], reverse=True)
|
||||
|
||||
total_count = len(combined)
|
||||
offset = (page - 1) * per_page
|
||||
return combined[offset : offset + per_page], total_count
|
||||
@@ -0,0 +1,56 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from flask import jsonify, request
|
||||
|
||||
from app.blueprints.seo import bp
|
||||
from app.blueprints.seo.queries import (
|
||||
get_bot_status_codes,
|
||||
get_bot_summary,
|
||||
get_crawl_chart_data,
|
||||
get_crawled_vs_visited,
|
||||
)
|
||||
from app.utils.dates import parse_date_range
|
||||
from app.utils.htmx import render_htmx_aware
|
||||
from app.utils.pagination import parse_pagination
|
||||
|
||||
|
||||
@bp.route("/seo")
|
||||
def seo():
|
||||
from_date, to_date = parse_date_range(request)
|
||||
return render_htmx_aware(
|
||||
request, full_template="seo/index.html", partial_template="seo/_content.html",
|
||||
from_date=from_date, to_date=to_date,
|
||||
)
|
||||
|
||||
|
||||
@bp.get("/api/seo/bot-summary")
|
||||
def api_bot_summary():
|
||||
from_date, to_date = parse_date_range(request)
|
||||
return jsonify(data=get_bot_summary(from_date, to_date), meta={"from": from_date.isoformat(), "to": to_date.isoformat()})
|
||||
|
||||
|
||||
@bp.get("/api/seo/crawl-chart-data")
|
||||
def api_crawl_chart_data():
|
||||
from_date, to_date = parse_date_range(request)
|
||||
bot_name = request.args.get("bot")
|
||||
return jsonify(
|
||||
data=get_crawl_chart_data(from_date, to_date, bot_name),
|
||||
meta={"from": from_date.isoformat(), "to": to_date.isoformat(), "bot": bot_name},
|
||||
)
|
||||
|
||||
|
||||
@bp.get("/api/seo/bot-status-codes")
|
||||
def api_bot_status_codes():
|
||||
from_date, to_date = parse_date_range(request)
|
||||
return jsonify(data=get_bot_status_codes(from_date, to_date), meta={"from": from_date.isoformat(), "to": to_date.isoformat()})
|
||||
|
||||
|
||||
@bp.get("/api/seo/crawled-vs-visited")
|
||||
def api_crawled_vs_visited():
|
||||
from_date, to_date = parse_date_range(request)
|
||||
page, per_page = parse_pagination(request)
|
||||
rows, total = get_crawled_vs_visited(from_date, to_date, page, per_page)
|
||||
return jsonify(
|
||||
data={"rows": rows, "total": total},
|
||||
meta={"from": from_date.isoformat(), "to": to_date.isoformat(), "page": page, "per_page": per_page},
|
||||
)
|
||||
@@ -0,0 +1,102 @@
|
||||
<div id="seo-content"
|
||||
hx-get="{{ url_for('seo.seo') }}"
|
||||
hx-trigger="change from:#seo-date-range-form"
|
||||
hx-include="#seo-date-range-form"
|
||||
hx-target="#seo-content"
|
||||
hx-swap="outerHTML"
|
||||
data-from="{{ from_date.isoformat() }}"
|
||||
data-to="{{ to_date.isoformat() }}">
|
||||
|
||||
<div class="flex flex-wrap items-end justify-between gap-4 mb-6">
|
||||
<div>
|
||||
<h1 class="font-display font-bold text-xl">SEO & Bot Behavior</h1>
|
||||
<p class="text-sm text-muted dark:text-muted-dark">How search engines are crawling your site</p>
|
||||
</div>
|
||||
<form id="seo-date-range-form" class="flex gap-3 items-end">
|
||||
<label class="text-sm text-muted dark:text-muted-dark">From
|
||||
<input type="date" name="from" value="{{ from_date.isoformat() }}"
|
||||
class="block border border-line dark:border-line-dark rounded-md px-2 py-1 mt-1 bg-surface dark:bg-surface-dark text-ink dark:text-ink-dark font-data text-sm">
|
||||
</label>
|
||||
<label class="text-sm text-muted dark:text-muted-dark">To
|
||||
<input type="date" name="to" value="{{ to_date.isoformat() }}"
|
||||
class="block border border-line dark:border-line-dark rounded-md px-2 py-1 mt-1 bg-surface dark:bg-surface-dark text-ink dark:text-ink-dark font-data text-sm">
|
||||
</label>
|
||||
</form>
|
||||
</div>
|
||||
|
||||
<div id="bot-summary-cards" class="grid grid-cols-2 sm:grid-cols-3 gap-3 mb-6"
|
||||
data-endpoint="{{ url_for('seo.api_bot_summary') }}"></div>
|
||||
|
||||
<div class="grid grid-cols-1 lg:grid-cols-2 gap-4 mb-6">
|
||||
<div class="border border-line dark:border-line-dark rounded-xl p-4 bg-surface dark:bg-surface-dark h-72">
|
||||
<h3 class="text-xs font-medium uppercase tracking-wide text-muted dark:text-muted-dark mb-2">Crawl frequency</h3>
|
||||
<div class="h-56"><canvas id="crawl-frequency-chart" data-endpoint="{{ url_for('seo.api_crawl_chart_data') }}"></canvas></div>
|
||||
</div>
|
||||
<div class="border border-line dark:border-line-dark rounded-xl p-4 bg-surface dark:bg-surface-dark h-72">
|
||||
<h3 class="text-xs font-medium uppercase tracking-wide text-muted dark:text-muted-dark mb-2">Status codes served to bots</h3>
|
||||
<div class="h-44"><canvas id="bot-status-codes-chart" data-endpoint="{{ url_for('seo.api_bot_status_codes') }}"></canvas></div>
|
||||
<p id="bot-404-note" class="text-xs text-muted dark:text-muted-dark mt-2"></p>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div>
|
||||
<h3 class="text-xs font-medium uppercase tracking-wide text-muted dark:text-muted-dark mb-2">Most-crawled vs. most-visited URLs</h3>
|
||||
<div id="crawled-vs-visited-grid" data-endpoint="{{ url_for('seo.api_crawled_vs_visited') }}"></div>
|
||||
</div>
|
||||
|
||||
<script>
|
||||
(function initSeoWidgets() {
|
||||
const root = document.getElementById('seo-content');
|
||||
const from = root.dataset.from, to = root.dataset.to;
|
||||
const withRange = (url) => `${url}?from=${from}&to=${to}`;
|
||||
|
||||
fetch(withRange(document.getElementById('bot-summary-cards').dataset.endpoint))
|
||||
.then((r) => r.json())
|
||||
.then(({ data }) => {
|
||||
document.getElementById('bot-summary-cards').innerHTML = data.length ? data.map((bot) => `
|
||||
<div class="bg-surface dark:bg-surface-dark rounded-lg border border-line dark:border-line-dark border-t-2 ${bot.verified_pct < 100 ? 'border-t-warn dark:border-t-warn-dark' : 'border-t-ok dark:border-t-ok-dark'} p-3">
|
||||
<p class="text-xs text-muted dark:text-muted-dark uppercase tracking-wide">${bot.bot_name}</p>
|
||||
<p class="font-data text-xl font-medium mt-0.5">${bot.hits.toLocaleString()} <span class="text-sm text-muted dark:text-muted-dark font-sans">hits</span></p>
|
||||
<p class="text-xs mt-1 ${bot.verified_pct < 100 ? 'text-warn dark:text-warn-dark' : 'text-muted dark:text-muted-dark'}">
|
||||
${bot.verified_pct}% verified${bot.verified_pct < 100 ? ' — some spoofed' : ''}
|
||||
</p>
|
||||
<p class="text-xs text-muted dark:text-muted-dark font-data mt-0.5">Last seen: ${bot.last_seen ?? '—'}</p>
|
||||
</div>`).join('') : `<p class="text-sm text-muted dark:text-muted-dark col-span-full">No bot activity in this range.</p>`;
|
||||
});
|
||||
|
||||
const crawlEl = document.getElementById('crawl-frequency-chart');
|
||||
fetch(withRange(crawlEl.dataset.endpoint))
|
||||
.then((r) => r.json())
|
||||
.then(({ data }) => window.initChart('crawl-frequency-chart', {
|
||||
type: 'line',
|
||||
data: {
|
||||
labels: [...new Set(data.series.flatMap((s) => s.points.map((p) => p.t)))].sort(),
|
||||
datasets: data.series.map((s, i) => ({
|
||||
label: s.bot_name,
|
||||
data: s.points.map((p) => ({ x: p.t, y: p.count })),
|
||||
tension: 0.3,
|
||||
borderColor: window.KAVOSH_CHART_PALETTE[i % window.KAVOSH_CHART_PALETTE.length],
|
||||
})),
|
||||
},
|
||||
}));
|
||||
|
||||
const statusEl = document.getElementById('bot-status-codes-chart');
|
||||
fetch(withRange(statusEl.dataset.endpoint))
|
||||
.then((r) => r.json())
|
||||
.then(({ data }) => {
|
||||
window.initChart('bot-status-codes-chart', {
|
||||
type: 'bar',
|
||||
data: { labels: Object.keys(data.breakdown), datasets: [{ label: 'Bot Requests', data: Object.values(data.breakdown), backgroundColor: '#0E7C86' }] },
|
||||
});
|
||||
document.getElementById('bot-404-note').textContent = `${data.not_found_404} 404s served to bots in range — wasted crawl budget.`;
|
||||
});
|
||||
|
||||
window.initGrid(
|
||||
'crawled-vs-visited-grid',
|
||||
document.getElementById('crawled-vs-visited-grid').dataset.endpoint,
|
||||
[{ name: 'Path' }, { name: 'Bot Hits' }, { name: 'Human Hits' }, { name: 'Bot Share %' }],
|
||||
{ from, to },
|
||||
);
|
||||
})();
|
||||
</script>
|
||||
</div>
|
||||
@@ -0,0 +1,5 @@
|
||||
{% extends "base.html" %}
|
||||
{% block title %}SEO & Bots — Kavosh{% endblock %}
|
||||
{% block content %}
|
||||
{% include "seo/_content.html" %}
|
||||
{% endblock %}
|
||||
@@ -0,0 +1,15 @@
|
||||
from flask import Blueprint
|
||||
from flask_login import login_required
|
||||
|
||||
bp = Blueprint("uploads", __name__, template_folder="templates")
|
||||
|
||||
|
||||
@bp.before_request
|
||||
@login_required
|
||||
def require_login():
|
||||
"""Ch01: dashboard reachable from one AUTHENTICATED shell; Ch12:
|
||||
single-admin login. All routes on this blueprint require a session."""
|
||||
pass
|
||||
|
||||
|
||||
from app.blueprints.uploads import routes # noqa: E402,F401 registers routes
|
||||
@@ -0,0 +1,51 @@
|
||||
"""Query/formatting helpers for the "Uploaded files" list (project-owner
|
||||
follow-up request). Kept separate from routes.py to match this project's
|
||||
established per-blueprint queries.py convention (Ch08/09/10).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from app.models.log_file import LogFile
|
||||
|
||||
|
||||
def get_uploaded_files(page: int, per_page: int) -> tuple[list[list], int]:
|
||||
"""Newest first, independent of the dashboard date-range picker —
|
||||
this lists uploads by when they arrived, not by which log dates they
|
||||
contain (a single file can span many dates).
|
||||
"""
|
||||
query = LogFile.query.order_by(LogFile.uploaded_at.desc())
|
||||
total = query.count()
|
||||
rows = query.offset((page - 1) * per_page).limit(per_page).all()
|
||||
|
||||
result = []
|
||||
for lf in rows:
|
||||
result.append([
|
||||
lf.id,
|
||||
lf.filename,
|
||||
lf.server_type,
|
||||
_status_display(lf),
|
||||
lf.uploaded_at.strftime("%Y-%m-%d %H:%M"),
|
||||
_human_size(lf.size_bytes),
|
||||
lf.status, # raw status (hidden column) — lets the client disable
|
||||
# the select checkbox for files still "processing"
|
||||
])
|
||||
return result, total
|
||||
|
||||
|
||||
def _status_display(lf: LogFile) -> str:
|
||||
if lf.status == "processing":
|
||||
if lf.total_lines:
|
||||
pct = round(lf.processed_lines / lf.total_lines * 100)
|
||||
return f"processing ({pct}%)"
|
||||
return "processing"
|
||||
if lf.status == "error":
|
||||
return f"error: {lf.error_message}" if lf.error_message else "error"
|
||||
return lf.status
|
||||
|
||||
|
||||
def _human_size(num_bytes: int) -> str:
|
||||
size = float(num_bytes)
|
||||
for unit in ("B", "KB", "MB", "GB"):
|
||||
if size < 1024 or unit == "GB":
|
||||
return f"{size:.0f} {unit}" if unit == "B" else f"{size:.1f} {unit}"
|
||||
size /= 1024
|
||||
return f"{size:.1f} GB"
|
||||
@@ -0,0 +1,188 @@
|
||||
"""Upload endpoint (Chapter 04): validated, streamed-to-disk save.
|
||||
|
||||
Parsing happens in a background thread triggered right after this request
|
||||
completes (app/services/background.py) — never synchronously inside this
|
||||
request (Chapter 03, rule 5 still holds: the response returns immediately
|
||||
regardless of file size).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import uuid
|
||||
from pathlib import Path
|
||||
|
||||
from flask import current_app, jsonify, render_template, request
|
||||
from werkzeug.utils import secure_filename
|
||||
|
||||
from app.blueprints.uploads import bp
|
||||
from app.blueprints.uploads.queries import get_uploaded_files
|
||||
from app.extensions import db, limiter
|
||||
from app.models.log_file import LogFile
|
||||
from app.services.background import trigger_processing
|
||||
from app.services.file_deletion import delete_log_files
|
||||
from app.utils.pagination import parse_pagination
|
||||
from app.utils.upload_paths import upload_path_for
|
||||
|
||||
_ALLOWED_EXTENSIONS = {".log", ".txt", ".gz"}
|
||||
_CHUNK_SIZE = 64 * 1024 # 64 KB per read — never buffer the whole upload
|
||||
_GZIP_MAGIC = b"\x1f\x8b"
|
||||
|
||||
|
||||
class UploadRejected(Exception):
|
||||
"""Raised when an upload fails extension/content validation."""
|
||||
|
||||
|
||||
def _validate_extension(filename: str) -> str:
|
||||
ext = Path(filename).suffix.lower()
|
||||
if ext not in _ALLOWED_EXTENSIONS:
|
||||
raise UploadRejected(f"Unsupported extension {ext!r}; allowed: {_ALLOWED_EXTENSIONS}")
|
||||
return ext
|
||||
|
||||
|
||||
def _sniff_content(first_chunk: bytes, ext: str) -> None:
|
||||
"""Light content sniff — don't just trust the client-supplied MIME type."""
|
||||
if ext == ".gz":
|
||||
if not first_chunk.startswith(_GZIP_MAGIC):
|
||||
raise UploadRejected("File has a .gz extension but isn't gzip-magic-prefixed.")
|
||||
return
|
||||
if b"\x00" in first_chunk:
|
||||
raise UploadRejected("File extension claims text but content looks binary.")
|
||||
|
||||
|
||||
def _stream_to_temp(file_storage, tmp_path: Path) -> tuple[int, str, bytes]:
|
||||
"""Stream the upload to disk in bounded chunks; return (size, sha256_hex, first_chunk).
|
||||
|
||||
Never calls file.read() on the whole stream (Chapter 03, rule 1).
|
||||
"""
|
||||
sha256 = hashlib.sha256()
|
||||
size = 0
|
||||
first_chunk: bytes | None = None
|
||||
with tmp_path.open("wb") as out:
|
||||
while True:
|
||||
chunk = file_storage.stream.read(_CHUNK_SIZE)
|
||||
if not chunk:
|
||||
break
|
||||
if first_chunk is None:
|
||||
first_chunk = chunk
|
||||
sha256.update(chunk)
|
||||
size += len(chunk)
|
||||
out.write(chunk)
|
||||
if first_chunk is None:
|
||||
raise UploadRejected("Uploaded file is empty.")
|
||||
return size, sha256.hexdigest(), first_chunk
|
||||
|
||||
|
||||
@bp.post("/uploads")
|
||||
@limiter.limit("20 per minute") # Chapter 12: rate limiting on /uploads at minimum
|
||||
def upload_log_file():
|
||||
"""Validate, stream, and register an uploaded access log (Ch04/Ch11)."""
|
||||
file_storage = request.files.get("logfile")
|
||||
if file_storage is None or not file_storage.filename:
|
||||
return render_template("uploads/_error.html", message="No file provided."), 400
|
||||
|
||||
upload_dir = Path(current_app.config["UPLOAD_DIR"])
|
||||
(upload_dir / "tmp").mkdir(parents=True, exist_ok=True)
|
||||
tmp_path = upload_dir / "tmp" / f"{uuid.uuid4().hex}.part"
|
||||
|
||||
try:
|
||||
ext = _validate_extension(file_storage.filename)
|
||||
size_bytes, checksum, first_chunk = _stream_to_temp(file_storage, tmp_path)
|
||||
_sniff_content(first_chunk, ext)
|
||||
|
||||
max_bytes = current_app.config["UPLOAD_MAX_SIZE_MB"] * 1024 * 1024
|
||||
if size_bytes > max_bytes:
|
||||
raise UploadRejected(f"File exceeds {current_app.config['UPLOAD_MAX_SIZE_MB']}MB limit.")
|
||||
except UploadRejected as exc:
|
||||
tmp_path.unlink(missing_ok=True)
|
||||
return render_template("uploads/_error.html", message=str(exc)), 400
|
||||
|
||||
existing = LogFile.query.filter_by(checksum=checksum).first()
|
||||
if existing is not None:
|
||||
tmp_path.unlink(missing_ok=True)
|
||||
return render_template("uploads/_duplicate.html", log_file=existing)
|
||||
|
||||
log_file = LogFile(
|
||||
filename=secure_filename(file_storage.filename),
|
||||
server_type=request.form.get("server_type", "apache"),
|
||||
format_string=request.form.get("format_string", ""),
|
||||
status="queued",
|
||||
size_bytes=size_bytes,
|
||||
checksum=checksum,
|
||||
)
|
||||
db.session.add(log_file)
|
||||
db.session.commit() # need the assigned id before the final rename
|
||||
|
||||
tmp_path.rename(upload_path_for(log_file))
|
||||
|
||||
# Chapter 12 simplification: no cron required — kick off processing
|
||||
# immediately in a background thread. The response below returns as
|
||||
# soon as the file is queued (Ch03 rule 5 still holds: this request
|
||||
# never blocks on parsing), while the thread runs independently.
|
||||
trigger_processing(current_app._get_current_object(), log_file.id)
|
||||
|
||||
return render_template("uploads/_queued.html", log_file=log_file)
|
||||
|
||||
|
||||
@bp.get("/api/uploads/<int:log_file_id>/status")
|
||||
def upload_status(log_file_id: int):
|
||||
"""Polled every 3s by the browser (hx-trigger) until done/error (Ch04)."""
|
||||
log_file = db.get_or_404(LogFile, log_file_id)
|
||||
|
||||
if request.headers.get("Accept") == "application/json":
|
||||
return jsonify(
|
||||
data={
|
||||
"id": log_file.id,
|
||||
"status": log_file.status,
|
||||
"processed_lines": log_file.processed_lines,
|
||||
"total_lines": log_file.total_lines,
|
||||
},
|
||||
meta={},
|
||||
)
|
||||
|
||||
template = {
|
||||
"done": "uploads/_status_done.html",
|
||||
"error": "uploads/_status_error.html",
|
||||
# BUG FIX: this key was missing, so "queued" fell through to the
|
||||
# "processing" fallback below — a file that hadn't been picked up
|
||||
# by `flask process-logs` yet displayed as "Processing X: 0 lines"
|
||||
# instead of "Queued — waiting for the next parse cycle", making a
|
||||
# cron job that simply hasn't run yet indistinguishable from one
|
||||
# that's actually hung mid-parse.
|
||||
"queued": "uploads/_queued.html",
|
||||
}.get(log_file.status, "uploads/_status_processing.html")
|
||||
return render_template(template, log_file=log_file)
|
||||
|
||||
|
||||
@bp.get("/api/uploads")
|
||||
def list_uploads():
|
||||
"""Uploaded-files list (project-owner follow-up request) — Grid.js-
|
||||
backed, same page/per_page convention as every other table (Ch11).
|
||||
Sorted by upload recency, independent of the dashboard date-range
|
||||
picker (a single file can span many log dates).
|
||||
"""
|
||||
page, per_page = parse_pagination(request)
|
||||
rows, total = get_uploaded_files(page, per_page)
|
||||
return jsonify(
|
||||
data={"rows": rows, "total": total},
|
||||
meta={"page": page, "per_page": per_page},
|
||||
)
|
||||
|
||||
|
||||
@bp.delete("/api/uploads")
|
||||
def bulk_delete_uploads():
|
||||
"""Delete one or more uploaded files and everything derived from them
|
||||
(log_entries/bot_hits/suspicious_events, the raw file on disk, and a
|
||||
correct rollup recompute for the affected dates — see
|
||||
app/services/file_deletion.py for why a rollup recompute is needed
|
||||
rather than a simple per-file delete).
|
||||
|
||||
Body: {"ids": [1, 2, 3]}. Files currently "processing" are skipped,
|
||||
not force-deleted, to avoid racing the background parse thread.
|
||||
"""
|
||||
body = request.get_json(silent=True) or {}
|
||||
ids = body.get("ids")
|
||||
if not isinstance(ids, list) or not ids or not all(isinstance(i, int) for i in ids):
|
||||
return jsonify(error={"code": "invalid_request", "message": "Expected {\"ids\": [int, ...]}."}), 400
|
||||
|
||||
result = delete_log_files(ids)
|
||||
return jsonify(data={"deleted": result.deleted, "skipped": result.skipped}, meta={})
|
||||
@@ -0,0 +1,4 @@
|
||||
<div class="border border-warn/30 dark:border-warn-dark/30 bg-warn/5 dark:bg-warn-dark/10 rounded-lg p-3 text-sm">
|
||||
<span class="font-data">{{ log_file.filename }}</span>
|
||||
<span class="text-muted dark:text-muted-dark">matches an already-uploaded file (status: {{ log_file.status }}); skipped re-upload.</span>
|
||||
</div>
|
||||
@@ -0,0 +1,3 @@
|
||||
<div class="border border-danger/30 dark:border-danger-dark/30 bg-danger/5 dark:bg-danger-dark/10 rounded-lg p-3 text-sm text-danger dark:text-danger-dark">
|
||||
{{ message }}
|
||||
</div>
|
||||
@@ -0,0 +1,10 @@
|
||||
<div id="upload-status-{{ log_file.id }}"
|
||||
hx-get="{{ url_for('uploads.upload_status', log_file_id=log_file.id) }}"
|
||||
hx-trigger="every 1s" hx-swap="outerHTML"
|
||||
class="border border-line dark:border-line-dark rounded-lg p-3 bg-paper dark:bg-paper-dark">
|
||||
<div class="flex items-center gap-2 text-sm">
|
||||
<svg class="w-4 h-4 text-muted dark:text-muted-dark animate-spin"><use href="/static/dist/icons.svg#loader-circle"/></svg>
|
||||
<span class="font-data">{{ log_file.filename }}</span>
|
||||
<span class="text-muted dark:text-muted-dark">— queued, starting shortly…</span>
|
||||
</div>
|
||||
</div>
|
||||
@@ -0,0 +1,7 @@
|
||||
<div id="upload-status-{{ log_file.id }}" class="border border-ok/30 dark:border-ok-dark/30 bg-ok/5 dark:bg-ok-dark/10 rounded-lg p-3">
|
||||
<div class="flex items-center gap-2 text-sm">
|
||||
<span class="w-2 h-2 rounded-full bg-ok dark:bg-ok-dark shrink-0"></span>
|
||||
<span class="font-data">{{ log_file.filename }}</span>
|
||||
<span class="text-ok dark:text-ok-dark">— done, {{ "{:,}".format(log_file.processed_lines) }} lines analyzed</span>
|
||||
</div>
|
||||
</div>
|
||||
@@ -0,0 +1,7 @@
|
||||
<div id="upload-status-{{ log_file.id }}" class="border border-danger/30 dark:border-danger-dark/30 bg-danger/5 dark:bg-danger-dark/10 rounded-lg p-3">
|
||||
<div class="flex items-center gap-2 text-sm">
|
||||
<span class="w-2 h-2 rounded-full bg-danger dark:bg-danger-dark shrink-0"></span>
|
||||
<span class="font-data">{{ log_file.filename }}</span>
|
||||
</div>
|
||||
<p class="text-danger dark:text-danger-dark text-xs mt-1">{{ log_file.error_message }}</p>
|
||||
</div>
|
||||
@@ -0,0 +1,19 @@
|
||||
{% set pct = ((log_file.processed_lines / log_file.total_lines) * 100) if log_file.total_lines else None %}
|
||||
<div id="upload-status-{{ log_file.id }}"
|
||||
hx-get="{{ url_for('uploads.upload_status', log_file_id=log_file.id) }}"
|
||||
hx-trigger="every 1s" hx-swap="outerHTML"
|
||||
class="border border-line dark:border-line-dark rounded-lg p-3 bg-paper dark:bg-paper-dark">
|
||||
<div class="flex items-center justify-between text-sm mb-2">
|
||||
<span class="font-data">{{ log_file.filename }}</span>
|
||||
<span class="font-data text-muted dark:text-muted-dark">
|
||||
{% if pct is not none %}{{ pct | round(0) | int }}%{% else %}analyzing…{% endif %}
|
||||
</span>
|
||||
</div>
|
||||
<div class="w-full h-2 rounded-full bg-line dark:bg-line-dark overflow-hidden">
|
||||
<div class="h-full rounded-full bg-accent dark:bg-accent-dark transition-all duration-500 ease-out"
|
||||
style="width: {{ pct | round(1) if pct is not none else 8 }}%"></div>
|
||||
</div>
|
||||
<p class="text-xs text-muted dark:text-muted-dark mt-1.5 font-data">
|
||||
{{ "{:,}".format(log_file.processed_lines) }}{% if log_file.total_lines %} / {{ "{:,}".format(log_file.total_lines) }}{% endif %} lines
|
||||
</p>
|
||||
</div>
|
||||
@@ -0,0 +1,82 @@
|
||||
"""Concrete, checkable translation of the cPanel account ceiling (Chapter 03).
|
||||
|
||||
Single source of truth for the account's literal resource limits. Other
|
||||
chapters should import BUDGET rather than re-hardcoding these numbers.
|
||||
validate_budget_config() is called from create_app() so an out-of-budget
|
||||
deployment fails loudly at startup instead of silently degrading under load
|
||||
— this is the concrete mechanism behind Chapter 03's "flag, don't silently
|
||||
accept" requirement.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
|
||||
from flask import Flask
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class AccountBudget:
|
||||
"""The hosting account's literal ceiling (Chapter 03).
|
||||
|
||||
iops/io_throughput are informational only here — Python config can't
|
||||
enforce them directly; they constrain how Ch06/07 batch reads/writes.
|
||||
"""
|
||||
|
||||
cpu_cores: int = 4
|
||||
max_entry_processes: int = 60
|
||||
memory_mb: int = 2048
|
||||
iops: int = 1024
|
||||
io_throughput_mb_s: int = 16
|
||||
max_total_processes: int = 150
|
||||
max_db_connections: int = 150
|
||||
|
||||
# Derived engineering targets from Chapter 03's budget table.
|
||||
db_pool_size_min: int = 2
|
||||
db_pool_size_max: int = 5
|
||||
|
||||
|
||||
BUDGET = AccountBudget()
|
||||
|
||||
_DISALLOWED_CACHE_BACKENDS = {"redis", "rediscache", "memcached", "memcachedcache"}
|
||||
|
||||
|
||||
def validate_budget_config(app: Flask) -> None:
|
||||
"""Raise at startup if config violates a Chapter 03 rule.
|
||||
|
||||
Cheap checks only (string/int comparisons) since this runs on every
|
||||
process boot — Passenger may recycle processes frequently (factor 9).
|
||||
"""
|
||||
# Validated as its own config key, not read out of
|
||||
# SQLALCHEMY_ENGINE_OPTIONS — that dict is empty for SQLite (its pool
|
||||
# classes reject pool_size/pool_recycle outright; see app/config.py's
|
||||
# _engine_options_for), so the *intended* setting must be checked
|
||||
# independently of whether the active engine actually consumes it.
|
||||
pool_size = app.config.get("DB_POOL_SIZE")
|
||||
if pool_size is None or not (BUDGET.db_pool_size_min <= pool_size <= BUDGET.db_pool_size_max):
|
||||
raise RuntimeError(
|
||||
f"DB_POOL_SIZE={pool_size} is outside the Chapter 03 budget "
|
||||
f"({BUDGET.db_pool_size_min}-{BUDGET.db_pool_size_max} per process; "
|
||||
f"{BUDGET.max_db_connections} total connections are shared across "
|
||||
f"up to {BUDGET.max_entry_processes} entry processes)."
|
||||
)
|
||||
|
||||
cache_type = str(app.config.get("CACHE_TYPE", "")).lower()
|
||||
if any(name in cache_type for name in _DISALLOWED_CACHE_BACKENDS):
|
||||
raise RuntimeError(
|
||||
f"CACHE_TYPE={app.config.get('CACHE_TYPE')!r} assumes a backend "
|
||||
"(Redis/Memcached) Chapter 03 says not to assume is available. "
|
||||
"Use FileSystemCache or a dashboard_cache DB table (Ch06)."
|
||||
)
|
||||
|
||||
batch_size = app.config.get("PARSE_BATCH_SIZE")
|
||||
if not batch_size or batch_size <= 0:
|
||||
raise RuntimeError(
|
||||
"PARSE_BATCH_SIZE must be a positive integer — unbounded/whole-file "
|
||||
"parsing per invocation violates the Chapter 03 memory budget."
|
||||
)
|
||||
|
||||
max_upload_mb = app.config.get("UPLOAD_MAX_SIZE_MB")
|
||||
if not max_upload_mb or max_upload_mb <= 0:
|
||||
raise RuntimeError(
|
||||
"UPLOAD_MAX_SIZE_MB must be a positive integer to bound disk/IOPS per upload."
|
||||
)
|
||||
+163
@@ -0,0 +1,163 @@
|
||||
"""Flask CLI admin commands (factor 12).
|
||||
|
||||
process-logs is now OPTIONAL (Chapter 12 simplification, per project
|
||||
owner request): file processing is triggered automatically in-app right
|
||||
after upload (app/services/background.py), so cron is no longer required.
|
||||
This command still exists for anyone who'd rather run it manually or via
|
||||
cron — it shares the exact same processing code (app/services/
|
||||
log_processor.py) as the automatic background trigger, so both paths
|
||||
behave identically. cleanup (Chapter 06/12) enforces the retention
|
||||
policy. create-admin (Chapter 12) bootstraps the single admin account.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import date, datetime, timedelta
|
||||
|
||||
import click
|
||||
from flask import Flask, current_app
|
||||
|
||||
from app.extensions import db
|
||||
from app.models.ip_traffic_stats import IpPathStatsDaily, IpStatusStatsDaily
|
||||
from app.models.log_entry import LogEntry
|
||||
from app.models.log_file import LogFile
|
||||
from app.models.request_stats import RequestStatsHourly
|
||||
from app.models.user import User
|
||||
from app.services import aggregator
|
||||
from app.services.log_processor import process_one_batch
|
||||
from app.utils.upload_paths import upload_path_for
|
||||
|
||||
# Chapter 06 retention policy — the constants `flask cleanup` enforces.
|
||||
LOG_ENTRIES_RETENTION_DAYS = 30
|
||||
HOURLY_STATS_RETENTION_DAYS = 90
|
||||
IP_TRAFFIC_STATS_RETENTION_DAYS = 30
|
||||
|
||||
|
||||
def register_commands(app: Flask) -> None:
|
||||
app.cli.add_command(process_logs)
|
||||
app.cli.add_command(rollup)
|
||||
app.cli.add_command(cleanup)
|
||||
app.cli.add_command(create_admin)
|
||||
|
||||
|
||||
@click.command("process-logs")
|
||||
@click.option("--batch-size", default=None, type=int, help="Override PARSE_BATCH_SIZE.")
|
||||
def process_logs(batch_size: int | None) -> None:
|
||||
"""Optional manual/cron fallback — processing now also runs
|
||||
automatically in-app after upload. Parses queued/processing
|
||||
log_files in bounded, checkpointed batches; one batch per file per
|
||||
invocation, same as before.
|
||||
"""
|
||||
batch = batch_size or current_app.config["PARSE_BATCH_SIZE"]
|
||||
pending = LogFile.query.filter(LogFile.status.in_(["queued", "processing"])).all()
|
||||
for log_file in pending:
|
||||
process_one_batch(log_file, batch)
|
||||
|
||||
|
||||
@click.command("rollup")
|
||||
@click.option("--from", "from_date", required=True, help="ISO date, e.g. 2026-07-01")
|
||||
@click.option("--to", "to_date", required=True, help="ISO date, e.g. 2026-07-26")
|
||||
def rollup(from_date: str, to_date: str) -> None:
|
||||
"""Manual rollup recompute for an explicit range (e.g. after a backfill).
|
||||
|
||||
Requires an explicit range — no "all time" default, mirroring Ch03 rule 7
|
||||
even for an admin command, to avoid an unbounded scan on constrained hosting.
|
||||
"""
|
||||
start = date.fromisoformat(from_date)
|
||||
end = date.fromisoformat(to_date)
|
||||
aggregator.compute_rollups_for_range(start, end)
|
||||
click.echo(f"Rolled up {start} .. {end}")
|
||||
|
||||
|
||||
@click.command("cleanup")
|
||||
def cleanup() -> None:
|
||||
"""Enforce the Chapter 06 retention policy (Chapter 12).
|
||||
|
||||
Previously a stub through every earlier chapter — implemented here as
|
||||
Chapter 12's non-functional/deployment concern. Cron-invoked (e.g.
|
||||
daily, off-peak), never a long-running daemon.
|
||||
"""
|
||||
now = datetime.utcnow()
|
||||
deleted_entries = _delete_old_log_entries(now)
|
||||
deleted_hourly = _collapse_old_hourly_stats(now)
|
||||
deleted_ip_stats = _delete_old_ip_traffic_stats(now)
|
||||
archived_files = _delete_parsed_upload_files()
|
||||
click.echo(
|
||||
f"Cleanup complete: {deleted_entries} log_entries, {deleted_hourly} "
|
||||
f"request_stats_hourly, {deleted_ip_stats} ip traffic-rollup rows "
|
||||
f"deleted; {archived_files} parsed upload file(s) removed from disk."
|
||||
)
|
||||
|
||||
|
||||
def _delete_old_log_entries(now: datetime) -> int:
|
||||
"""Ch06: raw log_entries retained ~30 days, then deleted."""
|
||||
cutoff = now - timedelta(days=LOG_ENTRIES_RETENTION_DAYS)
|
||||
count = db.session.query(LogEntry).filter(LogEntry.timestamp < cutoff).delete(synchronize_session=False)
|
||||
db.session.commit()
|
||||
return count
|
||||
|
||||
|
||||
def _collapse_old_hourly_stats(now: datetime) -> int:
|
||||
"""Ch06: request_stats_hourly retained ~90 days, then collapsed into
|
||||
request_stats_daily only. request_stats_daily is already computed
|
||||
independently by the aggregator straight from raw log_entries, so
|
||||
"collapsing" here just means deleting the now-redundant hourly rows
|
||||
once the retention window passes — no data is lost, since the daily
|
||||
rollup for that period was already written when the file was parsed.
|
||||
"""
|
||||
cutoff = now - timedelta(days=HOURLY_STATS_RETENTION_DAYS)
|
||||
count = db.session.query(RequestStatsHourly).filter(
|
||||
RequestStatsHourly.date_hour < cutoff
|
||||
).delete(synchronize_session=False)
|
||||
db.session.commit()
|
||||
return count
|
||||
|
||||
|
||||
def _delete_old_ip_traffic_stats(now: datetime) -> int:
|
||||
"""The bounded per-IP rollup (added as a Ch10 follow-up) was designed
|
||||
for ~30-day retention, matching log_entries — see aggregator.py.
|
||||
"""
|
||||
cutoff_date = (now - timedelta(days=IP_TRAFFIC_STATS_RETENTION_DAYS)).date()
|
||||
count = db.session.query(IpPathStatsDaily).filter(IpPathStatsDaily.date < cutoff_date).delete(synchronize_session=False)
|
||||
count += db.session.query(IpStatusStatsDaily).filter(IpStatusStatsDaily.date < cutoff_date).delete(synchronize_session=False)
|
||||
db.session.commit()
|
||||
return count
|
||||
|
||||
|
||||
def _delete_parsed_upload_files() -> int:
|
||||
"""Ch06: 'compress or delete after successful parse + rollup, rather
|
||||
than keeping both the raw file and a full raw-row copy.' Deletes
|
||||
(rather than compresses) — simpler, and avoids spending extra CPU/IOPS
|
||||
gzip-ing data that's already been fully parsed into the database.
|
||||
"""
|
||||
done_files = LogFile.query.filter_by(status="done").all()
|
||||
removed = 0
|
||||
for log_file in done_files:
|
||||
path = upload_path_for(log_file)
|
||||
if path.exists():
|
||||
path.unlink()
|
||||
removed += 1
|
||||
return removed
|
||||
|
||||
|
||||
@click.command("create-admin")
|
||||
@click.option("--email", default=None, help="Defaults to the ADMIN_EMAIL env var.")
|
||||
@click.password_option()
|
||||
def create_admin(email: str | None, password: str) -> None:
|
||||
"""Bootstrap the single admin account (Chapter 12).
|
||||
|
||||
Interactive password prompt (via --password-option's confirmation
|
||||
prompt) keeps the raw credential out of process env/config, unlike an
|
||||
ADMIN_PASSWORD env var would — Chapter 12 doesn't specify a bootstrap
|
||||
mechanism beyond documenting ADMIN_EMAIL, so this is a flagged addition.
|
||||
"""
|
||||
resolved_email = (email or current_app.config.get("ADMIN_EMAIL") or "").strip().lower()
|
||||
if not resolved_email:
|
||||
raise click.UsageError("No --email given and ADMIN_EMAIL is not set.")
|
||||
if User.query.filter_by(email=resolved_email).first():
|
||||
raise click.UsageError(f"User {resolved_email} already exists.")
|
||||
|
||||
user = User(email=resolved_email)
|
||||
user.set_password(password)
|
||||
db.session.add(user)
|
||||
db.session.commit()
|
||||
click.echo(f"Created admin user {resolved_email}")
|
||||
+109
@@ -0,0 +1,109 @@
|
||||
"""Environment-sourced configuration classes (12-factor factor 3).
|
||||
|
||||
Every value comes from os.environ. No secret, DB URL, or filesystem path
|
||||
is ever hardcoded; .env.example documents every variable a deployment
|
||||
must set.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from datetime import timedelta
|
||||
|
||||
|
||||
def _bool_env(name: str, default: bool = False) -> bool:
|
||||
"""Parse a boolean-ish environment variable."""
|
||||
val = os.environ.get(name)
|
||||
return default if val is None else val.strip().lower() in {"1", "true", "yes", "on"}
|
||||
|
||||
|
||||
def _engine_options_for(database_url: str) -> dict:
|
||||
"""pool_size/pool_recycle are QueuePool-only kwargs.
|
||||
|
||||
BUG FIX (caught by actually booting the app): SQLite's default pool
|
||||
classes (NullPool for file DBs, StaticPool for :memory:) raise
|
||||
TypeError if handed pool_size/pool_recycle at all — they're not
|
||||
silently ignored. Chapter 06 makes SQLite the default engine and
|
||||
MySQL the opt-in fallback, so these kwargs are only meaningful (and
|
||||
only passed) when DATABASE_URL actually points at a non-SQLite engine.
|
||||
"""
|
||||
if database_url.startswith("sqlite"):
|
||||
return {}
|
||||
return {
|
||||
"pool_size": int(os.environ.get("DB_POOL_SIZE", "3")),
|
||||
"pool_recycle": int(os.environ.get("DB_POOL_RECYCLE_SECONDS", "280")),
|
||||
"pool_pre_ping": True,
|
||||
}
|
||||
|
||||
|
||||
class BaseConfig:
|
||||
"""Shared config. Subclasses override only what differs per environment."""
|
||||
|
||||
SECRET_KEY: str | None = os.environ.get("SECRET_KEY")
|
||||
|
||||
# Chapter 06: SQLite/WAL default; swappable via DATABASE_URL without code changes.
|
||||
SQLALCHEMY_DATABASE_URI: str = os.environ.get("DATABASE_URL", "sqlite:///kavosh.db")
|
||||
SQLALCHEMY_ENGINE_OPTIONS: dict = _engine_options_for(SQLALCHEMY_DATABASE_URI)
|
||||
SQLALCHEMY_TRACK_MODIFICATIONS = False
|
||||
|
||||
# Chapter 03: 150 max DB connections shared across up to 60 entry
|
||||
# processes -> keep each process's pool small. Exposed as its own
|
||||
# config key (not just buried inside SQLALCHEMY_ENGINE_OPTIONS) so
|
||||
# budget.py can validate the *intended* setting regardless of whether
|
||||
# the active engine (SQLite) actually consumes it.
|
||||
DB_POOL_SIZE: int = int(os.environ.get("DB_POOL_SIZE", "3"))
|
||||
|
||||
# Chapter 03: filesystem cache, not Redis.
|
||||
CACHE_TYPE = os.environ.get("CACHE_TYPE", "FileSystemCache")
|
||||
CACHE_DIR = os.environ.get("CACHE_DIR", "/tmp/kavosh-cache")
|
||||
CACHE_DEFAULT_TIMEOUT = int(os.environ.get("CACHE_DEFAULT_TIMEOUT", "60"))
|
||||
|
||||
# Chapter 12: secure session cookie flags.
|
||||
SESSION_COOKIE_SECURE = _bool_env("SESSION_COOKIE_SECURE", True)
|
||||
SESSION_COOKIE_HTTPONLY = True
|
||||
SESSION_COOKIE_SAMESITE = "Lax"
|
||||
PERMANENT_SESSION_LIFETIME = timedelta(
|
||||
hours=int(os.environ.get("SESSION_LIFETIME_HOURS", "12"))
|
||||
)
|
||||
|
||||
WTF_CSRF_ENABLED = True
|
||||
|
||||
# Chapter 04/12: upload constraints.
|
||||
UPLOAD_MAX_SIZE_MB = int(os.environ.get("UPLOAD_MAX_SIZE_MB", "500"))
|
||||
MAX_CONTENT_LENGTH = UPLOAD_MAX_SIZE_MB * 1024 * 1024
|
||||
UPLOAD_DIR = os.environ.get("UPLOAD_DIR", "/home/kavosh/uploads")
|
||||
|
||||
PARSE_BATCH_SIZE = int(os.environ.get("PARSE_BATCH_SIZE", "5000"))
|
||||
ADMIN_EMAIL = os.environ.get("ADMIN_EMAIL")
|
||||
|
||||
|
||||
class DevConfig(BaseConfig):
|
||||
DEBUG = True
|
||||
SESSION_COOKIE_SECURE = False # allow plain-http local dev
|
||||
|
||||
|
||||
class ProdConfig(BaseConfig):
|
||||
DEBUG = False
|
||||
|
||||
|
||||
class TestConfig(BaseConfig):
|
||||
TESTING = True
|
||||
DEBUG = True # lets asset() tolerate a missing Vite manifest during tests
|
||||
SQLALCHEMY_DATABASE_URI = "sqlite:///:memory:"
|
||||
SQLALCHEMY_ENGINE_OPTIONS = {} # always in-memory SQLite regardless of DATABASE_URL
|
||||
WTF_CSRF_ENABLED = False
|
||||
|
||||
|
||||
_CONFIGS = {"development": DevConfig, "production": ProdConfig, "testing": TestConfig}
|
||||
|
||||
|
||||
def get_config(config_name: str | None = None):
|
||||
"""Resolve a config class from FLASK_ENV or an explicit name.
|
||||
|
||||
Defaults to `production` if unset, so an unconfigured deployment never
|
||||
silently runs with DEBUG on.
|
||||
"""
|
||||
name = config_name or os.environ.get("FLASK_ENV", "production")
|
||||
try:
|
||||
return _CONFIGS[name]
|
||||
except KeyError as exc:
|
||||
raise ValueError(f"Unknown config_name {name!r}; expected one of {list(_CONFIGS)}") from exc
|
||||
@@ -0,0 +1,28 @@
|
||||
"""Singleton Flask extension instances — initialized, not configured, here.
|
||||
|
||||
Configuration happens in create_app() via .init_app(), so nothing here
|
||||
holds app- or request-scoped state that must survive a process restart
|
||||
(factor 6: stateless processes).
|
||||
"""
|
||||
from flask_caching import Cache
|
||||
from flask_limiter import Limiter
|
||||
from flask_limiter.util import get_remote_address
|
||||
from flask_login import LoginManager
|
||||
from flask_migrate import Migrate
|
||||
from flask_sqlalchemy import SQLAlchemy
|
||||
from flask_wtf import CSRFProtect
|
||||
|
||||
db = SQLAlchemy()
|
||||
cache = Cache()
|
||||
csrf = CSRFProtect()
|
||||
login_manager = LoginManager()
|
||||
login_manager.login_view = "auth.login"
|
||||
migrate = Migrate()
|
||||
|
||||
# In-memory storage (Chapter 12: "in-memory or DB-backed... do not require
|
||||
# Redis"). CAVEAT (flagged): each of up to 60 entry processes (Ch03) keeps
|
||||
# its own counters, so the effective ceiling across the whole app is up to
|
||||
# (per-process limit x concurrent processes hit), not one hard global cap.
|
||||
# Acceptable for this app's threat model (slowing down /login and /uploads
|
||||
# brute-forcing), but not a strict global rate guarantee.
|
||||
limiter = Limiter(key_func=get_remote_address, storage_uri="memory://")
|
||||
@@ -0,0 +1,34 @@
|
||||
"""Structured stdout/stderr logging (Chapter 02 factor 11 / Chapter 12).
|
||||
|
||||
Writes structured (key=value) lines to stdout so the host's log capture
|
||||
picks them up, per 12-factor logs. Falls back to a size-capped rotating
|
||||
file only if LOG_FALLBACK_FILE is explicitly set (Ch12: "if stdout capture
|
||||
is unavailable on the specific hosting setup").
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
from logging.handlers import RotatingFileHandler
|
||||
|
||||
from flask import Flask
|
||||
|
||||
_FORMAT = "%(asctime)s level=%(levelname)s logger=%(name)s msg=%(message)s"
|
||||
|
||||
|
||||
def configure_logging(app: Flask) -> None:
|
||||
formatter = logging.Formatter(_FORMAT)
|
||||
|
||||
stdout_handler = logging.StreamHandler(sys.stdout)
|
||||
stdout_handler.setFormatter(formatter)
|
||||
|
||||
app.logger.handlers = [stdout_handler]
|
||||
app.logger.setLevel(logging.INFO if not app.debug else logging.DEBUG)
|
||||
app.logger.propagate = False
|
||||
|
||||
fallback_path = os.environ.get("LOG_FALLBACK_FILE")
|
||||
if fallback_path:
|
||||
file_handler = RotatingFileHandler(fallback_path, maxBytes=5 * 1024 * 1024, backupCount=3)
|
||||
file_handler.setFormatter(formatter)
|
||||
app.logger.addHandler(file_handler)
|
||||
@@ -0,0 +1,19 @@
|
||||
from app.models.blocklist_suggestion import BlocklistSuggestion
|
||||
from app.models.bot_hit import BotHit
|
||||
from app.models.browser_stats import BrowserStatsDaily
|
||||
from app.models.human_path_stats import HumanPathStatsDaily
|
||||
from app.models.ip_registry import IPRegistry
|
||||
from app.models.ip_traffic_stats import IpPathStatsDaily, IpStatusStatsDaily
|
||||
from app.models.log_entry import LogEntry
|
||||
from app.models.log_file import LogFile
|
||||
from app.models.referrer_stats import ReferrerStatsDaily
|
||||
from app.models.request_stats import RequestStatsDaily, RequestStatsHourly
|
||||
from app.models.suspicious_event import SuspiciousEvent
|
||||
from app.models.user import User
|
||||
|
||||
__all__ = [
|
||||
"LogFile", "LogEntry", "RequestStatsHourly", "RequestStatsDaily",
|
||||
"BotHit", "IPRegistry", "SuspiciousEvent", "BlocklistSuggestion",
|
||||
"ReferrerStatsDaily", "BrowserStatsDaily", "HumanPathStatsDaily",
|
||||
"IpPathStatsDaily", "IpStatusStatsDaily", "User",
|
||||
]
|
||||
@@ -0,0 +1,18 @@
|
||||
"""blocklist_suggestions table (Chapter 06). `id` isn't in Ch06's column
|
||||
list — added because `ip` alone can't be the key (the same IP may be
|
||||
re-flagged with a different reason later)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import datetime
|
||||
|
||||
from app.extensions import db
|
||||
|
||||
|
||||
class BlocklistSuggestion(db.Model):
|
||||
__tablename__ = "blocklist_suggestions"
|
||||
|
||||
id: int = db.Column(db.Integer, primary_key=True)
|
||||
ip: str = db.Column(db.String(45), nullable=False, index=True)
|
||||
reason: str = db.Column(db.String(255), nullable=False)
|
||||
created_at = db.Column(db.DateTime, nullable=False, default=datetime.utcnow)
|
||||
exported: bool = db.Column(db.Boolean, nullable=False, default=False)
|
||||
@@ -0,0 +1,22 @@
|
||||
"""bot_hits table (Chapter 06) — SEO section (Ch09)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from app.extensions import db
|
||||
|
||||
|
||||
class BotHit(db.Model):
|
||||
__tablename__ = "bot_hits"
|
||||
|
||||
id: int = db.Column(db.Integer, primary_key=True)
|
||||
log_file_id: int = db.Column(db.Integer, db.ForeignKey("log_files.id", ondelete="CASCADE"), nullable=False)
|
||||
timestamp = db.Column(db.DateTime, nullable=False)
|
||||
ip: str = db.Column(db.String(45), nullable=False)
|
||||
bot_name: str = db.Column(db.String(64), nullable=False)
|
||||
verified: bool = db.Column(db.Boolean, nullable=False, default=False)
|
||||
path: str = db.Column(db.Text, nullable=False)
|
||||
status_code: int = db.Column(db.SmallInteger, nullable=False)
|
||||
|
||||
__table_args__ = (
|
||||
db.Index("ix_bot_hits_timestamp", "timestamp"),
|
||||
db.Index("ix_bot_hits_bot_name", "bot_name"),
|
||||
)
|
||||
@@ -0,0 +1,14 @@
|
||||
"""NEW table (Method A, Ch08 follow-up) — not in Chapter 06's original
|
||||
list. Human-only browser/OS breakdown widget (Ch08)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from app.extensions import db
|
||||
|
||||
|
||||
class BrowserStatsDaily(db.Model):
|
||||
__tablename__ = "browser_stats_daily"
|
||||
|
||||
date = db.Column(db.Date, primary_key=True)
|
||||
browser: str = db.Column(db.String(64), primary_key=True)
|
||||
os: str = db.Column(db.String(64), primary_key=True)
|
||||
count: int = db.Column(db.Integer, nullable=False, default=0)
|
||||
@@ -0,0 +1,15 @@
|
||||
"""NEW table (Method A, Ch09 follow-up) — not in Chapter 06's original
|
||||
list. request_stats_hourly/_daily aggregate ALL traffic with no is_bot
|
||||
split, so Ch09's "most-visited-by-humans" comparison had no rollup to
|
||||
read. Bot hits excluded at rollup-write time (Ch07's is_bot flag)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from app.extensions import db
|
||||
|
||||
|
||||
class HumanPathStatsDaily(db.Model):
|
||||
__tablename__ = "human_path_stats_daily"
|
||||
|
||||
date = db.Column(db.Date, primary_key=True)
|
||||
path: str = db.Column(db.String(2048), primary_key=True)
|
||||
count: int = db.Column(db.Integer, nullable=False, default=0)
|
||||
@@ -0,0 +1,22 @@
|
||||
"""ip_registry table (Chapter 06 + Chapter 07 extension).
|
||||
|
||||
last_verified_bot_result: NOT in Ch06's literal column list — added in
|
||||
Chapter 07. last_verified_at alone can't tell a cache hit *what* was
|
||||
verified, only *when*; this stores the outcome.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from app.extensions import db
|
||||
|
||||
|
||||
class IPRegistry(db.Model):
|
||||
__tablename__ = "ip_registry"
|
||||
|
||||
ip: str = db.Column(db.String(45), primary_key=True)
|
||||
first_seen = db.Column(db.DateTime, nullable=False)
|
||||
last_seen = db.Column(db.DateTime, nullable=False)
|
||||
total_requests: int = db.Column(db.Integer, nullable=False, default=0)
|
||||
reputation_score: int = db.Column(db.Integer, nullable=False, default=0)
|
||||
is_flagged: bool = db.Column(db.Boolean, nullable=False, default=False)
|
||||
last_verified_at = db.Column(db.DateTime, nullable=True)
|
||||
last_verified_bot_result: bool | None = db.Column(db.Boolean, nullable=True)
|
||||
@@ -0,0 +1,35 @@
|
||||
"""NEW tables (explicit follow-up to Ch10's flagged scope gap): bounded
|
||||
per-IP traffic breakdown so the IP investigation panel reflects TRUE
|
||||
full traffic, not just bot_hits/suspicious_events activity.
|
||||
|
||||
CARDINALITY NOTE: unlike the other Method-A tables, this one's row count
|
||||
scales with distinct IPs per day, which is unbounded for a probed site.
|
||||
Two mitigations: ip_path_stats_daily keeps only the top N paths per IP
|
||||
per day (not every ip x path pair); both tables use log_entries' ~30-day
|
||||
retention (Ch06), enforced by `flask cleanup` (Chapter 12).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from app.extensions import db
|
||||
|
||||
|
||||
class IpPathStatsDaily(db.Model):
|
||||
__tablename__ = "ip_path_stats_daily"
|
||||
|
||||
date = db.Column(db.Date, primary_key=True)
|
||||
ip: str = db.Column(db.String(45), primary_key=True)
|
||||
path: str = db.Column(db.String(2048), primary_key=True)
|
||||
count: int = db.Column(db.Integer, nullable=False, default=0)
|
||||
|
||||
__table_args__ = (db.Index("ix_ip_path_stats_ip", "ip"),)
|
||||
|
||||
|
||||
class IpStatusStatsDaily(db.Model):
|
||||
__tablename__ = "ip_status_stats_daily"
|
||||
|
||||
date = db.Column(db.Date, primary_key=True)
|
||||
ip: str = db.Column(db.String(45), primary_key=True)
|
||||
status_bucket: str = db.Column(db.String(8), primary_key=True) # 2xx|3xx|4xx|5xx
|
||||
count: int = db.Column(db.Integer, nullable=False, default=0)
|
||||
|
||||
__table_args__ = (db.Index("ix_ip_status_stats_ip", "ip"),)
|
||||
@@ -0,0 +1,29 @@
|
||||
"""Optional, time-boxed raw storage (Chapter 06) — retention (~30 days)
|
||||
enforced by `flask cleanup` (Chapter 12)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from app.extensions import db
|
||||
|
||||
|
||||
class LogEntry(db.Model):
|
||||
__tablename__ = "log_entries"
|
||||
|
||||
id: int = db.Column(db.Integer, primary_key=True)
|
||||
log_file_id: int = db.Column(db.Integer, db.ForeignKey("log_files.id", ondelete="CASCADE"), nullable=False)
|
||||
timestamp = db.Column(db.DateTime, nullable=False) # normalized to UTC (Ch07)
|
||||
ip: str = db.Column(db.String(45), nullable=False) # IPv4 or IPv6
|
||||
method: str = db.Column(db.String(10), nullable=False)
|
||||
path: str = db.Column(db.Text, nullable=False)
|
||||
status_code: int = db.Column(db.SmallInteger, nullable=False)
|
||||
bytes_sent: int = db.Column(db.Integer, nullable=False, default=0)
|
||||
referrer: str | None = db.Column(db.Text, nullable=True)
|
||||
user_agent: str | None = db.Column(db.Text, nullable=True)
|
||||
is_bot: bool = db.Column(db.Boolean, nullable=False, default=False)
|
||||
flagged: bool = db.Column(db.Boolean, nullable=False, default=False)
|
||||
|
||||
__table_args__ = (
|
||||
db.Index("ix_log_entries_file_ts", "log_file_id", "timestamp"),
|
||||
db.Index("ix_log_entries_ip", "ip"),
|
||||
db.Index("ix_log_entries_path", "path"),
|
||||
db.Index("ix_log_entries_timestamp", "timestamp"),
|
||||
)
|
||||
@@ -0,0 +1,22 @@
|
||||
"""LogFile model (Chapter 06)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import datetime
|
||||
|
||||
from app.extensions import db
|
||||
|
||||
|
||||
class LogFile(db.Model):
|
||||
__tablename__ = "log_files"
|
||||
|
||||
id: int = db.Column(db.Integer, primary_key=True)
|
||||
filename: str = db.Column(db.String(255), nullable=False)
|
||||
server_type: str = db.Column(db.String(16), nullable=False) # apache|litespeed
|
||||
format_string: str = db.Column(db.Text, nullable=False)
|
||||
uploaded_at: datetime = db.Column(db.DateTime, nullable=False, default=datetime.utcnow)
|
||||
status: str = db.Column(db.String(16), nullable=False, default="queued")
|
||||
total_lines: int | None = db.Column(db.Integer, nullable=True)
|
||||
processed_lines: int = db.Column(db.Integer, nullable=False, default=0)
|
||||
size_bytes: int = db.Column(db.BigInteger, nullable=False)
|
||||
checksum: str = db.Column(db.String(64), nullable=False, index=True) # sha256 hex
|
||||
error_message: str | None = db.Column(db.Text, nullable=True)
|
||||
@@ -0,0 +1,14 @@
|
||||
"""NEW table (Method A, Ch08 follow-up) — not in Chapter 06's original
|
||||
list. Gives Top Referrers a rollup data source instead of scanning
|
||||
log_entries live. Domain-bucketed to keep cardinality bounded."""
|
||||
from __future__ import annotations
|
||||
|
||||
from app.extensions import db
|
||||
|
||||
|
||||
class ReferrerStatsDaily(db.Model):
|
||||
__tablename__ = "referrer_stats_daily"
|
||||
|
||||
date = db.Column(db.Date, primary_key=True)
|
||||
referrer_domain: str = db.Column(db.String(255), primary_key=True)
|
||||
count: int = db.Column(db.Integer, nullable=False, default=0)
|
||||
@@ -0,0 +1,26 @@
|
||||
"""Rollup tables (Chapter 06) — primary source for Overview widgets (Ch08).
|
||||
Both are site-wide (no log_file_id), consistent with Ch01's single-site scope.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from app.extensions import db
|
||||
|
||||
|
||||
class RequestStatsHourly(db.Model):
|
||||
__tablename__ = "request_stats_hourly"
|
||||
|
||||
date_hour = db.Column(db.DateTime, primary_key=True)
|
||||
path: str = db.Column(db.String(2048), primary_key=True)
|
||||
status_code: int = db.Column(db.SmallInteger, primary_key=True)
|
||||
count: int = db.Column(db.Integer, nullable=False, default=0)
|
||||
bytes_sent_sum: int = db.Column(db.BigInteger, nullable=False, default=0)
|
||||
|
||||
|
||||
class RequestStatsDaily(db.Model):
|
||||
__tablename__ = "request_stats_daily"
|
||||
|
||||
date = db.Column(db.Date, primary_key=True)
|
||||
count: int = db.Column(db.Integer, nullable=False, default=0)
|
||||
unique_ips: int = db.Column(db.Integer, nullable=False, default=0)
|
||||
bytes_sum: int = db.Column(db.BigInteger, nullable=False, default=0)
|
||||
error_count: int = db.Column(db.Integer, nullable=False, default=0)
|
||||
@@ -0,0 +1,22 @@
|
||||
"""suspicious_events table (Chapter 06) — Security section (Ch10)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from app.extensions import db
|
||||
|
||||
|
||||
class SuspiciousEvent(db.Model):
|
||||
__tablename__ = "suspicious_events"
|
||||
|
||||
id: int = db.Column(db.Integer, primary_key=True)
|
||||
log_file_id: int = db.Column(db.Integer, db.ForeignKey("log_files.id", ondelete="CASCADE"), nullable=False)
|
||||
ip: str = db.Column(db.String(45), nullable=False)
|
||||
timestamp = db.Column(db.DateTime, nullable=False)
|
||||
path: str = db.Column(db.Text, nullable=False)
|
||||
rule_matched: str = db.Column(db.String(128), nullable=False)
|
||||
severity: str = db.Column(db.String(8), nullable=False) # low|medium|high (provisional; Ch10 escalates)
|
||||
|
||||
__table_args__ = (
|
||||
db.Index("ix_suspicious_events_timestamp", "timestamp"),
|
||||
db.Index("ix_suspicious_events_ip", "ip"),
|
||||
db.Index("ix_suspicious_events_severity", "severity"),
|
||||
)
|
||||
@@ -0,0 +1,27 @@
|
||||
"""Single-admin user model (Chapter 12). Not in Chapter 06's table list —
|
||||
that chapter scopes log-analytics tables; auth is a separate concern this
|
||||
chapter owns. Chapter 01 confirms single-admin, no self-registration.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import datetime
|
||||
|
||||
import bcrypt
|
||||
from flask_login import UserMixin
|
||||
|
||||
from app.extensions import db
|
||||
|
||||
|
||||
class User(db.Model, UserMixin):
|
||||
__tablename__ = "users"
|
||||
|
||||
id: int = db.Column(db.Integer, primary_key=True)
|
||||
email: str = db.Column(db.String(255), unique=True, nullable=False, index=True)
|
||||
password_hash: str = db.Column(db.String(255), nullable=False)
|
||||
created_at = db.Column(db.DateTime, nullable=False, default=datetime.utcnow)
|
||||
|
||||
def set_password(self, raw_password: str) -> None:
|
||||
self.password_hash = bcrypt.hashpw(raw_password.encode("utf-8"), bcrypt.gensalt()).decode("utf-8")
|
||||
|
||||
def check_password(self, raw_password: str) -> bool:
|
||||
return bcrypt.checkpw(raw_password.encode("utf-8"), self.password_hash.encode("utf-8"))
|
||||
@@ -0,0 +1,199 @@
|
||||
"""Rollup computation (Chapter 06 + Method A extensions from Ch08/09/10).
|
||||
|
||||
Called once per parsed file (app/cli.py::process_logs) for the dates it
|
||||
touched, ad hoc via `flask rollup` for a manual recompute, and now also
|
||||
after a file deletion (app/services/file_deletion.py) for whatever dates
|
||||
the deleted file touched. Every _upsert_* function scans log_entries
|
||||
inside this background batch job, never at request time — that's what
|
||||
makes Ch03 rule 6 compliance possible.
|
||||
|
||||
CORRECTNESS FIX: every rollup writer below now deletes a day's existing
|
||||
rows before writing whatever the fresh scan finds (including writing
|
||||
nothing, if a day now has zero data). Three of the five writers
|
||||
previously only ever upserted-when-present and silently left stale rows
|
||||
behind when a day's data disappeared — unreachable before file deletion
|
||||
existed (rollups only ever grew), but a real correctness bug once
|
||||
deletion makes "this day now has less data than before" possible. Only
|
||||
the two per-IP writers already had this right (Ch10 follow-up); the
|
||||
other three are fixed here to match.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from collections import defaultdict
|
||||
from datetime import date, datetime, time, timedelta
|
||||
|
||||
from sqlalchemy import case, func
|
||||
|
||||
from app.extensions import db
|
||||
from app.models.browser_stats import BrowserStatsDaily
|
||||
from app.models.human_path_stats import HumanPathStatsDaily
|
||||
from app.models.ip_traffic_stats import IpPathStatsDaily, IpStatusStatsDaily
|
||||
from app.models.log_entry import LogEntry
|
||||
from app.models.referrer_stats import ReferrerStatsDaily
|
||||
from app.models.request_stats import RequestStatsDaily, RequestStatsHourly
|
||||
from app.services.blocklist import refresh_blocklist_suggestions
|
||||
from app.services.referrer import referrer_domain
|
||||
from app.services.ua_classifier import classify_browser, classify_os
|
||||
from app.utils.http_status import status_bucket
|
||||
|
||||
TOP_PATHS_PER_IP_PER_DAY = 15 # bounds ip_path_stats_daily row growth (Ch10 follow-up)
|
||||
|
||||
|
||||
def compute_rollups_for_range(start: date, end: date) -> None:
|
||||
"""Recompute every rollup for each day in [start, end]. Site-wide,
|
||||
not per-file (Ch01: single site) — recomputing from scratch per day
|
||||
avoids double-counting when two uploads cover the same period, and
|
||||
correctly shrinks a day's numbers back down when a file covering
|
||||
that day is deleted.
|
||||
"""
|
||||
current = start
|
||||
while current <= end:
|
||||
_upsert_hourly(current)
|
||||
_upsert_daily(current)
|
||||
_upsert_per_line_derived_stats(current)
|
||||
refresh_blocklist_suggestions(current)
|
||||
current += timedelta(days=1)
|
||||
|
||||
|
||||
def _day_bounds(day: date) -> tuple[datetime, datetime]:
|
||||
start = datetime.combine(day, time.min)
|
||||
return start, start + timedelta(days=1)
|
||||
|
||||
|
||||
def _upsert_hourly(day: date) -> None:
|
||||
start, end = _day_bounds(day)
|
||||
rows = (
|
||||
db.session.query(
|
||||
func.strftime("%Y-%m-%d %H:00:00", LogEntry.timestamp).label("date_hour"),
|
||||
LogEntry.path,
|
||||
LogEntry.status_code,
|
||||
func.count().label("count"),
|
||||
func.coalesce(func.sum(LogEntry.bytes_sent), 0).label("bytes_sent_sum"),
|
||||
)
|
||||
.filter(LogEntry.timestamp >= start, LogEntry.timestamp < end)
|
||||
.group_by("date_hour", LogEntry.path, LogEntry.status_code)
|
||||
.all()
|
||||
)
|
||||
|
||||
# Delete-then-insert: replaces the day's hourly rows entirely,
|
||||
# including leaving none behind if `rows` is now empty (e.g. the
|
||||
# only file covering this day was just deleted).
|
||||
db.session.query(RequestStatsHourly).filter(
|
||||
RequestStatsHourly.date_hour >= start, RequestStatsHourly.date_hour < end
|
||||
).delete()
|
||||
|
||||
if rows:
|
||||
payload = [
|
||||
{
|
||||
"date_hour": datetime.strptime(r.date_hour, "%Y-%m-%d %H:%M:%S"),
|
||||
"path": r.path,
|
||||
"status_code": r.status_code,
|
||||
"count": r.count,
|
||||
"bytes_sent_sum": r.bytes_sent_sum,
|
||||
}
|
||||
for r in rows
|
||||
]
|
||||
db.session.execute(RequestStatsHourly.__table__.insert(), payload)
|
||||
|
||||
db.session.commit()
|
||||
|
||||
|
||||
def _upsert_daily(day: date) -> None:
|
||||
start, end = _day_bounds(day)
|
||||
result = (
|
||||
db.session.query(
|
||||
func.count().label("count"),
|
||||
func.count(func.distinct(LogEntry.ip)).label("unique_ips"),
|
||||
func.coalesce(func.sum(LogEntry.bytes_sent), 0).label("bytes_sum"),
|
||||
func.coalesce(func.sum(case((LogEntry.status_code >= 400, 1), else_=0)), 0).label("error_count"),
|
||||
)
|
||||
.filter(LogEntry.timestamp >= start, LogEntry.timestamp < end)
|
||||
.one()
|
||||
)
|
||||
|
||||
# Delete-then-insert: if this day now has zero entries (its only
|
||||
# contributing file was deleted), the stale row is removed rather
|
||||
# than left behind — no rollup row is better than a wrong one.
|
||||
db.session.query(RequestStatsDaily).filter(RequestStatsDaily.date == day).delete()
|
||||
|
||||
if result.count > 0:
|
||||
db.session.execute(
|
||||
RequestStatsDaily.__table__.insert(),
|
||||
{
|
||||
"date": day,
|
||||
"count": result.count,
|
||||
"unique_ips": result.unique_ips,
|
||||
"bytes_sum": result.bytes_sum,
|
||||
"error_count": result.error_count,
|
||||
},
|
||||
)
|
||||
|
||||
db.session.commit()
|
||||
|
||||
|
||||
def _upsert_per_line_derived_stats(day: date) -> None:
|
||||
"""Referrer domain, browser/OS, human-only path counts (Ch08/09), and
|
||||
per-IP path/status counts (Ch10 follow-up) — one streamed pass over
|
||||
log_entries (Ch03 rule 1: bounded per-chunk memory via yield_per,
|
||||
never the whole day loaded at once). Every table here uses the same
|
||||
delete-then-insert pattern so a day's rows are fully replaced by
|
||||
whatever the fresh scan finds, including nothing.
|
||||
"""
|
||||
start, end = _day_bounds(day)
|
||||
referrer_counts: dict[str, int] = defaultdict(int)
|
||||
browser_counts: dict[tuple[str, str], int] = defaultdict(int)
|
||||
human_path_counts: dict[str, int] = defaultdict(int)
|
||||
ip_path_counts: dict[str, dict[str, int]] = defaultdict(lambda: defaultdict(int))
|
||||
ip_status_counts: dict[str, dict[str, int]] = defaultdict(lambda: defaultdict(int))
|
||||
|
||||
query = (
|
||||
db.session.query(
|
||||
LogEntry.referrer, LogEntry.user_agent, LogEntry.is_bot,
|
||||
LogEntry.path, LogEntry.ip, LogEntry.status_code,
|
||||
)
|
||||
.filter(LogEntry.timestamp >= start, LogEntry.timestamp < end)
|
||||
)
|
||||
for referrer, user_agent, is_bot, path, ip, status_code in query.yield_per(1000):
|
||||
domain = referrer_domain(referrer)
|
||||
if domain:
|
||||
referrer_counts[domain] += 1
|
||||
if not is_bot: # Ch08: bot traffic excluded from human browser/OS breakdown
|
||||
browser_counts[(classify_browser(user_agent), classify_os(user_agent))] += 1
|
||||
human_path_counts[path] += 1
|
||||
ip_path_counts[ip][path] += 1
|
||||
ip_status_counts[ip][status_bucket(status_code)] += 1
|
||||
|
||||
db.session.query(ReferrerStatsDaily).filter(ReferrerStatsDaily.date == day).delete()
|
||||
if referrer_counts:
|
||||
payload = [{"date": day, "referrer_domain": d, "count": c} for d, c in referrer_counts.items()]
|
||||
db.session.execute(ReferrerStatsDaily.__table__.insert(), payload)
|
||||
|
||||
db.session.query(BrowserStatsDaily).filter(BrowserStatsDaily.date == day).delete()
|
||||
if browser_counts:
|
||||
payload = [{"date": day, "browser": b, "os": o, "count": c} for (b, o), c in browser_counts.items()]
|
||||
db.session.execute(BrowserStatsDaily.__table__.insert(), payload)
|
||||
|
||||
db.session.query(HumanPathStatsDaily).filter(HumanPathStatsDaily.date == day).delete()
|
||||
if human_path_counts:
|
||||
payload = [{"date": day, "path": p, "count": c} for p, c in human_path_counts.items()]
|
||||
db.session.execute(HumanPathStatsDaily.__table__.insert(), payload)
|
||||
|
||||
db.session.query(IpPathStatsDaily).filter(IpPathStatsDaily.date == day).delete()
|
||||
if ip_path_counts:
|
||||
payload = []
|
||||
for ip, paths in ip_path_counts.items():
|
||||
top_paths = sorted(paths.items(), key=lambda kv: kv[1], reverse=True)[:TOP_PATHS_PER_IP_PER_DAY]
|
||||
payload.extend({"date": day, "ip": ip, "path": p, "count": c} for p, c in top_paths)
|
||||
if payload:
|
||||
db.session.execute(IpPathStatsDaily.__table__.insert(), payload)
|
||||
|
||||
db.session.query(IpStatusStatsDaily).filter(IpStatusStatsDaily.date == day).delete()
|
||||
if ip_status_counts:
|
||||
payload = [
|
||||
{"date": day, "ip": ip, "status_bucket": bucket, "count": c}
|
||||
for ip, buckets in ip_status_counts.items()
|
||||
for bucket, c in buckets.items()
|
||||
]
|
||||
db.session.execute(IpStatusStatsDaily.__table__.insert(), payload)
|
||||
|
||||
db.session.commit()
|
||||
@@ -0,0 +1,97 @@
|
||||
"""No-cron automatic processing (Chapter 12 simplification, per project
|
||||
owner request): triggers file parsing immediately in a background thread
|
||||
right after upload, and opportunistically resumes any incomplete files
|
||||
when the Overview page loads — replacing the cron-triggered model.
|
||||
`flask process-logs` still exists in app/cli.py for anyone who'd rather
|
||||
use cron, but nothing requires it anymore.
|
||||
|
||||
TRADEOFF (flagged, deviating from Chapter 02/03's "no persistent
|
||||
background workers, cron-triggered CLI only" stance): a background thread
|
||||
lives inside the same worker process that handled the upload request. If
|
||||
Passenger recycles that process mid-parse, the thread dies with it —
|
||||
progress up to the last commit is still safely checkpointed (same bounded-
|
||||
batch model as before), but nothing will automatically resume it without
|
||||
either cron or a page visit. The "resume on page load" hook below is the
|
||||
deliberate replacement for that guarantee: visiting the Overview tab
|
||||
re-triggers processing for anything left incomplete, so in the worst case
|
||||
a stuck file resumes the next time the admin looks at the dashboard,
|
||||
rather than never.
|
||||
|
||||
This is not a long-lived daemon: each thread terminates once its file
|
||||
reaches "done"/"error" (or the process is killed), and no thread survives
|
||||
a process restart — it just gets re-triggered fresh next time.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import threading
|
||||
|
||||
from flask import Flask
|
||||
|
||||
from app.extensions import db
|
||||
from app.models.log_file import LogFile
|
||||
from app.services.log_processor import process_one_batch
|
||||
|
||||
# In-process guard against launching two threads for the same file at
|
||||
# once (e.g. the upload trigger and a page-load resume firing close
|
||||
# together). Per-worker-process only — a different entry process picking
|
||||
# up the same file concurrently is a low-probability edge case accepted
|
||||
# for this simplification; each write is still a small checkpointed
|
||||
# commit, not a giant one, which limits how bad a collision could be.
|
||||
_active_file_ids: set[int] = set()
|
||||
_lock = threading.Lock()
|
||||
|
||||
|
||||
def _claim(log_file_id: int) -> bool:
|
||||
with _lock:
|
||||
if log_file_id in _active_file_ids:
|
||||
return False
|
||||
_active_file_ids.add(log_file_id)
|
||||
return True
|
||||
|
||||
|
||||
def _release(log_file_id: int) -> None:
|
||||
with _lock:
|
||||
_active_file_ids.discard(log_file_id)
|
||||
|
||||
|
||||
def _run_to_completion(app: Flask, log_file_id: int, batch_size: int) -> None:
|
||||
with app.app_context():
|
||||
try:
|
||||
log_file = db.session.get(LogFile, log_file_id)
|
||||
if log_file is None:
|
||||
return
|
||||
while log_file.status in ("queued", "processing"):
|
||||
process_one_batch(log_file, batch_size)
|
||||
db.session.refresh(log_file)
|
||||
except Exception:
|
||||
app.logger.exception("Background processing failed for log_file_id=%s", log_file_id)
|
||||
log_file = db.session.get(LogFile, log_file_id)
|
||||
if log_file is not None and log_file.status != "done":
|
||||
log_file.status = "error"
|
||||
log_file.error_message = "Processing failed unexpectedly; see server logs."
|
||||
db.session.commit()
|
||||
finally:
|
||||
_release(log_file_id)
|
||||
|
||||
|
||||
def trigger_processing(app: Flask, log_file_id: int) -> None:
|
||||
"""Start background processing for one file; no-ops if already running."""
|
||||
if not _claim(log_file_id):
|
||||
return
|
||||
batch_size = app.config["PARSE_BATCH_SIZE"]
|
||||
thread = threading.Thread(
|
||||
target=_run_to_completion, args=(app, log_file_id, batch_size), daemon=True
|
||||
)
|
||||
thread.start()
|
||||
|
||||
|
||||
def resume_incomplete_files(app: Flask) -> None:
|
||||
"""Opportunistic resume hook, called from the Overview page load —
|
||||
the deliberate replacement for cron's "there's always a next tick"
|
||||
guarantee. Cheap: one indexed status-filtered query.
|
||||
"""
|
||||
incomplete_ids = [
|
||||
lf.id for lf in LogFile.query.filter(LogFile.status.in_(["queued", "processing"])).all()
|
||||
]
|
||||
for log_file_id in incomplete_ids:
|
||||
trigger_processing(app, log_file_id)
|
||||
@@ -0,0 +1,67 @@
|
||||
"""Blocklist-suggestion generation (Chapter 10 follow-up): no earlier
|
||||
chapter assigned ownership of populating blocklist_suggestions or setting
|
||||
ip_registry.is_flagged. Runs in the background aggregator pass (batch, not
|
||||
request-time, per Ch02/03), reusing severity_scoring.py so there's exactly
|
||||
one scoring model between the dashboard display and the flagging decision.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from collections import defaultdict
|
||||
from datetime import date, datetime, timedelta
|
||||
|
||||
from app.extensions import db
|
||||
from app.models.blocklist_suggestion import BlocklistSuggestion
|
||||
from app.models.ip_registry import IPRegistry
|
||||
from app.models.suspicious_event import SuspiciousEvent
|
||||
from app.services.severity_scoring import SeverityInputs, compute_effective_severity
|
||||
|
||||
_RANK = {"low": 0, "medium": 1, "high": 2}
|
||||
|
||||
|
||||
def refresh_blocklist_suggestions(day: date) -> None:
|
||||
"""Flag an IP (is_flagged + a suggestion row) if its escalated severity
|
||||
for `day` reaches 'high'. Idempotent — skips IPs already suggested.
|
||||
"""
|
||||
start = datetime.combine(day, datetime.min.time())
|
||||
end = start + timedelta(days=1)
|
||||
|
||||
events = (
|
||||
db.session.query(SuspiciousEvent.ip, SuspiciousEvent.timestamp, SuspiciousEvent.severity)
|
||||
.filter(SuspiciousEvent.timestamp >= start, SuspiciousEvent.timestamp < end)
|
||||
.all()
|
||||
)
|
||||
if not events:
|
||||
return
|
||||
|
||||
by_ip: dict[str, list] = defaultdict(list)
|
||||
for ip, ts, sev in events:
|
||||
by_ip[ip].append((ts, sev))
|
||||
|
||||
already_suggested = {ip for (ip,) in db.session.query(BlocklistSuggestion.ip).distinct().all()}
|
||||
|
||||
for ip, ip_events in by_ip.items():
|
||||
if ip in already_suggested:
|
||||
continue
|
||||
timestamps = sorted(ts for ts, _ in ip_events)
|
||||
avg_interval = (
|
||||
(timestamps[-1] - timestamps[0]).total_seconds() / (len(timestamps) - 1)
|
||||
if len(timestamps) > 1 else None
|
||||
)
|
||||
worst_base = max((sev for _, sev in ip_events), key=lambda s: _RANK.get(s, 0))
|
||||
effective = compute_effective_severity(
|
||||
SeverityInputs(base_severity=worst_base, ip_event_count=len(ip_events), avg_interval_seconds=avg_interval)
|
||||
)
|
||||
if effective != "high":
|
||||
continue
|
||||
|
||||
db.session.add(BlocklistSuggestion(
|
||||
ip=ip,
|
||||
reason=f"{len(ip_events)} suspicious event(s) on {day.isoformat()}, escalated to high severity",
|
||||
created_at=datetime.utcnow(),
|
||||
exported=False,
|
||||
))
|
||||
ip_row = db.session.get(IPRegistry, ip)
|
||||
if ip_row is not None:
|
||||
ip_row.is_flagged = True
|
||||
|
||||
db.session.commit()
|
||||
@@ -0,0 +1,78 @@
|
||||
"""Bot signature matching + reverse-DNS verification (Chapter 07).
|
||||
|
||||
Signatures are data (JSON), not hardcoded logic, so the list grows without
|
||||
a code change. Verification does real DNS I/O — only ever called from the
|
||||
background process-logs batch job (app/services/classification.py), never
|
||||
synchronously inside a dashboard request, per Chapter 07's explicit rule.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import socket
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime, timedelta
|
||||
from pathlib import Path
|
||||
|
||||
SIGNATURES_PATH = Path(__file__).parent / "data" / "bot_signatures.json"
|
||||
|
||||
# Skip re-verifying the same IP more often than this (Ch07: DNS latency is
|
||||
# a real cost on constrained hosting).
|
||||
VERIFICATION_TTL = timedelta(days=7)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class BotSignature:
|
||||
name: str
|
||||
ua_substrings: tuple[str, ...]
|
||||
verify_suffixes: tuple[str, ...] # PTR hostname must end in one of these
|
||||
|
||||
|
||||
def _load_signatures() -> list[BotSignature]:
|
||||
raw = json.loads(SIGNATURES_PATH.read_text())
|
||||
return [
|
||||
BotSignature(name=e["name"], ua_substrings=tuple(e["ua_substrings"]), verify_suffixes=tuple(e["verify_suffixes"]))
|
||||
for e in raw
|
||||
]
|
||||
|
||||
|
||||
_SIGNATURES = _load_signatures()
|
||||
|
||||
|
||||
def classify_bot(user_agent: str | None) -> str | None:
|
||||
"""Return the claimed bot name via UA substring match, or None."""
|
||||
if not user_agent:
|
||||
return None
|
||||
ua_lower = user_agent.lower()
|
||||
for sig in _SIGNATURES:
|
||||
if any(sub.lower() in ua_lower for sub in sig.ua_substrings):
|
||||
return sig.name
|
||||
return None
|
||||
|
||||
|
||||
def _signature_for(bot_name: str) -> BotSignature | None:
|
||||
return next((s for s in _SIGNATURES if s.name == bot_name), None)
|
||||
|
||||
|
||||
def verify_bot_ip(ip: str, bot_name: str) -> bool:
|
||||
"""Reverse-DNS + forward-confirm that `ip` really belongs to `bot_name`."""
|
||||
sig = _signature_for(bot_name)
|
||||
if sig is None:
|
||||
return False
|
||||
try:
|
||||
hostname, _, _ = socket.gethostbyaddr(ip)
|
||||
except (socket.herror, socket.gaierror, OSError):
|
||||
return False
|
||||
if not any(hostname.lower().endswith(suffix) for suffix in sig.verify_suffixes):
|
||||
return False
|
||||
try:
|
||||
forward_ips = socket.gethostbyname_ex(hostname)[2]
|
||||
except (socket.herror, socket.gaierror, OSError):
|
||||
return False
|
||||
return ip in forward_ips
|
||||
|
||||
|
||||
def is_verification_stale(last_verified_at: datetime | None) -> bool:
|
||||
"""True if this IP needs a fresh DNS check (Ch07 caching rule)."""
|
||||
if last_verified_at is None:
|
||||
return True
|
||||
return datetime.utcnow() - last_verified_at > VERIFICATION_TTL
|
||||
@@ -0,0 +1,122 @@
|
||||
"""Per-batch classification pipeline (Chapter 07): ties bot_identifier and
|
||||
threat_scanner into the bot_hits/suspicious_events/ip_registry write path.
|
||||
Called once per bulk-insert chunk from app/cli.py — DNS-based verification
|
||||
belongs here (background batch job), never in a dashboard request.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime
|
||||
|
||||
from sqlalchemy import func, insert
|
||||
from sqlalchemy.dialects.sqlite import insert as sqlite_insert
|
||||
|
||||
from app.extensions import db
|
||||
from app.models.bot_hit import BotHit
|
||||
from app.models.ip_registry import IPRegistry
|
||||
from app.models.suspicious_event import SuspiciousEvent
|
||||
from app.services import bot_identifier, threat_scanner
|
||||
from app.services.log_parser import ParsedEntry
|
||||
|
||||
|
||||
@dataclass
|
||||
class EntryClassification:
|
||||
"""Cheap, no-I/O flags for the log_entries row itself."""
|
||||
is_bot: bool
|
||||
flagged: bool
|
||||
|
||||
|
||||
def classify_entry(entry: ParsedEntry) -> EntryClassification:
|
||||
"""No DNS I/O here — bot *verification* is batched separately below,
|
||||
since it's stateful (cached per IP) and only worth doing once per IP
|
||||
per batch, not once per line.
|
||||
"""
|
||||
return EntryClassification(
|
||||
is_bot=bot_identifier.classify_bot(entry.user_agent) is not None,
|
||||
flagged=threat_scanner.scan(entry) is not None,
|
||||
)
|
||||
|
||||
|
||||
def write_batch_side_effects(log_file_id: int, entries: list[ParsedEntry]) -> None:
|
||||
"""Derive and bulk-write bot_hits, suspicious_events, ip_registry upserts."""
|
||||
if not entries:
|
||||
return
|
||||
|
||||
bot_hit_rows: list[dict] = []
|
||||
suspicious_rows: list[dict] = []
|
||||
ip_agg: dict[str, dict] = {}
|
||||
verified_this_batch: dict[str, bool] = {} # avoid repeat DNS for the same IP in one batch
|
||||
|
||||
for entry in entries:
|
||||
agg = ip_agg.setdefault(entry.ip, {"first": entry.timestamp, "last": entry.timestamp, "count": 0})
|
||||
agg["count"] += 1
|
||||
agg["first"] = min(agg["first"], entry.timestamp)
|
||||
agg["last"] = max(agg["last"], entry.timestamp)
|
||||
|
||||
bot_name = bot_identifier.classify_bot(entry.user_agent)
|
||||
if bot_name is not None:
|
||||
verified = verified_this_batch.get(entry.ip)
|
||||
if verified is None:
|
||||
verified = _verify_with_cache(entry.ip, bot_name)
|
||||
verified_this_batch[entry.ip] = verified
|
||||
bot_hit_rows.append({
|
||||
"log_file_id": log_file_id, "timestamp": entry.timestamp, "ip": entry.ip,
|
||||
"bot_name": bot_name, "verified": verified, "path": entry.path,
|
||||
"status_code": entry.status_code,
|
||||
})
|
||||
if not verified:
|
||||
# Single detection, two dashboard consumers (Ch07): spoofed
|
||||
# bot also surfaces as a suspicious_events row.
|
||||
suspicious_rows.append({
|
||||
"log_file_id": log_file_id, "ip": entry.ip, "timestamp": entry.timestamp,
|
||||
"path": entry.path, "rule_matched": f"spoofed_bot:{bot_name}", "severity": "medium",
|
||||
})
|
||||
|
||||
threat = threat_scanner.scan(entry)
|
||||
if threat is not None:
|
||||
suspicious_rows.append({
|
||||
"log_file_id": log_file_id, "ip": entry.ip, "timestamp": entry.timestamp,
|
||||
"path": entry.path, "rule_matched": threat.rule_matched, "severity": threat.severity,
|
||||
})
|
||||
|
||||
if bot_hit_rows:
|
||||
db.session.execute(insert(BotHit.__table__), bot_hit_rows)
|
||||
if suspicious_rows:
|
||||
db.session.execute(insert(SuspiciousEvent.__table__), suspicious_rows)
|
||||
|
||||
_upsert_ip_registry(ip_agg, verified_this_batch)
|
||||
db.session.commit()
|
||||
|
||||
|
||||
def _verify_with_cache(ip: str, bot_name: str) -> bool:
|
||||
"""TTL-gated reverse/forward DNS check, cached via
|
||||
ip_registry.last_verified_bot_result (Ch07 schema addition).
|
||||
"""
|
||||
row = db.session.get(IPRegistry, ip)
|
||||
if row is not None and not bot_identifier.is_verification_stale(row.last_verified_at):
|
||||
return bool(row.last_verified_bot_result)
|
||||
return bot_identifier.verify_bot_ip(ip, bot_name)
|
||||
|
||||
|
||||
def _upsert_ip_registry(ip_agg: dict[str, dict], verified_this_batch: dict[str, bool]) -> None:
|
||||
now = datetime.utcnow()
|
||||
for ip, agg in ip_agg.items():
|
||||
values = {
|
||||
"ip": ip, "first_seen": agg["first"], "last_seen": agg["last"],
|
||||
"total_requests": agg["count"], "reputation_score": 0, "is_flagged": False,
|
||||
}
|
||||
if ip in verified_this_batch:
|
||||
values["last_verified_at"] = now
|
||||
values["last_verified_bot_result"] = verified_this_batch[ip]
|
||||
|
||||
stmt = sqlite_insert(IPRegistry.__table__).values(**values)
|
||||
update_set = {
|
||||
"last_seen": func.max(IPRegistry.last_seen, stmt.excluded.last_seen),
|
||||
"first_seen": func.min(IPRegistry.first_seen, stmt.excluded.first_seen),
|
||||
"total_requests": IPRegistry.total_requests + stmt.excluded.total_requests,
|
||||
}
|
||||
if ip in verified_this_batch:
|
||||
update_set["last_verified_at"] = stmt.excluded.last_verified_at
|
||||
update_set["last_verified_bot_result"] = stmt.excluded.last_verified_bot_result
|
||||
stmt = stmt.on_conflict_do_update(index_elements=["ip"], set_=update_set)
|
||||
db.session.execute(stmt)
|
||||
@@ -0,0 +1,9 @@
|
||||
[
|
||||
{"name": "Googlebot", "ua_substrings": ["Googlebot"], "verify_suffixes": [".googlebot.com", ".google.com"]},
|
||||
{"name": "Bingbot", "ua_substrings": ["bingbot"], "verify_suffixes": [".search.msn.com"]},
|
||||
{"name": "Yandex", "ua_substrings": ["YandexBot"], "verify_suffixes": [".yandex.ru", ".yandex.com", ".yandex.net"]},
|
||||
{"name": "Baidu", "ua_substrings": ["Baiduspider"], "verify_suffixes": [".baidu.com", ".baidu.jp"]},
|
||||
{"name": "DuckDuckBot", "ua_substrings": ["DuckDuckBot"], "verify_suffixes": [".duckduckgo.com"]},
|
||||
{"name": "AhrefsBot", "ua_substrings": ["AhrefsBot"], "verify_suffixes": [".ahrefs.com"]},
|
||||
{"name": "SemrushBot", "ua_substrings": ["SemrushBot"], "verify_suffixes": [".semrush.com"]}
|
||||
]
|
||||
@@ -0,0 +1,8 @@
|
||||
[
|
||||
{"name": "Edge", "ua_substrings": ["Edg/", "EdgA/", "EdgiOS/"]},
|
||||
{"name": "Opera", "ua_substrings": ["OPR/", "Opera"]},
|
||||
{"name": "Chrome", "ua_substrings": ["Chrome/", "CriOS/"]},
|
||||
{"name": "Firefox", "ua_substrings": ["Firefox/", "FxiOS/"]},
|
||||
{"name": "Safari", "ua_substrings": ["Safari/"]},
|
||||
{"name": "Internet Explorer", "ua_substrings": ["MSIE ", "Trident/"]}
|
||||
]
|
||||
@@ -0,0 +1,7 @@
|
||||
[
|
||||
{"name": "Windows", "ua_substrings": ["Windows NT"]},
|
||||
{"name": "iOS", "ua_substrings": ["iPhone", "iPad", "iPod"]},
|
||||
{"name": "macOS", "ua_substrings": ["Mac OS X", "Macintosh"]},
|
||||
{"name": "Android", "ua_substrings": ["Android"]},
|
||||
{"name": "Linux", "ua_substrings": ["Linux"]}
|
||||
]
|
||||
@@ -0,0 +1,11 @@
|
||||
{
|
||||
"sensitive_paths": [
|
||||
"/.env", "/.git/config", "wp-config.php", "/.htpasswd", "/phpmyadmin",
|
||||
"/xmlrpc.php", ".sql.gz", ".sql.bak", ".zip", ".bak", ".old"
|
||||
],
|
||||
"injection_markers": [
|
||||
"union select", "' or '1'='1", "<script>", "javascript:",
|
||||
"../", "..%2f", "%2e%2e%2f", "%252e%252e%252f"
|
||||
],
|
||||
"scanner_user_agents": ["sqlmap", "nikto", "nmap", "acunetix", "nessus"]
|
||||
}
|
||||
@@ -0,0 +1,93 @@
|
||||
"""Uploaded-file deletion (project-owner follow-up request).
|
||||
|
||||
Deleting a LogFile is more than one DELETE statement: log_entries,
|
||||
bot_hits, and suspicious_events all reference log_file_id and must be
|
||||
removed explicitly — this project's SQLite connections don't have
|
||||
`PRAGMA foreign_keys=ON`, so the `ondelete="CASCADE"` in the migrations
|
||||
(Ch06) is declarative documentation only, not an enforced behavior.
|
||||
Relying on it would silently leave orphaned rows behind.
|
||||
|
||||
The harder part is the rollup tables (request_stats_hourly/daily,
|
||||
referrer/browser/human-path/ip-path/ip-status): none of them have a
|
||||
log_file_id column (Ch06: they're site-wide, since two files can share a
|
||||
date). So instead of deleting rollup rows for the deleted file's dates
|
||||
directly (which could wipe out another file's contribution to the same
|
||||
date), this captures the affected dates BEFORE deleting, then calls
|
||||
aggregator.compute_rollups_for_range() AFTER deleting — which recomputes
|
||||
each affected day from whatever log_entries remain, correctly handling
|
||||
both "another file still covers this day" and "no file covers this day
|
||||
anymore" (the latter now handled correctly by the aggregator.py fix that
|
||||
shipped alongside this feature).
|
||||
|
||||
KNOWN LIMITATION (flagged, not fixed): ip_registry (total_requests,
|
||||
first_seen, last_seen, verification cache) and blocklist_suggestions are
|
||||
NOT per-file and are NOT recomputed on deletion — doing so correctly
|
||||
would mean re-scanning all remaining log_entries for every affected IP,
|
||||
which is unbounded work for a single delete action and would violate the
|
||||
same "no raw-row scan on a hot path" reasoning Ch03 applies elsewhere.
|
||||
After deleting a file, IP History numbers may include a deleted file's
|
||||
historical contribution until a full site-wide recompute is added as a
|
||||
separate feature. Doesn't affect correctness of Overview/SEO's date-
|
||||
scoped numbers, which is what this feature was actually asked to fix.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
|
||||
from app.extensions import db
|
||||
from app.models.bot_hit import BotHit
|
||||
from app.models.log_entry import LogEntry
|
||||
from app.models.log_file import LogFile
|
||||
from app.models.suspicious_event import SuspiciousEvent
|
||||
from app.services import aggregator
|
||||
from app.utils.upload_paths import upload_path_for
|
||||
|
||||
|
||||
@dataclass
|
||||
class BulkDeleteResult:
|
||||
deleted: list[int] = field(default_factory=list)
|
||||
skipped: list[dict] = field(default_factory=list) # [{"id": ..., "reason": ...}]
|
||||
|
||||
|
||||
def delete_log_files(log_file_ids: list[int]) -> BulkDeleteResult:
|
||||
"""Delete each id in `log_file_ids`; recompute rollups once at the end
|
||||
for the full span of dates any deleted file touched (cheaper than
|
||||
recomputing per-file when multiple files share dates).
|
||||
"""
|
||||
result = BulkDeleteResult()
|
||||
all_touched_dates: set = set()
|
||||
|
||||
for log_file_id in log_file_ids:
|
||||
log_file = db.session.get(LogFile, log_file_id)
|
||||
if log_file is None:
|
||||
result.skipped.append({"id": log_file_id, "reason": "not found"})
|
||||
continue
|
||||
if log_file.status == "processing":
|
||||
result.skipped.append({"id": log_file_id, "reason": "currently being analyzed"})
|
||||
continue
|
||||
|
||||
touched_dates = {
|
||||
row.date() for row, in db.session.query(LogEntry.timestamp)
|
||||
.filter(LogEntry.log_file_id == log_file_id).distinct()
|
||||
}
|
||||
# distinct() on a full timestamp rarely collapses much; reduce to
|
||||
# calendar dates in Python since SQLite's DATE() in a DISTINCT
|
||||
# clause is a bit more awkward to express portably here.
|
||||
all_touched_dates |= touched_dates
|
||||
|
||||
db.session.query(SuspiciousEvent).filter(SuspiciousEvent.log_file_id == log_file_id).delete()
|
||||
db.session.query(BotHit).filter(BotHit.log_file_id == log_file_id).delete()
|
||||
db.session.query(LogEntry).filter(LogEntry.log_file_id == log_file_id).delete()
|
||||
|
||||
raw_path = upload_path_for(log_file)
|
||||
if raw_path.exists():
|
||||
raw_path.unlink()
|
||||
|
||||
db.session.delete(log_file)
|
||||
db.session.commit()
|
||||
result.deleted.append(log_file_id)
|
||||
|
||||
if all_touched_dates:
|
||||
aggregator.compute_rollups_for_range(min(all_touched_dates), max(all_touched_dates))
|
||||
|
||||
return result
|
||||
@@ -0,0 +1,47 @@
|
||||
"""Log parser abstraction (Chapter 07)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime
|
||||
from typing import Protocol, TYPE_CHECKING
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from app.models.log_file import LogFile
|
||||
|
||||
|
||||
@dataclass
|
||||
class ParsedEntry:
|
||||
timestamp: datetime
|
||||
ip: str
|
||||
method: str
|
||||
path: str
|
||||
status_code: int
|
||||
bytes_sent: int
|
||||
referrer: str | None
|
||||
user_agent: str | None
|
||||
|
||||
|
||||
class LogParser(Protocol):
|
||||
def parse_line(self, line: str) -> ParsedEntry | None: ...
|
||||
|
||||
|
||||
def get_parser_for(log_file: "LogFile") -> LogParser:
|
||||
"""Build the right LogParser for a given upload's format_string.
|
||||
|
||||
ASSUMPTION (flagged): a blank format_string — which the upload form
|
||||
allows — falls back to Combined, since Chapter 07 states no default
|
||||
for that case. Anything else is compiled as-is; a real custom
|
||||
LiteSpeed format is never silently replaced by a preset.
|
||||
"""
|
||||
# Imported lazily to avoid a circular import (apache.py imports
|
||||
# ParsedEntry from this module).
|
||||
from app.services.log_parser.apache import FormatCompiledParser, combined_parser, common_parser
|
||||
from app.services.log_parser.format_compiler import compile_format
|
||||
|
||||
fmt = (log_file.format_string or "").strip()
|
||||
key = fmt.lower()
|
||||
if key in ("", "combined"):
|
||||
return combined_parser()
|
||||
if key == "common":
|
||||
return common_parser()
|
||||
return FormatCompiledParser(compile_format(fmt))
|
||||
@@ -0,0 +1,57 @@
|
||||
"""Apache Common/Combined presets + the generic parser built from
|
||||
format_compiler (Chapter 07)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from app.services.log_parser import ParsedEntry
|
||||
from app.services.log_parser.format_compiler import (
|
||||
CompiledFormat, compile_format, parse_apache_timestamp, parse_request_line, validate_ip,
|
||||
)
|
||||
|
||||
COMMON_LOG_FORMAT = '%h %l %u %t "%r" %>s %b'
|
||||
COMBINED_LOG_FORMAT = '%h %l %u %t "%r" %>s %b "%{Referer}i" "%{User-agent}i"'
|
||||
|
||||
|
||||
class FormatCompiledParser:
|
||||
"""LogParser implementation driven by any CompiledFormat (Ch04/07 protocol)."""
|
||||
|
||||
def __init__(self, compiled: CompiledFormat):
|
||||
self._compiled = compiled
|
||||
|
||||
def parse_line(self, line: str) -> ParsedEntry | None:
|
||||
match = self._compiled.pattern.match(line.rstrip("\n"))
|
||||
if match is None:
|
||||
return None # malformed line — skip, never raise (Ch07)
|
||||
fields = match.groupdict()
|
||||
|
||||
ip = validate_ip(fields["ip"])
|
||||
if ip is None:
|
||||
return None
|
||||
|
||||
req = parse_request_line(fields["request"])
|
||||
if req is None:
|
||||
return None
|
||||
method, path = req
|
||||
|
||||
try:
|
||||
timestamp = parse_apache_timestamp(fields["timestamp"])
|
||||
status_code = int(fields["status"])
|
||||
bytes_sent = 0 if fields["bytes"] == "-" else int(fields["bytes"])
|
||||
except (ValueError, KeyError):
|
||||
return None
|
||||
|
||||
referrer = fields.get("referrer")
|
||||
user_agent = fields.get("user_agent")
|
||||
return ParsedEntry(
|
||||
timestamp=timestamp, ip=ip, method=method, path=path,
|
||||
status_code=status_code, bytes_sent=bytes_sent,
|
||||
referrer=None if referrer in (None, "-") else referrer,
|
||||
user_agent=None if user_agent in (None, "-") else user_agent,
|
||||
)
|
||||
|
||||
|
||||
def common_parser() -> FormatCompiledParser:
|
||||
return FormatCompiledParser(compile_format(COMMON_LOG_FORMAT))
|
||||
|
||||
|
||||
def combined_parser() -> FormatCompiledParser:
|
||||
return FormatCompiledParser(compile_format(COMBINED_LOG_FORMAT))
|
||||
@@ -0,0 +1,95 @@
|
||||
"""LogFormat directive -> compiled regex + named groups (Chapter 07).
|
||||
|
||||
One code path for Common, Combined, and arbitrary custom formats — presets
|
||||
below are just specific format strings run through this same compiler.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime, timezone
|
||||
from ipaddress import ip_address
|
||||
|
||||
# Directive -> (group name, regex pattern). NOTE: %r and %{...}i patterns
|
||||
# deliberately do NOT include quote characters — the quotes around them
|
||||
# are literal text already present in the format string (e.g. `"%r"`),
|
||||
# and get regex-escaped by the literal-text path below. %t is different:
|
||||
# its brackets are part of what %t itself produces, not literal format-
|
||||
# string text, so they belong in this directive's own pattern.
|
||||
_SIMPLE_DIRECTIVES: dict[str, tuple[str, str]] = {
|
||||
"%h": ("ip", r"(?P<ip>\S+)"),
|
||||
"%l": ("ident", r"(?P<ident>\S+)"),
|
||||
"%u": ("user", r"(?P<user>\S+)"),
|
||||
"%t": ("timestamp", r"\[(?P<timestamp>[^\]]+)\]"),
|
||||
"%r": ("request", r'(?P<request>[^"]*)'),
|
||||
"%>s": ("status", r"(?P<status>\d{3})"),
|
||||
"%s": ("status", r"(?P<status>\d{3})"),
|
||||
"%b": ("bytes", r"(?P<bytes>\d+|-)"),
|
||||
}
|
||||
|
||||
_HEADER_DIRECTIVE_RE = re.compile(r'%\{([^}]+)\}i')
|
||||
_KNOWN_HEADER_GROUPS = {"referer": "referrer", "user-agent": "user_agent"}
|
||||
_DIRECTIVE_TOKEN_RE = re.compile(r"%>?\{[^}]+\}i|%>?[a-zA-Z]")
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class CompiledFormat:
|
||||
pattern: re.Pattern
|
||||
group_names: frozenset[str]
|
||||
|
||||
|
||||
def compile_format(format_string: str) -> CompiledFormat:
|
||||
"""Compile a LogFormat directive (Common/Combined preset or a pasted
|
||||
custom vhost format) into one regex."""
|
||||
regex_parts: list[str] = []
|
||||
group_names: set[str] = set()
|
||||
pos = 0
|
||||
for match in _DIRECTIVE_TOKEN_RE.finditer(format_string):
|
||||
literal = format_string[pos:match.start()]
|
||||
if literal:
|
||||
regex_parts.append(re.escape(literal))
|
||||
regex_parts.append(_directive_to_pattern(match.group(0), group_names))
|
||||
pos = match.end()
|
||||
regex_parts.append(re.escape(format_string[pos:]))
|
||||
|
||||
pattern = re.compile("^" + "".join(regex_parts) + r"\s*$")
|
||||
return CompiledFormat(pattern=pattern, group_names=frozenset(group_names))
|
||||
|
||||
|
||||
def _directive_to_pattern(token: str, group_names: set[str]) -> str:
|
||||
header_match = _HEADER_DIRECTIVE_RE.fullmatch(token)
|
||||
if header_match:
|
||||
header_name = header_match.group(1).lower()
|
||||
group = _KNOWN_HEADER_GROUPS.get(header_name, re.sub(r"[^a-z0-9]+", "_", header_name))
|
||||
group_names.add(group)
|
||||
return f'(?P<{group}>[^"]*)'
|
||||
|
||||
if token not in _SIMPLE_DIRECTIVES:
|
||||
raise ValueError(f"Unsupported LogFormat directive: {token!r}")
|
||||
group, pattern = _SIMPLE_DIRECTIVES[token]
|
||||
group_names.add(group)
|
||||
return pattern
|
||||
|
||||
|
||||
def parse_apache_timestamp(raw: str) -> datetime:
|
||||
"""'10/Oct/2026:13:55:36 -0700' -> naive UTC datetime (Ch07: store normalized to UTC)."""
|
||||
dt = datetime.strptime(raw, "%d/%b/%Y:%H:%M:%S %z")
|
||||
return dt.astimezone(timezone.utc).replace(tzinfo=None)
|
||||
|
||||
|
||||
def parse_request_line(raw: str) -> tuple[str, str] | None:
|
||||
"""Split '%r' ("GET /path HTTP/1.1") into (method, path); None if malformed."""
|
||||
parts = raw.split()
|
||||
if len(parts) != 3:
|
||||
return None
|
||||
method, path, _protocol = parts
|
||||
return method, path
|
||||
|
||||
|
||||
def validate_ip(raw: str) -> str | None:
|
||||
"""Return `raw` if a valid IPv4/IPv6 address, else None (Ch07)."""
|
||||
try:
|
||||
ip_address(raw)
|
||||
except ValueError:
|
||||
return None
|
||||
return raw
|
||||
@@ -0,0 +1,10 @@
|
||||
"""LiteSpeed uses the same LogFormat directives as Apache (Chapter 07):
|
||||
Combined generally works unmodified. get_parser_for() still always
|
||||
compiles the vhost's real format_string though — this module is just the
|
||||
named default, never a silent override of a custom format.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from app.services.log_parser.apache import combined_parser as default_litespeed_parser
|
||||
|
||||
__all__ = ["default_litespeed_parser"]
|
||||
@@ -0,0 +1,148 @@
|
||||
"""Core log-file processing pipeline (Chapter 07), factored out of
|
||||
app/cli.py so it has exactly one implementation shared by:
|
||||
- the optional `flask process-logs` CLI command (for anyone who still
|
||||
wants cron), and
|
||||
- the automatic background-thread trigger fired right after upload and
|
||||
opportunistically on page load (app/services/background.py) — the
|
||||
no-cron-required simplification.
|
||||
|
||||
process_one_batch() keeps the same bounded-batch, checkpointed-commit
|
||||
contract as the original cron design (Ch03 rule 1/4/9): it still never
|
||||
loads a whole file into memory, still commits progress incrementally, and
|
||||
still stops after `batch_size` lines so a very large file doesn't hold one
|
||||
enormous open transaction — the only thing that changed is *who* calls it
|
||||
again for the next batch.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import gzip
|
||||
import itertools
|
||||
from datetime import date
|
||||
from pathlib import Path
|
||||
|
||||
from sqlalchemy import insert
|
||||
|
||||
from app.extensions import db
|
||||
from app.models.log_entry import LogEntry
|
||||
from app.models.log_file import LogFile
|
||||
from app.services import aggregator
|
||||
from app.services.classification import classify_entry, write_batch_side_effects
|
||||
from app.services.log_parser import ParsedEntry, get_parser_for
|
||||
from app.utils.upload_paths import upload_path_for
|
||||
|
||||
|
||||
def open_log_stream(path: Path):
|
||||
"""Open a log file, transparently decompressing .gz, one line at a time."""
|
||||
if path.suffix == ".gz":
|
||||
return gzip.open(path, mode="rt", encoding="utf-8", errors="replace")
|
||||
return path.open("r", encoding="utf-8", errors="replace")
|
||||
|
||||
|
||||
def count_total_lines(path: Path) -> int:
|
||||
"""One streaming pass to count lines (bounded memory — never the whole
|
||||
file at once, Ch03 rule 1), so the UI can show a real
|
||||
processed/total percentage instead of an indeterminate spinner.
|
||||
|
||||
This is an extra sequential read of the file beyond the parse pass
|
||||
itself. That cost is now worth paying: processing used to be silently
|
||||
triggered by cron with no live audience watching, but now it runs
|
||||
automatically right after upload while the admin is looking at a
|
||||
progress bar — the UX value of a real percentage justifies the extra
|
||||
I/O pass.
|
||||
"""
|
||||
with open_log_stream(path) as fh:
|
||||
return sum(1 for _ in fh)
|
||||
|
||||
|
||||
def process_one_batch(log_file: LogFile, batch_size: int) -> None:
|
||||
"""Parse up to `batch_size` lines from `log_file`'s current checkpoint.
|
||||
|
||||
Leaves status as "processing" (with progress already committed) if
|
||||
the file isn't finished yet — the caller decides whether to invoke
|
||||
this again: the CLI calls it once per pending file per invocation
|
||||
(unchanged cron-tick semantics); the background thread loops it until
|
||||
the file reaches "done"/"error".
|
||||
"""
|
||||
path = upload_path_for(log_file)
|
||||
if not path.exists():
|
||||
log_file.status = "error"
|
||||
log_file.error_message = f"Upload file missing on disk: {path}"
|
||||
db.session.commit()
|
||||
return
|
||||
|
||||
if log_file.total_lines is None:
|
||||
# First pickup of this file — count once, not on every batch.
|
||||
log_file.total_lines = count_total_lines(path)
|
||||
|
||||
log_file.status = "processing"
|
||||
db.session.commit()
|
||||
parser = get_parser_for(log_file)
|
||||
|
||||
lines_seen_this_run = 0
|
||||
skipped_this_run = 0
|
||||
dates_touched: set[date] = set()
|
||||
|
||||
with open_log_stream(path) as fh:
|
||||
remainder = itertools.islice(fh, log_file.processed_lines, None)
|
||||
batch_entries: list[ParsedEntry] = []
|
||||
|
||||
for line in remainder:
|
||||
entry = parser.parse_line(line)
|
||||
if entry is not None:
|
||||
batch_entries.append(entry)
|
||||
dates_touched.add(entry.timestamp.date())
|
||||
else:
|
||||
skipped_this_run += 1 # malformed line — skip, don't abort the batch (Ch07)
|
||||
|
||||
log_file.processed_lines += 1
|
||||
lines_seen_this_run += 1
|
||||
|
||||
if len(batch_entries) >= 500:
|
||||
_bulk_insert_entries(log_file.id, batch_entries)
|
||||
batch_entries = []
|
||||
|
||||
if lines_seen_this_run >= batch_size:
|
||||
_record_skip_count(log_file, skipped_this_run)
|
||||
db.session.commit() # persist checkpoint; resumable if interrupted
|
||||
return
|
||||
|
||||
if batch_entries:
|
||||
_bulk_insert_entries(log_file.id, batch_entries)
|
||||
|
||||
if dates_touched:
|
||||
aggregator.compute_rollups_for_range(min(dates_touched), max(dates_touched))
|
||||
|
||||
_record_skip_count(log_file, skipped_this_run)
|
||||
log_file.status = "done"
|
||||
db.session.commit()
|
||||
|
||||
|
||||
def _record_skip_count(log_file: LogFile, skipped_this_run: int) -> None:
|
||||
"""Informational only — doesn't touch `status`.
|
||||
|
||||
STOPGAP (flagged in Ch07): log_files has no dedicated skip-counter
|
||||
column, so this overwrites error_message with the latest run's count
|
||||
rather than accumulating across runs.
|
||||
"""
|
||||
if skipped_this_run:
|
||||
log_file.error_message = f"{skipped_this_run} unparsable line(s) skipped in the most recent parse run."
|
||||
|
||||
|
||||
def _bulk_insert_entries(log_file_id: int, entries: list[ParsedEntry]) -> None:
|
||||
"""Bulk-insert log_entries (Ch03 rule 4), classifying is_bot/flagged
|
||||
per line, then derive bot_hits/suspicious_events/ip_registry (Ch07).
|
||||
"""
|
||||
if not entries:
|
||||
return
|
||||
payload = []
|
||||
for e in entries:
|
||||
classification = classify_entry(e)
|
||||
payload.append({
|
||||
"log_file_id": log_file_id, "timestamp": e.timestamp, "ip": e.ip,
|
||||
"method": e.method, "path": e.path, "status_code": e.status_code,
|
||||
"bytes_sent": e.bytes_sent, "referrer": e.referrer, "user_agent": e.user_agent,
|
||||
"is_bot": classification.is_bot, "flagged": classification.flagged,
|
||||
})
|
||||
db.session.execute(insert(LogEntry.__table__), payload)
|
||||
db.session.commit()
|
||||
write_batch_side_effects(log_file_id, entries)
|
||||
@@ -0,0 +1,19 @@
|
||||
"""Referrer -> domain bucketing (Chapter 08 rollup, Method A)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from urllib.parse import urlparse
|
||||
|
||||
|
||||
def referrer_domain(referrer: str | None) -> str | None:
|
||||
"""Bucket a raw referrer URL to its host, stripping a leading 'www.'.
|
||||
|
||||
Full-URL cardinality would make the daily rollup unbounded in row
|
||||
count; domain bucketing keeps it small (Ch03's bounded-resource bias).
|
||||
Empty/unparsable referrers return None and are excluded from the rollup.
|
||||
"""
|
||||
if not referrer:
|
||||
return None
|
||||
host = urlparse(referrer).netloc.lower()
|
||||
if not host:
|
||||
return None
|
||||
return host[4:] if host.startswith("www.") else host
|
||||
@@ -0,0 +1,49 @@
|
||||
"""Severity scoring (Chapter 10): a simple, transparent rank-escalation
|
||||
model — no ML, easy to explain to a non-expert user (Ch10's own requirement).
|
||||
|
||||
Chapter 07 assigns a PROVISIONAL severity per rule type at parse time.
|
||||
This module computes an EFFECTIVE severity by escalating that base rank
|
||||
for repeated/rapid hits from the same IP — the exact rule Ch10 names.
|
||||
|
||||
Escalation is rank-based and strictly non-decreasing: it starts at the
|
||||
base severity's rank and only ever moves up (repeat-count / hit-rate
|
||||
bonuses), clamped at "high". An earlier weighted-score-vs-fixed-threshold
|
||||
version could silently *downgrade* an isolated high-severity event (e.g.
|
||||
a single sqlmap hit) to "medium" purely because the thresholds weren't
|
||||
calibrated to the base weights — caught by running the pipeline against
|
||||
real sample data. A single dangerous event must never end up rated below
|
||||
its own base severity; only repetition/rate should push it higher.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
|
||||
_SEVERITY_RANKS = ["low", "medium", "high"]
|
||||
_RANK_BY_SEVERITY = {name: rank for rank, name in enumerate(_SEVERITY_RANKS)}
|
||||
|
||||
_REPEAT_COUNT_HIGH = 20
|
||||
_REPEAT_COUNT_MEDIUM = 5
|
||||
_RAPID_AVG_INTERVAL_SECONDS = 60 # >1 event/minute from one IP suggests automation
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class SeverityInputs:
|
||||
base_severity: str
|
||||
ip_event_count: int
|
||||
avg_interval_seconds: float | None # None if this IP has < 2 events in the window
|
||||
|
||||
|
||||
def compute_effective_severity(inputs: SeverityInputs) -> str:
|
||||
"""Two named, inspectable escalation rules: repeat-count and hit-rate."""
|
||||
rank = _RANK_BY_SEVERITY.get(inputs.base_severity, 0)
|
||||
|
||||
if inputs.ip_event_count >= _REPEAT_COUNT_HIGH:
|
||||
rank += 2
|
||||
elif inputs.ip_event_count >= _REPEAT_COUNT_MEDIUM:
|
||||
rank += 1
|
||||
|
||||
if inputs.avg_interval_seconds is not None and inputs.avg_interval_seconds < _RAPID_AVG_INTERVAL_SECONDS:
|
||||
rank += 1
|
||||
|
||||
rank = min(rank, len(_SEVERITY_RANKS) - 1)
|
||||
return _SEVERITY_RANKS[rank]
|
||||
@@ -0,0 +1,54 @@
|
||||
"""Suspicious-pattern matching (Chapter 07), reused as-is by Chapter 10.
|
||||
|
||||
Pattern dictionary is data (JSON), same reasoning as the bot signature
|
||||
table — editable without a code change. Severity here is a provisional
|
||||
per-rule-type default; Chapter 10 owns the full weighted/escalating score.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from app.services.log_parser import ParsedEntry
|
||||
|
||||
PATTERNS_PATH = Path(__file__).parent / "data" / "threat_patterns.json"
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ThreatMatch:
|
||||
rule_matched: str
|
||||
severity: str # low|medium|high — provisional; Ch10 may escalate
|
||||
|
||||
|
||||
def _load_patterns() -> dict:
|
||||
return json.loads(PATTERNS_PATH.read_text())
|
||||
|
||||
|
||||
_PATTERNS = _load_patterns()
|
||||
|
||||
|
||||
def scan(entry: "ParsedEntry") -> ThreatMatch | None:
|
||||
"""Check one parsed entry against the pattern dictionary.
|
||||
|
||||
Cheap substring checks only — no regex backtracking risk, no network
|
||||
calls (Ch03: CPU-cheap, explainable rule matching).
|
||||
"""
|
||||
path_lower = entry.path.lower()
|
||||
ua_lower = (entry.user_agent or "").lower()
|
||||
|
||||
for scanner_ua in _PATTERNS["scanner_user_agents"]:
|
||||
if scanner_ua in ua_lower:
|
||||
return ThreatMatch(rule_matched=f"scanner_ua:{scanner_ua}", severity="high")
|
||||
|
||||
for sensitive in _PATTERNS["sensitive_paths"]:
|
||||
if sensitive.lower() in path_lower:
|
||||
return ThreatMatch(rule_matched=f"sensitive_path:{sensitive}", severity="medium")
|
||||
|
||||
for marker in _PATTERNS["injection_markers"]:
|
||||
if marker.lower() in path_lower:
|
||||
return ThreatMatch(rule_matched=f"injection:{marker}", severity="high")
|
||||
|
||||
return None
|
||||
@@ -0,0 +1,54 @@
|
||||
"""Lightweight browser/OS classification for human traffic (Chapter 08).
|
||||
|
||||
Separate from Chapter 07's bot_identifier — that identifies automated
|
||||
crawlers via UA substring + DNS verification. This classifies ordinary
|
||||
browsers/OSes, same ordered-substring-match pattern, signatures as JSON
|
||||
data (not hardcoded), no third-party UA-parsing dependency — consistent
|
||||
with the low-footprint bias in Chapter 03.
|
||||
|
||||
Order matters within each signature list: e.g. Edge/Opera must be checked
|
||||
before Chrome (their UAs also contain "Chrome/"); iOS before macOS (iPhone
|
||||
UAs also contain "like Mac OS X"); Android before Linux (Android UAs also
|
||||
contain "Linux").
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
|
||||
_DATA_DIR = Path(__file__).parent / "data"
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class UASignature:
|
||||
name: str
|
||||
ua_substrings: tuple[str, ...]
|
||||
|
||||
|
||||
def _load(filename: str) -> list[UASignature]:
|
||||
raw = json.loads((_DATA_DIR / filename).read_text())
|
||||
return [UASignature(name=e["name"], ua_substrings=tuple(e["ua_substrings"])) for e in raw]
|
||||
|
||||
|
||||
_BROWSER_SIGNATURES = _load("browser_signatures.json")
|
||||
_OS_SIGNATURES = _load("os_signatures.json")
|
||||
|
||||
|
||||
def _match(user_agent: str, signatures: list[UASignature], default: str) -> str:
|
||||
for sig in signatures:
|
||||
if any(sub in user_agent for sub in sig.ua_substrings):
|
||||
return sig.name
|
||||
return default
|
||||
|
||||
|
||||
def classify_browser(user_agent: str | None) -> str:
|
||||
if not user_agent:
|
||||
return "Unknown"
|
||||
return _match(user_agent, _BROWSER_SIGNATURES, "Other")
|
||||
|
||||
|
||||
def classify_os(user_agent: str | None) -> str:
|
||||
if not user_agent:
|
||||
return "Unknown"
|
||||
return _match(user_agent, _OS_SIGNATURES, "Other")
|
||||
@@ -0,0 +1,67 @@
|
||||
/* app/static/src/css/main.css — Tailwind entry point (Chapter 05) */
|
||||
|
||||
/* Self-hosted fonts (Ch05: build-time only, no runtime CDN dependency).
|
||||
Display face used sparingly for headings/labels; body face for UI
|
||||
chrome; mono face for every data readout (IPs, counts, paths,
|
||||
timestamps) — reinforcing that this app's whole job is turning raw
|
||||
log lines into a readable instrument panel. */
|
||||
@import "@fontsource/space-grotesk/500.css";
|
||||
@import "@fontsource/space-grotesk/700.css";
|
||||
@import "@fontsource/public-sans/400.css";
|
||||
@import "@fontsource/public-sans/500.css";
|
||||
@import "@fontsource/public-sans/600.css";
|
||||
@import "@fontsource/jetbrains-mono/400.css";
|
||||
@import "@fontsource/jetbrains-mono/500.css";
|
||||
|
||||
@import "gridjs/dist/theme/mermaid.css";
|
||||
|
||||
@tailwind base;
|
||||
@tailwind components;
|
||||
@tailwind utilities;
|
||||
|
||||
@layer base {
|
||||
html {
|
||||
color-scheme: light;
|
||||
}
|
||||
html.dark {
|
||||
color-scheme: dark;
|
||||
}
|
||||
body {
|
||||
font-family: theme('fontFamily.sans');
|
||||
}
|
||||
h1, h2, h3, h4, .font-display {
|
||||
font-family: theme('fontFamily.display');
|
||||
}
|
||||
/* Every number/IP/path/timestamp reads like it came straight off the
|
||||
log line — the app's one signature typographic choice. */
|
||||
.font-data {
|
||||
font-family: theme('fontFamily.mono');
|
||||
font-variant-numeric: tabular-nums;
|
||||
}
|
||||
}
|
||||
|
||||
@layer components {
|
||||
/* Grid.js theme override to match the token system in both modes,
|
||||
without forking the whole "mermaid" stylesheet. */
|
||||
.gridjs-wrapper, .gridjs-container {
|
||||
@apply !border-line dark:!border-line-dark !rounded-lg;
|
||||
}
|
||||
.gridjs-table {
|
||||
@apply font-data text-sm;
|
||||
}
|
||||
.gridjs-th {
|
||||
@apply !bg-surface-raised dark:!bg-surface-raised-dark !text-muted dark:!text-muted-dark font-sans !font-medium !text-xs !uppercase !tracking-wide;
|
||||
}
|
||||
.gridjs-td {
|
||||
@apply !bg-surface dark:!bg-surface-dark !text-ink dark:!text-ink-dark !border-line dark:!border-line-dark;
|
||||
}
|
||||
.gridjs-footer {
|
||||
@apply !bg-surface dark:!bg-surface-dark !border-line dark:!border-line-dark;
|
||||
}
|
||||
.gridjs-pagination .gridjs-pages button {
|
||||
@apply !text-ink dark:!text-ink-dark;
|
||||
}
|
||||
.gridjs-search-input {
|
||||
@apply !bg-surface dark:!bg-surface-dark !text-ink dark:!text-ink-dark !border-line dark:!border-line-dark;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,62 @@
|
||||
/**
|
||||
* Shared Chart.js setup (Chapter 05) — avoids duplicating config per chart.
|
||||
* Callers must pass pre-aggregated series only (≤~200 points, per Ch05);
|
||||
* this file never fetches or aggregates raw rows itself.
|
||||
*
|
||||
* Dark-mode aware: reads the current theme at init time and colors grid
|
||||
* lines/ticks/legend accordingly. Charts drawn before a theme toggle
|
||||
* don't live-repaint — they pick up the new theme on their next redraw
|
||||
* (date-range change, tab revisit), which keeps this simple.
|
||||
*/
|
||||
import { Chart, registerables } from 'chart.js';
|
||||
|
||||
Chart.register(...registerables);
|
||||
|
||||
// Shared categorical palette, drawn from the Kavosh token system rather
|
||||
// than Chart.js's stock colors.
|
||||
export const CHART_PALETTE = ['#0E7C86', '#B8860B', '#C4372F', '#3F7D5C', '#5B6663', '#8A6FBE'];
|
||||
|
||||
function isDarkMode() {
|
||||
return document.documentElement.classList.contains('dark');
|
||||
}
|
||||
|
||||
function defaultOptions() {
|
||||
const dark = isDarkMode();
|
||||
const textColor = dark ? '#96A19D' : '#5B6663';
|
||||
const gridColor = dark ? '#2C3336' : '#E1E5E2';
|
||||
const tickFont = { family: '"JetBrains Mono", monospace', size: 11 };
|
||||
|
||||
return {
|
||||
responsive: true,
|
||||
maintainAspectRatio: false,
|
||||
animation: { duration: 200 },
|
||||
plugins: {
|
||||
legend: {
|
||||
display: true,
|
||||
position: 'bottom',
|
||||
labels: { color: textColor, font: { family: '"Public Sans", sans-serif', size: 12 }, boxWidth: 12, padding: 12 },
|
||||
},
|
||||
},
|
||||
scales: {
|
||||
x: { ticks: { color: textColor, font: tickFont }, grid: { color: gridColor } },
|
||||
y: { ticks: { color: textColor, font: tickFont }, grid: { color: gridColor } },
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
/** @param {string} canvasId @param {object} config @returns {Chart} */
|
||||
export function initChart(canvasId, config) {
|
||||
const ctx = document.getElementById(canvasId);
|
||||
if (!ctx) throw new Error(`initChart: no element #${canvasId}`);
|
||||
const base = defaultOptions();
|
||||
const callerOptions = config.options ?? {};
|
||||
return new Chart(ctx, {
|
||||
...config,
|
||||
options: {
|
||||
...base,
|
||||
...callerOptions,
|
||||
scales: config.type === 'doughnut' || config.type === 'pie' ? undefined : { ...base.scales, ...callerOptions.scales },
|
||||
plugins: { ...base.plugins, ...callerOptions.plugins },
|
||||
},
|
||||
});
|
||||
}
|
||||
@@ -0,0 +1,38 @@
|
||||
/**
|
||||
* Shared Grid.js setup (Chapter 05) — every table is server-side
|
||||
* paginated/sorted/searched against `?page=&per_page=&sort=&q=`.
|
||||
*/
|
||||
import { Grid } from 'gridjs';
|
||||
|
||||
/**
|
||||
* @param {string} elementId container element id
|
||||
* @param {string} endpoint JSON endpoint, e.g. '/api/overview/top-urls'
|
||||
* @param {Array} columns Grid.js column defs
|
||||
* @param {object} extraParams static extra query params (e.g. {from, to})
|
||||
*/
|
||||
export function initGrid(elementId, endpoint, columns, extraParams = {}) {
|
||||
const buildUrl = (prev, params) => {
|
||||
const url = new URL(prev, window.location.origin);
|
||||
Object.entries({ ...extraParams, ...params }).forEach(([k, v]) => url.searchParams.set(k, v));
|
||||
return url.toString();
|
||||
};
|
||||
|
||||
return new Grid({
|
||||
columns,
|
||||
server: {
|
||||
url: endpoint,
|
||||
then: (res) => res.data.rows,
|
||||
total: (res) => res.data.total,
|
||||
},
|
||||
pagination: {
|
||||
server: { url: (prev, page, per_page) => buildUrl(prev, { page: page + 1, per_page }) },
|
||||
limit: 20,
|
||||
},
|
||||
sort: {
|
||||
server: {
|
||||
url: (prev, cols) => (cols.length ? buildUrl(prev, { sort: cols[0].direction === 1 ? cols[0].index : `-${cols[0].index}` }) : prev),
|
||||
},
|
||||
},
|
||||
search: { server: { url: (prev, keyword) => buildUrl(prev, { q: keyword }) } },
|
||||
}).render(document.getElementById(elementId));
|
||||
}
|
||||
@@ -0,0 +1,36 @@
|
||||
/**
|
||||
* Vite entry point: HTMX + Alpine init + Chart.js/Grid.js glue (Chapter 05).
|
||||
* Bundled at build time — no CDN scripts, nothing compiled at request time.
|
||||
*/
|
||||
import htmx from 'htmx.org';
|
||||
import Alpine from 'alpinejs';
|
||||
import { html as gridHtml } from 'gridjs';
|
||||
|
||||
import { initChart, CHART_PALETTE } from './charts.js';
|
||||
import { initGrid } from './grids.js';
|
||||
import { initUploadForm } from './uploads.js';
|
||||
|
||||
window.htmx = htmx;
|
||||
window.Alpine = Alpine;
|
||||
window.initChart = initChart;
|
||||
window.initGrid = initGrid;
|
||||
window.gridHtml = gridHtml;
|
||||
window.initUploadForm = initUploadForm;
|
||||
window.KAVOSH_CHART_PALETTE = CHART_PALETTE;
|
||||
|
||||
// Chapter 12: HTMX requests must carry the CSRF token as a header, not
|
||||
// rely on the cookie alone.
|
||||
document.body.addEventListener('htmx:configRequest', (event) => {
|
||||
const token = document.querySelector('meta[name="csrf-token"]')?.content;
|
||||
if (token) event.detail.headers['X-CSRFToken'] = token;
|
||||
});
|
||||
|
||||
// Dark mode toggle (persisted; the initial class is set synchronously in
|
||||
// an inline <head> script in base.html to avoid a flash of the wrong
|
||||
// theme before this bundle loads).
|
||||
window.toggleTheme = function toggleTheme() {
|
||||
const isDark = document.documentElement.classList.toggle('dark');
|
||||
localStorage.setItem('kavosh-theme', isDark ? 'dark' : 'light');
|
||||
};
|
||||
|
||||
Alpine.start();
|
||||
@@ -0,0 +1,62 @@
|
||||
/**
|
||||
* Upload with real byte-transfer progress (Chapter 08/12).
|
||||
*
|
||||
* fetch() can't reliably report upload progress across browsers, so this
|
||||
* uses XMLHttpRequest directly rather than depending on htmx's internal
|
||||
* transport. On completion, the returned HTML fragment (queued/duplicate/
|
||||
* error) is injected manually, then handed to htmx.process() so any
|
||||
* hx-* polling attributes inside it (e.g. the processing-status poll)
|
||||
* activate normally, exactly as if htmx had performed the swap itself.
|
||||
*/
|
||||
export function initUploadForm(formId) {
|
||||
const form = document.getElementById(formId);
|
||||
if (!form) return;
|
||||
|
||||
const fileInput = form.querySelector('input[type="file"]');
|
||||
const progressWrap = document.getElementById('upload-progress-wrap');
|
||||
const progressBar = document.getElementById('upload-progress-bar');
|
||||
const progressLabel = document.getElementById('upload-progress-label');
|
||||
const resultEl = document.getElementById('upload-result');
|
||||
const submitBtn = form.querySelector('button[type="submit"]');
|
||||
|
||||
form.addEventListener('submit', (event) => {
|
||||
event.preventDefault();
|
||||
if (!fileInput.files.length) return;
|
||||
|
||||
const formData = new FormData(form);
|
||||
const xhr = new XMLHttpRequest();
|
||||
|
||||
submitBtn.disabled = true;
|
||||
resultEl.innerHTML = '';
|
||||
progressWrap.classList.remove('hidden');
|
||||
progressBar.style.width = '0%';
|
||||
progressLabel.textContent = 'Uploading… 0%';
|
||||
|
||||
xhr.upload.addEventListener('progress', (e) => {
|
||||
if (!e.lengthComputable) return;
|
||||
const pct = Math.round((e.loaded / e.total) * 100);
|
||||
progressBar.style.width = pct + '%';
|
||||
progressLabel.textContent = `Uploading… ${pct}%`;
|
||||
});
|
||||
|
||||
xhr.addEventListener('load', () => {
|
||||
submitBtn.disabled = false;
|
||||
progressWrap.classList.add('hidden');
|
||||
resultEl.innerHTML = xhr.responseText;
|
||||
if (window.htmx) window.htmx.process(resultEl);
|
||||
form.reset();
|
||||
});
|
||||
|
||||
xhr.addEventListener('error', () => {
|
||||
submitBtn.disabled = false;
|
||||
progressWrap.classList.add('hidden');
|
||||
resultEl.innerHTML =
|
||||
'<p class="text-danger dark:text-danger-dark text-sm">Upload failed — check your connection and try again.</p>';
|
||||
});
|
||||
|
||||
xhr.open('POST', form.getAttribute('action') || '/uploads');
|
||||
const token = document.querySelector('meta[name="csrf-token"]')?.content;
|
||||
if (token) xhr.setRequestHeader('X-CSRFToken', token);
|
||||
xhr.send(formData);
|
||||
});
|
||||
}
|
||||
@@ -0,0 +1,85 @@
|
||||
<!doctype html>
|
||||
<html lang="en" class="h-full">
|
||||
<head>
|
||||
<meta charset="utf-8">
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1">
|
||||
<meta name="csrf-token" content="{{ csrf_token() }}">
|
||||
<title>{% block title %}Kavosh{% endblock %}</title>
|
||||
<!--
|
||||
Anti-FOUC: set the theme class synchronously, before the stylesheet
|
||||
and before Alpine/htmx load, so there's no flash of the wrong theme
|
||||
on first paint. Kept as a tiny inline script rather than waiting for
|
||||
the bundled JS, which only arrives after the page has started
|
||||
rendering.
|
||||
-->
|
||||
<script>
|
||||
(function () {
|
||||
var stored = localStorage.getItem('kavosh-theme');
|
||||
var isDark = stored ? stored === 'dark' : window.matchMedia('(prefers-color-scheme: dark)').matches;
|
||||
document.documentElement.classList.toggle('dark', isDark);
|
||||
})();
|
||||
</script>
|
||||
<link rel="stylesheet" href="{{ asset('app/static/src/css/main.css') }}">
|
||||
</head>
|
||||
<body class="h-full bg-paper dark:bg-paper-dark text-ink dark:text-ink-dark antialiased">
|
||||
<div id="app-shell" x-data="{ mobileNavOpen: false }">
|
||||
<header class="border-b border-line dark:border-line-dark bg-surface dark:bg-surface-dark">
|
||||
<nav class="flex gap-1 px-4 py-3 items-center max-w-6xl mx-auto">
|
||||
<span class="font-display font-bold text-lg tracking-tight mr-4">Kavosh</span>
|
||||
|
||||
{% if current_user.is_authenticated %}
|
||||
{% set tabs = [
|
||||
('overview.overview', '/overview', 'house', 'Overview'),
|
||||
('seo.seo', '/seo', 'search', 'SEO & Bots'),
|
||||
('security.security', '/security', 'shield-alert', 'Security'),
|
||||
] %}
|
||||
{% for endpoint, path, icon, label in tabs %}
|
||||
<a href="{{ path }}" hx-get="{{ path }}" hx-target="#main-content" hx-push-url="true"
|
||||
class="flex items-center gap-1.5 px-3 py-1.5 rounded-md text-sm font-medium transition-colors
|
||||
{% if request.endpoint == endpoint %}
|
||||
bg-accent/10 text-accent dark:bg-accent-dark/15 dark:text-accent-dark
|
||||
{% else %}
|
||||
text-muted dark:text-muted-dark hover:text-ink dark:hover:text-ink-dark hover:bg-paper dark:hover:bg-paper-dark
|
||||
{% endif %}">
|
||||
<svg class="w-4 h-4 shrink-0"><use href="/static/dist/icons.svg#{{ icon }}"/></svg>
|
||||
<span class="hidden sm:inline">{{ label }}</span>
|
||||
</a>
|
||||
{% endfor %}
|
||||
|
||||
<div class="ml-auto flex items-center gap-2">
|
||||
<button type="button" onclick="toggleTheme()"
|
||||
aria-label="Toggle dark mode"
|
||||
class="p-2 rounded-md text-muted dark:text-muted-dark hover:text-ink dark:hover:text-ink-dark hover:bg-paper dark:hover:bg-paper-dark transition-colors">
|
||||
<svg class="w-4 h-4 block dark:hidden"><use href="/static/dist/icons.svg#moon"/></svg>
|
||||
<svg class="w-4 h-4 hidden dark:block"><use href="/static/dist/icons.svg#sun"/></svg>
|
||||
</button>
|
||||
<span class="text-sm text-muted dark:text-muted-dark font-data hidden md:inline">{{ current_user.email }}</span>
|
||||
<form method="post" action="{{ url_for('auth.logout') }}">
|
||||
<input type="hidden" name="csrf_token" value="{{ csrf_token() }}">
|
||||
<button type="submit" title="Log out"
|
||||
class="p-2 rounded-md text-muted dark:text-muted-dark hover:text-danger dark:hover:text-danger-dark hover:bg-paper dark:hover:bg-paper-dark transition-colors">
|
||||
<svg class="w-4 h-4"><use href="/static/dist/icons.svg#log-out"/></svg>
|
||||
</button>
|
||||
</form>
|
||||
</div>
|
||||
{% else %}
|
||||
<div class="ml-auto">
|
||||
<button type="button" onclick="toggleTheme()" aria-label="Toggle dark mode"
|
||||
class="p-2 rounded-md text-muted dark:text-muted-dark hover:text-ink dark:hover:text-ink-dark hover:bg-paper dark:hover:bg-paper-dark transition-colors">
|
||||
<svg class="w-4 h-4 block dark:hidden"><use href="/static/dist/icons.svg#moon"/></svg>
|
||||
<svg class="w-4 h-4 hidden dark:block"><use href="/static/dist/icons.svg#sun"/></svg>
|
||||
</button>
|
||||
</div>
|
||||
{% endif %}
|
||||
</nav>
|
||||
</header>
|
||||
|
||||
<!-- HTMX target swapped by the nav above (Ch01: single-page shell, no full reloads) -->
|
||||
<main id="main-content" class="max-w-6xl mx-auto p-4 sm:p-6">
|
||||
{% block content %}{% endblock %}
|
||||
</main>
|
||||
</div>
|
||||
|
||||
<script type="module" src="{{ asset('app/static/src/js/main.js') }}"></script>
|
||||
</body>
|
||||
</html>
|
||||
@@ -0,0 +1,39 @@
|
||||
"""Vite manifest reader — maps a source entry to its hashed dist path (Ch05)."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
from flask import Flask
|
||||
|
||||
|
||||
class ManifestNotFound(RuntimeError):
|
||||
"""app/static/dist/manifest.json is missing — `npm run build` hasn't been run."""
|
||||
|
||||
|
||||
def _load_manifest(static_dist_dir: Path) -> dict:
|
||||
for candidate in (static_dist_dir / ".vite" / "manifest.json", static_dist_dir / "manifest.json"):
|
||||
if candidate.exists():
|
||||
return json.loads(candidate.read_text())
|
||||
raise ManifestNotFound(f"No Vite manifest under {static_dist_dir} — run `npm run build`.")
|
||||
|
||||
|
||||
def register_asset_helper(app: Flask) -> None:
|
||||
"""Register `asset(name)` as a Jinja global (Chapter 05)."""
|
||||
static_dist_dir = Path(app.static_folder) / "dist"
|
||||
|
||||
@app.context_processor
|
||||
def inject_asset_helper():
|
||||
def asset(name: str) -> str:
|
||||
try:
|
||||
manifest = _load_manifest(static_dist_dir)
|
||||
except ManifestNotFound:
|
||||
if app.debug:
|
||||
return f"/static/dist/{name}" # tolerate an unbuilt dev checkout
|
||||
raise
|
||||
entry = manifest.get(name)
|
||||
if entry is None:
|
||||
raise KeyError(f"{name!r} not in Vite manifest.")
|
||||
return f"/static/dist/{entry['file']}"
|
||||
|
||||
return {"asset": asset}
|
||||
@@ -0,0 +1,32 @@
|
||||
"""Shared from/to date-range parsing (Chapter 11: consistent across every
|
||||
chart/KPI endpoint; never defaults to "all time" per Chapter 03 rule 7).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import date, datetime, timedelta
|
||||
|
||||
from flask import Request
|
||||
|
||||
DEFAULT_RANGE_DAYS = 7 # within Ch11's stated "7 or 30 day" default allowance
|
||||
|
||||
|
||||
def parse_date_range(request: Request) -> tuple[date, date]:
|
||||
"""Parse `?from=&to=` (ISO 8601 dates), defaulting to the last 7 days."""
|
||||
to_raw = request.args.get("to")
|
||||
from_raw = request.args.get("from")
|
||||
to_date = date.fromisoformat(to_raw) if to_raw else date.today()
|
||||
from_date = date.fromisoformat(from_raw) if from_raw else to_date - timedelta(days=DEFAULT_RANGE_DAYS - 1)
|
||||
if from_date > to_date:
|
||||
from_date, to_date = to_date, from_date
|
||||
return from_date, to_date
|
||||
|
||||
|
||||
def day_bounds(from_date: date, to_date: date) -> tuple[datetime, datetime]:
|
||||
"""Inclusive [from_date, to_date] -> half-open [start, end) datetime range.
|
||||
|
||||
Shared by overview/queries.py, seo/queries.py, and security/queries.py
|
||||
(consolidated here rather than each keeping its own local copy).
|
||||
"""
|
||||
start = datetime.combine(from_date, datetime.min.time())
|
||||
end = datetime.combine(to_date, datetime.min.time()) + timedelta(days=1)
|
||||
return start, end
|
||||
@@ -0,0 +1,18 @@
|
||||
"""Shared JSON envelope helpers (Chapter 11).
|
||||
|
||||
Every existing /api/... endpoint already hand-builds this exact shape via
|
||||
jsonify(data=..., meta=...) — audited for compliance in docs/api-contract-
|
||||
final.md. This module exists so NEW endpoints (e.g. this chapter's
|
||||
unauthorized_handler) have one call site instead of re-typing the shape.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from flask import jsonify
|
||||
|
||||
|
||||
def api_ok(data, meta: dict | None = None):
|
||||
return jsonify(data=data, meta=meta or {})
|
||||
|
||||
|
||||
def api_error(code: str, message: str, status: int):
|
||||
return jsonify(error={"code": code, "message": message}), status
|
||||
@@ -0,0 +1,20 @@
|
||||
"""Shared HTMX full-page-vs-fragment response helper (Chapter 04).
|
||||
|
||||
Every blueprint serving a dashboard tab (Ch08/09/10) uses this instead of
|
||||
duplicating the `if request.headers.get("HX-Request")` check, so the
|
||||
convention is identical everywhere.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from flask import Request, render_template
|
||||
|
||||
|
||||
def is_htmx(request: Request) -> bool:
|
||||
"""True if this request was triggered by an hx-* attribute."""
|
||||
return request.headers.get("HX-Request", "").lower() == "true"
|
||||
|
||||
|
||||
def render_htmx_aware(request: Request, *, full_template: str, partial_template: str, **ctx):
|
||||
"""Render `full_template` on first load, `partial_template` on HTMX swaps."""
|
||||
template = partial_template if is_htmx(request) else full_template
|
||||
return render_template(template, **ctx)
|
||||
@@ -0,0 +1,14 @@
|
||||
"""Shared HTTP status-code bucketing — used by the aggregator (per-IP
|
||||
rollup) and available for any endpoint needing the same 2xx/3xx/4xx/5xx
|
||||
buckets, so there's one bucketing rule, not several copies."""
|
||||
from __future__ import annotations
|
||||
|
||||
|
||||
def status_bucket(status_code: int) -> str:
|
||||
if status_code < 300:
|
||||
return "2xx"
|
||||
if status_code < 400:
|
||||
return "3xx"
|
||||
if status_code < 500:
|
||||
return "4xx"
|
||||
return "5xx"
|
||||
@@ -0,0 +1,14 @@
|
||||
"""Shared page/per_page parsing (Chapter 11: consistent across every
|
||||
Grid.js-backed endpoint)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from flask import Request
|
||||
|
||||
DEFAULT_PER_PAGE = 20
|
||||
MAX_PER_PAGE = 100
|
||||
|
||||
|
||||
def parse_pagination(request: Request) -> tuple[int, int]:
|
||||
page = max(1, request.args.get("page", 1, type=int))
|
||||
per_page = request.args.get("per_page", DEFAULT_PER_PAGE, type=int)
|
||||
return page, max(1, min(per_page, MAX_PER_PAGE))
|
||||
@@ -0,0 +1,20 @@
|
||||
"""Shared upload-path helper (Chapter 04) — used by the uploads blueprint,
|
||||
the optional CLI processor, and the automatic background-thread processor,
|
||||
so there's one definition of where a given LogFile's raw upload lives on
|
||||
disk. Pulled out of the uploads blueprint so services/ doesn't have to
|
||||
import from blueprints/ (the wrong direction) to reach it.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
from flask import current_app
|
||||
|
||||
from app.models.log_file import LogFile
|
||||
|
||||
|
||||
def upload_path_for(log_file: LogFile) -> Path:
|
||||
"""Deterministic on-disk path for a LogFile row — avoids a new DB column."""
|
||||
ext = Path(log_file.filename).suffix.lower()
|
||||
upload_dir = Path(current_app.config["UPLOAD_DIR"])
|
||||
return upload_dir / f"{log_file.id}{ext}"
|
||||
Reference in New Issue
Block a user