start project
This commit is contained in:
@@ -0,0 +1,25 @@
|
||||
# Copy to .env for local dev (python-dotenv). In production, set these via
|
||||
# cPanel's Python App "Environment Variables" UI — never commit a real .env.
|
||||
|
||||
FLASK_ENV=development
|
||||
SECRET_KEY=change-me
|
||||
DATABASE_URL=sqlite:///kavosh.db
|
||||
DB_POOL_SIZE=3
|
||||
DB_POOL_RECYCLE_SECONDS=280
|
||||
|
||||
CACHE_TYPE=FileSystemCache
|
||||
CACHE_DIR=/tmp/kavosh-cache
|
||||
CACHE_DEFAULT_TIMEOUT=60
|
||||
|
||||
SESSION_COOKIE_SECURE=true
|
||||
SESSION_LIFETIME_HOURS=12
|
||||
|
||||
UPLOAD_MAX_SIZE_MB=500
|
||||
UPLOAD_DIR=/home/kavosh/uploads
|
||||
PARSE_BATCH_SIZE=5000
|
||||
|
||||
ADMIN_EMAIL=admin@example.com
|
||||
|
||||
# Optional (Chapter 12): only used if you want a rotating-file logging
|
||||
# fallback in addition to stdout — leave unset to log to stdout only.
|
||||
# LOG_FALLBACK_FILE=/home/kavosh/logs/kavosh.log
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
__pycache__/
|
||||
*.pyc
|
||||
.env
|
||||
*.db
|
||||
instance/
|
||||
/app/static/dist/*
|
||||
!/app/static/dist/.gitkeep
|
||||
/node_modules/
|
||||
.pytest_cache/
|
||||
*.egg-info/
|
||||
.DS_Store
|
||||
@@ -0,0 +1,77 @@
|
||||
# Kavosh — cPanel Deployment Checklist
|
||||
|
||||
No cron job is required to run Kavosh. Log parsing starts automatically
|
||||
in the background as soon as a file is uploaded, and any file left
|
||||
incomplete (e.g. a worker process recycled mid-parse) resumes the next
|
||||
time the Overview tab is loaded. See "Optional: cron" at the bottom if
|
||||
you'd still like the extra resilience of a scheduled fallback.
|
||||
|
||||
1. **Setup Python App** (cPanel) — create the app; note the virtualenv
|
||||
path and set the startup file to `passenger_wsgi.py`.
|
||||
2. Activate the generated virtualenv and install dependencies:
|
||||
```
|
||||
pip install -r requirements.txt
|
||||
```
|
||||
(Dev-only tools — `pytest` etc. — live in `requirements-dev.txt` and
|
||||
are NOT installed on the host, per Ch02 factor 5's build/release/run
|
||||
separation.)
|
||||
3. Set every variable in `.env.example` via the Python App's
|
||||
**Environment Variables** UI — never a committed `.env` in production.
|
||||
4. Create the schema:
|
||||
```
|
||||
flask db upgrade
|
||||
```
|
||||
5. Bootstrap the single admin account (interactive password prompt keeps
|
||||
the raw credential out of process env/config):
|
||||
```
|
||||
flask create-admin --email you@example.com
|
||||
```
|
||||
6. Build frontend assets **locally or in CI**, not on the host:
|
||||
```
|
||||
npm install && npm run build
|
||||
```
|
||||
Deploy only the resulting `app/static/dist/` output alongside the
|
||||
Python app — Node never runs on the server (Chapter 05).
|
||||
7. Configure static file mapping (`.htaccess` or the panel's static-file
|
||||
rule) so `app/static/dist/*` bypasses Python entirely (Chapter 03).
|
||||
8. Verify `GET /healthz` returns `{"data": {"status": "ok"}, ...}`, then
|
||||
log in and upload a file — it should start analyzing within a second
|
||||
or two, with no further setup needed.
|
||||
|
||||
## Optional: cron
|
||||
|
||||
Two commands remain available for anyone who wants the extra resilience
|
||||
of a scheduled fallback instead of relying solely on automatic/on-visit
|
||||
processing:
|
||||
|
||||
```
|
||||
*/5 * * * * cd /home/YOURUSER/kavosh && /home/YOURUSER/virtualenv/kavosh/3.11/bin/flask process-logs >> /home/YOURUSER/logs/kavosh-process-logs.log 2>&1
|
||||
0 3 * * * cd /home/YOURUSER/kavosh && /home/YOURUSER/virtualenv/kavosh/3.11/bin/flask cleanup >> /home/YOURUSER/logs/kavosh-cleanup.log 2>&1
|
||||
```
|
||||
|
||||
`cleanup` enforces the Chapter 06 retention policy (below) — nothing in
|
||||
the UI depends on it having run; without it, old raw data simply
|
||||
accumulates instead of being pruned. `process-logs` is a fallback for
|
||||
the rare case a background thread dies before finishing (see
|
||||
`app/services/background.py` for why that can happen and how the app
|
||||
recovers without cron anyway).
|
||||
|
||||
## Local development
|
||||
```
|
||||
cp .env.example .env # fill in real values
|
||||
pip install -r requirements.txt -r requirements-dev.txt
|
||||
npm install && npm run build # or `npm run dev` while iterating on frontend
|
||||
flask db upgrade
|
||||
flask create-admin --email you@example.com
|
||||
flask run
|
||||
pytest # run the test suite
|
||||
```
|
||||
|
||||
## Retention policy enforced by `flask cleanup` (Chapter 06/12, optional)
|
||||
| Table | Retention |
|
||||
|---|---|
|
||||
| `log_entries` | ~30 days |
|
||||
| `request_stats_hourly` | ~90 days (daily rollup already retained indefinitely) |
|
||||
| `ip_path_stats_daily` / `ip_status_stats_daily` | ~30 days |
|
||||
| Raw uploaded log files | deleted once `status="done"` |
|
||||
| Everything else (`request_stats_daily`, `bot_hits`, `suspicious_events`, `referrer_stats_daily`, `browser_stats_daily`, `human_path_stats_daily`, `blocklist_suggestions`) | indefinite — small, bounded-cardinality row counts |
|
||||
@@ -0,0 +1,73 @@
|
||||
# Kavosh
|
||||
|
||||
Self-hosted dashboard that ingests Apache/LiteSpeed access logs and turns
|
||||
them into three sections — Overview, SEO & Bot Behavior, and Suspicious
|
||||
Requests/IP History — engineered to run inside a constrained cPanel
|
||||
shared-hosting account (4 CPU cores, 60 entry processes, 2GB RAM, 1,024
|
||||
IOPS, 16MB/s I/O, 150 processes, 150 DB connections).
|
||||
|
||||
Stack: Flask (application factory + blueprints) + HTMX/Alpine.js +
|
||||
Tailwind + Chart.js + Grid.js, served via Passenger, SQLite in WAL mode.
|
||||
No cron job is required — uploads process automatically in the
|
||||
background, with light/dark mode and live progress bars for both the
|
||||
upload and analysis phases.
|
||||
|
||||
## Project docs
|
||||
- `DEPLOYMENT.md` — cPanel deployment checklist + local dev setup (no cron needed)
|
||||
- `docs/api-contract-final.md` — consolidated route table + envelope audit
|
||||
- `.env.example` — every environment variable the app reads
|
||||
- `migrations/versions/0001`–`0006` — schema history, in order
|
||||
|
||||
## Layout
|
||||
```
|
||||
app/
|
||||
blueprints/ overview, seo, security, uploads, auth, api (routes)
|
||||
models/ SQLAlchemy models — one file per table
|
||||
services/ log parsing, bot/threat classification, aggregation,
|
||||
background.py (no-cron auto-processing)
|
||||
static/src Tailwind/JS source (built via Vite -> static/dist)
|
||||
templates Jinja base + HTMX partials
|
||||
migrations/ Alembic schema history
|
||||
tests/ pytest suite (parser + classifier + route smoke tests,
|
||||
real log-file fixtures under tests/fixtures/)
|
||||
```
|
||||
|
||||
## Quick start (local dev)
|
||||
```
|
||||
cp .env.example .env # fill in real values
|
||||
pip install -r requirements.txt -r requirements-dev.txt
|
||||
npm install && npm run build # or `npm run dev` while iterating on frontend
|
||||
flask db upgrade
|
||||
flask create-admin --email you@example.com
|
||||
flask run
|
||||
pytest
|
||||
```
|
||||
|
||||
Visit `/` — it redirects to `/overview` if you're logged in, or `/login`
|
||||
otherwise. Upload a log file and it starts analyzing immediately; no
|
||||
cron job or manual CLI command required (see `DEPLOYMENT.md` if you want
|
||||
the optional cron fallback anyway). The Overview page also lists every
|
||||
uploaded file with select-and-delete — deleting a file removes its raw
|
||||
data and correctly recomputes any shared date rollups, rather than just
|
||||
subtracting the file's contribution naively (see `app/services/
|
||||
file_deletion.py`).
|
||||
|
||||
## Design system
|
||||
- Colors/fonts are defined as Tailwind tokens in `tailwind.config.js`
|
||||
(`paper`/`surface`/`ink`/`muted`/`accent`/`danger`/`warn`/`ok`, each
|
||||
with a `-dark` counterpart) — not hardcoded `slate-*` classes.
|
||||
- Three type roles: `font-display` (Space Grotesk, headings), `font-sans`
|
||||
(Public Sans, UI chrome), `font-data` (JetBrains Mono, every number/IP/
|
||||
path/timestamp — the app's one deliberate signature touch, since the
|
||||
whole product is "raw log lines turned into a readout").
|
||||
- Dark mode toggles a `.dark` class on `<html>`, persisted in
|
||||
`localStorage`, set synchronously in `base.html`'s `<head>` to avoid a
|
||||
flash of the wrong theme on load.
|
||||
|
||||
## Notes on how this was built
|
||||
Built chapter-by-chapter against a 12-chapter project spec, then extended
|
||||
per project-owner follow-up requests (removing the cron requirement,
|
||||
dark mode, upload/analysis progress bars, root-URL auth redirect). Where
|
||||
an implementation choice extended or deviated from the original spec, it's
|
||||
marked inline with a comment — search for "flagged", "ASSUMPTION", or
|
||||
"TRADEOFF" to find every one of them.
|
||||
@@ -0,0 +1,84 @@
|
||||
"""Application factory (Chapter 02 / Chapter 04)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from flask import Flask, jsonify, redirect, request, url_for
|
||||
from flask_login import current_user
|
||||
from sqlalchemy import text
|
||||
|
||||
from app.budget import validate_budget_config
|
||||
from app.config import get_config
|
||||
from app.extensions import cache, csrf, db, limiter, login_manager, migrate
|
||||
from app.logging_setup import configure_logging
|
||||
from app.utils.assets import register_asset_helper
|
||||
|
||||
|
||||
def create_app(config_name: str | None = None) -> Flask:
|
||||
"""Build and configure the Flask application instance."""
|
||||
app = Flask(__name__)
|
||||
app.config.from_object(get_config(config_name))
|
||||
validate_budget_config(app) # Chapter 03: fail loudly on out-of-budget config
|
||||
configure_logging(app) # Chapter 02 factor 11 / Chapter 12
|
||||
|
||||
if not app.config.get("SECRET_KEY") and not app.testing:
|
||||
raise RuntimeError("SECRET_KEY must be set via environment variable")
|
||||
|
||||
db.init_app(app)
|
||||
cache.init_app(app)
|
||||
csrf.init_app(app)
|
||||
login_manager.init_app(app)
|
||||
migrate.init_app(app, db)
|
||||
limiter.init_app(app) # Chapter 12: rate limiting
|
||||
|
||||
register_blueprints(app)
|
||||
register_cli(app)
|
||||
register_asset_helper(app)
|
||||
|
||||
@login_manager.unauthorized_handler
|
||||
def handle_unauthorized():
|
||||
"""Chapter 11 envelope compliance: JSON 401 for /api/... paths
|
||||
instead of Flask-Login's default redirect (which would hand back
|
||||
login HTML to a fetch()/HTMX JSON caller). Full-page routes still
|
||||
redirect to the login form as normal.
|
||||
"""
|
||||
if request.path.startswith("/api/"):
|
||||
return jsonify(error={"code": "unauthorized", "message": "Authentication required."}), 401
|
||||
return redirect(url_for("auth.login", next=request.path))
|
||||
|
||||
@app.get("/healthz")
|
||||
def healthz():
|
||||
"""Liveness/readiness check (Chapter 12) — DB reachable, no long-running work."""
|
||||
db.session.execute(text("SELECT 1"))
|
||||
return {"data": {"status": "ok"}, "meta": {}}
|
||||
|
||||
@app.get("/")
|
||||
def index():
|
||||
"""Root URL: authenticated -> Overview, otherwise -> the login page."""
|
||||
if current_user.is_authenticated:
|
||||
return redirect(url_for("overview.overview"))
|
||||
return redirect(url_for("auth.login"))
|
||||
|
||||
return app
|
||||
|
||||
|
||||
def register_blueprints(app: Flask) -> None:
|
||||
"""Register one blueprint per dashboard section + support area (Ch. 04)."""
|
||||
from app.blueprints.api import bp as api_bp
|
||||
from app.blueprints.auth import bp as auth_bp
|
||||
from app.blueprints.overview import bp as overview_bp
|
||||
from app.blueprints.security import bp as security_bp
|
||||
from app.blueprints.seo import bp as seo_bp
|
||||
from app.blueprints.uploads import bp as uploads_bp
|
||||
|
||||
app.register_blueprint(overview_bp)
|
||||
app.register_blueprint(seo_bp)
|
||||
app.register_blueprint(security_bp)
|
||||
app.register_blueprint(uploads_bp)
|
||||
app.register_blueprint(api_bp)
|
||||
app.register_blueprint(auth_bp)
|
||||
|
||||
|
||||
def register_cli(app: Flask) -> None:
|
||||
"""Attach `flask <command>` admin processes (factor 12)."""
|
||||
from app.cli import register_commands
|
||||
|
||||
register_commands(app)
|
||||
@@ -0,0 +1,7 @@
|
||||
"""Deliberately minimal — Chapter 02's tree lists a standalone `api`
|
||||
blueprint, but Chapter 04/11 route each section's JSON endpoints through
|
||||
its own blueprint instead. Left unpopulated per the project owner's
|
||||
instruction to defer this branch."""
|
||||
from flask import Blueprint
|
||||
|
||||
bp = Blueprint("api", __name__)
|
||||
@@ -0,0 +1,5 @@
|
||||
from flask import Blueprint
|
||||
|
||||
bp = Blueprint("auth", __name__, template_folder="templates")
|
||||
|
||||
from app.blueprints.auth import routes # noqa: E402,F401 registers routes
|
||||
@@ -0,0 +1,18 @@
|
||||
"""Login form (Chapter 12). CSRF handled automatically via form.hidden_tag()
|
||||
(Flask-WTF, Ch12's CSRF requirement)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from flask_wtf import FlaskForm
|
||||
from wtforms import PasswordField, StringField, SubmitField
|
||||
from wtforms.validators import DataRequired
|
||||
|
||||
|
||||
class LoginForm(FlaskForm):
|
||||
# NOTE: no Email() validator — that validator requires the extra
|
||||
# `email-validator` package. Login checks the value against a stored
|
||||
# user row anyway, so a malformed email simply fails to match rather
|
||||
# than needing format validation up front; kept simple to avoid a new
|
||||
# dependency.
|
||||
email = StringField("Email", validators=[DataRequired()])
|
||||
password = PasswordField("Password", validators=[DataRequired()])
|
||||
submit = SubmitField("Log in")
|
||||
@@ -0,0 +1,45 @@
|
||||
"""Single-admin auth routes (Chapter 12 / Chapter 11's route table)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from flask import flash, redirect, render_template, request, url_for
|
||||
from flask_login import current_user, login_required, login_user, logout_user
|
||||
|
||||
from app.blueprints.auth import bp
|
||||
from app.blueprints.auth.forms import LoginForm
|
||||
from app.extensions import db, limiter, login_manager
|
||||
from app.models.user import User
|
||||
|
||||
|
||||
@login_manager.user_loader
|
||||
def load_user(user_id: str) -> User | None:
|
||||
return db.session.get(User, int(user_id))
|
||||
|
||||
|
||||
@bp.route("/login", methods=["GET", "POST"])
|
||||
@limiter.limit("10 per minute") # Ch12: rate limiting on /login at minimum
|
||||
def login():
|
||||
if current_user.is_authenticated:
|
||||
return redirect(url_for("overview.overview"))
|
||||
|
||||
form = LoginForm()
|
||||
if form.validate_on_submit():
|
||||
user = User.query.filter_by(email=form.email.data.strip().lower()).first()
|
||||
if user is not None and user.check_password(form.password.data):
|
||||
login_user(user)
|
||||
next_url = request.args.get("next") or url_for("overview.overview")
|
||||
return redirect(next_url)
|
||||
flash("Invalid email or password.")
|
||||
|
||||
return render_template("auth/login.html", form=form)
|
||||
|
||||
|
||||
@bp.post("/logout")
|
||||
@login_required
|
||||
def logout():
|
||||
# NOTE (flagged): Ch11's route table lists "/login, /logout" under a
|
||||
# shared "GET/POST" column. Logout is POST-only here — a state-changing
|
||||
# action behind a plain GET is a CSRF-adjacent anti-pattern Ch12's own
|
||||
# CSRF requirement argues against; login stays GET (show form) + POST
|
||||
# (submit), matching the table as-is.
|
||||
logout_user()
|
||||
return redirect(url_for("auth.login"))
|
||||
@@ -0,0 +1,27 @@
|
||||
{% extends "base.html" %}
|
||||
{% block title %}Log in — Kavosh{% endblock %}
|
||||
{% block content %}
|
||||
<div class="max-w-sm mx-auto mt-16 sm:mt-24">
|
||||
<div class="text-center mb-6">
|
||||
<h1 class="font-display font-bold text-2xl tracking-tight">Kavosh</h1>
|
||||
<p class="text-sm text-muted dark:text-muted-dark mt-1">Sign in to view your site's traffic</p>
|
||||
</div>
|
||||
<div class="bg-surface dark:bg-surface-dark border border-line dark:border-line-dark rounded-xl p-6 shadow-sm">
|
||||
{% for message in get_flashed_messages() %}
|
||||
<p class="text-danger dark:text-danger-dark text-sm mb-3 bg-danger/10 dark:bg-danger-dark/10 rounded-md px-3 py-2">{{ message }}</p>
|
||||
{% endfor %}
|
||||
<form method="post">
|
||||
{{ form.hidden_tag() }}
|
||||
<div class="mb-4">
|
||||
{{ form.email.label(class="block text-sm font-medium text-muted dark:text-muted-dark mb-1") }}
|
||||
{{ form.email(class="w-full border border-line dark:border-line-dark rounded-md px-3 py-2 bg-paper dark:bg-paper-dark text-ink dark:text-ink-dark focus:outline-none focus:ring-2 focus:ring-accent dark:focus:ring-accent-dark", autofocus=true) }}
|
||||
</div>
|
||||
<div class="mb-5">
|
||||
{{ form.password.label(class="block text-sm font-medium text-muted dark:text-muted-dark mb-1") }}
|
||||
{{ form.password(class="w-full border border-line dark:border-line-dark rounded-md px-3 py-2 bg-paper dark:bg-paper-dark text-ink dark:text-ink-dark focus:outline-none focus:ring-2 focus:ring-accent dark:focus:ring-accent-dark") }}
|
||||
</div>
|
||||
{{ form.submit(class="w-full bg-accent dark:bg-accent-dark text-white dark:text-paper-dark font-medium rounded-md px-4 py-2 hover:opacity-90 transition-opacity cursor-pointer") }}
|
||||
</form>
|
||||
</div>
|
||||
</div>
|
||||
{% endblock %}
|
||||
@@ -0,0 +1,13 @@
|
||||
from flask import Blueprint
|
||||
from flask_login import login_required
|
||||
|
||||
bp = Blueprint("overview", __name__, template_folder="templates")
|
||||
|
||||
|
||||
@bp.before_request
|
||||
@login_required
|
||||
def require_login():
|
||||
pass
|
||||
|
||||
|
||||
from app.blueprints.overview import routes # noqa: E402,F401 registers routes
|
||||
@@ -0,0 +1,167 @@
|
||||
"""Context-builder query functions for the Overview tab (Chapter 08).
|
||||
|
||||
Every function here reads request_stats_hourly / request_stats_daily /
|
||||
referrer_stats_daily / browser_stats_daily — never log_entries — per
|
||||
Ch03 rule 6 / Ch08's own data-source rule.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import date
|
||||
|
||||
from sqlalchemy import case, func
|
||||
|
||||
from app.extensions import db
|
||||
from app.models.browser_stats import BrowserStatsDaily
|
||||
from app.models.referrer_stats import ReferrerStatsDaily
|
||||
from app.models.request_stats import RequestStatsDaily, RequestStatsHourly
|
||||
from app.utils.dates import day_bounds
|
||||
|
||||
# ASSUMPTION (flagged in Ch08): Ch08 doesn't give a numeric threshold for
|
||||
# "hourly vs daily depending on range width" — picked 3 days.
|
||||
HOURLY_GRANULARITY_THRESHOLD_DAYS = 3
|
||||
|
||||
|
||||
def get_kpis(from_date: date, to_date: date) -> dict:
|
||||
"""All six Ch08 KPI-card fields, plus peak-day (folded in — Ch11 has
|
||||
no dedicated route for it and Ch08 calls for only a 'simple max-lookup').
|
||||
"""
|
||||
row = (
|
||||
db.session.query(
|
||||
func.coalesce(func.sum(RequestStatsDaily.count), 0).label("total_requests"),
|
||||
func.coalesce(func.sum(RequestStatsDaily.unique_ips), 0).label("unique_ips_sum"),
|
||||
func.coalesce(func.sum(RequestStatsDaily.bytes_sum), 0).label("total_bandwidth"),
|
||||
func.coalesce(func.sum(RequestStatsDaily.error_count), 0).label("total_errors"),
|
||||
)
|
||||
.filter(RequestStatsDaily.date >= from_date, RequestStatsDaily.date <= to_date)
|
||||
.one()
|
||||
)
|
||||
num_days = (to_date - from_date).days + 1
|
||||
avg_response_size = (row.total_bandwidth / row.total_requests) if row.total_requests else 0.0
|
||||
error_rate_pct = (row.total_errors / row.total_requests * 100) if row.total_requests else 0.0
|
||||
avg_requests_per_day = row.total_requests / num_days if num_days else 0.0
|
||||
|
||||
peak_row = (
|
||||
db.session.query(RequestStatsDaily.date, RequestStatsDaily.count)
|
||||
.filter(RequestStatsDaily.date >= from_date, RequestStatsDaily.date <= to_date)
|
||||
.order_by(RequestStatsDaily.count.desc())
|
||||
.first()
|
||||
)
|
||||
|
||||
return {
|
||||
"total_requests": row.total_requests,
|
||||
# APPROXIMATION (flagged in Ch08): sum of daily unique_ips over-counts
|
||||
# repeat visitors across days.
|
||||
"unique_ips": row.unique_ips_sum,
|
||||
"total_bandwidth_bytes": row.total_bandwidth,
|
||||
"avg_response_size_bytes": round(avg_response_size, 1),
|
||||
"error_rate_pct": round(error_rate_pct, 2),
|
||||
"avg_requests_per_day": round(avg_requests_per_day, 1),
|
||||
"peak_day": {"date": peak_row.date.isoformat(), "count": peak_row.count} if peak_row else None,
|
||||
}
|
||||
|
||||
|
||||
def get_traffic_chart_series(from_date: date, to_date: date) -> dict:
|
||||
span_days = (to_date - from_date).days + 1
|
||||
if span_days <= HOURLY_GRANULARITY_THRESHOLD_DAYS:
|
||||
start, end = day_bounds(from_date, to_date)
|
||||
rows = (
|
||||
db.session.query(RequestStatsHourly.date_hour, func.sum(RequestStatsHourly.count).label("count"))
|
||||
.filter(RequestStatsHourly.date_hour >= start, RequestStatsHourly.date_hour < end)
|
||||
.group_by(RequestStatsHourly.date_hour)
|
||||
.order_by(RequestStatsHourly.date_hour)
|
||||
.all()
|
||||
)
|
||||
return {"granularity": "hourly", "series": [{"t": r.date_hour.isoformat(), "count": r.count} for r in rows]}
|
||||
|
||||
rows = (
|
||||
db.session.query(RequestStatsDaily.date, RequestStatsDaily.count)
|
||||
.filter(RequestStatsDaily.date >= from_date, RequestStatsDaily.date <= to_date)
|
||||
.order_by(RequestStatsDaily.date)
|
||||
.all()
|
||||
)
|
||||
return {"granularity": "daily", "series": [{"t": r.date.isoformat(), "count": r.count} for r in rows]}
|
||||
|
||||
|
||||
def get_status_code_breakdown(from_date: date, to_date: date) -> dict:
|
||||
start, end = day_bounds(from_date, to_date)
|
||||
bucket = case(
|
||||
(RequestStatsHourly.status_code < 300, "2xx"),
|
||||
(RequestStatsHourly.status_code < 400, "3xx"),
|
||||
(RequestStatsHourly.status_code < 500, "4xx"),
|
||||
else_="5xx",
|
||||
)
|
||||
rows = (
|
||||
db.session.query(bucket.label("bucket"), func.sum(RequestStatsHourly.count).label("count"))
|
||||
.filter(RequestStatsHourly.date_hour >= start, RequestStatsHourly.date_hour < end)
|
||||
.group_by("bucket")
|
||||
.all()
|
||||
)
|
||||
breakdown = {"2xx": 0, "3xx": 0, "4xx": 0, "5xx": 0}
|
||||
for r in rows:
|
||||
breakdown[r.bucket] = r.count
|
||||
return breakdown
|
||||
|
||||
|
||||
def get_top_urls(from_date: date, to_date: date, page: int, per_page: int) -> tuple[list[list], int]:
|
||||
"""Top URLs by hits. NOTE (flagged in Ch08): only available within the
|
||||
~90-day hourly retention window (Ch06) — request_stats_daily has no
|
||||
path column, so a wider range returns nothing here.
|
||||
"""
|
||||
start, end = day_bounds(from_date, to_date)
|
||||
hits = func.sum(RequestStatsHourly.count)
|
||||
errors = func.sum(case((RequestStatsHourly.status_code >= 400, RequestStatsHourly.count), else_=0))
|
||||
bytes_sum = func.sum(RequestStatsHourly.bytes_sent_sum)
|
||||
|
||||
base_query = (
|
||||
db.session.query(RequestStatsHourly.path, hits.label("hits"), bytes_sum.label("bytes_sum"), errors.label("errors"))
|
||||
.filter(RequestStatsHourly.date_hour >= start, RequestStatsHourly.date_hour < end)
|
||||
.group_by(RequestStatsHourly.path)
|
||||
)
|
||||
total = base_query.count()
|
||||
rows = base_query.order_by(hits.desc()).offset((page - 1) * per_page).limit(per_page).all()
|
||||
|
||||
results = [
|
||||
[
|
||||
r.path,
|
||||
r.hits,
|
||||
round(r.bytes_sum / r.hits, 1) if r.hits else 0.0,
|
||||
round(r.errors / r.hits * 100, 2) if r.hits else 0.0,
|
||||
]
|
||||
for r in rows
|
||||
]
|
||||
return results, total
|
||||
|
||||
|
||||
def get_top_referrers(from_date: date, to_date: date, page: int, per_page: int) -> tuple[list[list], int]:
|
||||
"""Domain-bucketed referrers (Method A, Ch08 follow-up)."""
|
||||
hits = func.sum(ReferrerStatsDaily.count)
|
||||
base_query = (
|
||||
db.session.query(ReferrerStatsDaily.referrer_domain, hits.label("hits"))
|
||||
.filter(ReferrerStatsDaily.date >= from_date, ReferrerStatsDaily.date <= to_date)
|
||||
.group_by(ReferrerStatsDaily.referrer_domain)
|
||||
)
|
||||
total = base_query.count()
|
||||
rows = base_query.order_by(hits.desc()).offset((page - 1) * per_page).limit(per_page).all()
|
||||
return [[r.referrer_domain, r.hits] for r in rows], total
|
||||
|
||||
|
||||
def get_browser_breakdown(from_date: date, to_date: date) -> dict:
|
||||
"""Human-only (bots excluded at rollup-write time, Ch08)."""
|
||||
browser_rows = (
|
||||
db.session.query(BrowserStatsDaily.browser, func.sum(BrowserStatsDaily.count).label("count"))
|
||||
.filter(BrowserStatsDaily.date >= from_date, BrowserStatsDaily.date <= to_date)
|
||||
.group_by(BrowserStatsDaily.browser)
|
||||
.order_by(func.sum(BrowserStatsDaily.count).desc())
|
||||
.all()
|
||||
)
|
||||
os_rows = (
|
||||
db.session.query(BrowserStatsDaily.os, func.sum(BrowserStatsDaily.count).label("count"))
|
||||
.filter(BrowserStatsDaily.date >= from_date, BrowserStatsDaily.date <= to_date)
|
||||
.group_by(BrowserStatsDaily.os)
|
||||
.order_by(func.sum(BrowserStatsDaily.count).desc())
|
||||
.all()
|
||||
)
|
||||
return {
|
||||
"by_browser": [{"name": r.browser, "count": r.count} for r in browser_rows],
|
||||
"by_os": [{"name": r.os, "count": r.count} for r in os_rows],
|
||||
}
|
||||
@@ -0,0 +1,94 @@
|
||||
"""Overview blueprint routes (Chapter 08 / Chapter 11's route table)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from flask import current_app, jsonify, request
|
||||
|
||||
from app.blueprints.overview import bp
|
||||
from app.blueprints.overview.queries import (
|
||||
get_browser_breakdown,
|
||||
get_kpis,
|
||||
get_status_code_breakdown,
|
||||
get_top_referrers,
|
||||
get_top_urls,
|
||||
get_traffic_chart_series,
|
||||
)
|
||||
from app.services.background import resume_incomplete_files
|
||||
from app.utils.dates import parse_date_range
|
||||
from app.utils.htmx import render_htmx_aware
|
||||
from app.utils.pagination import parse_pagination
|
||||
|
||||
|
||||
@bp.route("/overview")
|
||||
def overview():
|
||||
"""Full page on first load, HTMX partial on tab switch / range change (Ch04).
|
||||
|
||||
Also opportunistically resumes any log_files stuck in queued/processing
|
||||
(Ch12 simplification: the replacement for cron's "there's always a
|
||||
next tick" guarantee — see app/services/background.py).
|
||||
"""
|
||||
resume_incomplete_files(current_app._get_current_object())
|
||||
from_date, to_date = parse_date_range(request)
|
||||
return render_htmx_aware(
|
||||
request,
|
||||
full_template="overview/index.html",
|
||||
partial_template="overview/_content.html",
|
||||
from_date=from_date,
|
||||
to_date=to_date,
|
||||
)
|
||||
|
||||
|
||||
@bp.get("/api/overview/kpis")
|
||||
def api_kpis():
|
||||
from_date, to_date = parse_date_range(request)
|
||||
return jsonify(data=get_kpis(from_date, to_date), meta={"from": from_date.isoformat(), "to": to_date.isoformat()})
|
||||
|
||||
|
||||
@bp.get("/api/overview/traffic-chart")
|
||||
def api_traffic_chart():
|
||||
from_date, to_date = parse_date_range(request)
|
||||
return jsonify(
|
||||
data=get_traffic_chart_series(from_date, to_date),
|
||||
meta={"from": from_date.isoformat(), "to": to_date.isoformat()},
|
||||
)
|
||||
|
||||
|
||||
@bp.get("/api/overview/status-codes")
|
||||
def api_status_codes():
|
||||
from_date, to_date = parse_date_range(request)
|
||||
return jsonify(
|
||||
data=get_status_code_breakdown(from_date, to_date),
|
||||
meta={"from": from_date.isoformat(), "to": to_date.isoformat()},
|
||||
)
|
||||
|
||||
|
||||
@bp.get("/api/overview/top-urls")
|
||||
def api_top_urls():
|
||||
from_date, to_date = parse_date_range(request)
|
||||
page, per_page = parse_pagination(request)
|
||||
rows, total = get_top_urls(from_date, to_date, page, per_page)
|
||||
return jsonify(
|
||||
data={"rows": rows, "total": total},
|
||||
meta={"from": from_date.isoformat(), "to": to_date.isoformat(), "page": page, "per_page": per_page},
|
||||
)
|
||||
|
||||
|
||||
@bp.get("/api/overview/top-referrers")
|
||||
def api_top_referrers():
|
||||
from_date, to_date = parse_date_range(request)
|
||||
page, per_page = parse_pagination(request)
|
||||
rows, total = get_top_referrers(from_date, to_date, page, per_page)
|
||||
return jsonify(
|
||||
data={"rows": rows, "total": total},
|
||||
meta={"from": from_date.isoformat(), "to": to_date.isoformat(), "page": page, "per_page": per_page},
|
||||
)
|
||||
|
||||
|
||||
@bp.get("/api/overview/browser-breakdown")
|
||||
def api_browser_breakdown():
|
||||
"""Chapter 11 addition (Method A, Ch08 follow-up) — not in the
|
||||
original route table; see docs/api-contract-final.md."""
|
||||
from_date, to_date = parse_date_range(request)
|
||||
return jsonify(
|
||||
data=get_browser_breakdown(from_date, to_date),
|
||||
meta={"from": from_date.isoformat(), "to": to_date.isoformat()},
|
||||
)
|
||||
@@ -0,0 +1,286 @@
|
||||
<div id="overview-content"
|
||||
hx-get="{{ url_for('overview.overview') }}"
|
||||
hx-trigger="change from:#date-range-form"
|
||||
hx-include="#date-range-form"
|
||||
hx-target="#overview-content"
|
||||
hx-swap="outerHTML"
|
||||
data-from="{{ from_date.isoformat() }}"
|
||||
data-to="{{ to_date.isoformat() }}">
|
||||
|
||||
<div class="flex flex-wrap items-end justify-between gap-4 mb-6">
|
||||
<div>
|
||||
<h1 class="font-display font-bold text-xl">Overview</h1>
|
||||
<p class="text-sm text-muted dark:text-muted-dark">What happened on your site recently</p>
|
||||
</div>
|
||||
<form id="date-range-form" class="flex gap-3 items-end">
|
||||
<label class="text-sm text-muted dark:text-muted-dark">From
|
||||
<input type="date" name="from" value="{{ from_date.isoformat() }}"
|
||||
class="block border border-line dark:border-line-dark rounded-md px-2 py-1 mt-1 bg-surface dark:bg-surface-dark text-ink dark:text-ink-dark font-data text-sm">
|
||||
</label>
|
||||
<label class="text-sm text-muted dark:text-muted-dark">To
|
||||
<input type="date" name="to" value="{{ to_date.isoformat() }}"
|
||||
class="block border border-line dark:border-line-dark rounded-md px-2 py-1 mt-1 bg-surface dark:bg-surface-dark text-ink dark:text-ink-dark font-data text-sm">
|
||||
</label>
|
||||
</form>
|
||||
</div>
|
||||
|
||||
<!-- Upload panel — real byte-transfer progress via XHR (uploads.js), then
|
||||
a real parse-progress bar (Ch12) once the file is queued. No cron
|
||||
required: processing starts automatically in the background. -->
|
||||
<div class="mb-6 border border-line dark:border-line-dark rounded-xl p-4 bg-surface dark:bg-surface-dark">
|
||||
<h2 class="font-display font-semibold text-sm mb-3">Upload a log file</h2>
|
||||
<form id="upload-form" action="{{ url_for('uploads.upload_log_file') }}"
|
||||
class="flex flex-wrap gap-3 items-center">
|
||||
<input type="file" name="logfile" accept=".log,.txt,.gz" required
|
||||
class="text-sm text-muted dark:text-muted-dark file:mr-3 file:py-1.5 file:px-3 file:rounded-md file:border-0 file:bg-accent/10 file:text-accent dark:file:bg-accent-dark/15 dark:file:text-accent-dark file:text-sm file:font-medium hover:file:bg-accent/20 dark:hover:file:bg-accent-dark/25 file:cursor-pointer cursor-pointer">
|
||||
<select name="server_type" class="border border-line dark:border-line-dark rounded-md px-2 py-1.5 text-sm bg-surface dark:bg-surface-dark text-ink dark:text-ink-dark">
|
||||
<option value="apache">Apache</option>
|
||||
<option value="litespeed">LiteSpeed</option>
|
||||
</select>
|
||||
<input type="text" name="format_string" placeholder="LogFormat (optional — defaults to Combined)"
|
||||
class="border border-line dark:border-line-dark rounded-md px-2 py-1.5 text-sm flex-1 min-w-[220px] bg-surface dark:bg-surface-dark text-ink dark:text-ink-dark placeholder:text-muted dark:placeholder:text-muted-dark">
|
||||
<button type="submit" class="bg-accent dark:bg-accent-dark text-white dark:text-paper-dark text-sm font-medium rounded-md px-4 py-1.5 hover:opacity-90 transition-opacity">
|
||||
Upload
|
||||
</button>
|
||||
</form>
|
||||
|
||||
<div id="upload-progress-wrap" class="hidden mt-3">
|
||||
<div class="flex items-center justify-between text-sm mb-1">
|
||||
<span id="upload-progress-label" class="text-muted dark:text-muted-dark">Uploading…</span>
|
||||
</div>
|
||||
<div class="w-full h-2 rounded-full bg-line dark:bg-line-dark overflow-hidden">
|
||||
<div id="upload-progress-bar" class="h-full rounded-full bg-accent dark:bg-accent-dark transition-all duration-150" style="width: 0%"></div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div id="upload-result" class="mt-3"></div>
|
||||
</div>
|
||||
|
||||
<!-- Uploaded files — select and delete (project-owner follow-up
|
||||
request). Independent of the date-range picker above: this lists
|
||||
uploads by when they arrived, not by which log dates they cover
|
||||
(a single file can span many dates). Selection is intentionally
|
||||
scoped to the currently-rendered page/sort/search view — it
|
||||
doesn't try to persist across a Grid.js re-render, which keeps
|
||||
"what's selected" always visually honest at the cost of losing
|
||||
selection if you page/sort/search mid-selection. -->
|
||||
<div class="mb-6 border border-line dark:border-line-dark rounded-xl bg-surface dark:bg-surface-dark overflow-hidden">
|
||||
<div class="flex items-center justify-between p-4 pb-3">
|
||||
<h2 class="font-display font-semibold text-sm">Uploaded files</h2>
|
||||
<button type="button" id="delete-selected-btn" disabled
|
||||
class="text-sm font-medium rounded-md px-3 py-1.5 border border-danger/40 dark:border-danger-dark/40 text-danger dark:text-danger-dark opacity-40 cursor-not-allowed disabled:opacity-40 enabled:opacity-100 enabled:hover:bg-danger/10 dark:enabled:hover:bg-danger-dark/10 transition-opacity">
|
||||
Delete selected
|
||||
</button>
|
||||
</div>
|
||||
<div id="uploaded-files-grid" class="px-4 pb-4" data-endpoint="{{ url_for('uploads.list_uploads') }}"></div>
|
||||
</div>
|
||||
|
||||
<!-- KPI readout — the one deliberate signature treatment: values set in
|
||||
tabular mono, like numbers straight off the log line, with a thin
|
||||
top rule that switches color when a metric needs attention
|
||||
(structure carries information, not just decoration). -->
|
||||
<div id="kpi-cards" class="grid grid-cols-2 sm:grid-cols-3 gap-3 mb-6" data-endpoint="{{ url_for('overview.api_kpis') }}"></div>
|
||||
|
||||
<div class="grid grid-cols-1 lg:grid-cols-2 gap-4 mb-6">
|
||||
<div class="border border-line dark:border-line-dark rounded-xl p-4 bg-surface dark:bg-surface-dark h-72">
|
||||
<h3 class="text-xs font-medium uppercase tracking-wide text-muted dark:text-muted-dark mb-2">Traffic over time</h3>
|
||||
<div class="h-56"><canvas id="traffic-chart" data-endpoint="{{ url_for('overview.api_traffic_chart') }}"></canvas></div>
|
||||
</div>
|
||||
<div class="border border-line dark:border-line-dark rounded-xl p-4 bg-surface dark:bg-surface-dark h-72">
|
||||
<h3 class="text-xs font-medium uppercase tracking-wide text-muted dark:text-muted-dark mb-2">Status codes</h3>
|
||||
<div class="h-56"><canvas id="status-code-chart" data-endpoint="{{ url_for('overview.api_status_codes') }}"></canvas></div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="grid grid-cols-1 lg:grid-cols-2 gap-4">
|
||||
<div>
|
||||
<h3 class="text-xs font-medium uppercase tracking-wide text-muted dark:text-muted-dark mb-2">Top URLs</h3>
|
||||
<div id="top-urls-grid" data-endpoint="{{ url_for('overview.api_top_urls') }}"></div>
|
||||
</div>
|
||||
<div>
|
||||
<h3 class="text-xs font-medium uppercase tracking-wide text-muted dark:text-muted-dark mb-2">Top referrers</h3>
|
||||
<div id="top-referrers-grid" data-endpoint="{{ url_for('overview.api_top_referrers') }}"></div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="grid grid-cols-1 lg:grid-cols-2 gap-4 mt-6">
|
||||
<div class="border border-line dark:border-line-dark rounded-xl p-4 bg-surface dark:bg-surface-dark h-64">
|
||||
<h3 class="text-xs font-medium uppercase tracking-wide text-muted dark:text-muted-dark mb-2">Browsers</h3>
|
||||
<div class="h-48"><canvas id="browser-breakdown-chart" data-endpoint="{{ url_for('overview.api_browser_breakdown') }}"></canvas></div>
|
||||
</div>
|
||||
<div class="border border-line dark:border-line-dark rounded-xl p-4 bg-surface dark:bg-surface-dark h-64">
|
||||
<h3 class="text-xs font-medium uppercase tracking-wide text-muted dark:text-muted-dark mb-2">Operating systems</h3>
|
||||
<div class="h-48"><canvas id="os-breakdown-chart"></canvas></div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<script>
|
||||
(function initOverviewWidgets() {
|
||||
const root = document.getElementById('overview-content');
|
||||
const from = root.dataset.from, to = root.dataset.to;
|
||||
const withRange = (url) => `${url}?from=${from}&to=${to}`;
|
||||
|
||||
window.initUploadForm('upload-form');
|
||||
initUploadedFilesGrid();
|
||||
|
||||
fetch(withRange(document.getElementById('kpi-cards').dataset.endpoint))
|
||||
.then((r) => r.json())
|
||||
.then(({ data }) => renderKpiCards(data));
|
||||
|
||||
const trafficEl = document.getElementById('traffic-chart');
|
||||
fetch(withRange(trafficEl.dataset.endpoint))
|
||||
.then((r) => r.json())
|
||||
.then(({ data }) => window.initChart('traffic-chart', {
|
||||
type: 'line',
|
||||
data: {
|
||||
labels: data.series.map((p) => p.t),
|
||||
datasets: [{
|
||||
label: 'Requests', data: data.series.map((p) => p.count), tension: 0.3,
|
||||
borderColor: '#0E7C86', backgroundColor: '#0E7C8622', fill: true,
|
||||
}],
|
||||
},
|
||||
}));
|
||||
|
||||
const statusEl = document.getElementById('status-code-chart');
|
||||
fetch(withRange(statusEl.dataset.endpoint))
|
||||
.then((r) => r.json())
|
||||
.then(({ data }) => window.initChart('status-code-chart', {
|
||||
type: 'doughnut',
|
||||
data: { labels: Object.keys(data), datasets: [{ data: Object.values(data), backgroundColor: window.KAVOSH_CHART_PALETTE }] },
|
||||
}));
|
||||
|
||||
window.initGrid(
|
||||
'top-urls-grid',
|
||||
document.getElementById('top-urls-grid').dataset.endpoint,
|
||||
[{ name: 'Path' }, { name: 'Hits' }, { name: 'Avg Size (B)' }, { name: 'Error Rate %' }],
|
||||
{ from, to },
|
||||
);
|
||||
|
||||
window.initGrid(
|
||||
'top-referrers-grid',
|
||||
document.getElementById('top-referrers-grid').dataset.endpoint,
|
||||
[{ name: 'Referrer Domain' }, { name: 'Hits' }],
|
||||
{ from, to },
|
||||
);
|
||||
|
||||
const browserEl = document.getElementById('browser-breakdown-chart');
|
||||
fetch(withRange(browserEl.dataset.endpoint))
|
||||
.then((r) => r.json())
|
||||
.then(({ data }) => {
|
||||
window.initChart('browser-breakdown-chart', {
|
||||
type: 'bar',
|
||||
data: { labels: data.by_browser.map((r) => r.name), datasets: [{ label: 'Browser', data: data.by_browser.map((r) => r.count), backgroundColor: '#0E7C86' }] },
|
||||
});
|
||||
window.initChart('os-breakdown-chart', {
|
||||
type: 'bar',
|
||||
data: { labels: data.by_os.map((r) => r.name), datasets: [{ label: 'OS', data: data.by_os.map((r) => r.count), backgroundColor: '#B8860B' }] },
|
||||
});
|
||||
});
|
||||
|
||||
function renderKpiCards(kpi) {
|
||||
const errorTone = kpi.error_rate_pct >= 5 ? 'danger' : kpi.error_rate_pct >= 1 ? 'warn' : 'ok';
|
||||
const cards = [
|
||||
['Total requests', kpi.total_requests.toLocaleString(), 'accent'],
|
||||
['Unique IPs (approx.)', kpi.unique_ips.toLocaleString(), 'accent'],
|
||||
['Bandwidth', `${(kpi.total_bandwidth_bytes / 1e6).toFixed(1)} MB`, 'accent'],
|
||||
['Avg response size', `${kpi.avg_response_size_bytes} B`, 'accent'],
|
||||
['Error rate', `${kpi.error_rate_pct}%`, errorTone],
|
||||
['Avg requests / day', kpi.avg_requests_per_day.toLocaleString(), 'accent'],
|
||||
];
|
||||
const toneBorder = {
|
||||
accent: 'border-t-accent dark:border-t-accent-dark',
|
||||
ok: 'border-t-ok dark:border-t-ok-dark',
|
||||
warn: 'border-t-warn dark:border-t-warn-dark',
|
||||
danger: 'border-t-danger dark:border-t-danger-dark',
|
||||
};
|
||||
const el = document.getElementById('kpi-cards');
|
||||
el.innerHTML = cards.map(([label, value, tone]) => `
|
||||
<div class="bg-surface dark:bg-surface-dark rounded-lg border border-line dark:border-line-dark border-t-2 ${toneBorder[tone]} p-3">
|
||||
<p class="text-xs text-muted dark:text-muted-dark uppercase tracking-wide">${label}</p>
|
||||
<p class="font-data text-2xl font-medium mt-0.5">${value}</p>
|
||||
</div>`).join('');
|
||||
if (kpi.peak_day) {
|
||||
el.innerHTML += `
|
||||
<div class="bg-surface dark:bg-surface-dark rounded-lg border border-line dark:border-line-dark border-t-2 border-t-accent dark:border-t-accent-dark p-3 col-span-2 sm:col-span-3">
|
||||
<p class="text-xs text-muted dark:text-muted-dark uppercase tracking-wide">Peak day</p>
|
||||
<p class="font-data text-xl font-medium mt-0.5">${kpi.peak_day.date} <span class="text-muted dark:text-muted-dark text-sm">— ${kpi.peak_day.count.toLocaleString()} requests</span></p>
|
||||
</div>`;
|
||||
}
|
||||
}
|
||||
|
||||
function initUploadedFilesGrid() {
|
||||
const container = document.getElementById('uploaded-files-grid');
|
||||
const deleteBtn = document.getElementById('delete-selected-btn');
|
||||
const selectedIds = new Set();
|
||||
|
||||
function updateDeleteButton() {
|
||||
deleteBtn.disabled = selectedIds.size === 0;
|
||||
deleteBtn.textContent = selectedIds.size > 0 ? `Delete selected (${selectedIds.size})` : 'Delete selected';
|
||||
}
|
||||
|
||||
// Selection is intentionally NOT restored across a Grid.js
|
||||
// re-render (page/sort/search) — simpler and always visually
|
||||
// honest, at the cost of losing selection if you page away
|
||||
// mid-selection. See the comment above the HTML for this panel.
|
||||
container.addEventListener('change', (e) => {
|
||||
if (!e.target.matches('.file-select-checkbox')) return;
|
||||
const id = parseInt(e.target.value, 10);
|
||||
if (e.target.checked) selectedIds.add(id); else selectedIds.delete(id);
|
||||
updateDeleteButton();
|
||||
});
|
||||
|
||||
const columns = [
|
||||
{
|
||||
name: '',
|
||||
formatter: (cell, row) => {
|
||||
const rawStatus = row.cells[6].data;
|
||||
const disabled = rawStatus === 'processing';
|
||||
return window.gridHtml(
|
||||
`<input type="checkbox" class="file-select-checkbox w-4 h-4 rounded border-line dark:border-line-dark text-accent focus:ring-accent cursor-pointer disabled:cursor-not-allowed disabled:opacity-40"
|
||||
value="${cell}" ${disabled ? 'disabled title="Still analyzing — wait for it to finish"' : ''}>`
|
||||
);
|
||||
},
|
||||
},
|
||||
{ name: 'Filename' },
|
||||
{ name: 'Server' },
|
||||
{ name: 'Status' },
|
||||
{ name: 'Uploaded' },
|
||||
{ name: 'Size' },
|
||||
{ name: 'Raw Status', hidden: true },
|
||||
];
|
||||
|
||||
const filesGrid = window.initGrid(
|
||||
'uploaded-files-grid',
|
||||
container.dataset.endpoint,
|
||||
columns,
|
||||
{},
|
||||
);
|
||||
|
||||
deleteBtn.addEventListener('click', () => {
|
||||
if (selectedIds.size === 0) return;
|
||||
if (!confirm(`Delete ${selectedIds.size} file(s) and all data derived from them? This can't be undone.`)) return;
|
||||
|
||||
const token = document.querySelector('meta[name="csrf-token"]')?.content;
|
||||
fetch('{{ url_for("uploads.bulk_delete_uploads") }}', {
|
||||
method: 'DELETE',
|
||||
headers: { 'Content-Type': 'application/json', 'X-CSRFToken': token || '' },
|
||||
body: JSON.stringify({ ids: Array.from(selectedIds) }),
|
||||
})
|
||||
.then((r) => r.json())
|
||||
.then(({ data }) => {
|
||||
selectedIds.clear();
|
||||
updateDeleteButton();
|
||||
filesGrid.forceRender();
|
||||
if (data.skipped && data.skipped.length) {
|
||||
alert('Some files were skipped:\n' + data.skipped.map((s) => `#${s.id} — ${s.reason}`).join('\n'));
|
||||
}
|
||||
// Deleting a file can change rollups for any date it
|
||||
// touched — refresh the whole page's KPIs/charts via the
|
||||
// same mechanism the date-range picker itself uses.
|
||||
window.htmx.trigger(document.getElementById('date-range-form'), 'change');
|
||||
});
|
||||
});
|
||||
}
|
||||
})();
|
||||
</script>
|
||||
</div>
|
||||
@@ -0,0 +1,5 @@
|
||||
{% extends "base.html" %}
|
||||
{% block title %}Overview — Kavosh{% endblock %}
|
||||
{% block content %}
|
||||
{% include "overview/_content.html" %}
|
||||
{% endblock %}
|
||||
@@ -0,0 +1,13 @@
|
||||
from flask import Blueprint
|
||||
from flask_login import login_required
|
||||
|
||||
bp = Blueprint("security", __name__, template_folder="templates")
|
||||
|
||||
|
||||
@bp.before_request
|
||||
@login_required
|
||||
def require_login():
|
||||
pass
|
||||
|
||||
|
||||
from app.blueprints.security import routes # noqa: E402,F401 registers routes
|
||||
@@ -0,0 +1,154 @@
|
||||
"""Context-builder query functions for the Security tab (Chapter 10),
|
||||
with the IP investigation panel's traffic breakdown upgraded to full-
|
||||
traffic data (bounded per-IP rollup, added as an explicit follow-up to
|
||||
Chapter 10's scope gap).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from collections import defaultdict
|
||||
from datetime import date
|
||||
|
||||
from sqlalchemy import func
|
||||
|
||||
from app.extensions import db
|
||||
from app.models.blocklist_suggestion import BlocklistSuggestion
|
||||
from app.models.bot_hit import BotHit
|
||||
from app.models.ip_registry import IPRegistry
|
||||
from app.models.ip_traffic_stats import IpPathStatsDaily, IpStatusStatsDaily
|
||||
from app.models.suspicious_event import SuspiciousEvent
|
||||
from app.services.severity_scoring import SeverityInputs, compute_effective_severity
|
||||
from app.utils.dates import day_bounds
|
||||
|
||||
IP_HISTORY_EVENT_LIMIT = 50
|
||||
IP_HISTORY_PATH_LIMIT = 20
|
||||
|
||||
|
||||
def get_suspicious_events(
|
||||
from_date: date, to_date: date, severity: str | None, rule_type: str | None, page: int, per_page: int,
|
||||
) -> tuple[list[list], int]:
|
||||
"""suspicious_events is small/indexed/indefinitely-retained (Ch06) —
|
||||
same precedent as Ch09's bot_hits queries, so loading + escalating in
|
||||
Python doesn't violate Ch03 rule 6 (that targets raw per-request rows).
|
||||
"""
|
||||
start, end = day_bounds(from_date, to_date)
|
||||
query = db.session.query(SuspiciousEvent).filter(
|
||||
SuspiciousEvent.timestamp >= start, SuspiciousEvent.timestamp < end
|
||||
)
|
||||
if rule_type:
|
||||
query = query.filter(SuspiciousEvent.rule_matched.like(f"{rule_type}:%"))
|
||||
events = query.order_by(SuspiciousEvent.timestamp.desc()).all()
|
||||
|
||||
by_ip: dict[str, list[SuspiciousEvent]] = defaultdict(list)
|
||||
for e in events:
|
||||
by_ip[e.ip].append(e)
|
||||
|
||||
enriched = []
|
||||
for e in events:
|
||||
ip_events = by_ip[e.ip]
|
||||
timestamps = sorted(ev.timestamp for ev in ip_events)
|
||||
avg_interval = (
|
||||
(timestamps[-1] - timestamps[0]).total_seconds() / (len(timestamps) - 1)
|
||||
if len(timestamps) > 1 else None
|
||||
)
|
||||
effective = compute_effective_severity(
|
||||
SeverityInputs(base_severity=e.severity, ip_event_count=len(ip_events), avg_interval_seconds=avg_interval)
|
||||
)
|
||||
if severity and effective != severity:
|
||||
continue
|
||||
enriched.append([e.timestamp.isoformat(), e.ip, e.path, e.rule_matched, effective])
|
||||
|
||||
total = len(enriched)
|
||||
offset = (page - 1) * per_page
|
||||
return enriched[offset : offset + per_page], total
|
||||
|
||||
|
||||
def get_sensitive_path_summary(from_date: date, to_date: date) -> list[dict]:
|
||||
"""Grouped by request path; filtered to Ch07's sensitive_path rule
|
||||
category. One ranked list, not sub-grouped into config/admin/VCS —
|
||||
Ch07's dictionary has no such taxonomy to reuse.
|
||||
"""
|
||||
start, end = day_bounds(from_date, to_date)
|
||||
rows = (
|
||||
db.session.query(
|
||||
SuspiciousEvent.path,
|
||||
func.count().label("hit_count"),
|
||||
func.count(func.distinct(SuspiciousEvent.ip)).label("distinct_ip_count"),
|
||||
)
|
||||
.filter(
|
||||
SuspiciousEvent.timestamp >= start, SuspiciousEvent.timestamp < end,
|
||||
SuspiciousEvent.rule_matched.like("sensitive_path:%"),
|
||||
)
|
||||
.group_by(SuspiciousEvent.path)
|
||||
.order_by(func.count().desc())
|
||||
.all()
|
||||
)
|
||||
return [{"path": r.path, "hit_count": r.hit_count, "distinct_ip_count": r.distinct_ip_count} for r in rows]
|
||||
|
||||
|
||||
def get_ip_history(ip: str) -> dict | None:
|
||||
"""Pulled from ip_registry (identity + true total_requests), plus
|
||||
bot_hits/suspicious_events (flagged activity), plus the bounded
|
||||
per-IP traffic rollup (top_paths / status_code_distribution — true
|
||||
full-traffic breakdown, added as a follow-up to Ch10's original scope
|
||||
gap). Retention caveat: the per-IP rollup covers roughly the last 30
|
||||
days (see aggregator.py / flask cleanup).
|
||||
"""
|
||||
registry = db.session.get(IPRegistry, ip)
|
||||
if registry is None:
|
||||
return None
|
||||
|
||||
path_rows = (
|
||||
db.session.query(IpPathStatsDaily.path, func.sum(IpPathStatsDaily.count).label("count"))
|
||||
.filter(IpPathStatsDaily.ip == ip)
|
||||
.group_by(IpPathStatsDaily.path)
|
||||
.order_by(func.sum(IpPathStatsDaily.count).desc())
|
||||
.limit(IP_HISTORY_PATH_LIMIT)
|
||||
.all()
|
||||
)
|
||||
status_rows = (
|
||||
db.session.query(IpStatusStatsDaily.status_bucket, func.sum(IpStatusStatsDaily.count).label("count"))
|
||||
.filter(IpStatusStatsDaily.ip == ip)
|
||||
.group_by(IpStatusStatsDaily.status_bucket)
|
||||
.all()
|
||||
)
|
||||
|
||||
bot_rows = (
|
||||
db.session.query(BotHit).filter(BotHit.ip == ip)
|
||||
.order_by(BotHit.timestamp.desc()).limit(IP_HISTORY_EVENT_LIMIT).all()
|
||||
)
|
||||
suspicious_rows = (
|
||||
db.session.query(SuspiciousEvent).filter(SuspiciousEvent.ip == ip)
|
||||
.order_by(SuspiciousEvent.timestamp.desc()).limit(IP_HISTORY_EVENT_LIMIT).all()
|
||||
)
|
||||
spoofed_bot_names = sorted({b.bot_name for b in bot_rows if not b.verified})
|
||||
|
||||
return {
|
||||
"ip": ip,
|
||||
"first_seen": registry.first_seen.isoformat(),
|
||||
"last_seen": registry.last_seen.isoformat(),
|
||||
"total_requests": registry.total_requests,
|
||||
"reputation_score": registry.reputation_score,
|
||||
"is_flagged": registry.is_flagged,
|
||||
"spoofed_bot_names": spoofed_bot_names,
|
||||
"top_paths": [[r.path, r.count] for r in path_rows],
|
||||
"status_code_distribution": {r.status_bucket: r.count for r in status_rows},
|
||||
"traffic_window_note": "Path/status breakdown reflects roughly the last 30 days (bounded retention).",
|
||||
"recent_suspicious_events": [
|
||||
{"timestamp": s.timestamp.isoformat(), "path": s.path, "rule_matched": s.rule_matched, "severity": s.severity}
|
||||
for s in suspicious_rows
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
def format_blocklist(suggestions: list[BlocklistSuggestion], fmt: str) -> str:
|
||||
"""Ch10: '.htaccess Deny/iptables/fail2ban-style'. 'plain' (a bare IP
|
||||
list) is the most portable interpretation of "fail2ban-style input"
|
||||
without assuming a specific fail2ban jail configuration Ch10 doesn't
|
||||
specify.
|
||||
"""
|
||||
ips = [s.ip for s in suggestions]
|
||||
if fmt == "htaccess":
|
||||
return "".join(f"Deny from {ip}\n" for ip in ips)
|
||||
if fmt == "iptables":
|
||||
return "".join(f"iptables -A INPUT -s {ip} -j DROP\n" for ip in ips)
|
||||
return "".join(f"{ip}\n" for ip in ips)
|
||||
@@ -0,0 +1,71 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from flask import Response, jsonify, render_template, request
|
||||
|
||||
from app.blueprints.security import bp
|
||||
from app.blueprints.security.queries import (
|
||||
format_blocklist, get_ip_history, get_sensitive_path_summary, get_suspicious_events,
|
||||
)
|
||||
from app.extensions import db
|
||||
from app.models.blocklist_suggestion import BlocklistSuggestion
|
||||
from app.utils.dates import parse_date_range
|
||||
from app.utils.htmx import render_htmx_aware
|
||||
from app.utils.pagination import parse_pagination
|
||||
|
||||
|
||||
@bp.route("/security")
|
||||
def security():
|
||||
from_date, to_date = parse_date_range(request)
|
||||
severity = request.args.get("severity") or ""
|
||||
rule_type = request.args.get("rule_type") or ""
|
||||
return render_htmx_aware(
|
||||
request, full_template="security/index.html", partial_template="security/_content.html",
|
||||
from_date=from_date, to_date=to_date, severity=severity, rule_type=rule_type,
|
||||
)
|
||||
|
||||
|
||||
@bp.get("/api/security/events")
|
||||
def api_security_events():
|
||||
from_date, to_date = parse_date_range(request)
|
||||
page, per_page = parse_pagination(request)
|
||||
severity = request.args.get("severity") or None
|
||||
rule_type = request.args.get("rule_type") or None
|
||||
rows, total = get_suspicious_events(from_date, to_date, severity, rule_type, page, per_page)
|
||||
return jsonify(
|
||||
data={"rows": rows, "total": total},
|
||||
meta={"from": from_date.isoformat(), "to": to_date.isoformat(), "page": page, "per_page": per_page},
|
||||
)
|
||||
|
||||
|
||||
@bp.get("/api/security/sensitive-paths")
|
||||
def api_sensitive_paths():
|
||||
from_date, to_date = parse_date_range(request)
|
||||
return jsonify(data=get_sensitive_path_summary(from_date, to_date), meta={"from": from_date.isoformat(), "to": to_date.isoformat()})
|
||||
|
||||
|
||||
@bp.get("/api/security/ip/<ip>")
|
||||
def api_ip_history(ip: str):
|
||||
history = get_ip_history(ip)
|
||||
if history is None:
|
||||
return render_template("security/_ip_not_found.html", ip=ip), 404
|
||||
return render_template("security/_ip_history.html", ip_data=history)
|
||||
|
||||
|
||||
@bp.get("/api/security/export-blocklist")
|
||||
def api_export_blocklist():
|
||||
fmt = request.args.get("format", "plain")
|
||||
include_all = request.args.get("all", "false").lower() == "true"
|
||||
query = BlocklistSuggestion.query
|
||||
if not include_all:
|
||||
query = query.filter_by(exported=False)
|
||||
suggestions = query.order_by(BlocklistSuggestion.created_at).all()
|
||||
|
||||
body = format_blocklist(suggestions, fmt)
|
||||
for s in suggestions:
|
||||
s.exported = True
|
||||
db.session.commit()
|
||||
|
||||
return Response(
|
||||
body, mimetype="text/plain",
|
||||
headers={"Content-Disposition": "attachment; filename=kavosh-blocklist.txt"},
|
||||
)
|
||||
@@ -0,0 +1,125 @@
|
||||
<div id="security-content"
|
||||
hx-get="{{ url_for('security.security') }}"
|
||||
hx-trigger="change from:#security-filter-form"
|
||||
hx-include="#security-filter-form"
|
||||
hx-target="#security-content"
|
||||
hx-swap="outerHTML"
|
||||
data-from="{{ from_date.isoformat() }}"
|
||||
data-to="{{ to_date.isoformat() }}"
|
||||
data-severity="{{ severity }}"
|
||||
data-rule-type="{{ rule_type }}">
|
||||
|
||||
<div class="flex flex-wrap items-end justify-between gap-4 mb-6">
|
||||
<div>
|
||||
<h1 class="font-display font-bold text-xl">Suspicious Requests & IP History</h1>
|
||||
<p class="text-sm text-muted dark:text-muted-dark">Who's poking at your site, and how hard</p>
|
||||
</div>
|
||||
<form id="security-filter-form" class="flex flex-wrap gap-3 items-end">
|
||||
<label class="text-sm text-muted dark:text-muted-dark">From
|
||||
<input type="date" name="from" value="{{ from_date.isoformat() }}" class="block border border-line dark:border-line-dark rounded-md px-2 py-1 mt-1 bg-surface dark:bg-surface-dark text-ink dark:text-ink-dark font-data text-sm">
|
||||
</label>
|
||||
<label class="text-sm text-muted dark:text-muted-dark">To
|
||||
<input type="date" name="to" value="{{ to_date.isoformat() }}" class="block border border-line dark:border-line-dark rounded-md px-2 py-1 mt-1 bg-surface dark:bg-surface-dark text-ink dark:text-ink-dark font-data text-sm">
|
||||
</label>
|
||||
<label class="text-sm text-muted dark:text-muted-dark">Severity
|
||||
<select name="severity" class="block border border-line dark:border-line-dark rounded-md px-2 py-1 mt-1 bg-surface dark:bg-surface-dark text-ink dark:text-ink-dark text-sm">
|
||||
<option value="" {{ 'selected' if not severity }}>All</option>
|
||||
<option value="low" {{ 'selected' if severity == 'low' }}>Low</option>
|
||||
<option value="medium" {{ 'selected' if severity == 'medium' }}>Medium</option>
|
||||
<option value="high" {{ 'selected' if severity == 'high' }}>High</option>
|
||||
</select>
|
||||
</label>
|
||||
<label class="text-sm text-muted dark:text-muted-dark">Rule Type
|
||||
<select name="rule_type" class="block border border-line dark:border-line-dark rounded-md px-2 py-1 mt-1 bg-surface dark:bg-surface-dark text-ink dark:text-ink-dark text-sm">
|
||||
<option value="" {{ 'selected' if not rule_type }}>All</option>
|
||||
<option value="sensitive_path" {{ 'selected' if rule_type == 'sensitive_path' }}>Sensitive Path</option>
|
||||
<option value="injection" {{ 'selected' if rule_type == 'injection' }}>Injection</option>
|
||||
<option value="scanner_ua" {{ 'selected' if rule_type == 'scanner_ua' }}>Scanner UA</option>
|
||||
<option value="spoofed_bot" {{ 'selected' if rule_type == 'spoofed_bot' }}>Spoofed Bot</option>
|
||||
</select>
|
||||
</label>
|
||||
</form>
|
||||
</div>
|
||||
|
||||
<div class="mb-6">
|
||||
<h3 class="text-xs font-medium uppercase tracking-wide text-muted dark:text-muted-dark mb-2">Suspicious events</h3>
|
||||
<div id="suspicious-events-grid" data-endpoint="{{ url_for('security.api_security_events') }}"></div>
|
||||
</div>
|
||||
|
||||
<div class="mb-6">
|
||||
<h3 class="text-xs font-medium uppercase tracking-wide text-muted dark:text-muted-dark mb-2">Sensitive-path probes</h3>
|
||||
<div id="sensitive-paths-panel" class="border border-line dark:border-line-dark rounded-xl bg-surface dark:bg-surface-dark overflow-hidden"
|
||||
data-endpoint="{{ url_for('security.api_sensitive_paths') }}"></div>
|
||||
</div>
|
||||
|
||||
<div class="mb-6 border border-line dark:border-line-dark rounded-xl p-4 bg-surface dark:bg-surface-dark">
|
||||
<h3 class="font-display font-semibold text-sm mb-3">Export blocklist</h3>
|
||||
<div class="flex flex-wrap gap-2 items-center">
|
||||
<a href="{{ url_for('security.api_export_blocklist') }}"
|
||||
class="bg-accent dark:bg-accent-dark text-white dark:text-paper-dark rounded-md px-3 py-1.5 text-sm font-medium hover:opacity-90 transition-opacity">Download new (.txt)</a>
|
||||
<a href="{{ url_for('security.api_export_blocklist', all='true') }}"
|
||||
class="border border-line dark:border-line-dark rounded-md px-3 py-1.5 text-sm hover:bg-paper dark:hover:bg-paper-dark transition-colors">Re-export all</a>
|
||||
<select id="blocklist-format" class="border border-line dark:border-line-dark rounded-md px-2 py-1.5 text-sm bg-surface dark:bg-surface-dark text-ink dark:text-ink-dark" onchange="updateBlocklistLinks(this.value)">
|
||||
<option value="plain">Plain IP list</option>
|
||||
<option value="htaccess">.htaccess Deny</option>
|
||||
<option value="iptables">iptables</option>
|
||||
</select>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div id="ip-history-modal" class="fixed inset-0 bg-ink/40 dark:bg-ink-dark/60 items-center justify-center empty:hidden flex z-50"></div>
|
||||
|
||||
<script>
|
||||
(function initSecurityWidgets() {
|
||||
const root = document.getElementById('security-content');
|
||||
const from = root.dataset.from, to = root.dataset.to;
|
||||
const severity = root.dataset.severity, ruleType = root.dataset.ruleType;
|
||||
|
||||
window.initGrid(
|
||||
'suspicious-events-grid',
|
||||
document.getElementById('suspicious-events-grid').dataset.endpoint,
|
||||
[
|
||||
{ name: 'Timestamp' },
|
||||
{
|
||||
name: 'IP',
|
||||
formatter: (cell) => window.gridHtml(
|
||||
`<button class="text-accent dark:text-accent-dark underline" hx-get="/api/security/ip/${cell}" hx-target="#ip-history-modal" hx-swap="innerHTML">${cell}</button>`
|
||||
),
|
||||
},
|
||||
{ name: 'Path' }, { name: 'Rule Matched' },
|
||||
{
|
||||
name: 'Severity',
|
||||
formatter: (cell) => {
|
||||
const tone = { low: 'text-muted dark:text-muted-dark', medium: 'text-warn dark:text-warn-dark', high: 'text-danger dark:text-danger-dark' }[cell] || '';
|
||||
return window.gridHtml(`<span class="font-medium ${tone}">${cell}</span>`);
|
||||
},
|
||||
},
|
||||
],
|
||||
{ from, to, severity, rule_type: ruleType },
|
||||
);
|
||||
|
||||
const pathsEl = document.getElementById('sensitive-paths-panel');
|
||||
fetch(`${pathsEl.dataset.endpoint}?from=${from}&to=${to}`)
|
||||
.then((r) => r.json())
|
||||
.then(({ data }) => {
|
||||
pathsEl.innerHTML = data.length
|
||||
? `<table class="w-full text-sm font-data">
|
||||
<thead><tr class="text-left text-muted dark:text-muted-dark text-xs uppercase tracking-wide bg-surface-raised dark:bg-surface-raised-dark font-sans">
|
||||
<th class="px-3 py-2">Path</th><th class="px-3 py-2">Hits</th><th class="px-3 py-2">Distinct IPs</th>
|
||||
</tr></thead>
|
||||
<tbody class="divide-y divide-line dark:divide-line-dark">${
|
||||
data.map((r) => `<tr><td class="px-3 py-2">${r.path}</td><td class="px-3 py-2">${r.hit_count}</td><td class="px-3 py-2">${r.distinct_ip_count}</td></tr>`).join('')
|
||||
}</tbody></table>`
|
||||
: `<p class="text-sm text-muted dark:text-muted-dark p-4">No sensitive-path probes in range.</p>`;
|
||||
});
|
||||
|
||||
window.updateBlocklistLinks = (fmt) => {
|
||||
document.querySelectorAll('a[href*="export-blocklist"]').forEach((a) => {
|
||||
const url = new URL(a.href, window.location.origin);
|
||||
url.searchParams.set('format', fmt);
|
||||
a.href = url.toString();
|
||||
});
|
||||
};
|
||||
})();
|
||||
</script>
|
||||
</div>
|
||||
@@ -0,0 +1,26 @@
|
||||
<div class="bg-surface dark:bg-surface-dark rounded-xl p-6 max-w-lg w-full relative border border-line dark:border-line-dark">
|
||||
<button class="absolute top-3 right-3 text-muted dark:text-muted-dark hover:text-ink dark:hover:text-ink-dark" onclick="document.getElementById('ip-history-modal').innerHTML=''">
|
||||
<svg class="w-4 h-4"><use href="/static/dist/icons.svg#x"/></svg>
|
||||
</button>
|
||||
<h3 class="font-display font-semibold text-lg mb-3 font-data">{{ ip_data.ip }}</h3>
|
||||
<dl class="text-sm grid grid-cols-2 gap-y-1.5 mb-4 font-data">
|
||||
<dt class="text-muted dark:text-muted-dark font-sans">First seen</dt><dd>{{ ip_data.first_seen }}</dd>
|
||||
<dt class="text-muted dark:text-muted-dark font-sans">Last seen</dt><dd>{{ ip_data.last_seen }}</dd>
|
||||
<dt class="text-muted dark:text-muted-dark font-sans">Total requests</dt><dd>{{ ip_data.total_requests }}</dd>
|
||||
<dt class="text-muted dark:text-muted-dark font-sans">Reputation score</dt><dd>{{ ip_data.reputation_score }}</dd>
|
||||
<dt class="text-muted dark:text-muted-dark font-sans">Flagged</dt>
|
||||
<dd class="{{ 'text-danger dark:text-danger-dark font-medium' if ip_data.is_flagged else '' }}">{{ 'Yes' if ip_data.is_flagged else 'No' }}</dd>
|
||||
{% if ip_data.spoofed_bot_names %}
|
||||
<dt class="text-muted dark:text-muted-dark font-sans">Spoofed bot claims</dt><dd class="text-danger dark:text-danger-dark">{{ ip_data.spoofed_bot_names | join(', ') }}</dd>
|
||||
{% endif %}
|
||||
</dl>
|
||||
<p class="text-xs text-muted dark:text-muted-dark mb-3">{{ ip_data.traffic_window_note }}</p>
|
||||
<h4 class="font-medium text-sm mb-1">Top paths</h4>
|
||||
<ul class="text-sm mb-3 font-data text-ink dark:text-ink-dark space-y-0.5">{% for path, count in ip_data.top_paths %}<li>{{ path }} <span class="text-muted dark:text-muted-dark">— {{ count }}</span></li>{% endfor %}</ul>
|
||||
<h4 class="font-medium text-sm mb-1">Status codes</h4>
|
||||
<ul class="text-sm mb-3 font-data text-ink dark:text-ink-dark space-y-0.5">{% for code, count in ip_data.status_code_distribution.items() %}<li>{{ code }} <span class="text-muted dark:text-muted-dark">— {{ count }}</span></li>{% endfor %}</ul>
|
||||
{% if ip_data.recent_suspicious_events %}
|
||||
<h4 class="font-medium text-sm mb-1">Recent flagged events</h4>
|
||||
<ul class="text-sm font-data text-ink dark:text-ink-dark space-y-0.5">{% for e in ip_data.recent_suspicious_events %}<li>{{ e.timestamp }} — {{ e.path }} <span class="text-muted dark:text-muted-dark">({{ e.rule_matched }}, {{ e.severity }})</span></li>{% endfor %}</ul>
|
||||
{% endif %}
|
||||
</div>
|
||||
@@ -0,0 +1,4 @@
|
||||
<div class="bg-surface dark:bg-surface-dark rounded-xl p-6 max-w-sm w-full border border-line dark:border-line-dark">
|
||||
<p class="text-sm text-ink dark:text-ink-dark">No history found for <span class="font-data">{{ ip }}</span> — it hasn't been seen yet.</p>
|
||||
<button onclick="document.getElementById('ip-history-modal').innerHTML=''" class="mt-3 text-sm text-accent dark:text-accent-dark hover:underline">Close</button>
|
||||
</div>
|
||||
@@ -0,0 +1,5 @@
|
||||
{% extends "base.html" %}
|
||||
{% block title %}Security — Kavosh{% endblock %}
|
||||
{% block content %}
|
||||
{% include "security/_content.html" %}
|
||||
{% endblock %}
|
||||
@@ -0,0 +1,13 @@
|
||||
from flask import Blueprint
|
||||
from flask_login import login_required
|
||||
|
||||
bp = Blueprint("seo", __name__, template_folder="templates")
|
||||
|
||||
|
||||
@bp.before_request
|
||||
@login_required
|
||||
def require_login():
|
||||
pass
|
||||
|
||||
|
||||
from app.blueprints.seo import routes # noqa: E402,F401 registers routes
|
||||
@@ -0,0 +1,140 @@
|
||||
"""Context-builder query functions for the SEO tab (Chapter 09).
|
||||
|
||||
Per Ch09's own Output instruction, bot-related widgets query bot_hits
|
||||
directly (small, indefinitely-retained, indexed — Ch06) rather than a new
|
||||
rollup; only the human-side of the crawled-vs-visited comparison needs
|
||||
the new human_path_stats_daily table.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from collections import defaultdict
|
||||
from datetime import date
|
||||
|
||||
from sqlalchemy import case, func
|
||||
|
||||
from app.extensions import db
|
||||
from app.models.bot_hit import BotHit
|
||||
from app.models.human_path_stats import HumanPathStatsDaily
|
||||
from app.utils.dates import day_bounds
|
||||
|
||||
# ASSUMPTION (flagged): Ch09 doesn't define "major crawler" for the
|
||||
# crawl-frequency chart's series cap.
|
||||
TOP_N_BOTS_FOR_CHART = 6
|
||||
|
||||
|
||||
def get_bot_summary(from_date: date, to_date: date) -> list[dict]:
|
||||
start, end = day_bounds(from_date, to_date)
|
||||
rows = (
|
||||
db.session.query(
|
||||
BotHit.bot_name,
|
||||
func.count().label("hits"),
|
||||
func.sum(case((BotHit.verified.is_(True), 1), else_=0)).label("verified_hits"),
|
||||
func.max(BotHit.timestamp).label("last_seen"),
|
||||
)
|
||||
.filter(BotHit.timestamp >= start, BotHit.timestamp < end)
|
||||
.group_by(BotHit.bot_name)
|
||||
.order_by(func.count().desc())
|
||||
.all()
|
||||
)
|
||||
return [
|
||||
{
|
||||
"bot_name": r.bot_name,
|
||||
"hits": r.hits,
|
||||
"verified_pct": round(r.verified_hits / r.hits * 100, 1) if r.hits else 0.0,
|
||||
"last_seen": r.last_seen.isoformat() if r.last_seen else None,
|
||||
}
|
||||
for r in rows
|
||||
]
|
||||
|
||||
|
||||
def get_crawl_chart_data(from_date: date, to_date: date, bot_name: str | None) -> dict:
|
||||
start, end = day_bounds(from_date, to_date)
|
||||
base_filters = [BotHit.timestamp >= start, BotHit.timestamp < end]
|
||||
|
||||
if bot_name:
|
||||
allowed_bots = [bot_name]
|
||||
else:
|
||||
top_rows = (
|
||||
db.session.query(BotHit.bot_name, func.count().label("hits"))
|
||||
.filter(*base_filters)
|
||||
.group_by(BotHit.bot_name)
|
||||
.order_by(func.count().desc())
|
||||
.limit(TOP_N_BOTS_FOR_CHART)
|
||||
.all()
|
||||
)
|
||||
allowed_bots = [r.bot_name for r in top_rows]
|
||||
|
||||
if not allowed_bots:
|
||||
return {"series": []}
|
||||
|
||||
rows = (
|
||||
db.session.query(func.date(BotHit.timestamp).label("day"), BotHit.bot_name, func.count().label("count"))
|
||||
.filter(*base_filters, BotHit.bot_name.in_(allowed_bots))
|
||||
.group_by("day", BotHit.bot_name)
|
||||
.order_by("day")
|
||||
.all()
|
||||
)
|
||||
points_by_bot: dict[str, list[dict]] = defaultdict(list)
|
||||
for r in rows:
|
||||
points_by_bot[r.bot_name].append({"t": r.day, "count": r.count})
|
||||
|
||||
return {"series": [{"bot_name": b, "points": points_by_bot.get(b, [])} for b in allowed_bots]}
|
||||
|
||||
|
||||
def get_bot_status_codes(from_date: date, to_date: date) -> dict:
|
||||
start, end = day_bounds(from_date, to_date)
|
||||
bucket = case(
|
||||
(BotHit.status_code < 300, "2xx"),
|
||||
(BotHit.status_code < 400, "3xx"),
|
||||
(BotHit.status_code < 500, "4xx"),
|
||||
else_="5xx",
|
||||
)
|
||||
rows = (
|
||||
db.session.query(bucket.label("bucket"), func.count().label("count"))
|
||||
.filter(BotHit.timestamp >= start, BotHit.timestamp < end)
|
||||
.group_by("bucket")
|
||||
.all()
|
||||
)
|
||||
breakdown = {"2xx": 0, "3xx": 0, "4xx": 0, "5xx": 0}
|
||||
for r in rows:
|
||||
breakdown[r.bucket] = r.count
|
||||
|
||||
# Ch09 calls out 404 by name specifically, not just the 4xx bucket.
|
||||
not_found_404 = (
|
||||
db.session.query(func.count())
|
||||
.filter(BotHit.timestamp >= start, BotHit.timestamp < end, BotHit.status_code == 404)
|
||||
.scalar()
|
||||
)
|
||||
return {"breakdown": breakdown, "not_found_404": not_found_404 or 0}
|
||||
|
||||
|
||||
def get_crawled_vs_visited(from_date: date, to_date: date, page: int, per_page: int) -> tuple[list[list], int]:
|
||||
"""Diffed table: one row per path, bot hits vs. human hits."""
|
||||
start, end = day_bounds(from_date, to_date)
|
||||
bot_rows = (
|
||||
db.session.query(BotHit.path, func.count().label("hits"))
|
||||
.filter(BotHit.timestamp >= start, BotHit.timestamp < end)
|
||||
.group_by(BotHit.path)
|
||||
.all()
|
||||
)
|
||||
human_rows = (
|
||||
db.session.query(HumanPathStatsDaily.path, func.sum(HumanPathStatsDaily.count).label("hits"))
|
||||
.filter(HumanPathStatsDaily.date >= from_date, HumanPathStatsDaily.date <= to_date)
|
||||
.group_by(HumanPathStatsDaily.path)
|
||||
.all()
|
||||
)
|
||||
bot_counts = {r.path: r.hits for r in bot_rows}
|
||||
human_counts = {r.path: r.hits for r in human_rows}
|
||||
|
||||
combined = []
|
||||
for path in set(bot_counts) | set(human_counts):
|
||||
b, h = bot_counts.get(path, 0), human_counts.get(path, 0)
|
||||
total = b + h
|
||||
combined.append([path, b, h, round(b / total * 100, 1) if total else 0.0])
|
||||
|
||||
# ASSUMPTION (flagged): sorted by bot hits desc — Ch09 doesn't specify.
|
||||
combined.sort(key=lambda row: row[1], reverse=True)
|
||||
|
||||
total_count = len(combined)
|
||||
offset = (page - 1) * per_page
|
||||
return combined[offset : offset + per_page], total_count
|
||||
@@ -0,0 +1,56 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from flask import jsonify, request
|
||||
|
||||
from app.blueprints.seo import bp
|
||||
from app.blueprints.seo.queries import (
|
||||
get_bot_status_codes,
|
||||
get_bot_summary,
|
||||
get_crawl_chart_data,
|
||||
get_crawled_vs_visited,
|
||||
)
|
||||
from app.utils.dates import parse_date_range
|
||||
from app.utils.htmx import render_htmx_aware
|
||||
from app.utils.pagination import parse_pagination
|
||||
|
||||
|
||||
@bp.route("/seo")
|
||||
def seo():
|
||||
from_date, to_date = parse_date_range(request)
|
||||
return render_htmx_aware(
|
||||
request, full_template="seo/index.html", partial_template="seo/_content.html",
|
||||
from_date=from_date, to_date=to_date,
|
||||
)
|
||||
|
||||
|
||||
@bp.get("/api/seo/bot-summary")
|
||||
def api_bot_summary():
|
||||
from_date, to_date = parse_date_range(request)
|
||||
return jsonify(data=get_bot_summary(from_date, to_date), meta={"from": from_date.isoformat(), "to": to_date.isoformat()})
|
||||
|
||||
|
||||
@bp.get("/api/seo/crawl-chart-data")
|
||||
def api_crawl_chart_data():
|
||||
from_date, to_date = parse_date_range(request)
|
||||
bot_name = request.args.get("bot")
|
||||
return jsonify(
|
||||
data=get_crawl_chart_data(from_date, to_date, bot_name),
|
||||
meta={"from": from_date.isoformat(), "to": to_date.isoformat(), "bot": bot_name},
|
||||
)
|
||||
|
||||
|
||||
@bp.get("/api/seo/bot-status-codes")
|
||||
def api_bot_status_codes():
|
||||
from_date, to_date = parse_date_range(request)
|
||||
return jsonify(data=get_bot_status_codes(from_date, to_date), meta={"from": from_date.isoformat(), "to": to_date.isoformat()})
|
||||
|
||||
|
||||
@bp.get("/api/seo/crawled-vs-visited")
|
||||
def api_crawled_vs_visited():
|
||||
from_date, to_date = parse_date_range(request)
|
||||
page, per_page = parse_pagination(request)
|
||||
rows, total = get_crawled_vs_visited(from_date, to_date, page, per_page)
|
||||
return jsonify(
|
||||
data={"rows": rows, "total": total},
|
||||
meta={"from": from_date.isoformat(), "to": to_date.isoformat(), "page": page, "per_page": per_page},
|
||||
)
|
||||
@@ -0,0 +1,102 @@
|
||||
<div id="seo-content"
|
||||
hx-get="{{ url_for('seo.seo') }}"
|
||||
hx-trigger="change from:#seo-date-range-form"
|
||||
hx-include="#seo-date-range-form"
|
||||
hx-target="#seo-content"
|
||||
hx-swap="outerHTML"
|
||||
data-from="{{ from_date.isoformat() }}"
|
||||
data-to="{{ to_date.isoformat() }}">
|
||||
|
||||
<div class="flex flex-wrap items-end justify-between gap-4 mb-6">
|
||||
<div>
|
||||
<h1 class="font-display font-bold text-xl">SEO & Bot Behavior</h1>
|
||||
<p class="text-sm text-muted dark:text-muted-dark">How search engines are crawling your site</p>
|
||||
</div>
|
||||
<form id="seo-date-range-form" class="flex gap-3 items-end">
|
||||
<label class="text-sm text-muted dark:text-muted-dark">From
|
||||
<input type="date" name="from" value="{{ from_date.isoformat() }}"
|
||||
class="block border border-line dark:border-line-dark rounded-md px-2 py-1 mt-1 bg-surface dark:bg-surface-dark text-ink dark:text-ink-dark font-data text-sm">
|
||||
</label>
|
||||
<label class="text-sm text-muted dark:text-muted-dark">To
|
||||
<input type="date" name="to" value="{{ to_date.isoformat() }}"
|
||||
class="block border border-line dark:border-line-dark rounded-md px-2 py-1 mt-1 bg-surface dark:bg-surface-dark text-ink dark:text-ink-dark font-data text-sm">
|
||||
</label>
|
||||
</form>
|
||||
</div>
|
||||
|
||||
<div id="bot-summary-cards" class="grid grid-cols-2 sm:grid-cols-3 gap-3 mb-6"
|
||||
data-endpoint="{{ url_for('seo.api_bot_summary') }}"></div>
|
||||
|
||||
<div class="grid grid-cols-1 lg:grid-cols-2 gap-4 mb-6">
|
||||
<div class="border border-line dark:border-line-dark rounded-xl p-4 bg-surface dark:bg-surface-dark h-72">
|
||||
<h3 class="text-xs font-medium uppercase tracking-wide text-muted dark:text-muted-dark mb-2">Crawl frequency</h3>
|
||||
<div class="h-56"><canvas id="crawl-frequency-chart" data-endpoint="{{ url_for('seo.api_crawl_chart_data') }}"></canvas></div>
|
||||
</div>
|
||||
<div class="border border-line dark:border-line-dark rounded-xl p-4 bg-surface dark:bg-surface-dark h-72">
|
||||
<h3 class="text-xs font-medium uppercase tracking-wide text-muted dark:text-muted-dark mb-2">Status codes served to bots</h3>
|
||||
<div class="h-44"><canvas id="bot-status-codes-chart" data-endpoint="{{ url_for('seo.api_bot_status_codes') }}"></canvas></div>
|
||||
<p id="bot-404-note" class="text-xs text-muted dark:text-muted-dark mt-2"></p>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div>
|
||||
<h3 class="text-xs font-medium uppercase tracking-wide text-muted dark:text-muted-dark mb-2">Most-crawled vs. most-visited URLs</h3>
|
||||
<div id="crawled-vs-visited-grid" data-endpoint="{{ url_for('seo.api_crawled_vs_visited') }}"></div>
|
||||
</div>
|
||||
|
||||
<script>
|
||||
(function initSeoWidgets() {
|
||||
const root = document.getElementById('seo-content');
|
||||
const from = root.dataset.from, to = root.dataset.to;
|
||||
const withRange = (url) => `${url}?from=${from}&to=${to}`;
|
||||
|
||||
fetch(withRange(document.getElementById('bot-summary-cards').dataset.endpoint))
|
||||
.then((r) => r.json())
|
||||
.then(({ data }) => {
|
||||
document.getElementById('bot-summary-cards').innerHTML = data.length ? data.map((bot) => `
|
||||
<div class="bg-surface dark:bg-surface-dark rounded-lg border border-line dark:border-line-dark border-t-2 ${bot.verified_pct < 100 ? 'border-t-warn dark:border-t-warn-dark' : 'border-t-ok dark:border-t-ok-dark'} p-3">
|
||||
<p class="text-xs text-muted dark:text-muted-dark uppercase tracking-wide">${bot.bot_name}</p>
|
||||
<p class="font-data text-xl font-medium mt-0.5">${bot.hits.toLocaleString()} <span class="text-sm text-muted dark:text-muted-dark font-sans">hits</span></p>
|
||||
<p class="text-xs mt-1 ${bot.verified_pct < 100 ? 'text-warn dark:text-warn-dark' : 'text-muted dark:text-muted-dark'}">
|
||||
${bot.verified_pct}% verified${bot.verified_pct < 100 ? ' — some spoofed' : ''}
|
||||
</p>
|
||||
<p class="text-xs text-muted dark:text-muted-dark font-data mt-0.5">Last seen: ${bot.last_seen ?? '—'}</p>
|
||||
</div>`).join('') : `<p class="text-sm text-muted dark:text-muted-dark col-span-full">No bot activity in this range.</p>`;
|
||||
});
|
||||
|
||||
const crawlEl = document.getElementById('crawl-frequency-chart');
|
||||
fetch(withRange(crawlEl.dataset.endpoint))
|
||||
.then((r) => r.json())
|
||||
.then(({ data }) => window.initChart('crawl-frequency-chart', {
|
||||
type: 'line',
|
||||
data: {
|
||||
labels: [...new Set(data.series.flatMap((s) => s.points.map((p) => p.t)))].sort(),
|
||||
datasets: data.series.map((s, i) => ({
|
||||
label: s.bot_name,
|
||||
data: s.points.map((p) => ({ x: p.t, y: p.count })),
|
||||
tension: 0.3,
|
||||
borderColor: window.KAVOSH_CHART_PALETTE[i % window.KAVOSH_CHART_PALETTE.length],
|
||||
})),
|
||||
},
|
||||
}));
|
||||
|
||||
const statusEl = document.getElementById('bot-status-codes-chart');
|
||||
fetch(withRange(statusEl.dataset.endpoint))
|
||||
.then((r) => r.json())
|
||||
.then(({ data }) => {
|
||||
window.initChart('bot-status-codes-chart', {
|
||||
type: 'bar',
|
||||
data: { labels: Object.keys(data.breakdown), datasets: [{ label: 'Bot Requests', data: Object.values(data.breakdown), backgroundColor: '#0E7C86' }] },
|
||||
});
|
||||
document.getElementById('bot-404-note').textContent = `${data.not_found_404} 404s served to bots in range — wasted crawl budget.`;
|
||||
});
|
||||
|
||||
window.initGrid(
|
||||
'crawled-vs-visited-grid',
|
||||
document.getElementById('crawled-vs-visited-grid').dataset.endpoint,
|
||||
[{ name: 'Path' }, { name: 'Bot Hits' }, { name: 'Human Hits' }, { name: 'Bot Share %' }],
|
||||
{ from, to },
|
||||
);
|
||||
})();
|
||||
</script>
|
||||
</div>
|
||||
@@ -0,0 +1,5 @@
|
||||
{% extends "base.html" %}
|
||||
{% block title %}SEO & Bots — Kavosh{% endblock %}
|
||||
{% block content %}
|
||||
{% include "seo/_content.html" %}
|
||||
{% endblock %}
|
||||
@@ -0,0 +1,15 @@
|
||||
from flask import Blueprint
|
||||
from flask_login import login_required
|
||||
|
||||
bp = Blueprint("uploads", __name__, template_folder="templates")
|
||||
|
||||
|
||||
@bp.before_request
|
||||
@login_required
|
||||
def require_login():
|
||||
"""Ch01: dashboard reachable from one AUTHENTICATED shell; Ch12:
|
||||
single-admin login. All routes on this blueprint require a session."""
|
||||
pass
|
||||
|
||||
|
||||
from app.blueprints.uploads import routes # noqa: E402,F401 registers routes
|
||||
@@ -0,0 +1,51 @@
|
||||
"""Query/formatting helpers for the "Uploaded files" list (project-owner
|
||||
follow-up request). Kept separate from routes.py to match this project's
|
||||
established per-blueprint queries.py convention (Ch08/09/10).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from app.models.log_file import LogFile
|
||||
|
||||
|
||||
def get_uploaded_files(page: int, per_page: int) -> tuple[list[list], int]:
|
||||
"""Newest first, independent of the dashboard date-range picker —
|
||||
this lists uploads by when they arrived, not by which log dates they
|
||||
contain (a single file can span many dates).
|
||||
"""
|
||||
query = LogFile.query.order_by(LogFile.uploaded_at.desc())
|
||||
total = query.count()
|
||||
rows = query.offset((page - 1) * per_page).limit(per_page).all()
|
||||
|
||||
result = []
|
||||
for lf in rows:
|
||||
result.append([
|
||||
lf.id,
|
||||
lf.filename,
|
||||
lf.server_type,
|
||||
_status_display(lf),
|
||||
lf.uploaded_at.strftime("%Y-%m-%d %H:%M"),
|
||||
_human_size(lf.size_bytes),
|
||||
lf.status, # raw status (hidden column) — lets the client disable
|
||||
# the select checkbox for files still "processing"
|
||||
])
|
||||
return result, total
|
||||
|
||||
|
||||
def _status_display(lf: LogFile) -> str:
|
||||
if lf.status == "processing":
|
||||
if lf.total_lines:
|
||||
pct = round(lf.processed_lines / lf.total_lines * 100)
|
||||
return f"processing ({pct}%)"
|
||||
return "processing"
|
||||
if lf.status == "error":
|
||||
return f"error: {lf.error_message}" if lf.error_message else "error"
|
||||
return lf.status
|
||||
|
||||
|
||||
def _human_size(num_bytes: int) -> str:
|
||||
size = float(num_bytes)
|
||||
for unit in ("B", "KB", "MB", "GB"):
|
||||
if size < 1024 or unit == "GB":
|
||||
return f"{size:.0f} {unit}" if unit == "B" else f"{size:.1f} {unit}"
|
||||
size /= 1024
|
||||
return f"{size:.1f} GB"
|
||||
@@ -0,0 +1,188 @@
|
||||
"""Upload endpoint (Chapter 04): validated, streamed-to-disk save.
|
||||
|
||||
Parsing happens in a background thread triggered right after this request
|
||||
completes (app/services/background.py) — never synchronously inside this
|
||||
request (Chapter 03, rule 5 still holds: the response returns immediately
|
||||
regardless of file size).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import uuid
|
||||
from pathlib import Path
|
||||
|
||||
from flask import current_app, jsonify, render_template, request
|
||||
from werkzeug.utils import secure_filename
|
||||
|
||||
from app.blueprints.uploads import bp
|
||||
from app.blueprints.uploads.queries import get_uploaded_files
|
||||
from app.extensions import db, limiter
|
||||
from app.models.log_file import LogFile
|
||||
from app.services.background import trigger_processing
|
||||
from app.services.file_deletion import delete_log_files
|
||||
from app.utils.pagination import parse_pagination
|
||||
from app.utils.upload_paths import upload_path_for
|
||||
|
||||
_ALLOWED_EXTENSIONS = {".log", ".txt", ".gz"}
|
||||
_CHUNK_SIZE = 64 * 1024 # 64 KB per read — never buffer the whole upload
|
||||
_GZIP_MAGIC = b"\x1f\x8b"
|
||||
|
||||
|
||||
class UploadRejected(Exception):
|
||||
"""Raised when an upload fails extension/content validation."""
|
||||
|
||||
|
||||
def _validate_extension(filename: str) -> str:
|
||||
ext = Path(filename).suffix.lower()
|
||||
if ext not in _ALLOWED_EXTENSIONS:
|
||||
raise UploadRejected(f"Unsupported extension {ext!r}; allowed: {_ALLOWED_EXTENSIONS}")
|
||||
return ext
|
||||
|
||||
|
||||
def _sniff_content(first_chunk: bytes, ext: str) -> None:
|
||||
"""Light content sniff — don't just trust the client-supplied MIME type."""
|
||||
if ext == ".gz":
|
||||
if not first_chunk.startswith(_GZIP_MAGIC):
|
||||
raise UploadRejected("File has a .gz extension but isn't gzip-magic-prefixed.")
|
||||
return
|
||||
if b"\x00" in first_chunk:
|
||||
raise UploadRejected("File extension claims text but content looks binary.")
|
||||
|
||||
|
||||
def _stream_to_temp(file_storage, tmp_path: Path) -> tuple[int, str, bytes]:
|
||||
"""Stream the upload to disk in bounded chunks; return (size, sha256_hex, first_chunk).
|
||||
|
||||
Never calls file.read() on the whole stream (Chapter 03, rule 1).
|
||||
"""
|
||||
sha256 = hashlib.sha256()
|
||||
size = 0
|
||||
first_chunk: bytes | None = None
|
||||
with tmp_path.open("wb") as out:
|
||||
while True:
|
||||
chunk = file_storage.stream.read(_CHUNK_SIZE)
|
||||
if not chunk:
|
||||
break
|
||||
if first_chunk is None:
|
||||
first_chunk = chunk
|
||||
sha256.update(chunk)
|
||||
size += len(chunk)
|
||||
out.write(chunk)
|
||||
if first_chunk is None:
|
||||
raise UploadRejected("Uploaded file is empty.")
|
||||
return size, sha256.hexdigest(), first_chunk
|
||||
|
||||
|
||||
@bp.post("/uploads")
|
||||
@limiter.limit("20 per minute") # Chapter 12: rate limiting on /uploads at minimum
|
||||
def upload_log_file():
|
||||
"""Validate, stream, and register an uploaded access log (Ch04/Ch11)."""
|
||||
file_storage = request.files.get("logfile")
|
||||
if file_storage is None or not file_storage.filename:
|
||||
return render_template("uploads/_error.html", message="No file provided."), 400
|
||||
|
||||
upload_dir = Path(current_app.config["UPLOAD_DIR"])
|
||||
(upload_dir / "tmp").mkdir(parents=True, exist_ok=True)
|
||||
tmp_path = upload_dir / "tmp" / f"{uuid.uuid4().hex}.part"
|
||||
|
||||
try:
|
||||
ext = _validate_extension(file_storage.filename)
|
||||
size_bytes, checksum, first_chunk = _stream_to_temp(file_storage, tmp_path)
|
||||
_sniff_content(first_chunk, ext)
|
||||
|
||||
max_bytes = current_app.config["UPLOAD_MAX_SIZE_MB"] * 1024 * 1024
|
||||
if size_bytes > max_bytes:
|
||||
raise UploadRejected(f"File exceeds {current_app.config['UPLOAD_MAX_SIZE_MB']}MB limit.")
|
||||
except UploadRejected as exc:
|
||||
tmp_path.unlink(missing_ok=True)
|
||||
return render_template("uploads/_error.html", message=str(exc)), 400
|
||||
|
||||
existing = LogFile.query.filter_by(checksum=checksum).first()
|
||||
if existing is not None:
|
||||
tmp_path.unlink(missing_ok=True)
|
||||
return render_template("uploads/_duplicate.html", log_file=existing)
|
||||
|
||||
log_file = LogFile(
|
||||
filename=secure_filename(file_storage.filename),
|
||||
server_type=request.form.get("server_type", "apache"),
|
||||
format_string=request.form.get("format_string", ""),
|
||||
status="queued",
|
||||
size_bytes=size_bytes,
|
||||
checksum=checksum,
|
||||
)
|
||||
db.session.add(log_file)
|
||||
db.session.commit() # need the assigned id before the final rename
|
||||
|
||||
tmp_path.rename(upload_path_for(log_file))
|
||||
|
||||
# Chapter 12 simplification: no cron required — kick off processing
|
||||
# immediately in a background thread. The response below returns as
|
||||
# soon as the file is queued (Ch03 rule 5 still holds: this request
|
||||
# never blocks on parsing), while the thread runs independently.
|
||||
trigger_processing(current_app._get_current_object(), log_file.id)
|
||||
|
||||
return render_template("uploads/_queued.html", log_file=log_file)
|
||||
|
||||
|
||||
@bp.get("/api/uploads/<int:log_file_id>/status")
|
||||
def upload_status(log_file_id: int):
|
||||
"""Polled every 3s by the browser (hx-trigger) until done/error (Ch04)."""
|
||||
log_file = db.get_or_404(LogFile, log_file_id)
|
||||
|
||||
if request.headers.get("Accept") == "application/json":
|
||||
return jsonify(
|
||||
data={
|
||||
"id": log_file.id,
|
||||
"status": log_file.status,
|
||||
"processed_lines": log_file.processed_lines,
|
||||
"total_lines": log_file.total_lines,
|
||||
},
|
||||
meta={},
|
||||
)
|
||||
|
||||
template = {
|
||||
"done": "uploads/_status_done.html",
|
||||
"error": "uploads/_status_error.html",
|
||||
# BUG FIX: this key was missing, so "queued" fell through to the
|
||||
# "processing" fallback below — a file that hadn't been picked up
|
||||
# by `flask process-logs` yet displayed as "Processing X: 0 lines"
|
||||
# instead of "Queued — waiting for the next parse cycle", making a
|
||||
# cron job that simply hasn't run yet indistinguishable from one
|
||||
# that's actually hung mid-parse.
|
||||
"queued": "uploads/_queued.html",
|
||||
}.get(log_file.status, "uploads/_status_processing.html")
|
||||
return render_template(template, log_file=log_file)
|
||||
|
||||
|
||||
@bp.get("/api/uploads")
|
||||
def list_uploads():
|
||||
"""Uploaded-files list (project-owner follow-up request) — Grid.js-
|
||||
backed, same page/per_page convention as every other table (Ch11).
|
||||
Sorted by upload recency, independent of the dashboard date-range
|
||||
picker (a single file can span many log dates).
|
||||
"""
|
||||
page, per_page = parse_pagination(request)
|
||||
rows, total = get_uploaded_files(page, per_page)
|
||||
return jsonify(
|
||||
data={"rows": rows, "total": total},
|
||||
meta={"page": page, "per_page": per_page},
|
||||
)
|
||||
|
||||
|
||||
@bp.delete("/api/uploads")
|
||||
def bulk_delete_uploads():
|
||||
"""Delete one or more uploaded files and everything derived from them
|
||||
(log_entries/bot_hits/suspicious_events, the raw file on disk, and a
|
||||
correct rollup recompute for the affected dates — see
|
||||
app/services/file_deletion.py for why a rollup recompute is needed
|
||||
rather than a simple per-file delete).
|
||||
|
||||
Body: {"ids": [1, 2, 3]}. Files currently "processing" are skipped,
|
||||
not force-deleted, to avoid racing the background parse thread.
|
||||
"""
|
||||
body = request.get_json(silent=True) or {}
|
||||
ids = body.get("ids")
|
||||
if not isinstance(ids, list) or not ids or not all(isinstance(i, int) for i in ids):
|
||||
return jsonify(error={"code": "invalid_request", "message": "Expected {\"ids\": [int, ...]}."}), 400
|
||||
|
||||
result = delete_log_files(ids)
|
||||
return jsonify(data={"deleted": result.deleted, "skipped": result.skipped}, meta={})
|
||||
@@ -0,0 +1,4 @@
|
||||
<div class="border border-warn/30 dark:border-warn-dark/30 bg-warn/5 dark:bg-warn-dark/10 rounded-lg p-3 text-sm">
|
||||
<span class="font-data">{{ log_file.filename }}</span>
|
||||
<span class="text-muted dark:text-muted-dark">matches an already-uploaded file (status: {{ log_file.status }}); skipped re-upload.</span>
|
||||
</div>
|
||||
@@ -0,0 +1,3 @@
|
||||
<div class="border border-danger/30 dark:border-danger-dark/30 bg-danger/5 dark:bg-danger-dark/10 rounded-lg p-3 text-sm text-danger dark:text-danger-dark">
|
||||
{{ message }}
|
||||
</div>
|
||||
@@ -0,0 +1,10 @@
|
||||
<div id="upload-status-{{ log_file.id }}"
|
||||
hx-get="{{ url_for('uploads.upload_status', log_file_id=log_file.id) }}"
|
||||
hx-trigger="every 1s" hx-swap="outerHTML"
|
||||
class="border border-line dark:border-line-dark rounded-lg p-3 bg-paper dark:bg-paper-dark">
|
||||
<div class="flex items-center gap-2 text-sm">
|
||||
<svg class="w-4 h-4 text-muted dark:text-muted-dark animate-spin"><use href="/static/dist/icons.svg#loader-circle"/></svg>
|
||||
<span class="font-data">{{ log_file.filename }}</span>
|
||||
<span class="text-muted dark:text-muted-dark">— queued, starting shortly…</span>
|
||||
</div>
|
||||
</div>
|
||||
@@ -0,0 +1,7 @@
|
||||
<div id="upload-status-{{ log_file.id }}" class="border border-ok/30 dark:border-ok-dark/30 bg-ok/5 dark:bg-ok-dark/10 rounded-lg p-3">
|
||||
<div class="flex items-center gap-2 text-sm">
|
||||
<span class="w-2 h-2 rounded-full bg-ok dark:bg-ok-dark shrink-0"></span>
|
||||
<span class="font-data">{{ log_file.filename }}</span>
|
||||
<span class="text-ok dark:text-ok-dark">— done, {{ "{:,}".format(log_file.processed_lines) }} lines analyzed</span>
|
||||
</div>
|
||||
</div>
|
||||
@@ -0,0 +1,7 @@
|
||||
<div id="upload-status-{{ log_file.id }}" class="border border-danger/30 dark:border-danger-dark/30 bg-danger/5 dark:bg-danger-dark/10 rounded-lg p-3">
|
||||
<div class="flex items-center gap-2 text-sm">
|
||||
<span class="w-2 h-2 rounded-full bg-danger dark:bg-danger-dark shrink-0"></span>
|
||||
<span class="font-data">{{ log_file.filename }}</span>
|
||||
</div>
|
||||
<p class="text-danger dark:text-danger-dark text-xs mt-1">{{ log_file.error_message }}</p>
|
||||
</div>
|
||||
@@ -0,0 +1,19 @@
|
||||
{% set pct = ((log_file.processed_lines / log_file.total_lines) * 100) if log_file.total_lines else None %}
|
||||
<div id="upload-status-{{ log_file.id }}"
|
||||
hx-get="{{ url_for('uploads.upload_status', log_file_id=log_file.id) }}"
|
||||
hx-trigger="every 1s" hx-swap="outerHTML"
|
||||
class="border border-line dark:border-line-dark rounded-lg p-3 bg-paper dark:bg-paper-dark">
|
||||
<div class="flex items-center justify-between text-sm mb-2">
|
||||
<span class="font-data">{{ log_file.filename }}</span>
|
||||
<span class="font-data text-muted dark:text-muted-dark">
|
||||
{% if pct is not none %}{{ pct | round(0) | int }}%{% else %}analyzing…{% endif %}
|
||||
</span>
|
||||
</div>
|
||||
<div class="w-full h-2 rounded-full bg-line dark:bg-line-dark overflow-hidden">
|
||||
<div class="h-full rounded-full bg-accent dark:bg-accent-dark transition-all duration-500 ease-out"
|
||||
style="width: {{ pct | round(1) if pct is not none else 8 }}%"></div>
|
||||
</div>
|
||||
<p class="text-xs text-muted dark:text-muted-dark mt-1.5 font-data">
|
||||
{{ "{:,}".format(log_file.processed_lines) }}{% if log_file.total_lines %} / {{ "{:,}".format(log_file.total_lines) }}{% endif %} lines
|
||||
</p>
|
||||
</div>
|
||||
@@ -0,0 +1,82 @@
|
||||
"""Concrete, checkable translation of the cPanel account ceiling (Chapter 03).
|
||||
|
||||
Single source of truth for the account's literal resource limits. Other
|
||||
chapters should import BUDGET rather than re-hardcoding these numbers.
|
||||
validate_budget_config() is called from create_app() so an out-of-budget
|
||||
deployment fails loudly at startup instead of silently degrading under load
|
||||
— this is the concrete mechanism behind Chapter 03's "flag, don't silently
|
||||
accept" requirement.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
|
||||
from flask import Flask
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class AccountBudget:
|
||||
"""The hosting account's literal ceiling (Chapter 03).
|
||||
|
||||
iops/io_throughput are informational only here — Python config can't
|
||||
enforce them directly; they constrain how Ch06/07 batch reads/writes.
|
||||
"""
|
||||
|
||||
cpu_cores: int = 4
|
||||
max_entry_processes: int = 60
|
||||
memory_mb: int = 2048
|
||||
iops: int = 1024
|
||||
io_throughput_mb_s: int = 16
|
||||
max_total_processes: int = 150
|
||||
max_db_connections: int = 150
|
||||
|
||||
# Derived engineering targets from Chapter 03's budget table.
|
||||
db_pool_size_min: int = 2
|
||||
db_pool_size_max: int = 5
|
||||
|
||||
|
||||
BUDGET = AccountBudget()
|
||||
|
||||
_DISALLOWED_CACHE_BACKENDS = {"redis", "rediscache", "memcached", "memcachedcache"}
|
||||
|
||||
|
||||
def validate_budget_config(app: Flask) -> None:
|
||||
"""Raise at startup if config violates a Chapter 03 rule.
|
||||
|
||||
Cheap checks only (string/int comparisons) since this runs on every
|
||||
process boot — Passenger may recycle processes frequently (factor 9).
|
||||
"""
|
||||
# Validated as its own config key, not read out of
|
||||
# SQLALCHEMY_ENGINE_OPTIONS — that dict is empty for SQLite (its pool
|
||||
# classes reject pool_size/pool_recycle outright; see app/config.py's
|
||||
# _engine_options_for), so the *intended* setting must be checked
|
||||
# independently of whether the active engine actually consumes it.
|
||||
pool_size = app.config.get("DB_POOL_SIZE")
|
||||
if pool_size is None or not (BUDGET.db_pool_size_min <= pool_size <= BUDGET.db_pool_size_max):
|
||||
raise RuntimeError(
|
||||
f"DB_POOL_SIZE={pool_size} is outside the Chapter 03 budget "
|
||||
f"({BUDGET.db_pool_size_min}-{BUDGET.db_pool_size_max} per process; "
|
||||
f"{BUDGET.max_db_connections} total connections are shared across "
|
||||
f"up to {BUDGET.max_entry_processes} entry processes)."
|
||||
)
|
||||
|
||||
cache_type = str(app.config.get("CACHE_TYPE", "")).lower()
|
||||
if any(name in cache_type for name in _DISALLOWED_CACHE_BACKENDS):
|
||||
raise RuntimeError(
|
||||
f"CACHE_TYPE={app.config.get('CACHE_TYPE')!r} assumes a backend "
|
||||
"(Redis/Memcached) Chapter 03 says not to assume is available. "
|
||||
"Use FileSystemCache or a dashboard_cache DB table (Ch06)."
|
||||
)
|
||||
|
||||
batch_size = app.config.get("PARSE_BATCH_SIZE")
|
||||
if not batch_size or batch_size <= 0:
|
||||
raise RuntimeError(
|
||||
"PARSE_BATCH_SIZE must be a positive integer — unbounded/whole-file "
|
||||
"parsing per invocation violates the Chapter 03 memory budget."
|
||||
)
|
||||
|
||||
max_upload_mb = app.config.get("UPLOAD_MAX_SIZE_MB")
|
||||
if not max_upload_mb or max_upload_mb <= 0:
|
||||
raise RuntimeError(
|
||||
"UPLOAD_MAX_SIZE_MB must be a positive integer to bound disk/IOPS per upload."
|
||||
)
|
||||
+163
@@ -0,0 +1,163 @@
|
||||
"""Flask CLI admin commands (factor 12).
|
||||
|
||||
process-logs is now OPTIONAL (Chapter 12 simplification, per project
|
||||
owner request): file processing is triggered automatically in-app right
|
||||
after upload (app/services/background.py), so cron is no longer required.
|
||||
This command still exists for anyone who'd rather run it manually or via
|
||||
cron — it shares the exact same processing code (app/services/
|
||||
log_processor.py) as the automatic background trigger, so both paths
|
||||
behave identically. cleanup (Chapter 06/12) enforces the retention
|
||||
policy. create-admin (Chapter 12) bootstraps the single admin account.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import date, datetime, timedelta
|
||||
|
||||
import click
|
||||
from flask import Flask, current_app
|
||||
|
||||
from app.extensions import db
|
||||
from app.models.ip_traffic_stats import IpPathStatsDaily, IpStatusStatsDaily
|
||||
from app.models.log_entry import LogEntry
|
||||
from app.models.log_file import LogFile
|
||||
from app.models.request_stats import RequestStatsHourly
|
||||
from app.models.user import User
|
||||
from app.services import aggregator
|
||||
from app.services.log_processor import process_one_batch
|
||||
from app.utils.upload_paths import upload_path_for
|
||||
|
||||
# Chapter 06 retention policy — the constants `flask cleanup` enforces.
|
||||
LOG_ENTRIES_RETENTION_DAYS = 30
|
||||
HOURLY_STATS_RETENTION_DAYS = 90
|
||||
IP_TRAFFIC_STATS_RETENTION_DAYS = 30
|
||||
|
||||
|
||||
def register_commands(app: Flask) -> None:
|
||||
app.cli.add_command(process_logs)
|
||||
app.cli.add_command(rollup)
|
||||
app.cli.add_command(cleanup)
|
||||
app.cli.add_command(create_admin)
|
||||
|
||||
|
||||
@click.command("process-logs")
|
||||
@click.option("--batch-size", default=None, type=int, help="Override PARSE_BATCH_SIZE.")
|
||||
def process_logs(batch_size: int | None) -> None:
|
||||
"""Optional manual/cron fallback — processing now also runs
|
||||
automatically in-app after upload. Parses queued/processing
|
||||
log_files in bounded, checkpointed batches; one batch per file per
|
||||
invocation, same as before.
|
||||
"""
|
||||
batch = batch_size or current_app.config["PARSE_BATCH_SIZE"]
|
||||
pending = LogFile.query.filter(LogFile.status.in_(["queued", "processing"])).all()
|
||||
for log_file in pending:
|
||||
process_one_batch(log_file, batch)
|
||||
|
||||
|
||||
@click.command("rollup")
|
||||
@click.option("--from", "from_date", required=True, help="ISO date, e.g. 2026-07-01")
|
||||
@click.option("--to", "to_date", required=True, help="ISO date, e.g. 2026-07-26")
|
||||
def rollup(from_date: str, to_date: str) -> None:
|
||||
"""Manual rollup recompute for an explicit range (e.g. after a backfill).
|
||||
|
||||
Requires an explicit range — no "all time" default, mirroring Ch03 rule 7
|
||||
even for an admin command, to avoid an unbounded scan on constrained hosting.
|
||||
"""
|
||||
start = date.fromisoformat(from_date)
|
||||
end = date.fromisoformat(to_date)
|
||||
aggregator.compute_rollups_for_range(start, end)
|
||||
click.echo(f"Rolled up {start} .. {end}")
|
||||
|
||||
|
||||
@click.command("cleanup")
|
||||
def cleanup() -> None:
|
||||
"""Enforce the Chapter 06 retention policy (Chapter 12).
|
||||
|
||||
Previously a stub through every earlier chapter — implemented here as
|
||||
Chapter 12's non-functional/deployment concern. Cron-invoked (e.g.
|
||||
daily, off-peak), never a long-running daemon.
|
||||
"""
|
||||
now = datetime.utcnow()
|
||||
deleted_entries = _delete_old_log_entries(now)
|
||||
deleted_hourly = _collapse_old_hourly_stats(now)
|
||||
deleted_ip_stats = _delete_old_ip_traffic_stats(now)
|
||||
archived_files = _delete_parsed_upload_files()
|
||||
click.echo(
|
||||
f"Cleanup complete: {deleted_entries} log_entries, {deleted_hourly} "
|
||||
f"request_stats_hourly, {deleted_ip_stats} ip traffic-rollup rows "
|
||||
f"deleted; {archived_files} parsed upload file(s) removed from disk."
|
||||
)
|
||||
|
||||
|
||||
def _delete_old_log_entries(now: datetime) -> int:
|
||||
"""Ch06: raw log_entries retained ~30 days, then deleted."""
|
||||
cutoff = now - timedelta(days=LOG_ENTRIES_RETENTION_DAYS)
|
||||
count = db.session.query(LogEntry).filter(LogEntry.timestamp < cutoff).delete(synchronize_session=False)
|
||||
db.session.commit()
|
||||
return count
|
||||
|
||||
|
||||
def _collapse_old_hourly_stats(now: datetime) -> int:
|
||||
"""Ch06: request_stats_hourly retained ~90 days, then collapsed into
|
||||
request_stats_daily only. request_stats_daily is already computed
|
||||
independently by the aggregator straight from raw log_entries, so
|
||||
"collapsing" here just means deleting the now-redundant hourly rows
|
||||
once the retention window passes — no data is lost, since the daily
|
||||
rollup for that period was already written when the file was parsed.
|
||||
"""
|
||||
cutoff = now - timedelta(days=HOURLY_STATS_RETENTION_DAYS)
|
||||
count = db.session.query(RequestStatsHourly).filter(
|
||||
RequestStatsHourly.date_hour < cutoff
|
||||
).delete(synchronize_session=False)
|
||||
db.session.commit()
|
||||
return count
|
||||
|
||||
|
||||
def _delete_old_ip_traffic_stats(now: datetime) -> int:
|
||||
"""The bounded per-IP rollup (added as a Ch10 follow-up) was designed
|
||||
for ~30-day retention, matching log_entries — see aggregator.py.
|
||||
"""
|
||||
cutoff_date = (now - timedelta(days=IP_TRAFFIC_STATS_RETENTION_DAYS)).date()
|
||||
count = db.session.query(IpPathStatsDaily).filter(IpPathStatsDaily.date < cutoff_date).delete(synchronize_session=False)
|
||||
count += db.session.query(IpStatusStatsDaily).filter(IpStatusStatsDaily.date < cutoff_date).delete(synchronize_session=False)
|
||||
db.session.commit()
|
||||
return count
|
||||
|
||||
|
||||
def _delete_parsed_upload_files() -> int:
|
||||
"""Ch06: 'compress or delete after successful parse + rollup, rather
|
||||
than keeping both the raw file and a full raw-row copy.' Deletes
|
||||
(rather than compresses) — simpler, and avoids spending extra CPU/IOPS
|
||||
gzip-ing data that's already been fully parsed into the database.
|
||||
"""
|
||||
done_files = LogFile.query.filter_by(status="done").all()
|
||||
removed = 0
|
||||
for log_file in done_files:
|
||||
path = upload_path_for(log_file)
|
||||
if path.exists():
|
||||
path.unlink()
|
||||
removed += 1
|
||||
return removed
|
||||
|
||||
|
||||
@click.command("create-admin")
|
||||
@click.option("--email", default=None, help="Defaults to the ADMIN_EMAIL env var.")
|
||||
@click.password_option()
|
||||
def create_admin(email: str | None, password: str) -> None:
|
||||
"""Bootstrap the single admin account (Chapter 12).
|
||||
|
||||
Interactive password prompt (via --password-option's confirmation
|
||||
prompt) keeps the raw credential out of process env/config, unlike an
|
||||
ADMIN_PASSWORD env var would — Chapter 12 doesn't specify a bootstrap
|
||||
mechanism beyond documenting ADMIN_EMAIL, so this is a flagged addition.
|
||||
"""
|
||||
resolved_email = (email or current_app.config.get("ADMIN_EMAIL") or "").strip().lower()
|
||||
if not resolved_email:
|
||||
raise click.UsageError("No --email given and ADMIN_EMAIL is not set.")
|
||||
if User.query.filter_by(email=resolved_email).first():
|
||||
raise click.UsageError(f"User {resolved_email} already exists.")
|
||||
|
||||
user = User(email=resolved_email)
|
||||
user.set_password(password)
|
||||
db.session.add(user)
|
||||
db.session.commit()
|
||||
click.echo(f"Created admin user {resolved_email}")
|
||||
+109
@@ -0,0 +1,109 @@
|
||||
"""Environment-sourced configuration classes (12-factor factor 3).
|
||||
|
||||
Every value comes from os.environ. No secret, DB URL, or filesystem path
|
||||
is ever hardcoded; .env.example documents every variable a deployment
|
||||
must set.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from datetime import timedelta
|
||||
|
||||
|
||||
def _bool_env(name: str, default: bool = False) -> bool:
|
||||
"""Parse a boolean-ish environment variable."""
|
||||
val = os.environ.get(name)
|
||||
return default if val is None else val.strip().lower() in {"1", "true", "yes", "on"}
|
||||
|
||||
|
||||
def _engine_options_for(database_url: str) -> dict:
|
||||
"""pool_size/pool_recycle are QueuePool-only kwargs.
|
||||
|
||||
BUG FIX (caught by actually booting the app): SQLite's default pool
|
||||
classes (NullPool for file DBs, StaticPool for :memory:) raise
|
||||
TypeError if handed pool_size/pool_recycle at all — they're not
|
||||
silently ignored. Chapter 06 makes SQLite the default engine and
|
||||
MySQL the opt-in fallback, so these kwargs are only meaningful (and
|
||||
only passed) when DATABASE_URL actually points at a non-SQLite engine.
|
||||
"""
|
||||
if database_url.startswith("sqlite"):
|
||||
return {}
|
||||
return {
|
||||
"pool_size": int(os.environ.get("DB_POOL_SIZE", "3")),
|
||||
"pool_recycle": int(os.environ.get("DB_POOL_RECYCLE_SECONDS", "280")),
|
||||
"pool_pre_ping": True,
|
||||
}
|
||||
|
||||
|
||||
class BaseConfig:
|
||||
"""Shared config. Subclasses override only what differs per environment."""
|
||||
|
||||
SECRET_KEY: str | None = os.environ.get("SECRET_KEY")
|
||||
|
||||
# Chapter 06: SQLite/WAL default; swappable via DATABASE_URL without code changes.
|
||||
SQLALCHEMY_DATABASE_URI: str = os.environ.get("DATABASE_URL", "sqlite:///kavosh.db")
|
||||
SQLALCHEMY_ENGINE_OPTIONS: dict = _engine_options_for(SQLALCHEMY_DATABASE_URI)
|
||||
SQLALCHEMY_TRACK_MODIFICATIONS = False
|
||||
|
||||
# Chapter 03: 150 max DB connections shared across up to 60 entry
|
||||
# processes -> keep each process's pool small. Exposed as its own
|
||||
# config key (not just buried inside SQLALCHEMY_ENGINE_OPTIONS) so
|
||||
# budget.py can validate the *intended* setting regardless of whether
|
||||
# the active engine (SQLite) actually consumes it.
|
||||
DB_POOL_SIZE: int = int(os.environ.get("DB_POOL_SIZE", "3"))
|
||||
|
||||
# Chapter 03: filesystem cache, not Redis.
|
||||
CACHE_TYPE = os.environ.get("CACHE_TYPE", "FileSystemCache")
|
||||
CACHE_DIR = os.environ.get("CACHE_DIR", "/tmp/kavosh-cache")
|
||||
CACHE_DEFAULT_TIMEOUT = int(os.environ.get("CACHE_DEFAULT_TIMEOUT", "60"))
|
||||
|
||||
# Chapter 12: secure session cookie flags.
|
||||
SESSION_COOKIE_SECURE = _bool_env("SESSION_COOKIE_SECURE", True)
|
||||
SESSION_COOKIE_HTTPONLY = True
|
||||
SESSION_COOKIE_SAMESITE = "Lax"
|
||||
PERMANENT_SESSION_LIFETIME = timedelta(
|
||||
hours=int(os.environ.get("SESSION_LIFETIME_HOURS", "12"))
|
||||
)
|
||||
|
||||
WTF_CSRF_ENABLED = True
|
||||
|
||||
# Chapter 04/12: upload constraints.
|
||||
UPLOAD_MAX_SIZE_MB = int(os.environ.get("UPLOAD_MAX_SIZE_MB", "500"))
|
||||
MAX_CONTENT_LENGTH = UPLOAD_MAX_SIZE_MB * 1024 * 1024
|
||||
UPLOAD_DIR = os.environ.get("UPLOAD_DIR", "/home/kavosh/uploads")
|
||||
|
||||
PARSE_BATCH_SIZE = int(os.environ.get("PARSE_BATCH_SIZE", "5000"))
|
||||
ADMIN_EMAIL = os.environ.get("ADMIN_EMAIL")
|
||||
|
||||
|
||||
class DevConfig(BaseConfig):
|
||||
DEBUG = True
|
||||
SESSION_COOKIE_SECURE = False # allow plain-http local dev
|
||||
|
||||
|
||||
class ProdConfig(BaseConfig):
|
||||
DEBUG = False
|
||||
|
||||
|
||||
class TestConfig(BaseConfig):
|
||||
TESTING = True
|
||||
DEBUG = True # lets asset() tolerate a missing Vite manifest during tests
|
||||
SQLALCHEMY_DATABASE_URI = "sqlite:///:memory:"
|
||||
SQLALCHEMY_ENGINE_OPTIONS = {} # always in-memory SQLite regardless of DATABASE_URL
|
||||
WTF_CSRF_ENABLED = False
|
||||
|
||||
|
||||
_CONFIGS = {"development": DevConfig, "production": ProdConfig, "testing": TestConfig}
|
||||
|
||||
|
||||
def get_config(config_name: str | None = None):
|
||||
"""Resolve a config class from FLASK_ENV or an explicit name.
|
||||
|
||||
Defaults to `production` if unset, so an unconfigured deployment never
|
||||
silently runs with DEBUG on.
|
||||
"""
|
||||
name = config_name or os.environ.get("FLASK_ENV", "production")
|
||||
try:
|
||||
return _CONFIGS[name]
|
||||
except KeyError as exc:
|
||||
raise ValueError(f"Unknown config_name {name!r}; expected one of {list(_CONFIGS)}") from exc
|
||||
@@ -0,0 +1,28 @@
|
||||
"""Singleton Flask extension instances — initialized, not configured, here.
|
||||
|
||||
Configuration happens in create_app() via .init_app(), so nothing here
|
||||
holds app- or request-scoped state that must survive a process restart
|
||||
(factor 6: stateless processes).
|
||||
"""
|
||||
from flask_caching import Cache
|
||||
from flask_limiter import Limiter
|
||||
from flask_limiter.util import get_remote_address
|
||||
from flask_login import LoginManager
|
||||
from flask_migrate import Migrate
|
||||
from flask_sqlalchemy import SQLAlchemy
|
||||
from flask_wtf import CSRFProtect
|
||||
|
||||
db = SQLAlchemy()
|
||||
cache = Cache()
|
||||
csrf = CSRFProtect()
|
||||
login_manager = LoginManager()
|
||||
login_manager.login_view = "auth.login"
|
||||
migrate = Migrate()
|
||||
|
||||
# In-memory storage (Chapter 12: "in-memory or DB-backed... do not require
|
||||
# Redis"). CAVEAT (flagged): each of up to 60 entry processes (Ch03) keeps
|
||||
# its own counters, so the effective ceiling across the whole app is up to
|
||||
# (per-process limit x concurrent processes hit), not one hard global cap.
|
||||
# Acceptable for this app's threat model (slowing down /login and /uploads
|
||||
# brute-forcing), but not a strict global rate guarantee.
|
||||
limiter = Limiter(key_func=get_remote_address, storage_uri="memory://")
|
||||
@@ -0,0 +1,34 @@
|
||||
"""Structured stdout/stderr logging (Chapter 02 factor 11 / Chapter 12).
|
||||
|
||||
Writes structured (key=value) lines to stdout so the host's log capture
|
||||
picks them up, per 12-factor logs. Falls back to a size-capped rotating
|
||||
file only if LOG_FALLBACK_FILE is explicitly set (Ch12: "if stdout capture
|
||||
is unavailable on the specific hosting setup").
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
from logging.handlers import RotatingFileHandler
|
||||
|
||||
from flask import Flask
|
||||
|
||||
_FORMAT = "%(asctime)s level=%(levelname)s logger=%(name)s msg=%(message)s"
|
||||
|
||||
|
||||
def configure_logging(app: Flask) -> None:
|
||||
formatter = logging.Formatter(_FORMAT)
|
||||
|
||||
stdout_handler = logging.StreamHandler(sys.stdout)
|
||||
stdout_handler.setFormatter(formatter)
|
||||
|
||||
app.logger.handlers = [stdout_handler]
|
||||
app.logger.setLevel(logging.INFO if not app.debug else logging.DEBUG)
|
||||
app.logger.propagate = False
|
||||
|
||||
fallback_path = os.environ.get("LOG_FALLBACK_FILE")
|
||||
if fallback_path:
|
||||
file_handler = RotatingFileHandler(fallback_path, maxBytes=5 * 1024 * 1024, backupCount=3)
|
||||
file_handler.setFormatter(formatter)
|
||||
app.logger.addHandler(file_handler)
|
||||
@@ -0,0 +1,19 @@
|
||||
from app.models.blocklist_suggestion import BlocklistSuggestion
|
||||
from app.models.bot_hit import BotHit
|
||||
from app.models.browser_stats import BrowserStatsDaily
|
||||
from app.models.human_path_stats import HumanPathStatsDaily
|
||||
from app.models.ip_registry import IPRegistry
|
||||
from app.models.ip_traffic_stats import IpPathStatsDaily, IpStatusStatsDaily
|
||||
from app.models.log_entry import LogEntry
|
||||
from app.models.log_file import LogFile
|
||||
from app.models.referrer_stats import ReferrerStatsDaily
|
||||
from app.models.request_stats import RequestStatsDaily, RequestStatsHourly
|
||||
from app.models.suspicious_event import SuspiciousEvent
|
||||
from app.models.user import User
|
||||
|
||||
__all__ = [
|
||||
"LogFile", "LogEntry", "RequestStatsHourly", "RequestStatsDaily",
|
||||
"BotHit", "IPRegistry", "SuspiciousEvent", "BlocklistSuggestion",
|
||||
"ReferrerStatsDaily", "BrowserStatsDaily", "HumanPathStatsDaily",
|
||||
"IpPathStatsDaily", "IpStatusStatsDaily", "User",
|
||||
]
|
||||
@@ -0,0 +1,18 @@
|
||||
"""blocklist_suggestions table (Chapter 06). `id` isn't in Ch06's column
|
||||
list — added because `ip` alone can't be the key (the same IP may be
|
||||
re-flagged with a different reason later)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import datetime
|
||||
|
||||
from app.extensions import db
|
||||
|
||||
|
||||
class BlocklistSuggestion(db.Model):
|
||||
__tablename__ = "blocklist_suggestions"
|
||||
|
||||
id: int = db.Column(db.Integer, primary_key=True)
|
||||
ip: str = db.Column(db.String(45), nullable=False, index=True)
|
||||
reason: str = db.Column(db.String(255), nullable=False)
|
||||
created_at = db.Column(db.DateTime, nullable=False, default=datetime.utcnow)
|
||||
exported: bool = db.Column(db.Boolean, nullable=False, default=False)
|
||||
@@ -0,0 +1,22 @@
|
||||
"""bot_hits table (Chapter 06) — SEO section (Ch09)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from app.extensions import db
|
||||
|
||||
|
||||
class BotHit(db.Model):
|
||||
__tablename__ = "bot_hits"
|
||||
|
||||
id: int = db.Column(db.Integer, primary_key=True)
|
||||
log_file_id: int = db.Column(db.Integer, db.ForeignKey("log_files.id", ondelete="CASCADE"), nullable=False)
|
||||
timestamp = db.Column(db.DateTime, nullable=False)
|
||||
ip: str = db.Column(db.String(45), nullable=False)
|
||||
bot_name: str = db.Column(db.String(64), nullable=False)
|
||||
verified: bool = db.Column(db.Boolean, nullable=False, default=False)
|
||||
path: str = db.Column(db.Text, nullable=False)
|
||||
status_code: int = db.Column(db.SmallInteger, nullable=False)
|
||||
|
||||
__table_args__ = (
|
||||
db.Index("ix_bot_hits_timestamp", "timestamp"),
|
||||
db.Index("ix_bot_hits_bot_name", "bot_name"),
|
||||
)
|
||||
@@ -0,0 +1,14 @@
|
||||
"""NEW table (Method A, Ch08 follow-up) — not in Chapter 06's original
|
||||
list. Human-only browser/OS breakdown widget (Ch08)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from app.extensions import db
|
||||
|
||||
|
||||
class BrowserStatsDaily(db.Model):
|
||||
__tablename__ = "browser_stats_daily"
|
||||
|
||||
date = db.Column(db.Date, primary_key=True)
|
||||
browser: str = db.Column(db.String(64), primary_key=True)
|
||||
os: str = db.Column(db.String(64), primary_key=True)
|
||||
count: int = db.Column(db.Integer, nullable=False, default=0)
|
||||
@@ -0,0 +1,15 @@
|
||||
"""NEW table (Method A, Ch09 follow-up) — not in Chapter 06's original
|
||||
list. request_stats_hourly/_daily aggregate ALL traffic with no is_bot
|
||||
split, so Ch09's "most-visited-by-humans" comparison had no rollup to
|
||||
read. Bot hits excluded at rollup-write time (Ch07's is_bot flag)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from app.extensions import db
|
||||
|
||||
|
||||
class HumanPathStatsDaily(db.Model):
|
||||
__tablename__ = "human_path_stats_daily"
|
||||
|
||||
date = db.Column(db.Date, primary_key=True)
|
||||
path: str = db.Column(db.String(2048), primary_key=True)
|
||||
count: int = db.Column(db.Integer, nullable=False, default=0)
|
||||
@@ -0,0 +1,22 @@
|
||||
"""ip_registry table (Chapter 06 + Chapter 07 extension).
|
||||
|
||||
last_verified_bot_result: NOT in Ch06's literal column list — added in
|
||||
Chapter 07. last_verified_at alone can't tell a cache hit *what* was
|
||||
verified, only *when*; this stores the outcome.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from app.extensions import db
|
||||
|
||||
|
||||
class IPRegistry(db.Model):
|
||||
__tablename__ = "ip_registry"
|
||||
|
||||
ip: str = db.Column(db.String(45), primary_key=True)
|
||||
first_seen = db.Column(db.DateTime, nullable=False)
|
||||
last_seen = db.Column(db.DateTime, nullable=False)
|
||||
total_requests: int = db.Column(db.Integer, nullable=False, default=0)
|
||||
reputation_score: int = db.Column(db.Integer, nullable=False, default=0)
|
||||
is_flagged: bool = db.Column(db.Boolean, nullable=False, default=False)
|
||||
last_verified_at = db.Column(db.DateTime, nullable=True)
|
||||
last_verified_bot_result: bool | None = db.Column(db.Boolean, nullable=True)
|
||||
@@ -0,0 +1,35 @@
|
||||
"""NEW tables (explicit follow-up to Ch10's flagged scope gap): bounded
|
||||
per-IP traffic breakdown so the IP investigation panel reflects TRUE
|
||||
full traffic, not just bot_hits/suspicious_events activity.
|
||||
|
||||
CARDINALITY NOTE: unlike the other Method-A tables, this one's row count
|
||||
scales with distinct IPs per day, which is unbounded for a probed site.
|
||||
Two mitigations: ip_path_stats_daily keeps only the top N paths per IP
|
||||
per day (not every ip x path pair); both tables use log_entries' ~30-day
|
||||
retention (Ch06), enforced by `flask cleanup` (Chapter 12).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from app.extensions import db
|
||||
|
||||
|
||||
class IpPathStatsDaily(db.Model):
|
||||
__tablename__ = "ip_path_stats_daily"
|
||||
|
||||
date = db.Column(db.Date, primary_key=True)
|
||||
ip: str = db.Column(db.String(45), primary_key=True)
|
||||
path: str = db.Column(db.String(2048), primary_key=True)
|
||||
count: int = db.Column(db.Integer, nullable=False, default=0)
|
||||
|
||||
__table_args__ = (db.Index("ix_ip_path_stats_ip", "ip"),)
|
||||
|
||||
|
||||
class IpStatusStatsDaily(db.Model):
|
||||
__tablename__ = "ip_status_stats_daily"
|
||||
|
||||
date = db.Column(db.Date, primary_key=True)
|
||||
ip: str = db.Column(db.String(45), primary_key=True)
|
||||
status_bucket: str = db.Column(db.String(8), primary_key=True) # 2xx|3xx|4xx|5xx
|
||||
count: int = db.Column(db.Integer, nullable=False, default=0)
|
||||
|
||||
__table_args__ = (db.Index("ix_ip_status_stats_ip", "ip"),)
|
||||
@@ -0,0 +1,29 @@
|
||||
"""Optional, time-boxed raw storage (Chapter 06) — retention (~30 days)
|
||||
enforced by `flask cleanup` (Chapter 12)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from app.extensions import db
|
||||
|
||||
|
||||
class LogEntry(db.Model):
|
||||
__tablename__ = "log_entries"
|
||||
|
||||
id: int = db.Column(db.Integer, primary_key=True)
|
||||
log_file_id: int = db.Column(db.Integer, db.ForeignKey("log_files.id", ondelete="CASCADE"), nullable=False)
|
||||
timestamp = db.Column(db.DateTime, nullable=False) # normalized to UTC (Ch07)
|
||||
ip: str = db.Column(db.String(45), nullable=False) # IPv4 or IPv6
|
||||
method: str = db.Column(db.String(10), nullable=False)
|
||||
path: str = db.Column(db.Text, nullable=False)
|
||||
status_code: int = db.Column(db.SmallInteger, nullable=False)
|
||||
bytes_sent: int = db.Column(db.Integer, nullable=False, default=0)
|
||||
referrer: str | None = db.Column(db.Text, nullable=True)
|
||||
user_agent: str | None = db.Column(db.Text, nullable=True)
|
||||
is_bot: bool = db.Column(db.Boolean, nullable=False, default=False)
|
||||
flagged: bool = db.Column(db.Boolean, nullable=False, default=False)
|
||||
|
||||
__table_args__ = (
|
||||
db.Index("ix_log_entries_file_ts", "log_file_id", "timestamp"),
|
||||
db.Index("ix_log_entries_ip", "ip"),
|
||||
db.Index("ix_log_entries_path", "path"),
|
||||
db.Index("ix_log_entries_timestamp", "timestamp"),
|
||||
)
|
||||
@@ -0,0 +1,22 @@
|
||||
"""LogFile model (Chapter 06)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import datetime
|
||||
|
||||
from app.extensions import db
|
||||
|
||||
|
||||
class LogFile(db.Model):
|
||||
__tablename__ = "log_files"
|
||||
|
||||
id: int = db.Column(db.Integer, primary_key=True)
|
||||
filename: str = db.Column(db.String(255), nullable=False)
|
||||
server_type: str = db.Column(db.String(16), nullable=False) # apache|litespeed
|
||||
format_string: str = db.Column(db.Text, nullable=False)
|
||||
uploaded_at: datetime = db.Column(db.DateTime, nullable=False, default=datetime.utcnow)
|
||||
status: str = db.Column(db.String(16), nullable=False, default="queued")
|
||||
total_lines: int | None = db.Column(db.Integer, nullable=True)
|
||||
processed_lines: int = db.Column(db.Integer, nullable=False, default=0)
|
||||
size_bytes: int = db.Column(db.BigInteger, nullable=False)
|
||||
checksum: str = db.Column(db.String(64), nullable=False, index=True) # sha256 hex
|
||||
error_message: str | None = db.Column(db.Text, nullable=True)
|
||||
@@ -0,0 +1,14 @@
|
||||
"""NEW table (Method A, Ch08 follow-up) — not in Chapter 06's original
|
||||
list. Gives Top Referrers a rollup data source instead of scanning
|
||||
log_entries live. Domain-bucketed to keep cardinality bounded."""
|
||||
from __future__ import annotations
|
||||
|
||||
from app.extensions import db
|
||||
|
||||
|
||||
class ReferrerStatsDaily(db.Model):
|
||||
__tablename__ = "referrer_stats_daily"
|
||||
|
||||
date = db.Column(db.Date, primary_key=True)
|
||||
referrer_domain: str = db.Column(db.String(255), primary_key=True)
|
||||
count: int = db.Column(db.Integer, nullable=False, default=0)
|
||||
@@ -0,0 +1,26 @@
|
||||
"""Rollup tables (Chapter 06) — primary source for Overview widgets (Ch08).
|
||||
Both are site-wide (no log_file_id), consistent with Ch01's single-site scope.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from app.extensions import db
|
||||
|
||||
|
||||
class RequestStatsHourly(db.Model):
|
||||
__tablename__ = "request_stats_hourly"
|
||||
|
||||
date_hour = db.Column(db.DateTime, primary_key=True)
|
||||
path: str = db.Column(db.String(2048), primary_key=True)
|
||||
status_code: int = db.Column(db.SmallInteger, primary_key=True)
|
||||
count: int = db.Column(db.Integer, nullable=False, default=0)
|
||||
bytes_sent_sum: int = db.Column(db.BigInteger, nullable=False, default=0)
|
||||
|
||||
|
||||
class RequestStatsDaily(db.Model):
|
||||
__tablename__ = "request_stats_daily"
|
||||
|
||||
date = db.Column(db.Date, primary_key=True)
|
||||
count: int = db.Column(db.Integer, nullable=False, default=0)
|
||||
unique_ips: int = db.Column(db.Integer, nullable=False, default=0)
|
||||
bytes_sum: int = db.Column(db.BigInteger, nullable=False, default=0)
|
||||
error_count: int = db.Column(db.Integer, nullable=False, default=0)
|
||||
@@ -0,0 +1,22 @@
|
||||
"""suspicious_events table (Chapter 06) — Security section (Ch10)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from app.extensions import db
|
||||
|
||||
|
||||
class SuspiciousEvent(db.Model):
|
||||
__tablename__ = "suspicious_events"
|
||||
|
||||
id: int = db.Column(db.Integer, primary_key=True)
|
||||
log_file_id: int = db.Column(db.Integer, db.ForeignKey("log_files.id", ondelete="CASCADE"), nullable=False)
|
||||
ip: str = db.Column(db.String(45), nullable=False)
|
||||
timestamp = db.Column(db.DateTime, nullable=False)
|
||||
path: str = db.Column(db.Text, nullable=False)
|
||||
rule_matched: str = db.Column(db.String(128), nullable=False)
|
||||
severity: str = db.Column(db.String(8), nullable=False) # low|medium|high (provisional; Ch10 escalates)
|
||||
|
||||
__table_args__ = (
|
||||
db.Index("ix_suspicious_events_timestamp", "timestamp"),
|
||||
db.Index("ix_suspicious_events_ip", "ip"),
|
||||
db.Index("ix_suspicious_events_severity", "severity"),
|
||||
)
|
||||
@@ -0,0 +1,27 @@
|
||||
"""Single-admin user model (Chapter 12). Not in Chapter 06's table list —
|
||||
that chapter scopes log-analytics tables; auth is a separate concern this
|
||||
chapter owns. Chapter 01 confirms single-admin, no self-registration.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import datetime
|
||||
|
||||
import bcrypt
|
||||
from flask_login import UserMixin
|
||||
|
||||
from app.extensions import db
|
||||
|
||||
|
||||
class User(db.Model, UserMixin):
|
||||
__tablename__ = "users"
|
||||
|
||||
id: int = db.Column(db.Integer, primary_key=True)
|
||||
email: str = db.Column(db.String(255), unique=True, nullable=False, index=True)
|
||||
password_hash: str = db.Column(db.String(255), nullable=False)
|
||||
created_at = db.Column(db.DateTime, nullable=False, default=datetime.utcnow)
|
||||
|
||||
def set_password(self, raw_password: str) -> None:
|
||||
self.password_hash = bcrypt.hashpw(raw_password.encode("utf-8"), bcrypt.gensalt()).decode("utf-8")
|
||||
|
||||
def check_password(self, raw_password: str) -> bool:
|
||||
return bcrypt.checkpw(raw_password.encode("utf-8"), self.password_hash.encode("utf-8"))
|
||||
@@ -0,0 +1,199 @@
|
||||
"""Rollup computation (Chapter 06 + Method A extensions from Ch08/09/10).
|
||||
|
||||
Called once per parsed file (app/cli.py::process_logs) for the dates it
|
||||
touched, ad hoc via `flask rollup` for a manual recompute, and now also
|
||||
after a file deletion (app/services/file_deletion.py) for whatever dates
|
||||
the deleted file touched. Every _upsert_* function scans log_entries
|
||||
inside this background batch job, never at request time — that's what
|
||||
makes Ch03 rule 6 compliance possible.
|
||||
|
||||
CORRECTNESS FIX: every rollup writer below now deletes a day's existing
|
||||
rows before writing whatever the fresh scan finds (including writing
|
||||
nothing, if a day now has zero data). Three of the five writers
|
||||
previously only ever upserted-when-present and silently left stale rows
|
||||
behind when a day's data disappeared — unreachable before file deletion
|
||||
existed (rollups only ever grew), but a real correctness bug once
|
||||
deletion makes "this day now has less data than before" possible. Only
|
||||
the two per-IP writers already had this right (Ch10 follow-up); the
|
||||
other three are fixed here to match.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from collections import defaultdict
|
||||
from datetime import date, datetime, time, timedelta
|
||||
|
||||
from sqlalchemy import case, func
|
||||
|
||||
from app.extensions import db
|
||||
from app.models.browser_stats import BrowserStatsDaily
|
||||
from app.models.human_path_stats import HumanPathStatsDaily
|
||||
from app.models.ip_traffic_stats import IpPathStatsDaily, IpStatusStatsDaily
|
||||
from app.models.log_entry import LogEntry
|
||||
from app.models.referrer_stats import ReferrerStatsDaily
|
||||
from app.models.request_stats import RequestStatsDaily, RequestStatsHourly
|
||||
from app.services.blocklist import refresh_blocklist_suggestions
|
||||
from app.services.referrer import referrer_domain
|
||||
from app.services.ua_classifier import classify_browser, classify_os
|
||||
from app.utils.http_status import status_bucket
|
||||
|
||||
TOP_PATHS_PER_IP_PER_DAY = 15 # bounds ip_path_stats_daily row growth (Ch10 follow-up)
|
||||
|
||||
|
||||
def compute_rollups_for_range(start: date, end: date) -> None:
|
||||
"""Recompute every rollup for each day in [start, end]. Site-wide,
|
||||
not per-file (Ch01: single site) — recomputing from scratch per day
|
||||
avoids double-counting when two uploads cover the same period, and
|
||||
correctly shrinks a day's numbers back down when a file covering
|
||||
that day is deleted.
|
||||
"""
|
||||
current = start
|
||||
while current <= end:
|
||||
_upsert_hourly(current)
|
||||
_upsert_daily(current)
|
||||
_upsert_per_line_derived_stats(current)
|
||||
refresh_blocklist_suggestions(current)
|
||||
current += timedelta(days=1)
|
||||
|
||||
|
||||
def _day_bounds(day: date) -> tuple[datetime, datetime]:
|
||||
start = datetime.combine(day, time.min)
|
||||
return start, start + timedelta(days=1)
|
||||
|
||||
|
||||
def _upsert_hourly(day: date) -> None:
|
||||
start, end = _day_bounds(day)
|
||||
rows = (
|
||||
db.session.query(
|
||||
func.strftime("%Y-%m-%d %H:00:00", LogEntry.timestamp).label("date_hour"),
|
||||
LogEntry.path,
|
||||
LogEntry.status_code,
|
||||
func.count().label("count"),
|
||||
func.coalesce(func.sum(LogEntry.bytes_sent), 0).label("bytes_sent_sum"),
|
||||
)
|
||||
.filter(LogEntry.timestamp >= start, LogEntry.timestamp < end)
|
||||
.group_by("date_hour", LogEntry.path, LogEntry.status_code)
|
||||
.all()
|
||||
)
|
||||
|
||||
# Delete-then-insert: replaces the day's hourly rows entirely,
|
||||
# including leaving none behind if `rows` is now empty (e.g. the
|
||||
# only file covering this day was just deleted).
|
||||
db.session.query(RequestStatsHourly).filter(
|
||||
RequestStatsHourly.date_hour >= start, RequestStatsHourly.date_hour < end
|
||||
).delete()
|
||||
|
||||
if rows:
|
||||
payload = [
|
||||
{
|
||||
"date_hour": datetime.strptime(r.date_hour, "%Y-%m-%d %H:%M:%S"),
|
||||
"path": r.path,
|
||||
"status_code": r.status_code,
|
||||
"count": r.count,
|
||||
"bytes_sent_sum": r.bytes_sent_sum,
|
||||
}
|
||||
for r in rows
|
||||
]
|
||||
db.session.execute(RequestStatsHourly.__table__.insert(), payload)
|
||||
|
||||
db.session.commit()
|
||||
|
||||
|
||||
def _upsert_daily(day: date) -> None:
|
||||
start, end = _day_bounds(day)
|
||||
result = (
|
||||
db.session.query(
|
||||
func.count().label("count"),
|
||||
func.count(func.distinct(LogEntry.ip)).label("unique_ips"),
|
||||
func.coalesce(func.sum(LogEntry.bytes_sent), 0).label("bytes_sum"),
|
||||
func.coalesce(func.sum(case((LogEntry.status_code >= 400, 1), else_=0)), 0).label("error_count"),
|
||||
)
|
||||
.filter(LogEntry.timestamp >= start, LogEntry.timestamp < end)
|
||||
.one()
|
||||
)
|
||||
|
||||
# Delete-then-insert: if this day now has zero entries (its only
|
||||
# contributing file was deleted), the stale row is removed rather
|
||||
# than left behind — no rollup row is better than a wrong one.
|
||||
db.session.query(RequestStatsDaily).filter(RequestStatsDaily.date == day).delete()
|
||||
|
||||
if result.count > 0:
|
||||
db.session.execute(
|
||||
RequestStatsDaily.__table__.insert(),
|
||||
{
|
||||
"date": day,
|
||||
"count": result.count,
|
||||
"unique_ips": result.unique_ips,
|
||||
"bytes_sum": result.bytes_sum,
|
||||
"error_count": result.error_count,
|
||||
},
|
||||
)
|
||||
|
||||
db.session.commit()
|
||||
|
||||
|
||||
def _upsert_per_line_derived_stats(day: date) -> None:
|
||||
"""Referrer domain, browser/OS, human-only path counts (Ch08/09), and
|
||||
per-IP path/status counts (Ch10 follow-up) — one streamed pass over
|
||||
log_entries (Ch03 rule 1: bounded per-chunk memory via yield_per,
|
||||
never the whole day loaded at once). Every table here uses the same
|
||||
delete-then-insert pattern so a day's rows are fully replaced by
|
||||
whatever the fresh scan finds, including nothing.
|
||||
"""
|
||||
start, end = _day_bounds(day)
|
||||
referrer_counts: dict[str, int] = defaultdict(int)
|
||||
browser_counts: dict[tuple[str, str], int] = defaultdict(int)
|
||||
human_path_counts: dict[str, int] = defaultdict(int)
|
||||
ip_path_counts: dict[str, dict[str, int]] = defaultdict(lambda: defaultdict(int))
|
||||
ip_status_counts: dict[str, dict[str, int]] = defaultdict(lambda: defaultdict(int))
|
||||
|
||||
query = (
|
||||
db.session.query(
|
||||
LogEntry.referrer, LogEntry.user_agent, LogEntry.is_bot,
|
||||
LogEntry.path, LogEntry.ip, LogEntry.status_code,
|
||||
)
|
||||
.filter(LogEntry.timestamp >= start, LogEntry.timestamp < end)
|
||||
)
|
||||
for referrer, user_agent, is_bot, path, ip, status_code in query.yield_per(1000):
|
||||
domain = referrer_domain(referrer)
|
||||
if domain:
|
||||
referrer_counts[domain] += 1
|
||||
if not is_bot: # Ch08: bot traffic excluded from human browser/OS breakdown
|
||||
browser_counts[(classify_browser(user_agent), classify_os(user_agent))] += 1
|
||||
human_path_counts[path] += 1
|
||||
ip_path_counts[ip][path] += 1
|
||||
ip_status_counts[ip][status_bucket(status_code)] += 1
|
||||
|
||||
db.session.query(ReferrerStatsDaily).filter(ReferrerStatsDaily.date == day).delete()
|
||||
if referrer_counts:
|
||||
payload = [{"date": day, "referrer_domain": d, "count": c} for d, c in referrer_counts.items()]
|
||||
db.session.execute(ReferrerStatsDaily.__table__.insert(), payload)
|
||||
|
||||
db.session.query(BrowserStatsDaily).filter(BrowserStatsDaily.date == day).delete()
|
||||
if browser_counts:
|
||||
payload = [{"date": day, "browser": b, "os": o, "count": c} for (b, o), c in browser_counts.items()]
|
||||
db.session.execute(BrowserStatsDaily.__table__.insert(), payload)
|
||||
|
||||
db.session.query(HumanPathStatsDaily).filter(HumanPathStatsDaily.date == day).delete()
|
||||
if human_path_counts:
|
||||
payload = [{"date": day, "path": p, "count": c} for p, c in human_path_counts.items()]
|
||||
db.session.execute(HumanPathStatsDaily.__table__.insert(), payload)
|
||||
|
||||
db.session.query(IpPathStatsDaily).filter(IpPathStatsDaily.date == day).delete()
|
||||
if ip_path_counts:
|
||||
payload = []
|
||||
for ip, paths in ip_path_counts.items():
|
||||
top_paths = sorted(paths.items(), key=lambda kv: kv[1], reverse=True)[:TOP_PATHS_PER_IP_PER_DAY]
|
||||
payload.extend({"date": day, "ip": ip, "path": p, "count": c} for p, c in top_paths)
|
||||
if payload:
|
||||
db.session.execute(IpPathStatsDaily.__table__.insert(), payload)
|
||||
|
||||
db.session.query(IpStatusStatsDaily).filter(IpStatusStatsDaily.date == day).delete()
|
||||
if ip_status_counts:
|
||||
payload = [
|
||||
{"date": day, "ip": ip, "status_bucket": bucket, "count": c}
|
||||
for ip, buckets in ip_status_counts.items()
|
||||
for bucket, c in buckets.items()
|
||||
]
|
||||
db.session.execute(IpStatusStatsDaily.__table__.insert(), payload)
|
||||
|
||||
db.session.commit()
|
||||
@@ -0,0 +1,97 @@
|
||||
"""No-cron automatic processing (Chapter 12 simplification, per project
|
||||
owner request): triggers file parsing immediately in a background thread
|
||||
right after upload, and opportunistically resumes any incomplete files
|
||||
when the Overview page loads — replacing the cron-triggered model.
|
||||
`flask process-logs` still exists in app/cli.py for anyone who'd rather
|
||||
use cron, but nothing requires it anymore.
|
||||
|
||||
TRADEOFF (flagged, deviating from Chapter 02/03's "no persistent
|
||||
background workers, cron-triggered CLI only" stance): a background thread
|
||||
lives inside the same worker process that handled the upload request. If
|
||||
Passenger recycles that process mid-parse, the thread dies with it —
|
||||
progress up to the last commit is still safely checkpointed (same bounded-
|
||||
batch model as before), but nothing will automatically resume it without
|
||||
either cron or a page visit. The "resume on page load" hook below is the
|
||||
deliberate replacement for that guarantee: visiting the Overview tab
|
||||
re-triggers processing for anything left incomplete, so in the worst case
|
||||
a stuck file resumes the next time the admin looks at the dashboard,
|
||||
rather than never.
|
||||
|
||||
This is not a long-lived daemon: each thread terminates once its file
|
||||
reaches "done"/"error" (or the process is killed), and no thread survives
|
||||
a process restart — it just gets re-triggered fresh next time.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import threading
|
||||
|
||||
from flask import Flask
|
||||
|
||||
from app.extensions import db
|
||||
from app.models.log_file import LogFile
|
||||
from app.services.log_processor import process_one_batch
|
||||
|
||||
# In-process guard against launching two threads for the same file at
|
||||
# once (e.g. the upload trigger and a page-load resume firing close
|
||||
# together). Per-worker-process only — a different entry process picking
|
||||
# up the same file concurrently is a low-probability edge case accepted
|
||||
# for this simplification; each write is still a small checkpointed
|
||||
# commit, not a giant one, which limits how bad a collision could be.
|
||||
_active_file_ids: set[int] = set()
|
||||
_lock = threading.Lock()
|
||||
|
||||
|
||||
def _claim(log_file_id: int) -> bool:
|
||||
with _lock:
|
||||
if log_file_id in _active_file_ids:
|
||||
return False
|
||||
_active_file_ids.add(log_file_id)
|
||||
return True
|
||||
|
||||
|
||||
def _release(log_file_id: int) -> None:
|
||||
with _lock:
|
||||
_active_file_ids.discard(log_file_id)
|
||||
|
||||
|
||||
def _run_to_completion(app: Flask, log_file_id: int, batch_size: int) -> None:
|
||||
with app.app_context():
|
||||
try:
|
||||
log_file = db.session.get(LogFile, log_file_id)
|
||||
if log_file is None:
|
||||
return
|
||||
while log_file.status in ("queued", "processing"):
|
||||
process_one_batch(log_file, batch_size)
|
||||
db.session.refresh(log_file)
|
||||
except Exception:
|
||||
app.logger.exception("Background processing failed for log_file_id=%s", log_file_id)
|
||||
log_file = db.session.get(LogFile, log_file_id)
|
||||
if log_file is not None and log_file.status != "done":
|
||||
log_file.status = "error"
|
||||
log_file.error_message = "Processing failed unexpectedly; see server logs."
|
||||
db.session.commit()
|
||||
finally:
|
||||
_release(log_file_id)
|
||||
|
||||
|
||||
def trigger_processing(app: Flask, log_file_id: int) -> None:
|
||||
"""Start background processing for one file; no-ops if already running."""
|
||||
if not _claim(log_file_id):
|
||||
return
|
||||
batch_size = app.config["PARSE_BATCH_SIZE"]
|
||||
thread = threading.Thread(
|
||||
target=_run_to_completion, args=(app, log_file_id, batch_size), daemon=True
|
||||
)
|
||||
thread.start()
|
||||
|
||||
|
||||
def resume_incomplete_files(app: Flask) -> None:
|
||||
"""Opportunistic resume hook, called from the Overview page load —
|
||||
the deliberate replacement for cron's "there's always a next tick"
|
||||
guarantee. Cheap: one indexed status-filtered query.
|
||||
"""
|
||||
incomplete_ids = [
|
||||
lf.id for lf in LogFile.query.filter(LogFile.status.in_(["queued", "processing"])).all()
|
||||
]
|
||||
for log_file_id in incomplete_ids:
|
||||
trigger_processing(app, log_file_id)
|
||||
@@ -0,0 +1,67 @@
|
||||
"""Blocklist-suggestion generation (Chapter 10 follow-up): no earlier
|
||||
chapter assigned ownership of populating blocklist_suggestions or setting
|
||||
ip_registry.is_flagged. Runs in the background aggregator pass (batch, not
|
||||
request-time, per Ch02/03), reusing severity_scoring.py so there's exactly
|
||||
one scoring model between the dashboard display and the flagging decision.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from collections import defaultdict
|
||||
from datetime import date, datetime, timedelta
|
||||
|
||||
from app.extensions import db
|
||||
from app.models.blocklist_suggestion import BlocklistSuggestion
|
||||
from app.models.ip_registry import IPRegistry
|
||||
from app.models.suspicious_event import SuspiciousEvent
|
||||
from app.services.severity_scoring import SeverityInputs, compute_effective_severity
|
||||
|
||||
_RANK = {"low": 0, "medium": 1, "high": 2}
|
||||
|
||||
|
||||
def refresh_blocklist_suggestions(day: date) -> None:
|
||||
"""Flag an IP (is_flagged + a suggestion row) if its escalated severity
|
||||
for `day` reaches 'high'. Idempotent — skips IPs already suggested.
|
||||
"""
|
||||
start = datetime.combine(day, datetime.min.time())
|
||||
end = start + timedelta(days=1)
|
||||
|
||||
events = (
|
||||
db.session.query(SuspiciousEvent.ip, SuspiciousEvent.timestamp, SuspiciousEvent.severity)
|
||||
.filter(SuspiciousEvent.timestamp >= start, SuspiciousEvent.timestamp < end)
|
||||
.all()
|
||||
)
|
||||
if not events:
|
||||
return
|
||||
|
||||
by_ip: dict[str, list] = defaultdict(list)
|
||||
for ip, ts, sev in events:
|
||||
by_ip[ip].append((ts, sev))
|
||||
|
||||
already_suggested = {ip for (ip,) in db.session.query(BlocklistSuggestion.ip).distinct().all()}
|
||||
|
||||
for ip, ip_events in by_ip.items():
|
||||
if ip in already_suggested:
|
||||
continue
|
||||
timestamps = sorted(ts for ts, _ in ip_events)
|
||||
avg_interval = (
|
||||
(timestamps[-1] - timestamps[0]).total_seconds() / (len(timestamps) - 1)
|
||||
if len(timestamps) > 1 else None
|
||||
)
|
||||
worst_base = max((sev for _, sev in ip_events), key=lambda s: _RANK.get(s, 0))
|
||||
effective = compute_effective_severity(
|
||||
SeverityInputs(base_severity=worst_base, ip_event_count=len(ip_events), avg_interval_seconds=avg_interval)
|
||||
)
|
||||
if effective != "high":
|
||||
continue
|
||||
|
||||
db.session.add(BlocklistSuggestion(
|
||||
ip=ip,
|
||||
reason=f"{len(ip_events)} suspicious event(s) on {day.isoformat()}, escalated to high severity",
|
||||
created_at=datetime.utcnow(),
|
||||
exported=False,
|
||||
))
|
||||
ip_row = db.session.get(IPRegistry, ip)
|
||||
if ip_row is not None:
|
||||
ip_row.is_flagged = True
|
||||
|
||||
db.session.commit()
|
||||
@@ -0,0 +1,78 @@
|
||||
"""Bot signature matching + reverse-DNS verification (Chapter 07).
|
||||
|
||||
Signatures are data (JSON), not hardcoded logic, so the list grows without
|
||||
a code change. Verification does real DNS I/O — only ever called from the
|
||||
background process-logs batch job (app/services/classification.py), never
|
||||
synchronously inside a dashboard request, per Chapter 07's explicit rule.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import socket
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime, timedelta
|
||||
from pathlib import Path
|
||||
|
||||
SIGNATURES_PATH = Path(__file__).parent / "data" / "bot_signatures.json"
|
||||
|
||||
# Skip re-verifying the same IP more often than this (Ch07: DNS latency is
|
||||
# a real cost on constrained hosting).
|
||||
VERIFICATION_TTL = timedelta(days=7)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class BotSignature:
|
||||
name: str
|
||||
ua_substrings: tuple[str, ...]
|
||||
verify_suffixes: tuple[str, ...] # PTR hostname must end in one of these
|
||||
|
||||
|
||||
def _load_signatures() -> list[BotSignature]:
|
||||
raw = json.loads(SIGNATURES_PATH.read_text())
|
||||
return [
|
||||
BotSignature(name=e["name"], ua_substrings=tuple(e["ua_substrings"]), verify_suffixes=tuple(e["verify_suffixes"]))
|
||||
for e in raw
|
||||
]
|
||||
|
||||
|
||||
_SIGNATURES = _load_signatures()
|
||||
|
||||
|
||||
def classify_bot(user_agent: str | None) -> str | None:
|
||||
"""Return the claimed bot name via UA substring match, or None."""
|
||||
if not user_agent:
|
||||
return None
|
||||
ua_lower = user_agent.lower()
|
||||
for sig in _SIGNATURES:
|
||||
if any(sub.lower() in ua_lower for sub in sig.ua_substrings):
|
||||
return sig.name
|
||||
return None
|
||||
|
||||
|
||||
def _signature_for(bot_name: str) -> BotSignature | None:
|
||||
return next((s for s in _SIGNATURES if s.name == bot_name), None)
|
||||
|
||||
|
||||
def verify_bot_ip(ip: str, bot_name: str) -> bool:
|
||||
"""Reverse-DNS + forward-confirm that `ip` really belongs to `bot_name`."""
|
||||
sig = _signature_for(bot_name)
|
||||
if sig is None:
|
||||
return False
|
||||
try:
|
||||
hostname, _, _ = socket.gethostbyaddr(ip)
|
||||
except (socket.herror, socket.gaierror, OSError):
|
||||
return False
|
||||
if not any(hostname.lower().endswith(suffix) for suffix in sig.verify_suffixes):
|
||||
return False
|
||||
try:
|
||||
forward_ips = socket.gethostbyname_ex(hostname)[2]
|
||||
except (socket.herror, socket.gaierror, OSError):
|
||||
return False
|
||||
return ip in forward_ips
|
||||
|
||||
|
||||
def is_verification_stale(last_verified_at: datetime | None) -> bool:
|
||||
"""True if this IP needs a fresh DNS check (Ch07 caching rule)."""
|
||||
if last_verified_at is None:
|
||||
return True
|
||||
return datetime.utcnow() - last_verified_at > VERIFICATION_TTL
|
||||
@@ -0,0 +1,122 @@
|
||||
"""Per-batch classification pipeline (Chapter 07): ties bot_identifier and
|
||||
threat_scanner into the bot_hits/suspicious_events/ip_registry write path.
|
||||
Called once per bulk-insert chunk from app/cli.py — DNS-based verification
|
||||
belongs here (background batch job), never in a dashboard request.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime
|
||||
|
||||
from sqlalchemy import func, insert
|
||||
from sqlalchemy.dialects.sqlite import insert as sqlite_insert
|
||||
|
||||
from app.extensions import db
|
||||
from app.models.bot_hit import BotHit
|
||||
from app.models.ip_registry import IPRegistry
|
||||
from app.models.suspicious_event import SuspiciousEvent
|
||||
from app.services import bot_identifier, threat_scanner
|
||||
from app.services.log_parser import ParsedEntry
|
||||
|
||||
|
||||
@dataclass
|
||||
class EntryClassification:
|
||||
"""Cheap, no-I/O flags for the log_entries row itself."""
|
||||
is_bot: bool
|
||||
flagged: bool
|
||||
|
||||
|
||||
def classify_entry(entry: ParsedEntry) -> EntryClassification:
|
||||
"""No DNS I/O here — bot *verification* is batched separately below,
|
||||
since it's stateful (cached per IP) and only worth doing once per IP
|
||||
per batch, not once per line.
|
||||
"""
|
||||
return EntryClassification(
|
||||
is_bot=bot_identifier.classify_bot(entry.user_agent) is not None,
|
||||
flagged=threat_scanner.scan(entry) is not None,
|
||||
)
|
||||
|
||||
|
||||
def write_batch_side_effects(log_file_id: int, entries: list[ParsedEntry]) -> None:
|
||||
"""Derive and bulk-write bot_hits, suspicious_events, ip_registry upserts."""
|
||||
if not entries:
|
||||
return
|
||||
|
||||
bot_hit_rows: list[dict] = []
|
||||
suspicious_rows: list[dict] = []
|
||||
ip_agg: dict[str, dict] = {}
|
||||
verified_this_batch: dict[str, bool] = {} # avoid repeat DNS for the same IP in one batch
|
||||
|
||||
for entry in entries:
|
||||
agg = ip_agg.setdefault(entry.ip, {"first": entry.timestamp, "last": entry.timestamp, "count": 0})
|
||||
agg["count"] += 1
|
||||
agg["first"] = min(agg["first"], entry.timestamp)
|
||||
agg["last"] = max(agg["last"], entry.timestamp)
|
||||
|
||||
bot_name = bot_identifier.classify_bot(entry.user_agent)
|
||||
if bot_name is not None:
|
||||
verified = verified_this_batch.get(entry.ip)
|
||||
if verified is None:
|
||||
verified = _verify_with_cache(entry.ip, bot_name)
|
||||
verified_this_batch[entry.ip] = verified
|
||||
bot_hit_rows.append({
|
||||
"log_file_id": log_file_id, "timestamp": entry.timestamp, "ip": entry.ip,
|
||||
"bot_name": bot_name, "verified": verified, "path": entry.path,
|
||||
"status_code": entry.status_code,
|
||||
})
|
||||
if not verified:
|
||||
# Single detection, two dashboard consumers (Ch07): spoofed
|
||||
# bot also surfaces as a suspicious_events row.
|
||||
suspicious_rows.append({
|
||||
"log_file_id": log_file_id, "ip": entry.ip, "timestamp": entry.timestamp,
|
||||
"path": entry.path, "rule_matched": f"spoofed_bot:{bot_name}", "severity": "medium",
|
||||
})
|
||||
|
||||
threat = threat_scanner.scan(entry)
|
||||
if threat is not None:
|
||||
suspicious_rows.append({
|
||||
"log_file_id": log_file_id, "ip": entry.ip, "timestamp": entry.timestamp,
|
||||
"path": entry.path, "rule_matched": threat.rule_matched, "severity": threat.severity,
|
||||
})
|
||||
|
||||
if bot_hit_rows:
|
||||
db.session.execute(insert(BotHit.__table__), bot_hit_rows)
|
||||
if suspicious_rows:
|
||||
db.session.execute(insert(SuspiciousEvent.__table__), suspicious_rows)
|
||||
|
||||
_upsert_ip_registry(ip_agg, verified_this_batch)
|
||||
db.session.commit()
|
||||
|
||||
|
||||
def _verify_with_cache(ip: str, bot_name: str) -> bool:
|
||||
"""TTL-gated reverse/forward DNS check, cached via
|
||||
ip_registry.last_verified_bot_result (Ch07 schema addition).
|
||||
"""
|
||||
row = db.session.get(IPRegistry, ip)
|
||||
if row is not None and not bot_identifier.is_verification_stale(row.last_verified_at):
|
||||
return bool(row.last_verified_bot_result)
|
||||
return bot_identifier.verify_bot_ip(ip, bot_name)
|
||||
|
||||
|
||||
def _upsert_ip_registry(ip_agg: dict[str, dict], verified_this_batch: dict[str, bool]) -> None:
|
||||
now = datetime.utcnow()
|
||||
for ip, agg in ip_agg.items():
|
||||
values = {
|
||||
"ip": ip, "first_seen": agg["first"], "last_seen": agg["last"],
|
||||
"total_requests": agg["count"], "reputation_score": 0, "is_flagged": False,
|
||||
}
|
||||
if ip in verified_this_batch:
|
||||
values["last_verified_at"] = now
|
||||
values["last_verified_bot_result"] = verified_this_batch[ip]
|
||||
|
||||
stmt = sqlite_insert(IPRegistry.__table__).values(**values)
|
||||
update_set = {
|
||||
"last_seen": func.max(IPRegistry.last_seen, stmt.excluded.last_seen),
|
||||
"first_seen": func.min(IPRegistry.first_seen, stmt.excluded.first_seen),
|
||||
"total_requests": IPRegistry.total_requests + stmt.excluded.total_requests,
|
||||
}
|
||||
if ip in verified_this_batch:
|
||||
update_set["last_verified_at"] = stmt.excluded.last_verified_at
|
||||
update_set["last_verified_bot_result"] = stmt.excluded.last_verified_bot_result
|
||||
stmt = stmt.on_conflict_do_update(index_elements=["ip"], set_=update_set)
|
||||
db.session.execute(stmt)
|
||||
@@ -0,0 +1,9 @@
|
||||
[
|
||||
{"name": "Googlebot", "ua_substrings": ["Googlebot"], "verify_suffixes": [".googlebot.com", ".google.com"]},
|
||||
{"name": "Bingbot", "ua_substrings": ["bingbot"], "verify_suffixes": [".search.msn.com"]},
|
||||
{"name": "Yandex", "ua_substrings": ["YandexBot"], "verify_suffixes": [".yandex.ru", ".yandex.com", ".yandex.net"]},
|
||||
{"name": "Baidu", "ua_substrings": ["Baiduspider"], "verify_suffixes": [".baidu.com", ".baidu.jp"]},
|
||||
{"name": "DuckDuckBot", "ua_substrings": ["DuckDuckBot"], "verify_suffixes": [".duckduckgo.com"]},
|
||||
{"name": "AhrefsBot", "ua_substrings": ["AhrefsBot"], "verify_suffixes": [".ahrefs.com"]},
|
||||
{"name": "SemrushBot", "ua_substrings": ["SemrushBot"], "verify_suffixes": [".semrush.com"]}
|
||||
]
|
||||
@@ -0,0 +1,8 @@
|
||||
[
|
||||
{"name": "Edge", "ua_substrings": ["Edg/", "EdgA/", "EdgiOS/"]},
|
||||
{"name": "Opera", "ua_substrings": ["OPR/", "Opera"]},
|
||||
{"name": "Chrome", "ua_substrings": ["Chrome/", "CriOS/"]},
|
||||
{"name": "Firefox", "ua_substrings": ["Firefox/", "FxiOS/"]},
|
||||
{"name": "Safari", "ua_substrings": ["Safari/"]},
|
||||
{"name": "Internet Explorer", "ua_substrings": ["MSIE ", "Trident/"]}
|
||||
]
|
||||
@@ -0,0 +1,7 @@
|
||||
[
|
||||
{"name": "Windows", "ua_substrings": ["Windows NT"]},
|
||||
{"name": "iOS", "ua_substrings": ["iPhone", "iPad", "iPod"]},
|
||||
{"name": "macOS", "ua_substrings": ["Mac OS X", "Macintosh"]},
|
||||
{"name": "Android", "ua_substrings": ["Android"]},
|
||||
{"name": "Linux", "ua_substrings": ["Linux"]}
|
||||
]
|
||||
@@ -0,0 +1,11 @@
|
||||
{
|
||||
"sensitive_paths": [
|
||||
"/.env", "/.git/config", "wp-config.php", "/.htpasswd", "/phpmyadmin",
|
||||
"/xmlrpc.php", ".sql.gz", ".sql.bak", ".zip", ".bak", ".old"
|
||||
],
|
||||
"injection_markers": [
|
||||
"union select", "' or '1'='1", "<script>", "javascript:",
|
||||
"../", "..%2f", "%2e%2e%2f", "%252e%252e%252f"
|
||||
],
|
||||
"scanner_user_agents": ["sqlmap", "nikto", "nmap", "acunetix", "nessus"]
|
||||
}
|
||||
@@ -0,0 +1,93 @@
|
||||
"""Uploaded-file deletion (project-owner follow-up request).
|
||||
|
||||
Deleting a LogFile is more than one DELETE statement: log_entries,
|
||||
bot_hits, and suspicious_events all reference log_file_id and must be
|
||||
removed explicitly — this project's SQLite connections don't have
|
||||
`PRAGMA foreign_keys=ON`, so the `ondelete="CASCADE"` in the migrations
|
||||
(Ch06) is declarative documentation only, not an enforced behavior.
|
||||
Relying on it would silently leave orphaned rows behind.
|
||||
|
||||
The harder part is the rollup tables (request_stats_hourly/daily,
|
||||
referrer/browser/human-path/ip-path/ip-status): none of them have a
|
||||
log_file_id column (Ch06: they're site-wide, since two files can share a
|
||||
date). So instead of deleting rollup rows for the deleted file's dates
|
||||
directly (which could wipe out another file's contribution to the same
|
||||
date), this captures the affected dates BEFORE deleting, then calls
|
||||
aggregator.compute_rollups_for_range() AFTER deleting — which recomputes
|
||||
each affected day from whatever log_entries remain, correctly handling
|
||||
both "another file still covers this day" and "no file covers this day
|
||||
anymore" (the latter now handled correctly by the aggregator.py fix that
|
||||
shipped alongside this feature).
|
||||
|
||||
KNOWN LIMITATION (flagged, not fixed): ip_registry (total_requests,
|
||||
first_seen, last_seen, verification cache) and blocklist_suggestions are
|
||||
NOT per-file and are NOT recomputed on deletion — doing so correctly
|
||||
would mean re-scanning all remaining log_entries for every affected IP,
|
||||
which is unbounded work for a single delete action and would violate the
|
||||
same "no raw-row scan on a hot path" reasoning Ch03 applies elsewhere.
|
||||
After deleting a file, IP History numbers may include a deleted file's
|
||||
historical contribution until a full site-wide recompute is added as a
|
||||
separate feature. Doesn't affect correctness of Overview/SEO's date-
|
||||
scoped numbers, which is what this feature was actually asked to fix.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
|
||||
from app.extensions import db
|
||||
from app.models.bot_hit import BotHit
|
||||
from app.models.log_entry import LogEntry
|
||||
from app.models.log_file import LogFile
|
||||
from app.models.suspicious_event import SuspiciousEvent
|
||||
from app.services import aggregator
|
||||
from app.utils.upload_paths import upload_path_for
|
||||
|
||||
|
||||
@dataclass
|
||||
class BulkDeleteResult:
|
||||
deleted: list[int] = field(default_factory=list)
|
||||
skipped: list[dict] = field(default_factory=list) # [{"id": ..., "reason": ...}]
|
||||
|
||||
|
||||
def delete_log_files(log_file_ids: list[int]) -> BulkDeleteResult:
|
||||
"""Delete each id in `log_file_ids`; recompute rollups once at the end
|
||||
for the full span of dates any deleted file touched (cheaper than
|
||||
recomputing per-file when multiple files share dates).
|
||||
"""
|
||||
result = BulkDeleteResult()
|
||||
all_touched_dates: set = set()
|
||||
|
||||
for log_file_id in log_file_ids:
|
||||
log_file = db.session.get(LogFile, log_file_id)
|
||||
if log_file is None:
|
||||
result.skipped.append({"id": log_file_id, "reason": "not found"})
|
||||
continue
|
||||
if log_file.status == "processing":
|
||||
result.skipped.append({"id": log_file_id, "reason": "currently being analyzed"})
|
||||
continue
|
||||
|
||||
touched_dates = {
|
||||
row.date() for row, in db.session.query(LogEntry.timestamp)
|
||||
.filter(LogEntry.log_file_id == log_file_id).distinct()
|
||||
}
|
||||
# distinct() on a full timestamp rarely collapses much; reduce to
|
||||
# calendar dates in Python since SQLite's DATE() in a DISTINCT
|
||||
# clause is a bit more awkward to express portably here.
|
||||
all_touched_dates |= touched_dates
|
||||
|
||||
db.session.query(SuspiciousEvent).filter(SuspiciousEvent.log_file_id == log_file_id).delete()
|
||||
db.session.query(BotHit).filter(BotHit.log_file_id == log_file_id).delete()
|
||||
db.session.query(LogEntry).filter(LogEntry.log_file_id == log_file_id).delete()
|
||||
|
||||
raw_path = upload_path_for(log_file)
|
||||
if raw_path.exists():
|
||||
raw_path.unlink()
|
||||
|
||||
db.session.delete(log_file)
|
||||
db.session.commit()
|
||||
result.deleted.append(log_file_id)
|
||||
|
||||
if all_touched_dates:
|
||||
aggregator.compute_rollups_for_range(min(all_touched_dates), max(all_touched_dates))
|
||||
|
||||
return result
|
||||
@@ -0,0 +1,47 @@
|
||||
"""Log parser abstraction (Chapter 07)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime
|
||||
from typing import Protocol, TYPE_CHECKING
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from app.models.log_file import LogFile
|
||||
|
||||
|
||||
@dataclass
|
||||
class ParsedEntry:
|
||||
timestamp: datetime
|
||||
ip: str
|
||||
method: str
|
||||
path: str
|
||||
status_code: int
|
||||
bytes_sent: int
|
||||
referrer: str | None
|
||||
user_agent: str | None
|
||||
|
||||
|
||||
class LogParser(Protocol):
|
||||
def parse_line(self, line: str) -> ParsedEntry | None: ...
|
||||
|
||||
|
||||
def get_parser_for(log_file: "LogFile") -> LogParser:
|
||||
"""Build the right LogParser for a given upload's format_string.
|
||||
|
||||
ASSUMPTION (flagged): a blank format_string — which the upload form
|
||||
allows — falls back to Combined, since Chapter 07 states no default
|
||||
for that case. Anything else is compiled as-is; a real custom
|
||||
LiteSpeed format is never silently replaced by a preset.
|
||||
"""
|
||||
# Imported lazily to avoid a circular import (apache.py imports
|
||||
# ParsedEntry from this module).
|
||||
from app.services.log_parser.apache import FormatCompiledParser, combined_parser, common_parser
|
||||
from app.services.log_parser.format_compiler import compile_format
|
||||
|
||||
fmt = (log_file.format_string or "").strip()
|
||||
key = fmt.lower()
|
||||
if key in ("", "combined"):
|
||||
return combined_parser()
|
||||
if key == "common":
|
||||
return common_parser()
|
||||
return FormatCompiledParser(compile_format(fmt))
|
||||
@@ -0,0 +1,57 @@
|
||||
"""Apache Common/Combined presets + the generic parser built from
|
||||
format_compiler (Chapter 07)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from app.services.log_parser import ParsedEntry
|
||||
from app.services.log_parser.format_compiler import (
|
||||
CompiledFormat, compile_format, parse_apache_timestamp, parse_request_line, validate_ip,
|
||||
)
|
||||
|
||||
COMMON_LOG_FORMAT = '%h %l %u %t "%r" %>s %b'
|
||||
COMBINED_LOG_FORMAT = '%h %l %u %t "%r" %>s %b "%{Referer}i" "%{User-agent}i"'
|
||||
|
||||
|
||||
class FormatCompiledParser:
|
||||
"""LogParser implementation driven by any CompiledFormat (Ch04/07 protocol)."""
|
||||
|
||||
def __init__(self, compiled: CompiledFormat):
|
||||
self._compiled = compiled
|
||||
|
||||
def parse_line(self, line: str) -> ParsedEntry | None:
|
||||
match = self._compiled.pattern.match(line.rstrip("\n"))
|
||||
if match is None:
|
||||
return None # malformed line — skip, never raise (Ch07)
|
||||
fields = match.groupdict()
|
||||
|
||||
ip = validate_ip(fields["ip"])
|
||||
if ip is None:
|
||||
return None
|
||||
|
||||
req = parse_request_line(fields["request"])
|
||||
if req is None:
|
||||
return None
|
||||
method, path = req
|
||||
|
||||
try:
|
||||
timestamp = parse_apache_timestamp(fields["timestamp"])
|
||||
status_code = int(fields["status"])
|
||||
bytes_sent = 0 if fields["bytes"] == "-" else int(fields["bytes"])
|
||||
except (ValueError, KeyError):
|
||||
return None
|
||||
|
||||
referrer = fields.get("referrer")
|
||||
user_agent = fields.get("user_agent")
|
||||
return ParsedEntry(
|
||||
timestamp=timestamp, ip=ip, method=method, path=path,
|
||||
status_code=status_code, bytes_sent=bytes_sent,
|
||||
referrer=None if referrer in (None, "-") else referrer,
|
||||
user_agent=None if user_agent in (None, "-") else user_agent,
|
||||
)
|
||||
|
||||
|
||||
def common_parser() -> FormatCompiledParser:
|
||||
return FormatCompiledParser(compile_format(COMMON_LOG_FORMAT))
|
||||
|
||||
|
||||
def combined_parser() -> FormatCompiledParser:
|
||||
return FormatCompiledParser(compile_format(COMBINED_LOG_FORMAT))
|
||||
@@ -0,0 +1,95 @@
|
||||
"""LogFormat directive -> compiled regex + named groups (Chapter 07).
|
||||
|
||||
One code path for Common, Combined, and arbitrary custom formats — presets
|
||||
below are just specific format strings run through this same compiler.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime, timezone
|
||||
from ipaddress import ip_address
|
||||
|
||||
# Directive -> (group name, regex pattern). NOTE: %r and %{...}i patterns
|
||||
# deliberately do NOT include quote characters — the quotes around them
|
||||
# are literal text already present in the format string (e.g. `"%r"`),
|
||||
# and get regex-escaped by the literal-text path below. %t is different:
|
||||
# its brackets are part of what %t itself produces, not literal format-
|
||||
# string text, so they belong in this directive's own pattern.
|
||||
_SIMPLE_DIRECTIVES: dict[str, tuple[str, str]] = {
|
||||
"%h": ("ip", r"(?P<ip>\S+)"),
|
||||
"%l": ("ident", r"(?P<ident>\S+)"),
|
||||
"%u": ("user", r"(?P<user>\S+)"),
|
||||
"%t": ("timestamp", r"\[(?P<timestamp>[^\]]+)\]"),
|
||||
"%r": ("request", r'(?P<request>[^"]*)'),
|
||||
"%>s": ("status", r"(?P<status>\d{3})"),
|
||||
"%s": ("status", r"(?P<status>\d{3})"),
|
||||
"%b": ("bytes", r"(?P<bytes>\d+|-)"),
|
||||
}
|
||||
|
||||
_HEADER_DIRECTIVE_RE = re.compile(r'%\{([^}]+)\}i')
|
||||
_KNOWN_HEADER_GROUPS = {"referer": "referrer", "user-agent": "user_agent"}
|
||||
_DIRECTIVE_TOKEN_RE = re.compile(r"%>?\{[^}]+\}i|%>?[a-zA-Z]")
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class CompiledFormat:
|
||||
pattern: re.Pattern
|
||||
group_names: frozenset[str]
|
||||
|
||||
|
||||
def compile_format(format_string: str) -> CompiledFormat:
|
||||
"""Compile a LogFormat directive (Common/Combined preset or a pasted
|
||||
custom vhost format) into one regex."""
|
||||
regex_parts: list[str] = []
|
||||
group_names: set[str] = set()
|
||||
pos = 0
|
||||
for match in _DIRECTIVE_TOKEN_RE.finditer(format_string):
|
||||
literal = format_string[pos:match.start()]
|
||||
if literal:
|
||||
regex_parts.append(re.escape(literal))
|
||||
regex_parts.append(_directive_to_pattern(match.group(0), group_names))
|
||||
pos = match.end()
|
||||
regex_parts.append(re.escape(format_string[pos:]))
|
||||
|
||||
pattern = re.compile("^" + "".join(regex_parts) + r"\s*$")
|
||||
return CompiledFormat(pattern=pattern, group_names=frozenset(group_names))
|
||||
|
||||
|
||||
def _directive_to_pattern(token: str, group_names: set[str]) -> str:
|
||||
header_match = _HEADER_DIRECTIVE_RE.fullmatch(token)
|
||||
if header_match:
|
||||
header_name = header_match.group(1).lower()
|
||||
group = _KNOWN_HEADER_GROUPS.get(header_name, re.sub(r"[^a-z0-9]+", "_", header_name))
|
||||
group_names.add(group)
|
||||
return f'(?P<{group}>[^"]*)'
|
||||
|
||||
if token not in _SIMPLE_DIRECTIVES:
|
||||
raise ValueError(f"Unsupported LogFormat directive: {token!r}")
|
||||
group, pattern = _SIMPLE_DIRECTIVES[token]
|
||||
group_names.add(group)
|
||||
return pattern
|
||||
|
||||
|
||||
def parse_apache_timestamp(raw: str) -> datetime:
|
||||
"""'10/Oct/2026:13:55:36 -0700' -> naive UTC datetime (Ch07: store normalized to UTC)."""
|
||||
dt = datetime.strptime(raw, "%d/%b/%Y:%H:%M:%S %z")
|
||||
return dt.astimezone(timezone.utc).replace(tzinfo=None)
|
||||
|
||||
|
||||
def parse_request_line(raw: str) -> tuple[str, str] | None:
|
||||
"""Split '%r' ("GET /path HTTP/1.1") into (method, path); None if malformed."""
|
||||
parts = raw.split()
|
||||
if len(parts) != 3:
|
||||
return None
|
||||
method, path, _protocol = parts
|
||||
return method, path
|
||||
|
||||
|
||||
def validate_ip(raw: str) -> str | None:
|
||||
"""Return `raw` if a valid IPv4/IPv6 address, else None (Ch07)."""
|
||||
try:
|
||||
ip_address(raw)
|
||||
except ValueError:
|
||||
return None
|
||||
return raw
|
||||
@@ -0,0 +1,10 @@
|
||||
"""LiteSpeed uses the same LogFormat directives as Apache (Chapter 07):
|
||||
Combined generally works unmodified. get_parser_for() still always
|
||||
compiles the vhost's real format_string though — this module is just the
|
||||
named default, never a silent override of a custom format.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from app.services.log_parser.apache import combined_parser as default_litespeed_parser
|
||||
|
||||
__all__ = ["default_litespeed_parser"]
|
||||
@@ -0,0 +1,148 @@
|
||||
"""Core log-file processing pipeline (Chapter 07), factored out of
|
||||
app/cli.py so it has exactly one implementation shared by:
|
||||
- the optional `flask process-logs` CLI command (for anyone who still
|
||||
wants cron), and
|
||||
- the automatic background-thread trigger fired right after upload and
|
||||
opportunistically on page load (app/services/background.py) — the
|
||||
no-cron-required simplification.
|
||||
|
||||
process_one_batch() keeps the same bounded-batch, checkpointed-commit
|
||||
contract as the original cron design (Ch03 rule 1/4/9): it still never
|
||||
loads a whole file into memory, still commits progress incrementally, and
|
||||
still stops after `batch_size` lines so a very large file doesn't hold one
|
||||
enormous open transaction — the only thing that changed is *who* calls it
|
||||
again for the next batch.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import gzip
|
||||
import itertools
|
||||
from datetime import date
|
||||
from pathlib import Path
|
||||
|
||||
from sqlalchemy import insert
|
||||
|
||||
from app.extensions import db
|
||||
from app.models.log_entry import LogEntry
|
||||
from app.models.log_file import LogFile
|
||||
from app.services import aggregator
|
||||
from app.services.classification import classify_entry, write_batch_side_effects
|
||||
from app.services.log_parser import ParsedEntry, get_parser_for
|
||||
from app.utils.upload_paths import upload_path_for
|
||||
|
||||
|
||||
def open_log_stream(path: Path):
|
||||
"""Open a log file, transparently decompressing .gz, one line at a time."""
|
||||
if path.suffix == ".gz":
|
||||
return gzip.open(path, mode="rt", encoding="utf-8", errors="replace")
|
||||
return path.open("r", encoding="utf-8", errors="replace")
|
||||
|
||||
|
||||
def count_total_lines(path: Path) -> int:
|
||||
"""One streaming pass to count lines (bounded memory — never the whole
|
||||
file at once, Ch03 rule 1), so the UI can show a real
|
||||
processed/total percentage instead of an indeterminate spinner.
|
||||
|
||||
This is an extra sequential read of the file beyond the parse pass
|
||||
itself. That cost is now worth paying: processing used to be silently
|
||||
triggered by cron with no live audience watching, but now it runs
|
||||
automatically right after upload while the admin is looking at a
|
||||
progress bar — the UX value of a real percentage justifies the extra
|
||||
I/O pass.
|
||||
"""
|
||||
with open_log_stream(path) as fh:
|
||||
return sum(1 for _ in fh)
|
||||
|
||||
|
||||
def process_one_batch(log_file: LogFile, batch_size: int) -> None:
|
||||
"""Parse up to `batch_size` lines from `log_file`'s current checkpoint.
|
||||
|
||||
Leaves status as "processing" (with progress already committed) if
|
||||
the file isn't finished yet — the caller decides whether to invoke
|
||||
this again: the CLI calls it once per pending file per invocation
|
||||
(unchanged cron-tick semantics); the background thread loops it until
|
||||
the file reaches "done"/"error".
|
||||
"""
|
||||
path = upload_path_for(log_file)
|
||||
if not path.exists():
|
||||
log_file.status = "error"
|
||||
log_file.error_message = f"Upload file missing on disk: {path}"
|
||||
db.session.commit()
|
||||
return
|
||||
|
||||
if log_file.total_lines is None:
|
||||
# First pickup of this file — count once, not on every batch.
|
||||
log_file.total_lines = count_total_lines(path)
|
||||
|
||||
log_file.status = "processing"
|
||||
db.session.commit()
|
||||
parser = get_parser_for(log_file)
|
||||
|
||||
lines_seen_this_run = 0
|
||||
skipped_this_run = 0
|
||||
dates_touched: set[date] = set()
|
||||
|
||||
with open_log_stream(path) as fh:
|
||||
remainder = itertools.islice(fh, log_file.processed_lines, None)
|
||||
batch_entries: list[ParsedEntry] = []
|
||||
|
||||
for line in remainder:
|
||||
entry = parser.parse_line(line)
|
||||
if entry is not None:
|
||||
batch_entries.append(entry)
|
||||
dates_touched.add(entry.timestamp.date())
|
||||
else:
|
||||
skipped_this_run += 1 # malformed line — skip, don't abort the batch (Ch07)
|
||||
|
||||
log_file.processed_lines += 1
|
||||
lines_seen_this_run += 1
|
||||
|
||||
if len(batch_entries) >= 500:
|
||||
_bulk_insert_entries(log_file.id, batch_entries)
|
||||
batch_entries = []
|
||||
|
||||
if lines_seen_this_run >= batch_size:
|
||||
_record_skip_count(log_file, skipped_this_run)
|
||||
db.session.commit() # persist checkpoint; resumable if interrupted
|
||||
return
|
||||
|
||||
if batch_entries:
|
||||
_bulk_insert_entries(log_file.id, batch_entries)
|
||||
|
||||
if dates_touched:
|
||||
aggregator.compute_rollups_for_range(min(dates_touched), max(dates_touched))
|
||||
|
||||
_record_skip_count(log_file, skipped_this_run)
|
||||
log_file.status = "done"
|
||||
db.session.commit()
|
||||
|
||||
|
||||
def _record_skip_count(log_file: LogFile, skipped_this_run: int) -> None:
|
||||
"""Informational only — doesn't touch `status`.
|
||||
|
||||
STOPGAP (flagged in Ch07): log_files has no dedicated skip-counter
|
||||
column, so this overwrites error_message with the latest run's count
|
||||
rather than accumulating across runs.
|
||||
"""
|
||||
if skipped_this_run:
|
||||
log_file.error_message = f"{skipped_this_run} unparsable line(s) skipped in the most recent parse run."
|
||||
|
||||
|
||||
def _bulk_insert_entries(log_file_id: int, entries: list[ParsedEntry]) -> None:
|
||||
"""Bulk-insert log_entries (Ch03 rule 4), classifying is_bot/flagged
|
||||
per line, then derive bot_hits/suspicious_events/ip_registry (Ch07).
|
||||
"""
|
||||
if not entries:
|
||||
return
|
||||
payload = []
|
||||
for e in entries:
|
||||
classification = classify_entry(e)
|
||||
payload.append({
|
||||
"log_file_id": log_file_id, "timestamp": e.timestamp, "ip": e.ip,
|
||||
"method": e.method, "path": e.path, "status_code": e.status_code,
|
||||
"bytes_sent": e.bytes_sent, "referrer": e.referrer, "user_agent": e.user_agent,
|
||||
"is_bot": classification.is_bot, "flagged": classification.flagged,
|
||||
})
|
||||
db.session.execute(insert(LogEntry.__table__), payload)
|
||||
db.session.commit()
|
||||
write_batch_side_effects(log_file_id, entries)
|
||||
@@ -0,0 +1,19 @@
|
||||
"""Referrer -> domain bucketing (Chapter 08 rollup, Method A)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from urllib.parse import urlparse
|
||||
|
||||
|
||||
def referrer_domain(referrer: str | None) -> str | None:
|
||||
"""Bucket a raw referrer URL to its host, stripping a leading 'www.'.
|
||||
|
||||
Full-URL cardinality would make the daily rollup unbounded in row
|
||||
count; domain bucketing keeps it small (Ch03's bounded-resource bias).
|
||||
Empty/unparsable referrers return None and are excluded from the rollup.
|
||||
"""
|
||||
if not referrer:
|
||||
return None
|
||||
host = urlparse(referrer).netloc.lower()
|
||||
if not host:
|
||||
return None
|
||||
return host[4:] if host.startswith("www.") else host
|
||||
@@ -0,0 +1,49 @@
|
||||
"""Severity scoring (Chapter 10): a simple, transparent rank-escalation
|
||||
model — no ML, easy to explain to a non-expert user (Ch10's own requirement).
|
||||
|
||||
Chapter 07 assigns a PROVISIONAL severity per rule type at parse time.
|
||||
This module computes an EFFECTIVE severity by escalating that base rank
|
||||
for repeated/rapid hits from the same IP — the exact rule Ch10 names.
|
||||
|
||||
Escalation is rank-based and strictly non-decreasing: it starts at the
|
||||
base severity's rank and only ever moves up (repeat-count / hit-rate
|
||||
bonuses), clamped at "high". An earlier weighted-score-vs-fixed-threshold
|
||||
version could silently *downgrade* an isolated high-severity event (e.g.
|
||||
a single sqlmap hit) to "medium" purely because the thresholds weren't
|
||||
calibrated to the base weights — caught by running the pipeline against
|
||||
real sample data. A single dangerous event must never end up rated below
|
||||
its own base severity; only repetition/rate should push it higher.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
|
||||
_SEVERITY_RANKS = ["low", "medium", "high"]
|
||||
_RANK_BY_SEVERITY = {name: rank for rank, name in enumerate(_SEVERITY_RANKS)}
|
||||
|
||||
_REPEAT_COUNT_HIGH = 20
|
||||
_REPEAT_COUNT_MEDIUM = 5
|
||||
_RAPID_AVG_INTERVAL_SECONDS = 60 # >1 event/minute from one IP suggests automation
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class SeverityInputs:
|
||||
base_severity: str
|
||||
ip_event_count: int
|
||||
avg_interval_seconds: float | None # None if this IP has < 2 events in the window
|
||||
|
||||
|
||||
def compute_effective_severity(inputs: SeverityInputs) -> str:
|
||||
"""Two named, inspectable escalation rules: repeat-count and hit-rate."""
|
||||
rank = _RANK_BY_SEVERITY.get(inputs.base_severity, 0)
|
||||
|
||||
if inputs.ip_event_count >= _REPEAT_COUNT_HIGH:
|
||||
rank += 2
|
||||
elif inputs.ip_event_count >= _REPEAT_COUNT_MEDIUM:
|
||||
rank += 1
|
||||
|
||||
if inputs.avg_interval_seconds is not None and inputs.avg_interval_seconds < _RAPID_AVG_INTERVAL_SECONDS:
|
||||
rank += 1
|
||||
|
||||
rank = min(rank, len(_SEVERITY_RANKS) - 1)
|
||||
return _SEVERITY_RANKS[rank]
|
||||
@@ -0,0 +1,54 @@
|
||||
"""Suspicious-pattern matching (Chapter 07), reused as-is by Chapter 10.
|
||||
|
||||
Pattern dictionary is data (JSON), same reasoning as the bot signature
|
||||
table — editable without a code change. Severity here is a provisional
|
||||
per-rule-type default; Chapter 10 owns the full weighted/escalating score.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from app.services.log_parser import ParsedEntry
|
||||
|
||||
PATTERNS_PATH = Path(__file__).parent / "data" / "threat_patterns.json"
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ThreatMatch:
|
||||
rule_matched: str
|
||||
severity: str # low|medium|high — provisional; Ch10 may escalate
|
||||
|
||||
|
||||
def _load_patterns() -> dict:
|
||||
return json.loads(PATTERNS_PATH.read_text())
|
||||
|
||||
|
||||
_PATTERNS = _load_patterns()
|
||||
|
||||
|
||||
def scan(entry: "ParsedEntry") -> ThreatMatch | None:
|
||||
"""Check one parsed entry against the pattern dictionary.
|
||||
|
||||
Cheap substring checks only — no regex backtracking risk, no network
|
||||
calls (Ch03: CPU-cheap, explainable rule matching).
|
||||
"""
|
||||
path_lower = entry.path.lower()
|
||||
ua_lower = (entry.user_agent or "").lower()
|
||||
|
||||
for scanner_ua in _PATTERNS["scanner_user_agents"]:
|
||||
if scanner_ua in ua_lower:
|
||||
return ThreatMatch(rule_matched=f"scanner_ua:{scanner_ua}", severity="high")
|
||||
|
||||
for sensitive in _PATTERNS["sensitive_paths"]:
|
||||
if sensitive.lower() in path_lower:
|
||||
return ThreatMatch(rule_matched=f"sensitive_path:{sensitive}", severity="medium")
|
||||
|
||||
for marker in _PATTERNS["injection_markers"]:
|
||||
if marker.lower() in path_lower:
|
||||
return ThreatMatch(rule_matched=f"injection:{marker}", severity="high")
|
||||
|
||||
return None
|
||||
@@ -0,0 +1,54 @@
|
||||
"""Lightweight browser/OS classification for human traffic (Chapter 08).
|
||||
|
||||
Separate from Chapter 07's bot_identifier — that identifies automated
|
||||
crawlers via UA substring + DNS verification. This classifies ordinary
|
||||
browsers/OSes, same ordered-substring-match pattern, signatures as JSON
|
||||
data (not hardcoded), no third-party UA-parsing dependency — consistent
|
||||
with the low-footprint bias in Chapter 03.
|
||||
|
||||
Order matters within each signature list: e.g. Edge/Opera must be checked
|
||||
before Chrome (their UAs also contain "Chrome/"); iOS before macOS (iPhone
|
||||
UAs also contain "like Mac OS X"); Android before Linux (Android UAs also
|
||||
contain "Linux").
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
|
||||
_DATA_DIR = Path(__file__).parent / "data"
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class UASignature:
|
||||
name: str
|
||||
ua_substrings: tuple[str, ...]
|
||||
|
||||
|
||||
def _load(filename: str) -> list[UASignature]:
|
||||
raw = json.loads((_DATA_DIR / filename).read_text())
|
||||
return [UASignature(name=e["name"], ua_substrings=tuple(e["ua_substrings"])) for e in raw]
|
||||
|
||||
|
||||
_BROWSER_SIGNATURES = _load("browser_signatures.json")
|
||||
_OS_SIGNATURES = _load("os_signatures.json")
|
||||
|
||||
|
||||
def _match(user_agent: str, signatures: list[UASignature], default: str) -> str:
|
||||
for sig in signatures:
|
||||
if any(sub in user_agent for sub in sig.ua_substrings):
|
||||
return sig.name
|
||||
return default
|
||||
|
||||
|
||||
def classify_browser(user_agent: str | None) -> str:
|
||||
if not user_agent:
|
||||
return "Unknown"
|
||||
return _match(user_agent, _BROWSER_SIGNATURES, "Other")
|
||||
|
||||
|
||||
def classify_os(user_agent: str | None) -> str:
|
||||
if not user_agent:
|
||||
return "Unknown"
|
||||
return _match(user_agent, _OS_SIGNATURES, "Other")
|
||||
@@ -0,0 +1,67 @@
|
||||
/* app/static/src/css/main.css — Tailwind entry point (Chapter 05) */
|
||||
|
||||
/* Self-hosted fonts (Ch05: build-time only, no runtime CDN dependency).
|
||||
Display face used sparingly for headings/labels; body face for UI
|
||||
chrome; mono face for every data readout (IPs, counts, paths,
|
||||
timestamps) — reinforcing that this app's whole job is turning raw
|
||||
log lines into a readable instrument panel. */
|
||||
@import "@fontsource/space-grotesk/500.css";
|
||||
@import "@fontsource/space-grotesk/700.css";
|
||||
@import "@fontsource/public-sans/400.css";
|
||||
@import "@fontsource/public-sans/500.css";
|
||||
@import "@fontsource/public-sans/600.css";
|
||||
@import "@fontsource/jetbrains-mono/400.css";
|
||||
@import "@fontsource/jetbrains-mono/500.css";
|
||||
|
||||
@import "gridjs/dist/theme/mermaid.css";
|
||||
|
||||
@tailwind base;
|
||||
@tailwind components;
|
||||
@tailwind utilities;
|
||||
|
||||
@layer base {
|
||||
html {
|
||||
color-scheme: light;
|
||||
}
|
||||
html.dark {
|
||||
color-scheme: dark;
|
||||
}
|
||||
body {
|
||||
font-family: theme('fontFamily.sans');
|
||||
}
|
||||
h1, h2, h3, h4, .font-display {
|
||||
font-family: theme('fontFamily.display');
|
||||
}
|
||||
/* Every number/IP/path/timestamp reads like it came straight off the
|
||||
log line — the app's one signature typographic choice. */
|
||||
.font-data {
|
||||
font-family: theme('fontFamily.mono');
|
||||
font-variant-numeric: tabular-nums;
|
||||
}
|
||||
}
|
||||
|
||||
@layer components {
|
||||
/* Grid.js theme override to match the token system in both modes,
|
||||
without forking the whole "mermaid" stylesheet. */
|
||||
.gridjs-wrapper, .gridjs-container {
|
||||
@apply !border-line dark:!border-line-dark !rounded-lg;
|
||||
}
|
||||
.gridjs-table {
|
||||
@apply font-data text-sm;
|
||||
}
|
||||
.gridjs-th {
|
||||
@apply !bg-surface-raised dark:!bg-surface-raised-dark !text-muted dark:!text-muted-dark font-sans !font-medium !text-xs !uppercase !tracking-wide;
|
||||
}
|
||||
.gridjs-td {
|
||||
@apply !bg-surface dark:!bg-surface-dark !text-ink dark:!text-ink-dark !border-line dark:!border-line-dark;
|
||||
}
|
||||
.gridjs-footer {
|
||||
@apply !bg-surface dark:!bg-surface-dark !border-line dark:!border-line-dark;
|
||||
}
|
||||
.gridjs-pagination .gridjs-pages button {
|
||||
@apply !text-ink dark:!text-ink-dark;
|
||||
}
|
||||
.gridjs-search-input {
|
||||
@apply !bg-surface dark:!bg-surface-dark !text-ink dark:!text-ink-dark !border-line dark:!border-line-dark;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,62 @@
|
||||
/**
|
||||
* Shared Chart.js setup (Chapter 05) — avoids duplicating config per chart.
|
||||
* Callers must pass pre-aggregated series only (≤~200 points, per Ch05);
|
||||
* this file never fetches or aggregates raw rows itself.
|
||||
*
|
||||
* Dark-mode aware: reads the current theme at init time and colors grid
|
||||
* lines/ticks/legend accordingly. Charts drawn before a theme toggle
|
||||
* don't live-repaint — they pick up the new theme on their next redraw
|
||||
* (date-range change, tab revisit), which keeps this simple.
|
||||
*/
|
||||
import { Chart, registerables } from 'chart.js';
|
||||
|
||||
Chart.register(...registerables);
|
||||
|
||||
// Shared categorical palette, drawn from the Kavosh token system rather
|
||||
// than Chart.js's stock colors.
|
||||
export const CHART_PALETTE = ['#0E7C86', '#B8860B', '#C4372F', '#3F7D5C', '#5B6663', '#8A6FBE'];
|
||||
|
||||
function isDarkMode() {
|
||||
return document.documentElement.classList.contains('dark');
|
||||
}
|
||||
|
||||
function defaultOptions() {
|
||||
const dark = isDarkMode();
|
||||
const textColor = dark ? '#96A19D' : '#5B6663';
|
||||
const gridColor = dark ? '#2C3336' : '#E1E5E2';
|
||||
const tickFont = { family: '"JetBrains Mono", monospace', size: 11 };
|
||||
|
||||
return {
|
||||
responsive: true,
|
||||
maintainAspectRatio: false,
|
||||
animation: { duration: 200 },
|
||||
plugins: {
|
||||
legend: {
|
||||
display: true,
|
||||
position: 'bottom',
|
||||
labels: { color: textColor, font: { family: '"Public Sans", sans-serif', size: 12 }, boxWidth: 12, padding: 12 },
|
||||
},
|
||||
},
|
||||
scales: {
|
||||
x: { ticks: { color: textColor, font: tickFont }, grid: { color: gridColor } },
|
||||
y: { ticks: { color: textColor, font: tickFont }, grid: { color: gridColor } },
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
/** @param {string} canvasId @param {object} config @returns {Chart} */
|
||||
export function initChart(canvasId, config) {
|
||||
const ctx = document.getElementById(canvasId);
|
||||
if (!ctx) throw new Error(`initChart: no element #${canvasId}`);
|
||||
const base = defaultOptions();
|
||||
const callerOptions = config.options ?? {};
|
||||
return new Chart(ctx, {
|
||||
...config,
|
||||
options: {
|
||||
...base,
|
||||
...callerOptions,
|
||||
scales: config.type === 'doughnut' || config.type === 'pie' ? undefined : { ...base.scales, ...callerOptions.scales },
|
||||
plugins: { ...base.plugins, ...callerOptions.plugins },
|
||||
},
|
||||
});
|
||||
}
|
||||
@@ -0,0 +1,38 @@
|
||||
/**
|
||||
* Shared Grid.js setup (Chapter 05) — every table is server-side
|
||||
* paginated/sorted/searched against `?page=&per_page=&sort=&q=`.
|
||||
*/
|
||||
import { Grid } from 'gridjs';
|
||||
|
||||
/**
|
||||
* @param {string} elementId container element id
|
||||
* @param {string} endpoint JSON endpoint, e.g. '/api/overview/top-urls'
|
||||
* @param {Array} columns Grid.js column defs
|
||||
* @param {object} extraParams static extra query params (e.g. {from, to})
|
||||
*/
|
||||
export function initGrid(elementId, endpoint, columns, extraParams = {}) {
|
||||
const buildUrl = (prev, params) => {
|
||||
const url = new URL(prev, window.location.origin);
|
||||
Object.entries({ ...extraParams, ...params }).forEach(([k, v]) => url.searchParams.set(k, v));
|
||||
return url.toString();
|
||||
};
|
||||
|
||||
return new Grid({
|
||||
columns,
|
||||
server: {
|
||||
url: endpoint,
|
||||
then: (res) => res.data.rows,
|
||||
total: (res) => res.data.total,
|
||||
},
|
||||
pagination: {
|
||||
server: { url: (prev, page, per_page) => buildUrl(prev, { page: page + 1, per_page }) },
|
||||
limit: 20,
|
||||
},
|
||||
sort: {
|
||||
server: {
|
||||
url: (prev, cols) => (cols.length ? buildUrl(prev, { sort: cols[0].direction === 1 ? cols[0].index : `-${cols[0].index}` }) : prev),
|
||||
},
|
||||
},
|
||||
search: { server: { url: (prev, keyword) => buildUrl(prev, { q: keyword }) } },
|
||||
}).render(document.getElementById(elementId));
|
||||
}
|
||||
@@ -0,0 +1,36 @@
|
||||
/**
|
||||
* Vite entry point: HTMX + Alpine init + Chart.js/Grid.js glue (Chapter 05).
|
||||
* Bundled at build time — no CDN scripts, nothing compiled at request time.
|
||||
*/
|
||||
import htmx from 'htmx.org';
|
||||
import Alpine from 'alpinejs';
|
||||
import { html as gridHtml } from 'gridjs';
|
||||
|
||||
import { initChart, CHART_PALETTE } from './charts.js';
|
||||
import { initGrid } from './grids.js';
|
||||
import { initUploadForm } from './uploads.js';
|
||||
|
||||
window.htmx = htmx;
|
||||
window.Alpine = Alpine;
|
||||
window.initChart = initChart;
|
||||
window.initGrid = initGrid;
|
||||
window.gridHtml = gridHtml;
|
||||
window.initUploadForm = initUploadForm;
|
||||
window.KAVOSH_CHART_PALETTE = CHART_PALETTE;
|
||||
|
||||
// Chapter 12: HTMX requests must carry the CSRF token as a header, not
|
||||
// rely on the cookie alone.
|
||||
document.body.addEventListener('htmx:configRequest', (event) => {
|
||||
const token = document.querySelector('meta[name="csrf-token"]')?.content;
|
||||
if (token) event.detail.headers['X-CSRFToken'] = token;
|
||||
});
|
||||
|
||||
// Dark mode toggle (persisted; the initial class is set synchronously in
|
||||
// an inline <head> script in base.html to avoid a flash of the wrong
|
||||
// theme before this bundle loads).
|
||||
window.toggleTheme = function toggleTheme() {
|
||||
const isDark = document.documentElement.classList.toggle('dark');
|
||||
localStorage.setItem('kavosh-theme', isDark ? 'dark' : 'light');
|
||||
};
|
||||
|
||||
Alpine.start();
|
||||
@@ -0,0 +1,62 @@
|
||||
/**
|
||||
* Upload with real byte-transfer progress (Chapter 08/12).
|
||||
*
|
||||
* fetch() can't reliably report upload progress across browsers, so this
|
||||
* uses XMLHttpRequest directly rather than depending on htmx's internal
|
||||
* transport. On completion, the returned HTML fragment (queued/duplicate/
|
||||
* error) is injected manually, then handed to htmx.process() so any
|
||||
* hx-* polling attributes inside it (e.g. the processing-status poll)
|
||||
* activate normally, exactly as if htmx had performed the swap itself.
|
||||
*/
|
||||
export function initUploadForm(formId) {
|
||||
const form = document.getElementById(formId);
|
||||
if (!form) return;
|
||||
|
||||
const fileInput = form.querySelector('input[type="file"]');
|
||||
const progressWrap = document.getElementById('upload-progress-wrap');
|
||||
const progressBar = document.getElementById('upload-progress-bar');
|
||||
const progressLabel = document.getElementById('upload-progress-label');
|
||||
const resultEl = document.getElementById('upload-result');
|
||||
const submitBtn = form.querySelector('button[type="submit"]');
|
||||
|
||||
form.addEventListener('submit', (event) => {
|
||||
event.preventDefault();
|
||||
if (!fileInput.files.length) return;
|
||||
|
||||
const formData = new FormData(form);
|
||||
const xhr = new XMLHttpRequest();
|
||||
|
||||
submitBtn.disabled = true;
|
||||
resultEl.innerHTML = '';
|
||||
progressWrap.classList.remove('hidden');
|
||||
progressBar.style.width = '0%';
|
||||
progressLabel.textContent = 'Uploading… 0%';
|
||||
|
||||
xhr.upload.addEventListener('progress', (e) => {
|
||||
if (!e.lengthComputable) return;
|
||||
const pct = Math.round((e.loaded / e.total) * 100);
|
||||
progressBar.style.width = pct + '%';
|
||||
progressLabel.textContent = `Uploading… ${pct}%`;
|
||||
});
|
||||
|
||||
xhr.addEventListener('load', () => {
|
||||
submitBtn.disabled = false;
|
||||
progressWrap.classList.add('hidden');
|
||||
resultEl.innerHTML = xhr.responseText;
|
||||
if (window.htmx) window.htmx.process(resultEl);
|
||||
form.reset();
|
||||
});
|
||||
|
||||
xhr.addEventListener('error', () => {
|
||||
submitBtn.disabled = false;
|
||||
progressWrap.classList.add('hidden');
|
||||
resultEl.innerHTML =
|
||||
'<p class="text-danger dark:text-danger-dark text-sm">Upload failed — check your connection and try again.</p>';
|
||||
});
|
||||
|
||||
xhr.open('POST', form.getAttribute('action') || '/uploads');
|
||||
const token = document.querySelector('meta[name="csrf-token"]')?.content;
|
||||
if (token) xhr.setRequestHeader('X-CSRFToken', token);
|
||||
xhr.send(formData);
|
||||
});
|
||||
}
|
||||
@@ -0,0 +1,85 @@
|
||||
<!doctype html>
|
||||
<html lang="en" class="h-full">
|
||||
<head>
|
||||
<meta charset="utf-8">
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1">
|
||||
<meta name="csrf-token" content="{{ csrf_token() }}">
|
||||
<title>{% block title %}Kavosh{% endblock %}</title>
|
||||
<!--
|
||||
Anti-FOUC: set the theme class synchronously, before the stylesheet
|
||||
and before Alpine/htmx load, so there's no flash of the wrong theme
|
||||
on first paint. Kept as a tiny inline script rather than waiting for
|
||||
the bundled JS, which only arrives after the page has started
|
||||
rendering.
|
||||
-->
|
||||
<script>
|
||||
(function () {
|
||||
var stored = localStorage.getItem('kavosh-theme');
|
||||
var isDark = stored ? stored === 'dark' : window.matchMedia('(prefers-color-scheme: dark)').matches;
|
||||
document.documentElement.classList.toggle('dark', isDark);
|
||||
})();
|
||||
</script>
|
||||
<link rel="stylesheet" href="{{ asset('app/static/src/css/main.css') }}">
|
||||
</head>
|
||||
<body class="h-full bg-paper dark:bg-paper-dark text-ink dark:text-ink-dark antialiased">
|
||||
<div id="app-shell" x-data="{ mobileNavOpen: false }">
|
||||
<header class="border-b border-line dark:border-line-dark bg-surface dark:bg-surface-dark">
|
||||
<nav class="flex gap-1 px-4 py-3 items-center max-w-6xl mx-auto">
|
||||
<span class="font-display font-bold text-lg tracking-tight mr-4">Kavosh</span>
|
||||
|
||||
{% if current_user.is_authenticated %}
|
||||
{% set tabs = [
|
||||
('overview.overview', '/overview', 'house', 'Overview'),
|
||||
('seo.seo', '/seo', 'search', 'SEO & Bots'),
|
||||
('security.security', '/security', 'shield-alert', 'Security'),
|
||||
] %}
|
||||
{% for endpoint, path, icon, label in tabs %}
|
||||
<a href="{{ path }}" hx-get="{{ path }}" hx-target="#main-content" hx-push-url="true"
|
||||
class="flex items-center gap-1.5 px-3 py-1.5 rounded-md text-sm font-medium transition-colors
|
||||
{% if request.endpoint == endpoint %}
|
||||
bg-accent/10 text-accent dark:bg-accent-dark/15 dark:text-accent-dark
|
||||
{% else %}
|
||||
text-muted dark:text-muted-dark hover:text-ink dark:hover:text-ink-dark hover:bg-paper dark:hover:bg-paper-dark
|
||||
{% endif %}">
|
||||
<svg class="w-4 h-4 shrink-0"><use href="/static/dist/icons.svg#{{ icon }}"/></svg>
|
||||
<span class="hidden sm:inline">{{ label }}</span>
|
||||
</a>
|
||||
{% endfor %}
|
||||
|
||||
<div class="ml-auto flex items-center gap-2">
|
||||
<button type="button" onclick="toggleTheme()"
|
||||
aria-label="Toggle dark mode"
|
||||
class="p-2 rounded-md text-muted dark:text-muted-dark hover:text-ink dark:hover:text-ink-dark hover:bg-paper dark:hover:bg-paper-dark transition-colors">
|
||||
<svg class="w-4 h-4 block dark:hidden"><use href="/static/dist/icons.svg#moon"/></svg>
|
||||
<svg class="w-4 h-4 hidden dark:block"><use href="/static/dist/icons.svg#sun"/></svg>
|
||||
</button>
|
||||
<span class="text-sm text-muted dark:text-muted-dark font-data hidden md:inline">{{ current_user.email }}</span>
|
||||
<form method="post" action="{{ url_for('auth.logout') }}">
|
||||
<input type="hidden" name="csrf_token" value="{{ csrf_token() }}">
|
||||
<button type="submit" title="Log out"
|
||||
class="p-2 rounded-md text-muted dark:text-muted-dark hover:text-danger dark:hover:text-danger-dark hover:bg-paper dark:hover:bg-paper-dark transition-colors">
|
||||
<svg class="w-4 h-4"><use href="/static/dist/icons.svg#log-out"/></svg>
|
||||
</button>
|
||||
</form>
|
||||
</div>
|
||||
{% else %}
|
||||
<div class="ml-auto">
|
||||
<button type="button" onclick="toggleTheme()" aria-label="Toggle dark mode"
|
||||
class="p-2 rounded-md text-muted dark:text-muted-dark hover:text-ink dark:hover:text-ink-dark hover:bg-paper dark:hover:bg-paper-dark transition-colors">
|
||||
<svg class="w-4 h-4 block dark:hidden"><use href="/static/dist/icons.svg#moon"/></svg>
|
||||
<svg class="w-4 h-4 hidden dark:block"><use href="/static/dist/icons.svg#sun"/></svg>
|
||||
</button>
|
||||
</div>
|
||||
{% endif %}
|
||||
</nav>
|
||||
</header>
|
||||
|
||||
<!-- HTMX target swapped by the nav above (Ch01: single-page shell, no full reloads) -->
|
||||
<main id="main-content" class="max-w-6xl mx-auto p-4 sm:p-6">
|
||||
{% block content %}{% endblock %}
|
||||
</main>
|
||||
</div>
|
||||
|
||||
<script type="module" src="{{ asset('app/static/src/js/main.js') }}"></script>
|
||||
</body>
|
||||
</html>
|
||||
@@ -0,0 +1,39 @@
|
||||
"""Vite manifest reader — maps a source entry to its hashed dist path (Ch05)."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
from flask import Flask
|
||||
|
||||
|
||||
class ManifestNotFound(RuntimeError):
|
||||
"""app/static/dist/manifest.json is missing — `npm run build` hasn't been run."""
|
||||
|
||||
|
||||
def _load_manifest(static_dist_dir: Path) -> dict:
|
||||
for candidate in (static_dist_dir / ".vite" / "manifest.json", static_dist_dir / "manifest.json"):
|
||||
if candidate.exists():
|
||||
return json.loads(candidate.read_text())
|
||||
raise ManifestNotFound(f"No Vite manifest under {static_dist_dir} — run `npm run build`.")
|
||||
|
||||
|
||||
def register_asset_helper(app: Flask) -> None:
|
||||
"""Register `asset(name)` as a Jinja global (Chapter 05)."""
|
||||
static_dist_dir = Path(app.static_folder) / "dist"
|
||||
|
||||
@app.context_processor
|
||||
def inject_asset_helper():
|
||||
def asset(name: str) -> str:
|
||||
try:
|
||||
manifest = _load_manifest(static_dist_dir)
|
||||
except ManifestNotFound:
|
||||
if app.debug:
|
||||
return f"/static/dist/{name}" # tolerate an unbuilt dev checkout
|
||||
raise
|
||||
entry = manifest.get(name)
|
||||
if entry is None:
|
||||
raise KeyError(f"{name!r} not in Vite manifest.")
|
||||
return f"/static/dist/{entry['file']}"
|
||||
|
||||
return {"asset": asset}
|
||||
@@ -0,0 +1,32 @@
|
||||
"""Shared from/to date-range parsing (Chapter 11: consistent across every
|
||||
chart/KPI endpoint; never defaults to "all time" per Chapter 03 rule 7).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import date, datetime, timedelta
|
||||
|
||||
from flask import Request
|
||||
|
||||
DEFAULT_RANGE_DAYS = 7 # within Ch11's stated "7 or 30 day" default allowance
|
||||
|
||||
|
||||
def parse_date_range(request: Request) -> tuple[date, date]:
|
||||
"""Parse `?from=&to=` (ISO 8601 dates), defaulting to the last 7 days."""
|
||||
to_raw = request.args.get("to")
|
||||
from_raw = request.args.get("from")
|
||||
to_date = date.fromisoformat(to_raw) if to_raw else date.today()
|
||||
from_date = date.fromisoformat(from_raw) if from_raw else to_date - timedelta(days=DEFAULT_RANGE_DAYS - 1)
|
||||
if from_date > to_date:
|
||||
from_date, to_date = to_date, from_date
|
||||
return from_date, to_date
|
||||
|
||||
|
||||
def day_bounds(from_date: date, to_date: date) -> tuple[datetime, datetime]:
|
||||
"""Inclusive [from_date, to_date] -> half-open [start, end) datetime range.
|
||||
|
||||
Shared by overview/queries.py, seo/queries.py, and security/queries.py
|
||||
(consolidated here rather than each keeping its own local copy).
|
||||
"""
|
||||
start = datetime.combine(from_date, datetime.min.time())
|
||||
end = datetime.combine(to_date, datetime.min.time()) + timedelta(days=1)
|
||||
return start, end
|
||||
@@ -0,0 +1,18 @@
|
||||
"""Shared JSON envelope helpers (Chapter 11).
|
||||
|
||||
Every existing /api/... endpoint already hand-builds this exact shape via
|
||||
jsonify(data=..., meta=...) — audited for compliance in docs/api-contract-
|
||||
final.md. This module exists so NEW endpoints (e.g. this chapter's
|
||||
unauthorized_handler) have one call site instead of re-typing the shape.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from flask import jsonify
|
||||
|
||||
|
||||
def api_ok(data, meta: dict | None = None):
|
||||
return jsonify(data=data, meta=meta or {})
|
||||
|
||||
|
||||
def api_error(code: str, message: str, status: int):
|
||||
return jsonify(error={"code": code, "message": message}), status
|
||||
@@ -0,0 +1,20 @@
|
||||
"""Shared HTMX full-page-vs-fragment response helper (Chapter 04).
|
||||
|
||||
Every blueprint serving a dashboard tab (Ch08/09/10) uses this instead of
|
||||
duplicating the `if request.headers.get("HX-Request")` check, so the
|
||||
convention is identical everywhere.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from flask import Request, render_template
|
||||
|
||||
|
||||
def is_htmx(request: Request) -> bool:
|
||||
"""True if this request was triggered by an hx-* attribute."""
|
||||
return request.headers.get("HX-Request", "").lower() == "true"
|
||||
|
||||
|
||||
def render_htmx_aware(request: Request, *, full_template: str, partial_template: str, **ctx):
|
||||
"""Render `full_template` on first load, `partial_template` on HTMX swaps."""
|
||||
template = partial_template if is_htmx(request) else full_template
|
||||
return render_template(template, **ctx)
|
||||
@@ -0,0 +1,14 @@
|
||||
"""Shared HTTP status-code bucketing — used by the aggregator (per-IP
|
||||
rollup) and available for any endpoint needing the same 2xx/3xx/4xx/5xx
|
||||
buckets, so there's one bucketing rule, not several copies."""
|
||||
from __future__ import annotations
|
||||
|
||||
|
||||
def status_bucket(status_code: int) -> str:
|
||||
if status_code < 300:
|
||||
return "2xx"
|
||||
if status_code < 400:
|
||||
return "3xx"
|
||||
if status_code < 500:
|
||||
return "4xx"
|
||||
return "5xx"
|
||||
@@ -0,0 +1,14 @@
|
||||
"""Shared page/per_page parsing (Chapter 11: consistent across every
|
||||
Grid.js-backed endpoint)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from flask import Request
|
||||
|
||||
DEFAULT_PER_PAGE = 20
|
||||
MAX_PER_PAGE = 100
|
||||
|
||||
|
||||
def parse_pagination(request: Request) -> tuple[int, int]:
|
||||
page = max(1, request.args.get("page", 1, type=int))
|
||||
per_page = request.args.get("per_page", DEFAULT_PER_PAGE, type=int)
|
||||
return page, max(1, min(per_page, MAX_PER_PAGE))
|
||||
@@ -0,0 +1,20 @@
|
||||
"""Shared upload-path helper (Chapter 04) — used by the uploads blueprint,
|
||||
the optional CLI processor, and the automatic background-thread processor,
|
||||
so there's one definition of where a given LogFile's raw upload lives on
|
||||
disk. Pulled out of the uploads blueprint so services/ doesn't have to
|
||||
import from blueprints/ (the wrong direction) to reach it.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
from flask import current_app
|
||||
|
||||
from app.models.log_file import LogFile
|
||||
|
||||
|
||||
def upload_path_for(log_file: LogFile) -> Path:
|
||||
"""Deterministic on-disk path for a LogFile row — avoids a new DB column."""
|
||||
ext = Path(log_file.filename).suffix.lower()
|
||||
upload_dir = Path(current_app.config["UPLOAD_DIR"])
|
||||
return upload_dir / f"{log_file.id}{ext}"
|
||||
@@ -0,0 +1,105 @@
|
||||
# Chapter 11 — Final API Contract (post Ch08–12 additions)
|
||||
|
||||
Consolidated per Chapter 11's own instruction to update/extend the route
|
||||
table as new routes are introduced elsewhere, flagging additions explicitly.
|
||||
|
||||
## Additions beyond the original Chapter 11 table
|
||||
|
||||
| Method | Path | Blueprint | Purpose | Response | Added in |
|
||||
|---|---|---|---|---|---|
|
||||
| GET | `/api/overview/browser-breakdown` | overview | Human-only browser/OS breakdown | JSON | Ch08 (Method A) |
|
||||
| GET | `/` | (root) | Redirect to `/overview` if authenticated, else `/login` | redirect | Ch12 follow-up |
|
||||
| GET | `/api/uploads` | uploads | List uploaded files, paginated | JSON (Grid.js) | Ch12 follow-up |
|
||||
| DELETE | `/api/uploads` | uploads | Bulk-delete uploaded files + derived data | JSON | Ch12 follow-up |
|
||||
|
||||
`/login`, `/logout`, and `/healthz` were already listed in the original
|
||||
table (Ch02/Ch04) and are now actually implemented (Ch12) — no table
|
||||
change needed, just noting they're no longer placeholders. One narrowing:
|
||||
`/logout` is implemented as POST-only, not GET/POST, so a state-changing
|
||||
action isn't reachable via a plain GET (CSRF-safety; see
|
||||
`app/blueprints/auth/routes.py`).
|
||||
|
||||
## Architecture note: rollup recompute is now delete-safe (Ch12 follow-up)
|
||||
|
||||
Deleting an uploaded file (`app/services/file_deletion.py`) removes its
|
||||
`log_entries`/`bot_hits`/`suspicious_events` rows explicitly (this
|
||||
project's SQLite connections don't have `PRAGMA foreign_keys=ON`, so the
|
||||
`ondelete="CASCADE"` in the migrations is documentation, not enforced
|
||||
behavior) and recomputes rollups for whatever dates the file touched.
|
||||
That recompute exposed a real bug in `app/services/aggregator.py`: three
|
||||
of its five writers only ever upserted-when-data-present and silently
|
||||
left stale rows behind when a day's data disappeared entirely — never
|
||||
reachable before deletion existed (rollups only ever grew). All five
|
||||
writers now use delete-then-insert consistently. See the module
|
||||
docstrings in `aggregator.py` and `file_deletion.py` for the full
|
||||
reasoning, including the deliberately-out-of-scope limitation around
|
||||
`ip_registry`/`blocklist_suggestions` not being recomputed on delete.
|
||||
|
||||
## Architecture note: cron is now optional (Ch12 follow-up)
|
||||
|
||||
`POST /uploads` now triggers processing automatically in a background
|
||||
thread (`app/services/background.py`), and `GET /overview` opportunistically
|
||||
resumes any file left incomplete. `flask process-logs` and `flask cleanup`
|
||||
(`app/cli.py`) still exist and work identically to before for anyone who
|
||||
wants a cron-based fallback, but nothing in the app requires it anymore.
|
||||
See the module docstring in `app/services/background.py` for the explicit
|
||||
tradeoff this introduces relative to the original Chapter 02/03 "no
|
||||
persistent background workers, cron-triggered CLI only" stance.
|
||||
|
||||
## Full current route table
|
||||
|
||||
| Method | Path | Blueprint | Purpose | Response | Auth |
|
||||
|---|---|---|---|---|---|
|
||||
| GET | `/overview` | overview | Overview tab | full page / HTMX partial | required |
|
||||
| GET | `/api/overview/kpis` | overview | KPI card values | JSON | required |
|
||||
| GET | `/api/overview/traffic-chart` | overview | Traffic-over-time series | JSON | required |
|
||||
| GET | `/api/overview/status-codes` | overview | Status-code breakdown | JSON | required |
|
||||
| GET | `/api/overview/top-urls` | overview | Top URLs table | JSON (Grid.js) | required |
|
||||
| GET | `/api/overview/top-referrers` | overview | Top referrers table | JSON (Grid.js) | required |
|
||||
| GET | `/api/overview/browser-breakdown` | overview | Human browser/OS breakdown | JSON | required |
|
||||
| GET | `/seo` | seo | SEO tab | full page / HTMX partial | required |
|
||||
| GET | `/api/seo/bot-summary` | seo | Bot summary cards | JSON | required |
|
||||
| GET | `/api/seo/crawl-chart-data` | seo | Crawl frequency series | JSON | required |
|
||||
| GET | `/api/seo/bot-status-codes` | seo | Status codes served to bots | JSON | required |
|
||||
| GET | `/api/seo/crawled-vs-visited` | seo | Bot vs. human URL comparison | JSON (Grid.js) | required |
|
||||
| GET | `/security` | security | Security tab | full page / HTMX partial | required |
|
||||
| GET | `/api/security/events` | security | Suspicious events table | JSON (Grid.js) | required |
|
||||
| GET | `/api/security/sensitive-paths` | security | Sensitive-path summary | JSON | required |
|
||||
| GET | `/api/security/ip/<ip>` | security | IP history drill-down | HTMX fragment | required |
|
||||
| GET | `/api/security/export-blocklist` | security | Blocklist export | text/plain download | required |
|
||||
| POST | `/uploads` | uploads | Upload a log file | HTMX fragment (queued state) | required |
|
||||
| GET | `/api/uploads/<id>/status` | uploads | Poll parse status | JSON or HTMX fragment | required |
|
||||
| GET | `/api/uploads` | uploads | List uploaded files | JSON (Grid.js) | required |
|
||||
| DELETE | `/api/uploads` | uploads | Bulk-delete uploaded files | JSON | required |
|
||||
| GET | `/login` | auth | Show login form | full page | public |
|
||||
| POST | `/login` | auth | Authenticate | redirect | public |
|
||||
| POST | `/logout` | auth | End session | redirect | required |
|
||||
| GET | `/healthz` | (root) | Liveness check | JSON | public |
|
||||
| GET | `/` | (root) | Auth-based redirect | redirect | public |
|
||||
|
||||
## Envelope compliance audit (Chapter 12)
|
||||
|
||||
Every `/api/...` JSON endpoint above was checked against Chapter 11's
|
||||
`{"data": ..., "meta": {...}}` / `{"error": {"code", "message"}}`
|
||||
convention. All conform. `app/utils/envelope.py` was added this chapter
|
||||
as a shared helper for *new* endpoints going forward (used by the
|
||||
`unauthorized_handler` in `app/__init__.py`) — existing endpoints already
|
||||
matched the shape by hand and weren't rewritten, to avoid churn on
|
||||
working code.
|
||||
|
||||
New this chapter: every `/api/...` path now returns a JSON `401` with
|
||||
`{"error": {"code": "unauthorized", ...}}` when unauthenticated, instead
|
||||
of Flask-Login's default redirect — consistent with the envelope even
|
||||
for auth failures. `/api/security/ip/<ip>` and `/api/security/export-
|
||||
blocklist` are the two exceptions where a *successful* response isn't
|
||||
JSON (HTMX fragment / text download, per the table above and Chapter 10)
|
||||
— their auth-failure response is still the JSON envelope for consistency.
|
||||
|
||||
## Auth model (Chapter 12)
|
||||
|
||||
Every blueprint except `auth` and the root `/healthz` route requires a
|
||||
logged-in session (`@bp.before_request` + `flask_login.login_required` in
|
||||
each blueprint's `__init__.py`). This closes a gap that existed from
|
||||
Chapter 04 through Chapter 10: all dashboard and `/api/...` routes were
|
||||
reachable without authentication until Chapter 12 wired in Flask-Login,
|
||||
even though Chapter 01 specifies "one authenticated single-page shell."
|
||||
@@ -0,0 +1 @@
|
||||
Single-database configuration for Flask-Migrate/Alembic.
|
||||
@@ -0,0 +1,44 @@
|
||||
# A generic, single database configuration.
|
||||
|
||||
[alembic]
|
||||
# template used to generate migration files
|
||||
# file_template = %%(rev)s_%%(slug)s
|
||||
|
||||
[loggers]
|
||||
keys = root,sqlalchemy,alembic,flask_migrate
|
||||
|
||||
[handlers]
|
||||
keys = console
|
||||
|
||||
[formatters]
|
||||
keys = generic
|
||||
|
||||
[logger_root]
|
||||
level = WARN
|
||||
handlers = console
|
||||
qualname =
|
||||
|
||||
[logger_sqlalchemy]
|
||||
level = WARN
|
||||
handlers =
|
||||
qualname = sqlalchemy.engine
|
||||
|
||||
[logger_alembic]
|
||||
level = INFO
|
||||
handlers =
|
||||
qualname = alembic
|
||||
|
||||
[logger_flask_migrate]
|
||||
level = INFO
|
||||
handlers =
|
||||
qualname = flask_migrate
|
||||
|
||||
[handler_console]
|
||||
class = StreamHandler
|
||||
args = (sys.stderr,)
|
||||
level = NOTSET
|
||||
formatter = generic
|
||||
|
||||
[formatter_generic]
|
||||
format = %(levelname)-5.5s [%(name)s] %(message)s
|
||||
datefmt = %H:%M:%S
|
||||
@@ -0,0 +1,78 @@
|
||||
import logging
|
||||
from logging.config import fileConfig
|
||||
|
||||
from flask import current_app
|
||||
|
||||
from alembic import context
|
||||
|
||||
# this is the Alembic Config object, which provides
|
||||
# access to the values within the .ini file in use.
|
||||
config = context.config
|
||||
|
||||
# Interpret the config file for Python logging.
|
||||
fileConfig(config.config_file_name)
|
||||
logger = logging.getLogger('alembic.env')
|
||||
|
||||
|
||||
def get_engine():
|
||||
try:
|
||||
# this works with Flask-SQLAlchemy<3 and Alchemical
|
||||
return current_app.extensions['migrate'].db.get_engine()
|
||||
except (TypeError, AttributeError):
|
||||
# this works with Flask-SQLAlchemy>=3
|
||||
return current_app.extensions['migrate'].db.engine
|
||||
|
||||
|
||||
def get_engine_url():
|
||||
try:
|
||||
return get_engine().url.render_as_string(hide_password=False).replace('%', '%%')
|
||||
except AttributeError:
|
||||
return str(get_engine().url).replace('%', '%%')
|
||||
|
||||
|
||||
config.set_main_option('sqlalchemy.url', get_engine_url())
|
||||
target_db = current_app.extensions['migrate'].db
|
||||
|
||||
|
||||
def get_metadata():
|
||||
if hasattr(target_db, 'metadatas'):
|
||||
return target_db.metadatas[None]
|
||||
return target_db.metadata
|
||||
|
||||
|
||||
def run_migrations_offline():
|
||||
"""Run migrations in 'offline' mode."""
|
||||
url = config.get_main_option("sqlalchemy.url")
|
||||
context.configure(url=url, target_metadata=get_metadata(), literal_binds=True)
|
||||
|
||||
with context.begin_transaction():
|
||||
context.run_migrations()
|
||||
|
||||
|
||||
def run_migrations_online():
|
||||
"""Run migrations in 'online' mode."""
|
||||
|
||||
def process_revision_directives(context, revision, directives):
|
||||
if getattr(config.cmd_opts, 'autogenerate', False):
|
||||
script = directives[0]
|
||||
if script.upgrade_ops.is_empty():
|
||||
directives[:] = []
|
||||
logger.info('No changes in schema detected.')
|
||||
|
||||
conf_args = current_app.extensions['migrate'].configure_args
|
||||
if conf_args.get("process_revision_directives") is None:
|
||||
conf_args["process_revision_directives"] = process_revision_directives
|
||||
|
||||
connectable = get_engine()
|
||||
|
||||
with connectable.connect() as connection:
|
||||
context.configure(connection=connection, target_metadata=get_metadata(), **conf_args)
|
||||
|
||||
with context.begin_transaction():
|
||||
context.run_migrations()
|
||||
|
||||
|
||||
if context.is_offline_mode():
|
||||
run_migrations_offline()
|
||||
else:
|
||||
run_migrations_online()
|
||||
@@ -0,0 +1,24 @@
|
||||
"""${message}
|
||||
|
||||
Revision ID: ${up_revision}
|
||||
Revises: ${down_revision | comma,n}
|
||||
Create Date: ${create_date}
|
||||
|
||||
"""
|
||||
from alembic import op
|
||||
import sqlalchemy as sa
|
||||
${imports if imports else ""}
|
||||
|
||||
# revision identifiers, used by Alembic.
|
||||
revision = ${repr(up_revision)}
|
||||
down_revision = ${repr(down_revision)}
|
||||
branch_labels = ${repr(branch_labels)}
|
||||
depends_on = ${repr(depends_on)}
|
||||
|
||||
|
||||
def upgrade():
|
||||
${upgrades if upgrades else "pass"}
|
||||
|
||||
|
||||
def downgrade():
|
||||
${downgrades if downgrades else "pass"}
|
||||
@@ -0,0 +1,126 @@
|
||||
"""Initial schema (Chapter 06): log_files, log_entries, request_stats_hourly/
|
||||
daily, bot_hits, ip_registry, suspicious_events, blocklist_suggestions.
|
||||
"""
|
||||
from alembic import op
|
||||
import sqlalchemy as sa
|
||||
|
||||
revision = "0001_initial_schema"
|
||||
down_revision = None
|
||||
branch_labels = None
|
||||
depends_on = None
|
||||
|
||||
|
||||
def upgrade():
|
||||
op.create_table(
|
||||
"log_files",
|
||||
sa.Column("id", sa.Integer, primary_key=True),
|
||||
sa.Column("filename", sa.String(255), nullable=False),
|
||||
sa.Column("server_type", sa.String(16), nullable=False),
|
||||
sa.Column("format_string", sa.Text, nullable=False),
|
||||
sa.Column("uploaded_at", sa.DateTime, nullable=False),
|
||||
sa.Column("status", sa.String(16), nullable=False, server_default="queued"),
|
||||
sa.Column("total_lines", sa.Integer, nullable=True),
|
||||
sa.Column("processed_lines", sa.Integer, nullable=False, server_default="0"),
|
||||
sa.Column("size_bytes", sa.BigInteger, nullable=False),
|
||||
sa.Column("checksum", sa.String(64), nullable=False),
|
||||
sa.Column("error_message", sa.Text, nullable=True),
|
||||
)
|
||||
op.create_index("ix_log_files_checksum", "log_files", ["checksum"])
|
||||
|
||||
op.create_table(
|
||||
"log_entries",
|
||||
sa.Column("id", sa.Integer, primary_key=True),
|
||||
sa.Column("log_file_id", sa.Integer, sa.ForeignKey("log_files.id", ondelete="CASCADE"), nullable=False),
|
||||
sa.Column("timestamp", sa.DateTime, nullable=False),
|
||||
sa.Column("ip", sa.String(45), nullable=False),
|
||||
sa.Column("method", sa.String(10), nullable=False),
|
||||
sa.Column("path", sa.Text, nullable=False),
|
||||
sa.Column("status_code", sa.SmallInteger, nullable=False),
|
||||
sa.Column("bytes_sent", sa.Integer, nullable=False, server_default="0"),
|
||||
sa.Column("referrer", sa.Text, nullable=True),
|
||||
sa.Column("user_agent", sa.Text, nullable=True),
|
||||
sa.Column("is_bot", sa.Boolean, nullable=False, server_default=sa.false()),
|
||||
sa.Column("flagged", sa.Boolean, nullable=False, server_default=sa.false()),
|
||||
)
|
||||
op.create_index("ix_log_entries_file_ts", "log_entries", ["log_file_id", "timestamp"])
|
||||
op.create_index("ix_log_entries_ip", "log_entries", ["ip"])
|
||||
op.create_index("ix_log_entries_path", "log_entries", ["path"])
|
||||
op.create_index("ix_log_entries_timestamp", "log_entries", ["timestamp"])
|
||||
|
||||
op.create_table(
|
||||
"request_stats_hourly",
|
||||
sa.Column("date_hour", sa.DateTime, primary_key=True),
|
||||
sa.Column("path", sa.String(2048), primary_key=True),
|
||||
sa.Column("status_code", sa.SmallInteger, primary_key=True),
|
||||
sa.Column("count", sa.Integer, nullable=False, server_default="0"),
|
||||
sa.Column("bytes_sent_sum", sa.BigInteger, nullable=False, server_default="0"),
|
||||
)
|
||||
|
||||
op.create_table(
|
||||
"request_stats_daily",
|
||||
sa.Column("date", sa.Date, primary_key=True),
|
||||
sa.Column("count", sa.Integer, nullable=False, server_default="0"),
|
||||
sa.Column("unique_ips", sa.Integer, nullable=False, server_default="0"),
|
||||
sa.Column("bytes_sum", sa.BigInteger, nullable=False, server_default="0"),
|
||||
sa.Column("error_count", sa.Integer, nullable=False, server_default="0"),
|
||||
)
|
||||
|
||||
op.create_table(
|
||||
"bot_hits",
|
||||
sa.Column("id", sa.Integer, primary_key=True),
|
||||
sa.Column("log_file_id", sa.Integer, sa.ForeignKey("log_files.id", ondelete="CASCADE"), nullable=False),
|
||||
sa.Column("timestamp", sa.DateTime, nullable=False),
|
||||
sa.Column("ip", sa.String(45), nullable=False),
|
||||
sa.Column("bot_name", sa.String(64), nullable=False),
|
||||
sa.Column("verified", sa.Boolean, nullable=False, server_default=sa.false()),
|
||||
sa.Column("path", sa.Text, nullable=False),
|
||||
sa.Column("status_code", sa.SmallInteger, nullable=False),
|
||||
)
|
||||
op.create_index("ix_bot_hits_timestamp", "bot_hits", ["timestamp"])
|
||||
op.create_index("ix_bot_hits_bot_name", "bot_hits", ["bot_name"])
|
||||
|
||||
op.create_table(
|
||||
"ip_registry",
|
||||
sa.Column("ip", sa.String(45), primary_key=True),
|
||||
sa.Column("first_seen", sa.DateTime, nullable=False),
|
||||
sa.Column("last_seen", sa.DateTime, nullable=False),
|
||||
sa.Column("total_requests", sa.Integer, nullable=False, server_default="0"),
|
||||
sa.Column("reputation_score", sa.Integer, nullable=False, server_default="0"),
|
||||
sa.Column("is_flagged", sa.Boolean, nullable=False, server_default=sa.false()),
|
||||
sa.Column("last_verified_at", sa.DateTime, nullable=True),
|
||||
)
|
||||
|
||||
op.create_table(
|
||||
"suspicious_events",
|
||||
sa.Column("id", sa.Integer, primary_key=True),
|
||||
sa.Column("log_file_id", sa.Integer, sa.ForeignKey("log_files.id", ondelete="CASCADE"), nullable=False),
|
||||
sa.Column("ip", sa.String(45), nullable=False),
|
||||
sa.Column("timestamp", sa.DateTime, nullable=False),
|
||||
sa.Column("path", sa.Text, nullable=False),
|
||||
sa.Column("rule_matched", sa.String(128), nullable=False),
|
||||
sa.Column("severity", sa.String(8), nullable=False),
|
||||
)
|
||||
op.create_index("ix_suspicious_events_timestamp", "suspicious_events", ["timestamp"])
|
||||
op.create_index("ix_suspicious_events_ip", "suspicious_events", ["ip"])
|
||||
op.create_index("ix_suspicious_events_severity", "suspicious_events", ["severity"])
|
||||
|
||||
op.create_table(
|
||||
"blocklist_suggestions",
|
||||
sa.Column("id", sa.Integer, primary_key=True),
|
||||
sa.Column("ip", sa.String(45), nullable=False),
|
||||
sa.Column("reason", sa.String(255), nullable=False),
|
||||
sa.Column("created_at", sa.DateTime, nullable=False),
|
||||
sa.Column("exported", sa.Boolean, nullable=False, server_default=sa.false()),
|
||||
)
|
||||
op.create_index("ix_blocklist_suggestions_ip", "blocklist_suggestions", ["ip"])
|
||||
|
||||
|
||||
def downgrade():
|
||||
op.drop_table("blocklist_suggestions")
|
||||
op.drop_table("suspicious_events")
|
||||
op.drop_table("ip_registry")
|
||||
op.drop_table("bot_hits")
|
||||
op.drop_table("request_stats_daily")
|
||||
op.drop_table("request_stats_hourly")
|
||||
op.drop_table("log_entries")
|
||||
op.drop_table("log_files")
|
||||
@@ -0,0 +1,16 @@
|
||||
"""Add ip_registry.last_verified_bot_result — flagged Ch07 schema deviation."""
|
||||
from alembic import op
|
||||
import sqlalchemy as sa
|
||||
|
||||
revision = "0002_ip_registry_verification_result"
|
||||
down_revision = "0001_initial_schema"
|
||||
branch_labels = None
|
||||
depends_on = None
|
||||
|
||||
|
||||
def upgrade():
|
||||
op.add_column("ip_registry", sa.Column("last_verified_bot_result", sa.Boolean, nullable=True))
|
||||
|
||||
|
||||
def downgrade():
|
||||
op.drop_column("ip_registry", "last_verified_bot_result")
|
||||
@@ -0,0 +1,29 @@
|
||||
"""Add referrer_stats_daily and browser_stats_daily (Method A, Ch08)."""
|
||||
from alembic import op
|
||||
import sqlalchemy as sa
|
||||
|
||||
revision = "0003_referrer_and_browser_stats"
|
||||
down_revision = "0002_ip_registry_verification_result"
|
||||
branch_labels = None
|
||||
depends_on = None
|
||||
|
||||
|
||||
def upgrade():
|
||||
op.create_table(
|
||||
"referrer_stats_daily",
|
||||
sa.Column("date", sa.Date, primary_key=True),
|
||||
sa.Column("referrer_domain", sa.String(255), primary_key=True),
|
||||
sa.Column("count", sa.Integer, nullable=False, server_default="0"),
|
||||
)
|
||||
op.create_table(
|
||||
"browser_stats_daily",
|
||||
sa.Column("date", sa.Date, primary_key=True),
|
||||
sa.Column("browser", sa.String(64), primary_key=True),
|
||||
sa.Column("os", sa.String(64), primary_key=True),
|
||||
sa.Column("count", sa.Integer, nullable=False, server_default="0"),
|
||||
)
|
||||
|
||||
|
||||
def downgrade():
|
||||
op.drop_table("browser_stats_daily")
|
||||
op.drop_table("referrer_stats_daily")
|
||||
@@ -0,0 +1,21 @@
|
||||
"""Add human_path_stats_daily (Method A, Ch09's most-crawled-vs-visited)."""
|
||||
from alembic import op
|
||||
import sqlalchemy as sa
|
||||
|
||||
revision = "0004_human_path_stats"
|
||||
down_revision = "0003_referrer_and_browser_stats"
|
||||
branch_labels = None
|
||||
depends_on = None
|
||||
|
||||
|
||||
def upgrade():
|
||||
op.create_table(
|
||||
"human_path_stats_daily",
|
||||
sa.Column("date", sa.Date, primary_key=True),
|
||||
sa.Column("path", sa.String(2048), primary_key=True),
|
||||
sa.Column("count", sa.Integer, nullable=False, server_default="0"),
|
||||
)
|
||||
|
||||
|
||||
def downgrade():
|
||||
op.drop_table("human_path_stats_daily")
|
||||
@@ -0,0 +1,34 @@
|
||||
"""Add ip_path_stats_daily and ip_status_stats_daily (bounded per-IP
|
||||
traffic rollup, explicit follow-up to Chapter 10)."""
|
||||
from alembic import op
|
||||
import sqlalchemy as sa
|
||||
|
||||
revision = "0005_ip_traffic_stats"
|
||||
down_revision = "0004_human_path_stats"
|
||||
branch_labels = None
|
||||
depends_on = None
|
||||
|
||||
|
||||
def upgrade():
|
||||
op.create_table(
|
||||
"ip_path_stats_daily",
|
||||
sa.Column("date", sa.Date, primary_key=True),
|
||||
sa.Column("ip", sa.String(45), primary_key=True),
|
||||
sa.Column("path", sa.String(2048), primary_key=True),
|
||||
sa.Column("count", sa.Integer, nullable=False, server_default="0"),
|
||||
)
|
||||
op.create_index("ix_ip_path_stats_ip", "ip_path_stats_daily", ["ip"])
|
||||
|
||||
op.create_table(
|
||||
"ip_status_stats_daily",
|
||||
sa.Column("date", sa.Date, primary_key=True),
|
||||
sa.Column("ip", sa.String(45), primary_key=True),
|
||||
sa.Column("status_bucket", sa.String(8), primary_key=True),
|
||||
sa.Column("count", sa.Integer, nullable=False, server_default="0"),
|
||||
)
|
||||
op.create_index("ix_ip_status_stats_ip", "ip_status_stats_daily", ["ip"])
|
||||
|
||||
|
||||
def downgrade():
|
||||
op.drop_table("ip_status_stats_daily")
|
||||
op.drop_table("ip_path_stats_daily")
|
||||
@@ -0,0 +1,23 @@
|
||||
"""Add users table (Chapter 12 single-admin auth)."""
|
||||
from alembic import op
|
||||
import sqlalchemy as sa
|
||||
|
||||
revision = "0006_users"
|
||||
down_revision = "0005_ip_traffic_stats"
|
||||
branch_labels = None
|
||||
depends_on = None
|
||||
|
||||
|
||||
def upgrade():
|
||||
op.create_table(
|
||||
"users",
|
||||
sa.Column("id", sa.Integer, primary_key=True),
|
||||
sa.Column("email", sa.String(255), nullable=False),
|
||||
sa.Column("password_hash", sa.String(255), nullable=False),
|
||||
sa.Column("created_at", sa.DateTime, nullable=False),
|
||||
)
|
||||
op.create_index("ix_users_email", "users", ["email"], unique=True)
|
||||
|
||||
|
||||
def downgrade():
|
||||
op.drop_table("users")
|
||||
Generated
+2235
File diff suppressed because it is too large
Load Diff
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user