perf(backend): close config/upload/export hot paths + stability fixes
Performance: - Cache the runtime `config` table in-memory (ConfigCache) with synchronous invalidation on every write (admin PATCH + test reseed). Was re-reading each key from Postgres on every request (~8 round-trips per upload). - Stream uploads chunk-by-chunk to a temp file instead of buffering the whole body in RAM (peak was up to the per-class cap, e.g. 500 MB/video); only 512 sniff-bytes are kept for magic-byte detection, then atomic rename into place. - Cache the media-filesystem disk snapshot (DiskCache, 15s TTL) shared by the quota check and admin stats; drop the discarded System::refresh_all(). - HTML export streams video (and small-image) originals straight into the ZIP via a manifest instead of copying them to a temp dir first (removed the transient 2x disk usage) and drops the double directory scan. - Auth extractor resolves session -> live user in one JOIN (was two queries), touching last_seen_at by token hash. Stability: - SSE: on broadcast lag, emit a `resync` event so the client runs a delta fetch instead of silently losing events; frontend reconciles adds, deletions, and (via an in-place refresh) like/comment counts on visible cards. - Storage quota fails OPEN when the disk can't be read (was a 0-byte limit that locked out all uploads). - Graceful shutdown drains in-flight requests on SIGTERM/SIGINT, bounded by a 10s backstop so open SSE streams can't stall a deploy. - Upload removes the persisted file if the DB transaction fails (no orphaned bytes with no row to reclaim them). Tests: - New pure select_disk() with 5 unit tests (longest-prefix, fallbacks, fail-open). - New e2e export-video spec covering the HTML export's video-streaming branch. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
@@ -161,7 +161,69 @@ async fn main() -> Result<()> {
|
||||
|
||||
let listener = tokio::net::TcpListener::bind(("0.0.0.0", config.app_port)).await?;
|
||||
tracing::info!("listening on {}", listener.local_addr()?);
|
||||
axum::serve(listener, router).await?;
|
||||
axum::serve(listener, router)
|
||||
.with_graceful_shutdown(shutdown_signal())
|
||||
.await?;
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Hard cap on how long we wait for in-flight connections to drain after a shutdown
|
||||
/// signal. Uploads (streamed to disk) finish in well under this; the cap exists because
|
||||
/// long-lived SSE streams never end on their own and would otherwise keep the graceful
|
||||
/// drain — and thus the process — pending until the orchestrator force-kills it.
|
||||
const SHUTDOWN_GRACE: std::time::Duration = std::time::Duration::from_secs(10);
|
||||
|
||||
/// Resolves on SIGINT (Ctrl-C) or SIGTERM (container stop / deploy). Letting
|
||||
/// `axum::serve` drain in-flight requests on this signal means a redeploy no longer
|
||||
/// truncates uploads mid-flight; any background compression/export half-states that a
|
||||
/// hard kill would leave are already reconciled by `startup_recovery` on the next boot.
|
||||
///
|
||||
/// Once the signal fires we also arm a detached backstop that force-exits after
|
||||
/// [`SHUTDOWN_GRACE`]. Without it, open SSE streams (which have no natural end) would
|
||||
/// hold the graceful drain open indefinitely; the backstop bounds shutdown regardless
|
||||
/// of the orchestrator's own kill timeout. If the drain completes first, `main` returns
|
||||
/// and the process exits before the timer ever fires.
|
||||
async fn shutdown_signal() {
|
||||
let ctrl_c = async {
|
||||
let _ = tokio::signal::ctrl_c().await;
|
||||
};
|
||||
|
||||
#[cfg(unix)]
|
||||
let terminate = async {
|
||||
match tokio::signal::unix::signal(tokio::signal::unix::SignalKind::terminate()) {
|
||||
Ok(mut sig) => {
|
||||
sig.recv().await;
|
||||
}
|
||||
Err(e) => {
|
||||
tracing::warn!(error = ?e, "failed to install SIGTERM handler");
|
||||
// Never resolve — fall back to ctrl_c only.
|
||||
std::future::pending::<()>().await;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
#[cfg(not(unix))]
|
||||
let terminate = std::future::pending::<()>();
|
||||
|
||||
tokio::select! {
|
||||
_ = ctrl_c => {}
|
||||
_ = terminate => {}
|
||||
}
|
||||
|
||||
tracing::info!(
|
||||
"shutdown signal received, draining in-flight requests (max {}s)",
|
||||
SHUTDOWN_GRACE.as_secs()
|
||||
);
|
||||
|
||||
// Backstop: if the graceful drain is still blocked after the grace window (almost
|
||||
// always because SSE streams are still open), exit anyway so deploys aren't stalled.
|
||||
tokio::spawn(async {
|
||||
tokio::time::sleep(SHUTDOWN_GRACE).await;
|
||||
tracing::warn!(
|
||||
"graceful drain exceeded {}s (likely open SSE streams); forcing exit",
|
||||
SHUTDOWN_GRACE.as_secs()
|
||||
);
|
||||
std::process::exit(0);
|
||||
});
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user