Files
EventSnap/backend/src/handlers/feed.rs
Fabian Hamm (Privat) 496dba5a1f fix(feed): filter on the server, exactly, and keep banned uploads out of the chips
Filtering was split across two independent client-side states and applied to whatever
page 1 happened to hold, by caption SUBSTRING. So a tag chip selected in the list view
was silently still applied in the grid without being shown; a filter matched photos
whose caption merely contained the text; and anything past the first page was invisible
to it. Verified against the seeded data: `hashtag=tanz` returned 6 photos by substring,
1 by tag.

`FeedQuery` now carries `hashtag` (single, list view), `hashtags` (CSV, OR'd, grid
chips) and `uploader` (exact, AND'd), normalised through one function that trims,
strips `#`, lowercases and dedupes, and yields None when empty — so an empty filter
means "no filter", never "match nothing". The two SQL branches collapse into one with
`h.tag = ANY($4)`. Tag-OR plus tag+user-AND is a specified feature, not an accident:
`e2e/specs/03-feed/filter-search.spec.ts` and USER_JOURNEYS §8 pin it, which is why the
semantics moved to the server rather than being simplified away.

Tags travel as CSV safely because the backend restricts them to ASCII alphanumerics and
`_`; `uploader` stays a single exact parameter because a display name can contain a
comma.

New `GET /api/v1/uploaders` reads `v_feed`, so banned and hidden uploaders are excluded
for free.

Migration 021 gives `v_hashtag_counts` the same treatment. It counted every upload
regardless of the uploader's ban state, so banning a guest left their tags in the chip
list as ghost filters that lead to an empty feed. Verified: after banning the guest who
owned all six `tanz*` photos, the chips went 6 -> 0.

`?limit=-5` returned a 500 — only the upper bound was clamped, so Postgres was asked
for `LIMIT -4`. Clamped at both ends.

`is_banned` is added to `/me/context` so the client can show a read-only notice instead
of letting a banned guest discover the ban one 403 toast at a time. `add_comment` sorts
and dedupes hashtags on the normalised key, matching the upload path — the two disagreed,
which is a lock-ordering deadlock between concurrent upserts.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-03 18:35:23 +02:00

491 lines
19 KiB
Rust

use std::time::Duration;
use axum::Json;
use axum::extract::{Query, State};
use chrono::{DateTime, Utc};
use serde::{Deserialize, Serialize};
use uuid::Uuid;
use crate::auth::middleware::AuthUser;
use crate::error::AppError;
use crate::services::config;
use crate::state::AppState;
#[derive(Deserialize)]
pub struct FeedQuery {
pub cursor: Option<Uuid>,
pub limit: Option<i64>,
/// Single tag (list view). Kept alongside `hashtags` so existing callers keep working.
pub hashtag: Option<String>,
/// Comma-separated tags, combined with **OR** — the grid's chip semantics
/// (USER_JOURNEYS §8). Filtering moved server-side because the client could only ever
/// filter the pages it had already loaded: with page 1 = 20 items out of a 1000-photo
/// event, selecting a tag showed a handful of tiles and looked complete. Matching is
/// now the exact `hashtag` row in both views, so the grid and the list can no longer
/// disagree about which photos carry a tag (the client matched a caption SUBSTRING, so
/// `#tanz` also matched `#tanzflaeche`).
pub hashtags: Option<String>,
/// Exact uploader display name, combined with the tag group using **AND**.
pub uploader: Option<String>,
}
/// Merge the single-tag and CSV tag params into one normalised, de-duplicated list.
///
/// Normalisation mirrors `Hashtag::upsert` exactly (trim, drop a leading `#`, lowercase), so
/// a chip built from a display string like `#Tanz` matches the stored `tanz` row. Returns
/// `None` when no usable tag was supplied, which makes the SQL predicate a no-op — an empty
/// list must mean "no tag filter", never "match nothing".
fn normalize_tags(single: Option<&str>, csv: Option<&str>) -> Option<Vec<String>> {
let mut out: Vec<String> = Vec::new();
let mut push = |raw: &str| {
let t = raw.trim().trim_start_matches('#').to_lowercase();
if !t.is_empty() && !out.contains(&t) {
out.push(t);
}
};
if let Some(s) = single {
push(s);
}
if let Some(s) = csv {
for part in s.split(',') {
push(part);
}
}
if out.is_empty() { None } else { Some(out) }
}
#[derive(Serialize)]
pub struct FeedUpload {
pub id: Uuid,
pub user_id: Uuid,
pub uploader_name: String,
pub preview_url: Option<String>,
pub thumbnail_url: Option<String>,
/// Big-screen (~2048px) variant for the diashow. Absent until the derivative exists.
pub display_url: Option<String>,
pub mime_type: String,
pub caption: Option<String>,
pub like_count: i64,
pub comment_count: i64,
pub liked_by_me: bool,
pub created_at: DateTime<Utc>,
}
#[derive(Serialize)]
pub struct FeedResponse {
pub uploads: Vec<FeedUpload>,
pub next_cursor: Option<Uuid>,
}
#[derive(sqlx::FromRow)]
struct FeedRow {
id: Uuid,
user_id: Uuid,
uploader_name: String,
preview_path: Option<String>,
thumbnail_path: Option<String>,
display_path: Option<String>,
mime_type: String,
caption: Option<String>,
like_count: i64,
comment_count: i64,
created_at: DateTime<Utc>,
}
pub async fn feed(
State(state): State<AppState>,
auth: AuthUser,
Query(q): Query<FeedQuery>,
) -> Result<Json<FeedResponse>, AppError> {
let rate_limits_on = config::get_bool(&state.config_cache, "rate_limits_enabled", true).await;
let feed_rate_on = config::get_bool(&state.config_cache, "feed_rate_enabled", true).await;
if rate_limits_on && feed_rate_on {
let rate_limit = config::get_usize(&state.config_cache, "feed_rate_per_min", 60).await;
// Keyed per-user, exactly like `feed_delta` below: at a venue every guest shares
// one public IP, so an IP key gave the whole party a single 60/min bucket and the
// fastest scroller starved everyone else.
if let Err(retry_after_secs) = state.rate_limiter.check_with_retry(
format!("feed:{}", auth.user_id),
rate_limit,
Duration::from_secs(60),
) {
return Err(AppError::TooManyRequests(
"Zu viele Anfragen. Bitte warte kurz und versuche es erneut.".into(),
Some(retry_after_secs),
));
}
}
// Clamped at BOTH ends: only the upper bound was enforced, so `?limit=-5` reached Postgres
// as `LIMIT -4` and answered a hand-written URL with a 500.
let limit = q.limit.unwrap_or(20).clamp(1, 100);
// Resolve the cursor to a (created_at, id) position. The pair is compared as a
// tuple so ties on created_at break on id — keyset pagination on created_at
// alone would silently drop rows sharing a timestamp across a page boundary.
let (cursor_time, cursor_id) = match q.cursor {
Some(c) => match get_cursor_pos(&state.pool, c).await {
Some((t, id)) => (Some(t), Some(id)),
None => (None, None),
},
None => (None, None),
};
// Tags from either param, normalised the same way `Hashtag::upsert` stores them
// (trimmed, leading `#` dropped, lowercased) so the comparison is exact.
let tags = normalize_tags(q.hashtag.as_deref(), q.hashtags.as_deref());
let uploader = q
.uploader
.as_deref()
.map(str::trim)
.filter(|s| !s.is_empty());
// ONE statement for every combination, rather than a branch per filter. `EXISTS` with
// `= ANY($4)` gives OR across the tag group without the row multiplication a JOIN would
// cause when a photo carries two selected tags; the uploader predicate ANDs on top. Both
// are no-ops when NULL, so the unfiltered feed takes the same path.
let rows = sqlx::query_as::<_, FeedRow>(
"SELECT v.id, v.user_id, v.uploader_name, v.preview_path, v.thumbnail_path,
v.display_path, v.mime_type, v.caption, v.like_count, v.comment_count,
v.created_at
FROM v_feed v
WHERE v.event_id = $1
AND ($2::timestamptz IS NULL OR (v.created_at, v.id) < ($2, $3))
AND ($4::text[] IS NULL OR EXISTS (
SELECT 1 FROM upload_hashtag uh
JOIN hashtag h ON h.id = uh.hashtag_id
WHERE uh.upload_id = v.id AND h.tag = ANY($4)))
AND ($5::text IS NULL OR v.uploader_name = $5)
ORDER BY v.created_at DESC, v.id DESC
LIMIT $6",
)
.bind(auth.event_id)
.bind(cursor_time)
.bind(cursor_id)
.bind(tags.as_deref())
.bind(uploader)
.bind(limit + 1)
.fetch_all(&state.pool)
.await?;
let has_more = rows.len() as i64 > limit;
let rows: Vec<FeedRow> = rows.into_iter().take(limit as usize).collect();
let next_cursor = if has_more {
rows.last().map(|r| r.id)
} else {
None
};
// Batch check which uploads the current user has liked
let upload_ids: Vec<Uuid> = rows.iter().map(|r| r.id).collect();
let liked_set = get_liked_set(&state.pool, auth.user_id, &upload_ids).await;
let uploads = rows
.into_iter()
.map(|r| {
// Gated media aliases (visibility-checked, direct /media blocked). Emit the
// URL only when the variant actually exists — the URL is what signals the
// client which variant to load.
let preview_url = r
.preview_path
.as_ref()
.map(|_| format!("/api/v1/upload/{}/preview", r.id));
let thumbnail_url = r
.thumbnail_path
.as_ref()
.map(|_| format!("/api/v1/upload/{}/thumbnail", r.id));
let display_url = r
.display_path
.as_ref()
.map(|_| format!("/api/v1/upload/{}/display", r.id));
FeedUpload {
liked_by_me: liked_set.contains(&r.id),
id: r.id,
user_id: r.user_id,
uploader_name: r.uploader_name,
preview_url,
thumbnail_url,
display_url,
mime_type: r.mime_type,
caption: r.caption,
like_count: r.like_count,
comment_count: r.comment_count,
created_at: r.created_at,
}
})
.collect();
Ok(Json(FeedResponse {
uploads,
next_cursor,
}))
}
#[derive(Deserialize)]
pub struct DeltaQuery {
pub since: DateTime<Utc>,
}
#[derive(Serialize)]
pub struct DeltaResponse {
pub uploads: Vec<FeedUpload>,
pub deleted_ids: Vec<Uuid>,
/// Users whose uploads became hidden (banned / uploads_hidden) since `since`. A ban is
/// not a soft-delete, so it never appears in `deleted_ids`; without this, a client that
/// missed the ephemeral `user-hidden` SSE (a reconnecting projector, most acutely) would
/// keep displaying the banned user's already-loaded slides. The client evicts every
/// upload from these users on receipt.
pub hidden_user_ids: Vec<Uuid>,
/// True when the upload query hit `DELTA_LIMIT`: the response carries only the
/// newest slice of the gap, so the client must fall back to a full feed refresh
/// rather than merging (the older missed uploads are absent and unrecoverable
/// via a later delta, which advances `since` past them).
pub truncated: bool,
/// The server's clock at the moment this delta was computed. The client advances its
/// reconnect cursor from THIS, never `new Date()` — a browser clock even seconds fast
/// would otherwise silently skip uploads whose server `created_at` falls in the skew
/// window (they'd never reappear without a hard refresh).
pub server_time: DateTime<Utc>,
}
pub async fn feed_delta(
State(state): State<AppState>,
auth: AuthUser,
Query(q): Query<DeltaQuery>,
) -> Result<Json<DeltaResponse>, AppError> {
// Rate-limit the delta the same way as the paginated feed. Without this, ~100 clients
// reconnecting at once (post-outage, or a flapping network) each fire an unbounded
// delta fetch — a reconnect stampede. Keyed per-user so one client can't starve others
// behind a shared NAT.
let rate_limits_on = config::get_bool(&state.config_cache, "rate_limits_enabled", true).await;
let feed_rate_on = config::get_bool(&state.config_cache, "feed_rate_enabled", true).await;
if rate_limits_on && feed_rate_on {
let rate_limit = config::get_usize(&state.config_cache, "feed_rate_per_min", 60).await;
if let Err(retry_after_secs) = state.rate_limiter.check_with_retry(
format!("feed_delta:{}", auth.user_id),
rate_limit,
Duration::from_secs(60),
) {
return Err(AppError::TooManyRequests(
"Zu viele Anfragen. Bitte warte kurz und versuche es erneut.".into(),
Some(retry_after_secs),
));
}
}
// Anchor the next cursor to the DB clock, captured *before* the queries so an upload
// committed during this handler is re-fetched next time rather than skipped (a
// duplicate id merges idempotently on the client; a miss is unrecoverable).
let server_time: DateTime<Utc> = sqlx::query_scalar("SELECT NOW()")
.fetch_one(&state.pool)
.await?;
// Bounded like the paginated feed: a stale `since` could otherwise pull the
// entire event's uploads in one response. If a client hits the cap it should
// fall back to a full feed refresh rather than another delta.
const DELTA_LIMIT: i64 = 200;
// `>= since` (not `>`): `created_at` is not unique, so a strict `>` anchored to the exact
// timestamp of the last-seen upload silently drops a SECOND upload committed in the same
// microsecond — an unrecoverable live-update miss. `>=` re-includes the boundary rows;
// the client merges by id (see the feed-delta handler's `seen` set), so the duplicate is
// harmless while the tied upload is no longer lost. The cursor still advances to the
// response's `server_time`, so this doesn't re-fetch on every subsequent delta.
let rows = sqlx::query_as::<_, FeedRow>(
"SELECT id, user_id, uploader_name, preview_path, thumbnail_path,
display_path, mime_type, caption, like_count, comment_count, created_at
FROM v_feed
WHERE event_id = $1 AND created_at >= $2
ORDER BY created_at DESC, id DESC
LIMIT $3",
)
.bind(auth.event_id)
.bind(q.since)
.bind(DELTA_LIMIT)
.fetch_all(&state.pool)
.await?;
// Hit the cap => this is only the newest slice of a larger gap. Signal the
// client to full-refresh instead of merging a partial delta.
let truncated = rows.len() as i64 >= DELTA_LIMIT;
// `>=` for the same tie-break reason as the uploads query above; re-signalling an
// already-removed id is idempotent on the client (it filters its list by these ids).
let deleted_ids: Vec<(Uuid,)> = sqlx::query_as(
"SELECT id FROM upload
WHERE event_id = $1 AND deleted_at IS NOT NULL AND deleted_at >= $2",
)
.bind(auth.event_id)
.bind(q.since)
.fetch_all(&state.pool)
.await?;
// Users hidden since the cursor (ban / uploads_hidden). Uncapped like `deleted_ids` and
// for the same reason — an eviction the client misses is unrecoverable via a later delta.
let hidden_user_ids: Vec<(Uuid,)> = sqlx::query_as(
"SELECT id FROM \"user\"
WHERE event_id = $1 AND uploads_hidden = TRUE
AND uploads_hidden_at IS NOT NULL AND uploads_hidden_at >= $2",
)
.bind(auth.event_id)
.bind(q.since)
.fetch_all(&state.pool)
.await?;
let upload_ids: Vec<Uuid> = rows.iter().map(|r| r.id).collect();
let liked_set = get_liked_set(&state.pool, auth.user_id, &upload_ids).await;
let uploads = rows
.into_iter()
.map(|r| FeedUpload {
liked_by_me: liked_set.contains(&r.id),
id: r.id,
user_id: r.user_id,
uploader_name: r.uploader_name,
preview_url: r
.preview_path
.as_ref()
.map(|_| format!("/api/v1/upload/{}/preview", r.id)),
thumbnail_url: r
.thumbnail_path
.as_ref()
.map(|_| format!("/api/v1/upload/{}/thumbnail", r.id)),
display_url: r
.display_path
.as_ref()
.map(|_| format!("/api/v1/upload/{}/display", r.id)),
mime_type: r.mime_type,
caption: r.caption,
like_count: r.like_count,
comment_count: r.comment_count,
created_at: r.created_at,
})
.collect();
Ok(Json(DeltaResponse {
uploads,
deleted_ids: deleted_ids.into_iter().map(|r| r.0).collect(),
hidden_user_ids: hidden_user_ids.into_iter().map(|r| r.0).collect(),
truncated,
server_time,
}))
}
#[derive(Serialize)]
pub struct HashtagCount {
pub tag: String,
pub count: i64,
}
pub async fn hashtags(
State(state): State<AppState>,
auth: AuthUser,
) -> Result<Json<Vec<HashtagCount>>, AppError> {
let rows: Vec<(String, i64)> =
sqlx::query_as("SELECT tag, upload_count FROM v_hashtag_counts WHERE event_id = $1")
.bind(auth.event_id)
.fetch_all(&state.pool)
.await?;
Ok(Json(
rows.into_iter()
.map(|(tag, count)| HashtagCount { tag, count })
.collect(),
))
}
/// Every uploader who has at least one visible upload, for the grid's "Nutzer suchen" picker.
///
/// The picker used to derive names from the uploads currently in memory — page 1, 20 items —
/// so typing a guest's name found nothing whenever their photos happened to sit below the
/// fold, which reads as "search is broken". This is the authoritative list.
///
/// Reads `v_feed`, so it inherits exactly the feed's visibility rules: soft-deleted uploads,
/// banned uploaders and hidden uploaders are all excluded, and a guest who has not uploaded
/// anything never appears. Uncapped on purpose — one short string per uploader, bounded by
/// the guest count, and truncating it would reintroduce the very bug this replaces.
/// Deliberately NOT the host-only `/host/users` route: that one lists every joined guest and
/// exposes moderation state.
pub async fn uploaders(
State(state): State<AppState>,
auth: AuthUser,
) -> Result<Json<Vec<String>>, AppError> {
let rows: Vec<(String,)> = sqlx::query_as(
"SELECT DISTINCT uploader_name FROM v_feed WHERE event_id = $1 ORDER BY uploader_name",
)
.bind(auth.event_id)
.fetch_all(&state.pool)
.await?;
Ok(Json(rows.into_iter().map(|(name,)| name).collect()))
}
/// Resolve a cursor id to its `(created_at, id)` position. Both are needed:
/// `created_at` alone isn't unique, so pagination must break ties on `id` to
/// avoid silently dropping rows that share a timestamp across a page boundary.
async fn get_cursor_pos(pool: &sqlx::PgPool, cursor_id: Uuid) -> Option<(DateTime<Utc>, Uuid)> {
let row: Option<(DateTime<Utc>, Uuid)> =
sqlx::query_as("SELECT created_at, id FROM upload WHERE id = $1")
.bind(cursor_id)
.fetch_optional(pool)
.await
.ok()?;
row
}
async fn get_liked_set(
pool: &sqlx::PgPool,
user_id: Uuid,
upload_ids: &[Uuid],
) -> std::collections::HashSet<Uuid> {
if upload_ids.is_empty() {
return std::collections::HashSet::new();
}
let rows: Vec<(Uuid,)> =
sqlx::query_as("SELECT upload_id FROM \"like\" WHERE user_id = $1 AND upload_id = ANY($2)")
.bind(user_id)
.bind(upload_ids)
.fetch_all(pool)
.await
.unwrap_or_default();
rows.into_iter().map(|r| r.0).collect()
}
#[cfg(test)]
mod tests {
use super::normalize_tags;
/// The chips carry display strings (`#Tanz`), the `hashtag` table stores `tanz`. If these
/// two drift the filter silently returns nothing, which is indistinguishable from "no
/// photos have this tag" — so pin the normalisation to `Hashtag::upsert`'s rule.
#[test]
fn tags_are_normalised_like_upsert_stores_them() {
assert_eq!(
normalize_tags(Some("#Tanz"), None),
Some(vec!["tanz".to_string()])
);
assert_eq!(
normalize_tags(None, Some(" #Buffet , reden ")),
Some(vec!["buffet".to_string(), "reden".to_string()])
);
}
/// An empty list must mean "no filter", never "match nothing" — returning `Some(vec![])`
/// would make `= ANY('{}')` false for every row and blank the feed.
#[test]
fn blank_input_disables_the_filter() {
assert_eq!(normalize_tags(None, None), None);
assert_eq!(normalize_tags(Some(" "), Some(" , ,#")), None);
}
/// Both params feed one list, de-duplicated: the list view sends `hashtag`, the grid sends
/// `hashtags`, and carrying a filter across views can legitimately set both to the same tag.
#[test]
fn single_and_csv_merge_without_duplicates() {
assert_eq!(
normalize_tags(Some("tanz"), Some("tanz,buffet")),
Some(vec!["tanz".to_string(), "buffet".to_string()])
);
}
}