Store: crawl generations as the sync cursor. Diffs compare generations
on entity identity plus a content digest, never on upstream timestamps —
GET /course-rooms/{id}/board returns request time as updatedAt for most
elements, so a timestamp cursor would report every board as changed on
every crawl. Identity diffing also yields deletions, which no timestamp
scheme can. A per-course crawl carries the other courses' rows forward
so every completed generation is a complete picture and any two diff
directly; without that a partial crawl reads as a mass deletion.
FTS uses the german dictionary with weighted title/body, plus a pg_trgm
arm because stemming will not match "Datenschutz" inside
"Datenschutzgrundverordnung" and German compounds make that the common
case. file_texts is keyed by file record id and deliberately outlives
generations: records are immutable upstream, so text extracted once is
valid forever and a re-crawl of unchanged content costs nothing.
Store.open returns undefined instead of throwing when Postgres is
unreachable — the index is an accelerator, and a Pi that loses its
database should get slower, not broken.
core/paths.ts is the security boundary for the mirror. Course titles,
card titles and filenames are all user-supplied upstream, so this is
where a hostile name stops being text and becomes a path. Two bugs found
by its own tests: "///" produced "---" instead of falling back, and dot
runs survived mid-component. Now no ".." can survive anywhere, which
makes the invariant checkable rather than a claim about ordering.
Indexer coalesces concurrent refreshes onto one run and enforces a
minimum interval, since a full crawl is ~270 requests from an account
that looks like a student.
9 store tests against a real Postgres (mocks would test nothing here)
and 13 path tests; 47 total.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
84 lines
3.2 KiB
TypeScript
84 lines
3.2 KiB
TypeScript
/**
|
|
* Runtime configuration, read once from the environment.
|
|
*
|
|
* The two Schulcloud values are named after the browser artefacts they come
|
|
* from (`TSC_URL`, `TSC_JWT_COOKIE`) so that copying a fresh token out of
|
|
* DevTools stays an obvious, mechanical step — see docs/AUTH.md.
|
|
*/
|
|
|
|
export interface Config {
|
|
/** Instance base URL, no trailing slash, e.g. `https://schulcloud-thueringen.de`. */
|
|
baseUrl: string;
|
|
/** Raw JWT from the instance's `jwt` cookie. Sent as `Authorization: Bearer`. */
|
|
jwt: string;
|
|
/** Shared secret callers must present to this MCP server. Unused in stdio mode. */
|
|
authToken: string | undefined;
|
|
port: number;
|
|
bindHost: string;
|
|
/** Hard ceiling on how many bytes `download_file` will pull from the instance. */
|
|
maxDownloadBytes: number;
|
|
/** Characters of extracted text returned before truncation kicks in. */
|
|
maxExtractedChars: number;
|
|
requestTimeoutMs: number;
|
|
/**
|
|
* How often to ping the instance to hold the session open. Must stay well
|
|
* under the instance's `JWT_TIMEOUT_SECONDS` (7200s here) — see
|
|
* src/core/keepalive.ts. Zero disables the keepalive.
|
|
*/
|
|
keepaliveIntervalMs: number;
|
|
|
|
/** Postgres for the search index and file mirror. Unset = live-only mode. */
|
|
databaseUrl: string | undefined;
|
|
/** Where mirrored file bytes live on disk. */
|
|
mirrorDir: string;
|
|
/** Files larger than this are indexed as metadata but not mirrored. */
|
|
mirrorMaxBytes: number;
|
|
/** How often to re-crawl on a timer. Zero = only on demand. */
|
|
crawlIntervalMs: number;
|
|
}
|
|
|
|
function required(name: string): string {
|
|
const value = process.env[name]?.trim();
|
|
if (!value) throw new Error(`Missing required environment variable ${name}`);
|
|
return value;
|
|
}
|
|
|
|
function int(name: string, fallback: number): number {
|
|
const raw = process.env[name]?.trim();
|
|
if (!raw) return fallback;
|
|
const parsed = Number.parseInt(raw, 10);
|
|
if (!Number.isFinite(parsed) || parsed <= 0) {
|
|
throw new Error(`Environment variable ${name} must be a positive integer, got ${raw}`);
|
|
}
|
|
return parsed;
|
|
}
|
|
|
|
/** Like `int`, but 0 is meaningful (it disables the feature) rather than invalid. */
|
|
function intAllowingZero(name: string, fallback: number): number {
|
|
const raw = process.env[name]?.trim();
|
|
if (!raw) return fallback;
|
|
const parsed = Number.parseInt(raw, 10);
|
|
if (!Number.isFinite(parsed) || parsed < 0) {
|
|
throw new Error(`Environment variable ${name} must be a non-negative integer, got ${raw}`);
|
|
}
|
|
return parsed;
|
|
}
|
|
|
|
export function loadConfig(): Config {
|
|
return {
|
|
baseUrl: required('TSC_URL').replace(/\/+$/, ''),
|
|
jwt: required('TSC_JWT_COOKIE'),
|
|
authToken: process.env.MCP_AUTH_TOKEN?.trim() || undefined,
|
|
port: int('PORT', 8080),
|
|
bindHost: process.env.BIND_HOST?.trim() || '0.0.0.0',
|
|
maxDownloadBytes: int('MAX_DOWNLOAD_BYTES', 25 * 1024 * 1024),
|
|
maxExtractedChars: int('MAX_EXTRACTED_CHARS', 120_000),
|
|
requestTimeoutMs: int('REQUEST_TIMEOUT_MS', 30_000),
|
|
keepaliveIntervalMs: intAllowingZero('KEEPALIVE_INTERVAL_MS', 30 * 60_000),
|
|
databaseUrl: process.env.DATABASE_URL?.trim() || undefined,
|
|
mirrorDir: process.env.MIRROR_DIR?.trim() || '/var/lib/schulcloud-mcp/mirror',
|
|
mirrorMaxBytes: int('MIRROR_MAX_BYTES', 64 * 1024 * 1024),
|
|
crawlIntervalMs: intAllowingZero('CRAWL_INTERVAL_MS', 6 * 60 * 60_000),
|
|
};
|
|
}
|