Files
Schulcloud-MCP/test/store.test.ts
MechaCat02 3e44e66dde Survive a first full crawl: poll, time out on silence, retry downloads
The first full crawl with the file manager ran 14 minutes, downloading every
file once, and broke in three ways:

- `schulcloud refresh` reported "fetch failed" for a crawl that was
  succeeding: Node's fetch abandons a response without headers after five
  minutes. POST /api/refresh takes wait:false and the CLI polls /api/status;
  refresh_index answers after 50 s and leaves the crawl running, and
  index_status says when a first crawl is under way.
- Downloads were bounded by the 30 s request timeout, which cut 11 MB scans
  off mid-transfer. They now time out on 30 s of silence instead.
- Failures were recorded once and never retried. A download failure is now
  retried on the next crawl while an extraction failure stays final, and PDF
  text containing NUL, which Postgres refuses, is stripped.

On the re-crawl all six failed files succeeded; only two videos above the
mirror cap stay metadata-only, by design. 137 tests.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-09-16 20:19:16 +02:00

222 lines
9.4 KiB
TypeScript

import assert from 'node:assert/strict';
import { after, before, describe, it } from 'node:test';
import { Store } from '../src/store/store.ts';
import type { Snapshot } from '../src/core/crawl.ts';
/**
* Exercises the store against a real Postgres — the generation/diff semantics
* are entirely SQL, so a mock would test nothing. Skipped when TEST_DATABASE_URL
* is unset so `npm test` stays offline by default.
*/
const DB_URL = process.env.TEST_DATABASE_URL;
/**
* These tests TRUNCATE. Pointing them at a real database destroys it — which
* happened once during development, when TEST_DATABASE_URL was aimed at the dev
* instance and the fixtures ended up in live data. Requiring "test" in the
* database name makes that mistake impossible to repeat by accident.
*/
function assertDisposable(url: string): void {
const name = new globalThis.URL(url).pathname.replace(/^\//, '');
if (!/test/i.test(name)) {
throw new Error(
`Refusing to run: TEST_DATABASE_URL points at database "${name}", which is not obviously ` +
`disposable. These tests TRUNCATE. Use a database with "test" in its name.`,
);
}
}
function snapshot(
courses: { id: string; title: string; boardText?: string; files?: { id: string; name: string; size: number }[] }[],
rooms: { id: string; name: string; boardText?: string }[] = [],
): Snapshot {
return {
crawledAt: new Date(),
schoolId: 'school1',
failures: [],
rooms: rooms.map((r) => ({
id: r.id,
name: r.name,
boards: r.boardText
? [{ id: `${r.id}-b`, title: 'Raum-Board', courseId: r.id, text: r.boardText,
board: { id: `${r.id}-b`, title: 'Raum-Board', columns: [], fileCount: 0 } }]
: [],
})),
courses: courses.map((c) => ({
course: { id: c.id, title: c.title, shortTitle: c.title.slice(0, 2), displayColor: '#000' },
title: c.title,
boards: c.boardText
? [{ id: `${c.id}-b`, title: 'Board', courseId: c.id, text: c.boardText,
board: { id: `${c.id}-b`, title: 'Board', columns: [], fileCount: 0 } }]
: [],
lessons: [],
tasks: [],
})),
files: courses.flatMap((c) =>
(c.files ?? []).map((f) => ({
record: { id: f.id, name: f.name, parentId: 'p', parentType: 'boardnodes' as const,
url: '', size: f.size, mimeType: 'application/pdf',
securityCheckStatus: 'verified', previewStatus: 'x' },
parentType: 'boardnodes' as const,
parentId: 'p',
at: { courseId: c.id, courseTitle: c.title, containerTitle: 'Board' },
})),
),
};
}
describe('Store', { skip: DB_URL ? false : 'set TEST_DATABASE_URL to run' }, () => {
let store: Store;
before(async () => {
assertDisposable(DB_URL!);
const opened = await Store.open(DB_URL);
assert.ok(opened, 'store should open');
store = opened;
// Start from a clean slate so generation ids are predictable.
await (store as never as { db: { query: (q: string) => Promise<unknown> } }).db.query(
'TRUNCATE crawls, file_texts RESTART IDENTITY CASCADE',
);
});
after(async () => {
await store?.close();
});
it('returns undefined rather than throwing when the database is unreachable', async () => {
const dead = await Store.open('postgresql://nobody@127.0.0.1:1/none');
assert.equal(dead, undefined);
});
it('saves a generation and reports stats', async () => {
const id = await store.saveSnapshot(snapshot([{ id: 'c1', title: 'Mathe', boardText: 'Bruchrechnung', files: [{ id: 'f1', name: 'a.pdf', size: 10 }] }]), 'full');
assert.ok(id > 0);
const stats = await store.stats();
assert.equal(stats.crawlId, id);
assert.equal(stats.files, 1);
});
it('diffs by identity, reporting additions and deletions', async () => {
const first = await store.latestCrawlId();
const second = await store.saveSnapshot(
snapshot([{ id: 'c1', title: 'Mathe', boardText: 'Bruchrechnung', files: [{ id: 'f2', name: 'b.pdf', size: 20 }] }]),
'full',
);
const diff = await store.diff(first!, second);
assert.ok(diff.added.some((n) => n.nodeId === 'f2'), 'f2 added');
assert.ok(diff.removed.some((n) => n.nodeId === 'f1'), 'f1 removed — timestamps could never show this');
});
it('reports a content change even when the id is unchanged', async () => {
const before = await store.latestCrawlId();
const after = await store.saveSnapshot(
snapshot([{ id: 'c1', title: 'Mathe', boardText: 'Prozentrechnung', files: [{ id: 'f2', name: 'b.pdf', size: 20 }] }]),
'full',
);
const diff = await store.diff(before!, after);
assert.ok(diff.changed.some((n) => n.nodeId === 'c1-b'), 'board body change detected via digest');
});
it('reports a renamed file, which keeps its id and size', async () => {
// `PATCH /file/rename/{id}` renames a record in place. The digest once
// covered only id and size on the assumption that file records never
// change, so a rename went unreported — the file just quietly appeared
// under a new name.
const before = await store.latestCrawlId();
const after = await store.saveSnapshot(
snapshot([{ id: 'c1', title: 'Mathe', boardText: 'Prozentrechnung', files: [{ id: 'f2', name: 'b-v2.pdf', size: 20 }] }]),
'full',
);
const diff = await store.diff(before!, after);
assert.ok(diff.changed.some((n) => n.nodeId === 'f2'), 'rename detected');
assert.ok(!diff.added.some((n) => n.nodeId === 'f2'), 'a rename is not a new file');
assert.ok(!diff.removed.some((n) => n.nodeId === 'f2'), 'and not a deleted one');
});
it('stores rooms alongside courses, and notices when one changes', async () => {
// Rooms are a separate space, not a kind of course: they must land in the
// index under their own kind, or room content becomes unsearchable the way
// submitted files once were.
const before = await store.saveSnapshot(
snapshot([{ id: 'c1', title: 'Mathe', boardText: 'Prozentrechnung' }], [{ id: 'r1', name: 'Projektraum', boardText: 'Projektsteuerung' }]),
'full',
);
const hits = await store.search('Projektsteuerung', { limit: 5 });
assert.ok(hits.some((h) => h.nodeId === 'r1-b'), 'a room board is searchable');
const after = await store.saveSnapshot(
snapshot([{ id: 'c1', title: 'Mathe', boardText: 'Prozentrechnung' }], [{ id: 'r1', name: 'Projektraum Informatik', boardText: 'Projektsteuerung' }]),
'full',
);
const diff = await store.diff(before, after);
assert.ok(diff.changed.some((n) => n.nodeId === 'r1' && n.kind === 'room'), 'a renamed room is reported as changed');
});
it('carries other courses forward on a per-course crawl', async () => {
await store.saveSnapshot(
snapshot([
{ id: 'c1', title: 'Mathe', boardText: 'Prozentrechnung' },
{ id: 'c2', title: 'Physik', boardText: 'Optik' },
]),
'full',
);
const before = await store.latestCrawlId();
// Re-crawl only c2; c1 must survive rather than looking deleted.
const after = await store.saveSnapshot(snapshot([{ id: 'c2', title: 'Physik', boardText: 'Mechanik' }]), 'c2');
const diff = await store.diff(before!, after);
assert.equal(diff.removed.length, 0, 'a partial crawl must not look like a mass deletion');
assert.ok(diff.changed.some((n) => n.nodeId === 'c2-b'));
});
it('finds German content with stemming', async () => {
const hits = await store.search('Mechanik');
assert.ok(hits.length > 0, 'expected a hit for Mechanik');
});
it('makes extracted file text searchable', async () => {
await store.saveSnapshot(snapshot([{ id: 'c3', title: 'Info', files: [{ id: 'f9', name: 'skript.pdf', size: 99 }] }]), 'full');
await store.recordFileText({
fileId: 'f9', name: 'skript.pdf', mimeType: 'application/pdf', size: 99,
content: 'Die Cäsar-Verschlüsselung verschiebt Buchstaben im Alphabet.',
note: 'ok', mirrorPath: 'Info/Board/skript.pdf', mirrorSize: 99,
});
const hits = await store.search('Verschlüsselung');
assert.ok(hits.some((h) => h.nodeId === 'f9'), 'PDF contents should be searchable, not just the filename');
});
it('stores extracted text containing NUL, which Postgres text refuses', async () => {
await store.recordFileText({
fileId: 'f9', name: 'skript.pdf', mimeType: 'application/pdf', size: 99,
content: 'Webserver\u0000 und Proxy', note: 'ok\u0000', mirrorPath: 'Info/Board/skript.pdf', mirrorSize: 99,
});
const hits = await store.search('Proxy');
assert.ok(hits.some((h) => h.nodeId === 'f9'), 'text with NUL bytes should still be stored and searchable');
});
it('keeps a file whose download failed queued for the next crawl, but not one that failed to extract', async () => {
await store.recordFileText({
fileId: 'f9', name: 'skript.pdf', mimeType: 'application/pdf', size: 99,
content: null, note: 'download failed, retried on the next crawl: timeout',
mirrorPath: null, mirrorSize: null, retry: true,
});
assert.ok((await store.filesNeedingText()).some((f) => f.fileId === 'f9'), 'a transient failure must be retried');
await store.recordFileText({
fileId: 'f9', name: 'skript.pdf', mimeType: 'application/pdf', size: 99,
content: null, note: 'extraction failed: bad xref', mirrorPath: null, mirrorSize: null,
});
assert.ok(!(await store.filesNeedingText()).some((f) => f.fileId === 'f9'), 'a parser failure is final');
});
it('resolves a timestamp cursor to a generation', async () => {
const id = await store.resolveCursor(new Date().toISOString());
assert.ok(id && id > 0);
assert.equal(await store.resolveCursor('not-a-date'), undefined);
});
it('builds a manifest with per-file change status', async () => {
const { entries } = await store.manifest();
assert.ok(entries.some((e) => e.fileId === 'f9' && e.path.startsWith('Info/')));
});
})