The first full crawl with the file manager ran 14 minutes, downloading every file once, and broke in three ways: - `schulcloud refresh` reported "fetch failed" for a crawl that was succeeding: Node's fetch abandons a response without headers after five minutes. POST /api/refresh takes wait:false and the CLI polls /api/status; refresh_index answers after 50 s and leaves the crawl running, and index_status says when a first crawl is under way. - Downloads were bounded by the 30 s request timeout, which cut 11 MB scans off mid-transfer. They now time out on 30 s of silence instead. - Failures were recorded once and never retried. A download failure is now retried on the next crawl while an extraction failure stays final, and PDF text containing NUL, which Postgres refuses, is stripped. On the re-crawl all six failed files succeeded; only two videos above the mirror cap stay metadata-only, by design. 137 tests. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
170 lines
5.8 KiB
TypeScript
170 lines
5.8 KiB
TypeScript
import assert from 'node:assert/strict';
|
|
import { afterEach, describe, it } from 'node:test';
|
|
import { SchulcloudClient, SchulcloudApiError } from '../src/core/client.ts';
|
|
|
|
/**
|
|
* Retry behaviour, driven by a stubbed fetch. Observed live: a full crawl made
|
|
* the instance answer 4 of 26 course pages with a 503 front-page, all of which
|
|
* succeeded on retry — so this is the difference between a complete index and a
|
|
* quietly incomplete one.
|
|
*/
|
|
const config = {
|
|
baseUrl: 'https://example.test',
|
|
jwt: 'x',
|
|
authToken: undefined,
|
|
port: 1,
|
|
bindHost: '127.0.0.1',
|
|
maxDownloadBytes: 1000,
|
|
maxExtractedChars: 1000,
|
|
requestTimeoutMs: 1000,
|
|
keepaliveIntervalMs: 0,
|
|
databaseUrl: undefined,
|
|
mirrorDir: '/tmp',
|
|
mirrorMaxBytes: 1000,
|
|
crawlIntervalMs: 0,
|
|
};
|
|
|
|
const realFetch = globalThis.fetch;
|
|
afterEach(() => {
|
|
globalThis.fetch = realFetch;
|
|
});
|
|
|
|
function stubFetch(responses: (Response | Error)[]): () => number {
|
|
let calls = 0;
|
|
globalThis.fetch = (async () => {
|
|
const next = responses[calls++] ?? responses.at(-1)!;
|
|
if (next instanceof Error) throw next;
|
|
return next;
|
|
}) as typeof fetch;
|
|
return () => calls;
|
|
}
|
|
|
|
const json = (body: unknown, status = 200) =>
|
|
new Response(JSON.stringify(body), { status, headers: { 'content-type': 'application/json' } });
|
|
|
|
describe('SchulcloudClient retries', () => {
|
|
it('retries a 503 and succeeds', async () => {
|
|
const calls = stubFetch([json({}, 503), json({ school: { id: 's' } })]);
|
|
const client = new SchulcloudClient(config);
|
|
await client.me();
|
|
assert.equal(calls(), 2);
|
|
});
|
|
|
|
it('retries transient network errors', async () => {
|
|
const calls = stubFetch([new Error('ECONNRESET'), json({ school: { id: 's' } })]);
|
|
await new SchulcloudClient(config).me();
|
|
assert.equal(calls(), 2);
|
|
});
|
|
|
|
it('does not retry a 401 — a dead token will not recover', async () => {
|
|
const calls = stubFetch([json({}, 401)]);
|
|
await assert.rejects(() => new SchulcloudClient(config).me(), (error: SchulcloudApiError) => {
|
|
assert.equal(error.status, 401);
|
|
return true;
|
|
});
|
|
assert.equal(calls(), 1, 'retrying an expired token just wastes time');
|
|
});
|
|
|
|
it('does not retry a 404', async () => {
|
|
const calls = stubFetch([json({}, 404)]);
|
|
await assert.rejects(() => new SchulcloudClient(config).me());
|
|
assert.equal(calls(), 1);
|
|
});
|
|
|
|
it('gives up after the retry budget and reports the real status', async () => {
|
|
const calls = stubFetch([json({}, 503)]);
|
|
await assert.rejects(() => new SchulcloudClient(config).me(), (error: SchulcloudApiError) => {
|
|
assert.equal(error.status, 503);
|
|
return true;
|
|
});
|
|
assert.equal(calls(), 4, 'one attempt plus three retries');
|
|
});
|
|
});
|
|
|
|
describe('getCards chunking', () => {
|
|
it('never sends more than 20 ids in one request', async () => {
|
|
// The API's query parser turns >20 repeated params into an object, and
|
|
// validation then rejects every id with a message blaming the ids rather
|
|
// than their number. Verified live: 20 -> 200, 21 -> 400.
|
|
const seen: number[] = [];
|
|
globalThis.fetch = (async (url: string | URL) => {
|
|
const count = [...new globalThis.URL(String(url)).searchParams.getAll('ids')].length;
|
|
seen.push(count);
|
|
return json({ data: [] });
|
|
}) as typeof fetch;
|
|
|
|
const ids = Array.from({ length: 47 }, (_, i) => String(i).padStart(24, '0'));
|
|
await new SchulcloudClient(config).getCards(ids);
|
|
|
|
assert.deepEqual(seen, [20, 20, 7], 'should split 47 ids into 20/20/7');
|
|
assert.ok(Math.max(...seen) <= 20);
|
|
});
|
|
|
|
it('merges the chunked responses into one list', async () => {
|
|
let call = 0;
|
|
globalThis.fetch = (async () =>
|
|
json({ data: [{ id: `card${call++}`, height: 1, elements: [] }] })) as typeof fetch;
|
|
const ids = Array.from({ length: 25 }, (_, i) => String(i).padStart(24, '0'));
|
|
const cards = await new SchulcloudClient(config).getCards(ids);
|
|
assert.equal(cards.length, 2, 'one card from each of the two chunks');
|
|
});
|
|
});
|
|
|
|
describe('download timeouts', () => {
|
|
/**
|
|
* A real local server, because the timing is the point: a stubbed fetch
|
|
* cannot model bytes that keep arriving slowly. requestTimeoutMs is 1000ms.
|
|
*/
|
|
async function withServer(
|
|
handler: (res: import('node:http').ServerResponse) => void,
|
|
run: (base: string) => Promise<void>,
|
|
) {
|
|
const { createServer } = await import('node:http');
|
|
const server = createServer((_req, res) => handler(res));
|
|
await new Promise<void>((resolve) => server.listen(0, '127.0.0.1', resolve));
|
|
const { port } = server.address() as import('node:net').AddressInfo;
|
|
try {
|
|
await run(`http://127.0.0.1:${port}`);
|
|
} finally {
|
|
server.closeAllConnections();
|
|
await new Promise<void>((resolve) => server.close(() => resolve()));
|
|
}
|
|
}
|
|
|
|
const slowButSteady = (res: import('node:http').ServerResponse) => {
|
|
res.writeHead(200, { 'content-type': 'application/pdf' });
|
|
let sent = 0;
|
|
const tick = setInterval(() => {
|
|
res.write(Buffer.alloc(10, 65));
|
|
if (++sent === 4) {
|
|
clearInterval(tick);
|
|
res.end();
|
|
}
|
|
}, 400);
|
|
};
|
|
|
|
it('lets a download run past the request timeout while bytes keep arriving', async () => {
|
|
await withServer(slowButSteady, async (base) => {
|
|
// 4 chunks, 400ms apart: 1.6s in total against a 1s timeout, never 1s of silence.
|
|
const client = new SchulcloudClient({ ...config, baseUrl: base } as never);
|
|
const file = await client.getBytes('/file', 'slow.pdf');
|
|
assert.equal(file.bytes.length, 40);
|
|
assert.equal(file.truncated, false);
|
|
});
|
|
});
|
|
|
|
it('abandons a download that stalls', async () => {
|
|
await withServer(
|
|
(res) => {
|
|
res.writeHead(200, { 'content-type': 'application/pdf' });
|
|
res.write(Buffer.alloc(10, 65));
|
|
// …and then nothing, well past the 1s idle limit.
|
|
},
|
|
async (base) => {
|
|
const client = new SchulcloudClient({ ...config, baseUrl: base } as never);
|
|
await assert.rejects(client.getBytes('/file', 'stalled.pdf'), /no data received for 1s/);
|
|
},
|
|
);
|
|
});
|
|
});
|