Files
Schulcloud-MCP/src/core/match.ts
MechaCat02 dc50b4bcd5 Write the notes in an app, a school day at a time
The notes existed but there was nowhere to write them: a CLI command on a
laptop, a tool call through Claude, or a file in a Docker volume. None of
those is reachable from a phone in a lesson, which is where notes are
actually taken.

So: `/app`, served only when WEB_PASSWORD is set. A login, the day's
notes, and a settings page for the Schulcloud token — the one surface
here meant for a person rather than a program.

The shape follows how the notes are written: one note per school day,
one `##` heading per lesson, prose and lists and tables beneath. That
turns out to be the design decision that matters, twice over.

First, it is what lets WebUntis earn its keep. Opening a day with no
note fills in that day's lessons — numbered, with times, teacher and
room, cancellations dropped and substitutions marked. Retyping the
timetable is exactly the work the second upstream exists to avoid, and
"Stunden ergänzen" tops up a note started before the day ended without
touching what is already written.

Second, it changes how notes are indexed. A day note is indexed per
lesson, not whole: search answers "my own note, Deutsch, 18.09.2026"
rather than "my own note, Friday", and `list_notes subject=Deutsch`
finds a day whose frontmatter names no subject at all. Indexed whole,
every hit would read as a weekday and "what did we do in Deutsch" would
match notes whose other five lessons were something else. `lessonHeading`
and `subjectFromHeading` are a loop — the app writes the heading, the
indexer reads the subject back out — and a test holds them to it.

Notes taken in a lesson cannot be retaken, so the editor is built
around not losing them: autosave, every keystroke mirrored to local
storage, a save when the phone locks, and a fallback to the local copy
when the request never arrives. A save that would overwrite a version
the editor never saw is refused and the choice handed back — the notes
folder is synced and open in more than one place, and a phone must not
silently win over a laptop. `replaceNote` is separate from `writeNote`
for that reason: never-overwrite is right for `add_note` and exactly
wrong for an editor.

WEB_PASSWORD is the first credential here a human types, so it is the
first that can be guessed: scrypt at startup, never stored or compared
in the clear, per-address rate limiting — which is not decoration, since
the scrypt cost is itself a denial-of-service vector without it. The
session is a signed HttpOnly SameSite=Strict cookie whose key is derived
from the password, so changing it logs everyone out and there is no
second secret to keep. It opens /api, because a session is the user, and
never /mcp, because nothing in a browser speaks MCP.

Also here, because the app made them matter: frontmatter now reads the
indented `- item` list form editors write, so an Obsidian vault
round-trips its tags; and a four-digit folder is a filing scheme, not a
subject, so `2026/` does not file a school year under one.

357 tests; 106/107 smoke against the local instance, the one failure
being the H5P service that instance does not run.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-19 17:21:17 +02:00

169 lines
5.6 KiB
TypeScript

import type { CrawledBoard, Snapshot } from './crawl.ts';
import { h5pSearchText } from './h5p.ts';
import { noteSearchText, noteSections } from './notes.ts';
import { matchesAll, snippet, tokenize } from './text.ts';
/**
* Keyword matching over a crawled snapshot.
*
* This is the fallback path: when the Postgres index is unavailable or a
* caller asks for guaranteed-fresh results, we crawl and match in memory
* instead. Pure and synchronous, so it is testable without a network or a
* database.
*/
export interface Hit {
courseId: string;
courseTitle: string;
/** Where the match was found, in human terms. */
where: string;
/** Id to pass to a follow-up tool, with the tool that takes it. */
targetId: string;
targetKind: 'course' | 'room' | 'board' | 'lesson' | 'task' | 'file' | 'note';
snippet: string;
}
/**
* Matches a container's boards. Shared by courses and rooms so a hit in a room
* looks and behaves exactly like a hit in a course.
*/
function matchBoards(
boards: CrawledBoard[],
base: { courseId: string; courseTitle: string },
terms: string[],
push: (hit: Hit) => void,
): void {
for (const board of boards) {
if (matchesAll(board.title, terms)) {
push({ ...base, where: 'board title', targetId: board.id, targetKind: 'board', snippet: board.title });
}
// Match per card, so the snippet points at the right part of the board.
for (const column of board.board.columns) {
for (const card of column.cards) {
const parts = [card.title];
for (const element of card.elements) {
if (element.text) parts.push(element.text);
// Pad contents are searched by the index; without this the live
// crawl path would quietly disagree with it.
if (element.padText) parts.push(element.padText);
// Same corpus as the index, or a live search would disagree with it.
if (element.h5p) parts.push(h5pSearchText(element.h5p));
if (element.url) parts.push(element.url);
for (const file of element.files) parts.push(file.name);
}
const haystack = parts.filter(Boolean).join('\n');
if (matchesAll(haystack, terms)) {
push({
...base,
where: `board "${board.title}" → card "${card.title}"`,
targetId: board.id,
targetKind: 'board',
snippet: snippet(haystack, terms),
});
}
}
}
}
}
export function searchSnapshot(snapshot: Snapshot, query: string, limit = 50): Hit[] {
const terms = tokenize(query);
if (terms.length === 0) return [];
const hits: Hit[] = [];
const push = (hit: Hit) => {
hits.push(hit);
};
for (const course of snapshot.courses) {
const base = { courseId: course.course.id, courseTitle: course.title };
if (matchesAll(course.title, terms)) {
push({ ...base, where: 'course title', targetId: course.course.id, targetKind: 'course', snippet: course.title });
}
matchBoards(course.boards, base, terms, push);
for (const lesson of course.lessons) {
const haystack = `${lesson.name}\n${lesson.text}`;
if (matchesAll(haystack, terms)) {
push({
...base,
where: lesson.text && matchesAll(lesson.text, terms) ? `lesson "${lesson.name}"` : 'lesson title',
targetId: lesson.id,
targetKind: 'lesson',
snippet: snippet(haystack, terms),
});
}
}
for (const task of course.tasks) {
const haystack = `${task.task.name}\n${task.text}`;
if (matchesAll(haystack, terms)) {
push({ ...base, where: 'task', targetId: task.id, targetKind: 'task', snippet: snippet(haystack, terms) });
}
}
}
// Rooms carry boards and nothing else, so they reuse the same matcher; a hit
// reads the same whether the board hangs off a course or a room.
for (const room of snapshot.rooms) {
const base = { courseId: room.id, courseTitle: room.name };
if (matchesAll(room.name, terms)) {
push({ ...base, where: 'room title', targetId: room.id, targetKind: 'room', snippet: room.name });
}
matchBoards(room.boards, base, terms, push);
}
// The user's own notes. `courseId` is the subject rather than an id: nothing
// follows a note back to a course, and the subject is what makes the hit
// readable — "Deutsch — my note" rather than a bare path.
for (const note of snapshot.notes) {
// Per lesson where the note has lessons, exactly as the index does it —
// otherwise fresh=true would report "Monday" where the index reports
// "Deutsch, Monday", and the two paths would disagree about the same file.
const sections = noteSections(note);
if (sections.length > 0) {
for (const section of sections) {
const haystack = [section.heading, section.text].filter(Boolean).join('\n');
if (!matchesAll(haystack, terms)) continue;
hits.push({
courseId: note.courseId ?? '',
courseTitle: section.subject ?? note.subject ?? 'Notizen',
where: `my own note, ${section.heading}${note.date ? `, ${note.date}` : ''}`,
targetId: note.path,
targetKind: 'note',
snippet: snippet(haystack, terms),
});
}
continue;
}
const haystack = noteSearchText(note);
if (!matchesAll(haystack, terms)) continue;
hits.push({
courseId: note.courseId ?? '',
courseTitle: note.subject ?? 'Notizen',
where: `my own note${note.date ? `, ${note.date}` : ''}`,
targetId: note.path,
targetKind: 'note',
snippet: snippet(haystack, terms),
});
}
for (const file of snapshot.files) {
if (matchesAll(file.record.name, terms)) {
hits.push({
courseId: file.at.courseId,
courseTitle: file.at.courseTitle,
where: `file in ${file.at.containerTitle ?? 'course'}`,
targetId: file.record.id,
targetKind: 'file',
snippet: file.record.name,
});
}
}
return hits.slice(0, limit);
}