import assert from 'node:assert/strict'; import { describe, it } from 'node:test'; import { extractContent, formatBytes } from '../src/core/extract.ts'; const MAX = 10_000; describe('extractContent', () => { it('returns images inline as base64 without touching the bytes', async () => { const png = Buffer.from('89504e470d0a1a0a', 'hex'); const result = await extractContent(png, 'image/png', 'a.png', MAX); assert.equal(result.kind, 'image'); assert.equal(result.image?.base64, png.toString('base64')); assert.equal(result.image?.mimeType, 'image/png'); }); it('reads plain text and normalises CRLF', async () => { const result = await extractContent(Buffer.from('a\r\nb\r\n\r\n\r\n\r\nc'), 'text/plain', 'a.txt', MAX); assert.equal(result.kind, 'text'); assert.equal(result.text, 'a\nb\n\nc'); }); it('recognises text even when the server mislabels it as octet-stream', async () => { const result = await extractContent(Buffer.from('hello world'), 'application/octet-stream', 'note', MAX); assert.equal(result.kind, 'text'); assert.equal(result.text, 'hello world'); }); it('reports binary content instead of emitting mojibake', async () => { const bytes = Buffer.from([0x00, 0x01, 0x02, 0xff, 0xfe, 0x00]); const result = await extractContent(bytes, 'application/octet-stream', 'blob.bin', MAX); assert.equal(result.kind, 'binary'); assert.match(result.note, /no text extractor/); }); it('truncates at the limit and says so', async () => { const result = await extractContent(Buffer.from('x'.repeat(5000)), 'text/plain', 'a.txt', 100); assert.equal(result.truncated, true); assert.equal(result.text?.length, 100); assert.match(result.note, /truncated to 100 characters \(of 5000\)/); }); it('turns a parser failure into a note rather than throwing', async () => { const result = await extractContent(Buffer.from('not really a pdf'), 'application/pdf', 'broken.pdf', MAX); assert.equal(result.kind, 'binary'); assert.match(result.note, /Could not extract text|no text extractor/); }); }); describe('formatBytes', () => { it('scales units', () => { assert.equal(formatBytes(512), '512 B'); assert.equal(formatBytes(2048), '2.0 KB'); assert.equal(formatBytes(5 * 1024 * 1024), '5.0 MB'); }); }); /** * Builds a structurally valid PDF with correct xref offsets — pdfjs rejects * anything less, so a hand-waved byte string would test the error path instead * of the one we care about. */ function minimalPdf(options: { withFont: boolean }): Buffer { const content = options.withFont ? 'BT /F1 12 Tf 10 100 Td (Hallo Welt) Tj ET' : 'q 100 0 0 100 10 10 cm /Im0 Do Q'; // draws an image, no text operators const objs = [ '<>', '<>', `<>' : '/XObject<>' }>>/Contents 4 0 R>>`, `<>stream\n${content}\nendstream`, options.withFont ? '<>' : '<>', ]; let out = '%PDF-1.4\n'; const offsets: number[] = []; objs.forEach((body, index) => { offsets.push(out.length); out += `${index + 1} 0 obj${body}endobj\n`; }); const xref = out.length; out += `xref\n0 ${objs.length + 1}\n0000000000 65535 f \n`; for (const offset of offsets) out += `${String(offset).padStart(10, '0')} 00000 n \n`; out += `trailer<>\nstartxref\n${xref}\n%%EOF`; return Buffer.from(out, 'latin1'); } describe('image-only PDFs', () => { it('extracts normally when the PDF has a text layer', async () => { const result = await extractContent(minimalPdf({ withFont: true }), 'application/pdf', 'doc.pdf', 10_000); assert.equal(result.kind, 'text'); assert.match(result.text ?? '', /Hallo Welt/); }); it('reports a missing text layer rather than a bare zero-character result', async () => { // Measured on the real account: 3 of 4 sampled course PDFs are image-only, // so "0 characters" must not look like a parser failure. const result = await extractContent(minimalPdf({ withFont: false }), 'application/pdf', 'scan.pdf', 10_000); assert.equal(result.kind, 'binary'); assert.match(result.note, /image-only PDF/); assert.match(result.note, /OCR/); }); });