All checks were successful
Build & Publish Docker Image / build-and-push (push) Successful in 55s
SearXNG liefert je nach Seite mal ein thumbnail/img_src mit, mal nicht — bei Chefkoch-Treffern hatten deshalb zufällig die Hälfte der Kacheln einen Platzhalter, obwohl die Vorschau dann sehr wohl ein Bild fand. searchWeb() holt jetzt für jeden Treffer ohne Thumbnail parallel (max. 6 gleichzeitig, 4 s Timeout pro Request) die Seite und extrahiert das og:image- oder twitter:image-Meta-Tag. Ergebnis wird 30 min in-memory gecacht, damit wiederholte Suchen nicht wieder die gleichen Seiten laden. Tests: - Neuer Test: Treffer ohne Thumbnail wird via og:image angereichert. - Neuer Test: Treffer mit Thumbnail bleibt unverändert (keine Fetch). - Bestehende Tests deaktivieren Enrichment via enrichThumbnails:false, damit sie keine echten Chefkoch-URLs aufrufen.
128 lines
4.7 KiB
TypeScript
128 lines
4.7 KiB
TypeScript
import { describe, it, expect, beforeEach, afterEach } from 'vitest';
|
|
import { createServer, type Server } from 'node:http';
|
|
import type { AddressInfo } from 'node:net';
|
|
import { openInMemoryForTest } from '../../src/lib/server/db';
|
|
import { addDomain } from '../../src/lib/server/domains/repository';
|
|
import { searchWeb } from '../../src/lib/server/search/searxng';
|
|
|
|
let server: Server;
|
|
let baseUrl: string;
|
|
|
|
function respondWith(results: Record<string, unknown>[]) {
|
|
server.on('request', (_req, res) => {
|
|
res.writeHead(200, { 'content-type': 'application/json' });
|
|
res.end(JSON.stringify({ results }));
|
|
});
|
|
}
|
|
|
|
beforeEach(async () => {
|
|
server = createServer();
|
|
await new Promise<void>((r) => server.listen(0, '127.0.0.1', r));
|
|
const addr = server.address() as AddressInfo;
|
|
baseUrl = `http://127.0.0.1:${addr.port}`;
|
|
});
|
|
|
|
afterEach(async () => {
|
|
await new Promise<void>((r) => server.close(() => r()));
|
|
});
|
|
|
|
describe('searchWeb', () => {
|
|
it('filters results by whitelist', async () => {
|
|
const db = openInMemoryForTest();
|
|
addDomain(db, 'chefkoch.de');
|
|
respondWith([
|
|
{
|
|
url: 'https://www.chefkoch.de/rezepte/123/a.html',
|
|
title: 'Carbonara',
|
|
content: 'Pasta'
|
|
},
|
|
{
|
|
url: 'https://fake.de/x',
|
|
title: 'Not allowed',
|
|
content: 'blocked'
|
|
}
|
|
]);
|
|
const hits = await searchWeb(db, 'carbonara', { searxngUrl: baseUrl, enrichThumbnails: false });
|
|
expect(hits.length).toBe(1);
|
|
expect(hits[0].domain).toBe('chefkoch.de');
|
|
expect(hits[0].title).toBe('Carbonara');
|
|
});
|
|
|
|
it('dedupes identical URLs', async () => {
|
|
const db = openInMemoryForTest();
|
|
addDomain(db, 'chefkoch.de');
|
|
respondWith([
|
|
{ url: 'https://www.chefkoch.de/a', title: 'A', content: '' },
|
|
{ url: 'https://www.chefkoch.de/a', title: 'A dup', content: '' }
|
|
]);
|
|
const hits = await searchWeb(db, 'a', { searxngUrl: baseUrl, enrichThumbnails: false });
|
|
expect(hits.length).toBe(1);
|
|
});
|
|
|
|
it('returns empty list when no domains configured', async () => {
|
|
const db = openInMemoryForTest();
|
|
const hits = await searchWeb(db, 'x', { searxngUrl: baseUrl, enrichThumbnails: false });
|
|
expect(hits).toEqual([]);
|
|
});
|
|
|
|
it('returns empty for empty query', async () => {
|
|
const db = openInMemoryForTest();
|
|
addDomain(db, 'chefkoch.de');
|
|
const hits = await searchWeb(db, ' ', { searxngUrl: baseUrl, enrichThumbnails: false });
|
|
expect(hits).toEqual([]);
|
|
});
|
|
|
|
it('enriches missing thumbnails from og:image', async () => {
|
|
const pageServer = createServer((_req, res) => {
|
|
res.writeHead(200, { 'content-type': 'text/html; charset=utf-8' });
|
|
res.end(
|
|
'<html><head><meta property="og:image" content="https://cdn.example/foo.jpg" /></head><body></body></html>'
|
|
);
|
|
});
|
|
await new Promise<void>((r) => pageServer.listen(0, '127.0.0.1', r));
|
|
const addr = pageServer.address() as AddressInfo;
|
|
const pageUrl = `http://127.0.0.1:${addr.port}/rezept`;
|
|
try {
|
|
const db = openInMemoryForTest();
|
|
addDomain(db, '127.0.0.1');
|
|
respondWith([{ url: pageUrl, title: 'Kuchen', content: '' }]);
|
|
const hits = await searchWeb(db, 'kuchen', { searxngUrl: baseUrl });
|
|
expect(hits.length).toBe(1);
|
|
expect(hits[0].thumbnail).toBe('https://cdn.example/foo.jpg');
|
|
} finally {
|
|
await new Promise<void>((r) => pageServer.close(() => r()));
|
|
}
|
|
});
|
|
|
|
it('leaves existing thumbnails untouched (no enrichment fetch)', async () => {
|
|
const db = openInMemoryForTest();
|
|
addDomain(db, 'chefkoch.de');
|
|
respondWith([
|
|
{
|
|
url: 'https://www.chefkoch.de/rezepte/1/x.html',
|
|
title: 'X',
|
|
thumbnail: 'https://cdn.chefkoch/x.jpg'
|
|
}
|
|
]);
|
|
// enrichment enabled, but thumbnail is set → no fetch expected, no hang
|
|
const hits = await searchWeb(db, 'x', { searxngUrl: baseUrl });
|
|
expect(hits[0].thumbnail).toBe('https://cdn.chefkoch/x.jpg');
|
|
});
|
|
|
|
it('filters out forum/magazine/listing URLs', async () => {
|
|
const db = openInMemoryForTest();
|
|
addDomain(db, 'chefkoch.de');
|
|
respondWith([
|
|
{ url: 'https://www.chefkoch.de/rezepte/123/Ravioli.html', title: 'Ravioli' },
|
|
{ url: 'https://www.chefkoch.de/forum/2,17,89865/ravioli.html', title: 'Forum Ravioli' },
|
|
{ url: 'https://www.chefkoch.de/magazin/artikel/x.html', title: 'Magazin' },
|
|
{ url: 'https://www.chefkoch.de/suche/ravioli', title: 'Suche' },
|
|
{ url: 'https://www.chefkoch.de/themen/ravioli/', title: 'Themen' },
|
|
{ url: 'https://www.chefkoch.de/rezepte/', title: 'Rezepte Übersicht' }
|
|
]);
|
|
const hits = await searchWeb(db, 'ravioli', { searxngUrl: baseUrl, enrichThumbnails: false });
|
|
expect(hits.length).toBe(1);
|
|
expect(hits[0].title).toBe('Ravioli');
|
|
});
|
|
});
|