Searching "Winkler H" returned all seven Winklers instead of the one Hannah. Each word was matched as a substring, so "H" hit T-h-omas, Kat-h-arina and CNC-Dre-h-er:in — every row. The shorter the input, the more useless the result, and an initial is the shortest input anyone would type. A word now has to match at the start of a word: either the haystack begins with it, or a space does. The haystack is first name, last name and job title joined, with hyphens, slashes, colons and dots flattened to spaces, so "dreher" still finds CNC-Dreher:in and "cnc" still finds both the Dreher and the Fräser. Checked against the live data before and after: "winkler h" now returns Hannah Winkler alone, "h winkler" the same in either order, "winkler kat" the two Katharinas, "dreher" the twelve CNC-Dreher. The trigram index on the concatenated name no longer applies, which is the price. At under nine hundred rows the scan is a few milliseconds; an index on the same expression brings it back when that stops being true. LIKE's own wildcards are escaped now — typing "100%" searched for everything before. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
60 lines
2.2 KiB
TypeScript
60 lines
2.2 KiB
TypeScript
import { describe, expect, it } from "vitest";
|
|
import { istPersonalnummer, suchMuster } from "@/lib/employee-search";
|
|
|
|
// Der Anlass: „Winkler H" gab alle sieben Winkler zurück statt der einen
|
|
// Hannah. Das „H" wurde als Teilzeichenkette gesucht und traf damit T-h-omas,
|
|
// Kat-h-arina und CNC-Dre-h-er:in.
|
|
//
|
|
// Was hier geprüft wird, ist die Musterbildung. Dass „am Wortanfang" auch
|
|
// wirklich am Wortanfang trifft, entscheidet das SQL der Seite — und ist
|
|
// gegen die laufende Datenbank belegt: „winkler h" liefert dort genau Hannah
|
|
// Winkler, „winkler kat" die beiden Katharinas, „dreher" die CNC-Dreher:innen.
|
|
|
|
describe("istPersonalnummer", () => {
|
|
it("erkennt reine Ziffern", () => {
|
|
expect(istPersonalnummer("3038")).toBe(true);
|
|
expect(istPersonalnummer(" 3038 ")).toBe(true);
|
|
});
|
|
|
|
it("hält alles andere für einen Namen", () => {
|
|
for (const t of ["winkler", "winkler 3038", "3038a", "", " "]) {
|
|
expect(istPersonalnummer(t)).toBe(false);
|
|
}
|
|
});
|
|
});
|
|
|
|
describe("suchMuster", () => {
|
|
it("macht aus jedem Wort ein Paar: am Anfang, und nach einem Leerzeichen", () => {
|
|
expect(suchMuster("winkler")).toEqual([["winkler%", "% winkler%"]]);
|
|
});
|
|
|
|
it("behandelt jedes Wort einzeln", () => {
|
|
expect(suchMuster("winkler h")).toEqual([
|
|
["winkler%", "% winkler%"],
|
|
["h%", "% h%"],
|
|
]);
|
|
});
|
|
|
|
it("kennt keinen Platzhalter mitten im Wort", () => {
|
|
// Das ist der Kern: „h%" trifft Hannah, „%h%" träfe auch Thomas.
|
|
const [, [amAnfang]] = suchMuster("winkler h");
|
|
expect(amAnfang.startsWith("%")).toBe(false);
|
|
});
|
|
|
|
it("schreibt klein, damit der Vergleich unabhängig von der Schreibweise ist", () => {
|
|
expect(suchMuster("WINKLER")).toEqual([["winkler%", "% winkler%"]]);
|
|
});
|
|
|
|
it("verträgt beliebig viel Abstand und Rand", () => {
|
|
expect(suchMuster(" winkler h ")).toHaveLength(2);
|
|
expect(suchMuster(" ")).toEqual([]);
|
|
});
|
|
|
|
it("entschärft die Platzhalter von LIKE", () => {
|
|
// Sonst wäre „%" eine Suche nach allem und „_" nach beliebigem Zeichen.
|
|
expect(suchMuster("100%")).toEqual([["100\\%%", "% 100\\%%"]]);
|
|
expect(suchMuster("a_b")).toEqual([["a\\_b%", "% a\\_b%"]]);
|
|
expect(suchMuster("a\\b")).toEqual([["a\\\\b%", "% a\\\\b%"]]);
|
|
});
|
|
});
|