gute
This commit is contained in:
@@ -0,0 +1,213 @@
|
||||
/**
|
||||
* S02 – Säubert + dedupliziert Phrasen aus raw-phrases.json → clean-phrases.json
|
||||
*/
|
||||
import { readFileSync, writeFileSync } from "fs";
|
||||
import { dirname, join } from "path";
|
||||
import { fileURLToPath } from "url";
|
||||
|
||||
const __dirname = dirname(fileURLToPath(import.meta.url));
|
||||
const RAW_PATH = join(__dirname, "raw-phrases.json");
|
||||
const OUT_PATH = join(__dirname, "clean-phrases.json");
|
||||
|
||||
function decodeEntities(s) {
|
||||
return s
|
||||
.replace(/&/g, "&")
|
||||
.replace(/</g, "<")
|
||||
.replace(/>/g, ">")
|
||||
.replace(/"/g, '"')
|
||||
.replace(/&#(\d+);/g, (_, n) => String.fromCharCode(Number(n)));
|
||||
}
|
||||
|
||||
/** trim + Whitespace normalisieren + optionale Endpunkte entfernen (für Dedup-Key) */
|
||||
function normalizeKey(text) {
|
||||
return decodeEntities(String(text))
|
||||
.replace(/\s+/g, " ")
|
||||
.trim()
|
||||
.replace(/\.+$/, "")
|
||||
.toLowerCase();
|
||||
}
|
||||
|
||||
/** Anzeigeform: Whitespace normalisieren, Endpunkt entfernen (Heft-Stil ohne Punkt) */
|
||||
function displayText(text) {
|
||||
return decodeEntities(String(text))
|
||||
.replace(/\s+/g, " ")
|
||||
.trim()
|
||||
.replace(/\.+$/, "")
|
||||
.trim();
|
||||
}
|
||||
|
||||
const HEADER_RE =
|
||||
/^(Abteilung|Ausbildungsnachweis|Für die Zeit|Name|Datum|Zweites|Erstes|Drittes|Betriebliche|Zuordnung|Thema der Woche|Berufsschule|Lfd\.?\s*Nr|Ausbilderin|Ausbilder)\b/i;
|
||||
|
||||
const JUNK_EXACT_RE =
|
||||
/^(43180|Nico|Baumann|Ferien|Feiertag|Urlaub|Schultag|Krank|Montag|Dienstag|Mittwoch|Donnerstag|Freitag|Samstag|Sonntag|Additive,?$|\d{4,}|\d+)$/i;
|
||||
|
||||
function isJunk(text) {
|
||||
const s = displayText(text);
|
||||
if (!s || s.length < 3) return true;
|
||||
if (HEADER_RE.test(s)) return true;
|
||||
if (JUNK_EXACT_RE.test(s)) return true;
|
||||
if (/^[.\-_\s…·•]+$/.test(s)) return true;
|
||||
if (/…{2,}|\.{3,}/.test(s)) return true;
|
||||
if (/Unterschrift/i.test(s)) return true;
|
||||
if (/Ausbildungsnachweis|Für die Zeit|Betriebliche Tätig|Zuordnung zum Lernziel|Thema der Woche|Berufsschule|Lfd\.?\s*Nr/i.test(s))
|
||||
return true;
|
||||
if (/^\d{5,}/.test(s)) return true;
|
||||
if (/^[\d\s./-]+$/.test(s)) return true; // reine Zahlen / Ziffernblöcke
|
||||
if (/<\/?w:|xml|xmlns/i.test(s)) return true;
|
||||
if (!/[a-zäöüß]/i.test(s)) return true;
|
||||
return false;
|
||||
}
|
||||
|
||||
function isFerienActivity(s) {
|
||||
return /^ferien\b/i.test(s) || /^ferien$/i.test(s);
|
||||
}
|
||||
|
||||
/**
|
||||
* @param {Map<string, { text: string, count: number, sources?: Set<string> }>} map
|
||||
* @param {string} raw
|
||||
* @param {string | null} sourceFile
|
||||
* @param {{ trackSources?: boolean, minLen?: number, maxLen?: number }} opts
|
||||
*/
|
||||
function addPhrase(map, raw, sourceFile, opts = {}) {
|
||||
const { trackSources = false, minLen = 4, maxLen = 180 } = opts;
|
||||
if (isJunk(raw)) return;
|
||||
const text = displayText(raw);
|
||||
if (!text || text.length < minLen || text.length > maxLen) return;
|
||||
if (trackSources && isFerienActivity(text)) return;
|
||||
|
||||
const key = normalizeKey(text);
|
||||
if (!key) return;
|
||||
|
||||
const prev = map.get(key);
|
||||
if (prev) {
|
||||
prev.count += 1;
|
||||
// kürzere / punktlose Form bevorzugen
|
||||
if (text.length < prev.text.length || (!/\.$/.test(raw) && /\.$/.test(prev.text))) {
|
||||
prev.text = text;
|
||||
}
|
||||
if (trackSources && sourceFile) prev.sources.add(sourceFile);
|
||||
} else {
|
||||
map.set(key, {
|
||||
text,
|
||||
count: 1,
|
||||
...(trackSources ? { sources: new Set(sourceFile ? [sourceFile] : []) } : {}),
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
function toSortedList(map, { withSources = false } = {}) {
|
||||
return [...map.values()]
|
||||
.map((entry) => {
|
||||
if (withSources) {
|
||||
return {
|
||||
text: entry.text,
|
||||
count: entry.count,
|
||||
sources: [...entry.sources].sort((a, b) => a.localeCompare(b, "de")),
|
||||
};
|
||||
}
|
||||
return { text: entry.text, count: entry.count };
|
||||
})
|
||||
.sort((a, b) => b.count - a.count || a.text.localeCompare(b.text, "de"));
|
||||
}
|
||||
|
||||
function main() {
|
||||
let raw;
|
||||
try {
|
||||
raw = JSON.parse(readFileSync(RAW_PATH, "utf8"));
|
||||
} catch (err) {
|
||||
console.error(
|
||||
"raw-phrases.json fehlt oder ist ungültig. Zuerst S01 (extract-phrases.mjs) ausführen.",
|
||||
);
|
||||
console.error(err.message);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const hefte = raw.hefte ?? [];
|
||||
if (!hefte.length) {
|
||||
console.error("raw-phrases.json enthält keine Hefte.");
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const activities = new Map();
|
||||
const themes = new Map();
|
||||
const school = new Map();
|
||||
|
||||
for (const heft of hefte) {
|
||||
const source = heft.sourceFile ?? null;
|
||||
|
||||
// Pro Heft einmal pro normalisierter Phrase zählen (Tabellen-Duplikate)
|
||||
const seenAct = new Set();
|
||||
for (const a of heft.activities ?? []) {
|
||||
const key = normalizeKey(a);
|
||||
if (!key || seenAct.has(key)) continue;
|
||||
seenAct.add(key);
|
||||
addPhrase(activities, a, source, {
|
||||
trackSources: true,
|
||||
minLen: 6,
|
||||
maxLen: 160,
|
||||
});
|
||||
}
|
||||
|
||||
const seenTheme = new Set();
|
||||
for (const t of heft.weekThemes ?? []) {
|
||||
const key = normalizeKey(t);
|
||||
if (!key || seenTheme.has(key)) continue;
|
||||
seenTheme.add(key);
|
||||
if (/ferien/i.test(displayText(t))) continue;
|
||||
addPhrase(themes, t, null, { minLen: 3, maxLen: 80 });
|
||||
}
|
||||
|
||||
const seenSchool = new Set();
|
||||
for (const s of heft.schoolTopics ?? []) {
|
||||
const key = normalizeKey(s);
|
||||
if (!key || seenSchool.has(key)) continue;
|
||||
seenSchool.add(key);
|
||||
// Ferien gehört ins Berufsschule-Feld der Woche, nicht als Suggestion-Phrase
|
||||
if (/ferien/i.test(displayText(s))) continue;
|
||||
addPhrase(school, s, null, { minLen: 4, maxLen: 160 });
|
||||
}
|
||||
}
|
||||
|
||||
const actList = toSortedList(activities, { withSources: true }).filter(
|
||||
(a) =>
|
||||
!/^\d/.test(a.text) &&
|
||||
!/Lfd/i.test(a.text) &&
|
||||
/[a-zäöüß]{3,}/i.test(a.text),
|
||||
);
|
||||
const themeList = toSortedList(themes).filter(
|
||||
(t) => t.text !== "-" && !/^[.\-]+$/.test(t.text),
|
||||
);
|
||||
const schoolList = toSortedList(school).filter(
|
||||
(s) => !/^krank$/i.test(s.text) && !/stillarbeit$/i.test(s.text),
|
||||
);
|
||||
|
||||
const clean = {
|
||||
activities: actList,
|
||||
themes: themeList,
|
||||
school: schoolList,
|
||||
meta: {
|
||||
source: "raw-phrases.json",
|
||||
hefte: hefte.length,
|
||||
activities: actList.length,
|
||||
themes: themeList.length,
|
||||
school: schoolList.length,
|
||||
generatedAt: new Date().toISOString(),
|
||||
},
|
||||
};
|
||||
|
||||
writeFileSync(OUT_PATH, JSON.stringify(clean, null, 2), "utf8");
|
||||
|
||||
console.log(
|
||||
`Clean: ${hefte.length} Hefte → ${actList.length} Activities, ${themeList.length} Themes, ${schoolList.length} School`,
|
||||
);
|
||||
console.log(
|
||||
"Top Activities:",
|
||||
actList
|
||||
.slice(0, 12)
|
||||
.map((a) => `${a.text} (${a.count})`)
|
||||
.join(" · "),
|
||||
);
|
||||
}
|
||||
|
||||
main();
|
||||
Reference in New Issue
Block a user