Add encoding repair utilities and tests:

- Introduced `repairText` and `containsSuspiciousEncoding` utilities for handling corrupted text encodings.
- Added unit tests (`encoding-repair.spec.js`) to validate encoding repair and encoding marker detection.
- Integrated encoding check into unit test workflows (`text:check-encoding` and `text:fix-encoding` scripts).
- Replaced corrupted strings across multiple components and locales with proper UTF-8 encodings.
This commit is contained in:
Jeppe Bundgaard
2026-04-21 10:24:35 +02:00
parent 1a784bbbfd
commit d7ff64bf4e
41 changed files with 2890 additions and 1283 deletions
+314
View File
@@ -0,0 +1,314 @@
import { promises as fs } from "node:fs";
import path from "node:path";
import process from "node:process";
import { pathToFileURL } from "node:url";
const SCAN_ROOTS = ["src", "tests"];
const SCAN_EXTENSIONS = new Set([".vue", ".js", ".ts", ".json"]);
const CP1252_EXTRA_BYTE_BY_CHAR = new Map([
["€", 0x80],
["", 0x82],
["ƒ", 0x83],
["„", 0x84],
["…", 0x85],
["†", 0x86],
["‡", 0x87],
["ˆ", 0x88],
["‰", 0x89],
["Š", 0x8a],
["", 0x8b],
["Œ", 0x8c],
["Ž", 0x8e],
["", 0x91],
["", 0x92],
["“", 0x93],
["”", 0x94],
["•", 0x95],
["", 0x96],
["—", 0x97],
["˜", 0x98],
["™", 0x99],
["š", 0x9a],
["", 0x9b],
["œ", 0x9c],
["ž", 0x9e],
["Ÿ", 0x9f],
]);
const SUSPICIOUS_PATTERNS = [
/ÃÂ/g,
/Ã./g,
/Â(?=\S)/g,
/â€(?:[™œžŸ"“”‘’•–—…])/g,
/’/g,
/“/g,
/”/g,
/–/g,
/—/g,
/…/g,
//g,
];
const REGIONAL_CHAR_REGEX = /[æøåÆØÅäöÄÖüÜßéÉèÈáÁàÀóÓúÚíÍñÑçÇ]/g;
const LATIN_EXTENDED_CHAR_REGEX = /[\u00C0-\u024F]/g;
const CONTROL_CHAR_REGEX = /[\u0000-\u0008\u000b\u000c\u000e-\u001f\u007f-\u009f]/g;
const utf8Decoder = new TextDecoder("utf-8", { fatal: true });
function countMatches(text, regex) {
const matches = text.match(regex);
return matches ? matches.length : 0;
}
export function countSuspiciousMarkers(text) {
return SUSPICIOUS_PATTERNS.reduce((count, pattern) => count + countMatches(text, pattern), 0);
}
export function countRegionalCharacters(text) {
return countMatches(text, REGIONAL_CHAR_REGEX);
}
function countLatinExtendedCharacters(text) {
return countMatches(text, LATIN_EXTENDED_CHAR_REGEX);
}
function countControlCharacters(text) {
return countMatches(text, CONTROL_CHAR_REGEX);
}
function encodeLatin1(text) {
const bytes = [];
for (const character of text) {
const codePoint = character.codePointAt(0);
if (codePoint > 0xff) {
return null;
}
bytes.push(codePoint);
}
return Uint8Array.from(bytes);
}
function encodeWindows1252(text) {
const bytes = [];
for (const character of text) {
if (CP1252_EXTRA_BYTE_BY_CHAR.has(character)) {
bytes.push(CP1252_EXTRA_BYTE_BY_CHAR.get(character));
continue;
}
const codePoint = character.codePointAt(0);
if (codePoint > 0xff) {
return null;
}
bytes.push(codePoint);
}
return Uint8Array.from(bytes);
}
function decodeUtf8(bytes) {
try {
return utf8Decoder.decode(bytes);
} catch {
return null;
}
}
function buildRepairCandidates(text) {
const candidateSet = new Set();
const encoders = [encodeLatin1, encodeWindows1252];
for (const encoder of encoders) {
const encoded = encoder(text);
if (!encoded) {
continue;
}
const decoded = decodeUtf8(encoded);
if (decoded && decoded !== text) {
candidateSet.add(decoded);
}
}
return [...candidateSet];
}
function isRepairImprovement(before, after) {
if (!after || after === before) {
return false;
}
const beforeSuspicious = countSuspiciousMarkers(before);
const afterSuspicious = countSuspiciousMarkers(after);
const beforeRegional = countRegionalCharacters(before);
const afterRegional = countRegionalCharacters(after);
const beforeLatinExtended = countLatinExtendedCharacters(before);
const afterLatinExtended = countLatinExtendedCharacters(after);
const beforeControls = countControlCharacters(before);
const afterControls = countControlCharacters(after);
if (afterControls > beforeControls) {
return false;
}
if (afterSuspicious < beforeSuspicious) {
return afterRegional >= beforeRegional || afterLatinExtended >= beforeLatinExtended || afterSuspicious === 0;
}
return (
afterSuspicious === beforeSuspicious && (afterRegional > beforeRegional || afterLatinExtended > beforeLatinExtended)
);
}
function scoreCandidate(text) {
return (
countSuspiciousMarkers(text) * 20 +
countControlCharacters(text) * 30 -
countLatinExtendedCharacters(text) * 2 -
countRegionalCharacters(text) * 3
);
}
function repairOnce(text) {
const candidates = buildRepairCandidates(text).filter((candidate) => isRepairImprovement(text, candidate));
if (candidates.length === 0) {
return text;
}
return candidates.reduce((bestCandidate, candidate) => {
return scoreCandidate(candidate) < scoreCandidate(bestCandidate) ? candidate : bestCandidate;
});
}
function repairSegment(text) {
let current = text;
for (let pass = 0; pass < 4; pass += 1) {
const repaired = repairOnce(current);
if (repaired === current) {
break;
}
current = repaired;
}
return current;
}
export function repairText(text) {
const wholeTextRepair = repairSegment(text);
const linewiseRepair = wholeTextRepair
.split(/(\r?\n)/)
.map((segment) => (segment.match(/\r?\n/) ? segment : repairSegment(segment)))
.join("");
return repairSegment(linewiseRepair);
}
export function containsSuspiciousEncoding(text) {
return countSuspiciousMarkers(text) > 0;
}
function getPreview(line) {
return line.replace(/\s+/g, " ").trim().slice(0, 180);
}
export function findSuspiciousLines(text) {
return text
.split(/\r?\n/)
.map((line, index) => ({
lineNumber: index + 1,
preview: getPreview(line),
suspiciousCount: countSuspiciousMarkers(line),
}))
.filter((entry) => entry.suspiciousCount > 0);
}
async function collectScanFiles(rootDir) {
const files = [];
async function walk(currentDir) {
let entries = [];
try {
entries = await fs.readdir(currentDir, { withFileTypes: true });
} catch {
return;
}
for (const entry of entries) {
const entryPath = path.join(currentDir, entry.name);
if (entry.isDirectory()) {
await walk(entryPath);
continue;
}
if (!entry.isFile() || !SCAN_EXTENSIONS.has(path.extname(entry.name))) {
continue;
}
files.push(entryPath);
}
}
for (const scanRoot of SCAN_ROOTS) {
await walk(path.join(rootDir, scanRoot));
}
return files.sort();
}
async function run(mode) {
const rootDir = process.cwd();
const files = await collectScanFiles(rootDir);
const suspiciousReports = [];
let changedFiles = 0;
for (const filePath of files) {
const originalText = await fs.readFile(filePath, "utf8");
const repairedText = repairText(originalText);
if (mode === "fix" && repairedText !== originalText) {
await fs.writeFile(filePath, repairedText, "utf8");
changedFiles += 1;
}
const inspectedText = mode === "fix" ? repairedText : originalText;
const suspiciousLines = findSuspiciousLines(inspectedText);
if (suspiciousLines.length > 0) {
suspiciousReports.push({
filePath: path.relative(rootDir, filePath),
lines: suspiciousLines.slice(0, 12),
});
}
}
if (mode === "fix") {
console.log(`Updated ${changedFiles} file(s).`);
}
if (suspiciousReports.length > 0) {
console.error(`Detected suspicious encoding markers in ${suspiciousReports.length} file(s):`);
suspiciousReports.forEach((report) => {
console.error(`- ${report.filePath}`);
report.lines.forEach((line) => {
console.error(` ${line.lineNumber}: ${line.preview}`);
});
});
process.exitCode = 1;
return;
}
console.log(`No suspicious encoding markers found across ${files.length} file(s).`);
}
const isDirectRun = typeof process.argv[1] === "string" && pathToFileURL(process.argv[1]).href === import.meta.url;
if (isDirectRun) {
const mode = process.argv[2] || "check";
if (!["check", "fix"].includes(mode)) {
console.error("Usage: node scripts/text-encoding.mjs <check|fix>");
process.exit(1);
}
await run(mode);
}