chore: initialize repository baseline

Import the existing Electron + React + TypeScript app as the version-control
baseline before the scanner rework (C# input/capture sidecar, resolution-anchored
layout profiles, OCR preprocessing, eval harness, rescan-merge, GOOD interop).

Housekeeping in this commit:
- Remove orphaned temp_inputhelper_block.ts (duplicate of the input-helper script).
- Ignore .claude/scheduled_tasks.lock local session state.
- Add .gitattributes to normalize line endings (LF in repo).

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
AzuTear
2026-07-05 20:31:01 +02:00
commit e76d88e0c7
147 changed files with 26220 additions and 0 deletions
+584
View File
@@ -0,0 +1,584 @@
import type { CaptureResult } from "../types/global.js";
import {
allowedMainStatsForSlot,
canonicalStatName,
fixedMainStatBySlot,
globalMainStats,
globalSubstats,
knownCharacters,
knownPieceNames,
knownSets,
mainStatValueReferences,
normalizeCharacterAlias,
normalizePieceAlias,
normalizeSetAlias,
normalizeSlotAlias,
pieceToSet,
pieceToSlot,
slotNames,
textReplacements,
} from "./genshinData.js";
import { fuzzyFindKnown, simplifyForMatch } from "./fuzzyMatch.js";
type MainStatValueReference = { stat: string; base: number; max: number };
export interface ParsedField {
value: string;
confidence: number;
source: "ocr" | "database" | "derived" | "fallback" | "missing";
}
export interface ParsedArtifactCandidate {
name: string;
slot: string;
level: number;
mainStat: string;
mainValue: string;
substats: string[];
setName: string;
equipped: string;
confidence: number;
notes: string[];
fields: {
name: ParsedField;
slot: ParsedField;
level?: ParsedField;
mainStat: ParsedField;
mainValue: ParsedField;
setName: ParsedField;
equipped: ParsedField;
substats: ParsedField;
};
}
const mainStatNames = sortLongestFirst(globalMainStats);
const substatNames = sortLongestFirst(globalSubstats);
export function parseArtifactCandidate(capture: CaptureResult | null): ParsedArtifactCandidate | null {
if (!capture?.ocr?.length) return null;
const byId = new Map(
(capture.ocr ?? []).map((entry: (typeof capture.ocr)[number]) => [entry.id, normalizeText(entry.text)]),
);
const allText = normalizeText((capture.ocr ?? []).map((entry: (typeof capture.ocr)[number]) => entry.text).join("\n"));
const titleText = byId.get("artifact-title") ?? "";
const mainText = byId.get("artifact-main-stat") ?? "";
const substatText = byId.get("artifact-substats") ?? "";
const setText = byId.get("artifact-set-effects") ?? "";
const footerText = byId.get("artifact-footer") ?? "";
const nameField = parseArtifactName(titleText);
const slotField = parseSlot(titleText + "\n" + allText, nameField.value);
const levelField = parseArtifactLevel(substatText + "\n" + mainText + "\n" + allText);
const parsedLevel = levelField.value ? Number.parseInt(levelField.value, 10) : null;
const level = parsedLevel ?? 0;
let mainStatField = inferMainStat(slotField.value, mainText);
let mainValueField = findMainValue(mainText, mainStatField.value, slotField.value, parsedLevel);
if (!mainStatField.value && mainValueField.value) {
const inferredFromValue = inferMainStatFromValue(slotField.value, mainValueField.value, mainText, parsedLevel);
if (inferredFromValue.value) {
mainStatField = inferredFromValue;
mainValueField = findMainValue(mainText, mainStatField.value, slotField.value, parsedLevel);
}
}
if (!mainStatField.value && mainValueField.value) {
const exactReferenceMatch = deriveMainStatFromExactReferenceValue(slotField.value, mainValueField.value, mainText, parsedLevel);
if (exactReferenceMatch.value) {
mainStatField = exactReferenceMatch;
mainValueField = findMainValue(mainText, mainStatField.value, slotField.value, parsedLevel);
}
}
if ((!mainStatField.value || !mainValueField.value) && slotField.value) {
const noisyReferenceMatch = deriveMainStatAndValueFromNoisyReference(slotField.value, mainText, parsedLevel);
if (!mainStatField.value && noisyReferenceMatch.mainStat.value) {
mainStatField = noisyReferenceMatch.mainStat;
}
if (!mainValueField.value && noisyReferenceMatch.mainValue.value) {
mainValueField = noisyReferenceMatch.mainValue;
}
}
const substats = parseSubstats([substatText, leadingSetEffectText(setText)].filter(Boolean).join("\n"));
const substatsField = field(substats.join(", "), substats.length >= 4 ? 96 : substats.length >= 3 ? 82 : substats.length > 0 ? 55 : 0, substats.length ? "ocr" : "missing");
const setField = parseSetName(setText, nameField.value);
const equippedField = parseEquippedCharacter(footerText + "\n" + allText);
const notes: string[] = [];
if (!nameField.value) notes.push("Artifact name not confidently parsed.");
if (nameField.value && nameField.confidence < 84) notes.push("Artifact name was fuzzy-matched; review if this piece matters.");
if (!slotField.value) notes.push("Slot not confidently parsed.");
if (!levelField.value) notes.push("Artifact level not confidently parsed.");
if (!mainStatField.value) notes.push("Main stat not confidently parsed.");
if (!mainValueField.value) notes.push("Main stat value not confidently parsed.");
if (substats.length < 3) notes.push("Substats look incomplete; crop or OCR needs tuning.");
if (!setField.value) notes.push("Set name not confidently parsed.");
for (const [label, parsedField] of Object.entries({
name: nameField,
slot: slotField,
level: levelField,
mainStat: mainStatField,
mainValue: mainValueField,
set: setField,
equipped: equippedField,
}) as Array<[string, ParsedField]>) {
if (parsedField.value && parsedField.confidence < 70) notes.push(`${label} confidence is low; review before trusting it.`);
}
const fields = {
name: nameField,
slot: slotField,
level: levelField,
mainStat: mainStatField,
mainValue: mainValueField,
setName: setField,
equipped: equippedField,
substats: substatsField,
};
const confidence = Math.round(Object.values(fields).reduce((sum, parsedField) => sum + parsedField.confidence, 0) / Object.values(fields).length);
return {
name: nameField.value || "Unknown artifact",
slot: slotField.value || "Unknown slot",
level,
mainStat: mainStatField.value || "Unknown main stat",
mainValue: mainValueField.value || "?",
substats,
setName: setField.value || "Unknown set",
equipped: equippedField.value || "Not detected",
confidence,
notes: [...new Set(notes)],
fields,
};
}
function parseArtifactName(titleText: string): ParsedField {
const titleLines = titleText
.split("\n")
.map((line) => cleanupOcrLabel(line))
.filter(Boolean);
for (const line of titleLines) {
const alias = normalizePieceAlias(line);
if (alias) return field(alias, 96, "database");
}
const knownPiece = fuzzyFindKnown(titleText, knownPieceNames, 0.72);
if (knownPiece) return field(knownPiece.value, Math.round(knownPiece.score * 100), knownPiece.score >= 0.98 ? "database" : "fallback");
const fallback = firstUsefulLine(titleText, slotNames);
return fallback ? field(fallback, 50, "fallback") : field("", 0, "missing");
}
function parseSlot(text: string, artifactName: string): ParsedField {
const slotLines = text
.split("\n")
.map((line) => cleanupOcrLabel(line))
.filter(Boolean);
for (const line of slotLines) {
const alias = normalizeSlotAlias(line);
if (alias) return field(alias, 96, "ocr");
}
const directSlot = fuzzyFindKnown(text, slotNames, 0.68);
if (directSlot) return field(directSlot.value, Math.round(directSlot.score * 100), directSlot.score >= 0.95 ? "ocr" : "fallback");
const derivedSlot = artifactName ? pieceToSlot.get(artifactName) ?? "" : "";
return derivedSlot ? field(derivedSlot, 94, "derived") : field("", 0, "missing");
}
function parseSetName(setText: string, artifactName: string): ParsedField {
const setFromPiece = artifactName ? pieceToSet.get(artifactName) : undefined;
const candidateLines = setText
.split("\n")
.map((line) => line.trim().replace(/:$/, ""))
.filter((line) => line.length > 3 && !/^\d/.test(line) && !/piece set/i.test(line));
for (const line of candidateLines) {
const alias = normalizeSetAlias(line);
if (alias) return field(alias, 96, "database");
}
const directLine = candidateLines.find((line) => line.length > 8);
const setFromText = fuzzyFindKnown(`${directLine ?? ""}\n${setText}`, knownSets, 0.64);
if (setFromText && (!setFromPiece || setFromText.score >= 0.78)) return field(setFromText.value, Math.round(setFromText.score * 100), setFromText.score >= 0.95 ? "ocr" : "fallback");
if (setFromPiece) return field(setFromPiece, 92, "derived");
return setFromText ? field(setFromText.value, Math.round(setFromText.score * 100), "fallback") : field("", 0, "missing");
}
function parseArtifactLevel(text: string): ParsedField {
const lines = normalizeText(text)
.split("\n")
.map((line) => line.trim())
.filter(Boolean);
for (const line of lines) {
const match = line.match(/^\+\s*(20|1[0-9]|[0-9])\b/);
if (match?.[1]) return field(match[1], 96, "ocr");
}
const anywhere = normalizeText(text).match(/(?:^|\s)\+\s*(20|1[0-9]|[0-9])\b/);
return anywhere?.[1] ? field(anywhere[1], 84, "ocr") : field("", 0, "missing");
}
function normalizeText(text: string) {
return applyTextReplacements(text)
.replace(/[\u201c\u201d]/g, '"')
.replace(/[\u2019]/g, "'")
.replace(/[\u00B7]/g, ".")
.replace(/\r/g, "")
.replace(/[|]/g, "I")
.replace(/\s+\n/g, "\n")
.trim();
}
function applyTextReplacements(text: string) {
return Object.entries(textReplacements as Record<string, string>).reduce(
(current, [from, to]) => current.replace(new RegExp(escapeRegex(from), "gi"), to),
text,
);
}
function firstUsefulLine(text: string, rejectIncludes: string[]) {
return text
.split("\n")
.map((line) => line.trim())
.find((line) => line.length > 5 && !rejectIncludes.some((reject) => simplifyForMatch(line).includes(simplifyForMatch(reject)))) ?? "";
}
function findMainValue(text: string, mainStat: string, slot: string, level: number | null): ParsedField {
const cleaned = text.replace(/\b20\b/g, " ").replace(/[Oo]/g, "0");
const percentValue = extractPercentValue(cleaned);
let ocrField = field("", 0, "missing");
if (percentValue && !mainStat) ocrField = field(percentValue, 84, "ocr");
if (percentValue && isPercentMainStat(mainStat)) ocrField = field(percentValue, 96, "ocr");
const flat = /\b([0-9]{1,2},[0-9]{3}|[0-9]{2,4})\b/.exec(cleaned);
if (!ocrField.value && flat) ocrField = field(flat[1], 92, "ocr");
const derivedField = deriveMainValueFromLevel(slot, mainStat, level);
if (!ocrField.value) return derivedField;
if (!derivedField.value) return ocrField;
if (["HP", "ATK", "DEF", "Elemental Mastery"].includes(mainStat)) {
const ocrNumeric = parseNumericValue(ocrField.value);
const derivedNumeric = parseNumericValue(derivedField.value);
const derivedIntDigits = String(Math.round(derivedNumeric)).length;
const ocrIntDigits = String(Math.round(ocrNumeric)).length;
if (!Number.isFinite(ocrNumeric) || ocrNumeric < derivedNumeric * 0.5 || ocrIntDigits + 1 < derivedIntDigits) return derivedField;
if (Math.abs(ocrNumeric - derivedNumeric) >= Math.max(4, derivedNumeric * 0.22)) return derivedField;
}
return ocrField;
}
function extractPercentValue(text: string) {
const lineMatches = text
.split("\n")
.map((line) => line.match(/([0-9]{1,3}(?:[.,:\u00B7][0-9])?)\s*%/))
.filter((match): match is RegExpMatchArray => Boolean(match));
const preferred = lineMatches[0] ?? text.match(/([0-9]{1,3}(?:[.,:\u00B7][0-9])?)\s*%/);
if (!preferred?.[1]) return "";
return `${preferred[1].replace(/[:,\u00B7]/g, ".")}%`;
}
function inferMainStat(slot: string, text: string): ParsedField {
if (fixedMainStatBySlot[slot]) return field(fixedMainStatBySlot[slot], 100, "derived");
const direct = findDirectMainStat(text);
if (direct) return field(promotePercentVariant(direct, text), 94, "ocr");
const allowedForSlot = sortLongestFirst(allowedMainStatsForSlot(slot));
const fuzzyAllowed = fuzzyFindKnown(text, allowedForSlot, 0.68);
if (fuzzyAllowed) return field(promotePercentVariant(fuzzyAllowed.value, text), Math.round(fuzzyAllowed.score * 100), "fallback");
const fuzzy = fuzzyFindKnown(text, mainStatNames, 0.72);
return fuzzy ? field(promotePercentVariant(fuzzy.value, text), Math.round(fuzzy.score * 100), "fallback") : field("", 0, "missing");
}
function findDirectMainStat(text: string) {
const compact = simplifyForMatch(text);
const hasPercentValue = /[0-9]{1,3}(?:\.[0-9])?\s*%/.test(text);
const priority = [
"Physical DMG Bonus",
"Elemental Mastery",
"Energy Recharge",
"Healing Bonus",
"Hydro DMG Bonus",
"Pyro DMG Bonus",
"Electro DMG Bonus",
"Cryo DMG Bonus",
"Dendro DMG Bonus",
"Anemo DMG Bonus",
"Geo DMG Bonus",
"CRIT Rate",
"CRIT DMG",
"ATK%",
"HP%",
"DEF%",
"ATK",
"HP",
"DEF",
];
const direct = priority.find((stat) => compact.includes(simplifyForMatch(stat))) ?? "";
if (direct) return direct;
if (hasPercentValue && /(^|\s)atk(\s|$)/i.test(text)) return "ATK%";
if (hasPercentValue && /(^|\s)hp(\s|$)/i.test(text)) return "HP%";
if (hasPercentValue && /(^|\s)def(\s|$)/i.test(text)) return "DEF%";
return "";
}
function inferMainStatFromValue(slot: string, mainValue: string, text: string, level: number | null): ParsedField {
const numeric = Number.parseFloat(normalizeMainValue(mainValue).replace("%", ""));
if (!Number.isFinite(numeric)) return field("", 0, "missing");
const candidates = getSlotMainStatValueReferences(slot)
.map((candidate) => {
const expected = expectedMainStatValue(candidate, level);
return {
...candidate,
expected,
delta: Math.abs(expected - numeric),
};
})
.sort((left, right) => left.delta - right.delta);
const best = candidates[0];
const tolerance = toleranceForMainStatValue(best?.stat ?? "", mainValue);
const competing = candidates.filter((candidate) => candidate.delta <= tolerance);
if (best && competing.length === 1) {
return field(promotePercentVariant(best.stat, text), Math.max(72, Math.round(92 - best.delta * 24)), "derived");
}
return field("", 0, "missing");
}
function deriveMainStatFromExactReferenceValue(slot: string, mainValue: string, text: string, level: number | null): ParsedField {
const numeric = Number.parseFloat(normalizeMainValue(mainValue).replace("%", ""));
if (!Number.isFinite(numeric)) return field("", 0, "missing");
const matches = getSlotMainStatValueReferences(slot)
.filter((candidate) => Math.abs(expectedMainStatValue(candidate, level) - numeric) <= toleranceForMainStatValue(candidate.stat, mainValue))
.sort((left, right) => Math.abs(expectedMainStatValue(left, level) - numeric) - Math.abs(expectedMainStatValue(right, level) - numeric));
if (matches.length !== 1) return field("", 0, "missing");
return field(promotePercentVariant(matches[0].stat, text || mainValue), 88, "derived");
}
function deriveMainStatAndValueFromNoisyReference(slot: string, text: string, level: number | null) {
const fragment = text.replace(/[^\d]/g, "");
if (fragment.length < 2) {
return {
mainStat: field("", 0, "missing"),
mainValue: field("", 0, "missing"),
};
}
const candidates = getSlotMainStatValueReferences(slot)
.map((candidate) => {
const formattedValue = formatExpectedValue(candidate, level);
const digits = formattedValue.replace(/[^\d]/g, "");
return {
candidate,
formattedValue,
score: digitReferenceScore(fragment, digits),
};
})
.filter((entry) => entry.score > 0)
.sort((left, right) => right.score - left.score);
const best = candidates[0];
const second = candidates[1];
if (!best) {
return {
mainStat: field("", 0, "missing"),
mainValue: field("", 0, "missing"),
};
}
if (second && best.score - second.score < 0.2) {
return {
mainStat: field("", 0, "missing"),
mainValue: field("", 0, "missing"),
};
}
return {
mainStat: field(promotePercentVariant(best.candidate.stat, text), Math.round(72 + best.score * 18), "derived"),
mainValue: field(best.formattedValue, Math.round(78 + best.score * 14), "derived"),
};
}
function parseSubstats(text: string) {
const normalized = normalizeText(text)
.replace(/CRIT\s*DMG/gi, "CRIT DMG")
.replace(/CRIT\s*Rate/gi, "CRIT Rate")
.replace(/Energy\s*Recharge/gi, "Energy Recharge")
.replace(/Elemental\s*Mastery/gi, "Elemental Mastery")
.replace(/([A-Z]{2,4})\s*\+/g, "$1+");
const statPattern = new RegExp(`(${substatNames.map(escapeRegex).join("|")})\\s*\\+\\s*([0-9]+(?:\\.[0-9])?%?)`, "gi");
const results: string[] = [];
let match: RegExpExecArray | null;
while ((match = statPattern.exec(normalized))) {
const stat = canonicalSubstatName(match[1], match[2]);
results.push(`${stat}+${match[2]}`);
}
return [...new Set(results)].slice(0, 4);
}
function leadingSetEffectText(text: string) {
const lines = normalizeText(text).split("\n");
const result: string[] = [];
for (const line of lines) {
if (/^\s*\d+\s*-\s*Piece Set/i.test(line) || /piece set/i.test(line)) break;
result.push(line);
}
return result.join("\n");
}
function canonicalSubstatName(rawStat: string, rawValue: string) {
const cleaned = rawStat.replace(/\s+/g, " ").trim();
const known = canonicalStatName(cleaned) || fuzzyFindKnown(cleaned, substatNames, 0.8)?.value || cleaned;
const canonical = canonicalStatName(known);
if (["ATK", "HP", "DEF"].includes(canonical) && rawValue.includes("%")) return `${canonical}%`;
return canonical;
}
function parseEquippedCharacter(text: string): ParsedField {
const equippedLine = text
.split("\n")
.map((line) => line.trim())
.find((line) => /equipped/i.test(line));
if (!equippedLine) return field("Not detected", 45, "missing");
const afterLabel = cleanupCharacterNoise(equippedLine);
const alias = afterLabel ? normalizeCharacterAlias(afterLabel) : "";
if (alias) return field(alias, 96, "database");
const known = afterLabel ? fuzzyFindKnown(afterLabel, knownCharacters, 0.6) : null;
if (known) return field(known.value, Math.round(known.score * 100), known.score >= 0.95 ? "ocr" : "fallback");
const fallbackSearch = cleanupCharacterNoise(text);
const wholeTextMatch = fallbackSearch ? fuzzyFindKnown(fallbackSearch, knownCharacters, 0.88) : null;
if (wholeTextMatch) return field(wholeTextMatch.value, Math.round(wholeTextMatch.score * 100), "fallback");
return afterLabel ? field(afterLabel, 50, "fallback") : field("Not detected", 45, "missing");
}
function field(value: string, confidence: number, source: ParsedField["source"]): ParsedField {
return { value, confidence: Math.max(0, Math.min(100, confidence)), source };
}
function isPercentMainStat(stat: string) {
return /%|Rate|DMG|Bonus|Recharge/i.test(stat);
}
function promotePercentVariant(stat: string, text: string) {
if (["ATK", "HP", "DEF"].includes(stat) && /[0-9]{1,3}(?:\.[0-9])?\s*%/.test(text)) return `${stat}%`;
return stat;
}
function getSlotMainStatValueReferences(slot: string): MainStatValueReference[] {
const valueReferences = mainStatValueReferences[slot];
return Array.isArray(valueReferences) ? valueReferences as MainStatValueReference[] : [];
}
function normalizeMainValue(value: string) {
return value.replace(/[:,\u00B7]/g, ".").replace(/\s+/g, "");
}
function deriveMainValueFromLevel(slot: string, mainStat: string, level: number | null): ParsedField {
if (!mainStat) return field("", 0, "missing");
const reference = getSlotMainStatValueReferences(slot).find((candidate) => candidate.stat === mainStat);
if (!reference) return field("", 0, "missing");
if (level === null) {
if (slot === "Flower of Life" || slot === "Plume of Death") return field(formatExpectedValue(reference, null), 72, "derived");
return field("", 0, "missing");
}
if (level < 0 || level > 20) return field("", 0, "missing");
return field(formatExpectedValue(reference, level), 88, "derived");
}
function expectedMainStatValue(reference: { base: number; max: number }, level: number | null) {
if (level === null || !Number.isFinite(level)) return reference.max;
const clampedLevel = Math.max(0, Math.min(20, level));
return reference.base + (reference.max - reference.base) * (clampedLevel / 20);
}
function formatExpectedValue(reference: { stat: string; base: number; max: number }, level: number | null) {
const numeric = expectedMainStatValue(reference, level);
const rounded = isPercentMainStat(reference.stat)
? roundTo(numeric, 1)
: Math.round(numeric);
return isPercentMainStat(reference.stat) ? `${rounded.toFixed(1)}%` : rounded.toLocaleString("en-US");
}
function toleranceForMainStatValue(stat: string, mainValue: string) {
if (mainValue.includes("%") || isPercentMainStat(stat)) return 0.45;
if (stat === "Elemental Mastery") return 2.5;
return 6;
}
function parseNumericValue(value: string) {
return Number.parseFloat(
value
.replace(/,/g, "")
.replace("%", "")
.trim(),
);
}
function roundTo(value: number, digits: number) {
const factor = 10 ** digits;
return Math.round(value * factor) / factor;
}
function digitReferenceScore(fragment: string, referenceDigits: string) {
if (!fragment || !referenceDigits) return 0;
if (fragment === referenceDigits) return 1;
if (referenceDigits.startsWith(fragment)) {
return Math.max(0, 0.95 - (referenceDigits.length - fragment.length) * 0.08);
}
if (fragment.length >= 3 && isDigitSubsequence(fragment, referenceDigits)) {
return 0.72;
}
return 0;
}
function isDigitSubsequence(fragment: string, referenceDigits: string) {
let index = 0;
for (const char of referenceDigits) {
if (char === fragment[index]) index++;
if (index >= fragment.length) return true;
}
return false;
}
function cleanupOcrLabel(line: string) {
return line
.replace(/^[^A-Za-z]+/, "")
.replace(/[^A-Za-z'\s]+$/g, "")
.replace(/\s+/g, " ")
.trim();
}
function cleanupCharacterNoise(text: string) {
return text
.replace(/^.*?equipped\s*:?\s*/i, "")
.replace(/[^A-Za-z'\-\s]/g, " ")
.replace(/\s+/g, " ")
.trim();
}
function sortLongestFirst(values: string[]) {
return [...values].sort((a, b) => b.length - a.length);
}
function escapeRegex(value: string) {
return value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
}