'use strict'; /** * Player-name normalization (Session 46, completed Session 47) — the ONE source * of truth for comparing/de-duplicating player names. PropLine sends the same * player many ways: "A.J. Ewing"/"AJ Ewing", "Jazz Chisholm"/"Jazz Chisholm Jr.", * "Iván Herrera"/"Ivan Herrera", "Matt Liberatore"/"Matthew Liberatore", * "Jose Fermin (STL)". This collapses them all. * * normalizeName("A.J. Ewing") → { display: "AJ Ewing", key: "aj ewing" } * normalizeName("Jose Fermin (STL)") → { display: "Jose Fermin", key: "jose fermin" } * normalizeName("Ronald Acuña Jr.") → { display: "Ronald Acuña Jr", key: "ronald acuna" } * nameKey("Matt Liberatore") === nameKey("Matthew Liberatore") === "matthew liberatore" * nameKey("Iván Herrera") === nameKey("Ivan Herrera") === "ivan herrera" * * `display` keeps proper casing + accents (periods, suffix dots, and parenthetical * team tags stripped). `key` is accent-folded, lowercased, suffix-stripped, and * first-name-nickname-resolved for comparison. * * NOTE: an identical copy lives at web/src/lib/playerName.js for the frontend * (the Next bundle can't import from src/). A test cross-checks they agree. */ const SUFFIXES = new Set(['jr', 'sr', 'ii', 'iii', 'iv', 'v']); // Common first-name nicknames in sports. KEY = short form, VALUE = canonical // long form. Both sides reduce to the long form so "Matt"/"Matthew" merge. const NICKNAMES = { matt: 'matthew', mike: 'michael', chris: 'christopher', jake: 'jacob', josh: 'joshua', nick: 'nicholas', nate: 'nathaniel', dan: 'daniel', danny: 'daniel', dave: 'david', rob: 'robert', bob: 'robert', joe: 'joseph', joey: 'joseph', jon: 'jonathan', johnny: 'john', tony: 'anthony', alex: 'alexander', andy: 'andrew', drew: 'andrew', ben: 'benjamin', will: 'william', bill: 'william', billy: 'william', willy: 'william', zach: 'zachary', zack: 'zachary', tom: 'thomas', tommy: 'thomas', ty: 'tyler', ed: 'edward', eddie: 'edward', teddy: 'theodore', charlie: 'charles', chuck: 'charles', rick: 'richard', dick: 'richard', jim: 'james', jimmy: 'james', ray: 'raymond', fred: 'frederick', kenny: 'kenneth', sam: 'samuel', pat: 'patrick', greg: 'gregory', steve: 'steven', tim: 'timothy', frank: 'francis', mickey: 'michael', richie: 'richard', }; // Collapse adjacent single-letter words: "J C Escarra" → "JC Escarra", // "J C E Escarra" → "JCE Escarra" (PropLine sends space-separated initials). function collapseInitials(s) { return s.replace(/\b([A-Za-z])(?: ([A-Za-z]))+\b(?=\s|$)/g, (m) => m.replace(/ /g, '')); } function normalizeName(raw) { const display = collapseInitials(String(raw == null ? '' : raw) .replace(/\s*\([^)]*\)\s*/g, ' ') // strip parenthetical team tags "(STL)" .replace(/\./g, '') // strip dots: "A.J." → "AJ", "Jr." → "Jr" .replace(/-/g, ' ') // "Jung-hoo" → "Jung hoo" (matches "Jung Hoo") .replace(/\s+/g, ' ') .trim()); const folded = display.normalize('NFD').replace(/[̀-ͯ]/g, '').toLowerCase(); const key = folded.split(' ').filter((t) => t && !SUFFIXES.has(t)).join(' '); return { display, key }; } /** * Comparison key: normalized + first-name nickname resolved to the canonical * form ("matt" → "matthew") so nicknames collapse. */ function nameKey(raw) { const { key } = normalizeName(raw); let parts = key.split(/\s+/).filter(Boolean); if (parts.length >= 2 && NICKNAMES[parts[0]]) parts[0] = NICKNAMES[parts[0]]; // Strip single-letter MIDDLE tokens ("josh h smith" → "josh smith"); keep the // first (may be a collapsed initial like "jc") and last token. if (parts.length >= 3) { parts = parts.filter((p, i, arr) => i === 0 || i === arr.length - 1 || p.length > 1); } return parts.join(' '); } module.exports = { normalizeName, nameKey, SUFFIXES, NICKNAMES };