/** * Converts a Ministry of Revenues EIMS location workbook into `src/config/mor-locations.data.ts`. * * pnpm --filter @edr/freight-api eims:import-locations [--sheet SHEET_NAME] * * Exists so a future MoR workbook replaces the dataset by rerunning one command and reviewing the * diff, instead of anyone hand-editing a thousand rows of tax reference data. Production never * parses the workbook: the committed TypeScript is what ships. * * The generated file reproduces the Ministry's own labels and IDs verbatim. This script validates * and reports; it does not correct. Spelling compatibility with EDR/e-Trade names lives in * `mor-location.resolver.ts`, so the dataset stays traceable back to the source sheet. */ import { writeFileSync } from "node:fs"; import { resolve } from "node:path"; import { Workbook } from "exceljs"; const DEFAULT_SHEET = "EIMS_COUNTRY_REGION_VW"; const OUTPUT = resolve(__dirname, "../config/mor-locations.data.ts"); const COLUMNS = [ "COUNTRY_NO", "COUNTRY_NAME", "PARISH_NO", "PARISH_NAME", "CITY_NO", "CITY_NAME", "LOCALITY_NO", "LOCALITY_DESC", ] as const; type Column = (typeof COLUMNS)[number]; type Row = [number, string, number, string, number, string, number, string]; /** Header cells arrive with stray casing, spaces and non-breaking spaces; compare on this form. */ const headerKey = (value: unknown): string => String(value ?? "") .toUpperCase() .replace(/[^A-Z0-9]+/g, "_") .replace(/^_+|_+$/g, ""); /** * Cell text with only the transport-layer damage removed (Excel's non-breaking spaces, and the * CR/LF a wrapped cell leaves behind). Deliberately keeps the Ministry's own leading/trailing * spaces and spelling — the resolver normalizes at comparison time, the dataset stays as supplied. */ const cellText = (value: unknown): string => { if (value === null || value === undefined) return ""; const raw = typeof value === "object" && "text" in (value as object) ? String((value as { text: unknown }).text ?? "") : String(value); return raw.replace(/\u00a0/g, " ").replace(/\r?\n/g, " "); }; const cellNumber = (value: unknown): number | null => { const text = cellText(value).trim(); if (!/^[0-9]+$/.test(text)) return null; return Number(text); }; async function main(): Promise { const args = process.argv.slice(2); const sheetFlag = args.indexOf("--sheet"); const sheetName = sheetFlag >= 0 ? args[sheetFlag + 1] : DEFAULT_SHEET; const workbookPath = args.find( (arg, i) => !arg.startsWith("--") && (sheetFlag < 0 || i !== sheetFlag + 1), ); if (!workbookPath) { throw new Error( "Usage: eims:import-locations [--sheet SHEET_NAME]\n" + `Defaults to sheet "${DEFAULT_SHEET}".`, ); } const workbook = new Workbook(); await workbook.xlsx.readFile(resolve(process.cwd(), workbookPath)); const sheet = workbook.getWorksheet(sheetName); if (!sheet) { const available = workbook.worksheets.map((w) => w.name).join(", "); throw new Error(`Sheet "${sheetName}" not found. Sheets in this workbook: ${available}`); } // The header is not guaranteed to be row 1 — find the first row carrying every required column. let headerRow = 0; let columnIndex: Partial> = {}; for (let r = 1; r <= Math.min(sheet.rowCount, 20); r++) { const found: Partial> = {}; sheet.getRow(r).eachCell({ includeEmpty: false }, (cell, colNumber) => { const key = headerKey(cell.value) as Column; if (COLUMNS.includes(key) && found[key] === undefined) found[key] = colNumber; }); if (COLUMNS.every((c) => found[c] !== undefined)) { headerRow = r; columnIndex = found; break; } } if (headerRow === 0) { throw new Error( `Sheet "${sheetName}" has no header row containing all required columns: ${COLUMNS.join(", ")}`, ); } const rows: Row[] = []; const problems: string[] = []; for (let r = headerRow + 1; r <= sheet.rowCount; r++) { const sheetRow = sheet.getRow(r); const at = (column: Column): unknown => sheetRow.getCell(columnIndex[column]!).value; // A sheet exported from a view is usually padded with blank rows at the end; skip silently. if (COLUMNS.every((c) => cellText(at(c)).trim() === "")) continue; const countryNo = cellNumber(at("COUNTRY_NO")); const parishNo = cellNumber(at("PARISH_NO")); const cityNo = cellNumber(at("CITY_NO")); const localityNo = cellNumber(at("LOCALITY_NO")); const countryName = cellText(at("COUNTRY_NAME")); const parishName = cellText(at("PARISH_NAME")); const cityName = cellText(at("CITY_NAME")); const localityDesc = cellText(at("LOCALITY_DESC")); const missingIds = ( [ ["COUNTRY_NO", countryNo], ["PARISH_NO", parishNo], ["CITY_NO", cityNo], ["LOCALITY_NO", localityNo], ] as const ) .filter(([, value]) => value === null) .map(([name]) => name); const blankNames = ( [ ["COUNTRY_NAME", countryName], ["PARISH_NAME", parishName], ["CITY_NAME", cityName], ["LOCALITY_DESC", localityDesc], ] as const ) .filter(([, value]) => value.trim() === "") .map(([name]) => name); if (missingIds.length > 0 || blankNames.length > 0) { problems.push( `row ${r}: ${[ missingIds.length ? `non-numeric/missing ${missingIds.join(", ")}` : "", blankNames.length ? `blank ${blankNames.join(", ")}` : "", ] .filter(Boolean) .join("; ")}`, ); continue; } rows.push([ countryNo!, countryName, parishNo!, parishName, cityNo!, cityName, localityNo!, localityDesc, ]); } if (problems.length > 0) { throw new Error( `Sheet "${sheetName}" has ${problems.length} unusable row(s); nothing was written:\n ` + problems.slice(0, 25).join("\n ") + (problems.length > 25 ? `\n ... and ${problems.length - 25} more` : ""), ); } if (rows.length === 0) { throw new Error(`Sheet "${sheetName}" has a valid header but no data rows.`); } // Exact duplicates carry no information and only inflate the file — collapsed here, and the // count reported, so the collapse is a stated decision rather than a silent one. IDs are never // touched: only whole identical rows are dropped. const seen = new Map(); for (const row of rows) { const key = JSON.stringify(row); if (!seen.has(key)) seen.set(key, row); } const unique = [...seen.values()]; const duplicates = rows.length - unique.length; // Deterministic output: same workbook in, byte-identical file out, so a regeneration diff shows // only what the Ministry actually changed. unique.sort( (a, b) => a[0] - b[0] || a[2] - b[2] || a[4] - b[4] || a[6] - b[6] || a[7].localeCompare(b[7]) || a[5].localeCompare(b[5]), ); // A name that resolves to two different codes under the same parent cannot be resolved by any // amount of normalization — the resolver refuses it as ambiguous at filing time rather than // picking one. Reported here so it can be raised with MoR instead of surfacing on a live invoice. const byPath = new Map>(); const collide = (level: string, path: string, name: string, no: number): void => { const key = [level, path, name.trim().toUpperCase().replace(/\s+/g, " ")].join(" | "); if (!byPath.has(key)) byPath.set(key, new Set()); byPath.get(key)!.add(no); }; for (const [cNo, cName, pNo, pName, tNo, tName, lNo, lName] of unique) { collide("COUNTRY_NAME", "", cName, cNo); collide("PARISH_NAME", String(cNo), pName, pNo); collide("CITY_NAME", `${cNo}/${pNo}`, tName, tNo); collide("LOCALITY_DESC", `${cNo}/${pNo}/${tNo}`, lName, lNo); } const conflicts = [...byPath.entries()] .filter(([, codes]) => codes.size > 1) .map(([key, codes]) => { const [level, path, name] = key.split(" | "); const where = path ? ` under ${path}` : ""; return ` ${level} "${name}"${where} -> codes ${[...codes].sort((a, b) => a - b).join(", ")}`; }) .sort(); const header = `/** * GENERATED FILE — do not hand-edit. * * MoR EIMS location master (\`${sheetName}\`), the Ministry's own geographic reference data. * Regenerate from a supplied workbook with: * * pnpm --filter @edr/freight-api eims:import-locations * * Values are reproduced verbatim from the Ministry sheet — original spelling, original casing, * original numbering. Nothing here is cleaned up or renumbered: this file is the traceable copy of * the source. Spelling compatibility between EDR/e-Trade names and MoR names belongs in * \`mor-location.resolver.ts\`'s normalization and alias layer, never here. */ /** \`[COUNTRY_NO, COUNTRY_NAME, PARISH_NO, PARISH_NAME, CITY_NO, CITY_NAME, LOCALITY_NO, LOCALITY_DESC]\` */ export type MorLocationTuple = [number, string, number, string, number, string, number, string]; export const MOR_LOCATIONS: MorLocationTuple[] = [ `; const body = unique .map( ([cNo, cName, pNo, pName, tNo, tName, lNo, lName]) => ` [${cNo}, ${JSON.stringify(cName)}, ${pNo}, ${JSON.stringify(pName)}, ${tNo}, ` + `${JSON.stringify(tName)}, ${lNo}, ${JSON.stringify(lName)}],`, ) .join("\n"); writeFileSync(OUTPUT, `${header}${body}\n];\n`, "utf8"); const countries = new Set(unique.map((r) => r[0])).size; const regions = new Set(unique.map((r) => `${r[0]}/${r[2]}`)).size; const zones = new Set(unique.map((r) => `${r[0]}/${r[2]}/${r[4]}`)).size; console.log(`Wrote ${OUTPUT}`); console.log( ` ${unique.length} rows - ${countries} countries, ${regions} regions, ${zones} zones` + (duplicates > 0 ? `; collapsed ${duplicates} exact duplicate row(s)` : ""), ); if (conflicts.length > 0) { console.log( ` ${conflicts.length} same-hierarchy name conflict(s) - these resolve to an ambiguity ` + "error at filing time, never to a guess:", ); console.log(conflicts.join("\n")); } } main().catch((err: Error) => { console.error(err.message); process.exit(1); });