diff --git a/scripts/ingest-deadlines/README.md b/scripts/ingest-deadlines/README.md new file mode 100644 index 0000000..169f040 --- /dev/null +++ b/scripts/ingest-deadlines/README.md @@ -0,0 +1,69 @@ +# Common App deadline ingest + +Fills in application deadlines on `public.colleges` from the +[Common App requirements grid](https://content.commonapp.org/Files/ReqGrid.pdf), +the PDF Common App publishes and keeps current through the cycle. + +**46** of 6,273 colleges had a real deadline. Everything else fell back to a +typical date for the round and was shown to the student as an estimate. This +brings that to **~980**. + +## Run + +```bash +cd scripts/ingest-deadlines +node --test parse.test.mjs # the parser is the fragile part; test it first +node ingest.mjs # dry run: prints what it would change, writes nothing +node ingest.mjs -v # the same, plus every row +node ingest.mjs --apply # write +``` + +Needs `pdftotext` (`brew install poppler`) on PATH. Reads Supabase credentials +from `web/.env.local`. Re-runnable: it only ever fills blanks. + +Current run: 1,127 schools parsed, 955 matched, **935 rows to fill**, 0 overwritten. + +## What it will not do + +**It never overwrites a value that is already there.** Spot-checking the +disagreements found the grid right about Stanford (Jan 5, where we held a wrong +Jan 2) and wrong about Georgetown (Jan 1, where Jan 10 is the real date), so +neither source wins outright. The 46 curated rows were checked by hand and a +bulk import does not get to quietly undo that. Disagreements print as +`differs:` lines for someone to settle by hand — there are 13, and 30 other +values agree exactly, which is the main evidence the parse is sound. + +**It does not guess between campuses.** A grid row matches a college by +normalised name, or by a name prefix when exactly one college matches. Arizona +State, Kent State and Colorado State each have six or more campus rows, so +those stay unmatched rather than risk putting the main campus's deadline on a +branch. Of 169 unmatched, most are foreign universities absent from the +Scorecard dataset (Aberystwyth, Dublin City, Esade, IE) and the rest are +multi-campus systems. + +**It does not store EDII or EAII.** 185 schools publish them; the app's +`AppDeadlineType` has no round for them, and a column nothing renders is data +nobody reads. + +**Schools listed more than once are skipped** — a few appear once per program +with different deadlines under one name, and there is no way to tell which +program a student means. + +## The parser + +`parse.mjs` is where everything fragile lives, which is why it is separate and +tested without a network or a database. + +The grid repeats its column header on every one of its 55 pages, and **the +columns are in a different place on each**. Across pages `ED` sits anywhere +from column 36 to 39 and the drift compounds rightward — by `RD` two pages can +be 28 characters apart. So a round is identified by calibrating against that +page's own header, never a fixed offset. + +This was not theoretical: fixed bands read Yale's restrictive-early date as its +regular-decision deadline, because on Yale's page REA sits where other pages +keep RD. `parse.test.mjs` pins that case and the other four that broke. + +If Common App changes the layout, the parser needs revisiting, and it is built +to fail loudly rather than mis-file a round: `ingest.mjs` throws if fewer than +900 schools parse, and refuses to let two grid rows claim one college. diff --git a/scripts/ingest-deadlines/ingest.mjs b/scripts/ingest-deadlines/ingest.mjs new file mode 100644 index 0000000..3c86fd8 --- /dev/null +++ b/scripts/ingest-deadlines/ingest.mjs @@ -0,0 +1,178 @@ +// Common App requirements grid -> Supabase `colleges` deadline columns. +// +// Usage: +// node ingest.mjs # parse, match, print what would change. Writes nothing. +// node ingest.mjs -v # the same, plus every row it would write +// node ingest.mjs --apply # write +// +// Writing needs --apply. It used to be the no-flag default, which meant one +// mistyped invocation went straight at production. +// +// Why this exists: 6,273 colleges are in the table and 46 had a real +// application deadline. Everything else fell back to a typical date for the +// round, shown to the student as "no date on file". Common App publishes a +// grid of its members' deadlines and keeps it current through the cycle, so +// this is re-runnable rather than a one-off import. +// +// Needs `pdftotext` (poppler) on PATH. Reads VITE_SUPABASE_URL and +// VITE_SUPABASE_ANON_KEY from ../../web/.env.local. + +import { readFileSync, writeFileSync, existsSync } from 'node:fs' +import { execFileSync } from 'node:child_process' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { createClient } from '@supabase/supabase-js' +import { parseGrid, toDisplayDate, normaliseName } from './parse.mjs' + +const APPLY = process.argv.includes('--apply') +const VERBOSE = process.argv.includes('-v') +const GRID_URL = 'https://content.commonapp.org/Files/ReqGrid.pdf' + +function loadEnv(path) { + const out = {} + for (const line of readFileSync(path, 'utf8').split('\n')) { + const m = line.match(/^([A-Z0-9_]+)=(.*)$/) + if (m) out[m[1]] = m[2].trim() + } + return out +} + +const web = loadEnv(new URL('../../web/.env.local', import.meta.url).pathname) +const supabase = createClient(web.VITE_SUPABASE_URL, web.VITE_SUPABASE_ANON_KEY) + +/** Fetch the grid and flatten it to text. Cached so a dry run is repeatable. */ +async function gridText() { + const pdf = join(tmpdir(), 'commonapp-reqgrid.pdf') + const txt = join(tmpdir(), 'commonapp-reqgrid.txt') + if (!existsSync(pdf)) { + const res = await fetch(GRID_URL, { signal: AbortSignal.timeout(60000) }) + if (!res.ok) throw new Error(`grid fetch failed: ${res.status}`) + writeFileSync(pdf, Buffer.from(await res.arrayBuffer())) + } + execFileSync('pdftotext', ['-layout', pdf, txt]) + return readFileSync(txt, 'utf8') +} + +/** + * The three rounds the app models. EDII and EAII are parsed and counted but + * not stored: AppDeadlineType has no round for them, and inventing a column + * the app cannot render would be data nobody reads. + */ +const COLUMNS = { ed: 'early_decision', ea: 'early_action', rd: 'regular_decision' } + +async function main() { + const rows = parseGrid(await gridText()) + console.log(`grid: ${rows.length} schools parsed`) + + // A layout change upstream would show up as a collapse in what parses. + // Common App has ~1,100 members and the grid lists every one. + if (rows.length < 900) { + throw new Error(`only ${rows.length} schools parsed — the grid layout has probably changed; check parse.mjs before trusting this`) + } + + // A few schools appear once per program, each with its own deadlines under + // the same name. There is no way to tell which program a student means, so + // take neither rather than pick. + const byName = new Map() + for (const row of rows) { + const key = normaliseName(row.name) + byName.set(key, [...(byName.get(key) ?? []), row]) + } + const ambiguous = [...byName.values()].filter((g) => g.length > 1) + const single = [...byName.values()].filter((g) => g.length === 1).map((g) => g[0]) + if (ambiguous.length) { + console.log(`skipped ${ambiguous.length} listed more than once: ${ambiguous.map((g) => g[0].name).join(', ')}`) + } + + // PostgREST caps a select at 1,000 rows. Without paging this silently saw + // a sixth of the table and matched a sixth of the grid. + const colleges = [] + for (let from = 0; ; from += 1000) { + const { data, error } = await supabase + .from('colleges') + .select('scorecard_id, name, early_decision, early_action, regular_decision') + .order('scorecard_id') + .range(from, from + 999) + if (error) throw new Error(`could not read colleges: ${error.message}`) + colleges.push(...data) + if (data.length < 1000) break + } + console.log(`colleges: ${colleges.length} in the table`) + + const index = new Map() + for (const c of colleges) { + const key = normaliseName(c.name) + if (!index.has(key)) index.set(key, c) + } + const keys = [...index.keys()] + + const updates = [] + const unmatched = [] + const claimed = new Map() + const collisions = [] + let unchanged = 0, conflicts = 0, extraRounds = 0, agreed = 0 + + for (const row of single) { + const key = normaliseName(row.name) + let college = index.get(key) + if (!college && key.length >= 10) { + const near = keys.filter((k) => k.startsWith(key)) + if (near.length === 1) college = index.get(near[0]) + } + if (!college) { unmatched.push(row.name); continue } + // Two grid rows landing on one college means the prefix fallback reached + // too far, and the second would overwrite the first without a word. + const seen = claimed.get(college.scorecard_id) + if (seen) { collisions.push(`${seen} + ${row.name} -> ${college.name}`); continue } + claimed.set(college.scorecard_id, row.name) + if (row.edii || row.eaii) extraRounds += 1 + + const patch = {} + for (const [round, column] of Object.entries(COLUMNS)) { + const raw = row[round] + if (!raw) continue + const value = raw === 'Rolling' ? 'Rolling' : toDisplayDate(raw) + if (!value) continue + const current = college[column] + if (current === value) { agreed += 1; continue } + // Only ever fill a blank. Spot-checking the disagreements showed the + // grid right about Stanford (Jan 5, where we held a wrong Jan 2) and + // wrong about Georgetown (Jan 1, where Jan 10 is the real date), so + // neither source wins outright. The 46 curated rows were checked by + // hand; a bulk import does not get to quietly undo that. Disagreements + // are printed for someone to settle. + if (current) { conflicts += 1; console.log(` differs: ${college.name} ${column}: ours ${current} | grid ${value}`); continue } + patch[column] = value + } + if (Object.keys(patch).length === 0) { unchanged += 1; continue } + updates.push({ scorecard_id: college.scorecard_id, name: college.name, patch }) + } + + console.log(`matched: ${single.length - unmatched.length}/${single.length}`) + console.log(`to write: ${updates.length}`) + console.log(`unchanged: ${unchanged}`) + console.log(`agrees with a value we already had: ${agreed}`) + console.log(`differs from an existing value (left alone): ${conflicts}`) + console.log(`have EDII/EAII we do not store: ${extraRounds}`) + console.log(`unmatched: ${unmatched.length}`) + if (collisions.length) console.log(`collisions: ${collisions.length}\n ${collisions.join('\n ')}`) + if (VERBOSE) { + for (const u of updates) console.log(' ', u.name, JSON.stringify(u.patch)) + console.log(' unmatched:', unmatched.join(' | ')) + } + + if (!APPLY) { console.log('\nDry run — nothing written. Re-run with --apply to write.'); return } + + let written = 0 + for (const u of updates) { + const { error: e } = await supabase + .from('colleges') + .update({ ...u.patch, verified_at: new Date().toISOString() }) + .eq('scorecard_id', u.scorecard_id) + if (e) { console.error(` failed ${u.name}: ${e.message}`); continue } + written += 1 + } + console.log(`\nwrote ${written}/${updates.length}`) +} + +main().catch((e) => { console.error(e.message); process.exit(1) }) diff --git a/scripts/ingest-deadlines/package-lock.json b/scripts/ingest-deadlines/package-lock.json new file mode 100644 index 0000000..2042d56 --- /dev/null +++ b/scripts/ingest-deadlines/package-lock.json @@ -0,0 +1,122 @@ +{ + "name": "ingest-deadlines", + "version": "1.0.0", + "lockfileVersion": 3, + "requires": true, + "packages": { + "": { + "name": "ingest-deadlines", + "version": "1.0.0", + "dependencies": { + "@supabase/supabase-js": "^2.110.3" + } + }, + "node_modules/@supabase/auth-js": { + "version": "2.117.1", + "resolved": "https://registry.npmjs.org/@supabase/auth-js/-/auth-js-2.117.1.tgz", + "integrity": "sha512-3vJpc5thxaovbZ2zrWagzp/ZtCcXpQk1xftinzneNB1XLdhAmjqChOnj2Y8wpriYEFVvlqC+sFbR0JBUi83hDA==", + "license": "MIT", + "dependencies": { + "tslib": "2.8.1" + }, + "engines": { + "node": ">=22.0.0" + } + }, + "node_modules/@supabase/functions-js": { + "version": "2.117.1", + "resolved": "https://registry.npmjs.org/@supabase/functions-js/-/functions-js-2.117.1.tgz", + "integrity": "sha512-yyMIbEPMDXmjpjWe7WKd7en3mn6q4eDGLHEQVPcqWQIyC1hC7xa3po+o4cWSyItI6a6GYY/voDUpwhZmkxjcPw==", + "license": "MIT", + "dependencies": { + "tslib": "2.8.1" + }, + "engines": { + "node": ">=22.0.0" + } + }, + "node_modules/@supabase/phoenix": { + "version": "0.4.5", + "resolved": "https://registry.npmjs.org/@supabase/phoenix/-/phoenix-0.4.5.tgz", + "integrity": "sha512-aAn9H9ovVyeApKy11OWOrrOGq8DV68yWeH4ud2lN9fzn4aO8Zb5GLL9m1pUg9nLqIcT+ZDfAcsZe0E/nqdv2lw==", + "license": "MIT" + }, + "node_modules/@supabase/postgrest-js": { + "version": "2.117.1", + "resolved": "https://registry.npmjs.org/@supabase/postgrest-js/-/postgrest-js-2.117.1.tgz", + "integrity": "sha512-2a7we4uKXA1d94CUc02MpUj+LzZWO3qm8fH2PDq60tVt0T6KQTIiMmEsRD53lrnwZA9C3SsaKL3mk4DiDH6GvQ==", + "license": "MIT", + "dependencies": { + "tslib": "2.8.1" + }, + "engines": { + "node": ">=22.0.0" + } + }, + "node_modules/@supabase/realtime-js": { + "version": "2.117.1", + "resolved": "https://registry.npmjs.org/@supabase/realtime-js/-/realtime-js-2.117.1.tgz", + "integrity": "sha512-GkbZfeh5F5aqnUhG1chrdaVQmnRK8ZUL8H16Km/eVVsX99x0w2AijfxjONCB9YKgrXArcnMsKSJzOIQGCv+qvQ==", + "license": "MIT", + "dependencies": { + "@supabase/phoenix": "0.4.5", + "tslib": "2.8.1" + }, + "engines": { + "node": ">=22.0.0" + } + }, + "node_modules/@supabase/storage-js": { + "version": "2.117.1", + "resolved": "https://registry.npmjs.org/@supabase/storage-js/-/storage-js-2.117.1.tgz", + "integrity": "sha512-ytDQbELKKzIVStLkShxTTK5FT4vds12eXF5QfW+gTHVvO7cB3VKZ7Hu2wkUaJlDZCaUfFdih3vznF6P8kNkJQg==", + "license": "MIT", + "dependencies": { + "iceberg-js": "^0.8.1", + "tslib": "2.8.1" + }, + "engines": { + "node": ">=22.0.0" + } + }, + "node_modules/@supabase/supabase-js": { + "version": "2.117.1", + "resolved": "https://registry.npmjs.org/@supabase/supabase-js/-/supabase-js-2.117.1.tgz", + "integrity": "sha512-VcyiuXF0jZFbN4GcHtytSGNwfXnRJwlxjGFZxeEmRVgSLSbaoHk2Avyg49/aEdTJGBSz5vK1L7uhYiYdfl0Axw==", + "license": "MIT", + "dependencies": { + "@supabase/auth-js": "2.117.1", + "@supabase/functions-js": "2.117.1", + "@supabase/postgrest-js": "2.117.1", + "@supabase/realtime-js": "2.117.1", + "@supabase/storage-js": "2.117.1" + }, + "engines": { + "node": ">=22.0.0" + }, + "peerDependencies": { + "@opentelemetry/api": ">=1.0.0" + }, + "peerDependenciesMeta": { + "@opentelemetry/api": { + "optional": true + } + } + }, + "node_modules/iceberg-js": { + "version": "0.8.1", + "resolved": "https://registry.npmjs.org/iceberg-js/-/iceberg-js-0.8.1.tgz", + "integrity": "sha512-1dhVQZXhcHje7798IVM+xoo/1ZdVfzOMIc8/rgVSijRK38EDqOJoGula9N/8ZI5RD8QTxNQtK/Gozpr+qUqRRA==", + "license": "MIT", + "engines": { + "node": ">=20.0.0" + } + }, + "node_modules/tslib": { + "version": "2.8.1", + "resolved": "https://registry.npmjs.org/tslib/-/tslib-2.8.1.tgz", + "integrity": "sha512-oJFu94HQb+KVduSUQL7wnpmqnfmLsOA/nAh6b6EH0wCEoK0/mPeXU6c3wKDV83MkOuHPRHtSXKKU99IBazS/2w==", + "license": "0BSD" + } + } +} diff --git a/scripts/ingest-deadlines/package.json b/scripts/ingest-deadlines/package.json new file mode 100644 index 0000000..6824a56 --- /dev/null +++ b/scripts/ingest-deadlines/package.json @@ -0,0 +1,10 @@ +{ + "name": "ingest-deadlines", + "version": "1.0.0", + "private": true, + "type": "module", + "description": "Common App requirements grid -> Supabase colleges deadline columns", + "dependencies": { + "@supabase/supabase-js": "^2.110.3" + } +} diff --git a/scripts/ingest-deadlines/parse.mjs b/scripts/ingest-deadlines/parse.mjs new file mode 100644 index 0000000..61e6bf1 --- /dev/null +++ b/scripts/ingest-deadlines/parse.mjs @@ -0,0 +1,182 @@ +// Parsing the Common App requirements grid. +// +// Kept separate from the ingest so it can be tested without a network or a +// database. Everything fragile about this job lives here. +// +// The grid is a PDF laid out as a table. `pdftotext -layout` preserves the +// columns as whitespace, but not perfectly: values drift a few characters +// from the header above them, and long school names wrap onto the line +// before or after their own row. So a round is identified by where its date +// sits on the line, not by slicing at fixed offsets. + +/** The rounds the grid publishes, in the order its columns appear. */ +export const ROUNDS = ['ed', 'edii', 'ea', 'eaii', 'rea', 'rd'] + +/** Header label for each round; 'Rolling' heads the RD column. */ +const HEADER_LABEL = { ed: 'ED', edii: 'EDII', ea: 'EA', eaii: 'EAII', rea: 'REA', rd: 'Rolling' } + +/** + * Column positions for one page, read off that page's own header. + * + * The grid repeats its header on every page and the columns are *not* in the + * same place on each: across 55 pages ED alone sits anywhere from column 36 + * to 39, and the drift compounds to the right — far enough that on one page + * Yale's restrictive-early date lands where another page keeps regular + * decision. Fixed offsets silently mis-file those; the header is the only + * thing that says where the columns actually are. + */ +export function columnsFrom(headerLine) { + const cols = {} + for (const round of ROUNDS) { + // \b so ED does not match inside EDII, and EA not inside EAII. + const m = new RegExp(`\\b${HEADER_LABEL[round]}\\b`).exec(headerLine) + if (m) cols[round] = m.index + } + if (Object.keys(cols).length !== ROUNDS.length) return null + // 'type' heads the school-type column. Names live to its left, so it tells + // us whether a line's first field is a name at all. + const type = /\btype\b/.exec(headerLine) + return type ? { ...cols, typeAt: type.index } : null +} + +/** + * The school name: the line's first whitespace-delimited field. + * + * Names use single spaces; the gap to the school-type column is always + * several. That holds whatever the page's column offsets are, which slicing + * at a fixed or even header-derived position does not — the type column can + * sit either side of its own header word, so a slice takes the 'C' off + * 'Coordinate' and leaves it glued to the name. + */ +/** The closed set of school-type values; never a school name. */ +const SCHOOL_TYPE = /^(Coed|Coordinate|Men|Women)[\d¹²³*]*$/ + +const nameOn = (line, cols) => { + const text = line.slice(0, ROW_END) + const start = text.search(/\S/) + // When a name wraps, its own row carries no name and starts at the school + // -type column instead. Taking the first field regardless yields 'Coed'. + if (start < 0 || start >= cols.typeAt) return '' + const first = text.trim().split(/\s{2,}/)[0].trim() + // 'Coordinate' is long enough to begin left of its own header word, so the + // column guard alone lets it through. + return SCHOOL_TYPE.test(first) ? '' : first +} + +/** True for a line that is one of the repeated column headers. */ +export const isHeaderLine = (line) => line.includes('EDII') && line.includes('EAII') + +/** The round whose column a date at `pos` belongs to: simply the nearest. */ +export function roundFor(pos, cols) { + let best = null, bestDist = Infinity + for (const [round, at] of Object.entries(cols)) { + const d = Math.abs(pos - at) + if (d < bestDist) { bestDist = d; best = round } + } + // Beyond this the line is into the fees columns, not a deadline. + return bestDist <= 12 ? best : null +} + +const DATE = /\d{1,2}\/\d{1,2}\/\d{4}/g +/** Dates and 'Rolling' never appear beyond here; fees and flags follow. */ +const ROW_END = 112 +/** Header, footnote and title fragments that are not schools. */ +const NOT_A_SCHOOL = [ + 'common app', 'member', '2026-27', 'first-year', 'school', 'deadlines', + 'application fee', 'updated:', 'see bottom', +] + +const isNoise = (s) => { + const t = s.toLowerCase() + return !t || t.startsWith('*') || t.startsWith('¹') || NOT_A_SCHOOL.some((p) => t.startsWith(p)) +} + +/** True for a line carrying only the tail of a school name. */ +const isNameOnly = (line, cols) => { + const fields = line.slice(0, ROW_END).trim().split(/\s{2,}/) + return fields.length === 1 && !!nameOn(line, cols) && !isNoise(fields[0]) +} + +/** + * Every school in the grid, with whichever rounds it publishes. + * + * `rd` may be the literal 'Rolling', which is not a date and is deliberately + * kept: "no fixed deadline" is a true and useful answer, and the app already + * understands Rolling as a round. + */ +export function parseGrid(text) { + const lines = text.split('\n') + const out = [] + let pending = [] + let cols = null + + for (let i = 0; i < lines.length; i++) { + const line = lines[i] + if (!line.trim()) { pending = []; continue } + + if (isHeaderLine(line)) { + const found = columnsFrom(line) + if (found) cols = found + pending = [] + continue + } + if (!cols) continue + + const head = line.slice(0, ROW_END) + const dates = [...head.matchAll(DATE)].map((m) => [m.index, m[0]]) + const rollingAt = head.indexOf('Rolling') + + if (dates.length === 0 && rollingAt < 0) { + const name = nameOn(line, cols) + if (!isNoise(name)) pending.push(name) + continue + } + + // A long name spills onto the lines before *and* after its own row. + const parts = [...pending, nameOn(line, cols)] + pending = [] + for (let j = i + 1; j <= i + 2 && j < lines.length; j++) { + if (!isNameOnly(lines[j], cols)) break + parts.push(nameOn(lines[j], cols)) + } + + const name = parts.filter(Boolean).join(' ').replace(/\s+/g, ' ').trim() + if (name.length < 3 || isNoise(name)) continue + + const row = { name } + for (const [pos, value] of dates) { + const round = roundFor(pos, cols) + if (round && !row[round]) row[round] = value + } + if (rollingAt >= 0 && roundFor(rollingAt, cols) === 'rd' && !row.rd) row.rd = 'Rolling' + if (ROUNDS.some((k) => row[k])) out.push(row) + } + return out +} + +const MONTHS = ['Jan', 'Feb', 'Mar', 'Apr', 'May', 'Jun', 'Jul', 'Aug', 'Sep', 'Oct', 'Nov', 'Dec'] + +/** + * `11/01/2026` to `Nov 1, 2026`, the shape the app already parses and the + * curated rows already use. Returns null for anything that is not a date, so + * 'Rolling' passes through untouched by the caller. + */ +export function toDisplayDate(value) { + if (!value || value === 'Rolling') return null + const m = /^(\d{1,2})\/(\d{1,2})\/(\d{4})$/.exec(value.trim()) + if (!m) return null + const month = Number(m[1]), day = Number(m[2]), year = Number(m[3]) + if (month < 1 || month > 12 || day < 1 || day > 31) return null + return `${MONTHS[month - 1]} ${day}, ${year}` +} + +/** Compare names loosely enough to survive "Univ." and "St." but no looser. */ +export function normaliseName(name) { + return name + .toLowerCase() + .replace(/\b(the|of|at|and)\b/g, ' ') + .replace(/[^a-z0-9]+/g, '') + .replace(/university/g, 'univ') + .replace(/college/g, 'coll') + .replace(/saint/g, 'st') +} diff --git a/scripts/ingest-deadlines/parse.test.mjs b/scripts/ingest-deadlines/parse.test.mjs new file mode 100644 index 0000000..cf3b24e --- /dev/null +++ b/scripts/ingest-deadlines/parse.test.mjs @@ -0,0 +1,85 @@ +// Run: node --test scripts/ingest-deadlines/ +// +// Fixtures are verbatim lines from the real PDF. The two headers below have +// genuinely different column offsets — that is the whole point: the grid +// repeats its header on every page and moves the columns each time. + +import test from 'node:test' +import assert from 'node:assert/strict' +import { parseGrid, toDisplayDate, normaliseName } from './parse.mjs' + +const HDR_NARROW = + " member type ¹ ED EDII EA EAII REA Rolling US Int'l" +const HDR_WIDE = + " member type ¹ ED EDII EA EAII REA Rolling US" + +const grid = (...lines) => parseGrid(lines.join('\n')) +const find = (rows, name) => rows.find((r) => r.name === name) + +test('a round is read from its own page’s header, not a fixed offset', () => { + // Yale publishes REA and RD. On the wide page its REA date sits where the + // narrow page keeps regular decision; fixed bands called this RD = Nov 1. + const rows = grid( + HDR_WIDE, + ' Yale University Coed 11/1/2026 01/02/2027 $85', + ) + assert.deepEqual(find(rows, 'Yale University'), { + name: 'Yale University', rea: '11/1/2026', rd: '01/02/2027', + }) +}) + +test('the same column position means different rounds on different pages', () => { + // Column 66 is REA on the narrow page and EA on the wide one; the two + // headers here sit 28 characters apart by the time they reach RD. + const head = ' A College Coed' + const row = head + ' '.repeat(66 - head.length) + '11/1/2026' + const roundOn = (hdr) => Object.keys(find(grid(hdr, row), 'A College'))[1] + assert.equal(roundOn(HDR_NARROW), 'rea') + assert.equal(roundOn(HDR_WIDE), 'ea') +}) + +test('a name wrapped across three lines is rejoined', () => { + const rows = grid( + " member type \u00b9 ED EDII EA EAII REA Rolling US", + ' Georgia Institute of', + ' Coed 10/15/2026 11/2/2026 01/06/2027 $75', + ' Technology', + ) + assert.deepEqual(find(rows, 'Georgia Institute of Technology'), { + name: 'Georgia Institute of Technology', ea: '10/15/2026', eaii: '11/2/2026', rd: '01/06/2027', + }) +}) + +test('the school-type value is never mistaken for a name', () => { + // 'Coordinate' is long enough to start left of its own header word. + const rows = grid( + " member type \u00b9 ED EDII EA EAII REA Rolling US", + ' College of Saint', + ' Coordinate 11/1/2026 12/1/2026 01/15/2027', + ' Benedict', + ) + assert.equal(rows.length, 1) + assert.equal(rows[0].name, 'College of Saint Benedict') +}) + +test('Rolling is kept, but only in the RD column', () => { + const rows = grid(HDR_NARROW, ' SUNY Delhi Coed Rolling $50 $50') + assert.equal(find(rows, 'SUNY Delhi').rd, 'Rolling') +}) + +test('rows before the first header are skipped rather than guessed at', () => { + assert.deepEqual(grid(' Some College Coed 11/1/2026'), []) +}) + +test('toDisplayDate matches the shape the app already stores', () => { + assert.equal(toDisplayDate('11/01/2026'), 'Nov 1, 2026') + assert.equal(toDisplayDate('1/4/2027'), 'Jan 4, 2027') + assert.equal(toDisplayDate('Rolling'), null) + assert.equal(toDisplayDate('13/1/2027'), null) +}) + +test('normaliseName survives the abbreviations the two sources disagree on', () => { + assert.equal(normaliseName('The University of Texas at Austin'), normaliseName('University of Texas-Austin')) + assert.equal(normaliseName('Saint Johns University'), normaliseName('St. Johns University')) + assert.notEqual(normaliseName('Miami University'), normaliseName('University of Miami')) +})