From fce95982a615ce8797439d994dad7377c6a27b17 Mon Sep 17 00:00:00 2001 From: Danial Beg Date: Thu, 24 Sep 2026 22:41:38 -0700 Subject: [PATCH] Fill in college deadlines from the Common App grid MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 46 of 6,273 colleges had a real application deadline. Everything else fell back to a typical date for the round and was shown to the student as an estimate, which is the one thing a deadline tool should not do. Common App publishes its members' deadlines as a PDF and keeps it current through the cycle, so this is re-runnable rather than a one-off import. A dry run fills 935 of them and overwrites nothing. The grid repeats its column header on all 55 pages and the columns land in a different place on each: ED sits anywhere from column 36 to 39, and the drift compounds rightward until two pages are 28 characters apart at RD. So each round is read against its own page's header. Fixed offsets are not merely untidy here — they read Yale's restrictive-early date as its regular-decision deadline, because on Yale's page REA sits where other pages keep RD. parse.test.mjs pins that case and the four others that broke. Existing values are never overwritten. Checking the disagreements found the grid right about Stanford (Jan 5, where we held a wrong Jan 2) and wrong about Georgetown (Jan 1, where Jan 10 is real), so neither source wins outright; the 46 curated rows were checked by hand and a bulk import does not get to undo that quietly. The 13 disagreements print for someone to settle. The 30 values that agree exactly are the main evidence the parse is sound. Nor does it guess between campuses: Arizona State, Kent State and Colorado State have six or more campus rows each, so they stay unmatched rather than risk putting a main campus's deadline on a branch. Writing takes --apply. It used to be the no-flag default, which puts one mistyped invocation straight into production. --- scripts/ingest-deadlines/README.md | 69 ++++++++ scripts/ingest-deadlines/ingest.mjs | 178 ++++++++++++++++++++ scripts/ingest-deadlines/package-lock.json | 122 ++++++++++++++ scripts/ingest-deadlines/package.json | 10 ++ scripts/ingest-deadlines/parse.mjs | 182 +++++++++++++++++++++ scripts/ingest-deadlines/parse.test.mjs | 85 ++++++++++ 6 files changed, 646 insertions(+) create mode 100644 scripts/ingest-deadlines/README.md create mode 100644 scripts/ingest-deadlines/ingest.mjs create mode 100644 scripts/ingest-deadlines/package-lock.json create mode 100644 scripts/ingest-deadlines/package.json create mode 100644 scripts/ingest-deadlines/parse.mjs create mode 100644 scripts/ingest-deadlines/parse.test.mjs diff --git a/scripts/ingest-deadlines/README.md b/scripts/ingest-deadlines/README.md new file mode 100644 index 0000000..169f040 --- /dev/null +++ b/scripts/ingest-deadlines/README.md @@ -0,0 +1,69 @@ +# Common App deadline ingest + +Fills in application deadlines on `public.colleges` from the +[Common App requirements grid](https://content.commonapp.org/Files/ReqGrid.pdf), +the PDF Common App publishes and keeps current through the cycle. + +**46** of 6,273 colleges had a real deadline. Everything else fell back to a +typical date for the round and was shown to the student as an estimate. This +brings that to **~980**. + +## Run + +```bash +cd scripts/ingest-deadlines +node --test parse.test.mjs # the parser is the fragile part; test it first +node ingest.mjs # dry run: prints what it would change, writes nothing +node ingest.mjs -v # the same, plus every row +node ingest.mjs --apply # write +``` + +Needs `pdftotext` (`brew install poppler`) on PATH. Reads Supabase credentials +from `web/.env.local`. Re-runnable: it only ever fills blanks. + +Current run: 1,127 schools parsed, 955 matched, **935 rows to fill**, 0 overwritten. + +## What it will not do + +**It never overwrites a value that is already there.** Spot-checking the +disagreements found the grid right about Stanford (Jan 5, where we held a wrong +Jan 2) and wrong about Georgetown (Jan 1, where Jan 10 is the real date), so +neither source wins outright. The 46 curated rows were checked by hand and a +bulk import does not get to quietly undo that. Disagreements print as +`differs:` lines for someone to settle by hand — there are 13, and 30 other +values agree exactly, which is the main evidence the parse is sound. + +**It does not guess between campuses.** A grid row matches a college by +normalised name, or by a name prefix when exactly one college matches. Arizona +State, Kent State and Colorado State each have six or more campus rows, so +those stay unmatched rather than risk putting the main campus's deadline on a +branch. Of 169 unmatched, most are foreign universities absent from the +Scorecard dataset (Aberystwyth, Dublin City, Esade, IE) and the rest are +multi-campus systems. + +**It does not store EDII or EAII.** 185 schools publish them; the app's +`AppDeadlineType` has no round for them, and a column nothing renders is data +nobody reads. + +**Schools listed more than once are skipped** — a few appear once per program +with different deadlines under one name, and there is no way to tell which +program a student means. + +## The parser + +`parse.mjs` is where everything fragile lives, which is why it is separate and +tested without a network or a database. + +The grid repeats its column header on every one of its 55 pages, and **the +columns are in a different place on each**. Across pages `ED` sits anywhere +from column 36 to 39 and the drift compounds rightward — by `RD` two pages can +be 28 characters apart. So a round is identified by calibrating against that +page's own header, never a fixed offset. + +This was not theoretical: fixed bands read Yale's restrictive-early date as its +regular-decision deadline, because on Yale's page REA sits where other pages +keep RD. `parse.test.mjs` pins that case and the other four that broke. + +If Common App changes the layout, the parser needs revisiting, and it is built +to fail loudly rather than mis-file a round: `ingest.mjs` throws if fewer than +900 schools parse, and refuses to let two grid rows claim one college. diff --git a/scripts/ingest-deadlines/ingest.mjs b/scripts/ingest-deadlines/ingest.mjs new file mode 100644 index 0000000..3c86fd8 --- /dev/null +++ b/scripts/ingest-deadlines/ingest.mjs @@ -0,0 +1,178 @@ +// Common App requirements grid -> Supabase `colleges` deadline columns. +// +// Usage: +// node ingest.mjs # parse, match, print what would change. Writes nothing. +// node ingest.mjs -v # the same, plus every row it would write +// node ingest.mjs --apply # write +// +// Writing needs --apply. It used to be the no-flag default, which meant one +// mistyped invocation went straight at production. +// +// Why this exists: 6,273 colleges are in the table and 46 had a real +// application deadline. Everything else fell back to a typical date for the +// round, shown to the student as "no date on file". Common App publishes a +// grid of its members' deadlines and keeps it current through the cycle, so +// this is re-runnable rather than a one-off import. +// +// Needs `pdftotext` (poppler) on PATH. Reads VITE_SUPABASE_URL and +// VITE_SUPABASE_ANON_KEY from ../../web/.env.local. + +import { readFileSync, writeFileSync, existsSync } from 'node:fs' +import { execFileSync } from 'node:child_process' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { createClient } from '@supabase/supabase-js' +import { parseGrid, toDisplayDate, normaliseName } from './parse.mjs' + +const APPLY = process.argv.includes('--apply') +const VERBOSE = process.argv.includes('-v') +const GRID_URL = 'https://content.commonapp.org/Files/ReqGrid.pdf' + +function loadEnv(path) { + const out = {} + for (const line of readFileSync(path, 'utf8').split('\n')) { + const m = line.match(/^([A-Z0-9_]+)=(.*)$/) + if (m) out[m[1]] = m[2].trim() + } + return out +} + +const web = loadEnv(new URL('../../web/.env.local', import.meta.url).pathname) +const supabase = createClient(web.VITE_SUPABASE_URL, web.VITE_SUPABASE_ANON_KEY) + +/** Fetch the grid and flatten it to text. Cached so a dry run is repeatable. */ +async function gridText() { + const pdf = join(tmpdir(), 'commonapp-reqgrid.pdf') + const txt = join(tmpdir(), 'commonapp-reqgrid.txt') + if (!existsSync(pdf)) { + const res = await fetch(GRID_URL, { signal: AbortSignal.timeout(60000) }) + if (!res.ok) throw new Error(`grid fetch failed: ${res.status}`) + writeFileSync(pdf, Buffer.from(await res.arrayBuffer())) + } + execFileSync('pdftotext', ['-layout', pdf, txt]) + return readFileSync(txt, 'utf8') +} + +/** + * The three rounds the app models. EDII and EAII are parsed and counted but + * not stored: AppDeadlineType has no round for them, and inventing a column + * the app cannot render would be data nobody reads. + */ +const COLUMNS = { ed: 'early_decision', ea: 'early_action', rd: 'regular_decision' } + +async function main() { + const rows = parseGrid(await gridText()) + console.log(`grid: ${rows.length} schools parsed`) + + // A layout change upstream would show up as a collapse in what parses. + // Common App has ~1,100 members and the grid lists every one. + if (rows.length < 900) { + throw new Error(`only ${rows.length} schools parsed — the grid layout has probably changed; check parse.mjs before trusting this`) + } + + // A few schools appear once per program, each with its own deadlines under + // the same name. There is no way to tell which program a student means, so + // take neither rather than pick. + const byName = new Map() + for (const row of rows) { + const key = normaliseName(row.name) + byName.set(key, [...(byName.get(key) ?? []), row]) + } + const ambiguous = [...byName.values()].filter((g) => g.length > 1) + const single = [...byName.values()].filter((g) => g.length === 1).map((g) => g[0]) + if (ambiguous.length) { + console.log(`skipped ${ambiguous.length} listed more than once: ${ambiguous.map((g) => g[0].name).join(', ')}`) + } + + // PostgREST caps a select at 1,000 rows. Without paging this silently saw + // a sixth of the table and matched a sixth of the grid. + const colleges = [] + for (let from = 0; ; from += 1000) { + const { data, error } = await supabase + .from('colleges') + .select('scorecard_id, name, early_decision, early_action, regular_decision') + .order('scorecard_id') + .range(from, from + 999) + if (error) throw new Error(`could not read colleges: ${error.message}`) + colleges.push(...data) + if (data.length < 1000) break + } + console.log(`colleges: ${colleges.length} in the table`) + + const index = new Map() + for (const c of colleges) { + const key = normaliseName(c.name) + if (!index.has(key)) index.set(key, c) + } + const keys = [...index.keys()] + + const updates = [] + const unmatched = [] + const claimed = new Map() + const collisions = [] + let unchanged = 0, conflicts = 0, extraRounds = 0, agreed = 0 + + for (const row of single) { + const key = normaliseName(row.name) + let college = index.get(key) + if (!college && key.length >= 10) { + const near = keys.filter((k) => k.startsWith(key)) + if (near.length === 1) college = index.get(near[0]) + } + if (!college) { unmatched.push(row.name); continue } + // Two grid rows landing on one college means the prefix fallback reached + // too far, and the second would overwrite the first without a word. + const seen = claimed.get(college.scorecard_id) + if (seen) { collisions.push(`${seen} + ${row.name} -> ${college.name}`); continue } + claimed.set(college.scorecard_id, row.name) + if (row.edii || row.eaii) extraRounds += 1 + + const patch = {} + for (const [round, column] of Object.entries(COLUMNS)) { + const raw = row[round] + if (!raw) continue + const value = raw === 'Rolling' ? 'Rolling' : toDisplayDate(raw) + if (!value) continue + const current = college[column] + if (current === value) { agreed += 1; continue } + // Only ever fill a blank. Spot-checking the disagreements showed the + // grid right about Stanford (Jan 5, where we held a wrong Jan 2) and + // wrong about Georgetown (Jan 1, where Jan 10 is the real date), so + // neither source wins outright. The 46 curated rows were checked by + // hand; a bulk import does not get to quietly undo that. Disagreements + // are printed for someone to settle. + if (current) { conflicts += 1; console.log(` differs: ${college.name} ${column}: ours ${current} | grid ${value}`); continue } + patch[column] = value + } + if (Object.keys(patch).length === 0) { unchanged += 1; continue } + updates.push({ scorecard_id: college.scorecard_id, name: college.name, patch }) + } + + console.log(`matched: ${single.length - unmatched.length}/${single.length}`) + console.log(`to write: ${updates.length}`) + console.log(`unchanged: ${unchanged}`) + console.log(`agrees with a value we already had: ${agreed}`) + console.log(`differs from an existing value (left alone): ${conflicts}`) + console.log(`have EDII/EAII we do not store: ${extraRounds}`) + console.log(`unmatched: ${unmatched.length}`) + if (collisions.length) console.log(`collisions: ${collisions.length}\n ${collisions.join('\n ')}`) + if (VERBOSE) { + for (const u of updates) console.log(' ', u.name, JSON.stringify(u.patch)) + console.log(' unmatched:', unmatched.join(' | ')) + } + + if (!APPLY) { console.log('\nDry run — nothing written. Re-run with --apply to write.'); return } + + let written = 0 + for (const u of updates) { + const { error: e } = await supabase + .from('colleges') + .update({ ...u.patch, verified_at: new Date().toISOString() }) + .eq('scorecard_id', u.scorecard_id) + if (e) { console.error(` failed ${u.name}: ${e.message}`); continue } + written += 1 + } + console.log(`\nwrote ${written}/${updates.length}`) +} + +main().catch((e) => { console.error(e.message); process.exit(1) }) diff --git a/scripts/ingest-deadlines/package-lock.json b/scripts/ingest-deadlines/package-lock.json new file mode 100644 index 0000000..2042d56 --- /dev/null +++ b/scripts/ingest-deadlines/package-lock.json @@ -0,0 +1,122 @@ +{ + "name": "ingest-deadlines", + "version": "1.0.0", + "lockfileVersion": 3, + "requires": true, + "packages": { + "": { + "name": "ingest-deadlines", + "version": "1.0.0", + "dependencies": { + "@supabase/supabase-js": "^2.110.3" + } + }, + "node_modules/@supabase/auth-js": { + "version": "2.117.1", + "resolved": "https://registry.npmjs.org/@supabase/auth-js/-/auth-js-2.117.1.tgz", + "integrity": "sha512-3vJpc5thxaovbZ2zrWagzp/ZtCcXpQk1xftinzneNB1XLdhAmjqChOnj2Y8wpriYEFVvlqC+sFbR0JBUi83hDA==", + "license": "MIT", + "dependencies": { + "tslib": "2.8.1" + }, + "engines": { + "node": ">=22.0.0" + } + }, + "node_modules/@supabase/functions-js": { + "version": "2.117.1", + "resolved": "https://registry.npmjs.org/@supabase/functions-js/-/functions-js-2.117.1.tgz", + "integrity": "sha512-yyMIbEPMDXmjpjWe7WKd7en3mn6q4eDGLHEQVPcqWQIyC1hC7xa3po+o4cWSyItI6a6GYY/voDUpwhZmkxjcPw==", + "license": "MIT", + "dependencies": { + "tslib": "2.8.1" + }, + "engines": { + "node": ">=22.0.0" + } + }, + "node_modules/@supabase/phoenix": { + "version": "0.4.5", + "resolved": "https://registry.npmjs.org/@supabase/phoenix/-/phoenix-0.4.5.tgz", + "integrity": "sha512-aAn9H9ovVyeApKy11OWOrrOGq8DV68yWeH4ud2lN9fzn4aO8Zb5GLL9m1pUg9nLqIcT+ZDfAcsZe0E/nqdv2lw==", + "license": "MIT" + }, + "node_modules/@supabase/postgrest-js": { + "version": "2.117.1", + "resolved": "https://registry.npmjs.org/@supabase/postgrest-js/-/postgrest-js-2.117.1.tgz", + "integrity": "sha512-2a7we4uKXA1d94CUc02MpUj+LzZWO3qm8fH2PDq60tVt0T6KQTIiMmEsRD53lrnwZA9C3SsaKL3mk4DiDH6GvQ==", + "license": "MIT", + "dependencies": { + "tslib": "2.8.1" + }, + "engines": { + "node": ">=22.0.0" + } + }, + "node_modules/@supabase/realtime-js": { + "version": "2.117.1", + "resolved": "https://registry.npmjs.org/@supabase/realtime-js/-/realtime-js-2.117.1.tgz", + "integrity": "sha512-GkbZfeh5F5aqnUhG1chrdaVQmnRK8ZUL8H16Km/eVVsX99x0w2AijfxjONCB9YKgrXArcnMsKSJzOIQGCv+qvQ==", + "license": "MIT", + "dependencies": { + "@supabase/phoenix": "0.4.5", + "tslib": "2.8.1" + }, + "engines": { + "node": ">=22.0.0" + } + }, + "node_modules/@supabase/storage-js": { + "version": "2.117.1", + "resolved": "https://registry.npmjs.org/@supabase/storage-js/-/storage-js-2.117.1.tgz", + "integrity": "sha512-ytDQbELKKzIVStLkShxTTK5FT4vds12eXF5QfW+gTHVvO7cB3VKZ7Hu2wkUaJlDZCaUfFdih3vznF6P8kNkJQg==", + "license": "MIT", + "dependencies": { + "iceberg-js": "^0.8.1", + "tslib": "2.8.1" + }, + "engines": { + "node": ">=22.0.0" + } + }, + "node_modules/@supabase/supabase-js": { + "version": "2.117.1", + "resolved": "https://registry.npmjs.org/@supabase/supabase-js/-/supabase-js-2.117.1.tgz", + "integrity": "sha512-VcyiuXF0jZFbN4GcHtytSGNwfXnRJwlxjGFZxeEmRVgSLSbaoHk2Avyg49/aEdTJGBSz5vK1L7uhYiYdfl0Axw==", + "license": "MIT", + "dependencies": { + "@supabase/auth-js": "2.117.1", + "@supabase/functions-js": "2.117.1", + "@supabase/postgrest-js": "2.117.1", + "@supabase/realtime-js": "2.117.1", + "@supabase/storage-js": "2.117.1" + }, + "engines": { + "node": ">=22.0.0" + }, + "peerDependencies": { + "@opentelemetry/api": ">=1.0.0" + }, + "peerDependenciesMeta": { + "@opentelemetry/api": { + "optional": true + } + } + }, + "node_modules/iceberg-js": { + "version": "0.8.1", + "resolved": "https://registry.npmjs.org/iceberg-js/-/iceberg-js-0.8.1.tgz", + "integrity": "sha512-1dhVQZXhcHje7798IVM+xoo/1ZdVfzOMIc8/rgVSijRK38EDqOJoGula9N/8ZI5RD8QTxNQtK/Gozpr+qUqRRA==", + "license": "MIT", + "engines": { + "node": ">=20.0.0" + } + }, + "node_modules/tslib": { + "version": "2.8.1", + "resolved": "https://registry.npmjs.org/tslib/-/tslib-2.8.1.tgz", + "integrity": "sha512-oJFu94HQb+KVduSUQL7wnpmqnfmLsOA/nAh6b6EH0wCEoK0/mPeXU6c3wKDV83MkOuHPRHtSXKKU99IBazS/2w==", + "license": "0BSD" + } + } +} diff --git a/scripts/ingest-deadlines/package.json b/scripts/ingest-deadlines/package.json new file mode 100644 index 0000000..6824a56 --- /dev/null +++ b/scripts/ingest-deadlines/package.json @@ -0,0 +1,10 @@ +{ + "name": "ingest-deadlines", + "version": "1.0.0", + "private": true, + "type": "module", + "description": "Common App requirements grid -> Supabase colleges deadline columns", + "dependencies": { + "@supabase/supabase-js": "^2.110.3" + } +} diff --git a/scripts/ingest-deadlines/parse.mjs b/scripts/ingest-deadlines/parse.mjs new file mode 100644 index 0000000..61e6bf1 --- /dev/null +++ b/scripts/ingest-deadlines/parse.mjs @@ -0,0 +1,182 @@ +// Parsing the Common App requirements grid. +// +// Kept separate from the ingest so it can be tested without a network or a +// database. Everything fragile about this job lives here. +// +// The grid is a PDF laid out as a table. `pdftotext -layout` preserves the +// columns as whitespace, but not perfectly: values drift a few characters +// from the header above them, and long school names wrap onto the line +// before or after their own row. So a round is identified by where its date +// sits on the line, not by slicing at fixed offsets. + +/** The rounds the grid publishes, in the order its columns appear. */ +export const ROUNDS = ['ed', 'edii', 'ea', 'eaii', 'rea', 'rd'] + +/** Header label for each round; 'Rolling' heads the RD column. */ +const HEADER_LABEL = { ed: 'ED', edii: 'EDII', ea: 'EA', eaii: 'EAII', rea: 'REA', rd: 'Rolling' } + +/** + * Column positions for one page, read off that page's own header. + * + * The grid repeats its header on every page and the columns are *not* in the + * same place on each: across 55 pages ED alone sits anywhere from column 36 + * to 39, and the drift compounds to the right — far enough that on one page + * Yale's restrictive-early date lands where another page keeps regular + * decision. Fixed offsets silently mis-file those; the header is the only + * thing that says where the columns actually are. + */ +export function columnsFrom(headerLine) { + const cols = {} + for (const round of ROUNDS) { + // \b so ED does not match inside EDII, and EA not inside EAII. + const m = new RegExp(`\\b${HEADER_LABEL[round]}\\b`).exec(headerLine) + if (m) cols[round] = m.index + } + if (Object.keys(cols).length !== ROUNDS.length) return null + // 'type' heads the school-type column. Names live to its left, so it tells + // us whether a line's first field is a name at all. + const type = /\btype\b/.exec(headerLine) + return type ? { ...cols, typeAt: type.index } : null +} + +/** + * The school name: the line's first whitespace-delimited field. + * + * Names use single spaces; the gap to the school-type column is always + * several. That holds whatever the page's column offsets are, which slicing + * at a fixed or even header-derived position does not — the type column can + * sit either side of its own header word, so a slice takes the 'C' off + * 'Coordinate' and leaves it glued to the name. + */ +/** The closed set of school-type values; never a school name. */ +const SCHOOL_TYPE = /^(Coed|Coordinate|Men|Women)[\d¹²³*]*$/ + +const nameOn = (line, cols) => { + const text = line.slice(0, ROW_END) + const start = text.search(/\S/) + // When a name wraps, its own row carries no name and starts at the school + // -type column instead. Taking the first field regardless yields 'Coed'. + if (start < 0 || start >= cols.typeAt) return '' + const first = text.trim().split(/\s{2,}/)[0].trim() + // 'Coordinate' is long enough to begin left of its own header word, so the + // column guard alone lets it through. + return SCHOOL_TYPE.test(first) ? '' : first +} + +/** True for a line that is one of the repeated column headers. */ +export const isHeaderLine = (line) => line.includes('EDII') && line.includes('EAII') + +/** The round whose column a date at `pos` belongs to: simply the nearest. */ +export function roundFor(pos, cols) { + let best = null, bestDist = Infinity + for (const [round, at] of Object.entries(cols)) { + const d = Math.abs(pos - at) + if (d < bestDist) { bestDist = d; best = round } + } + // Beyond this the line is into the fees columns, not a deadline. + return bestDist <= 12 ? best : null +} + +const DATE = /\d{1,2}\/\d{1,2}\/\d{4}/g +/** Dates and 'Rolling' never appear beyond here; fees and flags follow. */ +const ROW_END = 112 +/** Header, footnote and title fragments that are not schools. */ +const NOT_A_SCHOOL = [ + 'common app', 'member', '2026-27', 'first-year', 'school', 'deadlines', + 'application fee', 'updated:', 'see bottom', +] + +const isNoise = (s) => { + const t = s.toLowerCase() + return !t || t.startsWith('*') || t.startsWith('¹') || NOT_A_SCHOOL.some((p) => t.startsWith(p)) +} + +/** True for a line carrying only the tail of a school name. */ +const isNameOnly = (line, cols) => { + const fields = line.slice(0, ROW_END).trim().split(/\s{2,}/) + return fields.length === 1 && !!nameOn(line, cols) && !isNoise(fields[0]) +} + +/** + * Every school in the grid, with whichever rounds it publishes. + * + * `rd` may be the literal 'Rolling', which is not a date and is deliberately + * kept: "no fixed deadline" is a true and useful answer, and the app already + * understands Rolling as a round. + */ +export function parseGrid(text) { + const lines = text.split('\n') + const out = [] + let pending = [] + let cols = null + + for (let i = 0; i < lines.length; i++) { + const line = lines[i] + if (!line.trim()) { pending = []; continue } + + if (isHeaderLine(line)) { + const found = columnsFrom(line) + if (found) cols = found + pending = [] + continue + } + if (!cols) continue + + const head = line.slice(0, ROW_END) + const dates = [...head.matchAll(DATE)].map((m) => [m.index, m[0]]) + const rollingAt = head.indexOf('Rolling') + + if (dates.length === 0 && rollingAt < 0) { + const name = nameOn(line, cols) + if (!isNoise(name)) pending.push(name) + continue + } + + // A long name spills onto the lines before *and* after its own row. + const parts = [...pending, nameOn(line, cols)] + pending = [] + for (let j = i + 1; j <= i + 2 && j < lines.length; j++) { + if (!isNameOnly(lines[j], cols)) break + parts.push(nameOn(lines[j], cols)) + } + + const name = parts.filter(Boolean).join(' ').replace(/\s+/g, ' ').trim() + if (name.length < 3 || isNoise(name)) continue + + const row = { name } + for (const [pos, value] of dates) { + const round = roundFor(pos, cols) + if (round && !row[round]) row[round] = value + } + if (rollingAt >= 0 && roundFor(rollingAt, cols) === 'rd' && !row.rd) row.rd = 'Rolling' + if (ROUNDS.some((k) => row[k])) out.push(row) + } + return out +} + +const MONTHS = ['Jan', 'Feb', 'Mar', 'Apr', 'May', 'Jun', 'Jul', 'Aug', 'Sep', 'Oct', 'Nov', 'Dec'] + +/** + * `11/01/2026` to `Nov 1, 2026`, the shape the app already parses and the + * curated rows already use. Returns null for anything that is not a date, so + * 'Rolling' passes through untouched by the caller. + */ +export function toDisplayDate(value) { + if (!value || value === 'Rolling') return null + const m = /^(\d{1,2})\/(\d{1,2})\/(\d{4})$/.exec(value.trim()) + if (!m) return null + const month = Number(m[1]), day = Number(m[2]), year = Number(m[3]) + if (month < 1 || month > 12 || day < 1 || day > 31) return null + return `${MONTHS[month - 1]} ${day}, ${year}` +} + +/** Compare names loosely enough to survive "Univ." and "St." but no looser. */ +export function normaliseName(name) { + return name + .toLowerCase() + .replace(/\b(the|of|at|and)\b/g, ' ') + .replace(/[^a-z0-9]+/g, '') + .replace(/university/g, 'univ') + .replace(/college/g, 'coll') + .replace(/saint/g, 'st') +} diff --git a/scripts/ingest-deadlines/parse.test.mjs b/scripts/ingest-deadlines/parse.test.mjs new file mode 100644 index 0000000..cf3b24e --- /dev/null +++ b/scripts/ingest-deadlines/parse.test.mjs @@ -0,0 +1,85 @@ +// Run: node --test scripts/ingest-deadlines/ +// +// Fixtures are verbatim lines from the real PDF. The two headers below have +// genuinely different column offsets — that is the whole point: the grid +// repeats its header on every page and moves the columns each time. + +import test from 'node:test' +import assert from 'node:assert/strict' +import { parseGrid, toDisplayDate, normaliseName } from './parse.mjs' + +const HDR_NARROW = + " member type ¹ ED EDII EA EAII REA Rolling US Int'l" +const HDR_WIDE = + " member type ¹ ED EDII EA EAII REA Rolling US" + +const grid = (...lines) => parseGrid(lines.join('\n')) +const find = (rows, name) => rows.find((r) => r.name === name) + +test('a round is read from its own page’s header, not a fixed offset', () => { + // Yale publishes REA and RD. On the wide page its REA date sits where the + // narrow page keeps regular decision; fixed bands called this RD = Nov 1. + const rows = grid( + HDR_WIDE, + ' Yale University Coed 11/1/2026 01/02/2027 $85', + ) + assert.deepEqual(find(rows, 'Yale University'), { + name: 'Yale University', rea: '11/1/2026', rd: '01/02/2027', + }) +}) + +test('the same column position means different rounds on different pages', () => { + // Column 66 is REA on the narrow page and EA on the wide one; the two + // headers here sit 28 characters apart by the time they reach RD. + const head = ' A College Coed' + const row = head + ' '.repeat(66 - head.length) + '11/1/2026' + const roundOn = (hdr) => Object.keys(find(grid(hdr, row), 'A College'))[1] + assert.equal(roundOn(HDR_NARROW), 'rea') + assert.equal(roundOn(HDR_WIDE), 'ea') +}) + +test('a name wrapped across three lines is rejoined', () => { + const rows = grid( + " member type \u00b9 ED EDII EA EAII REA Rolling US", + ' Georgia Institute of', + ' Coed 10/15/2026 11/2/2026 01/06/2027 $75', + ' Technology', + ) + assert.deepEqual(find(rows, 'Georgia Institute of Technology'), { + name: 'Georgia Institute of Technology', ea: '10/15/2026', eaii: '11/2/2026', rd: '01/06/2027', + }) +}) + +test('the school-type value is never mistaken for a name', () => { + // 'Coordinate' is long enough to start left of its own header word. + const rows = grid( + " member type \u00b9 ED EDII EA EAII REA Rolling US", + ' College of Saint', + ' Coordinate 11/1/2026 12/1/2026 01/15/2027', + ' Benedict', + ) + assert.equal(rows.length, 1) + assert.equal(rows[0].name, 'College of Saint Benedict') +}) + +test('Rolling is kept, but only in the RD column', () => { + const rows = grid(HDR_NARROW, ' SUNY Delhi Coed Rolling $50 $50') + assert.equal(find(rows, 'SUNY Delhi').rd, 'Rolling') +}) + +test('rows before the first header are skipped rather than guessed at', () => { + assert.deepEqual(grid(' Some College Coed 11/1/2026'), []) +}) + +test('toDisplayDate matches the shape the app already stores', () => { + assert.equal(toDisplayDate('11/01/2026'), 'Nov 1, 2026') + assert.equal(toDisplayDate('1/4/2027'), 'Jan 4, 2027') + assert.equal(toDisplayDate('Rolling'), null) + assert.equal(toDisplayDate('13/1/2027'), null) +}) + +test('normaliseName survives the abbreviations the two sources disagree on', () => { + assert.equal(normaliseName('The University of Texas at Austin'), normaliseName('University of Texas-Austin')) + assert.equal(normaliseName('Saint Johns University'), normaliseName('St. Johns University')) + assert.notEqual(normaliseName('Miami University'), normaliseName('University of Miami')) +})