diff --git a/scripts/ingest-deadlines/README.md b/scripts/ingest-deadlines/README.md index 169f040..e0eab0d 100644 --- a/scripts/ingest-deadlines/README.md +++ b/scripts/ingest-deadlines/README.md @@ -21,6 +21,47 @@ node ingest.mjs --apply # write Needs `pdftotext` (`brew install poppler`) on PATH. Reads Supabase credentials from `web/.env.local`. Re-runnable: it only ever fills blanks. +## Writing: use emit-sql + +`public.colleges` is **RLS read-only** in normal operation, so the anon key +cannot update it. Rather than open a hole to let the script write, emit the +same changes as SQL and run them as the project owner: + +```bash +node emit-sql.mjs deadlines.sql # same selection as --apply, writes no rows +``` + +Paste the result into the Supabase SQL editor. It is one statement, so it +lands whole or not at all, and every column is written through +`coalesce(new, existing)` — a curated value survives even if the selection +logic were wrong. + +This is how the September 2026 run was applied: 935 rows, taking real +regular-decision dates from 46 colleges to 977. + +### --apply, and why it needs care + +`--apply` writes with the anon key, so it only works bracketed by a temporary +policy that must be dropped straight after — the procedure +`scripts/ingest-colleges` documents: + +```sql +create policy "colleges temp deadline update" on public.colleges + for update to anon using (true) with check (true); +-- ... run ... +drop policy if exists "colleges temp deadline update" on public.colleges; +``` + +That policy lets anyone holding the anon key — which is public in the JS +bundle — rewrite the table for as long as it exists. `emit-sql` avoids the +window entirely, which is why it is the path above. + +Without the policy PostgREST accepts every UPDATE and changes nothing: an +RLS-filtered update is zero rows, not an error. That is not hypothetical — a +run reported a confident `wrote 935/935` having written nothing at all. The +script now counts the rows that actually came back and stops on the first that +changed none. + Current run: 1,127 schools parsed, 955 matched, **935 rows to fill**, 0 overwritten. ## What it will not do diff --git a/scripts/ingest-deadlines/emit-sql.mjs b/scripts/ingest-deadlines/emit-sql.mjs new file mode 100644 index 0000000..e57cad6 --- /dev/null +++ b/scripts/ingest-deadlines/emit-sql.mjs @@ -0,0 +1,75 @@ +// Emit the same writes ingest.mjs --apply would make, as one SQL statement. +// +// Running them as the project owner rather than the anon key means no +// temporary "anyone may update colleges" policy has to exist even briefly. +// Identical selection logic: fills blanks only, never overwrites. +import { readFileSync, writeFileSync } from 'node:fs' +import { execFileSync } from 'node:child_process' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { createClient } from '@supabase/supabase-js' +import { parseGrid, toDisplayDate, normaliseName } from './parse.mjs' + +const env = {} +for (const l of readFileSync(new URL('../../web/.env.local', import.meta.url).pathname, 'utf8').split('\n')) { + const m = l.match(/^([A-Z0-9_]+)=(.*)$/); if (m) env[m[1]] = m[2].trim() +} +const supabase = createClient(env.VITE_SUPABASE_URL, env.VITE_SUPABASE_ANON_KEY) + +const pdf = join(tmpdir(), 'commonapp-reqgrid.pdf') +const txt = join(tmpdir(), 'commonapp-reqgrid.txt') +execFileSync('pdftotext', ['-layout', pdf, txt]) +const rows = parseGrid(readFileSync(txt, 'utf8')) + +const colleges = [] +for (let from = 0; ; from += 1000) { + const { data, error } = await supabase.from('colleges') + .select('scorecard_id, name, early_decision, early_action, regular_decision') + .order('scorecard_id').range(from, from + 999) + if (error) throw new Error(error.message) + colleges.push(...data); if (data.length < 1000) break +} + +const byName = new Map() +for (const r of rows) { const k = normaliseName(r.name); byName.set(k, [...(byName.get(k) ?? []), r]) } +const single = [...byName.values()].filter((g) => g.length === 1).map((g) => g[0]) + +const index = new Map() +for (const c of colleges) { const k = normaliseName(c.name); if (!index.has(k)) index.set(k, c) } +const keys = [...index.keys()] +const COLUMNS = { ed: 'early_decision', ea: 'early_action', rd: 'regular_decision' } + +const claimed = new Map(); const updates = [] +for (const row of single) { + const key = normaliseName(row.name) + let college = index.get(key) + if (!college && key.length >= 10) { + const near = keys.filter((k) => k.startsWith(key)); if (near.length === 1) college = index.get(near[0]) + } + if (!college || claimed.has(college.scorecard_id)) continue + claimed.set(college.scorecard_id, row.name) + const patch = {} + for (const [round, column] of Object.entries(COLUMNS)) { + const raw = row[round]; if (!raw) continue + const value = raw === 'Rolling' ? 'Rolling' : toDisplayDate(raw) + if (!value || college[column]) continue // never overwrite + patch[column] = value + } + if (Object.keys(patch).length) updates.push({ id: college.scorecard_id, patch }) +} + +const q = (v) => (v == null ? 'null' : `'${String(v).replace(/'/g, "''")}'`) +const values = updates.map((u) => + `(${u.id},${q(u.patch.early_decision)},${q(u.patch.early_action)},${q(u.patch.regular_decision)})`).join(',\n ') +// coalesce so a column this run has nothing for keeps whatever is there. +const sql = `update public.colleges c set + early_decision = coalesce(v.ed, c.early_decision), + early_action = coalesce(v.ea, c.early_action), + regular_decision = coalesce(v.rd, c.regular_decision), + verified_at = now() +from (values + ${values} +) as v(id, ed, ea, rd) +where c.scorecard_id = v.id;` +writeFileSync(process.argv[2] ?? 'deadlines.sql', sql) +console.log(`${updates.length} rows, ${sql.length} bytes -> ${process.argv[2] ?? 'deadlines.sql'}`) diff --git a/scripts/ingest-deadlines/ingest.mjs b/scripts/ingest-deadlines/ingest.mjs index 3c86fd8..fa2eb09 100644 --- a/scripts/ingest-deadlines/ingest.mjs +++ b/scripts/ingest-deadlines/ingest.mjs @@ -16,6 +16,11 @@ // // Needs `pdftotext` (poppler) on PATH. Reads VITE_SUPABASE_URL and // VITE_SUPABASE_ANON_KEY from ../../web/.env.local. +// +// `colleges` is RLS read-only in normal operation, so --apply needs a +// temporary anon update policy bracketing the run — see the README. Without +// it PostgREST accepts every UPDATE and changes nothing: an RLS-filtered +// update is zero rows, not an error. import { readFileSync, writeFileSync, existsSync } from 'node:fs' import { execFileSync } from 'node:child_process' @@ -165,11 +170,23 @@ async function main() { let written = 0 for (const u of updates) { - const { error: e } = await supabase + // `.select()` so the changed rows come back and can be counted. Without + // it a write blocked by RLS is indistinguishable from one that landed: + // PostgREST returns 204 and no error either way, and this reported a + // confident "wrote 935/935" having changed nothing at all. + const { data: changed, error: e } = await supabase .from('colleges') .update({ ...u.patch, verified_at: new Date().toISOString() }) .eq('scorecard_id', u.scorecard_id) + .select('scorecard_id') if (e) { console.error(` failed ${u.name}: ${e.message}`); continue } + if (!changed?.length) { + throw new Error( + `${u.name} matched no row. The anon key cannot write colleges without the\n` + + ' temporary RLS policy described in the README — add it, re-run, and drop it after.\n' + + ` Stopped on the first silent no-op; ${written} rows written so far.`, + ) + } written += 1 } console.log(`\nwrote ${written}/${updates.length}`)