diff --git a/.cursor/rules/project-conventions.mdc b/.cursor/rules/project-conventions.mdc index f965876..353ec35 100644 --- a/.cursor/rules/project-conventions.mdc +++ b/.cursor/rules/project-conventions.mdc @@ -27,7 +27,7 @@ bun run format # format only ### Monorepo structure ``` -apps/web — Next.js 15 (App Router) → @the-forum/web +apps/web — Next.js 16 (App Router) → @the-forum/web apps/database — Drizzle ORM → @the-forum/database backends/fastapi — FastAPI Python → @the-forum/fastapi ``` diff --git a/.env.example b/.env.example index f6f89ed..2293b74 100644 --- a/.env.example +++ b/.env.example @@ -12,36 +12,29 @@ POSTGRES_PORT=5434 # ----- Database URL (used by Drizzle & Next.js) ----- DATABASE_URL=postgresql://forum:forum_password@localhost:5434/the_forum -# ----- Auth.js (Microsoft Entra ID) ----- +# ----- Auth.js + Princeton CAS ----- +# Generate with: openssl rand -base64 32 AUTH_SECRET=your-auth-secret-here -AUTH_AZURE_AD_CLIENT_ID=your-client-id -AUTH_AZURE_AD_CLIENT_SECRET=your-client-secret -AUTH_AZURE_AD_TENANT_ID=YOUR_TENANT_ID_HERE +# Canonical public URL of the app. Required in production (pins the CAS service +# URL and Auth.js callbacks to this origin). Optional locally. +# AUTH_URL=https://forum.example.edu +# Trust X-Forwarded-Host/Proto from your reverse proxy (not needed on Vercel or when AUTH_URL is set). +# AUTH_TRUST_HOST=true +# Princeton CAS server (default shown). +# CAS_BASE_URL=https://fed.princeton.edu/cas/ # ----- AWS S3 (image uploads) ----- # AWS_S3_BUCKET=the-forum-uploads # AWS_REGION=us-east-1 # ----- Next.js public vars (prefix with NEXT_PUBLIC_) ----- -# NEXT_PUBLIC_API_URL=http://localhost:8000 +# Public origin for metadata / Open Graph / sitemap (defaults to https://forum.tigerapps.org) +# NEXT_PUBLIC_SITE_URL=https://forum.tigerapps.org NEXT_PUBLIC_MAPBOX_TOKEN=YOUR_CAMPUS_MAPBOX_TOKEN_HERE NEXT_PUBLIC_CAMPUS_MAP_TOKEN=YOUR_CAMPUS_MAP_TOKEN_HERE NEXT_PUBLIC_CAMPUS_MAP_STYLE=YOUR_CAMPUS_MAP_STYLE_HERE -# ----- FastAPI ----- -# FASTAPI_SECRET_KEY=changeme +# ----- InboxEngine (org/venue/event sync, apps/database) ----- +# INBOX_ENGINE_URL=https://inbox-engine.tigerapps.org +# INBOX_ENGINE_TOKEN=ask-a-tigerapps-admin -# ----- Pipeline (Phase 7) ----- -# Gmail API — OAuth2 credentials JSON (from Google Cloud Console) -# GMAIL_CREDENTIALS_JSON={"installed":{"client_id":"...","client_secret":"...",...}} -# GMAIL_TOKEN_PATH=./gmail_token.json - -# OpenRouter API key for LLM extraction (Gemini 3.1 Flash Lite) -# OPENROUTER_API_KEY=sk-or-... - -# Admin API key for pipeline management endpoints -# ADMIN_API_KEY=your-secret-admin-key - -# ----- Listserv Scraper ----- -LISTSERV_EMAIL=tigerapp@princeton.edu -LISTSERV_PASSWORD=YOUR_LISTSERV_PASSWORD_HERE diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml new file mode 100644 index 0000000..fc2377e --- /dev/null +++ b/.github/workflows/ci.yml @@ -0,0 +1,53 @@ +name: CI + +on: + pull_request: + push: + branches: [staging, main] + workflow_call: {} + +permissions: + contents: read + +concurrency: + group: ci-${{ github.workflow }}-${{ github.ref }} + cancel-in-progress: true + +jobs: + web: + name: lint / typecheck / test / build + runs-on: ubuntu-latest + timeout-minutes: 20 + env: + # No secrets needed: env validation is skipped and every value is a dummy. + SKIP_ENV_VALIDATION: "1" + AUTH_SECRET: ci-dummy-auth-secret + DATABASE_URL: postgresql://ci:ci@localhost:5432/ci + NEXT_PUBLIC_MAPBOX_TOKEN: pk.ci-dummy + NEXT_PUBLIC_CAMPUS_MAP_TOKEN: pk.ci-dummy + NEXT_PUBLIC_CAMPUS_MAP_STYLE: mapbox://styles/ci/dummy + NEXT_TELEMETRY_DISABLED: "1" + TURBO_TELEMETRY_DISABLED: "1" + steps: + - uses: actions/checkout@v4 + + - uses: oven-sh/setup-bun@v2 + with: + # bun.lock uses configVersion (Bun >= 1.3). + bun-version: 1.3.5 + + - name: Install dependencies + run: bun install --frozen-lockfile --ignore-scripts + + - name: Lint (Biome) + run: bun run lint + + - name: Typecheck (apps/web) + run: bunx tsc --noEmit -p apps/web + + - name: Unit tests (apps/web) + working-directory: apps/web + run: bun test + + - name: Build (apps/web) + run: bun run build --filter=@the-forum/web diff --git a/.github/workflows/deploy.yml b/.github/workflows/deploy.yml new file mode 100644 index 0000000..1e90d80 --- /dev/null +++ b/.github/workflows/deploy.yml @@ -0,0 +1,115 @@ +name: Deploy + +# staging → forumdev.tigerapps.org, main → forum.tigerapps.org, after the CI checks pass. +on: + push: + branches: [staging, main] + workflow_dispatch: + inputs: + environment: + description: Environment to deploy from this ref + type: choice + options: [staging, production] + +permissions: + contents: read + id-token: write + +jobs: + ci: + uses: ./.github/workflows/ci.yml + + deploy: + needs: ci + runs-on: ubuntu-latest + timeout-minutes: 30 + concurrency: + group: theforum-deploy-${{ github.ref_name }} + cancel-in-progress: false + env: + AWS_REGION: us-east-1 + BUCKET: theforum-deploy-104733724423-us-east-1 + INSTANCE_ID: i-00a811982e8784203 + SHA: ${{ github.sha }} + BRANCH: ${{ github.ref_name }} + steps: + - name: Choose environment + run: | + if [ "${{ github.event_name }}" = workflow_dispatch ]; then target="${{ inputs.environment }}"; + elif [ "$BRANCH" = main ]; then target=production; else target=staging; fi + if [ "$target" = production ]; then site=https://forum.tigerapps.org; else site=https://forumdev.tigerapps.org; fi + echo "TARGET=$target" >> "$GITHUB_ENV" + echo "NEXT_PUBLIC_SITE_URL=$site" >> "$GITHUB_ENV" + echo "Deploying $SHA ($BRANCH) to $target" + + - uses: actions/checkout@v4 + with: + ref: ${{ env.SHA }} + + - name: Fetch the InboxEngine submodule (private) + env: + KEY: ${{ secrets.INBOX_ENGINE_DEPLOY_KEY }} + run: | + install -m 700 -d ~/.ssh + printf '%s\n' "$KEY" > ~/.ssh/inbox_engine && chmod 600 ~/.ssh/inbox_engine + ssh-keyscan -t ed25519 github.com >> ~/.ssh/known_hosts 2>/dev/null + git -c url."git@github.com:TigerAppsOrg/InboxEngine".insteadOf="https://github.com/TigerAppsOrg/InboxEngine" \ + -c core.sshCommand="ssh -i ~/.ssh/inbox_engine -o IdentitiesOnly=yes" \ + submodule update --init --depth 1 + rm -f ~/.ssh/inbox_engine + + - uses: oven-sh/setup-bun@v2 + with: + bun-version: 1.3.5 + + - run: bun install --frozen-lockfile --ignore-scripts + + - name: Build + env: + SKIP_ENV_VALIDATION: "1" + # Placeholders for build-time module evaluation; nothing connects during the build. + DATABASE_URL: postgresql://build:build@localhost:5432/build + AUTH_SECRET: build-time-placeholder + NEXT_TELEMETRY_DISABLED: "1" + TURBO_TELEMETRY_DISABLED: "1" + NEXT_PUBLIC_MAPBOX_TOKEN: ${{ vars.NEXT_PUBLIC_MAPBOX_TOKEN }} + NEXT_PUBLIC_CAMPUS_MAP_TOKEN: ${{ vars.NEXT_PUBLIC_CAMPUS_MAP_TOKEN }} + NEXT_PUBLIC_CAMPUS_MAP_STYLE: ${{ vars.NEXT_PUBLIC_CAMPUS_MAP_STYLE }} + run: | + bun run build --filter=@the-forum/web + deploy/build-release.sh /tmp/release.tar.gz + echo "DIGEST=$(sha256sum /tmp/release.tar.gz | cut -d' ' -f1)" >> "$GITHUB_ENV" + + - uses: aws-actions/configure-aws-credentials@v4 + with: + role-to-assume: arn:aws:iam::104733724423:role/TheForumGitHubDeployRole + aws-region: ${{ env.AWS_REGION }} + + - name: Upload release + run: aws s3 cp /tmp/release.tar.gz "s3://$BUCKET/releases/theforum/$TARGET/$SHA.tar.gz" --only-show-errors + + - name: Deploy through SSM + run: | + set -euo pipefail + params=$(jq -n --arg e "$TARGET" --arg r "$SHA" --arg c "$DIGEST" '{Environment:[$e],Release:[$r],Checksum:[$c]}') + id=$(aws ssm send-command --instance-ids "$INSTANCE_ID" --document-name TheForumDeploy --parameters "$params" --query Command.CommandId --output text) + echo "SSM command $id" + for _ in $(seq 1 100); do + status=$(aws ssm get-command-invocation --command-id "$id" --instance-id "$INSTANCE_ID" --query Status --output text 2>/dev/null || echo Pending) + case "$status" in + Success) aws ssm get-command-invocation --command-id "$id" --instance-id "$INSTANCE_ID" --query StandardOutputContent --output text | tail -8; exit 0 ;; + Failed|Cancelled|TimedOut|Cancelling) + aws ssm get-command-invocation --command-id "$id" --instance-id "$INSTANCE_ID" --query '{out:StandardOutputContent,err:StandardErrorContent}'; exit 1 ;; + esac + sleep 10 + done + echo 'Deployment did not finish in time'; exit 1 + + - name: Smoke test through Cloudflare + run: | + if [ "$TARGET" = production ]; then url=https://forum.tigerapps.org; else url=https://forumdev.tigerapps.org; fi + for _ in $(seq 1 10); do + code=$(curl -s -o /dev/null -w '%{http_code}' "$url/") && [ "$code" = 200 ] && { echo "$url is up"; exit 0; } + sleep 6 + done + echo "$url did not return 200"; exit 1 diff --git a/.github/workflows/pipeline-poll.yml b/.github/workflows/pipeline-poll.yml deleted file mode 100644 index e1645cf..0000000 --- a/.github/workflows/pipeline-poll.yml +++ /dev/null @@ -1,16 +0,0 @@ -name: Pipeline Poll - -on: - schedule: - - cron: "*/15 * * * *" # Every 15 minutes - workflow_dispatch: {} # Allow manual trigger - -jobs: - poll: - runs-on: ubuntu-latest - steps: - - name: Trigger pipeline poll - run: | - curl -sf -X POST "${{ secrets.FASTAPI_URL }}/pipeline/poll" \ - -H "Authorization: Bearer ${{ secrets.ADMIN_API_KEY }}" \ - -H "Content-Type: application/json" diff --git a/.github/workflows/scrape_listserv.yml b/.github/workflows/scrape_listserv.yml deleted file mode 100644 index 2452246..0000000 --- a/.github/workflows/scrape_listserv.yml +++ /dev/null @@ -1,41 +0,0 @@ -name: Scrape WhitmanWire - -on: - schedule: - # Runs at 8:00 AM UTC (4:00 AM Eastern Time) every day - - cron: '0 8 * * *' - workflow_dispatch: # Allows you to click a "Run workflow" button in GitHub's UI - -jobs: - scrape-and-commit: - runs-on: ubuntu-latest - # This permission is required so the bot can push the JSON file back to your repo - permissions: - contents: write - - steps: - - name: Checkout repository - uses: actions/checkout@v4 - - - name: Set up Python - uses: actions/setup-python@v5 - with: - python-version: '3.11' # The script uses standard libraries, so this is fine - - - name: Run Scraper - env: - # Pulls securely from GitHub Secrets - LISTSERV_EMAIL: ${{ secrets.LISTSERV_EMAIL }} - LISTSERV_PASSWORD: ${{ secrets.LISTSERV_PASSWORD }} - run: | - python3 apps/listserv-scraper/src/scrape_listserv.py --list WHITMANWIRE --limit 100 --fetch-bodies - - - name: Commit and push changes - run: | - git config --global user.name "github-actions[bot]" - git config --global user.email "41898282+github-actions[bot]@users.noreply.github.com" - # Stage the specific data folder where the script saves its output - git add apps/listserv-scraper/data/ - # Commit the changes; if there are no new emails, it fails gracefully instead of crashing - git commit -m "chore: update WhitmanWire scraped data" || echo "No changes to commit" - git push \ No newline at end of file diff --git a/.gitignore b/.gitignore index 6399e95..6dcca7a 100644 --- a/.gitignore +++ b/.gitignore @@ -57,3 +57,8 @@ Thumbs.db *.ntvs* *.njsproj *.sln + +# ---- Secrets / local artifacts ---- +gmail_token.json +**/cookies.json +*.tsbuildinfo diff --git a/.gitmodules b/.gitmodules new file mode 100644 index 0000000..0a3c836 --- /dev/null +++ b/.gitmodules @@ -0,0 +1,3 @@ +[submodule "packages/inbox-engine"] + path = packages/inbox-engine + url = https://github.com/TigerAppsOrg/InboxEngine.git diff --git a/AGENTS.md b/AGENTS.md index 619996b..9b0d254 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -3,9 +3,9 @@ ## Project Overview Turborepo monorepo with: -- `apps/web` — Next.js 15 (App Router, Turbopack, Tailwind v4, shadcn/ui) +- `apps/web` — Next.js 16 (App Router, Turbopack, Tailwind v4, shadcn/ui), Auth.js v5 + Princeton CAS login - `apps/database` — Drizzle ORM + PostgreSQL -- `backends/fastapi` — FastAPI (Python 3.12, uv) +- `packages/inbox-engine` — git submodule (TigerAppsOrg/InboxEngine): shared org/venue/event source of truth; sync with `bun run db:sync-engine`. Don't edit it here — change InboxEngine and bump the submodule. --- diff --git a/CLAUDE.md b/CLAUDE.md index c9c577e..59b4a34 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -3,9 +3,9 @@ ## Project Overview Turborepo monorepo with: -- `apps/web` — Next.js 15 (App Router, Turbopack, Tailwind v4, shadcn/ui) +- `apps/web` — Next.js 16 (App Router, Turbopack, Tailwind v4, shadcn/ui), Auth.js v5 + Princeton CAS login - `apps/database` — Drizzle ORM + PostgreSQL -- `backends/fastapi` — FastAPI (Python 3.12, uv) +- `packages/inbox-engine` — git submodule (TigerAppsOrg/InboxEngine): shared org/venue/event source of truth; sync with `bun run db:sync-engine`. Don't edit it here — change InboxEngine and bump the submodule. --- diff --git a/README.md b/README.md index 61baefe..c0468d2 100644 --- a/README.md +++ b/README.md @@ -6,16 +6,16 @@ This is a [Turborepo](https://turbo.build) monorepo managed with [Bun](https://b | Package | What it is | |---|---| -| `apps/web` | **The main app** — Next.js 15 (App Router), React 19, Tailwind v4, shadcn/ui | +| `apps/web` | **The main app** — Next.js 16 (App Router), React 19, Tailwind v4, shadcn/ui | | `apps/database` | Shared Drizzle ORM schema + migrations (PostgreSQL) | -| `apps/admin-web` | Admin dashboard — Vite + React | -| `backends/fastapi` | FastAPI backend (Python 3.12, managed with `uv`) | -| `apps/listserv-scraper` | Python scraper for Princeton listserv archives | -| `apps/mpu-scraper` | Scraper for MyPrincetonU events | +| `packages/inbox-engine` | **Git submodule** → [TigerAppsOrg/InboxEngine](https://github.com/TigerAppsOrg/InboxEngine), the shared source of truth for Princeton organizations, campus venues and events (MyPrincetonU + listserv emails) | New to the project? Follow **Quick start** below — it gets `apps/web` running locally, -which is the primary thing you need. The Python backend and scrapers are optional -until you work on them. +which is the primary thing you need. + +Clone with submodules (`git clone --recurse-submodules …`), or run +`git submodule update --init` in an existing checkout. InboxEngine is a private repo; +ask a TigerApps admin for access if the submodule fails to fetch. --- @@ -60,17 +60,23 @@ Then fill in `apps/web/.env.local`. Env vars are validated at startup by one is missing, and that file is the source of truth for what's required. > **Can't obtain a value yourself? Ask Ibraheem.** He is the contact for all -> credentials that aren't self-serve (Entra ID, Mapbox tokens, AWS, etc.). +> credentials that aren't self-serve (Mapbox tokens, AWS, etc.). | Variable | Where to get it | |---|---| | `DATABASE_URL` | Default in the example file works as-is with the Docker database (port **5434**) | | `AUTH_SECRET` | Generate your own: `openssl rand -base64 32` | -| `AUTH_AZURE_AD_CLIENT_ID` / `AUTH_AZURE_AD_CLIENT_SECRET` | **Ask Ibraheem** — these are the Microsoft Entra ID app credentials for Princeton CAS login | -| `AUTH_AZURE_AD_TENANT_ID` | Princeton's tenant ID — already filled in the example file | +| `AUTH_URL` | Optional locally. **Required in production:** the app's canonical public URL (e.g. `https://forum.example.edu`) — pins Auth.js callbacks and the CAS service URL to that origin | +| `AUTH_TRUST_HOST` | Optional. Set to `true` only when running behind a reverse proxy without `AUTH_URL` (not needed on Vercel, which Auth.js trusts automatically) | +| `CAS_BASE_URL` | Optional — defaults to `https://fed.princeton.edu/cas/` | | `NEXT_PUBLIC_MAPBOX_TOKEN` / `NEXT_PUBLIC_CAMPUS_MAP_TOKEN` / `NEXT_PUBLIC_CAMPUS_MAP_STYLE` | **Ask Ibraheem** — Mapbox tokens + the Princeton campus map style URL | | `AWS_S3_BUCKET` / `AWS_REGION` | Optional (image uploads) — ask Ibraheem if you're working on that feature | +Login uses **Princeton CAS** — there are no OAuth client credentials to obtain. +Clicking "Log in" goes to `/api/auth/cas/login`, which redirects to +`fed.princeton.edu/cas`; CAS sends you back to `/api/auth/cas/callback`, where the +ticket is validated server-side and your user row is created from your NetID. + ### 3. Start the database Make sure **Docker Desktop is running**, then from the repo root: @@ -104,13 +110,7 @@ cd apps/web && bun run dev Open . You're set up. -To run **everything at once** (web + admin + FastAPI) from the repo root: - -```bash -bun run dev # Turborepo TUI: web :3000, admin-web :5173, FastAPI :8000 -``` - -(FastAPI will only start if you've done the [Python backend](#python-backend-optional) setup.) +To run the dev server through Turborepo from the repo root: `bun run dev`. --- @@ -124,7 +124,9 @@ bun run build # build all packages bun run db:up # start Postgres db:down stop it (data persists) bun run db:push # push schema (dev) db:generate generate SQL migrations bun run db:migrate # apply migrations db:studio visual DB browser -bun run db:seed # seed demo data (safe to re-run any time) +bun run db:seed # seed demo data (safe to re-run; refuses non-local DBs unless ALLOW_REMOTE_SEED=1) +bun run db:sync-engine # import orgs, venues and events from InboxEngine (needs INBOX_ENGINE_URL/TOKEN) +(cd apps/web && bun test) # unit tests ``` Pre-commit hooks (Husky + lint-staged) automatically run Biome on staged files — @@ -136,26 +138,58 @@ if your commit fails, read the Biome output, fix, and re-commit. New vars get added to `apps/web/src/env.ts` *and* the `.env.example` files. - **UI components:** use [shadcn/ui](https://ui.shadcn.com). Add new ones from `apps/web`: `bunx shadcn@latest add `. -- **Linting:** Biome only (no ESLint/Prettier). Python uses Ruff. +- **Linting:** Biome only (no ESLint/Prettier). --- -## Python backend (optional) +## Organizations and events from InboxEngine -Only needed if you're working on `backends/fastapi` or the scrapers. +Official organizations (all MyPrincetonU groups, with logos, descriptions, social links and +MyPrincetonU page links), campus venues and events come from +[InboxEngine](https://github.com/TigerAppsOrg/InboxEngine), which also powers TigerInbox. +It ingests MyPrincetonU's official events feed and residential/FreeFood listserv emails, +resolves the hosting organization, extracts time and place, and exposes a revisioned change feed. ```bash -cd backends/fastapi -cp .env.example .env # default DATABASE_URL works with the Docker database -uv sync # creates .venv and installs all dependencies -bun run dev # = uv run uvicorn app.main:app --reload --port 8000 +# apps/database/.env +INBOX_ENGINE_URL=https://inbox-engine.tigerapps.org +INBOX_ENGINE_TOKEN=… # ask a TigerApps admin + +bun run db:sync-engine # incremental; add -- --full to replay the whole feed ``` -API docs live at . Lint with `uv run ruff check .` -and format with `uv run ruff format .`. +Imported orgs have `source = 'myprincetonu'` and `external_id = 'mpu:'`; their +officers are managed on MyPrincetonU. Imported events are owned by the `_inboxengine` bot user, +carry `source` (`myprincetonu` or `listserv`) and a `source_url`, and are unpublished (never +deleted) when InboxEngine withdraws them. Production runs the sync every five minutes. + +To update the engine version: `cd packages/inbox-engine && git pull origin main`, then commit +the new submodule pointer. --- +## Deployment + +| Branch | URL | Service (on `the-forum-web` EC2) | Database | +|---|---|---|---| +| `staging` | https://forumdev.tigerapps.org | `theforum-staging` on :3100 | `theforum_staging` (RDS) | +| `main` | https://forum.tigerapps.org | `theforum-production` on :3200 | `theforum` (RDS) | + +Every push to `staging` or `main` runs CI, then `.github/workflows/deploy.yml`: + +1. Builds a Next.js standalone server with the public map tokens (repo variables) and the + environment's `NEXT_PUBLIC_SITE_URL`, and bundles `tools/migrate.js` and `tools/sync.js` with Bun + (`deploy/build-release.sh`). +2. Uploads the checksummed tarball to S3 through GitHub OIDC (`TheForumGitHubDeployRole`). +3. Runs the `TheForumDeploy` SSM document on the instance, which executes `deploy/run-release.sh`: + fetch the SecureString `/theforum//environment`, run migrations, switch the `current` + symlink, restart systemd, health-check with automatic rollback, install the nginx site, and + enable the five-minute InboxEngine sync timer. + +nginx serves each host on :80 behind Cloudflare (TLS at the edge); `theforumdev.tigerapps.org` +redirects to `forumdev`. InboxEngine runs on the same host (`inbox-engine.service`, :8300). +To change runtime configuration, update the SSM parameter and redeploy (or re-run the workflow). + ## Branching workflow `main` is protected — you cannot push to it directly, and pull requests into `main` @@ -183,8 +217,5 @@ Change `POSTGRES_PORT` in the root `.env` and update `DATABASE_URL` everywhere t **Husky hooks not running** Re-run `bun install` from the repo root (the `prepare` script reinstalls hooks). -**`bun run dev` doesn't start FastAPI** -Expected unless you've run `uv sync` in `backends/fastapi` and `uv` is on your PATH. - **Wipe the database and start fresh** `docker compose down -v` (deletes the data volume), then `bun run db:up && bun run db:push && bun run db:seed`. diff --git a/apps/admin-web/.gitignore b/apps/admin-web/.gitignore deleted file mode 100644 index a547bf3..0000000 --- a/apps/admin-web/.gitignore +++ /dev/null @@ -1,24 +0,0 @@ -# Logs -logs -*.log -npm-debug.log* -yarn-debug.log* -yarn-error.log* -pnpm-debug.log* -lerna-debug.log* - -node_modules -dist -dist-ssr -*.local - -# Editor directories and files -.vscode/* -!.vscode/extensions.json -.idea -.DS_Store -*.suo -*.ntvs* -*.njsproj -*.sln -*.sw? diff --git a/apps/admin-web/index.html b/apps/admin-web/index.html deleted file mode 100644 index 848282d..0000000 --- a/apps/admin-web/index.html +++ /dev/null @@ -1,13 +0,0 @@ - - - - - - - Pipeline Admin — The Forum - - -
- - - diff --git a/apps/admin-web/package.json b/apps/admin-web/package.json deleted file mode 100644 index 926701c..0000000 --- a/apps/admin-web/package.json +++ /dev/null @@ -1,25 +0,0 @@ -{ - "name": "@the-forum/admin-web", - "private": true, - "version": "0.1.0", - "type": "module", - "scripts": { - "dev": "vite", - "build": "tsc && vite build", - "preview": "vite preview" - }, - "devDependencies": { - "@tailwindcss/vite": "^4.2.1", - "@types/react": "^19.0.0", - "@types/react-dom": "^19.0.0", - "tailwindcss": "^4.2.1", - "typescript": "~5.9.3", - "vite": "^7.3.1" - }, - "dependencies": { - "@tanstack/react-query": "^5.90.21", - "react": "^19.2.4", - "react-dom": "^19.2.4", - "react-router-dom": "^7.13.1" - } -} diff --git a/apps/admin-web/public/vite.svg b/apps/admin-web/public/vite.svg deleted file mode 100644 index e7b8dfb..0000000 --- a/apps/admin-web/public/vite.svg +++ /dev/null @@ -1 +0,0 @@ - \ No newline at end of file diff --git a/apps/admin-web/src/lib/api.ts b/apps/admin-web/src/lib/api.ts deleted file mode 100644 index f65df10..0000000 --- a/apps/admin-web/src/lib/api.ts +++ /dev/null @@ -1,106 +0,0 @@ -const API_BASE = "/pipeline"; - -function getApiKey(): string { - return localStorage.getItem("admin_api_key") ?? ""; -} - -export function setApiKey(key: string) { - localStorage.setItem("admin_api_key", key); -} - -export function getStoredApiKey(): string { - return getApiKey(); -} - -async function request(path: string, init?: RequestInit): Promise { - const res = await fetch(`${API_BASE}${path}`, { - ...init, - headers: { - "Content-Type": "application/json", - Authorization: `Bearer ${getApiKey()}`, - ...init?.headers, - }, - }); - if (!res.ok) { - const text = await res.text().catch(() => "Unknown error"); - throw new Error(`${res.status}: ${text}`); - } - return res.json(); -} - -// ── Pipeline control ───────────────────────────────────── - -export const pollPipeline = () => request<{ ok: boolean }>("/poll", { method: "POST" }); - -export const getPipelineStatus = () => - request<{ - total: number; - success: number; - duplicates: number; - skipped: number; - errors: number; - last_poll: string | null; - }>("/status"); - -// ── Events ─────────────────────────────────────────────── - -export interface PipelineEvent { - id: string; - title: string; - description: string; - datetime: string; - end_datetime: string | null; - source_message_id: string; - org_id: string | null; - location_name: string | null; - location_id: string | null; - created_at: string; -} - -export const getPipelineEvents = (limit = 50, offset = 0) => - request(`/events?limit=${limit}&offset=${offset}`); - -export const updatePipelineEvent = (id: string, fields: Record) => - request(`/events/${id}`, { method: "PATCH", body: JSON.stringify(fields) }); - -export const deletePipelineEvent = (id: string) => request(`/events/${id}`, { method: "DELETE" }); - -// ── Duplicates ─────────────────────────────────────────── - -export interface DuplicateEntry { - id: string; - message_id: string; - status: string; - error_text: string | null; - listserv_label: string | null; - created_at: string; -} - -export const getDuplicates = (limit = 50, offset = 0) => - request(`/duplicates?limit=${limit}&offset=${offset}`); - -// ── Configs ────────────────────────────────────────────── - -export interface ListservConfigItem { - id: string; - address: string; - label: string; - org_id: string | null; - gmail_label: string | null; - enabled: boolean; - created_at: string; -} - -export const getConfigs = () => request("/configs"); - -export const createConfig = (data: { - address: string; - label: string; - gmail_label?: string; - org_id?: string; -}) => request("/configs", { method: "POST", body: JSON.stringify(data) }); - -export const updateConfig = (id: string, fields: Record) => - request(`/configs/${id}`, { method: "PATCH", body: JSON.stringify(fields) }); - -export const deleteConfig = (id: string) => request(`/configs/${id}`, { method: "DELETE" }); diff --git a/apps/admin-web/src/main.tsx b/apps/admin-web/src/main.tsx deleted file mode 100644 index 8a22a88..0000000 --- a/apps/admin-web/src/main.tsx +++ /dev/null @@ -1,34 +0,0 @@ -import { QueryClient, QueryClientProvider } from "@tanstack/react-query"; -import { StrictMode } from "react"; -import { createRoot } from "react-dom/client"; -import { BrowserRouter, Navigate, Route, Routes } from "react-router-dom"; -import { Shell } from "./shell"; -import "./style.css"; -import { Dashboard } from "./pages/dashboard"; -import { DedupLog } from "./pages/dedup-log"; -import { EventReview } from "./pages/event-review"; -import { ListservConfig } from "./pages/listserv-config"; - -const queryClient = new QueryClient({ - defaultOptions: { queries: { retry: 1, refetchOnWindowFocus: false } }, -}); - -const root = document.getElementById("app"); -if (!root) throw new Error("Root element not found"); -createRoot(root).render( - - - - - }> - } /> - } /> - } /> - } /> - } /> - - - - - , -); diff --git a/apps/admin-web/src/pages/dashboard.tsx b/apps/admin-web/src/pages/dashboard.tsx deleted file mode 100644 index eca1269..0000000 --- a/apps/admin-web/src/pages/dashboard.tsx +++ /dev/null @@ -1,87 +0,0 @@ -import { useMutation, useQuery, useQueryClient } from "@tanstack/react-query"; -import { getPipelineStatus, pollPipeline } from "../lib/api"; - -export function Dashboard() { - const queryClient = useQueryClient(); - const { data, isLoading, error } = useQuery({ - queryKey: ["pipeline-status"], - queryFn: getPipelineStatus, - refetchInterval: 30_000, - }); - - const pollMutation = useMutation({ - mutationFn: pollPipeline, - onSuccess: () => queryClient.invalidateQueries({ queryKey: ["pipeline-status"] }), - }); - - return ( -
-
-

Pipeline Dashboard

- -
- - {pollMutation.isSuccess && ( -
- Poll completed successfully. -
- )} - {pollMutation.isError && ( -
- Poll failed: {(pollMutation.error as Error).message} -
- )} - - {isLoading &&

Loading...

} - {error &&

Error: {(error as Error).message}

} - - {data && ( -
- - - - - -
- )} - - {data?.last_poll && ( -

- Last poll: {new Date(data.last_poll).toLocaleString()} -

- )} -
- ); -} - -function StatCard({ - label, - value, - color = "indigo", -}: { - label: string; - value: number; - color?: string; -}) { - const colors: Record = { - indigo: "bg-indigo-50 text-indigo-700", - green: "bg-green-50 text-green-700", - yellow: "bg-yellow-50 text-yellow-700", - red: "bg-red-50 text-red-700", - gray: "bg-gray-50 text-gray-700", - }; - - return ( -
-

{value}

-

{label}

-
- ); -} diff --git a/apps/admin-web/src/pages/dedup-log.tsx b/apps/admin-web/src/pages/dedup-log.tsx deleted file mode 100644 index 6151a5a..0000000 --- a/apps/admin-web/src/pages/dedup-log.tsx +++ /dev/null @@ -1,55 +0,0 @@ -import { useQuery } from "@tanstack/react-query"; -import { getDuplicates } from "../lib/api"; - -export function DedupLog() { - const { data: dupes, isLoading } = useQuery({ - queryKey: ["duplicates"], - queryFn: () => getDuplicates(), - }); - - return ( -
-

Dedup Log

- - {isLoading &&

Loading...

} - -
- - - - - - - - - - - {dupes?.map((d) => ( - - - - - - - ))} - {dupes?.length === 0 && ( - - - - )} - -
Message IDListservStatusDate
- {d.message_id} - {d.listserv_label ?? "—"} - - {d.status} - - - {new Date(d.created_at).toLocaleString()} -
- No duplicates logged yet. -
-
-
- ); -} diff --git a/apps/admin-web/src/pages/event-review.tsx b/apps/admin-web/src/pages/event-review.tsx deleted file mode 100644 index 77ac7df..0000000 --- a/apps/admin-web/src/pages/event-review.tsx +++ /dev/null @@ -1,117 +0,0 @@ -import { useMutation, useQuery, useQueryClient } from "@tanstack/react-query"; -import { useState } from "react"; -import { - type PipelineEvent, - deletePipelineEvent, - getPipelineEvents, - updatePipelineEvent, -} from "../lib/api"; - -export function EventReview() { - const queryClient = useQueryClient(); - const { data: events, isLoading } = useQuery({ - queryKey: ["pipeline-events"], - queryFn: () => getPipelineEvents(), - }); - - const deleteMutation = useMutation({ - mutationFn: deletePipelineEvent, - onSuccess: () => queryClient.invalidateQueries({ queryKey: ["pipeline-events"] }), - }); - - return ( -
-

Pipeline Events

- - {isLoading &&

Loading...

} - -
- - - - - - - - - - - - {events?.map((event) => ( - { - if (confirm(`Delete "${event.title}"?`)) { - deleteMutation.mutate(event.id); - } - }} - /> - ))} - {events?.length === 0 && ( - - - - )} - -
TitleDateLocationCreatedActions
- No pipeline events yet. -
-
-
- ); -} - -function EventRow({ event, onDelete }: { event: PipelineEvent; onDelete: () => void }) { - const queryClient = useQueryClient(); - const [editing, setEditing] = useState(false); - const [title, setTitle] = useState(event.title); - - const updateMutation = useMutation({ - mutationFn: (fields: Record) => updatePipelineEvent(event.id, fields), - onSuccess: () => { - queryClient.invalidateQueries({ queryKey: ["pipeline-events"] }); - setEditing(false); - }, - }); - - return ( - - - {editing ? ( - setTitle(e.target.value)} - onKeyDown={(e) => { - if (e.key === "Enter") updateMutation.mutate({ title }); - if (e.key === "Escape") setEditing(false); - }} - className="w-full px-2 py-1 border border-gray-300 rounded text-sm focus:outline-none focus:ring-2 focus:ring-indigo-300" - /> - ) : ( - - )} - - {new Date(event.datetime).toLocaleDateString()} - {event.location_name ?? "—"} - - {new Date(event.created_at).toLocaleDateString()} - - - - - - ); -} diff --git a/apps/admin-web/src/pages/listserv-config.tsx b/apps/admin-web/src/pages/listserv-config.tsx deleted file mode 100644 index cbd242b..0000000 --- a/apps/admin-web/src/pages/listserv-config.tsx +++ /dev/null @@ -1,138 +0,0 @@ -import { useMutation, useQuery, useQueryClient } from "@tanstack/react-query"; -import { useState } from "react"; -import { createConfig, deleteConfig, getConfigs, updateConfig } from "../lib/api"; - -export function ListservConfig() { - const queryClient = useQueryClient(); - const { data: configs, isLoading } = useQuery({ - queryKey: ["configs"], - queryFn: getConfigs, - }); - - const [showForm, setShowForm] = useState(false); - const [form, setForm] = useState({ address: "", label: "", gmail_label: "" }); - - const createMutation = useMutation({ - mutationFn: createConfig, - onSuccess: () => { - queryClient.invalidateQueries({ queryKey: ["configs"] }); - setShowForm(false); - setForm({ address: "", label: "", gmail_label: "" }); - }, - }); - - const deleteMutation = useMutation({ - mutationFn: deleteConfig, - onSuccess: () => queryClient.invalidateQueries({ queryKey: ["configs"] }), - }); - - const toggleMutation = useMutation({ - mutationFn: ({ id, enabled }: { id: string; enabled: boolean }) => - updateConfig(id, { enabled }), - onSuccess: () => queryClient.invalidateQueries({ queryKey: ["configs"] }), - }); - - return ( -
-
-

Listserv Configs

- -
- - {showForm && ( -
-
- setForm({ ...form, address: e.target.value })} - className="px-3 py-2 border border-gray-300 rounded-lg text-sm focus:outline-none focus:ring-2 focus:ring-indigo-300" - /> - setForm({ ...form, label: e.target.value })} - className="px-3 py-2 border border-gray-300 rounded-lg text-sm focus:outline-none focus:ring-2 focus:ring-indigo-300" - /> - setForm({ ...form, gmail_label: e.target.value })} - className="px-3 py-2 border border-gray-300 rounded-lg text-sm focus:outline-none focus:ring-2 focus:ring-indigo-300" - /> -
- -
- )} - - {isLoading &&

Loading...

} - -
- - - - - - - - - - - - {configs?.map((cfg) => ( - - - - - - - - ))} - {configs?.length === 0 && ( - - - - )} - -
LabelAddressGmail LabelStatusActions
{cfg.label}{cfg.address}{cfg.gmail_label ?? "—"} - - - -
- No listserv configs yet. -
-
-
- ); -} diff --git a/apps/admin-web/src/shell.tsx b/apps/admin-web/src/shell.tsx deleted file mode 100644 index f09ba14..0000000 --- a/apps/admin-web/src/shell.tsx +++ /dev/null @@ -1,87 +0,0 @@ -import { useEffect, useState } from "react"; -import { NavLink, Outlet } from "react-router-dom"; -import { getStoredApiKey, setApiKey } from "./lib/api"; - -const NAV_ITEMS = [ - { to: "/dashboard", label: "Dashboard" }, - { to: "/events", label: "Events" }, - { to: "/configs", label: "Listservs" }, - { to: "/duplicates", label: "Dedup Log" }, -]; - -export function Shell() { - const [key, setKey] = useState(getStoredApiKey()); - const [input, setInput] = useState(""); - - useEffect(() => { - if (!key) return; - setApiKey(key); - }, [key]); - - if (!key) { - return ( -
-
-

Admin Panel

-

Enter the admin API key to continue.

- setInput(e.target.value)} - onKeyDown={(e) => e.key === "Enter" && setKey(input)} - placeholder="API key" - className="w-full px-3 py-2 border border-gray-300 rounded-lg text-sm focus:outline-none focus:ring-2 focus:ring-indigo-300" - /> - -
-
- ); - } - - return ( -
- {/* Sidebar */} - - - {/* Main content */} -
- -
-
- ); -} diff --git a/apps/admin-web/src/style.css b/apps/admin-web/src/style.css deleted file mode 100644 index f1d8c73..0000000 --- a/apps/admin-web/src/style.css +++ /dev/null @@ -1 +0,0 @@ -@import "tailwindcss"; diff --git a/apps/admin-web/tsconfig.json b/apps/admin-web/tsconfig.json deleted file mode 100644 index 6e65fef..0000000 --- a/apps/admin-web/tsconfig.json +++ /dev/null @@ -1,27 +0,0 @@ -{ - "compilerOptions": { - "target": "ES2022", - "useDefineForClassFields": true, - "module": "ESNext", - "lib": ["ES2022", "DOM", "DOM.Iterable"], - "types": ["vite/client"], - "skipLibCheck": true, - "jsx": "react-jsx", - - /* Bundler mode */ - "moduleResolution": "bundler", - "allowImportingTsExtensions": true, - "verbatimModuleSyntax": true, - "moduleDetection": "force", - "noEmit": true, - - /* Linting */ - "strict": true, - "noUnusedLocals": true, - "noUnusedParameters": true, - "erasableSyntaxOnly": true, - "noFallthroughCasesInSwitch": true, - "noUncheckedSideEffectImports": true - }, - "include": ["src"] -} diff --git a/apps/admin-web/vite.config.ts b/apps/admin-web/vite.config.ts deleted file mode 100644 index c41071c..0000000 --- a/apps/admin-web/vite.config.ts +++ /dev/null @@ -1,15 +0,0 @@ -import tailwindcss from "@tailwindcss/vite"; -import { defineConfig } from "vite"; - -export default defineConfig({ - plugins: [tailwindcss()], - server: { - port: 5173, - proxy: { - "/pipeline": { - target: "http://localhost:8000", - changeOrigin: true, - }, - }, - }, -}); diff --git a/apps/database/.env.example b/apps/database/.env.example index 47929e6..b3c596a 100644 --- a/apps/database/.env.example +++ b/apps/database/.env.example @@ -1,2 +1,6 @@ # Copy to .env for local development (used by drizzle-kit) DATABASE_URL=postgresql://forum:forum_password@localhost:5434/the_forum + +# InboxEngine sync (`bun run db:sync-engine`) +# INBOX_ENGINE_URL=https://inbox-engine.tigerapps.org +# INBOX_ENGINE_TOKEN= diff --git a/apps/database/drizzle/0003_add_query_indexes.sql b/apps/database/drizzle/0003_add_query_indexes.sql new file mode 100644 index 0000000..f81acd6 --- /dev/null +++ b/apps/database/drizzle/0003_add_query_indexes.sql @@ -0,0 +1,13 @@ +CREATE INDEX "event_tags_tag_idx" ON "event_tags" USING btree ("tag");--> statement-breakpoint +CREATE INDEX "events_published_datetime_idx" ON "events" USING btree ("datetime") WHERE "events"."status" = 'published';--> statement-breakpoint +CREATE INDEX "events_org_id_idx" ON "events" USING btree ("org_id");--> statement-breakpoint +CREATE INDEX "events_creator_id_idx" ON "events" USING btree ("creator_id");--> statement-breakpoint +CREATE INDEX "events_location_id_idx" ON "events" USING btree ("location_id");--> statement-breakpoint +CREATE INDEX "friendships_friend_id_status_idx" ON "friendships" USING btree ("friend_id","status");--> statement-breakpoint +CREATE INDEX "interactions_user_id_created_at_idx" ON "interactions" USING btree ("user_id","created_at");--> statement-breakpoint +CREATE INDEX "interactions_item_id_type_idx" ON "interactions" USING btree ("item_id","interaction_type");--> statement-breakpoint +CREATE INDEX "notifications_user_id_created_at_idx" ON "notifications" USING btree ("user_id","created_at" DESC NULLS LAST);--> statement-breakpoint +CREATE INDEX "org_followers_user_id_idx" ON "org_followers" USING btree ("user_id");--> statement-breakpoint +CREATE INDEX "org_members_user_id_idx" ON "org_members" USING btree ("user_id");--> statement-breakpoint +CREATE INDEX "rsvps_event_id_idx" ON "rsvps" USING btree ("event_id");--> statement-breakpoint +CREATE INDEX "saved_events_event_id_idx" ON "saved_events" USING btree ("event_id"); \ No newline at end of file diff --git a/apps/database/drizzle/0004_inbox_engine_orgs_events.sql b/apps/database/drizzle/0004_inbox_engine_orgs_events.sql new file mode 100644 index 0000000..5827a26 --- /dev/null +++ b/apps/database/drizzle/0004_inbox_engine_orgs_events.sql @@ -0,0 +1,21 @@ +CREATE TABLE "sync_state" ( + "key" varchar(100) PRIMARY KEY NOT NULL, + "value" text NOT NULL, + "updated_at" timestamp DEFAULT now() NOT NULL +); +--> statement-breakpoint +ALTER TABLE "organizations" ALTER COLUMN "creator_id" DROP NOT NULL;--> statement-breakpoint +ALTER TABLE "events" ADD COLUMN "source_url" text;--> statement-breakpoint +ALTER TABLE "events" ADD COLUMN "location_detail" varchar(200);--> statement-breakpoint +ALTER TABLE "organizations" ADD COLUMN "source" varchar(20) DEFAULT 'manual' NOT NULL;--> statement-breakpoint +ALTER TABLE "organizations" ADD COLUMN "external_id" varchar(64);--> statement-breakpoint +ALTER TABLE "organizations" ADD COLUMN "acronym" varchar(40);--> statement-breakpoint +ALTER TABLE "organizations" ADD COLUMN "tagline" text;--> statement-breakpoint +ALTER TABLE "organizations" ADD COLUMN "group_type" varchar(120);--> statement-breakpoint +ALTER TABLE "organizations" ADD COLUMN "group_url" text;--> statement-breakpoint +ALTER TABLE "organizations" ADD COLUMN "website" text;--> statement-breakpoint +ALTER TABLE "organizations" ADD COLUMN "contact_email" varchar(255);--> statement-breakpoint +ALTER TABLE "organizations" ADD COLUMN "socials" jsonb DEFAULT '{}'::jsonb NOT NULL;--> statement-breakpoint +ALTER TABLE "organizations" ADD COLUMN "member_count" integer;--> statement-breakpoint +ALTER TABLE "organizations" ADD COLUMN "synced_at" timestamp;--> statement-breakpoint +ALTER TABLE "organizations" ADD CONSTRAINT "organizations_external_id_unique" UNIQUE("external_id"); \ No newline at end of file diff --git a/apps/database/drizzle/meta/0003_snapshot.json b/apps/database/drizzle/meta/0003_snapshot.json new file mode 100644 index 0000000..01f1c02 --- /dev/null +++ b/apps/database/drizzle/meta/0003_snapshot.json @@ -0,0 +1,1799 @@ +{ + "id": "ae793359-da5e-4223-b1b9-04c383fba037", + "prevId": "5bee482d-9232-416e-9059-145490932658", + "version": "7", + "dialect": "postgresql", + "tables": { + "public.campus_locations": { + "name": "campus_locations", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "varchar(100)", + "primaryKey": true, + "notNull": true + }, + "name": { + "name": "name", + "type": "varchar(255)", + "primaryKey": false, + "notNull": true + }, + "latitude": { + "name": "latitude", + "type": "double precision", + "primaryKey": false, + "notNull": true + }, + "longitude": { + "name": "longitude", + "type": "double precision", + "primaryKey": false, + "notNull": true + }, + "category": { + "name": "category", + "type": "location_category", + "typeSchema": "public", + "primaryKey": false, + "notNull": true + } + }, + "indexes": {}, + "foreignKeys": {}, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.event_tag_embeddings": { + "name": "event_tag_embeddings", + "schema": "", + "columns": { + "tag_name": { + "name": "tag_name", + "type": "event_tag", + "typeSchema": "public", + "primaryKey": true, + "notNull": true + }, + "embedding": { + "name": "embedding", + "type": "vector(1536)", + "primaryKey": false, + "notNull": true + } + }, + "indexes": {}, + "foreignKeys": {}, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.event_tags": { + "name": "event_tags", + "schema": "", + "columns": { + "event_id": { + "name": "event_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "tag": { + "name": "tag", + "type": "event_tag", + "typeSchema": "public", + "primaryKey": false, + "notNull": true + } + }, + "indexes": { + "event_tags_tag_idx": { + "name": "event_tags_tag_idx", + "columns": [ + { + "expression": "tag", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "event_tags_event_id_events_id_fk": { + "name": "event_tags_event_id_events_id_fk", + "tableFrom": "event_tags", + "tableTo": "events", + "columnsFrom": [ + "event_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": { + "event_tags_event_id_tag_pk": { + "name": "event_tags_event_id_tag_pk", + "columns": [ + "event_id", + "tag" + ] + } + }, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.events": { + "name": "events", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true, + "default": "gen_random_uuid()" + }, + "title": { + "name": "title", + "type": "varchar(200)", + "primaryKey": false, + "notNull": true + }, + "description": { + "name": "description", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "datetime": { + "name": "datetime", + "type": "timestamp", + "primaryKey": false, + "notNull": true + }, + "end_datetime": { + "name": "end_datetime", + "type": "timestamp", + "primaryKey": false, + "notNull": false + }, + "location_id": { + "name": "location_id", + "type": "varchar(100)", + "primaryKey": false, + "notNull": true + }, + "org_id": { + "name": "org_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "creator_id": { + "name": "creator_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "flyer_url": { + "name": "flyer_url", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "cover_preset": { + "name": "cover_preset", + "type": "varchar(50)", + "primaryKey": false, + "notNull": false + }, + "external_link": { + "name": "external_link", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "is_public": { + "name": "is_public", + "type": "boolean", + "primaryKey": false, + "notNull": true, + "default": true + }, + "status": { + "name": "status", + "type": "event_status", + "typeSchema": "public", + "primaryKey": false, + "notNull": true, + "default": "'published'" + }, + "source": { + "name": "source", + "type": "varchar(20)", + "primaryKey": false, + "notNull": true, + "default": "'manual'" + }, + "source_message_id": { + "name": "source_message_id", + "type": "varchar(255)", + "primaryKey": false, + "notNull": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "updated_at": { + "name": "updated_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": { + "events_published_datetime_idx": { + "name": "events_published_datetime_idx", + "columns": [ + { + "expression": "datetime", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "where": "\"events\".\"status\" = 'published'", + "concurrently": false, + "method": "btree", + "with": {} + }, + "events_org_id_idx": { + "name": "events_org_id_idx", + "columns": [ + { + "expression": "org_id", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + }, + "events_creator_id_idx": { + "name": "events_creator_id_idx", + "columns": [ + { + "expression": "creator_id", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + }, + "events_location_id_idx": { + "name": "events_location_id_idx", + "columns": [ + { + "expression": "location_id", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "events_location_id_campus_locations_id_fk": { + "name": "events_location_id_campus_locations_id_fk", + "tableFrom": "events", + "tableTo": "campus_locations", + "columnsFrom": [ + "location_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "no action", + "onUpdate": "no action" + }, + "events_org_id_organizations_id_fk": { + "name": "events_org_id_organizations_id_fk", + "tableFrom": "events", + "tableTo": "organizations", + "columnsFrom": [ + "org_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "set null", + "onUpdate": "no action" + }, + "events_creator_id_users_id_fk": { + "name": "events_creator_id_users_id_fk", + "tableFrom": "events", + "tableTo": "users", + "columnsFrom": [ + "creator_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": { + "events_source_message_id_unique": { + "name": "events_source_message_id_unique", + "nullsNotDistinct": false, + "columns": [ + "source_message_id" + ] + } + }, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.friendships": { + "name": "friendships", + "schema": "", + "columns": { + "user_id": { + "name": "user_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "friend_id": { + "name": "friend_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "status": { + "name": "status", + "type": "friendship_status", + "typeSchema": "public", + "primaryKey": false, + "notNull": true, + "default": "'pending'" + }, + "created_at": { + "name": "created_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": { + "friendships_friend_id_status_idx": { + "name": "friendships_friend_id_status_idx", + "columns": [ + { + "expression": "friend_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "status", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "friendships_user_id_users_id_fk": { + "name": "friendships_user_id_users_id_fk", + "tableFrom": "friendships", + "tableTo": "users", + "columnsFrom": [ + "user_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "friendships_friend_id_users_id_fk": { + "name": "friendships_friend_id_users_id_fk", + "tableFrom": "friendships", + "tableTo": "users", + "columnsFrom": [ + "friend_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": { + "friendships_user_id_friend_id_pk": { + "name": "friendships_user_id_friend_id_pk", + "columns": [ + "user_id", + "friend_id" + ] + } + }, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.interactions": { + "name": "interactions", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true, + "default": "gen_random_uuid()" + }, + "user_id": { + "name": "user_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "item_id": { + "name": "item_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "item_type": { + "name": "item_type", + "type": "item_type", + "typeSchema": "public", + "primaryKey": false, + "notNull": true, + "default": "'event'" + }, + "interaction_type": { + "name": "interaction_type", + "type": "interaction_type", + "typeSchema": "public", + "primaryKey": false, + "notNull": true + }, + "interaction_value": { + "name": "interaction_value", + "type": "double precision", + "primaryKey": false, + "notNull": true + }, + "metadata": { + "name": "metadata", + "type": "jsonb", + "primaryKey": false, + "notNull": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": { + "interactions_user_id_created_at_idx": { + "name": "interactions_user_id_created_at_idx", + "columns": [ + { + "expression": "user_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "created_at", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + }, + "interactions_item_id_type_idx": { + "name": "interactions_item_id_type_idx", + "columns": [ + { + "expression": "item_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "interaction_type", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "interactions_user_id_users_id_fk": { + "name": "interactions_user_id_users_id_fk", + "tableFrom": "interactions", + "tableTo": "users", + "columnsFrom": [ + "user_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.listserv_configs": { + "name": "listserv_configs", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true, + "default": "gen_random_uuid()" + }, + "address": { + "name": "address", + "type": "varchar(255)", + "primaryKey": false, + "notNull": true + }, + "label": { + "name": "label", + "type": "varchar(255)", + "primaryKey": false, + "notNull": true + }, + "org_id": { + "name": "org_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "gmail_label": { + "name": "gmail_label", + "type": "varchar(255)", + "primaryKey": false, + "notNull": false + }, + "enabled": { + "name": "enabled", + "type": "boolean", + "primaryKey": false, + "notNull": true, + "default": true + }, + "created_at": { + "name": "created_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "listserv_configs_org_id_organizations_id_fk": { + "name": "listserv_configs_org_id_organizations_id_fk", + "tableFrom": "listserv_configs", + "tableTo": "organizations", + "columnsFrom": [ + "org_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "set null", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": { + "listserv_configs_address_unique": { + "name": "listserv_configs_address_unique", + "nullsNotDistinct": false, + "columns": [ + "address" + ] + } + }, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.listserv_emails": { + "name": "listserv_emails", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true, + "default": "gen_random_uuid()" + }, + "message_id": { + "name": "message_id", + "type": "varchar(255)", + "primaryKey": false, + "notNull": true + }, + "listserv": { + "name": "listserv", + "type": "varchar(100)", + "primaryKey": false, + "notNull": true + }, + "subject": { + "name": "subject", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "author_name": { + "name": "author_name", + "type": "varchar(255)", + "primaryKey": false, + "notNull": false + }, + "author_email": { + "name": "author_email", + "type": "varchar(255)", + "primaryKey": false, + "notNull": false + }, + "date": { + "name": "date", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": false + }, + "body_text": { + "name": "body_text", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "body_html": { + "name": "body_html", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "is_hoagiemail": { + "name": "is_hoagiemail", + "type": "boolean", + "primaryKey": false, + "notNull": true, + "default": false + }, + "hoagiemail_sender_name": { + "name": "hoagiemail_sender_name", + "type": "varchar(255)", + "primaryKey": false, + "notNull": false + }, + "hoagiemail_sender_email": { + "name": "hoagiemail_sender_email", + "type": "varchar(255)", + "primaryKey": false, + "notNull": false + }, + "links": { + "name": "links", + "type": "jsonb", + "primaryKey": false, + "notNull": false, + "default": "'[]'::jsonb" + }, + "images": { + "name": "images", + "type": "jsonb", + "primaryKey": false, + "notNull": false, + "default": "'[]'::jsonb" + }, + "attachments": { + "name": "attachments", + "type": "jsonb", + "primaryKey": false, + "notNull": false, + "default": "'[]'::jsonb" + }, + "listserv_url": { + "name": "listserv_url", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": {}, + "compositePrimaryKeys": {}, + "uniqueConstraints": { + "listserv_emails_message_id_unique": { + "name": "listserv_emails_message_id_unique", + "nullsNotDistinct": false, + "columns": [ + "message_id" + ] + } + }, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.notifications": { + "name": "notifications", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true, + "default": "gen_random_uuid()" + }, + "user_id": { + "name": "user_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "type": { + "name": "type", + "type": "notification_type", + "typeSchema": "public", + "primaryKey": false, + "notNull": true + }, + "payload": { + "name": "payload", + "type": "jsonb", + "primaryKey": false, + "notNull": true + }, + "read": { + "name": "read", + "type": "boolean", + "primaryKey": false, + "notNull": true, + "default": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": { + "notifications_user_id_created_at_idx": { + "name": "notifications_user_id_created_at_idx", + "columns": [ + { + "expression": "user_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "created_at", + "isExpression": false, + "asc": false, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "notifications_user_id_users_id_fk": { + "name": "notifications_user_id_users_id_fk", + "tableFrom": "notifications", + "tableTo": "users", + "columnsFrom": [ + "user_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.org_followers": { + "name": "org_followers", + "schema": "", + "columns": { + "org_id": { + "name": "org_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "user_id": { + "name": "user_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "created_at": { + "name": "created_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": { + "org_followers_user_id_idx": { + "name": "org_followers_user_id_idx", + "columns": [ + { + "expression": "user_id", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "org_followers_org_id_organizations_id_fk": { + "name": "org_followers_org_id_organizations_id_fk", + "tableFrom": "org_followers", + "tableTo": "organizations", + "columnsFrom": [ + "org_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "org_followers_user_id_users_id_fk": { + "name": "org_followers_user_id_users_id_fk", + "tableFrom": "org_followers", + "tableTo": "users", + "columnsFrom": [ + "user_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": { + "org_followers_org_id_user_id_pk": { + "name": "org_followers_org_id_user_id_pk", + "columns": [ + "org_id", + "user_id" + ] + } + }, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.org_members": { + "name": "org_members", + "schema": "", + "columns": { + "org_id": { + "name": "org_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "user_id": { + "name": "user_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "role": { + "name": "role", + "type": "org_role", + "typeSchema": "public", + "primaryKey": false, + "notNull": true + } + }, + "indexes": { + "org_members_user_id_idx": { + "name": "org_members_user_id_idx", + "columns": [ + { + "expression": "user_id", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "org_members_org_id_organizations_id_fk": { + "name": "org_members_org_id_organizations_id_fk", + "tableFrom": "org_members", + "tableTo": "organizations", + "columnsFrom": [ + "org_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "org_members_user_id_users_id_fk": { + "name": "org_members_user_id_users_id_fk", + "tableFrom": "org_members", + "tableTo": "users", + "columnsFrom": [ + "user_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": { + "org_members_org_id_user_id_pk": { + "name": "org_members_org_id_user_id_pk", + "columns": [ + "org_id", + "user_id" + ] + } + }, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.organizations": { + "name": "organizations", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true, + "default": "gen_random_uuid()" + }, + "name": { + "name": "name", + "type": "varchar(255)", + "primaryKey": false, + "notNull": true + }, + "description": { + "name": "description", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "logo_url": { + "name": "logo_url", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "category": { + "name": "category", + "type": "org_category", + "typeSchema": "public", + "primaryKey": false, + "notNull": true + }, + "creator_id": { + "name": "creator_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "created_at": { + "name": "created_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "updated_at": { + "name": "updated_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "organizations_creator_id_users_id_fk": { + "name": "organizations_creator_id_users_id_fk", + "tableFrom": "organizations", + "tableTo": "users", + "columnsFrom": [ + "creator_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": { + "organizations_name_unique": { + "name": "organizations_name_unique", + "nullsNotDistinct": false, + "columns": [ + "name" + ] + } + }, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.pipeline_logs": { + "name": "pipeline_logs", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true, + "default": "gen_random_uuid()" + }, + "message_id": { + "name": "message_id", + "type": "varchar(255)", + "primaryKey": false, + "notNull": true + }, + "listserv_config_id": { + "name": "listserv_config_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "status": { + "name": "status", + "type": "pipeline_log_status", + "typeSchema": "public", + "primaryKey": false, + "notNull": true + }, + "error_text": { + "name": "error_text", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "extracted_event_id": { + "name": "extracted_event_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "pipeline_logs_listserv_config_id_listserv_configs_id_fk": { + "name": "pipeline_logs_listserv_config_id_listserv_configs_id_fk", + "tableFrom": "pipeline_logs", + "tableTo": "listserv_configs", + "columnsFrom": [ + "listserv_config_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "set null", + "onUpdate": "no action" + }, + "pipeline_logs_extracted_event_id_events_id_fk": { + "name": "pipeline_logs_extracted_event_id_events_id_fk", + "tableFrom": "pipeline_logs", + "tableTo": "events", + "columnsFrom": [ + "extracted_event_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "set null", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.rsvps": { + "name": "rsvps", + "schema": "", + "columns": { + "user_id": { + "name": "user_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "event_id": { + "name": "event_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "created_at": { + "name": "created_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": { + "rsvps_event_id_idx": { + "name": "rsvps_event_id_idx", + "columns": [ + { + "expression": "event_id", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "rsvps_user_id_users_id_fk": { + "name": "rsvps_user_id_users_id_fk", + "tableFrom": "rsvps", + "tableTo": "users", + "columnsFrom": [ + "user_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "rsvps_event_id_events_id_fk": { + "name": "rsvps_event_id_events_id_fk", + "tableFrom": "rsvps", + "tableTo": "events", + "columnsFrom": [ + "event_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": { + "rsvps_user_id_event_id_pk": { + "name": "rsvps_user_id_event_id_pk", + "columns": [ + "user_id", + "event_id" + ] + } + }, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.saved_events": { + "name": "saved_events", + "schema": "", + "columns": { + "user_id": { + "name": "user_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "event_id": { + "name": "event_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "created_at": { + "name": "created_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": { + "saved_events_event_id_idx": { + "name": "saved_events_event_id_idx", + "columns": [ + { + "expression": "event_id", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "saved_events_user_id_users_id_fk": { + "name": "saved_events_user_id_users_id_fk", + "tableFrom": "saved_events", + "tableTo": "users", + "columnsFrom": [ + "user_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "saved_events_event_id_events_id_fk": { + "name": "saved_events_event_id_events_id_fk", + "tableFrom": "saved_events", + "tableTo": "events", + "columnsFrom": [ + "event_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": { + "saved_events_user_id_event_id_pk": { + "name": "saved_events_user_id_event_id_pk", + "columns": [ + "user_id", + "event_id" + ] + } + }, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.user_interests": { + "name": "user_interests", + "schema": "", + "columns": { + "user_id": { + "name": "user_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "tag": { + "name": "tag", + "type": "event_tag", + "typeSchema": "public", + "primaryKey": false, + "notNull": true + } + }, + "indexes": {}, + "foreignKeys": { + "user_interests_user_id_users_id_fk": { + "name": "user_interests_user_id_users_id_fk", + "tableFrom": "user_interests", + "tableTo": "users", + "columnsFrom": [ + "user_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": { + "user_interests_user_id_tag_pk": { + "name": "user_interests_user_id_tag_pk", + "columns": [ + "user_id", + "tag" + ] + } + }, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.user_preference_vectors": { + "name": "user_preference_vectors", + "schema": "", + "columns": { + "user_id": { + "name": "user_id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "tag_weights": { + "name": "tag_weights", + "type": "jsonb", + "primaryKey": false, + "notNull": true, + "default": "'{}'::jsonb" + }, + "latent_vector": { + "name": "latent_vector", + "type": "double precision[]", + "primaryKey": false, + "notNull": false + }, + "interaction_count": { + "name": "interaction_count", + "type": "double precision", + "primaryKey": false, + "notNull": true, + "default": 0 + }, + "updated_at": { + "name": "updated_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "user_preference_vectors_user_id_users_id_fk": { + "name": "user_preference_vectors_user_id_users_id_fk", + "tableFrom": "user_preference_vectors", + "tableTo": "users", + "columnsFrom": [ + "user_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.user_regions": { + "name": "user_regions", + "schema": "", + "columns": { + "user_id": { + "name": "user_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "region": { + "name": "region", + "type": "campus_region", + "typeSchema": "public", + "primaryKey": false, + "notNull": true + } + }, + "indexes": {}, + "foreignKeys": { + "user_regions_user_id_users_id_fk": { + "name": "user_regions_user_id_users_id_fk", + "tableFrom": "user_regions", + "tableTo": "users", + "columnsFrom": [ + "user_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": { + "user_regions_user_id_region_pk": { + "name": "user_regions_user_id_region_pk", + "columns": [ + "user_id", + "region" + ] + } + }, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.users": { + "name": "users", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true, + "default": "gen_random_uuid()" + }, + "net_id": { + "name": "net_id", + "type": "varchar(50)", + "primaryKey": false, + "notNull": true + }, + "email": { + "name": "email", + "type": "varchar(255)", + "primaryKey": false, + "notNull": true + }, + "display_name": { + "name": "display_name", + "type": "varchar(255)", + "primaryKey": false, + "notNull": true + }, + "class_year": { + "name": "class_year", + "type": "varchar(10)", + "primaryKey": false, + "notNull": false + }, + "major": { + "name": "major", + "type": "varchar(255)", + "primaryKey": false, + "notNull": false + }, + "avatar_url": { + "name": "avatar_url", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "is_org_leader": { + "name": "is_org_leader", + "type": "boolean", + "primaryKey": false, + "notNull": true, + "default": false + }, + "onboarded": { + "name": "onboarded", + "type": "boolean", + "primaryKey": false, + "notNull": true, + "default": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "updated_at": { + "name": "updated_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": {}, + "compositePrimaryKeys": {}, + "uniqueConstraints": { + "users_net_id_unique": { + "name": "users_net_id_unique", + "nullsNotDistinct": false, + "columns": [ + "net_id" + ] + }, + "users_email_unique": { + "name": "users_email_unique", + "nullsNotDistinct": false, + "columns": [ + "email" + ] + } + }, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + } + }, + "enums": { + "public.campus_region": { + "name": "campus_region", + "schema": "public", + "values": [ + "central", + "east", + "west", + "south", + "north", + "off-campus" + ] + }, + "public.event_status": { + "name": "event_status", + "schema": "public", + "values": [ + "draft", + "published" + ] + }, + "public.event_tag": { + "name": "event_tag", + "schema": "public", + "values": [ + "free food", + "career", + "research", + "academics", + "tech", + "entrepreneurship", + "politics", + "visual arts", + "performing arts", + "literature", + "culture", + "music", + "gaming", + "athletics", + "religion", + "sustainability", + "outdoors", + "wellness", + "community service", + "speaker event", + "social event", + "stem" + ] + }, + "public.friendship_status": { + "name": "friendship_status", + "schema": "public", + "values": [ + "pending", + "accepted", + "declined" + ] + }, + "public.interaction_type": { + "name": "interaction_type", + "schema": "public", + "values": [ + "view", + "click", + "rsvp", + "save", + "share", + "hide" + ] + }, + "public.item_type": { + "name": "item_type", + "schema": "public", + "values": [ + "event", + "organization" + ] + }, + "public.location_category": { + "name": "location_category", + "schema": "public", + "values": [ + "academic", + "residential", + "athletic", + "social", + "administrative", + "library", + "dining", + "other" + ] + }, + "public.notification_type": { + "name": "notification_type", + "schema": "public", + "values": [ + "friend_request", + "event_reminder", + "org_new_event" + ] + }, + "public.org_category": { + "name": "org_category", + "schema": "public", + "values": [ + "career", + "affinity", + "performing arts", + "academics", + "athletics", + "social event", + "culture", + "religion", + "politics", + "community service" + ] + }, + "public.org_role": { + "name": "org_role", + "schema": "public", + "values": [ + "owner", + "officer", + "member" + ] + }, + "public.pipeline_log_status": { + "name": "pipeline_log_status", + "schema": "public", + "values": [ + "success", + "skipped_not_event", + "duplicate", + "error", + "needs_review" + ] + } + }, + "schemas": {}, + "sequences": {}, + "roles": {}, + "policies": {}, + "views": {}, + "_meta": { + "columns": {}, + "schemas": {}, + "tables": {} + } +} \ No newline at end of file diff --git a/apps/database/drizzle/meta/0004_snapshot.json b/apps/database/drizzle/meta/0004_snapshot.json new file mode 100644 index 0000000..0ba78f1 --- /dev/null +++ b/apps/database/drizzle/meta/0004_snapshot.json @@ -0,0 +1,1918 @@ +{ + "id": "5808d9c6-cd09-4f36-9ac1-b06891802056", + "prevId": "ae793359-da5e-4223-b1b9-04c383fba037", + "version": "7", + "dialect": "postgresql", + "tables": { + "public.campus_locations": { + "name": "campus_locations", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "varchar(100)", + "primaryKey": true, + "notNull": true + }, + "name": { + "name": "name", + "type": "varchar(255)", + "primaryKey": false, + "notNull": true + }, + "latitude": { + "name": "latitude", + "type": "double precision", + "primaryKey": false, + "notNull": true + }, + "longitude": { + "name": "longitude", + "type": "double precision", + "primaryKey": false, + "notNull": true + }, + "category": { + "name": "category", + "type": "location_category", + "typeSchema": "public", + "primaryKey": false, + "notNull": true + } + }, + "indexes": {}, + "foreignKeys": {}, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.event_tag_embeddings": { + "name": "event_tag_embeddings", + "schema": "", + "columns": { + "tag_name": { + "name": "tag_name", + "type": "event_tag", + "typeSchema": "public", + "primaryKey": true, + "notNull": true + }, + "embedding": { + "name": "embedding", + "type": "vector(1536)", + "primaryKey": false, + "notNull": true + } + }, + "indexes": {}, + "foreignKeys": {}, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.event_tags": { + "name": "event_tags", + "schema": "", + "columns": { + "event_id": { + "name": "event_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "tag": { + "name": "tag", + "type": "event_tag", + "typeSchema": "public", + "primaryKey": false, + "notNull": true + } + }, + "indexes": { + "event_tags_tag_idx": { + "name": "event_tags_tag_idx", + "columns": [ + { + "expression": "tag", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "event_tags_event_id_events_id_fk": { + "name": "event_tags_event_id_events_id_fk", + "tableFrom": "event_tags", + "tableTo": "events", + "columnsFrom": [ + "event_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": { + "event_tags_event_id_tag_pk": { + "name": "event_tags_event_id_tag_pk", + "columns": [ + "event_id", + "tag" + ] + } + }, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.events": { + "name": "events", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true, + "default": "gen_random_uuid()" + }, + "title": { + "name": "title", + "type": "varchar(200)", + "primaryKey": false, + "notNull": true + }, + "description": { + "name": "description", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "datetime": { + "name": "datetime", + "type": "timestamp", + "primaryKey": false, + "notNull": true + }, + "end_datetime": { + "name": "end_datetime", + "type": "timestamp", + "primaryKey": false, + "notNull": false + }, + "location_id": { + "name": "location_id", + "type": "varchar(100)", + "primaryKey": false, + "notNull": true + }, + "org_id": { + "name": "org_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "creator_id": { + "name": "creator_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "flyer_url": { + "name": "flyer_url", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "cover_preset": { + "name": "cover_preset", + "type": "varchar(50)", + "primaryKey": false, + "notNull": false + }, + "external_link": { + "name": "external_link", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "is_public": { + "name": "is_public", + "type": "boolean", + "primaryKey": false, + "notNull": true, + "default": true + }, + "status": { + "name": "status", + "type": "event_status", + "typeSchema": "public", + "primaryKey": false, + "notNull": true, + "default": "'published'" + }, + "source": { + "name": "source", + "type": "varchar(20)", + "primaryKey": false, + "notNull": true, + "default": "'manual'" + }, + "source_message_id": { + "name": "source_message_id", + "type": "varchar(255)", + "primaryKey": false, + "notNull": false + }, + "source_url": { + "name": "source_url", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "location_detail": { + "name": "location_detail", + "type": "varchar(200)", + "primaryKey": false, + "notNull": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "updated_at": { + "name": "updated_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": { + "events_published_datetime_idx": { + "name": "events_published_datetime_idx", + "columns": [ + { + "expression": "datetime", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "where": "\"events\".\"status\" = 'published'", + "concurrently": false, + "method": "btree", + "with": {} + }, + "events_org_id_idx": { + "name": "events_org_id_idx", + "columns": [ + { + "expression": "org_id", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + }, + "events_creator_id_idx": { + "name": "events_creator_id_idx", + "columns": [ + { + "expression": "creator_id", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + }, + "events_location_id_idx": { + "name": "events_location_id_idx", + "columns": [ + { + "expression": "location_id", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "events_location_id_campus_locations_id_fk": { + "name": "events_location_id_campus_locations_id_fk", + "tableFrom": "events", + "tableTo": "campus_locations", + "columnsFrom": [ + "location_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "no action", + "onUpdate": "no action" + }, + "events_org_id_organizations_id_fk": { + "name": "events_org_id_organizations_id_fk", + "tableFrom": "events", + "tableTo": "organizations", + "columnsFrom": [ + "org_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "set null", + "onUpdate": "no action" + }, + "events_creator_id_users_id_fk": { + "name": "events_creator_id_users_id_fk", + "tableFrom": "events", + "tableTo": "users", + "columnsFrom": [ + "creator_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": { + "events_source_message_id_unique": { + "name": "events_source_message_id_unique", + "nullsNotDistinct": false, + "columns": [ + "source_message_id" + ] + } + }, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.friendships": { + "name": "friendships", + "schema": "", + "columns": { + "user_id": { + "name": "user_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "friend_id": { + "name": "friend_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "status": { + "name": "status", + "type": "friendship_status", + "typeSchema": "public", + "primaryKey": false, + "notNull": true, + "default": "'pending'" + }, + "created_at": { + "name": "created_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": { + "friendships_friend_id_status_idx": { + "name": "friendships_friend_id_status_idx", + "columns": [ + { + "expression": "friend_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "status", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "friendships_user_id_users_id_fk": { + "name": "friendships_user_id_users_id_fk", + "tableFrom": "friendships", + "tableTo": "users", + "columnsFrom": [ + "user_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "friendships_friend_id_users_id_fk": { + "name": "friendships_friend_id_users_id_fk", + "tableFrom": "friendships", + "tableTo": "users", + "columnsFrom": [ + "friend_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": { + "friendships_user_id_friend_id_pk": { + "name": "friendships_user_id_friend_id_pk", + "columns": [ + "user_id", + "friend_id" + ] + } + }, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.interactions": { + "name": "interactions", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true, + "default": "gen_random_uuid()" + }, + "user_id": { + "name": "user_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "item_id": { + "name": "item_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "item_type": { + "name": "item_type", + "type": "item_type", + "typeSchema": "public", + "primaryKey": false, + "notNull": true, + "default": "'event'" + }, + "interaction_type": { + "name": "interaction_type", + "type": "interaction_type", + "typeSchema": "public", + "primaryKey": false, + "notNull": true + }, + "interaction_value": { + "name": "interaction_value", + "type": "double precision", + "primaryKey": false, + "notNull": true + }, + "metadata": { + "name": "metadata", + "type": "jsonb", + "primaryKey": false, + "notNull": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": { + "interactions_user_id_created_at_idx": { + "name": "interactions_user_id_created_at_idx", + "columns": [ + { + "expression": "user_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "created_at", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + }, + "interactions_item_id_type_idx": { + "name": "interactions_item_id_type_idx", + "columns": [ + { + "expression": "item_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "interaction_type", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "interactions_user_id_users_id_fk": { + "name": "interactions_user_id_users_id_fk", + "tableFrom": "interactions", + "tableTo": "users", + "columnsFrom": [ + "user_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.listserv_configs": { + "name": "listserv_configs", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true, + "default": "gen_random_uuid()" + }, + "address": { + "name": "address", + "type": "varchar(255)", + "primaryKey": false, + "notNull": true + }, + "label": { + "name": "label", + "type": "varchar(255)", + "primaryKey": false, + "notNull": true + }, + "org_id": { + "name": "org_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "gmail_label": { + "name": "gmail_label", + "type": "varchar(255)", + "primaryKey": false, + "notNull": false + }, + "enabled": { + "name": "enabled", + "type": "boolean", + "primaryKey": false, + "notNull": true, + "default": true + }, + "created_at": { + "name": "created_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "listserv_configs_org_id_organizations_id_fk": { + "name": "listserv_configs_org_id_organizations_id_fk", + "tableFrom": "listserv_configs", + "tableTo": "organizations", + "columnsFrom": [ + "org_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "set null", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": { + "listserv_configs_address_unique": { + "name": "listserv_configs_address_unique", + "nullsNotDistinct": false, + "columns": [ + "address" + ] + } + }, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.listserv_emails": { + "name": "listserv_emails", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true, + "default": "gen_random_uuid()" + }, + "message_id": { + "name": "message_id", + "type": "varchar(255)", + "primaryKey": false, + "notNull": true + }, + "listserv": { + "name": "listserv", + "type": "varchar(100)", + "primaryKey": false, + "notNull": true + }, + "subject": { + "name": "subject", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "author_name": { + "name": "author_name", + "type": "varchar(255)", + "primaryKey": false, + "notNull": false + }, + "author_email": { + "name": "author_email", + "type": "varchar(255)", + "primaryKey": false, + "notNull": false + }, + "date": { + "name": "date", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": false + }, + "body_text": { + "name": "body_text", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "body_html": { + "name": "body_html", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "is_hoagiemail": { + "name": "is_hoagiemail", + "type": "boolean", + "primaryKey": false, + "notNull": true, + "default": false + }, + "hoagiemail_sender_name": { + "name": "hoagiemail_sender_name", + "type": "varchar(255)", + "primaryKey": false, + "notNull": false + }, + "hoagiemail_sender_email": { + "name": "hoagiemail_sender_email", + "type": "varchar(255)", + "primaryKey": false, + "notNull": false + }, + "links": { + "name": "links", + "type": "jsonb", + "primaryKey": false, + "notNull": false, + "default": "'[]'::jsonb" + }, + "images": { + "name": "images", + "type": "jsonb", + "primaryKey": false, + "notNull": false, + "default": "'[]'::jsonb" + }, + "attachments": { + "name": "attachments", + "type": "jsonb", + "primaryKey": false, + "notNull": false, + "default": "'[]'::jsonb" + }, + "listserv_url": { + "name": "listserv_url", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": {}, + "compositePrimaryKeys": {}, + "uniqueConstraints": { + "listserv_emails_message_id_unique": { + "name": "listserv_emails_message_id_unique", + "nullsNotDistinct": false, + "columns": [ + "message_id" + ] + } + }, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.notifications": { + "name": "notifications", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true, + "default": "gen_random_uuid()" + }, + "user_id": { + "name": "user_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "type": { + "name": "type", + "type": "notification_type", + "typeSchema": "public", + "primaryKey": false, + "notNull": true + }, + "payload": { + "name": "payload", + "type": "jsonb", + "primaryKey": false, + "notNull": true + }, + "read": { + "name": "read", + "type": "boolean", + "primaryKey": false, + "notNull": true, + "default": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": { + "notifications_user_id_created_at_idx": { + "name": "notifications_user_id_created_at_idx", + "columns": [ + { + "expression": "user_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "created_at", + "isExpression": false, + "asc": false, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "notifications_user_id_users_id_fk": { + "name": "notifications_user_id_users_id_fk", + "tableFrom": "notifications", + "tableTo": "users", + "columnsFrom": [ + "user_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.org_followers": { + "name": "org_followers", + "schema": "", + "columns": { + "org_id": { + "name": "org_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "user_id": { + "name": "user_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "created_at": { + "name": "created_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": { + "org_followers_user_id_idx": { + "name": "org_followers_user_id_idx", + "columns": [ + { + "expression": "user_id", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "org_followers_org_id_organizations_id_fk": { + "name": "org_followers_org_id_organizations_id_fk", + "tableFrom": "org_followers", + "tableTo": "organizations", + "columnsFrom": [ + "org_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "org_followers_user_id_users_id_fk": { + "name": "org_followers_user_id_users_id_fk", + "tableFrom": "org_followers", + "tableTo": "users", + "columnsFrom": [ + "user_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": { + "org_followers_org_id_user_id_pk": { + "name": "org_followers_org_id_user_id_pk", + "columns": [ + "org_id", + "user_id" + ] + } + }, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.org_members": { + "name": "org_members", + "schema": "", + "columns": { + "org_id": { + "name": "org_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "user_id": { + "name": "user_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "role": { + "name": "role", + "type": "org_role", + "typeSchema": "public", + "primaryKey": false, + "notNull": true + } + }, + "indexes": { + "org_members_user_id_idx": { + "name": "org_members_user_id_idx", + "columns": [ + { + "expression": "user_id", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "org_members_org_id_organizations_id_fk": { + "name": "org_members_org_id_organizations_id_fk", + "tableFrom": "org_members", + "tableTo": "organizations", + "columnsFrom": [ + "org_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "org_members_user_id_users_id_fk": { + "name": "org_members_user_id_users_id_fk", + "tableFrom": "org_members", + "tableTo": "users", + "columnsFrom": [ + "user_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": { + "org_members_org_id_user_id_pk": { + "name": "org_members_org_id_user_id_pk", + "columns": [ + "org_id", + "user_id" + ] + } + }, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.organizations": { + "name": "organizations", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true, + "default": "gen_random_uuid()" + }, + "name": { + "name": "name", + "type": "varchar(255)", + "primaryKey": false, + "notNull": true + }, + "description": { + "name": "description", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "logo_url": { + "name": "logo_url", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "category": { + "name": "category", + "type": "org_category", + "typeSchema": "public", + "primaryKey": false, + "notNull": true + }, + "creator_id": { + "name": "creator_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "source": { + "name": "source", + "type": "varchar(20)", + "primaryKey": false, + "notNull": true, + "default": "'manual'" + }, + "external_id": { + "name": "external_id", + "type": "varchar(64)", + "primaryKey": false, + "notNull": false + }, + "acronym": { + "name": "acronym", + "type": "varchar(40)", + "primaryKey": false, + "notNull": false + }, + "tagline": { + "name": "tagline", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "group_type": { + "name": "group_type", + "type": "varchar(120)", + "primaryKey": false, + "notNull": false + }, + "group_url": { + "name": "group_url", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "website": { + "name": "website", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "contact_email": { + "name": "contact_email", + "type": "varchar(255)", + "primaryKey": false, + "notNull": false + }, + "socials": { + "name": "socials", + "type": "jsonb", + "primaryKey": false, + "notNull": true, + "default": "'{}'::jsonb" + }, + "member_count": { + "name": "member_count", + "type": "integer", + "primaryKey": false, + "notNull": false + }, + "synced_at": { + "name": "synced_at", + "type": "timestamp", + "primaryKey": false, + "notNull": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "updated_at": { + "name": "updated_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "organizations_creator_id_users_id_fk": { + "name": "organizations_creator_id_users_id_fk", + "tableFrom": "organizations", + "tableTo": "users", + "columnsFrom": [ + "creator_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": { + "organizations_name_unique": { + "name": "organizations_name_unique", + "nullsNotDistinct": false, + "columns": [ + "name" + ] + }, + "organizations_external_id_unique": { + "name": "organizations_external_id_unique", + "nullsNotDistinct": false, + "columns": [ + "external_id" + ] + } + }, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.pipeline_logs": { + "name": "pipeline_logs", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true, + "default": "gen_random_uuid()" + }, + "message_id": { + "name": "message_id", + "type": "varchar(255)", + "primaryKey": false, + "notNull": true + }, + "listserv_config_id": { + "name": "listserv_config_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "status": { + "name": "status", + "type": "pipeline_log_status", + "typeSchema": "public", + "primaryKey": false, + "notNull": true + }, + "error_text": { + "name": "error_text", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "extracted_event_id": { + "name": "extracted_event_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "pipeline_logs_listserv_config_id_listserv_configs_id_fk": { + "name": "pipeline_logs_listserv_config_id_listserv_configs_id_fk", + "tableFrom": "pipeline_logs", + "tableTo": "listserv_configs", + "columnsFrom": [ + "listserv_config_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "set null", + "onUpdate": "no action" + }, + "pipeline_logs_extracted_event_id_events_id_fk": { + "name": "pipeline_logs_extracted_event_id_events_id_fk", + "tableFrom": "pipeline_logs", + "tableTo": "events", + "columnsFrom": [ + "extracted_event_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "set null", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.rsvps": { + "name": "rsvps", + "schema": "", + "columns": { + "user_id": { + "name": "user_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "event_id": { + "name": "event_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "created_at": { + "name": "created_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": { + "rsvps_event_id_idx": { + "name": "rsvps_event_id_idx", + "columns": [ + { + "expression": "event_id", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "rsvps_user_id_users_id_fk": { + "name": "rsvps_user_id_users_id_fk", + "tableFrom": "rsvps", + "tableTo": "users", + "columnsFrom": [ + "user_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "rsvps_event_id_events_id_fk": { + "name": "rsvps_event_id_events_id_fk", + "tableFrom": "rsvps", + "tableTo": "events", + "columnsFrom": [ + "event_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": { + "rsvps_user_id_event_id_pk": { + "name": "rsvps_user_id_event_id_pk", + "columns": [ + "user_id", + "event_id" + ] + } + }, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.saved_events": { + "name": "saved_events", + "schema": "", + "columns": { + "user_id": { + "name": "user_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "event_id": { + "name": "event_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "created_at": { + "name": "created_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": { + "saved_events_event_id_idx": { + "name": "saved_events_event_id_idx", + "columns": [ + { + "expression": "event_id", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "saved_events_user_id_users_id_fk": { + "name": "saved_events_user_id_users_id_fk", + "tableFrom": "saved_events", + "tableTo": "users", + "columnsFrom": [ + "user_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "saved_events_event_id_events_id_fk": { + "name": "saved_events_event_id_events_id_fk", + "tableFrom": "saved_events", + "tableTo": "events", + "columnsFrom": [ + "event_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": { + "saved_events_user_id_event_id_pk": { + "name": "saved_events_user_id_event_id_pk", + "columns": [ + "user_id", + "event_id" + ] + } + }, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.sync_state": { + "name": "sync_state", + "schema": "", + "columns": { + "key": { + "name": "key", + "type": "varchar(100)", + "primaryKey": true, + "notNull": true + }, + "value": { + "name": "value", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "updated_at": { + "name": "updated_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": {}, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.user_interests": { + "name": "user_interests", + "schema": "", + "columns": { + "user_id": { + "name": "user_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "tag": { + "name": "tag", + "type": "event_tag", + "typeSchema": "public", + "primaryKey": false, + "notNull": true + } + }, + "indexes": {}, + "foreignKeys": { + "user_interests_user_id_users_id_fk": { + "name": "user_interests_user_id_users_id_fk", + "tableFrom": "user_interests", + "tableTo": "users", + "columnsFrom": [ + "user_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": { + "user_interests_user_id_tag_pk": { + "name": "user_interests_user_id_tag_pk", + "columns": [ + "user_id", + "tag" + ] + } + }, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.user_preference_vectors": { + "name": "user_preference_vectors", + "schema": "", + "columns": { + "user_id": { + "name": "user_id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "tag_weights": { + "name": "tag_weights", + "type": "jsonb", + "primaryKey": false, + "notNull": true, + "default": "'{}'::jsonb" + }, + "latent_vector": { + "name": "latent_vector", + "type": "double precision[]", + "primaryKey": false, + "notNull": false + }, + "interaction_count": { + "name": "interaction_count", + "type": "double precision", + "primaryKey": false, + "notNull": true, + "default": 0 + }, + "updated_at": { + "name": "updated_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "user_preference_vectors_user_id_users_id_fk": { + "name": "user_preference_vectors_user_id_users_id_fk", + "tableFrom": "user_preference_vectors", + "tableTo": "users", + "columnsFrom": [ + "user_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.user_regions": { + "name": "user_regions", + "schema": "", + "columns": { + "user_id": { + "name": "user_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "region": { + "name": "region", + "type": "campus_region", + "typeSchema": "public", + "primaryKey": false, + "notNull": true + } + }, + "indexes": {}, + "foreignKeys": { + "user_regions_user_id_users_id_fk": { + "name": "user_regions_user_id_users_id_fk", + "tableFrom": "user_regions", + "tableTo": "users", + "columnsFrom": [ + "user_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": { + "user_regions_user_id_region_pk": { + "name": "user_regions_user_id_region_pk", + "columns": [ + "user_id", + "region" + ] + } + }, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.users": { + "name": "users", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true, + "default": "gen_random_uuid()" + }, + "net_id": { + "name": "net_id", + "type": "varchar(50)", + "primaryKey": false, + "notNull": true + }, + "email": { + "name": "email", + "type": "varchar(255)", + "primaryKey": false, + "notNull": true + }, + "display_name": { + "name": "display_name", + "type": "varchar(255)", + "primaryKey": false, + "notNull": true + }, + "class_year": { + "name": "class_year", + "type": "varchar(10)", + "primaryKey": false, + "notNull": false + }, + "major": { + "name": "major", + "type": "varchar(255)", + "primaryKey": false, + "notNull": false + }, + "avatar_url": { + "name": "avatar_url", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "is_org_leader": { + "name": "is_org_leader", + "type": "boolean", + "primaryKey": false, + "notNull": true, + "default": false + }, + "onboarded": { + "name": "onboarded", + "type": "boolean", + "primaryKey": false, + "notNull": true, + "default": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "updated_at": { + "name": "updated_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": {}, + "compositePrimaryKeys": {}, + "uniqueConstraints": { + "users_net_id_unique": { + "name": "users_net_id_unique", + "nullsNotDistinct": false, + "columns": [ + "net_id" + ] + }, + "users_email_unique": { + "name": "users_email_unique", + "nullsNotDistinct": false, + "columns": [ + "email" + ] + } + }, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + } + }, + "enums": { + "public.campus_region": { + "name": "campus_region", + "schema": "public", + "values": [ + "central", + "east", + "west", + "south", + "north", + "off-campus" + ] + }, + "public.event_status": { + "name": "event_status", + "schema": "public", + "values": [ + "draft", + "published" + ] + }, + "public.event_tag": { + "name": "event_tag", + "schema": "public", + "values": [ + "free food", + "career", + "research", + "academics", + "tech", + "entrepreneurship", + "politics", + "visual arts", + "performing arts", + "literature", + "culture", + "music", + "gaming", + "athletics", + "religion", + "sustainability", + "outdoors", + "wellness", + "community service", + "speaker event", + "social event", + "stem" + ] + }, + "public.friendship_status": { + "name": "friendship_status", + "schema": "public", + "values": [ + "pending", + "accepted", + "declined" + ] + }, + "public.interaction_type": { + "name": "interaction_type", + "schema": "public", + "values": [ + "view", + "click", + "rsvp", + "save", + "share", + "hide" + ] + }, + "public.item_type": { + "name": "item_type", + "schema": "public", + "values": [ + "event", + "organization" + ] + }, + "public.location_category": { + "name": "location_category", + "schema": "public", + "values": [ + "academic", + "residential", + "athletic", + "social", + "administrative", + "library", + "dining", + "other" + ] + }, + "public.notification_type": { + "name": "notification_type", + "schema": "public", + "values": [ + "friend_request", + "event_reminder", + "org_new_event" + ] + }, + "public.org_category": { + "name": "org_category", + "schema": "public", + "values": [ + "career", + "affinity", + "performing arts", + "academics", + "athletics", + "social event", + "culture", + "religion", + "politics", + "community service" + ] + }, + "public.org_role": { + "name": "org_role", + "schema": "public", + "values": [ + "owner", + "officer", + "member" + ] + }, + "public.pipeline_log_status": { + "name": "pipeline_log_status", + "schema": "public", + "values": [ + "success", + "skipped_not_event", + "duplicate", + "error", + "needs_review" + ] + } + }, + "schemas": {}, + "sequences": {}, + "roles": {}, + "policies": {}, + "views": {}, + "_meta": { + "columns": {}, + "schemas": {}, + "tables": {} + } +} \ No newline at end of file diff --git a/apps/database/drizzle/meta/_journal.json b/apps/database/drizzle/meta/_journal.json index ba366a1..654b258 100644 --- a/apps/database/drizzle/meta/_journal.json +++ b/apps/database/drizzle/meta/_journal.json @@ -22,6 +22,20 @@ "when": 1785720843992, "tag": "0002_update_enums", "breakpoints": true + }, + { + "idx": 3, + "version": "7", + "when": 1790578328159, + "tag": "0003_add_query_indexes", + "breakpoints": true + }, + { + "idx": 4, + "version": "7", + "when": 1790579109504, + "tag": "0004_inbox_engine_orgs_events", + "breakpoints": true } ] -} +} \ No newline at end of file diff --git a/apps/database/package.json b/apps/database/package.json index 75e7f78..d5a9efc 100644 --- a/apps/database/package.json +++ b/apps/database/package.json @@ -12,7 +12,8 @@ "db:studio": "drizzle-kit studio", "db:seed": "bun run src/seed.ts", "lint": "biome check .", - "check-types": "tsc --noEmit" + "check-types": "tsc --noEmit", + "db:sync-engine": "bun run src/sync-inbox-engine.ts" }, "dependencies": { "drizzle-orm": "^0.45.2", diff --git a/apps/database/src/migrate.ts b/apps/database/src/migrate.ts new file mode 100644 index 0000000..8f89bd0 --- /dev/null +++ b/apps/database/src/migrate.ts @@ -0,0 +1,18 @@ +/** + * Apply Drizzle SQL migrations (apps/database/drizzle) with drizzle-orm's migrator. + * Used by deployments, where drizzle-kit isn't installed; bundled by deploy/build-release.sh. + * DATABASE_URL=… MIGRATIONS_DIR=… bun tools/migrate.js + */ +import { drizzle } from "drizzle-orm/postgres-js"; +import { migrate } from "drizzle-orm/postgres-js/migrator"; +import postgres from "postgres"; + +const url = process.env.DATABASE_URL; +if (!url) throw new Error("DATABASE_URL is required"); +const migrationsFolder = + process.env.MIGRATIONS_DIR ?? new URL("../drizzle", import.meta.url).pathname; + +const client = postgres(url, { max: 1, onnotice: () => {} }); +await migrate(drizzle(client), { migrationsFolder }); +await client.end(); +console.log(`Migrations applied from ${migrationsFolder}`); diff --git a/apps/database/src/schema/index.ts b/apps/database/src/schema/index.ts index 388f1f5..d399341 100644 --- a/apps/database/src/schema/index.ts +++ b/apps/database/src/schema/index.ts @@ -1,7 +1,9 @@ -import { relations } from "drizzle-orm"; +import { relations, sql } from "drizzle-orm"; import { boolean, doublePrecision, + index, + integer, jsonb, pgEnum, pgTable, @@ -152,9 +154,24 @@ export const organizations = pgTable("organizations", { description: text("description"), logoUrl: text("logo_url"), category: orgCategoryEnum("category").notNull(), - creatorId: uuid("creator_id") - .notNull() - .references(() => users.id), + /** Null for organizations imported from MyPrincetonU (no Forum user created them). */ + creatorId: uuid("creator_id").references(() => users.id), + /** 'myprincetonu' (synced from InboxEngine) or 'manual' (created in The Forum). */ + source: varchar("source", { length: 20 }).default("manual").notNull(), + /** Stable InboxEngine organization ID, e.g. "mpu:52941" for MyPrincetonU group 52941. */ + externalId: varchar("external_id", { length: 64 }).unique(), + acronym: varchar("acronym", { length: 40 }), + tagline: text("tagline"), + groupType: varchar("group_type", { length: 120 }), + /** The organization's MyPrincetonU group page. */ + groupUrl: text("group_url"), + website: text("website"), + contactEmail: varchar("contact_email", { length: 255 }), + /** { instagram?, facebook?, linkedin?, twitter?, youtube? } → URLs */ + socials: jsonb("socials").$type>().default({}).notNull(), + /** Member count reported by MyPrincetonU (not Forum users). */ + memberCount: integer("member_count"), + syncedAt: timestamp("synced_at"), createdAt: timestamp("created_at").defaultNow().notNull(), updatedAt: timestamp("updated_at").defaultNow().notNull(), }); @@ -170,7 +187,11 @@ export const orgMembers = pgTable( .references(() => users.id, { onDelete: "cascade" }), role: orgRoleEnum("role").notNull(), }, - (t) => [primaryKey({ columns: [t.orgId, t.userId] })], + (t) => [ + primaryKey({ columns: [t.orgId, t.userId] }), + // "Orgs I belong to / manage" — the PK leads with org_id, so it can't serve this. + index("org_members_user_id_idx").on(t.userId), + ], ); export const orgFollowers = pgTable( @@ -184,34 +205,58 @@ export const orgFollowers = pgTable( .references(() => users.id, { onDelete: "cascade" }), createdAt: timestamp("created_at").defaultNow().notNull(), }, - (t) => [primaryKey({ columns: [t.orgId, t.userId] })], + (t) => [ + primaryKey({ columns: [t.orgId, t.userId] }), + // "Orgs I follow" (feed org affinity, follow state). + index("org_followers_user_id_idx").on(t.userId), + ], ); -export const events = pgTable("events", { - id: uuid("id").defaultRandom().primaryKey(), - title: varchar("title", { length: 200 }).notNull(), - description: text("description").notNull(), - datetime: timestamp("datetime").notNull(), - endDatetime: timestamp("end_datetime"), - locationId: varchar("location_id", { length: 100 }) - .notNull() - .references(() => campusLocations.id), - orgId: uuid("org_id").references(() => organizations.id, { - onDelete: "set null", - }), - creatorId: uuid("creator_id") - .notNull() - .references(() => users.id), - flyerUrl: text("flyer_url"), - coverPreset: varchar("cover_preset", { length: 50 }), - externalLink: text("external_link"), - isPublic: boolean("is_public").default(true).notNull(), - status: eventStatusEnum("status").default("published").notNull(), - source: varchar("source", { length: 20 }).default("manual").notNull(), - sourceMessageId: varchar("source_message_id", { length: 255 }).unique(), - createdAt: timestamp("created_at").defaultNow().notNull(), - updatedAt: timestamp("updated_at").defaultNow().notNull(), -}); +export const events = pgTable( + "events", + { + id: uuid("id").defaultRandom().primaryKey(), + title: varchar("title", { length: 200 }).notNull(), + description: text("description").notNull(), + datetime: timestamp("datetime").notNull(), + endDatetime: timestamp("end_datetime"), + locationId: varchar("location_id", { length: 100 }) + .notNull() + .references(() => campusLocations.id), + orgId: uuid("org_id").references(() => organizations.id, { + onDelete: "set null", + }), + creatorId: uuid("creator_id") + .notNull() + .references(() => users.id), + flyerUrl: text("flyer_url"), + coverPreset: varchar("cover_preset", { length: 50 }), + externalLink: text("external_link"), + isPublic: boolean("is_public").default(true).notNull(), + status: eventStatusEnum("status").default("published").notNull(), + /** 'manual' | 'myprincetonu' (official) | 'listserv' (extracted from email) | legacy values. */ + source: varchar("source", { length: 20 }).default("manual").notNull(), + /** For imported events: "ie:". */ + sourceMessageId: varchar("source_message_id", { length: 255 }).unique(), + /** Where an imported event came from (MyPrincetonU page or the source email). */ + sourceUrl: text("source_url"), + /** Room or free-text place beyond the campus location ("Room 104", "Zoom"). */ + locationDetail: varchar("location_detail", { length: 200 }), + createdAt: timestamp("created_at").defaultNow().notNull(), + updatedAt: timestamp("updated_at").defaultNow().notNull(), + }, + (t) => [ + // Feed / map / search candidate scans: upcoming published events by time. + // Partial, so drafts never bloat it; discovery queries always filter on + // status = 'published', which lets the planner use it. + index("events_published_datetime_idx") + .on(t.datetime) + .where(sql`${t.status} = 'published'`), + index("events_org_id_idx").on(t.orgId), + index("events_creator_id_idx").on(t.creatorId), + index("events_location_id_idx").on(t.locationId), + ], +); export const eventTags = pgTable( "event_tags", @@ -221,7 +266,11 @@ export const eventTags = pgTable( .references(() => events.id, { onDelete: "cascade" }), tag: eventTagEnum("tag").notNull(), }, - (t) => [primaryKey({ columns: [t.eventId, t.tag] })], + (t) => [ + primaryKey({ columns: [t.eventId, t.tag] }), + // Tag filters and interest matching look events up by tag. + index("event_tags_tag_idx").on(t.tag), + ], ); export const eventTagEmbeddings = pgTable("event_tag_embeddings", { @@ -240,7 +289,11 @@ export const rsvps = pgTable( .references(() => events.id, { onDelete: "cascade" }), createdAt: timestamp("created_at").defaultNow().notNull(), }, - (t) => [primaryKey({ columns: [t.userId, t.eventId] })], + (t) => [ + primaryKey({ columns: [t.userId, t.eventId] }), + // Attendee lists / RSVP counts per event (PK leads with user_id). + index("rsvps_event_id_idx").on(t.eventId), + ], ); export const savedEvents = pgTable( @@ -254,7 +307,11 @@ export const savedEvents = pgTable( .references(() => events.id, { onDelete: "cascade" }), createdAt: timestamp("created_at").defaultNow().notNull(), }, - (t) => [primaryKey({ columns: [t.userId, t.eventId] })], + (t) => [ + primaryKey({ columns: [t.userId, t.eventId] }), + // Per-event lookups and ON DELETE CASCADE from events. + index("saved_events_event_id_idx").on(t.eventId), + ], ); export const friendships = pgTable( @@ -269,19 +326,30 @@ export const friendships = pgTable( status: friendshipStatusEnum("status").default("pending").notNull(), createdAt: timestamp("created_at").defaultNow().notNull(), }, - (t) => [primaryKey({ columns: [t.userId, t.friendId] })], + (t) => [ + primaryKey({ columns: [t.userId, t.friendId] }), + // Reverse direction: incoming requests / friends where I'm the recipient. + index("friendships_friend_id_status_idx").on(t.friendId, t.status), + ], ); -export const notifications = pgTable("notifications", { - id: uuid("id").defaultRandom().primaryKey(), - userId: uuid("user_id") - .notNull() - .references(() => users.id, { onDelete: "cascade" }), - type: notificationTypeEnum("type").notNull(), - payload: jsonb("payload").notNull(), - read: boolean("read").default(false).notNull(), - createdAt: timestamp("created_at").defaultNow().notNull(), -}); +export const notifications = pgTable( + "notifications", + { + id: uuid("id").defaultRandom().primaryKey(), + userId: uuid("user_id") + .notNull() + .references(() => users.id, { onDelete: "cascade" }), + type: notificationTypeEnum("type").notNull(), + payload: jsonb("payload").notNull(), + read: boolean("read").default(false).notNull(), + createdAt: timestamp("created_at").defaultNow().notNull(), + }, + (t) => [ + // The dropdown: a user's newest notifications first. + index("notifications_user_id_created_at_idx").on(t.userId, t.createdAt.desc()), + ], +); export const pipelineLogStatusEnum = pgEnum("pipeline_log_status", [ "success", @@ -339,21 +407,39 @@ export const listservEmails = pgTable("listserv_emails", { createdAt: timestamp("created_at").defaultNow().notNull(), }); -// ── Recommendation / ML tables ──────────────────────────── +// ── Integration state ──────────────────────────────────── -export const interactions = pgTable("interactions", { - id: uuid("id").defaultRandom().primaryKey(), - userId: uuid("user_id") - .notNull() - .references(() => users.id, { onDelete: "cascade" }), - itemId: uuid("item_id").notNull(), - itemType: itemTypeEnum("item_type").default("event").notNull(), - interactionType: interactionTypeEnum("interaction_type").notNull(), - interactionValue: doublePrecision("interaction_value").notNull(), - metadata: jsonb("metadata"), - createdAt: timestamp("created_at").defaultNow().notNull(), +/** Cursors and bookkeeping for background syncs (e.g. InboxEngine event revisions). */ +export const syncState = pgTable("sync_state", { + key: varchar("key", { length: 100 }).primaryKey(), + value: text("value").notNull(), + updatedAt: timestamp("updated_at").defaultNow().notNull(), }); +// ── Recommendation / ML tables ──────────────────────────── + +export const interactions = pgTable( + "interactions", + { + id: uuid("id").defaultRandom().primaryKey(), + userId: uuid("user_id") + .notNull() + .references(() => users.id, { onDelete: "cascade" }), + itemId: uuid("item_id").notNull(), + itemType: itemTypeEnum("item_type").default("event").notNull(), + interactionType: interactionTypeEnum("interaction_type").notNull(), + interactionValue: doublePrecision("interaction_value").notNull(), + metadata: jsonb("metadata"), + createdAt: timestamp("created_at").defaultNow().notNull(), + }, + (t) => [ + // A user's interaction history (recommendations / preference vectors). + index("interactions_user_id_created_at_idx").on(t.userId, t.createdAt), + // Per-event view counts for feed popularity (append-only, grows fastest). + index("interactions_item_id_type_idx").on(t.itemId, t.interactionType), + ], +); + export const userPreferenceVectors = pgTable("user_preference_vectors", { userId: uuid("user_id") .primaryKey() diff --git a/apps/database/src/seed.ts b/apps/database/src/seed.ts index c65e564..2becf24 100644 --- a/apps/database/src/seed.ts +++ b/apps/database/src/seed.ts @@ -31,6 +31,41 @@ type TagEmbeddingRecord = { embedding: number[]; }; +/* ── safety guard ── + * The seed inserts users with plausible real Princeton NetIDs. Never let it + * touch a production or remote database by accident. The postgres client is + * lazy, so no connection has been opened yet when this runs. + */ +function assertSafeSeedTarget() { + if (process.env.ALLOW_REMOTE_SEED === "1") return; + + const refuse = (reason: string): never => { + console.error(`\nRefusing to seed: ${reason}`); + console.error( + "The seed creates demo users with realistic NetIDs and must only run against a local dev database.", + ); + console.error("If you really mean it, re-run with ALLOW_REMOTE_SEED=1.\n"); + process.exit(1); + }; + + if (process.env.NODE_ENV === "production") { + refuse("NODE_ENV is 'production'."); + } + + let host: string; + try { + host = new URL(process.env.DATABASE_URL ?? "").hostname; + } catch { + return refuse("DATABASE_URL is missing or not a valid URL."); + } + const localHosts = new Set(["localhost", "127.0.0.1", "[::1]", "::1"]); + if (!localHosts.has(host)) { + refuse(`DATABASE_URL host '${host}' is not localhost/127.0.0.1.`); + } +} + +assertSafeSeedTarget(); + /* ── helpers ── */ function days(n: number) { return n * 24 * 60 * 60 * 1000; diff --git a/apps/database/src/sync-inbox-engine.ts b/apps/database/src/sync-inbox-engine.ts new file mode 100644 index 0000000..8b10f65 --- /dev/null +++ b/apps/database/src/sync-inbox-engine.ts @@ -0,0 +1,318 @@ +/** + * Sync The Forum from InboxEngine (packages/inbox-engine), the shared source of truth for + * Princeton organizations, campus venues and events. + * + * bun run db:sync-engine # orgs + venues + events (full pass; a few seconds) + * + * Env: DATABASE_URL, INBOX_ENGINE_URL, INBOX_ENGINE_TOKEN. + * + * - Organizations: every MyPrincetonU group, upserted on `external_id` ("mpu:"). A manual + * Forum org with the same name is linked instead of duplicated. Forum-only fields (followers, + * members, events) are never touched. + * - Venues: InboxEngine's campus gazetteer, upserted into campus_locations. + * - Events: the whole revisioned feed. Official MyPrincetonU events and publishable listserv + * extractions become published Forum events owned by the InboxEngine bot user; withdrawn or + * duplicate ones are unpublished (never deleted, so RSVPs survive). Recurring series (daily + * prayer, weekly office hours) show only their next SERIES_WINDOW occurrences, so each run + * re-reads the full feed to advance that window. + */ +import { and, eq, isNull, sql } from "drizzle-orm"; +import { + type EngineEvent, + type EngineOrganization, + InboxEngineClient, +} from "../../../packages/inbox-engine/src/client/index.ts"; +import { db } from "./db"; +import { + events, + campusLocations, + eventTagEnum, + eventTags, + locationCategoryEnum, + orgCategoryEnum, + organizations, + syncState, + users, +} from "./schema"; + +const TBA_LOCATION = { id: "tba", name: "Location TBA", latitude: 0, longitude: 0 } as const; +const BOT = { netId: "_inboxengine", email: "inboxengine@tigerapps.org", displayName: "The Forum" }; +const CURSOR_KEY = "inbox-engine:events"; +/** Upcoming occurrences shown per recurring series. */ +const SERIES_WINDOW = 2; + +type OrgCategory = (typeof orgCategoryEnum.enumValues)[number]; +type LocationCategory = (typeof locationCategoryEnum.enumValues)[number]; +type EventTag = (typeof eventTagEnum.enumValues)[number]; + +const orgCategories = new Set(orgCategoryEnum.enumValues); +const locationCategories = new Set(locationCategoryEnum.enumValues); +const eventTagValues = new Set(eventTagEnum.enumValues); + +function engineClient() { + const url = process.env.INBOX_ENGINE_URL; + const token = process.env.INBOX_ENGINE_TOKEN; + if (!url || !token) throw new Error("INBOX_ENGINE_URL and INBOX_ENGINE_TOKEN are required"); + return new InboxEngineClient(url, token, 60_000); +} + +async function botUserId(): Promise { + const [existing] = await db + .select({ id: users.id }) + .from(users) + .where(eq(users.netId, BOT.netId)); + if (existing) return existing.id; + const [created] = await db + .insert(users) + .values({ ...BOT, onboarded: true }) + .onConflictDoNothing() + .returning({ id: users.id }); + if (created) return created.id; + const [again] = await db.select({ id: users.id }).from(users).where(eq(users.netId, BOT.netId)); + if (!again) throw new Error("Could not create the InboxEngine bot user"); + return again.id; +} + +function orgValues(org: EngineOrganization) { + const category = ( + org.forumCategory && orgCategories.has(org.forumCategory) ? org.forumCategory : "social event" + ) as OrgCategory; + return { + name: org.name.slice(0, 255), + description: org.description ?? org.tagline ?? null, + logoUrl: org.logoUrl, + category, + source: "myprincetonu", + externalId: org.id, + acronym: org.acronym?.slice(0, 40) ?? null, + tagline: org.tagline, + groupType: org.groupType?.slice(0, 120) ?? null, + groupUrl: org.groupUrl, + website: org.website, + contactEmail: org.contactEmail?.slice(0, 255) ?? null, + socials: org.socials ?? {}, + memberCount: org.memberCount ?? null, + syncedAt: new Date(), + updatedAt: new Date(), + }; +} + +export async function syncOrganizations(client = engineClient()) { + const { organizations: remote } = await client.organizations("", 5000); + let created = 0; + let linked = 0; + let updated = 0; + for (const org of remote) { + if (!org.id.startsWith("mpu:")) continue; // curated non-directory aliases stay engine-only + const values = orgValues(org); + const [byExternal] = await db + .select({ id: organizations.id }) + .from(organizations) + .where(eq(organizations.externalId, org.id)); + if (byExternal) { + await db.update(organizations).set(values).where(eq(organizations.id, byExternal.id)); + updated++; + continue; + } + // A Forum-created org with the same name becomes the official record (keeps followers/events). + const [byName] = await db + .select({ id: organizations.id }) + .from(organizations) + .where(and(eq(organizations.name, values.name), isNull(organizations.externalId))); + if (byName) { + await db.update(organizations).set(values).where(eq(organizations.id, byName.id)); + linked++; + continue; + } + await db.insert(organizations).values(values).onConflictDoNothing(); + created++; + } + return { remote: remote.length, created, linked, updated }; +} + +export async function syncLocations(client = engineClient()) { + const { locations } = await client.locations(); + const rows = [ + ...locations.map((l) => ({ + id: l.id.slice(0, 100), + name: l.name.slice(0, 255), + latitude: l.latitude, + longitude: l.longitude, + category: (locationCategories.has(l.category) ? l.category : "other") as LocationCategory, + })), + { ...TBA_LOCATION, category: "other" as LocationCategory }, + ]; + for (const row of rows) + await db + .insert(campusLocations) + .values(row) + .onConflictDoUpdate({ + target: campusLocations.id, + set: { + name: row.name, + latitude: row.latitude, + longitude: row.longitude, + category: row.category, + }, + }); + return { locations: rows.length }; +} + +function describe(e: EngineEvent): string { + return (e.summary || e.title).slice(0, 5000); +} + +/** Events worth showing in The Forum: complete, not over, and (for series) coming up next. */ +function shouldPublish(e: EngineEvent, seriesRank: Map): boolean { + if (e.status !== "active" || !e.publishable) return false; + const end = new Date(e.endsAt ?? e.startsAt).getTime(); + if (end <= Date.now() - 6 * 3600_000) return false; + if (e.series && e.series.size > 2) + return (seriesRank.get(e.id) ?? Number.POSITIVE_INFINITY) < SERIES_WINDOW; + return true; +} + +/** Rank each upcoming occurrence within its series by start time (0 = next). */ +function rankSeries(all: EngineEvent[]): Map { + const bySeries = new Map(); + const now = Date.now() - 6 * 3600_000; + for (const e of all) { + if (!e.series || e.status !== "active") continue; + if (new Date(e.endsAt ?? e.startsAt).getTime() <= now) continue; + const list = bySeries.get(e.series.id) ?? []; + list.push(e); + bySeries.set(e.series.id, list); + } + const rank = new Map(); + for (const list of bySeries.values()) { + list.sort((a, b) => a.startsAt.localeCompare(b.startsAt)); + list.forEach((e, i) => rank.set(e.id, i)); + } + return rank; +} + +export async function syncEvents(client = engineClient()) { + const bot = await botUserId(); + // Latest state of every event, read from the start of the change feed. + const latest = new Map(); + let cursor = 0; + for (;;) { + const { changes, next } = await client.eventChanges(cursor, 1000); + if (!changes.length) break; + for (const e of changes) latest.set(e.id, e); + cursor = next; + } + const all = [...latest.values()]; + const seriesRank = rankSeries(all); + const knownLocations = new Set( + (await db.select({ id: campusLocations.id }).from(campusLocations)).map((l) => l.id), + ); + const orgByExternal = new Map( + ( + await db + .select({ id: organizations.id, externalId: organizations.externalId }) + .from(organizations) + ) + .filter((o) => o.externalId) + .map((o) => [o.externalId as string, o.id]), + ); + const stats = { seen: 0, published: 0, unpublished: 0, skipped: 0 }; + + { + const changes = all; + for (const e of changes) { + stats.seen++; + const key = `ie:${e.id}`; + const [existing] = await db + .select({ + id: events.id, + status: events.status, + isPublic: events.isPublic, + updatedAt: events.updatedAt, + }) + .from(events) + .where(eq(events.sourceMessageId, key)); + const publish = shouldPublish(e, seriesRank); + const isPublished = existing?.status === "published" && existing.isPublic; + // Unchanged upstream and already in the desired state: nothing to write. + if ( + existing && + publish === isPublished && + existing.updatedAt.getTime() >= new Date(e.updatedAt).getTime() + ) { + stats.skipped++; + continue; + } + if (!publish) { + if (existing && isPublished) { + await db + .update(events) + .set({ status: "draft", isPublic: false, updatedAt: new Date() }) + .where(eq(events.id, existing.id)); + stats.unpublished++; + } else stats.skipped++; + continue; + } + const locationId = + e.location.id && knownLocations.has(e.location.id) ? e.location.id : TBA_LOCATION.id; + const detail = + [ + e.location.room ? `Room ${e.location.room}` : null, + locationId === "tba" ? e.location.text : null, + ] + .filter(Boolean) + .join(" · ") || null; + const values = { + title: e.title.slice(0, 200), + description: describe(e), + datetime: new Date(e.startsAt), + endDatetime: e.endsAt ? new Date(e.endsAt) : null, + locationId, + locationDetail: detail?.slice(0, 200) ?? null, + orgId: e.host ? (orgByExternal.get(e.host.id) ?? null) : null, + creatorId: bot, + flyerUrl: e.imageUrl, + externalLink: e.rsvpUrl, + sourceUrl: e.source.url, + isPublic: true, + status: "published" as const, + source: e.source.kind, + sourceMessageId: key, + updatedAt: new Date(), + }; + let eventId = existing?.id; + if (eventId) await db.update(events).set(values).where(eq(events.id, eventId)); + else { + const [row] = await db.insert(events).values(values).returning({ id: events.id }); + eventId = row?.id; + } + if (eventId) { + const tags = e.tags.filter((t): t is EventTag => eventTagValues.has(t)); + await db.delete(eventTags).where(eq(eventTags.eventId, eventId)); + if (tags.length) + await db + .insert(eventTags) + .values(tags.map((tag) => ({ eventId: eventId as string, tag }))); + } + stats.published++; + } + } + await db + .insert(syncState) + .values({ key: CURSOR_KEY, value: String(cursor) }) + .onConflictDoUpdate({ + target: syncState.key, + set: { value: String(cursor), updatedAt: sql`now()` }, + }); + return { ...stats, cursor }; +} + +if (import.meta.main) { + const client = engineClient(); + const started = Date.now(); + console.log("organizations", await syncOrganizations(client)); + console.log("locations", await syncLocations(client)); + console.log("events", await syncEvents(client)); + console.log(`done in ${((Date.now() - started) / 1000).toFixed(1)}s`); + process.exit(0); +} diff --git a/apps/database/tsconfig.json b/apps/database/tsconfig.json index c4d2cb3..9e36bcd 100644 --- a/apps/database/tsconfig.json +++ b/apps/database/tsconfig.json @@ -5,13 +5,13 @@ "module": "ESNext", "moduleResolution": "bundler", "strict": true, - "declaration": true, - "outDir": "dist", - "rootDir": ".", + "rootDir": "../..", "skipLibCheck": true, "esModuleInterop": true, "resolveJsonModule": true, - "types": ["bun"] + "types": ["bun"], + "noEmit": true, + "allowImportingTsExtensions": true }, "include": ["src/**/*", "drizzle.config.ts"], "exclude": ["node_modules", "dist"] diff --git a/apps/listserv-scraper/package.json b/apps/listserv-scraper/package.json deleted file mode 100644 index a47ab09..0000000 --- a/apps/listserv-scraper/package.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "name": "@the-forum/listserv-scraper", - "version": "0.0.1", - "private": true, - "scripts": { - "scrape": "python3 src/scrape_listserv.py", - "scrape:all-lists": "python3 src/scrape_listserv.py --all", - "scrape:full": "bash scrape_all.sh" - } -} diff --git a/apps/listserv-scraper/scrape_all.sh b/apps/listserv-scraper/scrape_all.sh deleted file mode 100755 index 63affa2..0000000 --- a/apps/listserv-scraper/scrape_all.sh +++ /dev/null @@ -1,43 +0,0 @@ -#!/bin/bash -# Continuous WHITMANWIRE scraper — runs in 5k increments until all messages are fetched. -# Each batch skips messages already fetched in previous runs. -# Safe to Ctrl+C — resume with: python3 src/scrape_listserv.py --fetch-bodies --resume - -set -e -cd "$(dirname "$0")" - -echo "=== WHITMANWIRE Continuous Scraper ===" -echo "Started at $(date)" -echo "" - -# Wait for any existing scrape process to finish -EXISTING_PID=$(pgrep -f "scrape_listserv.py" 2>/dev/null || true) -if [ -n "$EXISTING_PID" ]; then - echo "Waiting for existing scrape (PID $EXISTING_PID) to finish..." - while kill -0 "$EXISTING_PID" 2>/dev/null; do - sleep 10 - done - echo "Previous scrape finished." - echo "" -fi - -LIMITS=(5000 10000 15000 20000 25000 30000 35000 40000) - -for LIMIT in "${LIMITS[@]}"; do - echo "============================================" - echo "Batch: limit=$LIMIT — $(date)" - echo "============================================" - - python3 src/scrape_listserv.py \ - --list WHITMANWIRE \ - --limit "$LIMIT" \ - --fetch-bodies \ - --batch-size 200 - - echo "" - echo "Batch limit=$LIMIT complete." - echo "" -done - -echo "=== All batches complete! ===" -echo "Finished at $(date)" diff --git a/apps/listserv-scraper/src/build_assessment_viewer.py b/apps/listserv-scraper/src/build_assessment_viewer.py deleted file mode 100644 index 8069eb6..0000000 --- a/apps/listserv-scraper/src/build_assessment_viewer.py +++ /dev/null @@ -1,423 +0,0 @@ -""" -Build an interactive HTML viewer for assessing classifier quality. - -Combines: UMAP coordinates + binary classification + tags + features + org info -Supports: filtering by event/not-event, tags, orgs, categories, search -""" - -import json -import html -import re -import numpy as np -from pathlib import Path -from collections import Counter - -DATA_DIR = Path(__file__).resolve().parent.parent / "data" - - -def load_all(): - """Load and merge all datasets.""" - # Core data - with open(DATA_DIR / "whitmanwire_emails.json") as f: - msgs = json.load(f)["messages"] - msg_map = {m["message_id"]: m for m in msgs} - - # UMAP coordinates (40k) - umap_2d = np.load(DATA_DIR / "whitmanwire_umap2d.npy") - hdbscan_labels = np.load(DATA_DIR / "whitmanwire_hdbscan_labels.npy") - - # LLM outputs (500 each) - classify = {} - if (DATA_DIR / "llm_classify_results.json").exists(): - with open(DATA_DIR / "llm_classify_results.json") as f: - for r in json.load(f): - classify[r["message_id"]] = r - - tagged = {} - if (DATA_DIR / "llm_tagged_results.json").exists(): - with open(DATA_DIR / "llm_tagged_results.json") as f: - for r in json.load(f): - tagged[r["message_id"]] = r - - features = {} - if (DATA_DIR / "llm_features.json").exists(): - with open(DATA_DIR / "llm_features.json") as f: - for r in json.load(f): - features[r["message_id"]] = r - - taxonomy = [] - if (DATA_DIR / "llm_taxonomy.json").exists(): - with open(DATA_DIR / "llm_taxonomy.json") as f: - taxonomy = json.load(f) - - return msgs, umap_2d, hdbscan_labels, classify, tagged, features, taxonomy - - -def build_org_categories(msgs): - """Infer org categories from author names.""" - author_counts = Counter(m["author_name"] for m in msgs) - - # Manual category mapping for top orgs - tech_keywords = ["tech", "code", "hack", "data", "computer", "software", "ai", "hoagie", "tigerapps", "quant", "blockchain", "crypto"] - dance_keywords = ["dance", "ballet", "bhangra", "naacho", "raqs", "hip hop", "choreograph"] - music_keywords = ["music", "choir", "glee", "a cappella", "acapella", "orchestra", "band", "pianists", "jazz"] - theater_keywords = ["theatre", "theater", "intime", "triangle", "improv", "comedy", "acting", "drama"] - food_keywords = ["food", "coffee", "baking", "challah", "culinary", "dining", "cheese"] - sports_keywords = ["sport", "athletic", "rugby", "fencing", "climbing", "soccer", "basketball", "tennis", "swim", "run", "track", "ultimate"] - political_keywords = ["democrat", "republican", "political", "whig", "clio", "government", "policy", "progressive"] - cultural_keywords = ["asian", "black", "latino", "latina", "hispanic", "chinese", "korean", "japanese", "indian", "african", "caribbean", "jewish", "muslim", "christian", "hindu", "sikh", "persian", "arab", "vietnamese", "filipino", "pakistani"] - service_keywords = ["service", "volunteer", "community", "civic", "habitat", "tutor"] - media_keywords = ["daily", "prince", "nassau", "magazine", "journal", "publication", "radio", "film", "photo"] - - def categorize_author(name): - n = name.lower() - cats = [] - if any(k in n for k in tech_keywords): cats.append("tech") - if any(k in n for k in dance_keywords): cats.append("dance") - if any(k in n for k in music_keywords): cats.append("music") - if any(k in n for k in theater_keywords): cats.append("theater") - if any(k in n for k in food_keywords): cats.append("food") - if any(k in n for k in sports_keywords): cats.append("sports") - if any(k in n for k in political_keywords): cats.append("political") - if any(k in n for k in cultural_keywords): cats.append("cultural") - if any(k in n for k in service_keywords): cats.append("service") - if any(k in n for k in media_keywords): cats.append("media") - return cats if cats else ["other"] - - org_cats = {} - for name in author_counts: - org_cats[name] = categorize_author(name) - - return org_cats - - -def main(): - print("Loading all data...", flush=True) - msgs, umap_2d, hdbscan_labels, classify, tagged, features, taxonomy = load_all() - - print("Building org categories...", flush=True) - org_cats = build_org_categories(msgs) - - # Get all unique tags and orgs for filters - all_tags = set() - for r in tagged.values(): - for t in r.get("tags", []): - tag = t["tag"] if isinstance(t, dict) else t - all_tags.add(tag) - all_tags = sorted(all_tags) - - all_orgs = Counter(m["author_name"] for m in msgs) - top_orgs = [name for name, _ in all_orgs.most_common(100)] - - all_org_cat_set = set() - for cats in org_cats.values(): - all_org_cat_set.update(cats) - all_org_cat_set = sorted(all_org_cat_set) - - print(f"Building HTML ({len(msgs)} points)...", flush=True) - - # Build point data - points = [] - for i, m in enumerate(msgs): - mid = m["message_id"] - cl = classify.get(mid, {}) - tg = tagged.get(mid, {}) - ft = features.get(mid, {}) - - tag_list = [] - for t in tg.get("tags", []): - tag_list.append(t["tag"] if isinstance(t, dict) else t) - - body_preview = html.escape(m.get("body_text", "")[:250].replace("\n", " ")) - subj = html.escape(m.get("subject", "")[:120]) - author = html.escape(m.get("author_name", "")) - sender = html.escape(m.get("hoagiemail_sender_name", "") or "") - url = m.get("listserv_url", "") - org_category = org_cats.get(m.get("author_name", ""), ["other"]) - - points.append({ - "x": round(float(umap_2d[i, 0]), 3), - "y": round(float(umap_2d[i, 1]), 3), - "s": subj, - "a": author, - "d": m.get("date", "")[:10], - "b": body_preview, - "u": url, - "h": int(hdbscan_labels[i]), - "ev": cl.get("is_event"), - "ec": cl.get("confidence", 0), - "er": html.escape(cl.get("reason", "")[:100]), - "t": tag_list, - "cat": ft.get("category", ""), - "urg": ft.get("urgency", ""), - "food": ft.get("has_food", False), - "oc": org_category, - "hm": m.get("is_hoagiemail", False), - "sn": sender, - }) - - points_json = json.dumps(points) - tags_json = json.dumps(all_tags) - orgs_json = json.dumps(top_orgs[:80]) - org_cats_json = json.dumps(all_org_cat_set) - taxonomy_json = json.dumps(taxonomy) - - html_content = f""" - - - -WHITMANWIRE — Classifier Assessment Viewer - - - -
-
-
- - -
- - - - -
- -
- - -
-
- -
-
-
-
- -
- - -""" - - out_path = DATA_DIR / "assessment_viewer.html" - with open(out_path, "w") as f: - f.write(html_content) - print(f"Saved to {out_path} ({out_path.stat().st_size / 1024 / 1024:.1f} MB)") - - -if __name__ == "__main__": - main() diff --git a/apps/listserv-scraper/src/export_data.py b/apps/listserv-scraper/src/export_data.py deleted file mode 100644 index 6324c9b..0000000 --- a/apps/listserv-scraper/src/export_data.py +++ /dev/null @@ -1,145 +0,0 @@ -""" -Export scraped listserv JSON data into ML-ready formats. - -Outputs: - data/whitmanwire.csv — flat CSV for quick exploration - data/whitmanwire.parquet — compressed columnar format for pandas/ML - data/whitmanwire_sample.csv — 200-row sample for labeling - -Usage: - python3 src/export_data.py - python3 src/export_data.py --input data/whitmanwire_emails.json -""" - -import argparse -import json -from pathlib import Path - -import pandas as pd - -DATA_DIR = Path(__file__).resolve().parent.parent / "data" - - -def load_and_flatten(json_path: Path) -> pd.DataFrame: - """Load scraped JSON and flatten into a DataFrame.""" - with open(json_path, encoding="utf-8") as f: - data = json.load(f) - - msgs = data["messages"] - print(f"Loaded {len(msgs):,} messages from {json_path.name}") - - df = pd.DataFrame(msgs) - - # Convert links/images arrays to counts + semicolon-joined strings - df["link_count"] = df["links"].apply(lambda x: len(x) if isinstance(x, list) else 0) - df["image_count"] = df["images"].apply(lambda x: len(x) if isinstance(x, list) else 0) - df["links_joined"] = df["links"].apply( - lambda x: ";".join(x) if isinstance(x, list) else "" - ) - df["images_joined"] = df["images"].apply( - lambda x: ";".join(x) if isinstance(x, list) else "" - ) - - # Body length features - df["body_text_len"] = df["body_text"].fillna("").str.len() - df["subject_len"] = df["subject"].fillna("").str.len() - - # Parse date as datetime - df["date"] = pd.to_datetime(df["date"], utc=True, errors="coerce") - - # Clean up columns for export - export_cols = [ - "message_id", - "subject", - "author_name", - "author_email", - "date", - "body_text", - "body_html", - "is_hoagiemail", - "hoagiemail_sender_name", - "hoagiemail_sender_email", - "link_count", - "image_count", - "links_joined", - "images_joined", - "body_text_len", - "subject_len", - "listserv_url", - ] - - # Only include columns that exist - export_cols = [c for c in export_cols if c in df.columns] - df = df[export_cols] - - return df - - -def print_summary(df: pd.DataFrame): - """Print dataset summary stats.""" - print(f"\n{'='*60}") - print(f"Dataset Summary") - print(f"{'='*60}") - print(f"Rows: {len(df):,}") - print(f"Columns: {len(df.columns)}") - print(f"Date range: {df['date'].min()} → {df['date'].max()}") - print(f"HoagieMail: {df['is_hoagiemail'].sum():,} ({df['is_hoagiemail'].mean()*100:.0f}%)") - print(f"Has sender name: {df['hoagiemail_sender_name'].notna().sum():,}") - print(f"Avg body length: {df['body_text_len'].mean():.0f} chars") - print(f"Avg links/email: {df['link_count'].mean():.1f}") - print(f"Unique authors: {df['author_name'].nunique():,}") - print() - print("Column types:") - for col in df.columns: - non_null = df[col].notna().sum() - print(f" {col:<30} {str(df[col].dtype):<15} {non_null:>5} non-null") - print() - - -def main(): - parser = argparse.ArgumentParser(description="Export listserv data for ML") - parser.add_argument( - "--input", - default=str(DATA_DIR / "whitmanwire_emails.json"), - help="Input JSON file", - ) - args = parser.parse_args() - - json_path = Path(args.input) - if not json_path.exists(): - print(f"Error: {json_path} not found") - return - - df = load_and_flatten(json_path) - print_summary(df) - - # Export full CSV - csv_path = DATA_DIR / "whitmanwire.csv" - df.to_csv(csv_path, index=False) - print(f"Saved CSV: {csv_path} ({csv_path.stat().st_size / 1024 / 1024:.1f} MB)") - - # Export Parquet (much smaller, preserves types) - parquet_path = DATA_DIR / "whitmanwire.parquet" - df.to_parquet(parquet_path, index=False) - print(f"Saved Parquet: {parquet_path} ({parquet_path.stat().st_size / 1024 / 1024:.1f} MB)") - - # Export 200-row sample for labeling - sample = df.sample(n=min(200, len(df)), random_state=42).sort_values("date") - sample_cols = [ - "message_id", "subject", "author_name", "date", - "body_text", "is_hoagiemail", "hoagiemail_sender_name", - "link_count", "image_count", - ] - sample_cols = [c for c in sample_cols if c in sample.columns] - sample_path = DATA_DIR / "whitmanwire_sample.csv" - sample[sample_cols].to_csv(sample_path, index=False) - print(f"Saved sample: {sample_path} ({len(sample)} rows for labeling)") - - print("\nDone. Load in Python with:") - print(' import pandas as pd') - print(f' df = pd.read_parquet("{parquet_path}")') - print(f' df = pd.read_csv("{csv_path}")') - - -if __name__ == "__main__": - main() diff --git a/apps/listserv-scraper/src/import_to_postgres.py b/apps/listserv-scraper/src/import_to_postgres.py deleted file mode 100644 index 5667574..0000000 --- a/apps/listserv-scraper/src/import_to_postgres.py +++ /dev/null @@ -1,154 +0,0 @@ -""" -Import scraped listserv JSON data into PostgreSQL listserv_emails table. - -Usage: - python3 src/import_to_postgres.py - python3 src/import_to_postgres.py --input data/whitmanwire_emails.json - -Requires: DATABASE_URL environment variable (or reads from root .env) -""" - -import argparse -import json -import os -import re -from datetime import datetime, timezone -from pathlib import Path - -try: - import psycopg2 - from psycopg2.extras import Json, execute_values -except ImportError: - print("Install psycopg2: pip3 install psycopg2-binary") - exit(1) - -DATA_DIR = Path(__file__).resolve().parent.parent / "data" -ROOT_DIR = Path(__file__).resolve().parent.parent.parent.parent - - -def get_database_url() -> str: - """Get DATABASE_URL from env or root .env file.""" - url = os.environ.get("DATABASE_URL") - if url: - return url - - env_file = ROOT_DIR / ".env" - if env_file.exists(): - with open(env_file) as f: - for line in f: - line = line.strip() - if line.startswith("DATABASE_URL="): - return line.split("=", 1)[1].strip().strip('"').strip("'") - - raise RuntimeError("DATABASE_URL not found in environment or .env file") - - -def parse_date(date_str: str) -> datetime | None: - """Parse ISO date string to datetime.""" - if not date_str: - return None - try: - return datetime.fromisoformat(date_str) - except (ValueError, TypeError): - return None - - -def main(): - parser = argparse.ArgumentParser(description="Import listserv data to Postgres") - parser.add_argument( - "--input", - default=str(DATA_DIR / "whitmanwire_emails.json"), - help="Input JSON file", - ) - parser.add_argument( - "--listserv", - default="WHITMANWIRE", - help="Listserv name to tag rows with", - ) - args = parser.parse_args() - - json_path = Path(args.input) - if not json_path.exists(): - print(f"Error: {json_path} not found") - return - - # Load data - print(f"Loading {json_path.name}...", end=" ", flush=True) - with open(json_path, encoding="utf-8") as f: - data = json.load(f) - msgs = data["messages"] - print(f"{len(msgs):,} messages") - - # Connect to Postgres - db_url = get_database_url() - print(f"Connecting to Postgres...", end=" ", flush=True) - conn = psycopg2.connect(db_url) - cur = conn.cursor() - print("OK") - - # Prepare rows - rows = [] - for m in msgs: - rows.append(( - m.get("message_id", ""), - args.listserv, - m.get("subject", ""), - m.get("author_name"), - m.get("author_email"), - parse_date(m.get("date", "")), - m.get("body_text"), - m.get("body_html"), - m.get("is_hoagiemail", False), - m.get("hoagiemail_sender_name"), - m.get("hoagiemail_sender_email"), - Json(m.get("links", [])), - Json(m.get("images", [])), - Json(m.get("attachments", [])), - m.get("listserv_url"), - )) - - # Bulk insert with ON CONFLICT skip - print(f"Inserting {len(rows):,} rows...", end=" ", flush=True) - execute_values( - cur, - """ - INSERT INTO listserv_emails ( - message_id, listserv, subject, author_name, author_email, - date, body_text, body_html, is_hoagiemail, - hoagiemail_sender_name, hoagiemail_sender_email, - links, images, attachments, listserv_url - ) VALUES %s - ON CONFLICT (message_id) DO NOTHING - """, - rows, - page_size=500, - ) - conn.commit() - - # Verify - cur.execute("SELECT COUNT(*) FROM listserv_emails WHERE listserv = %s", (args.listserv,)) - count = cur.fetchone()[0] - print(f"done") - print(f"\nTotal rows in listserv_emails for {args.listserv}: {count:,}") - - cur.execute(""" - SELECT - COUNT(*) as total, - COUNT(*) FILTER (WHERE is_hoagiemail) as hoagiemail, - COUNT(*) FILTER (WHERE hoagiemail_sender_name IS NOT NULL) as has_sender, - MIN(date) as oldest, - MAX(date) as newest - FROM listserv_emails WHERE listserv = %s - """, (args.listserv,)) - row = cur.fetchone() - print(f"HoagieMail: {row[1]:,} ({row[1]/row[0]*100:.0f}%)") - print(f"Has sender: {row[2]:,}") - print(f"Date range: {row[3]} → {row[4]}") - - cur.close() - conn.close() - print("\nDone.") - - -if __name__ == "__main__": - main() diff --git a/apps/listserv-scraper/src/label_all.py b/apps/listserv-scraper/src/label_all.py deleted file mode 100644 index 3326b95..0000000 --- a/apps/listserv-scraper/src/label_all.py +++ /dev/null @@ -1,219 +0,0 @@ -""" -Label all emails since June 2022 using Gemini 3.1 Flash Lite. -Runs concurrent requests for speed. Saves checkpoints. - -Usage: - python3 label_all.py --test # test with 10 emails - python3 label_all.py # full run - python3 label_all.py --resume # resume from checkpoint - python3 label_all.py --workers 5 # concurrency level -""" - -import argparse -import json -import re -import time -import os -from pathlib import Path -from concurrent.futures import ThreadPoolExecutor, as_completed -from threading import Lock - -import httpx - -OPENROUTER_KEY = os.environ.get("OPENROUTER_API_KEY", "") -MODEL = "google/gemini-3.1-flash-lite-preview" -API_URL = "https://openrouter.ai/api/v1/chat/completions" -DATA_DIR = Path(__file__).resolve().parent.parent / "data" - -SYSTEM_PROMPT = """You classify Princeton University listserv emails as EVENT or NOT_EVENT. - -An EVENT is a specific physical or virtual GATHERING at a time and place that students can ATTEND: -- Talks, lectures, panels, speaker events -- Performances, concerts, shows, recitals -- Workshops, classes, tutorials -- Club meetings, info sessions, interest meetings (with a specific time/place) -- Parties, socials, mixers -- Food events, bake sales, free food giveaways -- Sports games, tournaments -- Hackathons, competitions -- Auditions, tryouts, open calls -- Study breaks, craft sessions -- Fundraiser events (bake sales, auctions) -- Movie screenings -- Career fairs, networking events (in-person) - -NOT_EVENT: -- Service availability hours (hotlines, chat services saying "open tonight") — this is NOT a gathering -- Club recruitment/applications with NO specific gathering time -- Lost & found, selling items, rideshare requests -- Job/internship postings, application deadlines -- Surveys, petitions, announcements -- Newsletters, digests, weekly roundups -- Merch drops without a physical sale event -- General information (new club formed, new website launched) - -KEY DISTINCTION: "We're open tonight from 8-12" for a chat/phone service is NOT an event. -"Come to Frist tonight at 8pm" for a gathering IS an event. - -Respond with ONLY valid JSON: -{"is_event": true/false, "confidence": 0.0-1.0, "reason": "brief reason"}""" - -# Thread-safe results -results_lock = Lock() -results = [] -save_counter = 0 - - -def classify_email(client, m): - """Classify a single email. Returns result dict.""" - body = m.get("body_text", "")[:400] - body = re.sub(r"This email was instantly sent.*", "", body, flags=re.DOTALL | re.I) - body = re.sub(r"Email composed by.*", "", body, flags=re.DOTALL | re.I) - - email_text = f"Subject: {m.get('subject', '')}\nFrom: {m.get('author_name', '')}\nDate: {m.get('date', '')[:10]}\nBody: {body.strip()}" - - for attempt in range(3): - try: - resp = client.post( - API_URL, - headers={ - "Authorization": f"Bearer {OPENROUTER_KEY}", - "Content-Type": "application/json", - }, - json={ - "model": MODEL, - "messages": [ - {"role": "system", "content": SYSTEM_PROMPT}, - {"role": "user", "content": email_text}, - ], - "temperature": 0.1, - "max_tokens": 200, - }, - timeout=20, - ) - resp.raise_for_status() - content = resp.json()["choices"][0]["message"]["content"] - - json_match = re.search(r"\{.*\}", content, re.DOTALL) - if json_match: - parsed = json.loads(json_match.group()) - return { - "message_id": m["message_id"], - "subject": m["subject"][:120], - "date": m.get("date", "")[:10], - "author_name": m.get("author_name", ""), - "is_event": parsed.get("is_event"), - "confidence": parsed.get("confidence", 0), - "reason": parsed.get("reason", ""), - } - except Exception as e: - if attempt < 2: - time.sleep(2 ** attempt) - else: - return { - "message_id": m["message_id"], - "subject": m["subject"][:120], - "date": m.get("date", "")[:10], - "is_event": None, - "confidence": 0, - "reason": f"error: {str(e)[:80]}", - } - - -def save_checkpoint(results, output_path): - with open(output_path, "w") as f: - json.dump(results, f, ensure_ascii=False) - - -def main(): - parser = argparse.ArgumentParser() - parser.add_argument("--test", action="store_true", help="Test with 10 emails") - parser.add_argument("--resume", action="store_true", help="Resume from checkpoint") - parser.add_argument("--workers", type=int, default=5, help="Concurrent workers") - parser.add_argument("--since", default="2022-06", help="Label emails since this date") - args = parser.parse_args() - - # Load emails - print("Loading emails...", flush=True) - with open(DATA_DIR / "whitmanwire_emails.json") as f: - all_msgs = json.load(f)["messages"] - - to_label = [m for m in all_msgs if m.get("date", "") >= args.since] - print(f"Emails since {args.since}: {len(to_label):,}", flush=True) - - if args.test: - to_label = to_label[:10] - print(f"TEST MODE: labeling {len(to_label)} emails", flush=True) - - output_path = DATA_DIR / "llm_labels_full.json" - - # Resume - done_ids = set() - existing = [] - if args.resume and output_path.exists(): - with open(output_path) as f: - existing = json.load(f) - done_ids = {r["message_id"] for r in existing} - print(f"Resuming: {len(done_ids)} already done", flush=True) - - remaining = [m for m in to_label if m["message_id"] not in done_ids] - print(f"Remaining: {len(remaining):,} | Workers: {args.workers}", flush=True) - - if not remaining: - print("Nothing to do!") - return - - results = list(existing) - checkpoint_interval = 200 - client = httpx.Client(timeout=20) - - t0 = time.time() - completed = 0 - events = sum(1 for r in results if r.get("is_event") is True) - not_events = sum(1 for r in results if r.get("is_event") is False) - errors = sum(1 for r in results if r.get("is_event") is None) - - with ThreadPoolExecutor(max_workers=args.workers) as pool: - futures = {pool.submit(classify_email, client, m): m for m in remaining} - - for future in as_completed(futures): - result = future.result() - if result: - with results_lock: - results.append(result) - completed += 1 - - if result["is_event"] is True: - events += 1 - elif result["is_event"] is False: - not_events += 1 - else: - errors += 1 - - if completed % 50 == 0: - elapsed = time.time() - t0 - rate = completed / elapsed - eta = (len(remaining) - completed) / rate / 60 if rate > 0 else 0 - print( - f" [{completed}/{len(remaining)}] " - f"events={events} not={not_events} err={errors} " - f"({rate:.1f}/s, ~{eta:.0f}m left)", - flush=True, - ) - - if completed % checkpoint_interval == 0: - save_checkpoint(results, output_path) - print(f" ** Checkpoint at {completed} **", flush=True) - - # Final save - save_checkpoint(results, output_path) - - elapsed = time.time() - t0 - print(f"\nDone in {elapsed/60:.1f} minutes", flush=True) - print(f"Total: {len(results):,} labeled", flush=True) - print(f"Events: {events:,} | Not-events: {not_events:,} | Errors: {errors}", flush=True) - print(f"Saved to {output_path}", flush=True) - - -if __name__ == "__main__": - main() diff --git a/apps/listserv-scraper/src/llm_explore.py b/apps/listserv-scraper/src/llm_explore.py deleted file mode 100644 index 7f7b785..0000000 --- a/apps/listserv-scraper/src/llm_explore.py +++ /dev/null @@ -1,524 +0,0 @@ -""" -Explore Gemini 3.1 Flash Lite via OpenRouter for: -1. Binary event classification (labeled dataset) -2. Multi-tag classification (unsupervised tag discovery → taxonomy) -3. LLM-enriched features for better clustering - -Usage: - python3 src/llm_explore.py --phase test # test API connection - python3 src/llm_explore.py --phase classify # binary event classifier on sample - python3 src/llm_explore.py --phase tags # multi-tag classifier - python3 src/llm_explore.py --phase features # LLM feature extraction for clustering - python3 src/llm_explore.py --phase all # run everything -""" - -import argparse -import json -import os -import re -import time -from pathlib import Path -from datetime import datetime - -import httpx - -OPENROUTER_KEY = os.environ.get("OPENROUTER_API_KEY", "") -MODEL = "google/gemini-3.1-flash-lite-preview" # $0.25/M in, $1.50/M out -API_URL = "https://openrouter.ai/api/v1/chat/completions" -DATA_DIR = Path(__file__).resolve().parent.parent / "data" - -# Rate limiting -REQUEST_DELAY = 0.15 # seconds between requests -client = httpx.Client(timeout=30) - - -def call_llm(system_prompt: str, user_prompt: str, temperature: float = 0.1, max_tokens: int = 500) -> str | None: - """Call Gemini Flash Lite via OpenRouter.""" - try: - resp = client.post( - API_URL, - headers={ - "Authorization": f"Bearer {OPENROUTER_KEY}", - "Content-Type": "application/json", - }, - json={ - "model": MODEL, - "messages": [ - {"role": "system", "content": system_prompt}, - {"role": "user", "content": user_prompt}, - ], - "temperature": temperature, - "max_tokens": max_tokens, - }, - ) - resp.raise_for_status() - data = resp.json() - return data["choices"][0]["message"]["content"] - except Exception as e: - print(f" [ERR] {e}") - return None - - -def load_emails(limit: int | None = None) -> list[dict]: - """Load emails from JSON.""" - with open(DATA_DIR / "whitmanwire_emails.json") as f: - msgs = json.load(f)["messages"] - if limit: - msgs = msgs[:limit] - return msgs - - -def format_email_for_llm(m: dict, max_body: int = 400) -> str: - """Format an email for LLM input.""" - body = m.get("body_text", "")[:max_body] - # Strip hoagiemail footer - body = re.sub(r"This email was instantly sent.*", "", body, flags=re.DOTALL | re.I) - body = re.sub(r"Email composed by.*", "", body, flags=re.DOTALL | re.I) - - parts = [f"Subject: {m.get('subject', '')}"] - if m.get("author_name"): - parts.append(f"From: {m['author_name']}") - if m.get("date"): - parts.append(f"Date: {m['date'][:10]}") - parts.append(f"Body: {body.strip()}") - if m.get("links"): - parts.append(f"Links: {len(m['links'])} link(s)") - if m.get("images"): - parts.append(f"Images: {len(m['images'])} image(s)") - return "\n".join(parts) - - -# ── Phase: Test ────────────────────────────────────────── - -def phase_test(): - """Test API connectivity and model response.""" - print("=== Phase: Test API ===") - resp = call_llm( - "You are a helpful assistant.", - "Say 'API working' and nothing else.", - ) - if resp and "working" in resp.lower(): - print(f" API response: {resp.strip()}") - print(" ✓ Connection OK") - else: - print(f" ✗ Unexpected response: {resp}") - return False - - # Test with a real email classification - test_email = "Subject: FREE PIZZA IN FRIST TONIGHT\nFrom: PSEC\nBody: Come grab a slice at Frist MPR tonight from 8-10pm. All are welcome!" - resp = call_llm( - "Classify this email as 'event' or 'not_event'. Respond with just the label.", - test_email, - ) - print(f" Test classification: {resp.strip()}") - - # Check model info - resp2 = call_llm( - "You are a helpful assistant.", - "What model are you? One sentence.", - ) - print(f" Model: {resp2.strip() if resp2 else 'unknown'}") - return True - - -# ── Phase: Binary Classification ───────────────────────── - -CLASSIFY_SYSTEM = """You are classifying Princeton University listserv emails as either an EVENT or NOT_EVENT. - -An EVENT is something with a specific time/place where people gather: talks, performances, workshops, meetings, parties, food events, sports games, etc. - -NOT_EVENT includes: lost & found, selling items, rideshare requests, job/internship postings, club recruitment without a specific event, surveys, newsletters/digests, general announcements. - -Respond with ONLY valid JSON: -{"is_event": true/false, "confidence": 0.0-1.0, "reason": "brief reason"}""" - - -def phase_classify(sample_size: int = 500): - """Run binary event classification on a sample.""" - print(f"=== Phase: Binary Classification ({sample_size} emails) ===") - - msgs = load_emails() - - # Use a spread sample across the dataset - import random - random.seed(42) - indices = sorted(random.sample(range(len(msgs)), min(sample_size, len(msgs)))) - sample = [(i, msgs[i]) for i in indices] - - results = [] - output_path = DATA_DIR / "llm_classify_results.json" - - # Resume from existing results - if output_path.exists(): - with open(output_path) as f: - results = json.load(f) - done_ids = {r["message_id"] for r in results} - sample = [(i, m) for i, m in sample if m["message_id"] not in done_ids] - print(f" Resuming: {len(results)} done, {len(sample)} remaining") - - for count, (idx, m) in enumerate(sample): - email_text = format_email_for_llm(m) - resp = call_llm(CLASSIFY_SYSTEM, email_text) - - if resp: - # Parse JSON response - try: - # Extract JSON from response - json_match = re.search(r"\{.*\}", resp, re.DOTALL) - if json_match: - parsed = json.loads(json_match.group()) - else: - parsed = {"is_event": None, "confidence": 0, "reason": "parse_error"} - except json.JSONDecodeError: - parsed = {"is_event": None, "confidence": 0, "reason": "json_error"} - - results.append({ - "message_id": m["message_id"], - "index": idx, - "subject": m["subject"], - "date": m.get("date", "")[:10], - "author_name": m.get("author_name", ""), - "is_event": parsed.get("is_event"), - "confidence": parsed.get("confidence", 0), - "reason": parsed.get("reason", ""), - "raw_response": resp.strip(), - }) - else: - results.append({ - "message_id": m["message_id"], - "index": idx, - "subject": m["subject"], - "is_event": None, - "confidence": 0, - "reason": "api_error", - }) - - if (count + 1) % 20 == 0: - events = sum(1 for r in results if r["is_event"] is True) - not_events = sum(1 for r in results if r["is_event"] is False) - print(f" [{count+1}/{len(sample)}] events={events} not_events={not_events}") - # Save checkpoint - with open(output_path, "w") as f: - json.dump(results, f, indent=2, ensure_ascii=False) - - time.sleep(REQUEST_DELAY) - - # Final save - with open(output_path, "w") as f: - json.dump(results, f, indent=2, ensure_ascii=False) - - # Stats - events = sum(1 for r in results if r["is_event"] is True) - not_events = sum(1 for r in results if r["is_event"] is False) - errors = sum(1 for r in results if r["is_event"] is None) - high_conf = sum(1 for r in results if r.get("confidence", 0) >= 0.8) - - print(f"\n Results: {events} events, {not_events} not-events, {errors} errors") - print(f" High confidence (≥0.8): {high_conf}") - print(f" Event rate: {events/(events+not_events)*100:.1f}%") - print(f" Saved to {output_path}") - - # Show some examples - print("\n Sample EVENTS:") - for r in [r for r in results if r["is_event"] is True][:5]: - print(f" [{r['confidence']:.1f}] {r['subject'][:60]}") - print("\n Sample NOT-EVENTS:") - for r in [r for r in results if r["is_event"] is False][:5]: - print(f" [{r['confidence']:.1f}] {r['subject'][:60]}") - - -# ── Phase: Multi-Tag Classification ────────────────────── - -DISCOVER_TAGS_SYSTEM = """You are analyzing Princeton University event emails to discover natural categories/tags. - -For this email, suggest 1-3 short tags (2-3 words max each) that describe what kind of event or content this is. Tags should be general enough to apply to other similar emails. - -Respond with ONLY valid JSON: -{"tags": ["tag1", "tag2"], "primary_tag": "tag1"}""" - -APPLY_TAGS_SYSTEM = """You are tagging Princeton University event emails with categories from this taxonomy: - -{taxonomy} - -Assign 1-3 tags from the list above. Also assign a confidence (0-1) for each. - -Respond with ONLY valid JSON: -{{"tags": [{{"tag": "tag-name", "confidence": 0.9}}]}}""" - - -def phase_tags(discover_size: int = 200, apply_size: int = 500): - """Phase 1: Discover tags from sample. Phase 2: Apply consolidated taxonomy.""" - print(f"=== Phase: Multi-Tag Discovery ({discover_size} emails) ===") - - msgs = load_emails() - - # --- Step 1: Discover tags from a sample --- - discovery_path = DATA_DIR / "llm_tag_discovery.json" - - import random - random.seed(42) - - if discovery_path.exists(): - with open(discovery_path) as f: - discovery_results = json.load(f) - print(f" Loaded existing discovery: {len(discovery_results)} emails") - else: - # Use emails likely to be events (from classification results if available) - classify_path = DATA_DIR / "llm_classify_results.json" - if classify_path.exists(): - with open(classify_path) as f: - classified = json.load(f) - event_ids = {r["message_id"] for r in classified if r["is_event"] is True} - event_msgs = [m for m in msgs if m["message_id"] in event_ids] - other_msgs = [m for m in msgs if m["message_id"] not in event_ids] - # Mix: 70% events, 30% non-events for discovery - sample_events = random.sample(event_msgs, min(int(discover_size * 0.7), len(event_msgs))) - sample_other = random.sample(other_msgs, min(int(discover_size * 0.3), len(other_msgs))) - discover_sample = sample_events + sample_other - random.shuffle(discover_sample) - else: - discover_sample = random.sample(msgs, min(discover_size, len(msgs))) - - discovery_results = [] - for count, m in enumerate(discover_sample): - email_text = format_email_for_llm(m) - resp = call_llm(DISCOVER_TAGS_SYSTEM, email_text) - - if resp: - try: - json_match = re.search(r"\{.*\}", resp, re.DOTALL) - parsed = json.loads(json_match.group()) if json_match else {} - except json.JSONDecodeError: - parsed = {} - - discovery_results.append({ - "message_id": m["message_id"], - "subject": m["subject"][:80], - "tags": parsed.get("tags", []), - "primary_tag": parsed.get("primary_tag", ""), - }) - - if (count + 1) % 20 == 0: - print(f" Discovery [{count+1}/{len(discover_sample)}]") - - time.sleep(REQUEST_DELAY) - - with open(discovery_path, "w") as f: - json.dump(discovery_results, f, indent=2, ensure_ascii=False) - print(f" Saved discovery to {discovery_path}") - - # --- Step 2: Analyze discovered tags → taxonomy --- - from collections import Counter - all_tags = [] - for r in discovery_results: - all_tags.extend([t.lower().strip() for t in r.get("tags", [])]) - - tag_counts = Counter(all_tags) - print(f"\n Discovered {len(tag_counts)} unique tags from {len(discovery_results)} emails") - print(f" Top 30 tags:") - for tag, count in tag_counts.most_common(30): - print(f" {count:>4} {tag}") - - # --- Step 3: Ask LLM to consolidate tags into a taxonomy --- - print("\n Asking LLM to consolidate into taxonomy...") - top_tags = [f"{tag} ({count})" for tag, count in tag_counts.most_common(60)] - consolidate_resp = call_llm( - "You are designing an event tagging system for a college campus events platform.", - f"""Here are the most common tags discovered from analyzing {len(discovery_results)} Princeton listserv emails: - -{chr(10).join(top_tags)} - -Consolidate these into a clean taxonomy of 12-20 tags. Each tag should be: -- Kebab-case (e.g., "free-food") -- Distinct (no overlapping categories) -- Useful for students browsing events - -Respond with ONLY valid JSON: -{{"taxonomy": [{{"tag": "tag-name", "description": "what it covers", "maps_from": ["original tags it absorbs"]}}]}}""", - max_tokens=1500, - ) - - taxonomy = [] - if consolidate_resp: - try: - json_match = re.search(r"\{.*\}", consolidate_resp, re.DOTALL) - parsed = json.loads(json_match.group()) if json_match else {} - taxonomy = parsed.get("taxonomy", []) - except json.JSONDecodeError: - pass - - taxonomy_path = DATA_DIR / "llm_taxonomy.json" - with open(taxonomy_path, "w") as f: - json.dump(taxonomy, f, indent=2, ensure_ascii=False) - - print(f"\n Consolidated taxonomy ({len(taxonomy)} tags):") - for t in taxonomy: - print(f" {t['tag']:<25} {t.get('description', '')[:50]}") - - # --- Step 4: Apply taxonomy to a larger sample --- - print(f"\n=== Phase: Apply Tags ({apply_size} emails) ===") - taxonomy_str = "\n".join(f"- {t['tag']}: {t.get('description', '')}" for t in taxonomy) - apply_prompt = APPLY_TAGS_SYSTEM.format(taxonomy=taxonomy_str) - - apply_path = DATA_DIR / "llm_tagged_results.json" - tagged_results = [] - - if apply_path.exists(): - with open(apply_path) as f: - tagged_results = json.load(f) - done_ids = {r["message_id"] for r in tagged_results} - print(f" Resuming: {len(tagged_results)} done") - else: - done_ids = set() - - apply_sample = random.sample(msgs, min(apply_size, len(msgs))) - apply_sample = [m for m in apply_sample if m["message_id"] not in done_ids] - - for count, m in enumerate(apply_sample): - email_text = format_email_for_llm(m) - resp = call_llm(apply_prompt, email_text) - - if resp: - try: - json_match = re.search(r"\{.*\}", resp, re.DOTALL) - parsed = json.loads(json_match.group()) if json_match else {} - except json.JSONDecodeError: - parsed = {} - - tagged_results.append({ - "message_id": m["message_id"], - "subject": m["subject"][:80], - "date": m.get("date", "")[:10], - "author_name": m.get("author_name", ""), - "tags": parsed.get("tags", []), - }) - - if (count + 1) % 20 == 0: - print(f" Tagging [{count+1}/{len(apply_sample)}]") - with open(apply_path, "w") as f: - json.dump(tagged_results, f, indent=2, ensure_ascii=False) - - time.sleep(REQUEST_DELAY) - - with open(apply_path, "w") as f: - json.dump(tagged_results, f, indent=2, ensure_ascii=False) - - # Stats - all_applied = [] - for r in tagged_results: - for t in r.get("tags", []): - tag_name = t["tag"] if isinstance(t, dict) else t - all_applied.append(tag_name) - - applied_counts = Counter(all_applied) - print(f"\n Tag distribution across {len(tagged_results)} emails:") - for tag, count in applied_counts.most_common(): - bar = "█" * (count // 5) - print(f" {tag:<25} {count:>4} {bar}") - print(f" Saved to {apply_path}") - - -# ── Phase: LLM Feature Extraction for Clustering ──────── - -FEATURES_SYSTEM = """Analyze this Princeton listserv email and extract structured features for clustering. - -Respond with ONLY valid JSON: -{"category": "one of: event-announcement, recruitment, lost-and-found, selling, rideshare, digest, social, academic, performance, career, activism, religious, wellness, sports, other", - "urgency": "one of: happening-now, today, this-week, upcoming, timeless", - "audience": "one of: all-students, specific-club, specific-year, grad-students, everyone", - "has_food": true/false, - "has_rsvp": true/false, - "formality": "one of: casual, semi-formal, formal", - "sentiment": "one of: excited, neutral, urgent, informational", - "topic_keywords": ["keyword1", "keyword2", "keyword3"]}""" - - -def phase_features(sample_size: int = 500): - """Extract structured LLM features for clustering improvement.""" - print(f"=== Phase: LLM Feature Extraction ({sample_size} emails) ===") - - msgs = load_emails() - - import random - random.seed(42) - sample = random.sample(msgs, min(sample_size, len(msgs))) - - features_path = DATA_DIR / "llm_features.json" - results = [] - - if features_path.exists(): - with open(features_path) as f: - results = json.load(f) - done_ids = {r["message_id"] for r in results} - sample = [m for m in sample if m["message_id"] not in done_ids] - print(f" Resuming: {len(results)} done, {len(sample)} remaining") - - for count, m in enumerate(sample): - email_text = format_email_for_llm(m) - resp = call_llm(FEATURES_SYSTEM, email_text) - - if resp: - try: - json_match = re.search(r"\{.*\}", resp, re.DOTALL) - parsed = json.loads(json_match.group()) if json_match else {} - except json.JSONDecodeError: - parsed = {} - - results.append({ - "message_id": m["message_id"], - "subject": m["subject"][:80], - "date": m.get("date", "")[:10], - **parsed, - }) - - if (count + 1) % 20 == 0: - print(f" Features [{count+1}/{len(sample)}]") - with open(features_path, "w") as f: - json.dump(results, f, indent=2, ensure_ascii=False) - - time.sleep(REQUEST_DELAY) - - with open(features_path, "w") as f: - json.dump(results, f, indent=2, ensure_ascii=False) - - # Stats - from collections import Counter - cats = Counter(r.get("category", "unknown") for r in results) - urgencies = Counter(r.get("urgency", "unknown") for r in results) - food = sum(1 for r in results if r.get("has_food")) - - print(f"\n Category distribution:") - for cat, count in cats.most_common(): - print(f" {cat:<25} {count:>4}") - print(f"\n Urgency: {dict(urgencies.most_common())}") - print(f" Has food: {food}/{len(results)} ({food/len(results)*100:.0f}%)") - print(f" Saved to {features_path}") - - -# ── Main ───────────────────────────────────────────────── - -def main(): - parser = argparse.ArgumentParser(description="LLM exploration on listserv data") - parser.add_argument("--phase", default="all", choices=["test", "classify", "tags", "features", "all"]) - parser.add_argument("--sample", type=int, default=500, help="Sample size for classification/features") - parser.add_argument("--discover-size", type=int, default=200, help="Sample size for tag discovery") - args = parser.parse_args() - - phases = ["test", "classify", "tags", "features"] if args.phase == "all" else [args.phase] - - for phase in phases: - if phase == "test": - if not phase_test(): - print("API test failed, aborting.") - return - elif phase == "classify": - phase_classify(args.sample) - elif phase == "tags": - phase_tags(args.discover_size, args.sample) - elif phase == "features": - phase_features(args.sample) - print() - - -if __name__ == "__main__": - main() diff --git a/apps/listserv-scraper/src/scrape_listserv.py b/apps/listserv-scraper/src/scrape_listserv.py deleted file mode 100644 index b935f7d..0000000 --- a/apps/listserv-scraper/src/scrape_listserv.py +++ /dev/null @@ -1,599 +0,0 @@ -""" -Scrape Princeton LISTSERV archives via RSS feed. - -Usage: - python3 src/scrape_listserv.py # scrape WHITMANWIRE (RSS only) - python3 src/scrape_listserv.py --list ROCKYWIRE # scrape a specific list - python3 src/scrape_listserv.py --fetch-bodies # also fetch full HTML per message - python3 src/scrape_listserv.py --fetch-bodies --batch-size 500 # save progress every 500 messages - python3 src/scrape_listserv.py --fetch-bodies --resume # resume from last checkpoint - python3 src/scrape_listserv.py --all # scrape all known listservs - python3 src/scrape_listserv.py --limit 5000 # cap RSS fetch at 5000 - -Outputs JSON to data/_emails.json -Progress checkpoints saved to data/_progress.json - -Required environment variables: - LISTSERV_EMAIL — LISTSERV login email (e.g. tigerapp@princeton.edu) - LISTSERV_PASSWORD — LISTSERV password -""" - -import argparse -import html -import http.client -import json -import os -import re -import time -import urllib.parse -import urllib.request -import xml.etree.ElementTree as ET -from datetime import datetime -from pathlib import Path - -BASE_URL = "https://lists.princeton.edu/cgi-bin/wa" - -LISTSERV_EMAIL = os.environ.get("LISTSERV_EMAIL", "") -LISTSERV_PASSWORD = os.environ.get("LISTSERV_PASSWORD", "") - -# Set by login() -COOKIE = "" -AUTH_PARAMS = "" - -KNOWN_LISTS = [ - "WHITMANWIRE", - "ROCKYWIRE", - "BUTLERBUZZ", - "MATHEYMAIL", - "RE-INNFORMER", - "WILSONWIRE", - "FREEFOOD", -] - -DATA_DIR = Path(__file__).resolve().parent.parent / "data" - - -def login() -> None: - """Log in to LISTSERV and set global COOKIE and AUTH_PARAMS.""" - global COOKIE, AUTH_PARAMS - - if not LISTSERV_EMAIL or not LISTSERV_PASSWORD: - raise RuntimeError( - "LISTSERV_EMAIL and LISTSERV_PASSWORD environment variables are required" - ) - - print("Logging in to LISTSERV...", end=" ", flush=True) - conn = http.client.HTTPSConnection("lists.princeton.edu") - data = urllib.parse.urlencode({ - "LOGIN1": "", - "Y": LISTSERV_EMAIL, - "p": LISTSERV_PASSWORD, - "e": "Log In", - "X": "", - }) - conn.request( - "POST", - "/cgi-bin/wa", - body=data, - headers={ - "Content-Type": "application/x-www-form-urlencoded", - "User-Agent": "TheForum-Scraper/1.0", - }, - ) - resp = conn.getresponse() - body = resp.read().decode("utf-8", errors="replace") - - # Extract WALOGIN cookie - cookie_val = None - for header, value in resp.getheaders(): - if header.lower() == "set-cookie" and "WALOGIN=" in value: - match = re.search(r"WALOGIN=([^;]+)", value) - if match: - cookie_val = match.group(1) - - # Extract session X token - x_match = re.search(r"X=([A-F0-9]{16,})", body) - - if not cookie_val or not x_match: - raise RuntimeError("Login failed — check LISTSERV_EMAIL/LISTSERV_PASSWORD") - - COOKIE = f"WALOGIN={cookie_val}" - AUTH_PARAMS = f"X={x_match.group(1)}&Y={urllib.parse.quote(LISTSERV_EMAIL)}" - print("OK") - - -def make_request(url: str, retries: int = 3) -> bytes: - """Make an authenticated request with retries.""" - for attempt in range(retries): - try: - req = urllib.request.Request(url) - req.add_header("Cookie", COOKIE) - req.add_header("User-Agent", "TheForum-Scraper/1.0") - with urllib.request.urlopen(req, timeout=120) as resp: - return resp.read() - except Exception as e: - if attempt < retries - 1: - time.sleep(2 ** attempt) - else: - raise e - - -def parse_hoagiemail(body_html: str) -> dict | None: - """Extract HoagieMail sender info from email footer.""" - match = re.search( - r"Email composed by\s+(.+?)\s+\((\S+@\S+)\)", body_html - ) - if match: - return {"name": match.group(1), "email": match.group(2)} - return None - - -def strip_html(html_str: str) -> str: - """Basic HTML to plain text conversion.""" - text = re.sub(r"", "\n", html_str, flags=re.IGNORECASE) - text = re.sub(r"]*>", "\n", text, flags=re.IGNORECASE) - text = re.sub(r"

", "", text, flags=re.IGNORECASE) - text = re.sub(r"]*>", "\n• ", text, flags=re.IGNORECASE) - text = re.sub(r"<[^>]+>", "", text) - text = html.unescape(text) - text = re.sub(r"\n{3,}", "\n\n", text) - return text.strip() - - -def extract_links(html_str: str) -> list[str]: - """Extract unique URLs from HTML.""" - urls = re.findall(r'href="(https?://[^"]+)"', html_str) - seen = set() - unique = [] - for url in urls: - if url not in seen and "lists.princeton.edu" not in url: - seen.add(url) - unique.append(url) - return unique - - -def extract_images(html_str: str) -> list[str]: - """Extract image URLs from HTML.""" - imgs = re.findall(r']+src="(https?://[^"]+)"', html_str) - seen = set() - unique = [] - for url in imgs: - clean = re.sub(r"#https?://.*$", "", url) - if clean not in seen: - seen.add(clean) - unique.append(clean) - return unique - - -def fetch_rss(listname: str, limit: int = 5000) -> list[dict]: - """Fetch and parse the RSS feed for a listserv.""" - url = f"{BASE_URL}?RSS&L={listname}&v=2.0&LIMIT={limit}&{AUTH_PARAMS}" - print(f" Fetching RSS feed (limit={limit})...", end=" ", flush=True) - data = make_request(url) - print(f"received {len(data):,} bytes") - - print(f" Parsing XML...", end=" ", flush=True) - root = ET.fromstring(data) - items = root.findall(".//item") - print(f"{len(items):,} messages") - - emails = [] - for item in items: - title = item.findtext("title", "") - link = item.findtext("link", "") - description = item.findtext("description", "") - author_raw = item.findtext("author", "") - pub_date = item.findtext("pubDate", "") - - # Parse author: "Name " format - author_match = re.match(r"(.+?)\s*<(.+?)>", author_raw) - author_name = author_match.group(1).strip() if author_match else author_raw - author_email = author_match.group(2).strip() if author_match else "" - - # Extract message ID from link - msg_id_match = re.search(r"A2=([^&]+)", link) - msg_id = msg_id_match.group(1) if msg_id_match else "" - - # Parse the HTML description - body_text = strip_html(description) - links = extract_links(description) - images = extract_images(description) - hoagiemail = parse_hoagiemail(description) - - # Detect HoagieMail from author email as fallback - is_hoagiemail = ( - author_email.upper() == "HOAGIE@PRINCETON.EDU" - or hoagiemail is not None - ) - - # Parse date - try: - parsed_date = datetime.strptime( - re.sub(r"\s+", " ", pub_date).strip(), - "%a, %d %b %Y %H:%M:%S %z", - ) - iso_date = parsed_date.isoformat() - except (ValueError, TypeError): - iso_date = pub_date - - emails.append( - { - "message_id": msg_id, - "subject": title, - "author_name": author_name, - "author_email": author_email, - "date": iso_date, - "date_raw": pub_date, - "body_html": description, - "body_text": body_text, - "links": links, - "images": images, - "is_hoagiemail": is_hoagiemail, - "hoagiemail_sender_name": hoagiemail["name"] if hoagiemail else None, - "hoagiemail_sender_email": hoagiemail["email"] if hoagiemail else None, - "listserv_url": link, - } - ) - - return emails - - -def fetch_full_message(msg_url: str) -> dict | None: - """Fetch the full message page, extract attachments, and full HTML body.""" - try: - sep = "&" if "?" in msg_url else "?" - url = f"{msg_url}{sep}{AUTH_PARAMS}" - data = make_request(url) - page = data.decode("utf-8", errors="replace") - - # Detect session expiry — re-login and retry once - if "Login Required" in page: - login() - url = f"{msg_url}{sep}{AUTH_PARAMS}" - data = make_request(url) - page = data.decode("utf-8", errors="replace") - if "Login Required" in page: - print(" [WARN] Re-login failed") - return None - - result = {} - - # Find all A3 attachment links on the page - a3_links = re.findall( - r'href="(/cgi-bin/wa\?A3=[^"]+)"[^>]*>([^<]+)', page - ) - attachments = [] - html_attachment_url = None - seen_urls = set() - for link, label in a3_links: - full_url = f"https://lists.princeton.edu{link}" - if full_url not in seen_urls: - seen_urls.add(full_url) - attachments.append( - {"url": full_url, "type": label.strip()} - ) - if "text/html" in label and html_attachment_url is None: - html_attachment_url = full_url - - result["attachments"] = attachments - - # Fetch the text/html attachment for the full rich body - if html_attachment_url: - try: - clean_url = html_attachment_url.replace("&header=1", "") - html_data = make_request(clean_url) - result["body_html"] = html_data.decode( - "utf-8", errors="replace" - ) - except Exception: - pass - - # Fallback: fetch text/plain attachment if no HTML available - if "body_html" not in result: - plain_url = None - for att in attachments: - if "text/plain" in att["type"]: - plain_url = att["url"] - break - if plain_url: - try: - clean_url = plain_url.replace("&header=1", "") - plain_data = make_request(clean_url) - result["body_html"] = plain_data.decode( - "utf-8", errors="replace" - ) - except Exception: - pass - - if "body_html" not in result: - return None - - return result - except Exception as e: - print(f" [WARN] Failed to fetch {msg_url}: {e}") - return None - - -def enrich_email(email: dict) -> None: - """Fetch full body for a single email and update it in place.""" - if not email.get("listserv_url"): - email["body_complete"] = False - return - - full_msg = fetch_full_message(email["listserv_url"]) - if full_msg: - if "body_html" in full_msg: - email["body_html"] = full_msg["body_html"] - email["body_text"] = strip_html(full_msg["body_html"]) - email["links"] = extract_links(full_msg["body_html"]) - email["images"] = extract_images(full_msg["body_html"]) - hoagiemail = parse_hoagiemail(full_msg["body_html"]) - if hoagiemail: - email["hoagiemail_sender_name"] = hoagiemail["name"] - email["hoagiemail_sender_email"] = hoagiemail["email"] - email["is_hoagiemail"] = True - if "attachments" in full_msg: - email["attachments"] = full_msg["attachments"] - email["body_complete"] = True - else: - email["body_complete"] = False - - -def load_progress(listname: str) -> dict | None: - """Load progress checkpoint if it exists.""" - progress_file = DATA_DIR / f"{listname.lower()}_progress.json" - if progress_file.exists(): - with open(progress_file, encoding="utf-8") as f: - return json.load(f) - return None - - -def save_progress(listname: str, emails: list[dict], completed_idx: int): - """Save progress checkpoint.""" - DATA_DIR.mkdir(parents=True, exist_ok=True) - progress_file = DATA_DIR / f"{listname.lower()}_progress.json" - with open(progress_file, "w", encoding="utf-8") as f: - json.dump( - { - "listserv": listname, - "saved_at": datetime.now().isoformat(), - "total_messages": len(emails), - "bodies_completed": completed_idx, - "messages": emails, - }, - f, - ensure_ascii=False, - ) - - -def load_existing(listname: str) -> dict[str, dict]: - """Load already-completed messages from the output file, keyed by message_id.""" - filepath = DATA_DIR / f"{listname.lower()}_emails.json" - if filepath.exists(): - with open(filepath, encoding="utf-8") as f: - data = json.load(f) - return { - m["message_id"]: m - for m in data.get("messages", []) - if m.get("body_complete") - } - return {} - - -def scrape_list( - listname: str, - limit: int = 5000, - fetch_bodies: bool = False, - batch_size: int = 200, - resume: bool = False, -) -> dict: - """Scrape a single listserv and return structured data.""" - print(f"\nScraping {listname}...") - - start_idx = 0 - emails = None - - # Check for in-progress resume first - if resume and fetch_bodies: - progress = load_progress(listname) - if progress: - emails = progress["messages"] - start_idx = progress["bodies_completed"] - print( - f" Resuming from checkpoint: {start_idx}/{len(emails)} bodies completed" - ) - - # Fetch RSS if not resuming from checkpoint - if emails is None: - emails = fetch_rss(listname, limit) - - # Load already-completed messages to skip re-fetching - existing = load_existing(listname) if fetch_bodies else {} - if existing: - merged = 0 - for email in emails: - if email["message_id"] in existing: - email.update(existing[email["message_id"]]) - # Re-run HoagieMail extraction in case prior run missed it - if not email.get("hoagiemail_sender_name") and email.get("body_html"): - hoagiemail = parse_hoagiemail(email["body_html"]) - if hoagiemail: - email["hoagiemail_sender_name"] = hoagiemail["name"] - email["hoagiemail_sender_email"] = hoagiemail["email"] - email["is_hoagiemail"] = True - merged += 1 - print(f" Loaded {merged} already-fetched bodies from previous run") - - # Fetch full bodies in batches - if fetch_bodies and emails: - # Count how many still need fetching - need_fetch = [ - i for i in range(start_idx, len(emails)) - if not emails[i].get("body_complete") - ] - - if not need_fetch: - print(f" All {len(emails)} bodies already fetched") - else: - est_minutes = len(need_fetch) * 0.4 / 60 - print( - f" Fetching full bodies: {len(need_fetch)} remaining " - f"(~{est_minutes:.0f} min)" - ) - - fetched = 0 - for i in range(start_idx, len(emails)): - if emails[i].get("body_complete"): - continue - - enrich_email(emails[i]) - fetched += 1 - - # Progress logging - if fetched % 10 == 0: - print( - f" [{fetched}/{len(need_fetch)}] — " - f"{emails[i]['subject'][:60]}", - flush=True, - ) - - # Save checkpoint every batch_size messages - if fetched % batch_size == 0: - save_progress(listname, emails, i + 1) - print(f" ** Checkpoint saved ({fetched} new bodies) **") - - time.sleep(0.15) - - # Final checkpoint - save_progress(listname, emails, len(emails)) - - # Compute stats - dates = [e["date"] for e in emails if e["date"]] - hoagiemail_count = sum(1 for e in emails if e["is_hoagiemail"]) - unique_authors = len(set(e["author_name"] for e in emails)) - - result = { - "listserv": listname, - "scraped_at": datetime.now().isoformat(), - "total_messages": len(emails), - "date_range": { - "oldest": dates[-1] if dates else None, - "newest": dates[0] if dates else None, - }, - "unique_authors": unique_authors, - "hoagiemail_messages": hoagiemail_count, - "messages": emails, - } - - return result - - -def save_result(result: dict, listname: str): - """Save final scrape results to JSON file.""" - DATA_DIR.mkdir(parents=True, exist_ok=True) - filepath = DATA_DIR / f"{listname.lower()}_emails.json" - with open(filepath, "w", encoding="utf-8") as f: - json.dump(result, f, ensure_ascii=False, indent=2) - size_mb = filepath.stat().st_size / (1024 * 1024) - print(f" Saved to {filepath} ({size_mb:.1f} MB)") - - # Clean up progress file - progress_file = DATA_DIR / f"{listname.lower()}_progress.json" - if progress_file.exists(): - progress_file.unlink() - print(f" Cleaned up progress checkpoint") - - -def main(): - parser = argparse.ArgumentParser( - description="Scrape Princeton LISTSERV archives" - ) - parser.add_argument( - "--list", - default="WHITMANWIRE", - help="Listserv name to scrape (default: WHITMANWIRE)", - ) - parser.add_argument( - "--all", - action="store_true", - help="Scrape all known listservs", - ) - parser.add_argument( - "--limit", - type=int, - default=5000, - help="Max messages to fetch from RSS (default: 5000)", - ) - parser.add_argument( - "--fetch-bodies", - action="store_true", - help="Fetch full message bodies from individual pages", - ) - parser.add_argument( - "--batch-size", - type=int, - default=200, - help="Save checkpoint every N messages (default: 200)", - ) - parser.add_argument( - "--resume", - action="store_true", - help="Resume from last checkpoint (requires --fetch-bodies)", - ) - args = parser.parse_args() - - lists_to_scrape = KNOWN_LISTS if args.all else [args.list.upper()] - - login() - - print(f"LISTSERV Scraper — {len(lists_to_scrape)} list(s) to scrape") - print(f"RSS limit: {args.limit} | Batch size: {args.batch_size}") - print(f"Fetch bodies: {args.fetch_bodies} | Resume: {args.resume}") - - all_stats = [] - for listname in lists_to_scrape: - try: - result = scrape_list( - listname, - args.limit, - args.fetch_bodies, - args.batch_size, - args.resume, - ) - save_result(result, listname) - all_stats.append( - { - "list": listname, - "messages": result["total_messages"], - "date_range": result["date_range"], - "unique_authors": result["unique_authors"], - "hoagiemail_pct": ( - f"{result['hoagiemail_messages'] / result['total_messages'] * 100:.0f}%" - if result["total_messages"] > 0 - else "0%" - ), - } - ) - except Exception as e: - print(f" [ERROR] Failed to scrape {listname}: {e}") - all_stats.append({"list": listname, "error": str(e)}) - - print("\n" + "=" * 60) - print("Summary:") - print("=" * 60) - for stat in all_stats: - if "error" in stat: - print(f" {stat['list']}: ERROR — {stat['error']}") - else: - print( - f" {stat['list']}: {stat['messages']:,} messages, " - f"{stat['unique_authors']} authors, " - f"{stat['hoagiemail_pct']} via HoagieMail, " - f"range: {stat['date_range']['oldest'][:10] if stat['date_range']['oldest'] else '?'} → " - f"{stat['date_range']['newest'][:10] if stat['date_range']['newest'] else '?'}" - ) - print() - - -if __name__ == "__main__": - main() diff --git a/apps/mpu-scraper/.gitignore b/apps/mpu-scraper/.gitignore deleted file mode 100644 index e7eec9a..0000000 --- a/apps/mpu-scraper/.gitignore +++ /dev/null @@ -1,2 +0,0 @@ -cookies.json -data/ diff --git a/apps/mpu-scraper/package.json b/apps/mpu-scraper/package.json deleted file mode 100644 index 7b3c7ca..0000000 --- a/apps/mpu-scraper/package.json +++ /dev/null @@ -1,8 +0,0 @@ -{ - "name": "@the-forum/mpu-scraper", - "version": "0.0.1", - "private": true, - "scripts": { - "scrape": "bun run src/index.ts" - } -} diff --git a/apps/mpu-scraper/src/index.ts b/apps/mpu-scraper/src/index.ts deleted file mode 100644 index e7836a6..0000000 --- a/apps/mpu-scraper/src/index.ts +++ /dev/null @@ -1,524 +0,0 @@ -/** - * MyPrincetonU Club Officers Scraper - * - * Scrapes all club/org officer data from my.princeton.edu's CampusGroups API. - * - * Usage: - * 1. Export your browser cookies to apps/mpu-scraper/cookies.json - * (only CG.SessionID and cg_uid are required) - * 2. bun run scrape - * - * Output: apps/mpu-scraper/data/club_officers.json - */ - -import { execSync } from "node:child_process"; -import { existsSync, mkdirSync, readFileSync, writeFileSync } from "node:fs"; -import { join } from "node:path"; - -// --------------------------------------------------------------------------- -// Types -// --------------------------------------------------------------------------- - -interface CookieEntry { - name: string; - value: string; - domain?: string; - path?: string; -} - -interface ClubListItem { - id: number; - name: string; - email: string | null; - groupTypeValue: string; - categoryTagIds: string[]; - published: boolean; - countMembers: number; - countOfficers: number; -} - -interface ClubAboutOfficer { - position: string; - student: { - id: number; - firstName: string; - lastName: string; - profilePhotoFileName: string | null; - profilePhotoSubFolder: string; - uid: string; - }; -} - -interface ClubAboutResponse { - club: { - id: number; - name: string; - email: string | null; - mission: string | null; - whatWeDo: string | null; - goals: string | null; - websiteUrl: string | null; - instagram: string | null; - facebook: string | null; - twitter: string | null; - youtube: string | null; - linkedin: string | null; - discord: string | null; - groupTypeValue: string; - categories: { name: string }[]; - officers: ClubAboutOfficer[] | null; - }; -} - -interface UserProfile { - email: string | null; - accountType: string | null; - yearOfGraduation: string | null; -} - -interface OfficerEntry { - position: string; - first_name: string; - last_name: string; - student_id: number; - uid: string; - email: string | null; - net_id: string | null; -} - -interface MemberEntry { - first_name: string; - last_name: string; - email: string; - student_type: string; - year_of_graduation: string; - student_id: string; - uid: string; -} - -interface ClubOfficerRecord { - club_id: number; - club_name: string; - club_email: string | null; - group_type: string; - categories: string[]; - mission: string | null; - website: string | null; - instagram: string | null; - member_count: number; - officer_count: number; - officers: OfficerEntry[]; - president: (OfficerEntry & { email: string | null }) | null; - members: MemberEntry[]; -} - -// --------------------------------------------------------------------------- -// Config -// --------------------------------------------------------------------------- - -const BASE = "https://my.princeton.edu"; -const ROOT_DIR = join(import.meta.dir, ".."); -const COOKIE_FILE = join(ROOT_DIR, "cookies.json"); -const DATA_DIR = join(ROOT_DIR, "data"); -const OUTPUT_FILE = join(DATA_DIR, "club_officers.json"); -const PAGE_SIZE = 50; -const DELAY_MS = 200; - -// --------------------------------------------------------------------------- -// Cookie loading -// --------------------------------------------------------------------------- - -function loadCookieString(): string { - if (!existsSync(COOKIE_FILE)) { - console.error(`ERROR: Cookie file not found at ${COOKIE_FILE}`); - console.error("Export your my.princeton.edu cookies from the browser and save them there."); - process.exit(1); - } - - const raw: CookieEntry[] = JSON.parse(readFileSync(COOKIE_FILE, "utf-8")); - const essential = new Set(["CG.SessionID", "cg_uid", "AWSALB", "AWSALBCORS"]); - return raw - .filter((c) => essential.has(c.name)) - .map((c) => `${c.name}=${c.value}`) - .join("; "); -} - -// --------------------------------------------------------------------------- -// HTTP helpers -// --------------------------------------------------------------------------- - -const cookies = loadCookieString(); - -async function fetchJson(path: string): Promise { - const url = path.startsWith("http") ? path : `${BASE}${path}`; - try { - const res = await fetch(url, { - headers: { - Cookie: cookies, - "X-Requested-With": "XMLHttpRequest", - Accept: "application/json", - }, - redirect: "follow", - signal: AbortSignal.timeout(15_000), - }); - if (!res.ok) return null; - const text = await res.text(); - return JSON.parse(text) as T; - } catch { - return null; - } -} - -function sleep(ms: number): Promise { - return new Promise((r) => setTimeout(r, ms)); -} - -// --------------------------------------------------------------------------- -// LDAP lookup — resolves email → Princeton netID -// --------------------------------------------------------------------------- - -function ldapLookupNetId(email: string): string | null { - try { - const result = execSync( - `ldapsearch -x -H ldaps://ldap.princeton.edu -b "o=Princeton University,c=US" "(mail=${email})" uid`, - { timeout: 10_000, encoding: "utf-8", stdio: ["pipe", "pipe", "pipe"] }, - ); - const match = result.match(/^uid:\s*(\S+)/m); - return match?.[1]?.toLowerCase() ?? null; - } catch { - return null; - } -} - -// --------------------------------------------------------------------------- -// Scraping logic -// --------------------------------------------------------------------------- - -async function verifySession(): Promise { - const data = await fetchJson<{ jwt: string; csrf_token: string }>( - "/mobile_ws/v18/mobile_session", - ); - return data !== null && "jwt" in data; -} - -async function fetchAllClubs(): Promise { - const clubs: ClubListItem[] = []; - let page = 1; - - while (true) { - const data = await fetchJson<{ - data: ClubListItem[]; - _metadata: { total: number }; - }>(`/mobile_ws/v18/mobile_clubs_listing?page=${page}&pageSize=${PAGE_SIZE}`); - - if (!data?.data) { - console.error(` ERROR: unexpected response on page ${page}`); - break; - } - - const total = data._metadata.total; - clubs.push(...data.data); - console.log(` Page ${page}: got ${data.data.length} clubs (${clubs.length}/${total})`); - - if (data.data.length < PAGE_SIZE || clubs.length >= total) break; - page++; - await sleep(DELAY_MS); - } - - return clubs; -} - -async function fetchClubOfficers(club: ClubListItem): Promise { - const about = await fetchJson( - `/mobile_ws/v18/mobile_club_about?id=${club.id}`, - ); - - if (!about?.club) return null; - - const info = about.club; - const officers = info.officers ?? []; - - const president = officers.find((o) => o.position?.toLowerCase() === "president"); - - return { - club_id: club.id, - club_name: info.name.trim(), - club_email: info.email, - group_type: info.groupTypeValue, - categories: (info.categories ?? []).map((c) => c.name), - mission: info.mission, - website: info.websiteUrl, - instagram: info.instagram, - member_count: club.countMembers ?? 0, - officer_count: club.countOfficers ?? 0, - officers: officers.map((o) => ({ - position: o.position, - first_name: o.student.firstName, - last_name: o.student.lastName, - student_id: o.student.id, - uid: o.student.uid, - email: null, // populated in email resolution pass - net_id: null, // populated in LDAP resolution pass - })), - president: president - ? { - first_name: president.student.firstName, - last_name: president.student.lastName, - student_id: president.student.id, - uid: president.student.uid, - email: null, - net_id: null, - position: president.position, - } - : null, - members: [], // populated in member fetching pass - }; -} - -interface MemberListingItem { - fields: string; - counter: string; - [key: `p${number}`]: string; -} - -async function fetchClubMembers(clubId: number): Promise { - const data = await fetchJson( - `/mobile_ws/v17/mobile_group_page_members?range=0&limit=5000¶m=${clubId}`, - ); - - if (!data || !Array.isArray(data) || data.length === 0) return []; - - const fields = data[0].fields.replace(/,$/, "").split(","); - const idx = (name: string) => fields.indexOf(name); - - return data.map((item) => ({ - first_name: item[`p${idx("firstName")}`] ?? "", - last_name: item[`p${idx("lastName")}`] ?? "", - email: item[`p${idx("student_email")}`] ?? "", - student_type: item[`p${idx("studentType")}`] ?? "", - year_of_graduation: item[`p${idx("yearOfGraduation")}`] ?? "", - student_id: item[`p${idx("studentId")}`] ?? "", - uid: item[`p${idx("studentUid")}`] ?? "", - })); -} - -// --------------------------------------------------------------------------- -// Main -// --------------------------------------------------------------------------- - -async function main() { - console.log("MyPrincetonU Club Officers Scraper"); - console.log("==================================\n"); - - // 1. Verify session - console.log("Verifying session..."); - const alive = await verifySession(); - if (!alive) { - console.error("Session invalid or expired. Re-export your cookies."); - process.exit(1); - } - console.log(" Session OK\n"); - - // 2. Fetch all clubs - console.log("Fetching club list..."); - const clubs = await fetchAllClubs(); - console.log(`\nTotal clubs: ${clubs.length}\n`); - - // 3. Fetch officers for each club - console.log(`Fetching officer details for ${clubs.length} clubs...`); - const results: ClubOfficerRecord[] = []; - const errors: { id: number; name: string }[] = []; - - for (let i = 0; i < clubs.length; i++) { - const club = clubs[i]; - const record = await fetchClubOfficers(club); - - if (record) { - results.push(record); - } else { - errors.push({ id: club.id, name: club.name.trim() }); - } - - if ((i + 1) % 50 === 0 || i === clubs.length - 1) { - const withPres = results.filter((r) => r.president).length; - console.log( - ` [${i + 1}/${clubs.length}] ${results.length} ok, ${withPres} have president, ${errors.length} errors`, - ); - } - - await sleep(DELAY_MS); - } - - // 4. Fetch member lists for each club - console.log(`\nFetching member lists for ${results.length} clubs...`); - let totalMembers = 0; - - for (let i = 0; i < results.length; i++) { - const r = results[i]; - r.members = await fetchClubMembers(r.club_id); - totalMembers += r.members.length; - - if ((i + 1) % 50 === 0 || i === results.length - 1) { - console.log(` [${i + 1}/${results.length}] ${totalMembers} total members scraped`); - } - - await sleep(DELAY_MS); - } - - // 5. Resolve officer emails via profile endpoint - // Deduplicate by uid so we don't fetch the same person multiple times - const uniqueUids: string[] = []; - const seen = new Set(); - - for (const r of results) { - for (const o of r.officers) { - if (o.uid && !seen.has(o.uid)) { - seen.add(o.uid); - uniqueUids.push(o.uid); - } - } - } - - const uidToEmail = new Map(); - - console.log(`\nResolving emails for ${uniqueUids.length} unique officers...`); - - let resolved = 0; - let emailErrors = 0; - - for (let i = 0; i < uniqueUids.length; i++) { - const uid = uniqueUids[i]; - - const profile = await fetchJson( - `/mobile_ws/v18/mobile_profile2?uid=${uid}&view=full`, - ); - - if (profile?.email) { - uidToEmail.set(uid, profile.email); - resolved++; - } else { - emailErrors++; - } - - if ((i + 1) % 100 === 0 || i === uniqueUids.length - 1) { - console.log( - ` [${i + 1}/${uniqueUids.length}] ${resolved} emails resolved, ${emailErrors} failed`, - ); - } - - await sleep(DELAY_MS); - } - - // Populate emails back into all officer records - for (const r of results) { - for (const o of r.officers) { - o.email = uidToEmail.get(o.uid) ?? null; - } - if (r.president) { - r.president.email = uidToEmail.get(r.president.uid) ?? null; - } - } - - // 6. Resolve netIDs via Princeton LDAP - // Deduplicate emails so we don't query the same one twice - const emailToNetId = new Map(); - const uniqueEmails: string[] = []; - - for (const r of results) { - for (const o of r.officers) { - if (o.email && !emailToNetId.has(o.email)) { - emailToNetId.set(o.email, null); - uniqueEmails.push(o.email); - } - } - } - - console.log(`\nResolving netIDs via LDAP for ${uniqueEmails.length} unique emails...`); - - let netIdResolved = 0; - let netIdFailed = 0; - - for (let i = 0; i < uniqueEmails.length; i++) { - const email = uniqueEmails[i]; - const netId = ldapLookupNetId(email); - - if (netId) { - emailToNetId.set(email, netId); - netIdResolved++; - } else { - netIdFailed++; - } - - if ((i + 1) % 100 === 0 || i === uniqueEmails.length - 1) { - console.log( - ` [${i + 1}/${uniqueEmails.length}] ${netIdResolved} resolved, ${netIdFailed} failed`, - ); - } - } - - // Populate netIDs back into all officer records - for (const r of results) { - for (const o of r.officers) { - if (o.email) { - o.net_id = emailToNetId.get(o.email) ?? null; - } - } - if (r.president?.email) { - r.president.net_id = emailToNetId.get(r.president.email) ?? null; - } - } - - // 7. Write output - mkdirSync(DATA_DIR, { recursive: true }); - - const output = { - scraped_at: new Date().toISOString(), - total_clubs: clubs.length, - clubs_with_officers: results.filter((r) => r.officers.length > 0).length, - clubs_with_president: results.filter((r) => r.president).length, - unique_officers: uidToEmail.size, - officers_with_email: resolved, - officers_with_net_id: netIdResolved, - total_members_scraped: totalMembers, - clubs_with_members: results.filter((r) => r.members.length > 0).length, - errors: errors.length, - clubs: results, - }; - - writeFileSync(OUTPUT_FILE, JSON.stringify(output, null, 2)); - console.log(`\nResults saved to ${OUTPUT_FILE}`); - - // 8. Summary - console.log(`\n${"=".repeat(60)}`); - console.log("SUMMARY"); - console.log("=".repeat(60)); - console.log(`Total clubs: ${output.total_clubs}`); - console.log(`Clubs with officers: ${output.clubs_with_officers}`); - console.log(`Clubs with president: ${output.clubs_with_president}`); - console.log(`Unique officers: ${output.unique_officers}`); - console.log(`Officers with email: ${output.officers_with_email}`); - console.log(`Officers with netID: ${output.officers_with_net_id}`); - console.log(`Total members scraped: ${output.total_members_scraped}`); - console.log(`Clubs with members: ${output.clubs_with_members}`); - console.log(`Errors: ${output.errors}`); - - const presidents = results.filter((r) => r.president); - if (presidents.length > 0) { - console.log("\nSample presidents:"); - for (const r of presidents.slice(0, 10)) { - const p = r.president as NonNullable; - console.log( - ` ${r.club_name.padEnd(40)} → ${p.first_name} ${p.last_name} <${p.email ?? "?"}> (netID: ${p.net_id ?? "?"})`, - ); - } - } - - if (errors.length > 0) { - console.log("\nFirst few errors:"); - for (const e of errors.slice(0, 5)) { - console.log(` ${e.name} (ID ${e.id})`); - } - } -} - -main(); diff --git a/apps/mpu-scraper/tsconfig.json b/apps/mpu-scraper/tsconfig.json deleted file mode 100644 index ac3c17e..0000000 --- a/apps/mpu-scraper/tsconfig.json +++ /dev/null @@ -1,14 +0,0 @@ -{ - "compilerOptions": { - "target": "ESNext", - "module": "ESNext", - "moduleResolution": "bundler", - "strict": true, - "esModuleInterop": true, - "skipLibCheck": true, - "outDir": "dist", - "rootDir": "src", - "types": ["bun-types"] - }, - "include": ["src"] -} diff --git a/apps/web/.env.local.example b/apps/web/.env.local.example index b4438dc..ce63720 100644 --- a/apps/web/.env.local.example +++ b/apps/web/.env.local.example @@ -5,14 +5,17 @@ # ----- Database (Docker default — see `bun run db:up`) ----- DATABASE_URL=postgresql://forum:forum_password@localhost:5434/the_forum -# ----- Auth.js (Microsoft Entra ID / Princeton login) ----- +# ----- Auth.js + Princeton CAS login ----- # Generate AUTH_SECRET yourself: openssl rand -base64 32 AUTH_SECRET=YOUR_AUTH_SECRET_HERE -# Ask Ibraheem for the client ID + secret -AUTH_AZURE_AD_CLIENT_ID=YOUR_CLIENT_ID_HERE -AUTH_AZURE_AD_CLIENT_SECRET=YOUR_CLIENT_SECRET_HERE -# Princeton tenant ID (not secret) -AUTH_AZURE_AD_TENANT_ID=2ff60116-7431-425d-b5af-077d7791bda4 +# Login uses Princeton CAS (no client ID/secret needed). CAS redirects back to +# http://localhost:3000/api/auth/cas/callback in dev. +# Canonical public URL — set this in production (e.g. https://forum.example.edu). +# AUTH_URL=http://localhost:3000 +# Only needed behind a reverse proxy without AUTH_URL (not needed on Vercel). +# AUTH_TRUST_HOST=true +# CAS server (default shown) +# CAS_BASE_URL=https://fed.princeton.edu/cas/ # ----- Mapbox (ask Ibraheem for these) ----- NEXT_PUBLIC_MAPBOX_TOKEN=YOUR_MAPBOX_TOKEN_HERE @@ -20,7 +23,8 @@ NEXT_PUBLIC_CAMPUS_MAP_TOKEN=YOUR_CAMPUS_MAP_TOKEN_HERE NEXT_PUBLIC_CAMPUS_MAP_STYLE=YOUR_CAMPUS_MAP_STYLE_HERE # ----- Optional ----- -NEXT_PUBLIC_API_URL=http://localhost:8000 +# Public origin for metadata / Open Graph / sitemap (defaults to https://forum.tigerapps.org) +# NEXT_PUBLIC_SITE_URL=http://localhost:3000 # AWS S3 (image uploads) # AWS_S3_BUCKET=the-forum-uploads # AWS_REGION=us-east-1 diff --git a/apps/web/next.config.ts b/apps/web/next.config.ts index 85c7b0f..fd72621 100644 --- a/apps/web/next.config.ts +++ b/apps/web/next.config.ts @@ -1,9 +1,14 @@ // Validate env vars at build time — throws if required vars are missing. import "./src/env.ts"; +import path from "node:path"; import type { NextConfig } from "next"; const nextConfig: NextConfig = { + // Self-contained server bundle for EC2 deploys (deploy/build-release.sh). + output: "standalone", + // Monorepo: trace workspace packages from the repo root. + outputFileTracingRoot: path.join(import.meta.dirname, "../.."), // Allow Next.js to transpile the workspace database package (TypeScript sources) transpilePackages: ["@the-forum/database"], images: { diff --git a/apps/web/package.json b/apps/web/package.json index 5a9e26b..c152d92 100644 --- a/apps/web/package.json +++ b/apps/web/package.json @@ -7,10 +7,10 @@ "build": "next build", "start": "next start", "lint": "biome check .", + "test": "bun test", "check-types": "tsc --noEmit" }, "dependencies": { - "@auth/drizzle-adapter": "^1.11.1", "@aws-sdk/client-s3": "^3.1004.0", "@aws-sdk/s3-request-presigner": "^3.1004.0", "@t3-oss/env-nextjs": "^0.13.10", @@ -20,6 +20,7 @@ "clsx": "^2.1.1", "cmdk": "^1.1.1", "date-fns": "^4.1.0", + "fast-xml-parser": "^5.4.1", "lucide-react": "^0.575.0", "mapbox-gl": "^3.20.0", "next": "^16.1.6", diff --git a/apps/web/src/actions/events.ts b/apps/web/src/actions/events.ts index 4d00d52..010ce66 100644 --- a/apps/web/src/actions/events.ts +++ b/apps/web/src/actions/events.ts @@ -8,11 +8,8 @@ import { desc, eq, eventTags, - friendships, gt, - ilike, inArray, - lt, ne, notifications, or, @@ -22,12 +19,34 @@ import { rsvps, savedEvents, sql, - userInterests, users, } from "@the-forum/database"; import { revalidatePath } from "next/cache"; +import { z } from "zod"; import { auth } from "~/auth"; import { formatEventDateTime } from "~/lib/date-format"; +import { + canEditEvent, + canViewEvent, + eventDiscoverableBy, + eventEditableBy, + eventVisibleTo, +} from "~/lib/event-visibility"; +import { type FeedPage, loadRankedFeed } from "~/lib/feed"; +import { enforceRateLimit } from "~/lib/rate-limit"; +import { isOurImageUrl, uploadedImageUrlSchema } from "~/lib/s3"; +import { loadFriendIds } from "~/lib/social-graph"; +import { + dateInputSchema, + eventStatusSchema, + eventTagSchema, + httpsUrlSchema, + idSchema, + locationIdSchema, + orgCategorySchema, + parseInput, + uniqueEnumArray, +} from "~/lib/validation"; export interface FeedEvent { id: string; @@ -35,10 +54,14 @@ export interface FeedEvent { description: string | null; orgId: string | null; orgName: string | null; + /** Organization logo (MyPrincetonU groups often have one), else null. */ + orgLogoUrl?: string | null; datetime: string; /** ISO timestamp, for building calendar links client-side. */ rawDatetime?: string; location: string; + /** Room or free-text place within the venue, e.g. "Room 207". */ + locationDetail?: string | null; tags: string[]; flyerUrl: string | null; rsvpCount: number; @@ -54,21 +77,6 @@ export interface FeedEvent { isSaved: boolean; } -/** Accepted friendships are stored one-directional, so both columns are read. */ -async function loadFriendIds(userId: string): Promise { - const [outgoing, incoming] = await Promise.all([ - db - .select({ friendId: friendships.friendId }) - .from(friendships) - .where(and(eq(friendships.userId, userId), eq(friendships.status, "accepted"))), - db - .select({ friendId: friendships.userId }) - .from(friendships) - .where(and(eq(friendships.friendId, userId), eq(friendships.status, "accepted"))), - ]); - return [...outgoing, ...incoming].map((r) => r.friendId); -} - interface EventEnrichment { tags: Map; attendees: Map>; @@ -158,7 +166,19 @@ async function loadEventEnrichment( return result; } -export async function getFeedEvents(params?: { +const feedParamsSchema = z.object({ + search: z.string().trim().max(200).optional(), + tags: z.array(eventTagSchema).max(eventTagSchema.options.length).optional(), + orgCategory: orgCategorySchema.optional(), + locationId: z.string().max(100).optional(), + dateRange: z.enum(["today", "week", "month"]).optional(), + limit: z.number().int().min(1).max(50).default(20), + offset: z.number().int().min(0).max(5000).default(0), + asOf: z.iso.datetime().optional(), +}); + +/** Loosely typed on purpose — the schema above is what actually validates it. */ +export interface FeedParams { search?: string; tags?: string[]; orgCategory?: string; @@ -166,284 +186,33 @@ export async function getFeedEvents(params?: { dateRange?: "today" | "week" | "month"; limit?: number; offset?: number; -}): Promise<{ events: FeedEvent[]; total: number }> { + asOf?: string; +} + +/** + * One page of the ranked Explore feed. Ranking, org-diversity and the + * soon-event quota are applied to the whole candidate list before paginating; + * see `~/lib/feed.ts` and docs/ranking.md. + * + * Pass the returned `asOf` back with the next `offset` so later pages are + * slices of the same ranking. + */ +export async function getFeedEvents(params?: FeedParams): Promise { const session = await auth(); if (!session?.user?.id) throw new Error("Unauthorized"); - const userId = session.user.id; - const limit = params?.limit ?? 20; - const offset = params?.offset ?? 0; - - // Get user's interests for scoring - const myInterests = await db - .select({ tag: userInterests.tag }) - .from(userInterests) - .where(eq(userInterests.userId, userId)); - const myInterestTags = myInterests.map((i) => i.tag); - - // Get user's friend IDs - const friendRows = await db - .select({ friendId: friendships.friendId }) - .from(friendships) - .where(and(eq(friendships.userId, userId), eq(friendships.status, "accepted"))); - const reverseFriendRows = await db - .select({ friendId: friendships.userId }) - .from(friendships) - .where(and(eq(friendships.friendId, userId), eq(friendships.status, "accepted"))); - const friendIds = [ - ...friendRows.map((f) => f.friendId), - ...reverseFriendRows.map((f) => f.friendId), - ]; - - // Get orgs the user follows or belongs to, for the org-affinity signal - const followedOrgRows = await db - .select({ orgId: orgFollowers.orgId }) - .from(orgFollowers) - .where(eq(orgFollowers.userId, userId)); - const memberOrgRows = await db - .select({ orgId: orgMembers.orgId }) - .from(orgMembers) - .where(eq(orgMembers.userId, userId)); - const myOrgIds = new Set([ - ...followedOrgRows.map((o) => o.orgId), - ...memberOrgRows.map((o) => o.orgId), - ]); - - // Build base query conditions — only show published events in the feed - const conditions = [gt(events.datetime, new Date()), eq(events.status, "published")]; - - if (params?.search) { - const searchCondition = or( - ilike(events.title, `%${params.search}%`), - ilike(events.description, `%${params.search}%`), - ); - - if (searchCondition) { - conditions.push(searchCondition); - } - } - - if (params?.tags && params.tags.length > 0) { - const typedTags = params.tags as (typeof eventTags.$inferSelect.tag)[]; - const eventsWithTags = db - .select({ eventId: eventTags.eventId }) - .from(eventTags) - .where(inArray(eventTags.tag, typedTags)); - conditions.push(inArray(events.id, eventsWithTags)); - } - - if (params?.orgCategory) { - const orgsInCategory = db - .select({ id: organizations.id }) - .from(organizations) - .where( - eq( - organizations.category, - params.orgCategory as typeof organizations.$inferSelect.category, - ), - ); - conditions.push(inArray(events.orgId, orgsInCategory)); - } - - if (params?.locationId) { - conditions.push(eq(events.locationId, params.locationId)); - } - - if (params?.dateRange) { - const now = new Date(); - let end: Date; - if (params.dateRange === "today") { - end = new Date(now); - end.setHours(23, 59, 59, 999); - } else if (params.dateRange === "week") { - end = new Date(now.getTime() + 7 * 24 * 60 * 60 * 1000); - } else { - end = new Date(now.getTime() + 30 * 24 * 60 * 60 * 1000); - } - conditions.push(lt(events.datetime, end)); - } - - // Get total count - const [countResult] = await db - .select({ count: sql`count(*)::int` }) - .from(events) - .where(and(...conditions)); - const total = countResult?.count ?? 0; - - // Fetch events with scoring - const rawEvents = await db - .select({ - id: events.id, - title: events.title, - description: events.description, - datetime: events.datetime, - flyerUrl: events.flyerUrl, - locationName: campusLocations.name, - orgId: events.orgId, - orgName: organizations.name, - createdAt: events.createdAt, - }) - .from(events) - .leftJoin(campusLocations, eq(events.locationId, campusLocations.id)) - .leftJoin(organizations, eq(events.orgId, organizations.id)) - .where(and(...conditions)) - .orderBy(events.datetime) - .limit(limit) - .offset(offset); - - // Enrich each event with tags, rsvp counts, friend attendance, user state - // - // Ranking, in plain English: an event scores higher if (1) its tags match - // your interests, (2) it's happening soon, (3) friends of yours are - // attending, (4) it belongs to an org you follow or belong to, or (5) it - // was posted recently. Scores are deterministic — no randomness — so - // refreshing Explore without new data (RSVPs, new events, etc.) never - // reorders the feed. Ties break by soonest event first. - const enriched: (FeedEvent & { score: number; _rawDatetime: Date })[] = await Promise.all( - rawEvents.map(async (event) => { - // Get tags - const tags = await db - .select({ tag: eventTags.tag }) - .from(eventTags) - .where(eq(eventTags.eventId, event.id)); - - // Get RSVP count - const [rsvpCount] = await db - .select({ count: sql`count(*)::int` }) - .from(rsvps) - .where(eq(rsvps.eventId, event.id)); - - // Get friends attending - let friendsAttending: { id: string; displayName: string; avatarUrl: string | null }[] = []; - if (friendIds.length > 0) { - friendsAttending = await db - .select({ - id: users.id, - displayName: users.displayName, - avatarUrl: users.avatarUrl, - }) - .from(rsvps) - .innerJoin(users, eq(rsvps.userId, users.id)) - .where(and(eq(rsvps.eventId, event.id), inArray(rsvps.userId, friendIds))); - } - - // Check if user has RSVP'd or saved - const [userRsvp] = await db - .select() - .from(rsvps) - .where(and(eq(rsvps.userId, userId), eq(rsvps.eventId, event.id))) - .limit(1); - - const [userSave] = await db - .select() - .from(savedEvents) - .where(and(eq(savedEvents.userId, userId), eq(savedEvents.eventId, event.id))) - .limit(1); - - // Score for sorting - const tagNames = tags.map((t) => t.tag); - - // Fraction of this event's tags that match the user's interests — how - // relevant is this event to you, not how much of your profile it covers. - const matchedTags = tagNames.filter((t) => myInterestTags.includes(t)).length; - const interestRelevance = - myInterestTags.length === 0 - ? 0.5 - : tagNames.length === 0 - ? 0 - : matchedTags / tagNames.length; - - const now = Date.now(); - const eventTime = event.datetime.getTime(); - const daysUntil = (eventTime - now) / (1000 * 60 * 60 * 24); - const timeProximity = - daysUntil <= 1 - ? 1.0 - : daysUntil <= 3 - ? 0.8 - : daysUntil <= 7 - ? 0.6 - : daysUntil <= 14 - ? 0.3 - : 0.1; - - const friendRsvpScore = Math.min(1.0, friendsAttending.length / 3.0); - - const orgAffinity = event.orgId && myOrgIds.has(event.orgId) ? 1.0 : 0.0; - - const hoursSinceCreated = (now - event.createdAt.getTime()) / (1000 * 60 * 60); - const recencyBoost = hoursSinceCreated <= 24 ? 1.0 : hoursSinceCreated <= 72 ? 0.5 : 0.0; - - const score = - 3.0 * interestRelevance + - 2.0 * timeProximity + - 4.0 * friendRsvpScore + - 1.0 * orgAffinity + - 1.0 * recencyBoost; - - return { - id: event.id, - title: event.title, - description: event.description, - orgId: event.orgId, - orgName: event.orgName, - datetime: formatEventDateTime(event.datetime), - location: event.locationName ?? "TBD", - tags: tagNames, - flyerUrl: event.flyerUrl, - rsvpCount: rsvpCount?.count ?? 0, - friendsAttending, - isRsvped: !!userRsvp, - isSaved: !!userSave, - score, - _rawDatetime: event.datetime, - }; - }), - ); - - /* - * Attendees for the "N attending" control on each card. - * - * One query for the whole page rather than one per event — the enrichment - * above already issues several queries per event, and adding another to that - * loop is what pushes the connection pool over on a full feed. - */ - const feedIds = enriched.map((e) => e.id); - const attendeeRows = - feedIds.length === 0 - ? [] - : await db - .select({ - eventId: rsvps.eventId, - id: users.id, - displayName: users.displayName, - avatarUrl: users.avatarUrl, - }) - .from(rsvps) - .innerJoin(users, eq(rsvps.userId, users.id)) - .where(inArray(rsvps.eventId, feedIds)); - - const attendeesByEvent = new Map(); - for (const { eventId, ...person } of attendeeRows) { - const list = attendeesByEvent.get(eventId); - if (list) list.push(person); - else attendeesByEvent.set(eventId, [person]); - } - - // Sort by score descending; ties break by soonest event first, so - // refreshing Explore with no new data never reorders the feed. - enriched.sort((a, b) => { - if (b.score !== a.score) return b.score - a.score; - return a._rawDatetime.getTime() - b._rawDatetime.getTime(); + const input = parseInput(feedParamsSchema, params ?? {}); + + return loadRankedFeed(session.user.id, { + search: input.search || undefined, + tags: input.tags, + orgCategory: input.orgCategory, + locationId: input.locationId, + dateRange: input.dateRange, + limit: input.limit, + offset: input.offset, + asOf: input.asOf ? new Date(input.asOf) : undefined, }); - - return { - events: enriched.map(({ score: _score, _rawDatetime, ...event }) => ({ - ...event, - attendees: attendeesByEvent.get(event.id) ?? [], - })), - total, - }; } /** The attendee shape shared by the feed, the detail page and `toggleRsvp`. */ @@ -463,24 +232,26 @@ export async function toggleRsvp(eventId: string): Promise<{ if (!session?.user?.id) throw new Error("Unauthorized"); const userId = session.user.id; + const id = parseInput(idSchema, eventId); - // If the event doesn't exist in the DB (e.g. a demo/local-only event), no-op - const [eventRow] = await db.select().from(events).where(eq(events.id, eventId)).limit(1); - if (!eventRow) { - // Return zero count and no-op rsvp change to avoid FK constraint errors + // Missing, draft or private events the viewer can't see: no-op, exactly as + // if the event didn't exist (so this can't be used to probe for them). + if (!(await canViewEvent(id, userId))) { return { rsvped: false, count: 0, attendees: [] }; } const [existing] = await db - .select() + .select({ eventId: rsvps.eventId }) .from(rsvps) - .where(and(eq(rsvps.userId, userId), eq(rsvps.eventId, eventId))) + .where(and(eq(rsvps.userId, userId), eq(rsvps.eventId, id))) .limit(1); + // Both branches are idempotent, so a double-click (two concurrent toggles + // that both read "not RSVP'd") converges instead of throwing on the PK. if (existing) { - await db.delete(rsvps).where(and(eq(rsvps.userId, userId), eq(rsvps.eventId, eventId))); + await db.delete(rsvps).where(and(eq(rsvps.userId, userId), eq(rsvps.eventId, id))); } else { - await db.insert(rsvps).values({ userId, eventId }); + await db.insert(rsvps).values({ userId, eventId: id }).onConflictDoNothing(); } // Re-read the roster rather than counting: the count and the avatar stack are @@ -493,12 +264,12 @@ export async function toggleRsvp(eventId: string): Promise<{ }) .from(rsvps) .innerJoin(users, eq(rsvps.userId, users.id)) - .where(eq(rsvps.eventId, eventId)); + .where(eq(rsvps.eventId, id)); revalidatePath("/explore"); return { - rsvped: !existing, + rsvped: attendees.some((a) => a.id === userId), count: attendees.length, attendees, }; @@ -509,25 +280,26 @@ export async function toggleSave(eventId: string): Promise<{ saved: boolean }> { if (!session?.user?.id) throw new Error("Unauthorized"); const userId = session.user.id; + const id = parseInput(idSchema, eventId); - // If the event doesn't exist in the DB (e.g. demo/local-only event), no-op - const [eventRow] = await db.select().from(events).where(eq(events.id, eventId)).limit(1); - if (!eventRow) { + // Missing or not visible to this viewer: no-op. + if (!(await canViewEvent(id, userId))) { return { saved: false }; } const [existing] = await db - .select() + .select({ eventId: savedEvents.eventId }) .from(savedEvents) - .where(and(eq(savedEvents.userId, userId), eq(savedEvents.eventId, eventId))) + .where(and(eq(savedEvents.userId, userId), eq(savedEvents.eventId, id))) .limit(1); + // Idempotent either way — see toggleRsvp. if (existing) { await db .delete(savedEvents) - .where(and(eq(savedEvents.userId, userId), eq(savedEvents.eventId, eventId))); + .where(and(eq(savedEvents.userId, userId), eq(savedEvents.eventId, id))); } else { - await db.insert(savedEvents).values({ userId, eventId }); + await db.insert(savedEvents).values({ userId, eventId: id }).onConflictDoNothing(); } revalidatePath("/explore"); @@ -547,18 +319,27 @@ export interface EventDetail { locationName: string; orgId: string | null; orgName: string | null; + orgLogoUrl: string | null; creatorId: string; creatorName: string; flyerUrl: string | null; externalLink: string | null; isPublic: boolean; + /** 'manual', or where an imported event came from: 'myprincetonu' | 'listserv'. */ + source: string; + sourceUrl: string | null; + /** Room or free-text place, e.g. "Room 104". */ + locationDetail: string | null; tags: string[]; rsvpCount: number; attendees: { id: string; displayName: string; avatarUrl: string | null }[]; friendsAttending: { id: string; displayName: string; avatarUrl: string | null }[]; isRsvped: boolean; isSaved: boolean; + /** The viewer created this event (may delete it). */ isOwner: boolean; + /** The viewer may edit it: creator, or owner/officer of its org. */ + canEdit: boolean; } export async function getEvent(eventId: string): Promise { @@ -567,6 +348,9 @@ export async function getEvent(eventId: string): Promise { const userId = session.user.id; + // Not a well-formed id → not found (rather than a Postgres cast error). + if (!idSchema.safeParse(eventId).success) return null; + const [event] = await db .select({ id: events.id, @@ -578,68 +362,33 @@ export async function getEvent(eventId: string): Promise { locationName: campusLocations.name, orgId: events.orgId, orgName: organizations.name, + orgLogoUrl: organizations.logoUrl, creatorId: events.creatorId, creatorName: users.displayName, flyerUrl: events.flyerUrl, externalLink: events.externalLink, isPublic: events.isPublic, + source: events.source, + sourceUrl: events.sourceUrl, + locationDetail: events.locationDetail, }) .from(events) .leftJoin(campusLocations, eq(events.locationId, campusLocations.id)) .leftJoin(organizations, eq(events.orgId, organizations.id)) .innerJoin(users, eq(events.creatorId, users.id)) - .where(eq(events.id, eventId)) + // Drafts and private events 404 for anyone who isn't allowed to see them. + .where(and(eq(events.id, eventId), eventVisibleTo(userId))) .limit(1); if (!event) return null; - // Get tags - const tags = await db - .select({ tag: eventTags.tag }) - .from(eventTags) - .where(eq(eventTags.eventId, eventId)); - - // Get RSVP count + attendees - const attendees = await db - .select({ - id: users.id, - displayName: users.displayName, - avatarUrl: users.avatarUrl, - }) - .from(rsvps) - .innerJoin(users, eq(rsvps.userId, users.id)) - .where(eq(rsvps.eventId, eventId)); - - // Get friend IDs - const friendRows = await db - .select({ friendId: friendships.friendId }) - .from(friendships) - .where(and(eq(friendships.userId, userId), eq(friendships.status, "accepted"))); - const reverseFriendRows = await db - .select({ friendId: friendships.userId }) - .from(friendships) - .where(and(eq(friendships.friendId, userId), eq(friendships.status, "accepted"))); - const friendIdSet = new Set([ - ...friendRows.map((f) => f.friendId), - ...reverseFriendRows.map((f) => f.friendId), + const [friendIds, canEdit] = await Promise.all([ + loadFriendIds(userId), + event.creatorId === userId ? Promise.resolve(true) : canEditEvent(eventId, userId), ]); + const extra = await loadEventEnrichment([eventId], userId, friendIds); + const attendees = extra.attendees.get(eventId) ?? []; - const friendsAttending = attendees.filter((a) => friendIdSet.has(a.id)); - - // Check user RSVP + save - const [userRsvp] = await db - .select() - .from(rsvps) - .where(and(eq(rsvps.userId, userId), eq(rsvps.eventId, eventId))) - .limit(1); - - const [userSave] = await db - .select() - .from(savedEvents) - .where(and(eq(savedEvents.userId, userId), eq(savedEvents.eventId, eventId))) - .limit(1); - - // Similar events (same tags or same org) return { id: event.id, title: event.title, @@ -650,21 +399,32 @@ export async function getEvent(eventId: string): Promise { locationName: event.locationName ?? "TBD", orgId: event.orgId, orgName: event.orgName, + orgLogoUrl: event.orgLogoUrl ?? null, creatorId: event.creatorId, creatorName: event.creatorName, flyerUrl: event.flyerUrl, externalLink: event.externalLink, isPublic: event.isPublic, - tags: tags.map((t) => t.tag), + source: event.source, + sourceUrl: event.sourceUrl, + locationDetail: event.locationDetail, + tags: extra.tags.get(eventId) ?? [], rsvpCount: attendees.length, attendees, - friendsAttending, - isRsvped: !!userRsvp, - isSaved: !!userSave, + friendsAttending: extra.friends.get(eventId) ?? [], + isRsvped: extra.rsvpedByMe.has(eventId), + isSaved: extra.savedByMe.has(eventId), isOwner: event.creatorId === userId, + canEdit, }; } +const similarEventsSchema = z.object({ + eventId: idSchema, + tags: z.array(eventTagSchema).max(eventTagSchema.options.length), + orgId: idSchema.nullable(), +}); + export async function getSimilarEvents( eventId: string, tags: string[], @@ -674,22 +434,27 @@ export async function getSimilarEvents( if (!session?.user?.id) throw new Error("Unauthorized"); const userId = session.user.id; + const input = parseInput(similarEventsSchema, { eventId, tags, orgId }); - const conditions = [gt(events.datetime, new Date())]; + const conditions = [ + gt(events.datetime, new Date()), + eventDiscoverableBy(userId), + ne(events.id, input.eventId), + ]; // Events with matching tags or same org, excluding current event const tagFilter = - tags.length > 0 + input.tags.length > 0 ? inArray( events.id, db .select({ eventId: eventTags.eventId }) .from(eventTags) - .where(inArray(eventTags.tag, tags as (typeof eventTags.$inferSelect.tag)[])), + .where(inArray(eventTags.tag, input.tags)), ) : undefined; - const orgFilter = orgId ? eq(events.orgId, orgId) : undefined; + const orgFilter = input.orgId ? eq(events.orgId, input.orgId) : undefined; const matchFilter = tagFilter && orgFilter ? or(tagFilter, orgFilter) : (tagFilter ?? orgFilter); @@ -707,11 +472,13 @@ export async function getSimilarEvents( locationName: campusLocations.name, orgId: events.orgId, orgName: organizations.name, + orgLogoUrl: organizations.logoUrl, + locationDetail: events.locationDetail, }) .from(events) .leftJoin(campusLocations, eq(events.locationId, campusLocations.id)) .leftJoin(organizations, eq(events.orgId, organizations.id)) - .where(and(...conditions, sql`${events.id} != ${eventId}`)) + .where(and(...conditions)) .orderBy(events.datetime) .limit(4); @@ -730,9 +497,11 @@ export async function getSimilarEvents( description: event.description, orgId: event.orgId, orgName: event.orgName, + orgLogoUrl: event.orgLogoUrl ?? null, datetime: formatEventDateTime(event.datetime), rawDatetime: event.datetime.toISOString(), location: event.locationName ?? "TBD", + locationDetail: event.locationDetail ?? null, tags: extra.tags.get(event.id) ?? [], flyerUrl: event.flyerUrl, rsvpCount: attendees.length, @@ -744,7 +513,40 @@ export async function getSimilarEvents( }); } -type EventTagValue = typeof eventTags.$inferSelect.tag; +/** Fields shared by create and update. */ +const eventFieldsSchema = z.object({ + title: z.string().trim().min(1, { message: "Title is required" }).max(200), + description: z.string().trim().max(20_000), + datetime: dateInputSchema, + endDatetime: dateInputSchema.nullish(), + locationId: locationIdSchema, + tags: uniqueEnumArray(eventTagSchema).default([]), + externalLink: httpsUrlSchema, + isPublic: z.boolean().default(true), +}); + +type EventFields = z.output; + +function endNotBeforeStart(data: Pick) { + return !data.endDatetime || data.endDatetime.getTime() >= data.datetime.getTime(); +} +const END_BEFORE_START = { + message: "End time must be after the start time", + path: ["endDatetime"], +}; + +const createEventSchema = eventFieldsSchema + .extend({ + orgId: idSchema.nullish(), + flyerUrl: uploadedImageUrlSchema("event-flyers"), + coverPreset: z + .string() + .max(50) + .regex(/^[a-z0-9-]+$/) + .nullish(), + status: eventStatusSchema.default("published"), + }) + .refine(endNotBeforeStart, END_BEFORE_START); export async function createEvent(data: { title: string; @@ -764,14 +566,16 @@ export async function createEvent(data: { if (!session?.user?.id) throw new Error("Unauthorized"); const creatorId = session.user.id; + const input = parseInput(createEventSchema, data); + enforceRateLimit("createEvent", creatorId); - if (data.orgId) { + if (input.orgId) { const [membership] = await db .select({ orgId: orgMembers.orgId }) .from(orgMembers) .where( and( - eq(orgMembers.orgId, data.orgId), + eq(orgMembers.orgId, input.orgId), eq(orgMembers.userId, creatorId), inArray(orgMembers.role, ["owner", "officer"]), ), @@ -783,49 +587,51 @@ export async function createEvent(data: { } } - const [event] = await db - .insert(events) - .values({ - title: data.title, - description: data.description, - datetime: new Date(data.datetime), - endDatetime: data.endDatetime ? new Date(data.endDatetime) : null, - locationId: data.locationId, - orgId: data.orgId ?? null, - creatorId, - flyerUrl: data.flyerUrl ?? null, - coverPreset: data.coverPreset ?? null, - externalLink: data.externalLink ?? null, - isPublic: data.isPublic ?? true, - status: data.status ?? "published", - }) - .returning({ id: events.id }); - - if (!event) throw new Error("Failed to create event"); - - // Insert tags - if (data.tags.length > 0) { - await db.insert(eventTags).values( - data.tags.map((tag) => ({ - eventId: event.id, - tag: tag as EventTagValue, - })), - ); - } + // Event + tags commit together, so a failure can't leave an untagged event. + const eventId = await db.transaction(async (tx) => { + const [event] = await tx + .insert(events) + .values({ + title: input.title, + description: input.description, + datetime: input.datetime, + endDatetime: input.endDatetime ?? null, + locationId: input.locationId, + orgId: input.orgId ?? null, + creatorId, + flyerUrl: input.flyerUrl, + coverPreset: input.coverPreset ?? null, + externalLink: input.externalLink, + isPublic: input.isPublic, + status: input.status, + }) + .returning({ id: events.id }); - // Notify org followers about new event (exclude creator) — only for published events - if (data.orgId && (data.status ?? "published") === "published") { + if (!event) throw new Error("Failed to create event"); + + if (input.tags.length > 0) { + await tx + .insert(eventTags) + .values(input.tags.map((tag) => ({ eventId: event.id, tag }))) + .onConflictDoNothing(); + } + return event.id; + }); + + // Notify org followers about new event (exclude creator) — only for events + // followers can actually open: published AND public. + if (input.orgId && input.status === "published" && input.isPublic) { const followers = await db .select({ userId: orgFollowers.userId }) .from(orgFollowers) - .where(and(eq(orgFollowers.orgId, data.orgId), ne(orgFollowers.userId, creatorId))); + .where(and(eq(orgFollowers.orgId, input.orgId), ne(orgFollowers.userId, creatorId))); if (followers.length > 0) { await db.insert(notifications).values( followers.map((f) => ({ userId: f.userId, type: "org_new_event" as const, - payload: { eventId: event.id, eventTitle: data.title, orgId: data.orgId }, + payload: { eventId, eventTitle: input.title, orgId: input.orgId }, })), ); } @@ -834,9 +640,23 @@ export async function createEvent(data: { revalidatePath("/explore"); revalidatePath("/events"); - return { id: event.id }; + return { id: eventId }; } +const updateEventSchema = eventFieldsSchema + .extend({ + // Checked against the event's current flyer below: an unchanged flyer is + // always accepted (it may predate uploads, e.g. listserv-ingested events); + // a new one must be one of our uploads. + flyerUrl: z + .string() + .trim() + .max(2048) + .nullish() + .transform((v) => (v ? v : null)), + }) + .refine(endNotBeforeStart, END_BEFORE_START); + export async function updateEvent( eventId: string, data: { @@ -854,44 +674,56 @@ export async function updateEvent( const session = await auth(); if (!session?.user?.id) throw new Error("Unauthorized"); - // Verify ownership + const userId = session.user.id; + const id = parseInput(idSchema, eventId); + const input = parseInput(updateEventSchema, data); + + // Creator, or owner/officer of the event's org. const [event] = await db - .select({ creatorId: events.creatorId }) + .select({ flyerUrl: events.flyerUrl }) .from(events) - .where(eq(events.id, eventId)) + .where(and(eq(events.id, id), eventEditableBy(userId))) .limit(1); - if (!event || event.creatorId !== session.user.id) { + if (!event) { throw new Error("Not authorized to edit this event"); } - await db - .update(events) - .set({ - title: data.title, - description: data.description, - datetime: new Date(data.datetime), - endDatetime: data.endDatetime ? new Date(data.endDatetime) : null, - locationId: data.locationId, - flyerUrl: data.flyerUrl ?? null, - externalLink: data.externalLink ?? null, - isPublic: data.isPublic ?? true, - updatedAt: new Date(), - }) - .where(eq(events.id, eventId)); - - // Replace tags - await db.delete(eventTags).where(eq(eventTags.eventId, eventId)); - if (data.tags.length > 0) { - await db.insert(eventTags).values( - data.tags.map((tag) => ({ - eventId, - tag: tag as EventTagValue, - })), - ); + if ( + input.flyerUrl !== null && + input.flyerUrl !== event.flyerUrl && + !isOurImageUrl(input.flyerUrl, "event-flyers") + ) { + throw new Error("Invalid input — flyerUrl: Image must be uploaded through The Forum"); } - revalidatePath(`/events/${eventId}`); + await db.transaction(async (tx) => { + await tx + .update(events) + .set({ + title: input.title, + description: input.description, + datetime: input.datetime, + endDatetime: input.endDatetime ?? null, + locationId: input.locationId, + flyerUrl: input.flyerUrl, + externalLink: input.externalLink, + isPublic: input.isPublic, + updatedAt: new Date(), + }) + .where(eq(events.id, id)); + + // Replace tags + await tx.delete(eventTags).where(eq(eventTags.eventId, id)); + if (input.tags.length > 0) { + await tx + .insert(eventTags) + .values(input.tags.map((tag) => ({ eventId: id, tag }))) + .onConflictDoNothing(); + } + }); + + revalidatePath(`/events/${id}`); revalidatePath("/explore"); revalidatePath("/events"); } @@ -900,17 +732,19 @@ export async function deleteEvent(eventId: string): Promise { const session = await auth(); if (!session?.user?.id) throw new Error("Unauthorized"); + const id = parseInput(idSchema, eventId); + const [event] = await db .select({ creatorId: events.creatorId }) .from(events) - .where(eq(events.id, eventId)) + .where(eq(events.id, id)) .limit(1); if (!event || event.creatorId !== session.user.id) { throw new Error("Not authorized to delete this event"); } - await db.delete(events).where(eq(events.id, eventId)); + await db.delete(events).where(eq(events.id, id)); revalidatePath("/explore"); revalidatePath("/events"); @@ -937,6 +771,8 @@ export async function getMyEvents(): Promise<{ locationName: campusLocations.name, orgId: events.orgId, orgName: organizations.name, + orgLogoUrl: organizations.logoUrl, + locationDetail: events.locationDetail, }) .from(events) .leftJoin(campusLocations, eq(events.locationId, campusLocations.id)) @@ -955,12 +791,15 @@ export async function getMyEvents(): Promise<{ locationName: campusLocations.name, orgId: events.orgId, orgName: organizations.name, + orgLogoUrl: organizations.logoUrl, + locationDetail: events.locationDetail, }) .from(rsvps) .innerJoin(events, eq(rsvps.eventId, events.id)) .leftJoin(campusLocations, eq(events.locationId, campusLocations.id)) .leftJoin(organizations, eq(events.orgId, organizations.id)) - .where(eq(rsvps.userId, userId)) + // An RSVP'd event that has since been unpublished or made private drops out. + .where(and(eq(rsvps.userId, userId), eventVisibleTo(userId))) .orderBy(events.datetime); // Events the user saved @@ -974,12 +813,14 @@ export async function getMyEvents(): Promise<{ locationName: campusLocations.name, orgId: events.orgId, orgName: organizations.name, + orgLogoUrl: organizations.logoUrl, + locationDetail: events.locationDetail, }) .from(savedEvents) .innerJoin(events, eq(savedEvents.eventId, events.id)) .leftJoin(campusLocations, eq(events.locationId, campusLocations.id)) .leftJoin(organizations, eq(events.orgId, organizations.id)) - .where(eq(savedEvents.userId, userId)) + .where(and(eq(savedEvents.userId, userId), eventVisibleTo(userId))) .orderBy(events.datetime); /* @@ -1001,9 +842,11 @@ export async function getMyEvents(): Promise<{ description: e.description, orgId: e.orgId, orgName: e.orgName, + orgLogoUrl: e.orgLogoUrl ?? null, datetime: formatEventDateTime(e.datetime), rawDatetime: e.datetime.toISOString(), location: e.locationName ?? "TBD", + locationDetail: e.locationDetail ?? null, tags: extra.tags.get(e.id) ?? [], flyerUrl: e.flyerUrl, rsvpCount: attendees.length, @@ -1050,6 +893,8 @@ export async function getSavedEvents(): Promise { locationName: campusLocations.name, orgId: events.orgId, orgName: organizations.name, + orgLogoUrl: organizations.logoUrl, + locationDetail: events.locationDetail, }) .from(savedEvents) .innerJoin(events, eq(savedEvents.eventId, events.id)) @@ -1060,7 +905,9 @@ export async function getSavedEvents(): Promise { * the datetime bound a saved event from last week surfaced there — with * `formatRelativeDay` cheerfully announcing it was happening "yesterday". */ - .where(and(eq(savedEvents.userId, userId), gt(events.datetime, new Date()))) + .where( + and(eq(savedEvents.userId, userId), gt(events.datetime, new Date()), eventVisibleTo(userId)), + ) .orderBy(events.datetime) .limit(5); @@ -1079,9 +926,11 @@ export async function getSavedEvents(): Promise { description: event.description, orgId: event.orgId, orgName: event.orgName, + orgLogoUrl: event.orgLogoUrl ?? null, datetime: formatEventDateTime(event.datetime), rawDatetime: event.datetime.toISOString(), location: event.locationName ?? "TBD", + locationDetail: event.locationDetail ?? null, tags: extra.tags.get(event.id) ?? [], flyerUrl: event.flyerUrl, rsvpCount: attendees.length, @@ -1104,20 +953,7 @@ export async function getFriendsEvents(): Promise { const userId = session.user.id; - // Get friend IDs (bidirectional) - const friendRows = await db - .select({ friendId: friendships.friendId }) - .from(friendships) - .where(and(eq(friendships.userId, userId), eq(friendships.status, "accepted"))); - const reverseFriendRows = await db - .select({ friendId: friendships.userId }) - .from(friendships) - .where(and(eq(friendships.friendId, userId), eq(friendships.status, "accepted"))); - const friendIds = [ - ...friendRows.map((f) => f.friendId), - ...reverseFriendRows.map((f) => f.friendId), - ]; - + const friendIds = await loadFriendIds(userId); if (friendIds.length === 0) return []; // Find upcoming events where friends have RSVP'd, with friend count @@ -1131,13 +967,22 @@ export async function getFriendsEvents(): Promise { locationName: campusLocations.name, orgId: events.orgId, orgName: organizations.name, + orgLogoUrl: organizations.logoUrl, + locationDetail: events.locationDetail, friendCount: sql`count(distinct ${rsvps.userId})::int`.as("friend_count"), }) .from(rsvps) .innerJoin(events, eq(rsvps.eventId, events.id)) .leftJoin(campusLocations, eq(events.locationId, campusLocations.id)) .leftJoin(organizations, eq(events.orgId, organizations.id)) - .where(and(inArray(rsvps.userId, friendIds), gt(events.datetime, new Date()))) + // A friend's RSVP to a draft/private event must not leak it to the viewer. + .where( + and( + inArray(rsvps.userId, friendIds), + gt(events.datetime, new Date()), + eventDiscoverableBy(userId), + ), + ) .groupBy( events.id, events.title, @@ -1164,9 +1009,11 @@ export async function getFriendsEvents(): Promise { description: event.description, orgId: event.orgId, orgName: event.orgName, + orgLogoUrl: event.orgLogoUrl ?? null, datetime: formatEventDateTime(event.datetime), rawDatetime: event.datetime.toISOString(), location: event.locationName ?? "TBD", + locationDetail: event.locationDetail ?? null, tags: extra.tags.get(event.id) ?? [], flyerUrl: event.flyerUrl, rsvpCount: attendees.length, diff --git a/apps/web/src/actions/friends.ts b/apps/web/src/actions/friends.ts index 4f3dc19..bf92859 100644 --- a/apps/web/src/actions/friends.ts +++ b/apps/web/src/actions/friends.ts @@ -1,18 +1,12 @@ "use server"; -import { - and, - db, - eq, - friendships, - ilike, - notifications, - or, - sql, - users, -} from "@the-forum/database"; +import { and, db, eq, friendships, ilike, ne, notifications, or, users } from "@the-forum/database"; import { revalidatePath } from "next/cache"; +import { z } from "zod"; import { auth } from "~/auth"; +import { enforceRateLimit } from "~/lib/rate-limit"; +import { containsPattern } from "~/lib/sql-helpers"; +import { idSchema, parseInput } from "~/lib/validation"; export interface FriendProfile { id: string; @@ -23,6 +17,12 @@ export interface FriendProfile { major: string | null; } +/** Just what the search results UI renders — nothing more leaves the server. */ +export type UserSearchResult = Pick< + FriendProfile, + "id" | "displayName" | "netId" | "avatarUrl" | "classYear" +>; + export interface FriendRequest { id: string; displayName: string; @@ -31,30 +31,38 @@ export interface FriendRequest { createdAt: Date; } -export async function searchUsers(query: string): Promise { +const SEARCH_MIN_LENGTH = 2; +const SEARCH_MAX_RESULTS = 20; +const searchQuerySchema = z.string().max(200); + +export async function searchUsers(query: string): Promise { const session = await auth(); if (!session?.user?.id) throw new Error("Unauthorized"); - if (!query.trim()) return []; - const results = await db + const q = parseInput(searchQuerySchema, query).trim().slice(0, 100); + // Too short to be selective — would page through the whole user table. + if (q.length < SEARCH_MIN_LENGTH) return []; + + enforceRateLimit("searchUsers", session.user.id); + + const pattern = containsPattern(q); + return db .select({ id: users.id, displayName: users.displayName, netId: users.netId, avatarUrl: users.avatarUrl, classYear: users.classYear, - major: users.major, }) .from(users) .where( and( - or(ilike(users.displayName, `%${query}%`), ilike(users.netId, `%${query}%`)), - sql`${users.id} != ${session.user.id}`, + or(ilike(users.displayName, pattern), ilike(users.netId, pattern)), + ne(users.id, session.user.id), ), ) - .limit(20); - - return results; + .orderBy(users.displayName) + .limit(SEARCH_MAX_RESULTS); } export async function getFriends(): Promise { @@ -137,11 +145,12 @@ export async function getFriendshipStatus( if (!session?.user?.id) throw new Error("Unauthorized"); const userId = session.user.id; + const otherId = parseInput(idSchema, otherUserId); const [sent] = await db .select({ status: friendships.status }) .from(friendships) - .where(and(eq(friendships.userId, userId), eq(friendships.friendId, otherUserId))) + .where(and(eq(friendships.userId, userId), eq(friendships.friendId, otherId))) .limit(1); if (sent) { @@ -151,7 +160,7 @@ export async function getFriendshipStatus( const [received] = await db .select({ status: friendships.status }) .from(friendships) - .where(and(eq(friendships.userId, otherUserId), eq(friendships.friendId, userId))) + .where(and(eq(friendships.userId, otherId), eq(friendships.friendId, userId))) .limit(1); if (received) { @@ -161,42 +170,75 @@ export async function getFriendshipStatus( return "none"; } +/** + * Idempotent: repeating a request (double-click, retry) is a no-op, and + * requesting someone who already requested you accepts theirs. + */ export async function sendFriendRequest(friendId: string): Promise { const session = await auth(); if (!session?.user?.id) throw new Error("Unauthorized"); const userId = session.user.id; - if (userId === friendId) throw new Error("Cannot send friend request to yourself"); + const targetId = parseInput(idSchema, friendId); + if (userId === targetId) throw new Error("Cannot send friend request to yourself"); // Check if friendship already exists in either direction const [existing] = await db - .select() + .select({ userId: friendships.userId, status: friendships.status }) .from(friendships) .where( or( - and(eq(friendships.userId, userId), eq(friendships.friendId, friendId)), - and(eq(friendships.userId, friendId), eq(friendships.friendId, userId)), + and(eq(friendships.userId, userId), eq(friendships.friendId, targetId)), + and(eq(friendships.userId, targetId), eq(friendships.friendId, userId)), ), ) .limit(1); - if (existing) throw new Error("Friend request already exists"); + if (existing) { + // They already asked us: treat this as accepting. + if (existing.userId === targetId && existing.status === "pending") { + await db + .update(friendships) + .set({ status: "accepted" }) + .where( + and( + eq(friendships.userId, targetId), + eq(friendships.friendId, userId), + eq(friendships.status, "pending"), + ), + ); + revalidatePath("/friends"); + } + return; + } - await db.insert(friendships).values({ - userId, - friendId, - status: "pending", - }); + enforceRateLimit("friendRequest", userId); - // Create notification for the recipient - await db.insert(notifications).values({ - userId: friendId, - type: "friend_request", - payload: { - fromUserId: userId, - fromDisplayName: session.user.name ?? "Someone", - }, - }); + const [target] = await db + .select({ id: users.id }) + .from(users) + .where(eq(users.id, targetId)) + .limit(1); + if (!target) throw new Error("User not found"); + + // A concurrent duplicate hits the primary key and inserts nothing — and then + // sends no second notification. + const inserted = await db + .insert(friendships) + .values({ userId, friendId: targetId, status: "pending" }) + .onConflictDoNothing() + .returning({ userId: friendships.userId }); + + if (inserted.length > 0) { + await db.insert(notifications).values({ + userId: targetId, + type: "friend_request", + payload: { + fromUserId: userId, + fromDisplayName: session.user.name ?? "Someone", + }, + }); + } revalidatePath("/friends"); } @@ -205,12 +247,14 @@ export async function acceptFriendRequest(fromUserId: string): Promise { const session = await auth(); if (!session?.user?.id) throw new Error("Unauthorized"); + const fromId = parseInput(idSchema, fromUserId); + await db .update(friendships) .set({ status: "accepted" }) .where( and( - eq(friendships.userId, fromUserId), + eq(friendships.userId, fromId), eq(friendships.friendId, session.user.id), eq(friendships.status, "pending"), ), @@ -223,11 +267,13 @@ export async function declineFriendRequest(fromUserId: string): Promise { const session = await auth(); if (!session?.user?.id) throw new Error("Unauthorized"); + const fromId = parseInput(idSchema, fromUserId); + await db .delete(friendships) .where( and( - eq(friendships.userId, fromUserId), + eq(friendships.userId, fromId), eq(friendships.friendId, session.user.id), eq(friendships.status, "pending"), ), @@ -241,13 +287,14 @@ export async function removeFriend(friendId: string): Promise { if (!session?.user?.id) throw new Error("Unauthorized"); const userId = session.user.id; + const otherId = parseInput(idSchema, friendId); await db .delete(friendships) .where( or( - and(eq(friendships.userId, userId), eq(friendships.friendId, friendId)), - and(eq(friendships.userId, friendId), eq(friendships.friendId, userId)), + and(eq(friendships.userId, userId), eq(friendships.friendId, otherId)), + and(eq(friendships.userId, otherId), eq(friendships.friendId, userId)), ), ); diff --git a/apps/web/src/actions/interactions.ts b/apps/web/src/actions/interactions.ts index 75a0328..2ad86a4 100644 --- a/apps/web/src/actions/interactions.ts +++ b/apps/web/src/actions/interactions.ts @@ -1,10 +1,13 @@ "use server"; import { db, interactions } from "@the-forum/database"; +import { z } from "zod"; import { auth } from "~/auth"; +import { checkRateLimit } from "~/lib/rate-limit"; +import { idSchema, interactionTypeSchema, itemTypeSchema } from "~/lib/validation"; /** Interaction weights — maps type to implicit feedback score */ -const INTERACTION_WEIGHTS: Record = { +const INTERACTION_WEIGHTS: Record, number> = { view: 1.0, click: 2.0, share: 2.0, @@ -13,9 +16,26 @@ const INTERACTION_WEIGHTS: Record = { hide: -1.0, }; +/** Serialized metadata cap. Callers send small context (source, position). */ +const MAX_METADATA_BYTES = 1024; + +const interactionSchema = z.object({ + itemId: idSchema, + itemType: itemTypeSchema.default("event"), + interactionType: interactionTypeSchema, + metadata: z + .record(z.string().max(64), z.unknown()) + .optional() + .refine( + (m) => + m === undefined || new TextEncoder().encode(JSON.stringify(m)).length <= MAX_METADATA_BYTES, + "metadata too large", + ), +}); + /** - * Log an implicit user interaction. Fire-and-forget — failures are silently - * dropped so logging never breaks the UX. + * Log an implicit user interaction. Fire-and-forget — invalid, oversized or + * rate-limited calls are dropped silently so logging never breaks the UX. */ export async function logInteraction(data: { itemId: string; @@ -27,13 +47,18 @@ export async function logInteraction(data: { const session = await auth(); if (!session?.user?.id) return; // not logged in — skip silently + const parsed = interactionSchema.safeParse(data); + if (!parsed.success) return; + if (!checkRateLimit("interaction", session.user.id).ok) return; + + const input = parsed.data; await db.insert(interactions).values({ userId: session.user.id, - itemId: data.itemId, - itemType: data.itemType ?? "event", - interactionType: data.interactionType, - interactionValue: INTERACTION_WEIGHTS[data.interactionType] ?? 1.0, - metadata: data.metadata ?? null, + itemId: input.itemId, + itemType: input.itemType, + interactionType: input.interactionType, + interactionValue: INTERACTION_WEIGHTS[input.interactionType], + metadata: input.metadata ?? null, }); } catch { // Silently drop — logging should never break the app diff --git a/apps/web/src/actions/map.ts b/apps/web/src/actions/map.ts index 04044a2..040cf97 100644 --- a/apps/web/src/actions/map.ts +++ b/apps/web/src/actions/map.ts @@ -7,15 +7,21 @@ import { db, eq, eventTags, - friendships, gte, inArray, lt, + ne, + or, organizations, rsvps, users, } from "@the-forum/database"; +import { z } from "zod"; import { auth } from "~/auth"; +import { formatTime } from "~/lib/date-format"; +import { eventDiscoverableBy } from "~/lib/event-visibility"; +import { loadFriendIds } from "~/lib/social-graph"; +import { dateInputSchema, parseInput } from "~/lib/validation"; export interface MapEvent { id: string; @@ -33,6 +39,11 @@ export interface MapEvent { friendsAttending: { id: string; displayName: string; avatarUrl: string | null }[]; } +const mapEventsSchema = z.object({ + from: dateInputSchema.optional(), + days: z.number().int().min(1).max(31).default(7), +}); + /** Fetch events for a date range (defaults to next 7 days). */ export async function getMapEvents(opts?: { from?: string; @@ -42,25 +53,15 @@ export async function getMapEvents(opts?: { if (!session?.user?.id) throw new Error("Unauthorized"); const userId = session.user.id; + const input = parseInput(mapEventsSchema, opts ?? {}); - // Accepted friendships are stored one-directional, so both columns are read. - const [outgoing, incoming] = await Promise.all([ - db - .select({ friendId: friendships.friendId }) - .from(friendships) - .where(and(eq(friendships.userId, userId), eq(friendships.status, "accepted"))), - db - .select({ friendId: friendships.userId }) - .from(friendships) - .where(and(eq(friendships.friendId, userId), eq(friendships.status, "accepted"))), - ]); - const friendIds = [...outgoing, ...incoming].map((r) => r.friendId); + const friendIds = await loadFriendIds(userId); - const startDate = opts?.from ? new Date(opts.from) : new Date(); + const startDate = input.from ?? new Date(); startDate.setHours(0, 0, 0, 0); const endDate = new Date(startDate); - endDate.setDate(endDate.getDate() + (opts?.days ?? 7)); + endDate.setDate(endDate.getDate() + input.days); endDate.setHours(23, 59, 59, 999); const results = await db @@ -78,7 +79,17 @@ export async function getMapEvents(opts?: { .from(events) .innerJoin(campusLocations, eq(events.locationId, campusLocations.id)) .leftJoin(organizations, eq(events.orgId, organizations.id)) - .where(and(gte(events.datetime, startDate), lt(events.datetime, endDate))) + .where( + and( + gte(events.datetime, startDate), + lt(events.datetime, endDate), + // Published AND visible to this viewer — see ~/lib/event-visibility.ts. + eventDiscoverableBy(userId), + // The "other" placeholder location sits at (0, 0) — off the coast of + // Africa. Events there have no real coordinates, so keep them off the map. + or(ne(campusLocations.latitude, 0), ne(campusLocations.longitude, 0)), + ), + ) .orderBy(events.datetime); /* @@ -129,10 +140,7 @@ export async function getMapEvents(opts?: { locationId: r.locationId ?? "", locationName: r.locationName ?? "TBD", rawDatetime: r.datetime.toISOString(), - datetime: r.datetime.toLocaleTimeString("en-US", { - hour: "numeric", - minute: "2-digit", - }), + datetime: formatTime(r.datetime), tags: tagsByEvent.get(r.id) ?? [], })); } diff --git a/apps/web/src/actions/notifications.ts b/apps/web/src/actions/notifications.ts index 305ac92..31aa7dd 100644 --- a/apps/web/src/actions/notifications.ts +++ b/apps/web/src/actions/notifications.ts @@ -1,7 +1,23 @@ "use server"; -import { events, and, db, desc, eq, gte, lt, notifications, rsvps, sql } from "@the-forum/database"; +import { + events, + and, + db, + desc, + eq, + exists, + gte, + inArray, + lt, + notifications, + or, + rsvps, + sql, +} from "@the-forum/database"; import { auth } from "~/auth"; +import { eventVisibleTo } from "~/lib/event-visibility"; +import { idSchema, parseInput } from "~/lib/validation"; export interface NotificationItem { id: string; @@ -11,6 +27,31 @@ export interface NotificationItem { createdAt: string; } +/** + * WHERE fragment: the notification either references no event, or references + * an event the viewer can still see. An org_new_event notification for an + * event that was later unpublished, made private or deleted is hidden rather + * than leaking its title. + */ +function referencedEventStillVisible(userId: string) { + return or( + sql`${notifications.payload}->>'eventId' IS NULL`, + exists( + db + .select({ one: sql`1` }) + .from(events) + .where( + and( + // Cast (guarded, so a malformed payload can't error) rather than + // comparing as text, so the lookup can use the events primary key. + sql`${events.id} = (CASE WHEN ${notifications.payload}->>'eventId' ~* '^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$' THEN (${notifications.payload}->>'eventId')::uuid END)`, + eventVisibleTo(userId), + ), + ), + ), + ); +} + export async function getNotifications(): Promise<{ items: NotificationItem[]; unreadCount: number; @@ -20,7 +61,7 @@ export async function getNotifications(): Promise<{ const userId = session.user.id; - // On-demand: generate event reminders for RSVP'd events in next 24h + // On-demand: generate event reminders for RSVP'd, still-visible events in the next 24h. const now = new Date(); const in24h = new Date(now.getTime() + 24 * 60 * 60 * 1000); @@ -28,42 +69,58 @@ export async function getNotifications(): Promise<{ .select({ eventId: events.id, title: events.title }) .from(rsvps) .innerJoin(events, eq(rsvps.eventId, events.id)) - .where(and(eq(rsvps.userId, userId), gte(events.datetime, now), lt(events.datetime, in24h))); + .where( + and( + eq(rsvps.userId, userId), + gte(events.datetime, now), + lt(events.datetime, in24h), + eventVisibleTo(userId), + ), + ); - for (const r of upcomingRsvps) { - // Check if reminder already exists for this user+event pair - const [existing] = await db - .select({ id: notifications.id }) + if (upcomingRsvps.length > 0) { + // One lookup for every existing reminder instead of one per event. + const existing = await db + .select({ eventId: sql`${notifications.payload}->>'eventId'` }) .from(notifications) .where( and( eq(notifications.userId, userId), eq(notifications.type, "event_reminder"), - sql`${notifications.payload}->>'eventId' = ${r.eventId}`, + inArray( + sql`${notifications.payload}->>'eventId'`, + upcomingRsvps.map((r) => r.eventId), + ), ), - ) - .limit(1); - - if (!existing) { - await db.insert(notifications).values({ - userId, - type: "event_reminder", - payload: { eventId: r.eventId, eventTitle: r.title }, - }); + ); + const alreadyReminded = new Set(existing.map((r) => r.eventId)); + const missing = upcomingRsvps.filter((r) => !alreadyReminded.has(r.eventId)); + + if (missing.length > 0) { + await db.insert(notifications).values( + missing.map((r) => ({ + userId, + type: "event_reminder" as const, + payload: { eventId: r.eventId, eventTitle: r.title }, + })), + ); } } - const items = await db - .select() - .from(notifications) - .where(eq(notifications.userId, userId)) - .orderBy(desc(notifications.createdAt)) - .limit(20); + const visibleToMe = and(eq(notifications.userId, userId), referencedEventStillVisible(userId)); - const [countResult] = await db - .select({ count: sql`count(*)::int` }) - .from(notifications) - .where(and(eq(notifications.userId, userId), eq(notifications.read, false))); + const [items, [countResult]] = await Promise.all([ + db + .select() + .from(notifications) + .where(visibleToMe) + .orderBy(desc(notifications.createdAt)) + .limit(20), + db + .select({ count: sql`count(*)::int` }) + .from(notifications) + .where(and(visibleToMe, eq(notifications.read, false))), + ]); return { items: items.map((n) => ({ @@ -81,10 +138,12 @@ export async function markNotificationRead(notificationId: string): Promise { diff --git a/apps/web/src/actions/orgs.ts b/apps/web/src/actions/orgs.ts index e3015f7..7fbf64b 100644 --- a/apps/web/src/actions/orgs.ts +++ b/apps/web/src/actions/orgs.ts @@ -19,8 +19,15 @@ import { users, } from "@the-forum/database"; import { revalidatePath } from "next/cache"; +import { z } from "zod"; import { auth } from "~/auth"; import { formatEventDateTime } from "~/lib/date-format"; +import { eventVisibleTo } from "~/lib/event-visibility"; +import { enforceRateLimit } from "~/lib/rate-limit"; +import { uploadedImageUrlSchema } from "~/lib/s3"; +import { loadFriendIds } from "~/lib/social-graph"; +import { containsPattern } from "~/lib/sql-helpers"; +import { idSchema, orgCategorySchema, parseInput } from "~/lib/validation"; export interface OrgListItem { id: string; @@ -32,22 +39,38 @@ export interface OrgListItem { isFollowing: boolean; } +export interface OrgPerson { + id: string; + displayName: string; + avatarUrl: string | null; +} + export interface OrgDetail { id: string; name: string; description: string | null; logoUrl: string | null; category: string; - creatorId: string; + creatorId: string | null; + /** 'myprincetonu' for official groups imported via InboxEngine, 'manual' for Forum-created. */ + source: string; + acronym: string | null; + tagline: string | null; + groupType: string | null; + /** The group's MyPrincetonU page. */ + groupUrl: string | null; + website: string | null; + contactEmail: string | null; + socials: Record; + /** Membership count reported by MyPrincetonU. */ + memberCount: number | null; followerCount: number; isFollowing: boolean; isOwner: boolean; - members: { - id: string; - displayName: string; - avatarUrl: string | null; - role: string; - }[]; + /** The viewer's friends who follow this org (social proof; never the full follower list). */ + friendsFollowing: OrgPerson[]; + /** Forum officers, shown only for Forum-created orgs (official rosters live on MyPrincetonU). */ + team: (OrgPerson & { role: string })[]; upcomingEvents: { id: string; title: string; @@ -55,10 +78,16 @@ export interface OrgDetail { locationName: string; flyerUrl: string | null; tags: string[]; + /** Only owners/officers (and creators) ever receive drafts or private events here. */ + status: "draft" | "published"; + isPublic: boolean; }[]; } -type OrgCategoryValue = typeof organizations.$inferSelect.category; +const getOrgsSchema = z.object({ + search: z.string().trim().max(100).optional(), + category: z.union([orgCategorySchema, z.literal("")]).optional(), +}); export async function getOrgs(params?: { search?: string; @@ -68,19 +97,18 @@ export async function getOrgs(params?: { if (!session?.user?.id) throw new Error("Unauthorized"); const userId = session.user.id; + const input = parseInput(getOrgsSchema, params ?? {}); const conditions = []; - if (params?.search) { + if (input.search) { + const pattern = containsPattern(input.search); conditions.push( - or( - ilike(organizations.name, `%${params.search}%`), - ilike(organizations.description, `%${params.search}%`), - ), + or(ilike(organizations.name, pattern), ilike(organizations.description, pattern)), ); } - if (params?.category) { - conditions.push(eq(organizations.category, params.category as OrgCategoryValue)); + if (input.category) { + conditions.push(eq(organizations.category, input.category)); } const orgs = await db @@ -95,27 +123,30 @@ export async function getOrgs(params?: { .where(conditions.length > 0 ? and(...conditions) : undefined) .orderBy(organizations.name); - // Enrich with follower counts + follow status - return Promise.all( - orgs.map(async (org) => { - const [countResult] = await db - .select({ count: sql`count(*)::int` }) - .from(orgFollowers) - .where(eq(orgFollowers.orgId, org.id)); - - const [following] = await db - .select() - .from(orgFollowers) - .where(and(eq(orgFollowers.orgId, org.id), eq(orgFollowers.userId, userId))) - .limit(1); - - return { - ...org, - followerCount: countResult?.count ?? 0, - isFollowing: !!following, - }; - }), - ); + if (orgs.length === 0) return []; + + // Follower counts + the viewer's follow state: two grouped queries for the + // whole list, instead of two queries per org. + const orgIds = orgs.map((o) => o.id); + const [countRows, myFollows] = await Promise.all([ + db + .select({ orgId: orgFollowers.orgId, count: sql`count(*)::int` }) + .from(orgFollowers) + .where(inArray(orgFollowers.orgId, orgIds)) + .groupBy(orgFollowers.orgId), + db + .select({ orgId: orgFollowers.orgId }) + .from(orgFollowers) + .where(and(eq(orgFollowers.userId, userId), inArray(orgFollowers.orgId, orgIds))), + ]); + const countByOrg = new Map(countRows.map((r) => [r.orgId, r.count])); + const followed = new Set(myFollows.map((r) => r.orgId)); + + return orgs.map((org) => ({ + ...org, + followerCount: countByOrg.get(org.id) ?? 0, + isFollowing: followed.has(org.id), + })); } export async function getOrg(orgId: string): Promise { @@ -124,34 +155,46 @@ export async function getOrg(orgId: string): Promise { const userId = session.user.id; + // Not a well-formed id → not found (rather than a Postgres cast error). + if (!idSchema.safeParse(orgId).success) return null; + const [org] = await db.select().from(organizations).where(eq(organizations.id, orgId)).limit(1); if (!org) return null; - // Follower count - const [countResult] = await db - .select({ count: sql`count(*)::int` }) - .from(orgFollowers) - .where(eq(orgFollowers.orgId, orgId)); - - // Is following - const [following] = await db - .select() - .from(orgFollowers) - .where(and(eq(orgFollowers.orgId, orgId), eq(orgFollowers.userId, userId))) - .limit(1); - - // Members - const members = await db - .select({ - id: users.id, - displayName: users.displayName, - avatarUrl: users.avatarUrl, - role: orgMembers.role, - }) - .from(orgMembers) - .innerJoin(users, eq(orgMembers.userId, users.id)) - .where(eq(orgMembers.orgId, orgId)); + const friendIds = await loadFriendIds(userId); + const [[countResult], [following], friendsFollowing, team] = await Promise.all([ + db + .select({ count: sql`count(*)::int` }) + .from(orgFollowers) + .where(eq(orgFollowers.orgId, orgId)), + db + .select({ userId: orgFollowers.userId }) + .from(orgFollowers) + .where(and(eq(orgFollowers.orgId, orgId), eq(orgFollowers.userId, userId))) + .limit(1), + friendIds.length === 0 + ? Promise.resolve([] as OrgPerson[]) + : db + .select({ id: users.id, displayName: users.displayName, avatarUrl: users.avatarUrl }) + .from(orgFollowers) + .innerJoin(users, eq(orgFollowers.userId, users.id)) + .where(and(eq(orgFollowers.orgId, orgId), inArray(orgFollowers.userId, friendIds))) + .orderBy(users.displayName) + .limit(12), + org.source === "myprincetonu" + ? Promise.resolve([] as (OrgPerson & { role: string })[]) + : db + .select({ + id: users.id, + displayName: users.displayName, + avatarUrl: users.avatarUrl, + role: orgMembers.role, + }) + .from(orgMembers) + .innerJoin(users, eq(orgMembers.userId, users.id)) + .where(and(eq(orgMembers.orgId, orgId), inArray(orgMembers.role, ["owner", "officer"]))), + ]); // Upcoming events const upcomingEventsRaw = await db @@ -161,29 +204,52 @@ export async function getOrg(orgId: string): Promise { datetime: events.datetime, locationName: campusLocations.name, flyerUrl: events.flyerUrl, + status: events.status, + isPublic: events.isPublic, }) .from(events) .leftJoin(campusLocations, eq(events.locationId, campusLocations.id)) - .where(and(eq(events.orgId, orgId), sql`${events.datetime} > now()`)) + .where( + and( + eq(events.orgId, orgId), + sql`${events.datetime} > now()`, + // Drafts/private events only for the org's owners/officers (and creators). + eventVisibleTo(userId), + ), + ) .orderBy(events.datetime) - .limit(10); - - const upcomingEvents = await Promise.all( - upcomingEventsRaw.map(async (e) => { - const tags = await db - .select({ tag: eventTags.tag }) - .from(eventTags) - .where(eq(eventTags.eventId, e.id)); - return { - id: e.id, - title: e.title, - datetime: formatEventDateTime(e.datetime), - locationName: e.locationName ?? "TBD", - flyerUrl: e.flyerUrl, - tags: tags.map((t) => t.tag), - }; - }), - ); + .limit(20); + + // One query for every event's tags rather than one per event. + const tagRows = + upcomingEventsRaw.length === 0 + ? [] + : await db + .select({ eventId: eventTags.eventId, tag: eventTags.tag }) + .from(eventTags) + .where( + inArray( + eventTags.eventId, + upcomingEventsRaw.map((e) => e.id), + ), + ); + const tagsByEvent = new Map(); + for (const row of tagRows) { + const list = tagsByEvent.get(row.eventId); + if (list) list.push(row.tag); + else tagsByEvent.set(row.eventId, [row.tag]); + } + + const upcomingEvents = upcomingEventsRaw.map((e) => ({ + id: e.id, + title: e.title, + datetime: formatEventDateTime(e.datetime), + locationName: e.locationName ?? "TBD", + flyerUrl: e.flyerUrl, + tags: tagsByEvent.get(e.id) ?? [], + status: e.status, + isPublic: e.isPublic, + })); return { id: org.id, @@ -192,14 +258,31 @@ export async function getOrg(orgId: string): Promise { logoUrl: org.logoUrl, category: org.category, creatorId: org.creatorId, + source: org.source, + acronym: org.acronym, + tagline: org.tagline, + groupType: org.groupType, + groupUrl: org.groupUrl, + website: org.website, + contactEmail: org.contactEmail, + socials: org.socials ?? {}, + memberCount: org.memberCount, followerCount: countResult?.count ?? 0, isFollowing: !!following, - isOwner: org.creatorId === userId, - members, + isOwner: !!org.creatorId && org.creatorId === userId, + friendsFollowing, + team, upcomingEvents, }; } +const createOrgSchema = z.object({ + name: z.string().trim().min(1, { message: "Name is required" }).max(120), + description: z.string().trim().max(5_000), + category: orgCategorySchema, + logoUrl: uploadedImageUrlSchema("org-logos"), +}); + export async function createOrg(data: { name: string; description: string; @@ -209,39 +292,44 @@ export async function createOrg(data: { const session = await auth(); if (!session?.user?.id) throw new Error("Unauthorized"); + const userId = session.user.id; + const input = parseInput(createOrgSchema, data); + enforceRateLimit("createOrg", userId); + // Verify user is an org leader const [user] = await db .select({ isOrgLeader: users.isOrgLeader }) .from(users) - .where(eq(users.id, session.user.id)) + .where(eq(users.id, userId)) .limit(1); if (!user?.isOrgLeader) { throw new Error("Only org leaders can create organizations"); } - const [org] = await db - .insert(organizations) - .values({ - name: data.name, - description: data.description, - category: data.category as OrgCategoryValue, - logoUrl: data.logoUrl ?? null, - creatorId: session.user.id, - }) - .returning({ id: organizations.id }); - - if (!org) throw new Error("Failed to create organization"); - - // Add creator as owner member - await db.insert(orgMembers).values({ - orgId: org.id, - userId: session.user.id, - role: "owner", + // Org + owner membership commit together; a name clash (unique) is reported + // as a readable error rather than a raw constraint violation. + const orgId = await db.transaction(async (tx) => { + const [org] = await tx + .insert(organizations) + .values({ + name: input.name, + description: input.description, + category: input.category, + logoUrl: input.logoUrl, + creatorId: userId, + }) + .onConflictDoNothing({ target: organizations.name }) + .returning({ id: organizations.id }); + + if (!org) throw new Error("An organization with that name already exists"); + + await tx.insert(orgMembers).values({ orgId: org.id, userId, role: "owner" }); + return org.id; }); revalidatePath("/orgs"); - return { id: org.id }; + return { id: orgId }; } export async function toggleFollowOrg(orgId: string): Promise<{ following: boolean }> { @@ -249,75 +337,93 @@ export async function toggleFollowOrg(orgId: string): Promise<{ following: boole if (!session?.user?.id) throw new Error("Unauthorized"); const userId = session.user.id; + const id = parseInput(idSchema, orgId); const [existing] = await db - .select() + .select({ orgId: orgFollowers.orgId }) .from(orgFollowers) - .where(and(eq(orgFollowers.orgId, orgId), eq(orgFollowers.userId, userId))) + .where(and(eq(orgFollowers.orgId, id), eq(orgFollowers.userId, userId))) .limit(1); + // Idempotent either way, so a double-click can't throw on the primary key. if (existing) { await db .delete(orgFollowers) - .where(and(eq(orgFollowers.orgId, orgId), eq(orgFollowers.userId, userId))); + .where(and(eq(orgFollowers.orgId, id), eq(orgFollowers.userId, userId))); } else { - await db.insert(orgFollowers).values({ orgId, userId }); + await db.insert(orgFollowers).values({ orgId: id, userId }).onConflictDoNothing(); } - revalidatePath(`/orgs/${orgId}`); + revalidatePath(`/orgs/${id}`); revalidatePath("/orgs"); return { following: !existing }; } -export async function addOfficer(orgId: string, userId: string): Promise { - const session = await auth(); - if (!session?.user?.id) throw new Error("Unauthorized"); +const officerSchema = z.object({ orgId: idSchema, userId: idSchema }); - // Verify caller is org owner +async function assertOrgCreator(orgId: string, callerId: string, message: string) { const [org] = await db .select({ creatorId: organizations.creatorId }) .from(organizations) .where(eq(organizations.id, orgId)) .limit(1); - if (!org || org.creatorId !== session.user.id) { - throw new Error("Only the org owner can add officers"); - } + if (!org || org.creatorId !== callerId) throw new Error(message); +} - // Check if already a member - const [existing] = await db - .select() - .from(orgMembers) - .where(and(eq(orgMembers.orgId, orgId), eq(orgMembers.userId, userId))) +/** Idempotent: adding an existing officer is a no-op; a plain member is promoted. */ +export async function addOfficer(orgId: string, userId: string): Promise { + const session = await auth(); + if (!session?.user?.id) throw new Error("Unauthorized"); + + const input = parseInput(officerSchema, { orgId, userId }); + await assertOrgCreator(input.orgId, session.user.id, "Only the org owner can add officers"); + + const [target] = await db + .select({ id: users.id }) + .from(users) + .where(eq(users.id, input.userId)) .limit(1); + if (!target) throw new Error("User not found"); - if (existing) throw new Error("User is already a member"); + await db + .insert(orgMembers) + .values({ orgId: input.orgId, userId: input.userId, role: "officer" }) + .onConflictDoNothing(); + await db + .update(orgMembers) + .set({ role: "officer" }) + .where( + and( + eq(orgMembers.orgId, input.orgId), + eq(orgMembers.userId, input.userId), + eq(orgMembers.role, "member"), + ), + ); - await db.insert(orgMembers).values({ orgId, userId, role: "officer" }); - revalidatePath(`/orgs/${orgId}`); + revalidatePath(`/orgs/${input.orgId}`); } export async function removeOfficer(orgId: string, userId: string): Promise { const session = await auth(); if (!session?.user?.id) throw new Error("Unauthorized"); - // Verify caller is org owner - const [org] = await db - .select({ creatorId: organizations.creatorId }) - .from(organizations) - .where(eq(organizations.id, orgId)) - .limit(1); - - if (!org || org.creatorId !== session.user.id) { - throw new Error("Only the org owner can remove officers"); - } + const input = parseInput(officerSchema, { orgId, userId }); + await assertOrgCreator(input.orgId, session.user.id, "Only the org owner can remove officers"); + // Officers only — never the owner row, so an org can't be left ownerless. await db .delete(orgMembers) - .where(and(eq(orgMembers.orgId, orgId), eq(orgMembers.userId, userId))); + .where( + and( + eq(orgMembers.orgId, input.orgId), + eq(orgMembers.userId, input.userId), + eq(orgMembers.role, "officer"), + ), + ); - revalidatePath(`/orgs/${orgId}`); + revalidatePath(`/orgs/${input.orgId}`); } export async function getUserOrgs(): Promise<{ id: string; name: string }[]> { @@ -376,6 +482,9 @@ export async function getRecommendedOrgs(): Promise { .where( and( inArray(eventTags.tag, interestTags), + // Only publicly listed events count as evidence of what an org hosts. + eq(events.status, "published"), + eq(events.isPublic, true), followedOrgIds.length > 0 ? sql`${organizations.id} NOT IN (${sql.join( followedOrgIds.map((id) => sql`${id}`), diff --git a/apps/web/src/actions/upload.ts b/apps/web/src/actions/upload.ts index 99a72f3..2d5f5b8 100644 --- a/apps/web/src/actions/upload.ts +++ b/apps/web/src/actions/upload.ts @@ -2,11 +2,18 @@ import { PutObjectCommand, S3Client } from "@aws-sdk/client-s3"; import { getSignedUrl } from "@aws-sdk/s3-request-presigner"; +import { z } from "zod"; import { auth } from "~/auth"; import { env } from "~/env"; - -const ALLOWED_TYPES = ["image/jpeg", "image/png", "image/webp"]; -const MAX_SIZE = 5 * 1024 * 1024; // 5MB +import { enforceRateLimit } from "~/lib/rate-limit"; +import { + IMAGE_EXTENSIONS, + type ImageContentType, + MAX_UPLOAD_BYTES, + UPLOAD_FOLDERS, + publicUrlForKey, +} from "~/lib/s3"; +import { parseInput } from "~/lib/validation"; function getS3Client() { return new S3Client({ @@ -15,8 +22,22 @@ function getS3Client() { }); } +const presignSchema = z.object({ + /** Accepted for backwards compatibility but never used — keys are generated server-side. */ + filename: z.string().max(255).optional(), + contentType: z.enum(Object.keys(IMAGE_EXTENSIONS) as [ImageContentType, ...ImageContentType[]], { + message: `Invalid file type. Allowed: ${Object.keys(IMAGE_EXTENSIONS).join(", ")}`, + }), + size: z + .number() + .int() + .positive() + .max(MAX_UPLOAD_BYTES, { message: "File too large. Maximum 5MB." }), + folder: z.enum(UPLOAD_FOLDERS), +}); + export async function getPresignedUploadUrl(input: { - filename: string; + filename?: string; contentType: string; size: number; folder: string; @@ -25,34 +46,38 @@ export async function getPresignedUploadUrl(input: { if (!session?.user?.id) { throw new Error("Unauthorized"); } + const userId = session.user.id; - if (!ALLOWED_TYPES.includes(input.contentType)) { - throw new Error(`Invalid file type. Allowed: ${ALLOWED_TYPES.join(", ")}`); - } - - if (input.size > MAX_SIZE) { - throw new Error("File too large. Maximum 5MB."); - } + const { contentType, size, folder } = parseInput(presignSchema, input); + enforceRateLimit("upload", userId); if (!env.AWS_S3_BUCKET) { throw new Error("S3 bucket not configured"); } - const ext = input.filename.split(".").pop() ?? "jpg"; - const key = `${input.folder}/${crypto.randomUUID()}.${ext}`; + // Key and extension are derived server-side only: the client's filename + // never reaches S3, so it can't pick a path, overwrite an object, or upload + // an .html/.svg under an image content type. + const key = `${folder}/${userId}/${crypto.randomUUID()}.${IMAGE_EXTENSIONS[contentType]}`; const command = new PutObjectCommand({ Bucket: env.AWS_S3_BUCKET, Key: key, - ContentType: input.contentType, - ContentLength: input.size, + ContentType: contentType, + ContentLength: size, }); - const url = await getSignedUrl(getS3Client(), command, { expiresIn: 300 }); + // The presigner leaves Content-Type unsigned by default; signing both it and + // Content-Length makes S3 reject a PUT with a different type or size than + // the one validated above. + const url = await getSignedUrl(getS3Client(), command, { + expiresIn: 300, + signableHeaders: new Set(["content-type", "content-length"]), + }); return { uploadUrl: url, key, - publicUrl: `https://${env.AWS_S3_BUCKET}.s3.amazonaws.com/${key}`, + publicUrl: publicUrlForKey(key), }; } diff --git a/apps/web/src/actions/users.ts b/apps/web/src/actions/users.ts index 51b04e2..0bd2b1e 100644 --- a/apps/web/src/actions/users.ts +++ b/apps/web/src/actions/users.ts @@ -1,20 +1,51 @@ "use server"; -import { - type campusRegionEnum, - db, - eq, - type eventTagEnum, - userInterests, - userRegions, - users, -} from "@the-forum/database"; +import { db, eq, userInterests, userRegions, users } from "@the-forum/database"; import { revalidatePath } from "next/cache"; +import { z } from "zod"; import { auth } from "~/auth"; +import { uploadedImageUrlSchema } from "~/lib/s3"; +import { campusRegionSchema, eventTagSchema, parseInput, uniqueEnumArray } from "~/lib/validation"; + +// Enum arrays are validated against the pgEnum values (never cast), and +// de-duplicated so a repeated value can't trip the composite primary key. +const profileFieldsSchema = z.object({ + classYear: z.string().trim().max(10), + major: z.string().trim().max(255), + isOrgLeader: z.boolean(), + interests: uniqueEnumArray(eventTagSchema), + regions: uniqueEnumArray(campusRegionSchema), + // Optional everywhere: omitted means "leave the stored name alone". + displayName: z.string().trim().min(1).max(255).optional(), +}); + +const onboardingSchema = profileFieldsSchema; +const updateProfileSchema = profileFieldsSchema.partial(); + +type InterestTag = z.output; +type CampusRegion = z.output; + +type Tx = Parameters[0]>[0]; + +async function replaceInterests(tx: Tx, userId: string, interests: InterestTag[]) { + await tx.delete(userInterests).where(eq(userInterests.userId, userId)); + if (interests.length > 0) { + await tx + .insert(userInterests) + .values(interests.map((tag) => ({ userId, tag }))) + .onConflictDoNothing(); + } +} -// Derived from the DB schema so these can never drift from the pgEnum values -type InterestTag = (typeof eventTagEnum.enumValues)[number]; -type CampusRegion = (typeof campusRegionEnum.enumValues)[number]; +async function replaceRegions(tx: Tx, userId: string, regions: CampusRegion[]) { + await tx.delete(userRegions).where(eq(userRegions.userId, userId)); + if (regions.length > 0) { + await tx + .insert(userRegions) + .values(regions.map((region) => ({ userId, region }))) + .onConflictDoNothing(); + } +} export async function completeOnboarding(data: { interests: string[]; @@ -22,45 +53,32 @@ export async function completeOnboarding(data: { major: string; regions: string[]; isOrgLeader: boolean; + displayName?: string; }) { const session = await auth(); if (!session?.user?.id) throw new Error("Unauthorized"); const userId = session.user.id; - - // Update user profile - await db - .update(users) - .set({ - classYear: data.classYear, - major: data.major, - isOrgLeader: data.isOrgLeader, - onboarded: true, - updatedAt: new Date(), - }) - .where(eq(users.id, userId)); - - // Insert interests - if (data.interests.length > 0) { - await db.delete(userInterests).where(eq(userInterests.userId, userId)); - await db.insert(userInterests).values( - data.interests.map((tag) => ({ - userId, - tag: tag as InterestTag, - })), - ); - } - - // Insert regions - if (data.regions.length > 0) { - await db.delete(userRegions).where(eq(userRegions.userId, userId)); - await db.insert(userRegions).values( - data.regions.map((region) => ({ - userId, - region: region as CampusRegion, - })), - ); - } + const input = parseInput(onboardingSchema, data); + + // One transaction, so a double-submit can't interleave two delete+insert + // sequences and a failure can't leave a half-onboarded profile. + await db.transaction(async (tx) => { + await tx + .update(users) + .set({ + displayName: input.displayName, + classYear: input.classYear, + major: input.major, + isOrgLeader: input.isOrgLeader, + onboarded: true, + updatedAt: new Date(), + }) + .where(eq(users.id, userId)); + + if (input.interests.length > 0) await replaceInterests(tx, userId, input.interests); + if (input.regions.length > 0) await replaceRegions(tx, userId, input.regions); + }); revalidatePath("/"); } @@ -154,53 +172,44 @@ export async function updateProfile(data: { isOrgLeader?: boolean; interests?: string[]; regions?: string[]; + displayName?: string; }): Promise { const user = await getCurrentUser(); const userId = user.id; - - await db - .update(users) - .set({ - classYear: data.classYear, - major: data.major, - isOrgLeader: data.isOrgLeader, - updatedAt: new Date(), - }) - .where(eq(users.id, userId)); - - if (data.interests) { - await db.delete(userInterests).where(eq(userInterests.userId, userId)); - if (data.interests.length > 0) { - await db.insert(userInterests).values( - data.interests.map((tag) => ({ - userId, - tag: tag as InterestTag, - })), - ); - } - } - - if (data.regions) { - await db.delete(userRegions).where(eq(userRegions.userId, userId)); - if (data.regions.length > 0) { - await db.insert(userRegions).values( - data.regions.map((region) => ({ - userId, - region: region as CampusRegion, - })), - ); - } - } + const input = parseInput(updateProfileSchema, data); + + await db.transaction(async (tx) => { + await tx + .update(users) + .set({ + displayName: input.displayName, + classYear: input.classYear, + major: input.major, + isOrgLeader: input.isOrgLeader, + updatedAt: new Date(), + }) + .where(eq(users.id, userId)); + + if (input.interests) await replaceInterests(tx, userId, input.interests); + if (input.regions) await replaceRegions(tx, userId, input.regions); + }); revalidatePath("/settings"); revalidatePath("/profile"); revalidatePath("/explore"); } +const avatarUrlSchema = uploadedImageUrlSchema("avatars"); + export async function updateAvatar(avatarUrl: string): Promise { const user = await getCurrentUser(); + // Must be one of our uploads (or empty to clear it). + const url = parseInput(avatarUrlSchema, avatarUrl); - await db.update(users).set({ avatarUrl, updatedAt: new Date() }).where(eq(users.id, user.id)); + await db + .update(users) + .set({ avatarUrl: url, updatedAt: new Date() }) + .where(eq(users.id, user.id)); revalidatePath("/settings"); revalidatePath("/profile"); diff --git a/apps/web/src/app/(app)/error.tsx b/apps/web/src/app/(app)/error.tsx new file mode 100644 index 0000000..03256d3 --- /dev/null +++ b/apps/web/src/app/(app)/error.tsx @@ -0,0 +1,26 @@ +"use client"; + +import Link from "next/link"; +import { ErrorState } from "~/components/common/states"; +import { PageShell } from "~/components/layout/page-shell"; + +/** In-app error boundary: keeps the nav rail and top bar around the failure. */ +export default function AppError({ + reset, +}: { error: Error & { digest?: string }; reset: () => void }) { + return ( + + + + Back to Explore + + + ); +} diff --git a/apps/web/src/app/(app)/events/[id]/edit/edit-event-form.tsx b/apps/web/src/app/(app)/events/[id]/edit/edit-event-form.tsx index 73f88b4..ded5d6f 100644 --- a/apps/web/src/app/(app)/events/[id]/edit/edit-event-form.tsx +++ b/apps/web/src/app/(app)/events/[id]/edit/edit-event-form.tsx @@ -1,245 +1,190 @@ "use client"; -import { format } from "date-fns"; -import { - ArrowUp, - CalendarIcon, - ExternalLink, - Eye, - Globe, - ImagePlus, - Link2, - Lock, - MapPin, - Pencil, - Plus, - Search, - Upload, - X, -} from "lucide-react"; +import { ArrowUp, Eye, Globe, Link2, Lock, Pencil, Upload, X } from "lucide-react"; import { useRouter } from "next/navigation"; import { useCallback, useRef, useState, useTransition } from "react"; +import { toast } from "sonner"; import { type EventDetail, updateEvent } from "~/actions/events"; -import { getPresignedUploadUrl } from "~/actions/upload"; +import { + type CampusLocation, + type EventWhen, + EventWhenFields, + FORM_ERROR, + FORM_LABEL, + LocationPicker, + OTHER_LOCATION_ID, + TagPicker, + VisibilityToggle, + type WhenErrors, + describeEventWhen, + normalizeExternalLink, + resolveEventWhen, +} from "~/components/events/event-form-fields"; +import { EventPreviewModal } from "~/components/events/event-preview-modal"; import { PageShell } from "~/components/layout/page-shell"; import { Button } from "~/components/ui/button"; -import { Calendar } from "~/components/ui/calendar"; -import { Input } from "~/components/ui/input"; -import { Popover, PopoverContent, PopoverTrigger } from "~/components/ui/popover"; import { Textarea } from "~/components/ui/textarea"; +import { toZonedDateKey, toZonedTimeValue } from "~/lib/date-format"; +import { isInterestValue } from "~/lib/profile-options"; +import { IMAGE_ACCEPT, uploadImage } from "~/lib/upload-image"; import { cn } from "~/lib/utils"; -const TIMELINE_SECTIONS = [ - { id: "cover", label: "Cover & Title", color: "bg-forum-cerulean" }, - { id: "details", label: "Details & Description", color: "bg-forum-coral" }, - { id: "when-where", label: "When & Where", color: "bg-forum-coral" }, - { id: "uploads", label: "Uploads & Links", color: "bg-forum-coral" }, -]; - interface EditEventFormProps { event: EventDetail; - locations: { id: string; name: string; category: string }[]; + locations: CampusLocation[]; } -function pad(n: number) { - return n.toString().padStart(2, "0"); -} +type FormErrors = WhenErrors & + Partial>; export function EditEventForm({ event, locations }: EditEventFormProps) { const router = useRouter(); const [isPending, startTransition] = useTransition(); const fileInputRef = useRef(null); - const [activeSection, setActiveSection] = useState("cover"); const [title, setTitle] = useState(event.title); const [description, setDescription] = useState(event.description); - const [date, setDate] = useState(event.datetime); - const [startTime, setStartTime] = useState( - `${pad(event.datetime.getHours())}:${pad(event.datetime.getMinutes())}`, - ); - const [endTime, setEndTime] = useState( - event.endDatetime - ? `${pad(event.endDatetime.getHours())}:${pad(event.endDatetime.getMinutes())}` - : "", - ); + // Seeded from the stored instant, read back as Princeton wall-clock time. + const [when, setWhen] = useState(() => ({ + dateKey: toZonedDateKey(event.datetime), + startTime: toZonedTimeValue(event.datetime), + endTime: event.endDatetime ? toZonedTimeValue(event.endDatetime) : "", + })); const [locationId, setLocationId] = useState(event.locationId); - const [locationSearch, setLocationSearch] = useState(""); - const [locationOpen, setLocationOpen] = useState(false); - const [tags, setTags] = useState(event.tags); - const [tagSearch, setTagSearch] = useState(""); + const [tags, setTags] = useState(event.tags.filter(isInterestValue)); const [flyerUrl, setFlyerUrl] = useState(event.flyerUrl); const [flyerPreview, setFlyerPreview] = useState(event.flyerUrl); const [isUploading, setIsUploading] = useState(false); const [externalLink, setExternalLink] = useState(event.externalLink ?? ""); const [isPublic, setIsPublic] = useState(event.isPublic); - const [errors, setErrors] = useState>({}); - - const filteredLocations = locations.filter( - (loc) => !locationSearch || loc.name.toLowerCase().includes(locationSearch.toLowerCase()), - ); - const selectedLocationName = locations.find((l) => l.id === locationId)?.name ?? ""; + const [showPreview, setShowPreview] = useState(false); + const [errors, setErrors] = useState({}); - const addTag = (tag: string) => { - if (tag && !tags.includes(tag)) setTags((prev) => [...prev, tag]); - setTagSearch(""); - }; - const removeTag = (tag: string) => setTags((prev) => prev.filter((t) => t !== tag)); + const clearError = (key: keyof FormErrors) => + setErrors((prev) => (prev[key] ? { ...prev, [key]: undefined } : prev)); const handleImageUpload = useCallback( async (file: File) => { if (isUploading) return; setIsUploading(true); + const previousUrl = flyerUrl; + const localPreview = URL.createObjectURL(file); + setFlyerPreview(localPreview); try { - const reader = new FileReader(); - reader.onload = (e) => setFlyerPreview(e.target?.result as string); - reader.readAsDataURL(file); - const { uploadUrl, publicUrl } = await getPresignedUploadUrl({ - filename: file.name, - contentType: file.type, - size: file.size, - folder: "event-flyers", - }); - await fetch(uploadUrl, { - method: "PUT", - body: file, - headers: { "Content-Type": file.type }, - }); + const publicUrl = await uploadImage(file, "event-flyers"); setFlyerUrl(publicUrl); + setFlyerPreview(publicUrl); } catch (err) { - console.error("Upload failed:", err); - setFlyerPreview(event.flyerUrl); + setFlyerPreview(previousUrl); + toast.error(err instanceof Error ? err.message : "Couldn't upload that image."); } finally { + URL.revokeObjectURL(localPreview); setIsUploading(false); + if (fileInputRef.current) fileInputRef.current.value = ""; } }, - [isUploading, event.flyerUrl], + [isUploading, flyerUrl], ); const handleDrop = useCallback( (e: React.DragEvent) => { e.preventDefault(); const file = e.dataTransfer.files[0]; - if (file?.type.startsWith("image/")) handleImageUpload(file); + if (file) handleImageUpload(file); }, [handleImageUpload], ); - const validate = (): boolean => { - const newErrors: Record = {}; - if (!title.trim()) newErrors.title = "Title is required"; - if (!description.trim()) newErrors.description = "Description is required"; - if (!date) newErrors.date = "Date is required"; - if (!locationId) newErrors.location = "Location is required"; - setErrors(newErrors); - return Object.keys(newErrors).length === 0; - }; - const handleSubmit = () => { - if (!validate()) return; - startTransition(async () => { - const datetime = new Date(date as Date); - const [h, m] = startTime.split(":").map(Number); - datetime.setHours(h ?? 0, m ?? 0); + const nextErrors: FormErrors = {}; + if (!title.trim()) nextErrors.title = "Title is required"; + if (!description.trim()) nextErrors.description = "Description is required"; + if (!locationId) nextErrors.location = "Choose a location, or “Other / off-campus / TBA”"; + const link = normalizeExternalLink(externalLink); + if (link === null) nextErrors.link = "Enter a valid https:// link, e.g. https://example.com"; + const resolved = resolveEventWhen(when); + if (!resolved.ok) Object.assign(nextErrors, resolved.errors); - let endDatetime: string | undefined; - if (endTime) { - const end = new Date(date as Date); - const [eh, em] = endTime.split(":").map(Number); - end.setHours(eh ?? 0, em ?? 0); - endDatetime = end.toISOString(); - } + setErrors(nextErrors); + if (Object.values(nextErrors).some(Boolean) || !resolved.ok) { + toast.error("Please fix the highlighted fields."); + return; + } + if (isUploading) { + toast.error("Please wait for the cover image to finish uploading."); + return; + } - await updateEvent(event.id, { - title: title.trim(), - description: description.trim(), - datetime: datetime.toISOString(), - endDatetime, - locationId, - tags, - flyerUrl: flyerUrl ?? undefined, - externalLink: externalLink.trim() || undefined, - isPublic, - }); - router.push(`/events/${event.id}`); + startTransition(async () => { + try { + await updateEvent(event.id, { + title: title.trim(), + description: description.trim(), + datetime: resolved.datetime.toISOString(), + endDatetime: resolved.endDatetime?.toISOString(), + locationId, + tags, + flyerUrl: flyerUrl ?? undefined, + externalLink: link || undefined, + isPublic, + }); + toast.success("Changes saved"); + router.push(`/events/${event.id}`); + } catch { + toast.error("Couldn't save your changes. Please try again."); + } }); }; + const previewWhen = describeEventWhen(when); + return ( - + {/* Top buttons */} -
+
- -
-
-
- {/* Timeline sidebar */} -
-
- {TIMELINE_SECTIONS.map(({ id, label, color }) => ( - - ))} -
-
- +
{/* Form body */}
{/* Cover Image */} -
+
{flyerPreview ? ( -
+
Cover preview {isUploading && ( -
- -
+ + + Uploading… + )}
) : ( @@ -248,18 +193,22 @@ export function EditEventForm({ event, locations }: EditEventFormProps) { onClick={() => fileInputRef.current?.click()} onDragOver={(e) => e.preventDefault()} onDrop={handleDrop} - className="w-full h-[280px] rounded-[10px] bg-forum-turquoise/15 border-2 border-dashed border-forum-turquoise/40 flex items-end justify-end p-[20px] cursor-pointer hover:bg-forum-turquoise/20 transition-colors mb-[20px]" + className="w-full h-[120px] sm:h-[150px] rounded-[10px] bg-forum-turquoise/15 border-2 border-dashed border-forum-turquoise/40 flex flex-col items-end justify-end gap-1 p-4 cursor-pointer hover:bg-forum-turquoise/20 transition-colors mb-4" > - Add Cover Image + Add Cover Image + + + JPEG, PNG or WebP, up to 5 MB )} { const file = e.target.files?.[0]; if (file) handleImageUpload(file); @@ -268,39 +217,44 @@ export function EditEventForm({ event, locations }: EditEventFormProps) {
{/* Event Title */} -
- +
+