diff --git a/.env.example b/.env.example index 7a2e21f7d..6719b1185 100644 --- a/.env.example +++ b/.env.example @@ -1,9 +1,69 @@ -# AWS S3 (optional — for clip backup/gallery) -AWS_ACCESS_KEY_ID=your_aws_access_key_here -AWS_SECRET_ACCESS_KEY=your_aws_secret_key_here +# ============================================================================= +# OpenShorts environment configuration +# +# Copy this file to `.env` (gitignored) and fill in the values you need. +# Most API keys are stored encrypted in the browser and sent via headers — they +# do NOT need to be set here unless you want a server-side fallback. +# ============================================================================= + +# --- Required (server-side reads via os.getenv) ----------------------------- + +# Google Gemini API key — used by the viral-clip extractor in main.py. +# https://ai.google.dev/gemini-api/docs/api-key +GEMINI_API_KEY= + +# --- Optional: AWS S3 (clip backup + public gallery) ------------------------ + +AWS_ACCESS_KEY_ID= +AWS_SECRET_ACCESS_KEY= AWS_REGION=eu-west-3 -AWS_S3_BUCKET=your-bucket-name -AWS_S3_PUBLIC_BUCKET=your-public-bucket-name +AWS_S3_BUCKET= +AWS_S3_PUBLIC_BUCKET= + +# --- Optional: YouTube ingestion -------------------------------------------- + +# Disable the YouTube URL ingest tab entirely (uploads-only mode). +DISABLE_YOUTUBE_URL=false + +# Netscape-format cookies (concatenated into one line) to bypass YouTube's +# bot-detection on server IPs. yt-dlp writes this to /app/cookies.txt at +# container startup. +# YOUTUBE_COOKIES= + +# --- Optional: Remotion render service -------------------------------------- + +# Used by the backend to call the renderer. +# Default uses Docker's internal network (service name + container port). +# If you run the backend OUTSIDE Docker, change to: http://localhost:3003 +RENDER_SERVICE_URL=http://renderer:3100 + +# --- Tuning ----------------------------------------------------------------- + +# Max concurrent video-processing jobs (asyncio semaphore in the job queue). +MAX_CONCURRENT_JOBS=5 + +# AI Restyle upload cap (megabytes). The /api/restyle route does a Content-Length +# preflight before touching disk; requests above this return 413 immediately. +AI_RESTYLE_MAX_FILE_SIZE_MB=250 + +# ============================================================================= +# Frontend (dashboard/) — Vite reads these at build time +# ============================================================================= + +# Production API URL override (defaults to relative paths in dev). +VITE_API_URL=http://localhost:3002 + +# Optional salt for localStorage API-key encryption. +# VITE_ENCRYPTION_KEY= + +# ============================================================================= +# Client-side keys (stored encrypted in the browser, sent via headers per-call) +# +# These are listed here for reference. The Python code does NOT read them from +# .env — set them in the dashboard UI instead. Listed below in case you want +# a server-side default (would require code changes in app.py to honor them). +# ============================================================================= -# YouTube cookies (optional — paste Netscape-format cookies to bypass bot detection) -# YOUTUBE_COOKIES=... +# ELEVENLABS_API_KEY= +# UPLOAD_POST_API_KEY= +# FAL_KEY= diff --git a/.gitignore b/.gitignore index f2df51cd8..e49f13f08 100644 --- a/.gitignore +++ b/.gitignore @@ -12,6 +12,11 @@ __pycache__/ *.pyc +# Editable-install / build artifacts +*.egg-info/ +build/ +dist/ + # Temporary files / runtime dirs temp_* uploads/ @@ -33,6 +38,11 @@ output/ # Cache dirs .cache/ .config/ +.pytest_cache/ + +# Test ephemera (baseline.openapi.json IS committed; current is not) +backend/tests/snapshots/current.openapi.json +backend/tests/fixtures/smoke.mp4 # Multi-agent Skills .agents/ .agent/ @@ -40,3 +50,7 @@ output/ skills/ skills-lock.json +# Session-local artifacts (handovers, test screenshots) — not for git +.session/ +.compact-ultra + diff --git a/.pre-commit-config.yaml b/.pre-commit-config.yaml new file mode 100644 index 000000000..5a0f67093 --- /dev/null +++ b/.pre-commit-config.yaml @@ -0,0 +1,10 @@ +repos: + - repo: local + hooks: + - id: update-claude-md + name: Regenerate CLAUDE.md auto-managed sections + entry: python scripts/update_claude_md.py + language: system + pass_filenames: false + always_run: true + stages: [pre-commit] diff --git a/CLAUDE.md b/CLAUDE.md deleted file mode 100644 index 093db8afe..000000000 --- a/CLAUDE.md +++ /dev/null @@ -1,102 +0,0 @@ -# CLAUDE.md - -This file provides guidance to Claude Code (claude.ai/code) when working with code in this repository. - -## Project Overview - -OpenShorts is an AI-powered vertical video generator that transforms long YouTube videos or local uploads into viral-ready short clips (9:16 format) for TikTok, Instagram Reels, and YouTube Shorts. Uses Google Gemini 2.0 Flash for viral moment detection and title generation. - -## Development Commands - -### Local Development (Docker) -```bash -docker compose up --build # Build and run full stack -``` -- Backend: http://localhost:8000 (FastAPI/Uvicorn) -- Frontend: http://localhost:5175 (Vite proxies API calls to backend) - -### Frontend Only (Dashboard) -```bash -cd dashboard -npm install -npm run dev # Dev server with HMR (port 5173) -npm run build # Production build -npm run lint # ESLint (strict, --max-warnings 0) -``` - -### Backend Only -```bash -pip install -r requirements.txt -uvicorn app:app --host 0.0.0.0 --port 8000 -``` - -## Architecture - -### Core Processing Pipeline -1. **Ingest** - YouTube download (yt-dlp) or local upload -2. **Transcription** - faster-whisper with word-level timestamps -3. **Scene Detection** - PySceneDetect for segment boundaries -4. **AI Analysis** - Gemini identifies 3-15 viral moments (15-60 sec each) -5. **FFmpeg Extraction** - Precise clip cutting -6. **AI Cropping** - Vertical reframing with subject tracking -7. **Effects/Subtitles** - Optional AI-generated FFmpeg filters -8. **Hook Overlay** - Text overlays with styled fonts -9. **Voice Dubbing** - Optional ElevenLabs AI translation (30+ languages) -10. **S3 Backup** - Silent background upload -11. **Social Distribution** - Upload-Post API (async upload) - -### Key Files -| File | Purpose | -|------|---------| -| `main.py` | Core video processing: transcription, scene detection, clip extraction, vertical reframing | -| `app.py` | FastAPI server with async job queue and REST endpoints | -| `editor.py` | Gemini AI integration for dynamic video effects (FFmpeg filter generation) | -| `hooks.py` | Hook text overlay generation with font rendering | -| `s3_uploader.py` | AWS S3 upload with caching | -| `subtitles.py` | SRT generation, FFmpeg subtitle burning, and dubbed video transcription | -| `translate.py` | ElevenLabs dubbing API for AI voice translation | -| `dashboard/src/App.jsx` | Main React component with state management | -| `dashboard/src/components/TranslateModal.jsx` | Voice dubbing UI with language selection | - -### Dual-Mode Video Reframing -- **TRACK Mode** (single subject): MediaPipe face detection + YOLOv8 fallback with "Heavy Tripod" stabilization -- **GENERAL Mode** (groups/landscapes): Blurred background layout preserving full width - -### Key Classes -- `SmoothedCameraman` - Stabilized camera movement with safe zone logic (prevents jitter) -- `SpeakerTracker` - Prevents rapid speaker switching, handles temporary occlusions - -### API Endpoints -| Method | Route | Purpose | -|--------|-------|---------| -| POST | `/api/process` | Submit video for processing | -| GET | `/api/status/{job_id}` | Poll job status and logs | -| POST | `/api/edit` | Apply AI video effects | -| POST | `/api/subtitle` | Generate and apply subtitles (auto-transcribes dubbed videos) | -| POST | `/api/hook` | Add text hook overlays | -| POST | `/api/translate` | AI voice dubbing via ElevenLabs | -| GET | `/api/translate/languages` | List supported dubbing languages | -| POST | `/api/social/post` | Post to social media (async upload) | - -### Concurrency Model -Async job queue with semaphore-based concurrency control. Configure via `MAX_CONCURRENT_JOBS` env var (default: 5). Jobs auto-cleanup after 1 hour. - -## Environment Variables - -**Server-side (.env):** -- `AWS_ACCESS_KEY_ID`, `AWS_SECRET_ACCESS_KEY`, `AWS_REGION`, `AWS_S3_BUCKET` - For S3 backup -- `MAX_CONCURRENT_JOBS` - Concurrent processing limit (default: 5) -- `VITE_API_URL` - Production API URL override - -**Client-side (localStorage, encrypted):** -- `GEMINI_API_KEY` - Google Gemini API key (required) -- `ELEVENLABS_API_KEY` - ElevenLabs API key for voice dubbing (optional) -- `UPLOAD_POST_API_KEY` - Upload-Post API key for social posting (optional) - -> API keys are stored encrypted in the browser and sent via headers only when needed. Never stored server-side. - -## Tech Stack -- **Backend:** Python 3.11, FastAPI, google-genai, faster-whisper, ultralytics (YOLOv8), mediapipe, opencv-python, yt-dlp, FFmpeg, httpx -- **Frontend:** React 18, Vite 4, Tailwind CSS 3.4 -- **External APIs:** Google Gemini, ElevenLabs Dubbing, Upload-Post -- **Infrastructure:** Docker + Docker Compose, AWS S3 diff --git a/HANDOFF.md b/HANDOFF.md new file mode 100644 index 000000000..4d39af2e9 --- /dev/null +++ b/HANDOFF.md @@ -0,0 +1,629 @@ +# OpenShorts — Handoff + +A self-contained briefing for the next agent (human or LLM). Reads top-to-bottom. + +If you read nothing else, read **§3 Critical bugs already fixed**, **§5 Outstanding work**, and **§6 Operating rules** — those keep you from re-walking past landmines. + +--- + +## 1. What OpenShorts is + +OpenShorts is an **AI-powered vertical short-video generator**. Drop in a long video (YouTube URL or local upload) and it produces 3–15 viral 9:16 clips ready for TikTok / Reels / Shorts. + +Hot path of the Clip Generator: + +1. Transcribe audio locally (faster-whisper, INT8). +2. Detect scene boundaries (PySceneDetect). +3. Send transcript to **Gemini 2.5 Flash** → returns 3–15 viral moments with start/end times and titles. +4. Cut each clip with FFmpeg. +5. Per-scene reframe to 9:16 — either tracking the active speaker (MediaPipe face + YOLOv8 person) or a panoramic blurred-background ("General") composite. +6. Optional layers: AI-generated FFmpeg effects, text hook PNGs, burn-in subtitles, ElevenLabs voice dubbing, S3 backup, Upload-Post distribution. + +The frontend now wraps this in a **multi-page platform shell**: + +- **Short-form** — 4-step wizard for up to 5 source videos in one batch. Each runs the same `/api/process` pipeline in parallel. +- **Long-form** — 4-step wizard with a chapter-aware editor for re-exporting segments as shorts. The pipeline that actually generates chapters is stubbed; the wizard simulates progress and seeds placeholder chapters. +- **Clip Generator** — the original single-job flow at `/api/process` (still works; this is what the wizards are layered on top of). +- **Dashboard** — StatCards + scheduled-uploads list + recent activity, derived from localStorage history + the notifications store. +- **Settings** — Brand Kit, API Keys, and placeholder section pages for Subtitle style / Color presets / Export defaults / per-platform. +- **Legacy** — SaaSShorts UGC pipeline, YouTube thumbnails, UGC gallery, AI Agent terminal — hidden from the sidebar but reachable at `/legacy/*` URLs. + +See `CLAUDE.md` for the full decision table on **where new things go** (routes, FFmpeg ops, layouts, motion graphics, etc.) — auto-managed sections are regenerated by the pre-commit hook. + +--- + +## 2. Current branch state + +| | | +| --- | --- | +| **Branch** | `chore/restructure-and-docs` (27 commits ahead of `main`) | +| **HEAD** | `7d073cb fix(short-form): normalize backend job status + result key` | +| **Working tree** | **NOT clean** — see §4 | +| **Revert point** | `git reset --hard pre-restructure-20260519-1526` | +| **Tests** | 61/62 green (`cd backend && pytest -m "not e2e"`). 1 pre-existing OpenAPI snapshot drift — see §5. | +| **OpenAPI baseline** | `backend/tests/snapshots/baseline.openapi.json` (35 endpoints) | +| **Frontend build** | Green; 1616 modules, ~1289 KB JS chunk, 0 warnings | +| **Docker stack** | All three services run via `docker compose up --build` (read §6 first — there's a known volume gotcha) | +| **Frontend URL** | http://localhost:3001 | +| **Backend API** | http://localhost:3002 | +| **Renderer** | http://localhost:3003 | + +### Commit graph (top of branch) + +``` +7d073cb fix(short-form): normalize backend job status + result key +93f5907 docs(roadmap): add product roadmap + smoke-test follow-ups +43c2d96 fix(smoke-test): runtime bugs + Codex H1/H2/M3 remediation +95ca831 feat(ui): phase 4 — long-form 4-step wizard + Dashboard +97b7eff feat(ui): phase 3 — short-form 4-step wizard + UI primitives +337b509 feat(ui): phase 2 — Settings VS-Code layout + notifications + tooltips +667a88e feat(ui): phase 1 — shell + theme + routing skeleton +3d2b4f8 feat(brand-kit): brand kit settings + font upload + port refresh +55f0ef1 chore(restructure): split repo into backend/ + frontend/ + renderer/ + assets/ +1dd4b9a docs(roadmap): design future features + document deferred refactors +``` + +### Nothing has been pushed + +The user explicitly said this branch stays local until they decide otherwise. `mutonby/openshorts` is read-only for the active gh account — don't try to push there. If they ask to push, fork first or have them switch gh accounts. + +--- + +## 3. Critical bugs already fixed + +These were all surfaced by the post-Phase-4 browser smoke test (the previous agent shipped Phases 1–4 without ever exercising the UI in a browser — `npm run build` doesn't catch runtime). Don't reintroduce them. + +### 🔴 BLOCKER: `run_job` invoked a missing entry point + +`backend/app/main.py:365` was running `python -u main.py` but no top-level `main.py` exists post-restructure — the CLI was moved to `backend/app/cli.py`. **Every short-form Processing job exited with code 2** (`python: can't open file '/app/main.py'`). Fixed → `python -u -m app.cli`. The CLI now actually runs. + +### 🔴 BLOCKER: Long-form simulated progress stuck at 0 % in dev + +`LongForm/steps/Processing.jsx` used a `startedRef` gate combined with a cleanup `clearInterval`. Under React 18 StrictMode (dev mode in docker), mount #1 set the ref + timer, the auto-cleanup cleared the timer, mount #2 bailed early — no timer running. Fixed by removing the gate and making the initial `setData` idempotent. + +**Pattern rule (don't repeat):** if a useEffect has cleanup AND a ref-based "run once" gate, StrictMode dev will silently break it. Either no cleanup OR no gate — pick one. + +### 🟠 Codex H1: file upload had no MIME / signature check + +`POST /api/process` accepted any file with `.mp4` in the name. Added `_ensure_video_upload(filename, first_chunk)` at `backend/app/main.py` — checks extension (`.mp4`/`.mov`) AND the `ftyp` box at byte offset 4 before writing to disk. Returns 415 with precise reason on mismatch. + +**Verified via curl:** text-content-with-.mp4 → 415 (ftyp), real-mp4-with-.txt → 415 (extension), real .mp4 → 200. + +### 🟠 Codex H2: wizard let users march past lost File handles + +File objects don't survive `JSON.stringify`. Both wizards persist `wizard.data` to localStorage. After a reload, `data.files[0].file` is a plain `{}` instead of a real File — and the wizards happily let you advance past Upload into a step that always fails. + +Fixed by adding an optional `resetOnRehydrate(mergedData)` predicate to `useWizard`. Both `Wizard.jsx` callers pass a File-presence check; rehydrate detects degraded state and forces step=0 + clears persistence. Verified end-to-end: upload → advance → reload → wizard now sits on Upload with cleared state. + +### 🟡 Codex M3: short-form polling could race on stale responses + +`ShortForm/steps/Processing.jsx` setInterval callback was async — if a `/api/status` call took longer than the 2 s poll interval, an older response could arrive after a newer one and overwrite `complete`/`error` back to `processing`. Cleanup also didn't abort in-flight fetches. + +Fixed by adding `AbortController` + `cancelled` flag, and a terminal-status guard in the setData updater (`if cur.status === 'complete' || cur.status === 'error' return prev`). Tested under the dev StrictMode double-mount. + +### 🔴 Backend/frontend contract mismatch (caught only by the real-key run) + +**This one only surfaced when a real Gemini key let the pipeline actually finish.** Dummy-key runs failed at Gemini and never exercised the success path. + +| backend sends | wizard was reading | +| --- | --- | +| `status: "completed"` | `'complete'` (so done-check never fired) | +| `status: "failed"` | `'error'` (so error rows never matched) | +| `result: {...}` | `data.results` (typo — the clips never reached `j.result`) | + +Without these mappings, a backend job that **finished cleanly and produced a clip** left the wizard sitting on "Process finished successfully." with Skip + Review both disabled. + +Fixed by adding `normalizeJobPayload(data)` next to `fetchStatus` in `ShortForm/steps/Processing.jsx` (mirrors what the legacy `frontend/src/hooks/useJobPolling.js` does for the Clip Generator). Backend vocab is `queued | processing | completed | failed` with `result` (singular); the wizard speaks `queued | processing | complete | error` with `result`. + +**Rule:** the legacy `useJobPolling.js` is the canonical reference for how to consume `/api/status` — copy the mapping when adding a new poll site. + +--- + +## 4. Uncommitted working-tree changes + +When this handoff was written, the tree was dirty with **a follow-up UX pass on Short-form Processing** plus the corresponding CLAUDE.md rule. The user has not yet asked for these to be committed — confirm before committing. + +| File | What it does | +| --- | --- | +| `CLAUDE.md` | Adds Convention #7: short-form and long-form code MUST stay isolated. No cross-imports between `pages/ShortForm/` and `pages/LongForm/`. Shared things go in `frontend/src/hooks/`, `components/ui/`, `state/`, or `lib/`. | +| `frontend/src/pages/ShortForm/steps/Upload.jsx` | Probes video duration via a hidden `