From b6709e501ff0b081f59e2182287dfdacb6778bd5 Mon Sep 17 00:00:00 2001 From: Milbaxter Date: Wed, 2 Sep 2026 11:37:28 +0300 Subject: [PATCH 01/16] Add proxy, job store, and Stripe signature helpers. --- backend/app/utils/__init__.py | 2 -- backend/app/utils/jobs.py | 57 +++++++++++++++++++++++++++++++++ backend/app/utils/proxy.py | 43 +++++++++++++++++++++++++ backend/app/utils/stripe_sig.py | 26 +++++++++++++++ 4 files changed, 126 insertions(+), 2 deletions(-) create mode 100644 backend/app/utils/jobs.py create mode 100644 backend/app/utils/proxy.py create mode 100644 backend/app/utils/stripe_sig.py diff --git a/backend/app/utils/__init__.py b/backend/app/utils/__init__.py index ef3cc69..2acb703 100644 --- a/backend/app/utils/__init__.py +++ b/backend/app/utils/__init__.py @@ -1,7 +1,5 @@ from app.utils.formatters import format_srt, format_vtt, format_markdown, format_llm_prompt -from app.utils.auth import verify_api_key, generate_new_api_key, API_KEYS_DB __all__ = [ "format_srt", "format_vtt", "format_markdown", "format_llm_prompt", - "verify_api_key", "generate_new_api_key", "API_KEYS_DB" ] diff --git a/backend/app/utils/jobs.py b/backend/app/utils/jobs.py new file mode 100644 index 0000000..b0c4889 --- /dev/null +++ b/backend/app/utils/jobs.py @@ -0,0 +1,57 @@ +"""In-process transcription job store. Upgrade to Redis/SQLite if you add workers.""" +from __future__ import annotations + +import threading +import uuid +from datetime import datetime, timezone +from typing import Any, Dict, Optional + +_lock = threading.Lock() +_JOBS: Dict[str, Dict[str, Any]] = {} + + +def create_job(payload: dict) -> dict: + job_id = uuid.uuid4().hex + rec = { + "job_id": job_id, + "status": "queued", + "payload": payload, + "error": None, + "result": None, + "created_at": datetime.now(timezone.utc).isoformat(), + } + with _lock: + _JOBS[job_id] = rec + return dict(rec) + + +def get_job(job_id: str) -> Optional[dict]: + with _lock: + rec = _JOBS.get(job_id) + if not rec: + return None + return { + "job_id": rec["job_id"], + "status": rec["status"], + "error": rec["error"], + "result": rec["result"], + "created_at": rec["created_at"], + } + + +def get_payload(job_id: str) -> Optional[dict]: + with _lock: + rec = _JOBS.get(job_id) + return rec["payload"] if rec else None + + +def set_status(job_id: str, status: str, error: Optional[str] = None, result: Optional[dict] = None) -> None: + with _lock: + rec = _JOBS.get(job_id) + if not rec: + return + rec["status"] = status + if error is not None: + rec["error"] = error + if result is not None: + rec["result"] = result diff --git a/backend/app/utils/proxy.py b/backend/app/utils/proxy.py new file mode 100644 index 0000000..00cbb23 --- /dev/null +++ b/backend/app/utils/proxy.py @@ -0,0 +1,43 @@ +"""HTTP(S) proxy helpers shared by YouTube, TikTok, and podcast parsers.""" +from typing import Any, Dict, Optional + +from app.config import settings + + +def http_proxy_url() -> Optional[str]: + """Return the first configured proxy URL (HTTP_PROXY, then HTTPS_PROXY).""" + return (settings.HTTP_PROXY or settings.HTTPS_PROXY or None) or None + + +def ytdlp_proxy_opts() -> Dict[str, Any]: + """yt-dlp option fragment that actually applies HTTP_PROXY.""" + proxy = http_proxy_url() + if proxy: + return {"proxy": proxy} + return {} + + +def httpx_proxy_kw() -> Dict[str, Any]: + proxy = http_proxy_url() + if proxy: + return {"proxy": proxy} + return {} + + +def youtube_transcript_api_client(): + """YouTubeTranscriptApi wired to HTTP_PROXY via GenericProxyConfig when set.""" + from youtube_transcript_api import YouTubeTranscriptApi + + proxy = http_proxy_url() + if not proxy: + return YouTubeTranscriptApi() + try: + from youtube_transcript_api.proxies import GenericProxyConfig + + https = settings.HTTPS_PROXY or proxy + return YouTubeTranscriptApi( + proxy_config=GenericProxyConfig(http_url=proxy, https_url=https) + ) + except Exception as exc: + print(f"[proxy] Could not attach GenericProxyConfig ({exc}); continuing without proxy.") + return YouTubeTranscriptApi() diff --git a/backend/app/utils/stripe_sig.py b/backend/app/utils/stripe_sig.py new file mode 100644 index 0000000..af68074 --- /dev/null +++ b/backend/app/utils/stripe_sig.py @@ -0,0 +1,26 @@ +"""Verify Stripe-Signature without adding the Stripe SDK to the backend.""" +import hashlib +import hmac +import time + + +def verify_stripe_signature(payload: bytes, header: str, secret: str, tolerance: int = 300) -> bool: + if not header or not secret: + return False + parsed: dict[str, list[str]] = {} + for part in header.split(","): + key, _, value = part.strip().partition("=") + parsed.setdefault(key, []).append(value) + timestamp = (parsed.get("t") or [None])[0] + signatures = parsed.get("v1") or [] + if not timestamp or not signatures: + return False + try: + ts = int(timestamp) + except ValueError: + return False + if abs(time.time() - ts) > tolerance: + return False + signed = f"{timestamp}.".encode("utf-8") + payload + expected = hmac.new(secret.encode("utf-8"), signed, hashlib.sha256).hexdigest() + return any(hmac.compare_digest(expected, sig) for sig in signatures) From 22a2998c5f38c17b3eaefc6aea4046f911659b43 Mon Sep 17 00:00:00 2001 From: Milbaxter Date: Wed, 2 Sep 2026 11:38:17 +0300 Subject: [PATCH 02/16] Close poly_ prefix Pro hole; name search as keyword overlap. --- backend/app/ai/searcher.py | 30 +++++------ backend/app/utils/auth.py | 103 ++++++++++++------------------------- 2 files changed, 47 insertions(+), 86 deletions(-) diff --git a/backend/app/ai/searcher.py b/backend/app/ai/searcher.py index 2410a4a..7ee990e 100644 --- a/backend/app/ai/searcher.py +++ b/backend/app/ai/searcher.py @@ -1,11 +1,18 @@ -import re +"""Keyword / token-overlap search over transcript segments. + +This is NOT embedding-based semantic search. Scores are exact-phrase bonuses +plus per-token overlap. Swap in a vector index later if you need true semantics. +""" from typing import List from app.models import TranscriptSegment, SearchHit, SearchResponse + class TranscriptSearcher: + SEARCH_METHOD = "keyword_overlap" + def search(self, segments: List[TranscriptSegment], query: str, top_k: int = 10) -> SearchResponse: if not segments or not query.strip(): - return SearchResponse(query=query, total_matches=0, hits=[]) + return SearchResponse(query=query, total_matches=0, hits=[], method=self.SEARCH_METHOD) query_tokens = [q.lower().strip() for q in query.split() if q.strip()] hits: List[SearchHit] = [] @@ -13,38 +20,31 @@ def search(self, segments: List[TranscriptSegment], query: str, top_k: int = 10) for idx, seg in enumerate(segments): text_lower = seg.text.lower() score = 0.0 - - # Exact phrase match bonus + if query.lower() in text_lower: score += 5.0 - # Token match scoring - matched_tokens = 0 for token in query_tokens: if token in text_lower: - matched_tokens += 1 score += 1.0 if score > 0: - # Highlight or format snippet - snippet = seg.text hits.append(SearchHit( segment_index=idx, start=seg.start, end=seg.end, - text=snippet, + text=seg.text, score=score, - formatted_start=seg.formatted_start + formatted_start=seg.formatted_start, )) - # Sort by relevance score descending hits.sort(key=lambda h: h.score, reverse=True) - top_hits = hits[:top_k] - return SearchResponse( query=query, total_matches=len(hits), - hits=top_hits + hits=hits[:top_k], + method=self.SEARCH_METHOD, ) + searcher = TranscriptSearcher() diff --git a/backend/app/utils/auth.py b/backend/app/utils/auth.py index 03df49c..e83956f 100644 --- a/backend/app/utils/auth.py +++ b/backend/app/utils/auth.py @@ -1,99 +1,60 @@ -from datetime import datetime -from typing import Optional, Dict +from typing import Optional from fastapi import Header, HTTPException, status from app.config import settings from app.models import APIKeyInfo +from app.utils.key_store import get_key, increment_usage, mint_key + +PAID_TIERS = {"starter", "pro", "scale", "enterprise"} -# In-memory API key registry with preloaded demo / testing keys -API_KEYS_DB: Dict[str, APIKeyInfo] = { - "poly_free_demo_key": APIKeyInfo( - key="poly_free_demo_key", - tier="free", - monthly_limit=50, - used_this_month=12, - active=True - ), - "poly_starter_live_key": APIKeyInfo( - key="poly_starter_live_key", - tier="starter", - monthly_limit=500, - used_this_month=45, - active=True - ), - "poly_pro_live_key": APIKeyInfo( - key="poly_pro_live_key", - tier="pro", - monthly_limit=3000, - used_this_month=210, - active=True - ), - "poly_scale_live_key": APIKeyInfo( - key="poly_scale_live_key", - tier="scale", - monthly_limit=15000, - used_this_month=1400, - active=True - ) -} async def verify_api_key(x_api_key: Optional[str] = Header(None, alias="X-API-Key")) -> Optional[APIKeyInfo]: """ - Validate API key from request headers. - If no key is provided, returns a free anonymous guest tier object. + Validate API key from request headers against the persistent store. + A missing key is treated as an anonymous free guest (rate-limited). + Keys are NEVER auto-promoted to Pro based on a `poly_` prefix. """ if not x_api_key: return APIKeyInfo( key="guest_anonymous", tier="free", monthly_limit=settings.DEFAULT_FREE_DAILY_LIMIT, - used_this_month=1, - active=True + used_this_month=0, + active=True, ) - # Check key existence - key_info = API_KEYS_DB.get(x_api_key) + key_info = get_key(x_api_key) if not key_info or not key_info.active: - # For open developer flexibility, dynamically register valid prefix keys - if x_api_key.startswith("poly_"): - API_KEYS_DB[x_api_key] = APIKeyInfo( - key=x_api_key, - tier="pro", - monthly_limit=settings.PRO_MONTHLY_LIMIT, - used_this_month=1, - active=True - ) - return API_KEYS_DB[x_api_key] raise HTTPException( status_code=status.HTTP_401_UNAUTHORIZED, - detail="Invalid or revoked API key. Pass 'X-API-Key: poly_free_demo_key' or visit /pricing to generate one." + detail="Invalid or revoked API key. Generate a free key at /api-keys or complete Stripe checkout for a paid tier.", ) - # Check rate limits if key_info.used_this_month >= key_info.monthly_limit: raise HTTPException( status_code=status.HTTP_429_TOO_MANY_REQUESTS, - detail=f"Rate limit exceeded for tier '{key_info.tier}'. Upgrade your tier at /pricing to continue." + detail=f"Rate limit exceeded for tier '{key_info.tier}'. Upgrade at /pricing.", ) - key_info.used_this_month += 1 - return key_info + return increment_usage(x_api_key) or key_info + -def generate_new_api_key(tier: str = "starter") -> APIKeyInfo: - import uuid - new_key = f"poly_{tier}_{uuid.uuid4().hex[:12]}" - limit_map = { - "free": 50, - "starter": settings.STARTER_MONTHLY_LIMIT, - "pro": settings.PRO_MONTHLY_LIMIT, - "scale": settings.SCALE_MONTHLY_LIMIT, - "enterprise": 100000 - } - info = APIKeyInfo( - key=new_key, +def generate_new_api_key( + tier: str = "free", + *, + paid_verified: bool = False, + stripe_session_id: Optional[str] = None, + customer_email: Optional[str] = None, +) -> APIKeyInfo: + """Mint a key. Paid tiers require paid_verified=True (Stripe webhook/fulfill).""" + if tier not in {"free", "starter", "pro", "scale", "enterprise"}: + raise HTTPException(status_code=status.HTTP_400_BAD_REQUEST, detail="Unknown tier.") + if tier in PAID_TIERS and not paid_verified: + raise HTTPException( + status_code=status.HTTP_403_FORBIDDEN, + detail="Paid API keys are issued only after a verified Stripe checkout (webhook). Use tier=free or complete payment.", + ) + return mint_key( tier=tier, - monthly_limit=limit_map.get(tier, 500), - used_this_month=0, - active=True + stripe_session_id=stripe_session_id, + customer_email=customer_email, ) - API_KEYS_DB[new_key] = info - return info From b2f1fe3b4d470b9b119e867e0577f510a55448c4 Mon Sep 17 00:00:00 2001 From: Milbaxter Date: Wed, 2 Sep 2026 11:38:35 +0300 Subject: [PATCH 03/16] Persist API keys in a JSON file store. --- backend/app/utils/key_store.py | 141 +++++++++++++++++++++++++++++++++ 1 file changed, 141 insertions(+) create mode 100644 backend/app/utils/key_store.py diff --git a/backend/app/utils/key_store.py b/backend/app/utils/key_store.py new file mode 100644 index 0000000..92395a7 --- /dev/null +++ b/backend/app/utils/key_store.py @@ -0,0 +1,141 @@ +"""JSON-file API key store. Swap the backend of this module for SQLite later.""" +from __future__ import annotations + +import fcntl +import json +import os +from datetime import datetime, timezone +from typing import Dict, Optional + +from app.config import settings +from app.models import APIKeyInfo + + +def _limits() -> Dict[str, int]: + return { + "free": 50, + "starter": settings.STARTER_MONTHLY_LIMIT, + "pro": settings.PRO_MONTHLY_LIMIT, + "scale": settings.SCALE_MONTHLY_LIMIT, + "enterprise": 100000, + } + + +def _path() -> str: + os.makedirs(os.path.dirname(settings.API_KEYS_PATH) or ".", exist_ok=True) + return settings.API_KEYS_PATH + + +def _empty() -> dict: + return {"keys": {}, "by_session": {}} + + +def _load_unlocked(fh) -> dict: + fh.seek(0) + raw = fh.read() + if not raw: + return _empty() + try: + data = json.loads(raw) + except json.JSONDecodeError: + return _empty() + data.setdefault("keys", {}) + data.setdefault("by_session", {}) + return data + + +def _save_unlocked(fh, data: dict) -> None: + fh.seek(0) + fh.truncate() + json.dump(data, fh, indent=2) + fh.flush() + os.fsync(fh.fileno()) + + +def _with_lock(write: bool): + path = _path() + os.makedirs(os.path.dirname(path) or ".", exist_ok=True) + mode = "r+" if os.path.exists(path) else "w+" + fh = open(path, mode) + fcntl.flock(fh, fcntl.LOCK_EX if write else fcntl.LOCK_SH) + if os.path.getsize(path) == 0: + data = _empty() + if write: + _save_unlocked(fh, data) + else: + data = _load_unlocked(fh) + return fh, data + + +def get_key(key: str) -> Optional[APIKeyInfo]: + fh, data = _with_lock(False) + try: + rec = data["keys"].get(key) + if not rec: + return None + return APIKeyInfo(**{k: rec[k] for k in APIKeyInfo.model_fields if k in rec}) + finally: + fcntl.flock(fh, fcntl.LOCK_UN) + fh.close() + + +def get_by_session(session_id: str) -> Optional[APIKeyInfo]: + fh, data = _with_lock(False) + try: + key = data["by_session"].get(session_id) + if not key: + return None + rec = data["keys"].get(key) + if not rec: + return None + return APIKeyInfo(**{k: rec[k] for k in APIKeyInfo.model_fields if k in rec}) + finally: + fcntl.flock(fh, fcntl.LOCK_UN) + fh.close() + + +def put_key(info: APIKeyInfo) -> APIKeyInfo: + fh, data = _with_lock(True) + try: + rec = info.model_dump() + data["keys"][info.key] = rec + if info.stripe_session_id: + data["by_session"][info.stripe_session_id] = info.key + _save_unlocked(fh, data) + return info + finally: + fcntl.flock(fh, fcntl.LOCK_UN) + fh.close() + + +def increment_usage(key: str) -> Optional[APIKeyInfo]: + fh, data = _with_lock(True) + try: + rec = data["keys"].get(key) + if not rec: + return None + rec["used_this_month"] = int(rec.get("used_this_month") or 0) + 1 + rec["last_used_at"] = datetime.now(timezone.utc).isoformat() + data["keys"][key] = rec + _save_unlocked(fh, data) + return APIKeyInfo(**{k: rec[k] for k in APIKeyInfo.model_fields if k in rec}) + finally: + fcntl.flock(fh, fcntl.LOCK_UN) + fh.close() + + +def mint_key(tier: str, stripe_session_id: Optional[str] = None, customer_email: Optional[str] = None) -> APIKeyInfo: + import uuid + + limits = _limits() + new_key = f"poly_{tier}_{uuid.uuid4().hex[:16]}" + info = APIKeyInfo( + key=new_key, + tier=tier, # type: ignore[arg-type] + monthly_limit=limits.get(tier, 50), + used_this_month=0, + active=True, + stripe_session_id=stripe_session_id, + customer_email=customer_email, + ) + return put_key(info) From 6782e7182ca8c37e2bc4946717cd73c77cd41b3b Mon Sep 17 00:00:00 2001 From: Milbaxter Date: Wed, 2 Sep 2026 11:39:08 +0300 Subject: [PATCH 04/16] Env contract, CORS origins, drop enable_diarization, add job/key models. --- backend/app/config.py | 40 +++++++++++++++++++++++++++++++++------- backend/app/models.py | 36 +++++++++++++++++++++++++++++++++++- 2 files changed, 68 insertions(+), 8 deletions(-) diff --git a/backend/app/config.py b/backend/app/config.py index 43f0fa5..85e3a7b 100644 --- a/backend/app/config.py +++ b/backend/app/config.py @@ -2,6 +2,7 @@ from typing import List, Optional from pydantic_settings import BaseSettings, SettingsConfigDict + class Settings(BaseSettings): model_config = SettingsConfigDict(env_file=".env", env_file_encoding="utf-8", extra="ignore") @@ -10,6 +11,7 @@ class Settings(BaseSettings): APP_VERSION: str = "1.0.0" API_V1_PREFIX: str = "/api/v1" DEBUG: bool = False + APP_URL: str = os.getenv("APP_URL", "https://polytranscript.com") # AI API Keys OPENAI_API_KEY: Optional[str] = os.getenv("OPENAI_API_KEY", "") @@ -21,9 +23,11 @@ class Settings(BaseSettings): DEFAULT_LLM_MODEL: str = "gpt-4o-mini" DEFAULT_FAST_LLM_MODEL: str = "llama-3.3-70b-versatile" DEFAULT_WHISPER_MODEL: str = "whisper-large-v3" + # If True, try openai-whisper / faster-whisper locally when cloud keys are absent. + # Mock/demo transcripts are never returned regardless of this flag. LOCAL_WHISPER_FALLBACK: bool = True - # Proxy Configuration (For high-volume scraping without 429 IP bans) + # Proxy Configuration (passed into youtube-transcript-api + yt-dlp) HTTP_PROXY: Optional[str] = os.getenv("HTTP_PROXY", None) HTTPS_PROXY: Optional[str] = os.getenv("HTTPS_PROXY", None) @@ -33,17 +37,39 @@ class Settings(BaseSettings): PRO_MONTHLY_LIMIT: int = 3000 SCALE_MONTHLY_LIMIT: int = 15000 - # Sponsor Banner Configuration (Replicating YouTubeToTranscript's $11k/mo ad slot model) + # Sponsor Banner — keep on the live site, never polytranscript.dev (that host 500s) SPONSOR_ENABLED: bool = True SPONSOR_TEXT: str = "🚀 Sponsor this slot — Reach 100K+ AI builders & researchers monthly" - SPONSOR_LINK: str = "https://polytranscript.dev/pricing#sponsor" + SPONSOR_LINK: str = os.getenv("SPONSOR_LINK", "/pricing#sponsor") SPONSOR_BADGE: str = "Featured Sponsor" - # Storage / Temp paths - TEMP_STORAGE_DIR: str = "/tmp/polytranscript" + # Storage + TEMP_STORAGE_DIR: str = os.getenv("TEMP_STORAGE_DIR", "/tmp/polytranscript") + DATA_DIR: str = os.getenv("DATA_DIR", "/tmp/polytranscript/data") + API_KEYS_PATH: str = os.getenv("API_KEYS_PATH", "") + + # CORS. Wildcard + credentials is invalid; default to explicit origins. + CORS_ORIGINS: List[str] = [ + "http://localhost:3000", + "http://127.0.0.1:3000", + "https://polytranscript.com", + "https://www.polytranscript.com", + ] + + # Billing — never invent live Stripe keys. Operator must set these. + STRIPE_WEBHOOK_SECRET: Optional[str] = os.getenv("STRIPE_WEBHOOK_SECRET", "") + STRIPE_SECRET_KEY: Optional[str] = os.getenv("STRIPE_SECRET_KEY", "") + STRIPE_PRICE_STARTER_MONTHLY: Optional[str] = os.getenv("STRIPE_PRICE_STARTER_MONTHLY", "") + STRIPE_PRICE_STARTER_YEARLY: Optional[str] = os.getenv("STRIPE_PRICE_STARTER_YEARLY", "") + STRIPE_PRICE_PRO_MONTHLY: Optional[str] = os.getenv("STRIPE_PRICE_PRO_MONTHLY", "") + STRIPE_PRICE_PRO_YEARLY: Optional[str] = os.getenv("STRIPE_PRICE_PRO_YEARLY", "") + STRIPE_PRICE_SCALE_MONTHLY: Optional[str] = os.getenv("STRIPE_PRICE_SCALE_MONTHLY", "") + STRIPE_PRICE_SCALE_YEARLY: Optional[str] = os.getenv("STRIPE_PRICE_SCALE_YEARLY", "") + STRIPE_PRICE_SPONSOR: Optional[str] = os.getenv("STRIPE_PRICE_SPONSOR", "") - # CORS Allowed Origins - CORS_ORIGINS: List[str] = ["*"] settings = Settings() +if not settings.API_KEYS_PATH: + settings.API_KEYS_PATH = os.path.join(settings.DATA_DIR, "api_keys.json") os.makedirs(settings.TEMP_STORAGE_DIR, exist_ok=True) +os.makedirs(settings.DATA_DIR, exist_ok=True) diff --git a/backend/app/models.py b/backend/app/models.py index bf100c9..d5eb84c 100644 --- a/backend/app/models.py +++ b/backend/app/models.py @@ -4,6 +4,7 @@ PlatformType = Literal["youtube", "tiktok", "podcast", "direct_audio", "file_upload", "unknown"] + class TranscriptSegment(BaseModel): start: float = Field(..., description="Start timestamp in seconds") end: float = Field(..., description="End timestamp in seconds") @@ -18,6 +19,7 @@ def formatted_start(self) -> str: return f"{hours:02d}:{mins:02d}:{secs:02d}" return f"{mins:02d}:{secs:02d}" + class MediaMetadata(BaseModel): title: str = "Untitled Media" author: str = "Unknown Author" @@ -29,6 +31,7 @@ class MediaMetadata(BaseModel): url: str = "" description: Optional[str] = None + class Chapter(BaseModel): start: float end: float @@ -44,6 +47,7 @@ def formatted_start(self) -> str: return f"{hours:02d}:{mins:02d}:{secs:02d}" return f"{mins:02d}:{secs:02d}" + class SummaryResponse(BaseModel): tldr: str key_takeaways: List[str] = [] @@ -51,6 +55,7 @@ class SummaryResponse(BaseModel): soundbites: List[str] = [] social_post: Optional[str] = None + class SearchHit(BaseModel): segment_index: int start: float @@ -59,33 +64,43 @@ class SearchHit(BaseModel): score: float formatted_start: str + class SearchResponse(BaseModel): query: str total_matches: int hits: List[SearchHit] + method: str = Field( + default="keyword_overlap", + description="Search is token/phrase overlap, not embedding-based semantic search.", + ) + class ChatMessage(BaseModel): role: Literal["user", "assistant", "system"] content: str + class ChatRequest(BaseModel): url: Optional[str] = None transcript_text: Optional[str] = None question: str history: List[ChatMessage] = [] + segments: List[TranscriptSegment] = [] + class ChatResponse(BaseModel): answer: str relevant_timestamps: List[Dict[str, Any]] = [] + class TranscribeRequest(BaseModel): url: str = Field(..., description="URL to YouTube, TikTok, Podcast RSS/Episode, or Audio file") language: Optional[str] = Field("en", description="Target or source language code (e.g., 'en', 'auto')") include_chapters: bool = Field(True, description="Generate AI chapters automatically") include_summary: bool = Field(True, description="Generate AI summary automatically") - enable_diarization: bool = Field(False, description="Attempt speaker diarization") output_format: Optional[str] = Field("json", description="Output format: json, srt, vtt, markdown, text") + class TranscriptResponse(BaseModel): metadata: MediaMetadata language: str @@ -98,12 +113,22 @@ class TranscriptResponse(BaseModel): processing_time_ms: float = 0.0 created_at: str = Field(default_factory=lambda: datetime.now(timezone.utc).isoformat()) + +class JobStatusResponse(BaseModel): + job_id: str + status: Literal["queued", "running", "completed", "failed"] + error: Optional[str] = None + result: Optional[TranscriptResponse] = None + created_at: Optional[str] = None + + class SponsorInfo(BaseModel): enabled: bool = True text: str link: str badge: str + class HealthResponse(BaseModel): status: str = "ok" version: str @@ -111,9 +136,18 @@ class HealthResponse(BaseModel): ai_providers: Dict[str, bool] timestamp: str + class APIKeyInfo(BaseModel): key: str tier: Literal["free", "starter", "pro", "scale", "enterprise"] monthly_limit: int used_this_month: int active: bool + stripe_session_id: Optional[str] = None + customer_email: Optional[str] = None + + +class KeyFulfillRequest(BaseModel): + stripe_session_id: str + tier: Literal["starter", "pro", "scale", "enterprise"] + customer_email: Optional[str] = None From 801b86c3b5a697f4f23715fd9f582365aa3f4aaa Mon Sep 17 00:00:00 2001 From: Milbaxter Date: Wed, 2 Sep 2026 11:39:32 +0300 Subject: [PATCH 05/16] Fail closed: real Whisper only, never mock demo transcripts. --- backend/app/ai/transcriber.py | 174 ++++++++++++++++++++++++---------- 1 file changed, 125 insertions(+), 49 deletions(-) diff --git a/backend/app/ai/transcriber.py b/backend/app/ai/transcriber.py index 1958a05..710fe06 100644 --- a/backend/app/ai/transcriber.py +++ b/backend/app/ai/transcriber.py @@ -1,11 +1,35 @@ import os +import shutil import subprocess import tempfile -import time from typing import List, Tuple + from app.config import settings from app.models import TranscriptSegment +DEMO_FORBIDDEN_SNIPPETS = ( + "Spoken audio content is parsed and indexed natively", + "multi-platform media intelligence and agentic workflows", + "Model Context Protocol (MCP) transform unstructured audio", + "Welcome to this TikTok clip", + "agentic architectures and automated multi-modal pipelines", +) + + +def local_whisper_available() -> bool: + try: + import faster_whisper # noqa: F401 + return True + except Exception: + pass + try: + import whisper # noqa: F401 + return True + except Exception: + pass + return shutil.which("whisper") is not None + + class AudioTranscriber: def __init__(self): self.groq_api_key = settings.GROQ_API_KEY @@ -16,39 +40,40 @@ def has_ai_credentials(self) -> bool: async def transcribe_audio_file(self, audio_path: str, language: str = "en") -> Tuple[str, List[TranscriptSegment]]: """ - Transcribe an audio file using Groq Whisper, OpenAI Whisper, or a local audio pipeline. - Returns full text and timestamped segments. + Transcribe an audio file using Groq Whisper, OpenAI Whisper, or a real local Whisper install. + Never returns canned/demo transcript copy. """ if not os.path.exists(audio_path): raise FileNotFoundError(f"Audio file not found: {audio_path}") - # Ensure audio is optimized for Whisper (<25MB, 16kHz mono mp3) processed_path = self._preprocess_audio(audio_path) - # 1. Try Groq Whisper (Ultra fast, cost-effective) if self.groq_api_key: try: return await self._transcribe_with_groq(processed_path, language) except Exception as e: print(f"[Transcriber] Groq failed, trying fallback: {e}") - # 2. Try OpenAI Whisper if self.openai_api_key: try: return await self._transcribe_with_openai(processed_path, language) except Exception as e: print(f"[Transcriber] OpenAI Whisper failed: {e}") - # 3. Fallback: Local Whisper CLI / Mock parser for testing environment - return self._local_or_mock_transcribe(processed_path) + if settings.LOCAL_WHISPER_FALLBACK and local_whisper_available(): + return self._local_whisper_transcribe(processed_path, language) + + raise RuntimeError( + "No transcription provider available. Set GROQ_API_KEY or OPENAI_API_KEY, " + "or install openai-whisper / faster-whisper. Demo/mock transcripts are disabled." + ) def _preprocess_audio(self, input_path: str) -> str: - """Convert any audio/video file to 16kHz mono MP3 for high compression and Whisper compatibility.""" output_path = tempfile.mktemp(suffix=".mp3", dir=settings.TEMP_STORAGE_DIR) cmd = [ "ffmpeg", "-y", "-i", input_path, "-vn", "-ar", "16000", "-ac", "1", "-b:a", "64k", - output_path + output_path, ] try: subprocess.run(cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE, check=True) @@ -60,74 +85,125 @@ def _preprocess_audio(self, input_path: str) -> str: async def _transcribe_with_groq(self, audio_path: str, language: str) -> Tuple[str, List[TranscriptSegment]]: from groq import AsyncGroq client = AsyncGroq(api_key=self.groq_api_key) - + with open(audio_path, "rb") as f: transcription = await client.audio.transcriptions.create( file=(os.path.basename(audio_path), f.read()), model="whisper-large-v3", response_format="verbose_json", language=language if language != "auto" else None, - temperature=0.0 + temperature=0.0, ) - full_text = transcription.text.strip() - segments = [] - raw_segments = getattr(transcription, "segments", []) or [] - for s in raw_segments: - seg_dict = s if isinstance(s, dict) else s.model_dump() - segments.append(TranscriptSegment( - start=float(seg_dict.get("start", 0.0)), - end=float(seg_dict.get("end", 0.0)), - text=seg_dict.get("text", "").strip() - )) - - return full_text, segments + return self._segments_from_whisper_result(transcription) async def _transcribe_with_openai(self, audio_path: str, language: str) -> Tuple[str, List[TranscriptSegment]]: from openai import AsyncOpenAI client = AsyncOpenAI(api_key=self.openai_api_key) - + with open(audio_path, "rb") as f: transcription = await client.audio.transcriptions.create( file=f, model="whisper-1", response_format="verbose_json", language=language if language != "auto" else None, - timestamp_granularities=["segment"] + timestamp_granularities=["segment"], ) - full_text = transcription.text.strip() - segments = [] + return self._segments_from_whisper_result(transcription) + + def _segments_from_whisper_result(self, transcription) -> Tuple[str, List[TranscriptSegment]]: + full_text = (transcription.text or "").strip() + segments: List[TranscriptSegment] = [] raw_segments = getattr(transcription, "segments", []) or [] for s in raw_segments: seg_dict = s if isinstance(s, dict) else s.model_dump() segments.append(TranscriptSegment( start=float(seg_dict.get("start", 0.0)), end=float(seg_dict.get("end", 0.0)), - text=seg_dict.get("text", "").strip() + text=(seg_dict.get("text") or "").strip(), )) - + self._assert_not_demo(full_text) return full_text, segments - def _local_or_mock_transcribe(self, audio_path: str) -> Tuple[str, List[TranscriptSegment]]: - """Generate structured transcript if no external API key is active.""" - # Check audio length via ffprobe - duration = 60.0 + def _local_whisper_transcribe(self, audio_path: str, language: str) -> Tuple[str, List[TranscriptSegment]]: + """Real local Whisper only. Raises if no engine is installed.""" + lang = None if language == "auto" else language + try: - probe = subprocess.run( - ["ffprobe", "-v", "error", "-show_entries", "format=duration", "-of", "default=noprint_wrappers=1:nokey=1", audio_path], - stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True - ) - duration = float(probe.stdout.strip()) - except Exception: - duration = 60.0 - - sample_segments = [ - TranscriptSegment(start=0.0, end=min(15.0, duration), text="Welcome to this episode. Today we are breaking down multi-platform media intelligence and agentic workflows."), - TranscriptSegment(start=min(15.0, duration), end=min(35.0, duration), text="We are exploring how automated transcript extraction and Model Context Protocol (MCP) transform unstructured audio into actionable knowledge."), - TranscriptSegment(start=min(35.0, duration), end=duration, text="By indexing YouTube, TikTok, and podcasts natively, autonomous agents can search soundbites and reason over rich media in real-time.") - ] - full_text = " ".join([s.text for s in sample_segments]) - return full_text, sample_segments + from faster_whisper import WhisperModel + + model = WhisperModel("base", device="cpu", compute_type="int8") + segments_iter, _info = model.transcribe(audio_path, language=lang) + segments: List[TranscriptSegment] = [] + parts = [] + for s in segments_iter: + text = (s.text or "").strip() + if not text: + continue + segments.append(TranscriptSegment(start=float(s.start or 0.0), end=float(s.end or 0.0), text=text)) + parts.append(text) + full_text = " ".join(parts) + self._assert_not_demo(full_text) + return full_text, segments + except ImportError: + pass + + try: + import whisper + + model = whisper.load_model("base") + result = model.transcribe(audio_path, language=lang) + full_text = (result.get("text") or "").strip() + segments = [] + for s in result.get("segments") or []: + segments.append(TranscriptSegment( + start=float(s.get("start", 0.0)), + end=float(s.get("end", 0.0)), + text=(s.get("text") or "").strip(), + )) + self._assert_not_demo(full_text) + return full_text, segments + except ImportError: + pass + + whisper_bin = shutil.which("whisper") + if whisper_bin: + outdir = tempfile.mkdtemp(dir=settings.TEMP_STORAGE_DIR) + cmd = [whisper_bin, audio_path, "--model", "base", "--output_format", "json", "--output_dir", outdir] + if lang: + cmd.extend(["--language", lang]) + proc = subprocess.run(cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True) + if proc.returncode != 0: + raise RuntimeError(f"whisper CLI failed: {proc.stderr[-500:]}") + import json + json_files = [os.path.join(outdir, f) for f in os.listdir(outdir) if f.endswith(".json")] + if not json_files: + raise RuntimeError("whisper CLI produced no JSON output") + with open(json_files[0]) as fh: + result = json.load(fh) + full_text = (result.get("text") or "").strip() + segments = [ + TranscriptSegment( + start=float(s.get("start", 0.0)), + end=float(s.get("end", 0.0)), + text=(s.get("text") or "").strip(), + ) + for s in result.get("segments") or [] + ] + self._assert_not_demo(full_text) + return full_text, segments + + raise RuntimeError( + "Local Whisper is not installed. Set GROQ_API_KEY or OPENAI_API_KEY, " + "or `pip install openai-whisper` / `faster-whisper`. Mock transcripts are disabled." + ) + + def _assert_not_demo(self, full_text: str) -> None: + lower = (full_text or "").lower() + for snippet in DEMO_FORBIDDEN_SNIPPETS: + if snippet.lower() in lower: + raise RuntimeError("Refusing to return canned/demo transcript copy.") + transcriber = AudioTranscriber() From ef3c88a46561c45eacdb742c4a72715a45ab914d Mon Sep 17 00:00:00 2001 From: Milbaxter Date: Wed, 2 Sep 2026 11:39:57 +0300 Subject: [PATCH 06/16] Wire BACKEND_URL in compose, document env contract, list all MCP tools. --- .env.example | 29 +++++++++++++++++++ docker-compose.yml | 18 +++++++++++- smithery.json | 71 +++++++++++++++++++++++++++++++++------------- 3 files changed, 98 insertions(+), 20 deletions(-) create mode 100644 .env.example diff --git a/.env.example b/.env.example new file mode 100644 index 0000000..9c54e33 --- /dev/null +++ b/.env.example @@ -0,0 +1,29 @@ +# Frontend (Next.js server). /api/transcribe reads BACKEND_URL, not NEXT_PUBLIC_API_URL. +BACKEND_URL=http://localhost:8000 +APP_URL=http://localhost:3000 + +# Backend AI (Whisper). Without these, local openai-whisper/faster-whisper must be installed. +OPENAI_API_KEY= +GROQ_API_KEY= + +# YouTube / yt-dlp proxy (passed into youtube-transcript-api + yt-dlp) +HTTP_PROXY= +HTTPS_PROXY= + +# Persistent API key JSON store (upgrade path: swap key_store.py for SQLite) +DATA_DIR=./data +API_KEYS_PATH=./data/api_keys.json + +# Stripe. Yearly checkout requires annual Price IDs. Do not invent live keys. +STRIPE_SECRET_KEY= +STRIPE_WEBHOOK_SECRET= +STRIPE_PRICE_STARTER_MONTHLY= +STRIPE_PRICE_STARTER_YEARLY= +STRIPE_PRICE_PRO_MONTHLY= +STRIPE_PRICE_PRO_YEARLY= +STRIPE_PRICE_SCALE_MONTHLY= +STRIPE_PRICE_SCALE_YEARLY= +STRIPE_PRICE_SPONSOR= + +# Sponsor banner (keep off polytranscript.dev) +SPONSOR_LINK=/pricing#sponsor diff --git a/docker-compose.yml b/docker-compose.yml index 0d0a011..6fe61d5 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -11,6 +11,11 @@ services: - OPENAI_API_KEY=${OPENAI_API_KEY} - GROQ_API_KEY=${GROQ_API_KEY} - SPONSOR_ENABLED=true + - SPONSOR_LINK=/pricing#sponsor + - HTTP_PROXY=${HTTP_PROXY:-} + - HTTPS_PROXY=${HTTPS_PROXY:-} + - STRIPE_WEBHOOK_SECRET=${STRIPE_WEBHOOK_SECRET:-} + - DATA_DIR=/tmp/polytranscript/data volumes: - /tmp/polytranscript:/tmp/polytranscript restart: unless-stopped @@ -22,7 +27,18 @@ services: ports: - "3000:3000" environment: - - NEXT_PUBLIC_API_URL=http://localhost:8000 + # /api/transcribe reads BACKEND_URL (server-side). NEXT_PUBLIC_API_URL is unused by that route. + - BACKEND_URL=http://backend:8000 + - APP_URL=${APP_URL:-http://localhost:3000} + - STRIPE_SECRET_KEY=${STRIPE_SECRET_KEY:-} + - STRIPE_WEBHOOK_SECRET=${STRIPE_WEBHOOK_SECRET:-} + - STRIPE_PRICE_STARTER_MONTHLY=${STRIPE_PRICE_STARTER_MONTHLY:-} + - STRIPE_PRICE_STARTER_YEARLY=${STRIPE_PRICE_STARTER_YEARLY:-} + - STRIPE_PRICE_PRO_MONTHLY=${STRIPE_PRICE_PRO_MONTHLY:-} + - STRIPE_PRICE_PRO_YEARLY=${STRIPE_PRICE_PRO_YEARLY:-} + - STRIPE_PRICE_SCALE_MONTHLY=${STRIPE_PRICE_SCALE_MONTHLY:-} + - STRIPE_PRICE_SCALE_YEARLY=${STRIPE_PRICE_SCALE_YEARLY:-} + - STRIPE_PRICE_SPONSOR=${STRIPE_PRICE_SPONSOR:-} depends_on: - backend restart: unless-stopped diff --git a/smithery.json b/smithery.json index ecfdae7..e95dc3e 100644 --- a/smithery.json +++ b/smithery.json @@ -2,7 +2,7 @@ "$schema": "https://smithery.ai/docs/config/schema.json", "name": "polytranscript", "version": "1.0.0", - "description": "Multi-platform media transcript & audio intelligence MCP server (YouTube + TikTok + Podcasts).", + "description": "Multi-platform media transcript and audio intelligence MCP server (YouTube + TikTok + Podcasts).", "startCommand": { "type": "stdio", "config": { @@ -13,23 +13,13 @@ "tools": [ { "name": "poly_transcribe", - "description": "Transcribe any YouTube video/short, TikTok clip, Apple/Spotify podcast, or direct audio URL into timestamped text.", + "description": "Transcribe any YouTube video/short, TikTok clip, Apple/RSS podcast, or direct audio URL into timestamped text.", "parameters": { "type": "object", "properties": { - "url": { - "type": "string", - "description": "The media URL" - }, - "language": { - "type": "string", - "description": "Language code (default 'en')" - }, - "format": { - "type": "string", - "enum": ["markdown", "text", "json"], - "description": "Output format" - } + "url": { "type": "string", "description": "The media URL" }, + "language": { "type": "string", "description": "Language code (default en)" }, + "format": { "type": "string", "enum": ["markdown", "text", "json"], "description": "Output format" } }, "required": ["url"] } @@ -40,13 +30,56 @@ "parameters": { "type": "object", "properties": { - "url": { - "type": "string", - "description": "The media URL" - } + "url": { "type": "string", "description": "The media URL" } }, "required": ["url"] } + }, + { + "name": "poly_get_chapters", + "description": "Generate timestamped chapters with summaries for any supported media URL.", + "parameters": { + "type": "object", + "properties": { + "url": { "type": "string", "description": "The media URL" } + }, + "required": ["url"] + } + }, + { + "name": "poly_summarize", + "description": "Executive TL;DR, key takeaways, action items, and soundbites from a media URL.", + "parameters": { + "type": "object", + "properties": { + "url": { "type": "string", "description": "The media URL" } + }, + "required": ["url"] + } + }, + { + "name": "poly_search_soundbites", + "description": "Keyword/token-overlap search across a transcript (not embedding semantic search) with timestamps.", + "parameters": { + "type": "object", + "properties": { + "url": { "type": "string", "description": "The media URL" }, + "query": { "type": "string", "description": "Search term or phrase" } + }, + "required": ["url", "query"] + } + }, + { + "name": "poly_ask_media", + "description": "Ask a question about a video or podcast and get a grounded answer with timestamp citations.", + "parameters": { + "type": "object", + "properties": { + "url": { "type": "string", "description": "The media URL" }, + "question": { "type": "string", "description": "The user's question" } + }, + "required": ["url", "question"] + } } ] } From 9cacdad623c520553807486efbb4dab80142acc1 Mon Sep 17 00:00:00 2001 From: Milbaxter Date: Wed, 2 Sep 2026 11:41:30 +0300 Subject: [PATCH 07/16] Pass HTTP_PROXY into TikTok yt-dlp downloads. --- backend/app/parsers/tiktok.py | 32 +++++++++++++------------------- 1 file changed, 13 insertions(+), 19 deletions(-) diff --git a/backend/app/parsers/tiktok.py b/backend/app/parsers/tiktok.py index 580d1e5..ac4349f 100644 --- a/backend/app/parsers/tiktok.py +++ b/backend/app/parsers/tiktok.py @@ -2,19 +2,19 @@ import os import tempfile import asyncio -from typing import Optional, List import yt_dlp -import httpx -from app.models import MediaMetadata, TranscriptResponse, TranscriptSegment +from app.models import MediaMetadata, TranscriptResponse from app.parsers.base import BaseMediaParser from app.ai.transcriber import transcriber from app.config import settings +from app.utils.proxy import ytdlp_proxy_opts TIKTOK_REGEX = re.compile( r'(?:https?://)?(?:www\.|vm\.|vt\.|m\.)?tiktok\.com/(?:@[\w.-]+/video/\d+|[\w-]+|t/\w+)' ) + class TikTokParser(BaseMediaParser): def can_handle(self, url: str) -> bool: return bool(TIKTOK_REGEX.search(url)) @@ -24,6 +24,7 @@ async def extract_metadata(self, url: str) -> MediaMetadata: 'skip_download': True, 'quiet': True, 'no_warnings': True, + **ytdlp_proxy_opts(), } def _fetch(): @@ -31,15 +32,15 @@ def _fetch(): with yt_dlp.YoutubeDL(ydl_opts) as ydl: info = ydl.extract_info(url, download=False) return MediaMetadata( - title=info.get('title') or info.get('description', 'TikTok Video')[:80], + title=info.get('title') or (info.get('description') or 'TikTok Video')[:80], author=info.get('uploader') or info.get('channel', 'TikTok Creator'), - duration_seconds=float(info.get('duration', 0.0)), + duration_seconds=float(info.get('duration', 0.0) or 0.0), thumbnail_url=info.get('thumbnail'), view_count=info.get('view_count'), upload_date=info.get('upload_date'), platform="tiktok", url=url, - description=info.get('description', '')[:500] + description=(info.get('description') or '')[:500] ) except Exception: return MediaMetadata( @@ -67,6 +68,7 @@ async def extract_transcript(self, url: str, language: str = "en") -> Transcript 'no_warnings': True, 'writesubtitles': True, 'allsubtitles': True, + **ytdlp_proxy_opts(), } def _fetch_and_download(): @@ -76,27 +78,18 @@ def _fetch_and_download(): return info, actual_file try: - info, audio_file = await asyncio.to_thread(_fetch_and_download) + _info, audio_file = await asyncio.to_thread(_fetch_and_download) metadata = await metadata_task - - # Check if direct subtitles exist in info - subtitles = info.get('subtitles') or info.get('automatic_captions') or {} - if subtitles and (language in subtitles or 'en' in subtitles): - # Try parsing direct subtitle json if available - sub_lang = language if language in subtitles else 'en' - # Subtitles extracted or fallback to audio transcription - - # Transcribe audio file with Whisper + if not os.path.exists(audio_file): + raise RuntimeError("Failed to download TikTok audio for transcription.") full_text, segments = await transcriber.transcribe_audio_file(audio_file, language=language) - word_count = len(full_text.split()) - return TranscriptResponse( metadata=metadata, language=language, full_text=full_text, segments=segments, source_type="tiktok_whisper_ai", - word_count=word_count + word_count=len(full_text.split()) ) finally: for p in [temp_audio, temp_audio + ".mp3"]: @@ -106,4 +99,5 @@ def _fetch_and_download(): except Exception: pass + tiktok_parser = TikTokParser() From 261660cd378f52afe9018008126203a82b23be18 Mon Sep 17 00:00:00 2001 From: Milbaxter Date: Wed, 2 Sep 2026 11:41:56 +0300 Subject: [PATCH 08/16] Wire HTTP_PROXY into YouTube captions and yt-dlp; captions-only sync path. --- backend/app/parsers/youtube.py | 158 +++++++++++++++++---------------- 1 file changed, 82 insertions(+), 76 deletions(-) diff --git a/backend/app/parsers/youtube.py b/backend/app/parsers/youtube.py index e00f575..f2ee4be 100644 --- a/backend/app/parsers/youtube.py +++ b/backend/app/parsers/youtube.py @@ -4,18 +4,19 @@ import asyncio from typing import Optional, List import yt_dlp -from youtube_transcript_api import YouTubeTranscriptApi from youtube_transcript_api._errors import TranscriptsDisabled, NoTranscriptFound, VideoUnavailable from app.models import MediaMetadata, TranscriptResponse, TranscriptSegment from app.parsers.base import BaseMediaParser from app.ai.transcriber import transcriber from app.config import settings +from app.utils.proxy import ytdlp_proxy_opts, youtube_transcript_api_client YOUTUBE_REGEX = re.compile( r'(?:https?://)?(?:www\.|m\.|music\.)?(?:youtube\.com/(?:watch\?v=|embed/|v/|shorts/)|youtu\.be/)([\w-]{11})' ) + class YouTubeParser(BaseMediaParser): def can_handle(self, url: str) -> bool: return bool(YOUTUBE_REGEX.search(url)) @@ -33,7 +34,8 @@ async def extract_metadata(self, url: str) -> MediaMetadata: 'skip_download': True, 'quiet': True, 'no_warnings': True, - 'extract_flat': False + 'extract_flat': False, + **ytdlp_proxy_opts(), } def _fetch_meta(): @@ -49,7 +51,7 @@ def _fetch_meta(): upload_date=info.get('upload_date'), platform="youtube", url=f"https://www.youtube.com/watch?v={video_id}", - description=info.get('description', '')[:500] + description=(info.get('description') or '')[:500] ) except Exception: return MediaMetadata( @@ -62,93 +64,95 @@ def _fetch_meta(): return await asyncio.to_thread(_fetch_meta) + def _fetch_captions_sync(self, video_id: str, language: str): + try: + ytt = youtube_transcript_api_client() + transcript_list = ytt.list(video_id) + t = None + + for lang_code in [language, 'en', 'en-US', 'en-GB']: + try: + t = transcript_list.find_transcript([lang_code]) + break + except Exception: + continue + + if not t: + for tr in transcript_list: + if not tr.is_generated: + t = tr + break + + if not t: + t = next(iter(transcript_list)) + + raw_data = t.fetch() + return raw_data, t.language_code + except (TranscriptsDisabled, NoTranscriptFound, VideoUnavailable, Exception): + return None, None + + def _captions_to_response(self, raw_captions, lang, language, metadata) -> TranscriptResponse: + segments: List[TranscriptSegment] = [] + text_parts = [] + for item in raw_captions: + if isinstance(item, dict): + start = float(item.get('start', 0.0)) + duration = float(item.get('duration', 0.0)) + text = str(item.get('text', '')).replace('\n', ' ').strip() + else: + start = float(getattr(item, 'start', 0.0)) + duration = float(getattr(item, 'duration', 0.0)) + text = str(getattr(item, 'text', '')).replace('\n', ' ').strip() + + if text: + segments.append(TranscriptSegment( + start=round(start, 2), + end=round(start + duration, 2), + text=text + )) + text_parts.append(text) + + full_text = " ".join(text_parts) + return TranscriptResponse( + metadata=metadata, + language=lang or language, + full_text=full_text, + segments=segments, + source_type="youtube_captions", + word_count=len(full_text.split()) + ) + + async def extract_captions_only(self, url: str, language: str = "en") -> Optional[TranscriptResponse]: + """TimedText/captions only — no audio download. Returns None if captions are missing.""" + video_id = self.extract_video_id(url) + if not video_id: + return None + raw_captions, lang = await asyncio.to_thread(self._fetch_captions_sync, video_id, language) + if not raw_captions: + return None + metadata = await self.extract_metadata(url) + return self._captions_to_response(raw_captions, lang, language, metadata) + async def extract_transcript(self, url: str, language: str = "en") -> TranscriptResponse: video_id = self.extract_video_id(url) if not video_id: raise ValueError(f"Invalid YouTube URL: {url}") - metadata_task = self.extract_metadata(url) - - # 1. Try fast direct caption extraction via YouTubeTranscriptApi - def _get_yt_captions(): - try: - ytt = YouTubeTranscriptApi() - transcript_list = ytt.list(video_id) - t = None - - # Priority 1: Exact requested language - for lang_code in [language, 'en', 'en-US', 'en-GB']: - try: - t = transcript_list.find_transcript([lang_code]) - break - except Exception: - continue - - # Priority 2: Try finding manually created transcript - if not t: - for tr in transcript_list: - if not tr.is_generated: - t = tr - break - - # Priority 3: Any available transcript - if not t: - t = next(iter(transcript_list)) - - raw_data = t.fetch() - actual_lang = t.language_code - return raw_data, actual_lang - except (TranscriptsDisabled, NoTranscriptFound, VideoUnavailable, Exception) as e: - return None, None - - raw_captions, lang = await asyncio.to_thread(_get_yt_captions) - metadata = await metadata_task - - if raw_captions: - segments: List[TranscriptSegment] = [] - text_parts = [] - for item in raw_captions: - if isinstance(item, dict): - start = float(item.get('start', 0.0)) - duration = float(item.get('duration', 0.0)) - text = str(item.get('text', '')).replace('\n', ' ').strip() - else: - start = float(getattr(item, 'start', 0.0)) - duration = float(getattr(item, 'duration', 0.0)) - text = str(getattr(item, 'text', '')).replace('\n', ' ').strip() - - if text: - segments.append(TranscriptSegment( - start=round(start, 2), - end=round(start + duration, 2), - text=text - )) - text_parts.append(text) - - full_text = " ".join(text_parts) - word_count = len(full_text.split()) - - return TranscriptResponse( - metadata=metadata, - language=lang or language, - full_text=full_text, - segments=segments, - source_type="youtube_captions", - word_count=word_count - ) + captions = await self.extract_captions_only(url, language=language) + if captions: + return captions - # 2. Fallback: Download audio stream and transcribe via Whisper + metadata = await self.extract_metadata(url) audio_path = await self._download_youtube_audio(video_id) try: full_text, segments = await transcriber.transcribe_audio_file(audio_path, language=language) - word_count = len(full_text.split()) return TranscriptResponse( metadata=metadata, language=language, full_text=full_text, segments=segments, source_type="whisper_ai_audio", - word_count=word_count + word_count=len(full_text.split()) ) finally: if os.path.exists(audio_path): @@ -168,7 +172,8 @@ async def _download_youtube_audio(self, video_id: str) -> str: 'preferredquality': '64', }], 'quiet': True, - 'no_warnings': True + 'no_warnings': True, + **ytdlp_proxy_opts(), } def _download(): @@ -182,4 +187,5 @@ def _download(): return await asyncio.to_thread(_download) + youtube_parser = YouTubeParser() From e60512bb39dffb79f2ba0fc616b68dd4d58bdb22 Mon Sep 17 00:00:00 2001 From: Milbaxter Date: Wed, 2 Sep 2026 11:42:38 +0300 Subject: [PATCH 09/16] Add webhook key fulfill, async jobs, and CORS fix. --- backend/app/main.py | 254 +++++++++++++++++++++++++++++++------------- 1 file changed, 182 insertions(+), 72 deletions(-) diff --git a/backend/app/main.py b/backend/app/main.py index aaec862..14464ab 100644 --- a/backend/app/main.py +++ b/backend/app/main.py @@ -1,43 +1,81 @@ +import hmac import time import os from datetime import datetime, timezone -from typing import Optional, List -from fastapi import FastAPI, Depends, HTTPException, UploadFile, File, Form, Query +from typing import Optional + +from fastapi import FastAPI, Depends, HTTPException, UploadFile, File, Form, Query, Request, BackgroundTasks, Header from fastapi.middleware.cors import CORSMiddleware from fastapi.responses import PlainTextResponse, JSONResponse from app.config import settings from app.models import ( - TranscribeRequest, TranscriptResponse, Chapter, SummaryResponse, + TranscribeRequest, TranscriptResponse, SearchResponse, ChatRequest, ChatResponse, HealthResponse, SponsorInfo, - APIKeyInfo, MediaMetadata + APIKeyInfo, KeyFulfillRequest ) from app.parsers.universal import universal_parser -from app.parsers.audio_file import audio_file_parser +from app.parsers.youtube import youtube_parser from app.ai.chapterer import chapter_generator from app.ai.summarizer import summarizer from app.ai.searcher import searcher from app.ai.chat import chat_engine +from app.ai.transcriber import local_whisper_available from app.utils.formatters import format_srt, format_vtt, format_markdown, format_llm_prompt -from app.utils.auth import verify_api_key, generate_new_api_key, API_KEYS_DB +from app.utils.auth import verify_api_key, generate_new_api_key, PAID_TIERS +from app.utils import key_store, jobs as job_store +from app.utils.stripe_sig import verify_stripe_signature app = FastAPI( title=settings.APP_NAME, version=settings.APP_VERSION, - description="Multi-platform transcript & audio intelligence API (YouTube + TikTok + Podcasts) with AI chaptering, semantic soundbite search, and Model Context Protocol (MCP) server.", + description="Multi-platform transcript and audio intelligence API with AI chaptering, keyword soundbite search, and MCP.", docs_url="/docs", redoc_url="/redoc" ) -# CORS middleware for Next.js frontend and external developer access +_origins = list(settings.CORS_ORIGINS or []) +_wildcard = any(o.strip() == "*" for o in _origins) app.add_middleware( CORSMiddleware, - allow_origins=settings.CORS_ORIGINS, - allow_credentials=True, + allow_origins=["*"] if _wildcard else _origins, + allow_credentials=not _wildcard, allow_methods=["*"], allow_headers=["*"], ) + +async def _apply_intelligence(transcript: TranscriptResponse, include_chapters: bool, include_summary: bool) -> TranscriptResponse: + if include_chapters and transcript.segments: + transcript.chapters = await chapter_generator.generate_chapters( + transcript.segments, + transcript.full_text + ) + if include_summary and transcript.full_text: + transcript.summary = await summarizer.generate_summary(transcript) + return transcript + + +async def _run_transcription_job(job_id: str) -> None: + payload = job_store.get_payload(job_id) or {} + job_store.set_status(job_id, "running") + start_time = time.time() + try: + transcript = await universal_parser.extract_transcript( + url=payload.get("url"), + language=payload.get("language") or "en", + ) + transcript = await _apply_intelligence( + transcript, + include_chapters=bool(payload.get("include_chapters", True)), + include_summary=bool(payload.get("include_summary", True)), + ) + transcript.processing_time_ms = round((time.time() - start_time) * 1000, 2) + job_store.set_status(job_id, "completed", result=transcript.model_dump()) + except Exception as exc: + job_store.set_status(job_id, "failed", error=str(exc)) + + @app.get("/", tags=["General"]) async def root(): return { @@ -49,24 +87,25 @@ async def root(): "supported_platforms": ["youtube", "tiktok", "podcast", "direct_audio", "file_upload"] } + @app.get("/api/v1/health", response_model=HealthResponse, tags=["General"]) async def health_check(): return HealthResponse( status="ok", version=settings.APP_VERSION, - platforms_supported=["YouTube", "TikTok", "Podcasts (Apple, Spotify, RSS)", "Direct Audio URLs", "File Upload"], + platforms_supported=["YouTube", "TikTok", "Podcasts (Apple, RSS)", "Direct Audio URLs", "File Upload"], ai_providers={ "groq_whisper": bool(settings.GROQ_API_KEY), "openai_whisper": bool(settings.OPENAI_API_KEY), "anthropic": bool(settings.ANTHROPIC_API_KEY), - "local_whisper_engine": True + "local_whisper_engine": local_whisper_available(), }, timestamp=datetime.now(timezone.utc).isoformat() ) + @app.get("/api/v1/sponsor", response_model=SponsorInfo, tags=["Monetization"]) async def get_sponsor_info(): - """Return active sponsor slot info (Replicating YouTubeToTranscript's $11k/mo ad banner).""" return SponsorInfo( enabled=settings.SPONSOR_ENABLED, text=settings.SPONSOR_TEXT, @@ -74,108 +113,178 @@ async def get_sponsor_info(): badge=settings.SPONSOR_BADGE ) + @app.post("/api/v1/keys/generate", response_model=APIKeyInfo, tags=["Monetization"]) -async def generate_api_key(tier: str = Query("starter", enum=["free", "starter", "pro", "scale"])): - """Generate a new developer API key.""" +async def generate_api_key(tier: str = Query("free", enum=["free", "starter", "pro", "scale"])): + """Generate a free developer API key. Paid tiers are minted only by Stripe webhook fulfillment.""" return generate_new_api_key(tier) -@app.post("/api/v1/transcribe", response_model=TranscriptResponse, tags=["Transcription"]) + +@app.post("/api/v1/keys/fulfill", response_model=APIKeyInfo, tags=["Monetization"]) +async def fulfill_paid_key( + body: KeyFulfillRequest, + x_webhook_secret: Optional[str] = Header(None, alias="X-Webhook-Secret"), +): + """Mint a paid key after Stripe checkout. Requires X-Webhook-Secret == STRIPE_WEBHOOK_SECRET.""" + expected = settings.STRIPE_WEBHOOK_SECRET or "" + if not expected: + raise HTTPException(status_code=503, detail="STRIPE_WEBHOOK_SECRET is not configured.") + if not x_webhook_secret or not hmac.compare_digest(x_webhook_secret, expected): + raise HTTPException(status_code=401, detail="Invalid webhook secret.") + existing = key_store.get_by_session(body.stripe_session_id) + if existing: + return existing + return generate_new_api_key( + body.tier, + paid_verified=True, + stripe_session_id=body.stripe_session_id, + customer_email=body.customer_email, + ) + + +@app.get("/api/v1/keys/by-session/{session_id}", response_model=APIKeyInfo, tags=["Monetization"]) +async def key_by_session(session_id: str): + info = key_store.get_by_session(session_id) + if not info: + raise HTTPException(status_code=404, detail="No key issued for this checkout session yet. Wait for the Stripe webhook.") + return info + + +@app.post("/api/v1/webhooks/stripe", tags=["Monetization"]) +async def stripe_webhook(request: Request): + secret = settings.STRIPE_WEBHOOK_SECRET or "" + if not secret: + raise HTTPException(status_code=503, detail="STRIPE_WEBHOOK_SECRET is not configured.") + payload = await request.body() + header = request.headers.get("stripe-signature") or "" + if not verify_stripe_signature(payload, header, secret): + raise HTTPException(status_code=400, detail="Invalid Stripe signature.") + import json + try: + event = json.loads(payload.decode("utf-8")) + except json.JSONDecodeError: + raise HTTPException(status_code=400, detail="Invalid JSON payload.") + if event.get("type") == "checkout.session.completed": + session = event.get("data", {}).get("object") or {} + metadata = session.get("metadata") or {} + tier = metadata.get("tier") or "starter" + if tier not in PAID_TIERS: + tier = "starter" + session_id = session.get("id") + email = session.get("customer_email") or (session.get("customer_details") or {}).get("email") + if session_id: + existing = key_store.get_by_session(session_id) + if not existing: + generate_new_api_key( + tier, + paid_verified=True, + stripe_session_id=session_id, + customer_email=email, + ) + return {"received": True} + + +@app.post("/api/v1/transcribe", tags=["Transcription"]) async def transcribe_media( request: TranscribeRequest, - api_key: Optional[APIKeyInfo] = Depends(verify_api_key) + background_tasks: BackgroundTasks, + api_key: Optional[APIKeyInfo] = Depends(verify_api_key), ): - """ - Transcribe any YouTube, TikTok, Podcast, or Audio URL into timestamped text, - with automatic AI chaptering and executive summary. - """ + """YouTube captions return 200. Whisper/download returns 202 + job_id. Never canned demo copy.""" start_time = time.time() try: - # 1. Parse and extract transcript - transcript = await universal_parser.extract_transcript( - url=request.url, - language=request.language or "en" - ) - - # 2. AI Chaptering (if requested) - if request.include_chapters and transcript.segments: - transcript.chapters = await chapter_generator.generate_chapters( - transcript.segments, - transcript.full_text + if youtube_parser.can_handle(request.url): + captions = await youtube_parser.extract_captions_only( + request.url, language=request.language or "en" ) + if captions: + captions = await _apply_intelligence( + captions, request.include_chapters, request.include_summary + ) + captions.processing_time_ms = round((time.time() - start_time) * 1000, 2) + return captions - # 3. AI Summary (if requested) - if request.include_summary and transcript.full_text: - transcript.summary = await summarizer.generate_summary(transcript) + job = job_store.create_job(request.model_dump()) + background_tasks.add_task(_run_transcription_job, job["job_id"]) + return JSONResponse( + { + "job_id": job["job_id"], + "status": "queued", + "created_at": job["created_at"], + }, + status_code=202, + ) + except HTTPException: + raise + except Exception as e: + raise HTTPException(status_code=502, detail=f"Transcription failed: {str(e)}") - # 4. Processing metrics - transcript.processing_time_ms = round((time.time() - start_time) * 1000, 2) - return transcript - except Exception as e: - raise HTTPException(status_code=400, detail=f"Transcription failed: {str(e)}") +@app.get("/api/v1/jobs/{job_id}", tags=["Transcription"]) +async def get_job(job_id: str): + rec = job_store.get_job(job_id) + if not rec: + raise HTTPException(status_code=404, detail="Unknown job_id.") + return rec + -@app.post("/api/v1/upload", response_model=TranscriptResponse, tags=["Transcription"]) +@app.post("/api/v1/upload", tags=["Transcription"]) async def upload_and_transcribe( + background_tasks: BackgroundTasks, file: UploadFile = File(...), language: str = Form("en"), include_chapters: bool = Form(True), include_summary: bool = Form(True), - api_key: Optional[APIKeyInfo] = Depends(verify_api_key) + api_key: Optional[APIKeyInfo] = Depends(verify_api_key), ): - """Upload and transcribe local audio/video file.""" - start_time = time.time() + """Upload and transcribe local audio/video file (async job).""" temp_path = os.path.join(settings.TEMP_STORAGE_DIR, f"upload_{file.filename}") try: with open(temp_path, "wb") as f: f.write(await file.read()) + except Exception as e: + raise HTTPException(status_code=500, detail=f"Failed to store upload: {e}") - transcript = await audio_file_parser.extract_transcript(temp_path, language=language) - - if include_chapters and transcript.segments: - transcript.chapters = await chapter_generator.generate_chapters(transcript.segments, transcript.full_text) - - if include_summary and transcript.full_text: - transcript.summary = await summarizer.generate_summary(transcript) + job = job_store.create_job({ + "url": temp_path, + "language": language, + "include_chapters": include_chapters, + "include_summary": include_summary, + "source": "upload", + }) + background_tasks.add_task(_run_transcription_job, job["job_id"]) + return JSONResponse( + {"job_id": job["job_id"], "status": "queued", "created_at": job["created_at"]}, + status_code=202, + ) - transcript.processing_time_ms = round((time.time() - start_time) * 1000, 2) - return transcript - finally: - if os.path.exists(temp_path): - try: - os.remove(temp_path) - except Exception: - pass @app.post("/api/v1/search", response_model=SearchResponse, tags=["AI Intelligence"]) -async def search_transcript( - query: str = Query(..., description="Query phrase or keywords to find in audio"), - segments: List[dict] = Form(None) -): - """Search for exact soundbites and timestamp moments within transcript segments.""" +async def search_transcript(body: dict): + """Keyword/token-overlap search (not embedding semantic search) over provided segments.""" from app.models import TranscriptSegment - segs = [TranscriptSegment(**s) for s in segments] if segments else [] + query = body.get("query") or "" + raw_segments = body.get("segments") or [] + segs = [TranscriptSegment(**s) for s in raw_segments] return searcher.search(segs, query) + @app.post("/api/v1/chat", response_model=ChatResponse, tags=["AI Intelligence"]) async def chat_with_media(request: ChatRequest): - """Ask questions directly grounded on the video/podcast transcript with timestamp citations.""" - from app.models import TranscriptSegment - # If URL provided, transcribe first or use provided text full_text = request.transcript_text or "" - segments = [] - if request.url: + segments = list(request.segments or []) + if request.url and not full_text: t = await universal_parser.extract_transcript(request.url) full_text = t.full_text segments = t.segments - return await chat_engine.answer_question(request.question, full_text, segments) + @app.post("/api/v1/export/{format_type}", tags=["Export"]) async def export_transcript( format_type: str, transcript: TranscriptResponse ): - """Export transcript to Markdown, SRT, VTT, LLM Prompt, or JSON.""" fmt = format_type.lower() if fmt == "srt": return PlainTextResponse(format_srt(transcript.segments), media_type="text/plain") @@ -190,6 +299,7 @@ async def export_transcript( else: return PlainTextResponse(transcript.full_text, media_type="text/plain") + if __name__ == "__main__": import uvicorn uvicorn.run("app.main:app", host="0.0.0.0", port=8000, reload=True) From 1b4058ce41021a3fbf357c927231044f2eeb4a10 Mon Sep 17 00:00:00 2001 From: Milbaxter Date: Wed, 2 Sep 2026 11:43:51 +0300 Subject: [PATCH 10/16] Reject Spotify DRM, select requested RSS episode, pass HTTP_PROXY. --- backend/app/parsers/podcast.py | 170 ++++++++++++++++++++++++--------- 1 file changed, 127 insertions(+), 43 deletions(-) diff --git a/backend/app/parsers/podcast.py b/backend/app/parsers/podcast.py index 7d85e17..8bd0677 100644 --- a/backend/app/parsers/podcast.py +++ b/backend/app/parsers/podcast.py @@ -2,21 +2,30 @@ import os import tempfile import asyncio -from typing import Optional, List +from typing import Optional +from urllib.parse import parse_qs, urlparse import feedparser import httpx import yt_dlp -from app.models import MediaMetadata, TranscriptResponse, TranscriptSegment +from app.models import MediaMetadata, TranscriptResponse from app.parsers.base import BaseMediaParser from app.ai.transcriber import transcriber from app.config import settings +from app.utils.proxy import ytdlp_proxy_opts, httpx_proxy_kw APPLE_PODCAST_REGEX = re.compile(r'https?://podcasts\.apple\.com/[\w-]+/podcast/[^/]+/id(\d+)(?:\?i=(\d+))?') -SPOTIFY_REGEX = re.compile(r'https?://open\.spotify\.com/episode/([a-zA-Z0-9]+)') +SPOTIFY_REGEX = re.compile(r'https?://open\.spotify\.com/(episode|show)/([a-zA-Z0-9]+)') DIRECT_AUDIO_REGEX = re.compile(r'https?://.+\.(?:mp3|m4a|wav|ogg|aac|flac|webm)(?:\?.*)?$', re.IGNORECASE) RSS_REGEX = re.compile(r'https?://.+/(?:feed|rss|podcast|\.xml)', re.IGNORECASE) +SPOTIFY_DRM_MSG = ( + "Spotify episodes are DRM-protected and cannot be downloaded as MP3. " + "Provide an Apple Podcasts episode link, a public RSS episode permalink " + "(or ?episode=), or a direct MP3/M4A enclosure URL." +) + + class PodcastParser(BaseMediaParser): def can_handle(self, url: str) -> bool: return bool( @@ -29,8 +38,12 @@ def can_handle(self, url: str) -> bool: url.endswith(".mp3") ) + def _reject_spotify(self, url: str) -> None: + if SPOTIFY_REGEX.search(url): + raise ValueError(SPOTIFY_DRM_MSG) + async def extract_metadata(self, url: str) -> MediaMetadata: - # 1. Apple Podcast URL + self._reject_spotify(url) apple_match = APPLE_PODCAST_REGEX.search(url) if apple_match: show_id = apple_match.group(1) @@ -38,14 +51,10 @@ async def extract_metadata(self, url: str) -> MediaMetadata: meta = await self._fetch_apple_podcast_meta(show_id, episode_id, url) if meta: return meta - - # 2. RSS Feed URL if RSS_REGEX.search(url) or url.endswith(".xml") or "feed" in url: meta = await self._fetch_rss_meta(url) if meta: return meta - - # 3. Direct Audio URL audio_name = os.path.basename(url.split("?")[0]) return MediaMetadata( title=audio_name.replace("-", " ").replace("_", " ").title(), @@ -57,13 +66,14 @@ async def extract_metadata(self, url: str) -> MediaMetadata: async def _fetch_apple_podcast_meta(self, show_id: str, episode_id: Optional[str], original_url: str) -> Optional[MediaMetadata]: try: lookup_url = f"https://itunes.apple.com/lookup?id={episode_id or show_id}&entity=podcastEpisode" - async with httpx.AsyncClient(timeout=10.0) as client: + async with httpx.AsyncClient(timeout=10.0, **httpx_proxy_kw()) as client: resp = await client.get(lookup_url) if resp.status_code == 200: data = resp.json() results = data.get("results", []) if results: item = results[0] + audio = item.get("episodeUrl") or original_url return MediaMetadata( title=item.get("trackName") or item.get("collectionName", "Apple Podcast Episode"), author=item.get("artistName", "Unknown Host"), @@ -71,73 +81,143 @@ async def _fetch_apple_podcast_meta(self, show_id: str, episode_id: Optional[str thumbnail_url=item.get("artworkUrl600") or item.get("artworkUrl100"), upload_date=item.get("releaseDate"), platform="podcast", - url=item.get("episodeUrl") or original_url, - description=item.get("description", "")[:500] + url=audio, + description=(item.get("description") or "")[:500] ) except Exception as e: print(f"[PodcastParser] Apple lookup error: {e}") return None + def _select_rss_entry(self, feed, requested_url: str): + parsed = urlparse(requested_url) + qs = parse_qs(parsed.query) + hints = [] + for key in ("i", "episode", "guid", "id"): + if qs.get(key): + hints.extend(qs[key]) + if parsed.fragment: + hints.append(parsed.fragment) + + def _entry_urls(entry): + urls = [entry.get("link") or "", str(entry.get("id") or ""), str(entry.get("guid") or "")] + for enc in entry.get("enclosures") or []: + urls.append(enc.get("href") or "") + return urls + + for entry in feed.entries: + for candidate in _entry_urls(entry): + if candidate and (candidate == requested_url or requested_url in candidate or candidate in requested_url): + if candidate.endswith((".mp3", ".m4a", ".ogg", ".wav")) or requested_url.endswith((".mp3", ".m4a")): + return entry + if entry.get("link") == requested_url or str(entry.get("id") or "") == requested_url: + return entry + + if hints: + lowered = [h.lower() for h in hints] + for entry in feed.entries: + blob = " ".join(_entry_urls(entry) + [entry.get("title") or ""]).lower() + if any(h in blob for h in lowered if h): + return entry + + looks_like_feed = bool( + RSS_REGEX.search(requested_url) + or requested_url.endswith(".xml") + or "/feed" in requested_url + or requested_url.rstrip("/").endswith("rss") + ) + if looks_like_feed: + titles = [] + for entry in feed.entries[:5]: + titles.append(f"- {entry.get('title', 'untitled')} ({entry.get('link') or 'no permalink'})") + listing = "\n".join(titles) if titles else "(no entries)" + raise ValueError( + "This looks like an RSS feed URL, not a specific episode. " + "Pass an episode permalink, enclosure MP3, or ?episode=. " + f"Recent entries:\n{listing}" + ) + + raise ValueError( + "Could not identify the requested episode in this RSS feed. " + "Pass a specific episode permalink, enclosure URL, or ?episode=." + ) + async def _fetch_rss_meta(self, feed_url: str) -> Optional[MediaMetadata]: def _parse(): feed = feedparser.parse(feed_url) - if feed.entries: - latest = feed.entries[0] - show_title = feed.feed.get("title", "Podcast Show") - ep_title = latest.get("title", "Latest Episode") - author = feed.feed.get("author", "Podcast Host") - image = feed.feed.get("image", {}).get("href") - audio_url = None - for enc in latest.get("enclosures", []): - if "audio" in enc.get("type", "") or enc.get("href", "").endswith(".mp3"): - audio_url = enc.get("href") - break - - return MediaMetadata( - title=f"{show_title}: {ep_title}", - author=author, - thumbnail_url=image, - platform="podcast", - url=audio_url or feed_url, - description=latest.get("summary", "")[:500] - ) - return None + if not feed.entries: + return None + entry = self._select_rss_entry(feed, feed_url) + show_title = feed.feed.get("title", "Podcast Show") + ep_title = entry.get("title", "Episode") + author = feed.feed.get("author", "Podcast Host") + image = feed.feed.get("image", {}).get("href") + audio_url = None + for enc in entry.get("enclosures", []): + href = enc.get("href") or "" + if "audio" in enc.get("type", "") or href.lower().endswith((".mp3", ".m4a", ".ogg", ".wav")): + audio_url = href + break + if not audio_url: + raise ValueError(f"Episode '{ep_title}' has no audio enclosure.") + return MediaMetadata( + title=f"{show_title}: {ep_title}", + author=author, + thumbnail_url=image, + platform="podcast", + url=audio_url, + description=(entry.get("summary") or "")[:500] + ) return await asyncio.to_thread(_parse) async def extract_transcript(self, url: str, language: str = "en") -> TranscriptResponse: + self._reject_spotify(url) metadata = await self.extract_metadata(url) resolved_audio_url = metadata.url or url + parsed = urlparse(resolved_audio_url) + if "spotify.com" in (parsed.netloc or ""): + raise ValueError(SPOTIFY_DRM_MSG) - # Download audio stream to temporary file temp_audio = tempfile.mktemp(suffix=".mp3", dir=settings.TEMP_STORAGE_DIR) - - async with httpx.AsyncClient(timeout=60.0, follow_redirects=True) as client: + async with httpx.AsyncClient(timeout=60.0, follow_redirects=True, **httpx_proxy_kw()) as client: try: - # If direct audio stream async with client.stream("GET", resolved_audio_url) as response: - if response.status_code == 200: + content_type = (response.headers.get("content-type") or "").lower() + if response.status_code == 200 and ( + "audio" in content_type + or "octet-stream" in content_type + or re.search(r'\.(mp3|m4a|wav|ogg|aac|flac)(\?|$)', resolved_audio_url, re.I) + ): + if "html" in content_type: + raise ValueError( + f"URL returned HTML, not audio ({resolved_audio_url}). " + "Spotify/web pages cannot be fetched as MP3." + ) with open(temp_audio, "wb") as f: async for chunk in response.aiter_bytes(chunk_size=1024 * 64): f.write(chunk) else: - # Fallback to yt-dlp to download await self._download_with_ytdlp(resolved_audio_url, temp_audio) + except ValueError: + raise except Exception: - # Fallback to yt-dlp await self._download_with_ytdlp(resolved_audio_url, temp_audio) try: actual_file = temp_audio if os.path.exists(temp_audio) else temp_audio + ".mp3" + if not os.path.exists(actual_file) or os.path.getsize(actual_file) < 64: + raise RuntimeError("Failed to download audio for transcription.") + with open(actual_file, "rb") as fh: + head = fh.read(64).lstrip().lower() + if head.startswith(b" Transcript pass async def _download_with_ytdlp(self, url: str, output_path: str): + if "spotify.com" in url: + raise ValueError(SPOTIFY_DRM_MSG) ydl_opts = { 'format': 'bestaudio/best', 'outtmpl': output_path.replace(".mp3", ""), @@ -157,11 +239,13 @@ async def _download_with_ytdlp(self, url: str, output_path: str): 'preferredquality': '64', }], 'quiet': True, - 'no_warnings': True + 'no_warnings': True, + **ytdlp_proxy_opts(), } def _exec(): with yt_dlp.YoutubeDL(ydl_opts) as ydl: ydl.download([url]) await asyncio.to_thread(_exec) + podcast_parser = PodcastParser() From 4cf354b90330da3a47f7d6c28d34fff5bb046d59 Mon Sep 17 00:00:00 2001 From: Milbaxter Date: Wed, 2 Sep 2026 11:44:31 +0300 Subject: [PATCH 11/16] Add P0 tests; stop unverified key mint; send yearly Stripe Price IDs. --- backend/tests/test_api.py | 15 +- backend/tests/test_honest_pipeline.py | 151 ++++++++++++++++++++ frontend/src/app/api/checkout/route.ts | 67 +++++---- frontend/src/app/api/keys/generate/route.ts | 41 +++--- 4 files changed, 225 insertions(+), 49 deletions(-) create mode 100644 backend/tests/test_honest_pipeline.py diff --git a/backend/tests/test_api.py b/backend/tests/test_api.py index 2c2746d..40c6a4c 100644 --- a/backend/tests/test_api.py +++ b/backend/tests/test_api.py @@ -4,6 +4,7 @@ client = TestClient(app) + def test_root_and_health(): res = client.get("/") assert res.status_code == 200 @@ -14,18 +15,22 @@ def test_root_and_health(): data = res_health.json() assert data["status"] == "ok" assert "YouTube" in data["platforms_supported"][0] + assert "local_whisper_engine" in data["ai_providers"] + assert isinstance(data["ai_providers"]["local_whisper_engine"], bool) + def test_sponsor_endpoint(): res = client.get("/api/v1/sponsor") assert res.status_code == 200 data = res.json() assert data["enabled"] is True - assert "Sponsor" in data["badge"] or "Featured" in data["badge"] + assert "polytranscript.dev" not in (data.get("link") or "") + -def test_generate_api_key(): - res = client.post("/api/v1/keys/generate?tier=starter") +def test_generate_free_api_key(): + res = client.post("/api/v1/keys/generate?tier=free") assert res.status_code == 200 data = res.json() - assert data["tier"] == "starter" - assert data["key"].startswith("poly_starter_") + assert data["tier"] == "free" + assert data["key"].startswith("poly_free_") assert data["monthly_limit"] > 0 diff --git a/backend/tests/test_honest_pipeline.py b/backend/tests/test_honest_pipeline.py new file mode 100644 index 0000000..c6185dc --- /dev/null +++ b/backend/tests/test_honest_pipeline.py @@ -0,0 +1,151 @@ +"""P0 tests: no demo transcripts; unverified requests cannot mint Pro keys.""" +import asyncio + +import pytest +from fastapi.testclient import TestClient + +from app.main import app +from app.ai.transcriber import AudioTranscriber, DEMO_FORBIDDEN_SNIPPETS +from app.config import settings +from app.parsers.podcast import podcast_parser +from app.parsers.youtube import youtube_parser +from app.utils.proxy import ytdlp_proxy_opts + +client = TestClient(app) + +DEMO_SENTENCES = list(DEMO_FORBIDDEN_SNIPPETS) + [ + "This transcript was extracted and processed via PolyTranscript multi-platform intelligence", + "Welcome to this episode. Today we are breaking down multi-platform media intelligence", +] + + +def test_forged_poly_prefix_is_not_pro(): + res = client.post( + "/api/v1/transcribe", + json={"url": "https://www.youtube.com/watch?v=dQw4w9WgXcQ", "include_chapters": False, "include_summary": False}, + headers={"X-API-Key": "poly_pro_forgedkey123456"}, + ) + assert res.status_code == 401 + body = res.json() + assert "detail" in body + blob = str(body).lower() + for snippet in DEMO_SENTENCES: + assert snippet.lower() not in blob + + +@pytest.mark.parametrize("tier", ["starter", "pro", "scale"]) +def test_cannot_mint_paid_key_without_webhook(tier): + res = client.post(f"/api/v1/keys/generate?tier={tier}") + assert res.status_code == 403 + data = res.json() + assert "detail" in data + assert "poly_pro_" not in str(data.get("key") or "") + + +def test_fulfill_rejects_missing_or_wrong_secret(monkeypatch, tmp_path): + monkeypatch.setattr(settings, "STRIPE_WEBHOOK_SECRET", "whsec_test_secret") + monkeypatch.setattr(settings, "API_KEYS_PATH", str(tmp_path / "keys.json")) + res = client.post( + "/api/v1/keys/fulfill", + json={"stripe_session_id": "cs_test_abc", "tier": "pro"}, + ) + assert res.status_code == 401 + + res = client.post( + "/api/v1/keys/fulfill", + json={"stripe_session_id": "cs_test_abc", "tier": "pro"}, + headers={"X-Webhook-Secret": "wrong"}, + ) + assert res.status_code == 401 + + +def test_fulfill_with_secret_mints_pro_and_lookup(monkeypatch, tmp_path): + monkeypatch.setattr(settings, "STRIPE_WEBHOOK_SECRET", "whsec_test_secret") + monkeypatch.setattr(settings, "API_KEYS_PATH", str(tmp_path / "keys.json")) + (tmp_path / "keys.json").write_text('{"keys":{},"by_session":{}}') + + res = client.post( + "/api/v1/keys/fulfill", + json={"stripe_session_id": "cs_test_lookup", "tier": "pro", "customer_email": "a@b.c"}, + headers={"X-Webhook-Secret": "whsec_test_secret"}, + ) + assert res.status_code == 200 + data = res.json() + assert data["tier"] == "pro" + assert data["key"].startswith("poly_pro_") + assert data["active"] is True + + lookup = client.get("/api/v1/keys/by-session/cs_test_lookup") + assert lookup.status_code == 200 + assert lookup.json()["key"] == data["key"] + + res2 = client.post( + "/api/v1/keys/fulfill", + json={"stripe_session_id": "cs_test_lookup", "tier": "pro"}, + headers={"X-Webhook-Secret": "whsec_test_secret"}, + ) + assert res2.json()["key"] == data["key"] + + +def test_transcriber_never_returns_demo_sentences(tmp_path, monkeypatch): + t = AudioTranscriber() + t.groq_api_key = "" + t.openai_api_key = "" + monkeypatch.setattr("app.ai.transcriber.settings.LOCAL_WHISPER_FALLBACK", False) + monkeypatch.setattr("app.ai.transcriber.local_whisper_available", lambda: False) + audio = tmp_path / "silence.mp3" + audio.write_bytes(b"ID3fake") + monkeypatch.setattr(t, "_preprocess_audio", lambda p: str(audio)) + + with pytest.raises(RuntimeError) as ei: + asyncio.run(t.transcribe_audio_file(str(audio))) + err = str(ei.value) + for snippet in DEMO_SENTENCES: + assert snippet.lower() not in err.lower() + assert "demo" in err.lower() or "GROQ_API_KEY" in err or "provider" in err.lower() + + +def test_spotify_is_rejected_as_drm(): + url = "https://open.spotify.com/episode/7makk4oTQel546B6mAzWCO" + assert podcast_parser.can_handle(url) + with pytest.raises(ValueError) as ei: + asyncio.run(podcast_parser.extract_metadata(url)) + assert "DRM" in str(ei.value) + + +def test_rss_selects_requested_episode_not_first(): + import feedparser + rss = ( + '' + "" + "Demo Show" + "Episode Oneep-1" + "https://example.com/ep1" + "" + "Episode Twoep-2" + "https://example.com/ep2" + "" + "" + ) + feed = feedparser.parse(rss) + with pytest.raises(ValueError) as ei: + podcast_parser._select_rss_entry(feed, "https://example.com/feed.xml") + msg = str(ei.value).lower() + assert "episode" in msg or "permalink" in msg + + chosen = podcast_parser._select_rss_entry(feed, "https://example.com/ep2") + assert chosen.get("title") == "Episode Two" + + chosen2 = podcast_parser._select_rss_entry(feed, "https://example.com/feed.xml?episode=ep-1") + assert chosen2.get("title") == "Episode One" + + +def test_ytdlp_proxy_opts_honor_http_proxy(monkeypatch): + monkeypatch.setattr(settings, "HTTP_PROXY", "http://127.0.0.1:8888") + monkeypatch.setattr(settings, "HTTPS_PROXY", None) + opts = ytdlp_proxy_opts() + assert opts.get("proxy") == "http://127.0.0.1:8888" + + +def test_youtube_captions_method_exists(): + assert hasattr(youtube_parser, "extract_captions_only") diff --git a/frontend/src/app/api/checkout/route.ts b/frontend/src/app/api/checkout/route.ts index f697409..1205f2b 100644 --- a/frontend/src/app/api/checkout/route.ts +++ b/frontend/src/app/api/checkout/route.ts @@ -3,52 +3,65 @@ import Stripe from 'stripe'; const STRIPE_SECRET_KEY = process.env.STRIPE_SECRET_KEY || ''; const stripe = STRIPE_SECRET_KEY ? new Stripe(STRIPE_SECRET_KEY) : null; +const APP_URL = (process.env.APP_URL || process.env.NEXT_PUBLIC_APP_URL || 'https://polytranscript.com').replace(/\/$/, ''); -const PRICE_MAP: Record = { - starter: 'price_1UB8uhCr8oInGVYkVVsC27db', - pro: 'price_1UB8ujCr8oInGVYk64rfLsJk', - scale: 'price_1UB8ukCr8oInGVYk7rgDj98D', - sponsor: 'price_1UB8umCr8oInGVYkHqQcO6Ig', -}; +type Cycle = 'monthly' | 'yearly'; -const LINK_MAP: Record = { - starter: 'https://buy.stripe.com/3cI00i47Ad0K7qP5re2880k', - pro: 'https://buy.stripe.com/9B6cN48nQaSC9yXaLy2880l', - scale: 'https://buy.stripe.com/dRm14m1Zsf8S5iH1aY2880m', - sponsor: 'https://buy.stripe.com/5kQbJ0eMeaSC8uTcTG2880n', -}; +function priceIdFor(tier: string, cycle: Cycle): string | undefined { + const envKey = `STRIPE_PRICE_${tier.toUpperCase()}_${cycle.toUpperCase()}`; + const fromEnv = process.env[envKey]; + if (fromEnv) return fromEnv; + if (cycle === 'monthly') { + const legacy: Record = { + starter: process.env.STRIPE_PRICE_STARTER_MONTHLY || 'price_1UB8uhCr8oInGVYkVVsC27db', + pro: process.env.STRIPE_PRICE_PRO_MONTHLY || 'price_1UB8ujCr8oInGVYk64rfLsJk', + scale: process.env.STRIPE_PRICE_SCALE_MONTHLY || 'price_1UB8ukCr8oInGVYk7rgDj98D', + sponsor: process.env.STRIPE_PRICE_SPONSOR || 'price_1UB8umCr8oInGVYkHqQcO6Ig', + }; + return legacy[tier]; + } + return undefined; +} export async function POST(req: NextRequest) { - let tier = 'pro'; try { const body = await req.json(); - tier = body.tier || 'pro'; - const priceId = PRICE_MAP[tier]; + const tier = (body.tier || 'pro') as string; + const billingCycle: Cycle = body.billingCycle === 'yearly' ? 'yearly' : 'monthly'; - if (!priceId) { + if (!['starter', 'pro', 'scale', 'sponsor'].includes(tier)) { return NextResponse.json({ error: 'Invalid tier' }, { status: 400 }); } + const priceId = priceIdFor(tier, billingCycle); + if (!priceId) { + return NextResponse.json( + { + error: `Missing Stripe Price ID for ${tier} ${billingCycle}. Set STRIPE_PRICE_${tier.toUpperCase()}_${billingCycle.toUpperCase()} in the environment. Yearly checkout cannot reuse monthly Price IDs.`, + }, + { status: 400 } + ); + } + if (!stripe) { - return NextResponse.json({ url: LINK_MAP[tier] }); + return NextResponse.json( + { error: 'STRIPE_SECRET_KEY is not configured. Cannot start checkout.' }, + { status: 503 } + ); } const session = await stripe.checkout.sessions.create({ mode: 'subscription', payment_method_types: ['card'], - line_items: [ - { - price: priceId, - quantity: 1, - }, - ], - success_url: `https://polytranscript.com/api-keys?tier=${tier}&paid=true&session_id={CHECKOUT_SESSION_ID}`, - cancel_url: `https://polytranscript.com/pricing`, + line_items: [{ price: priceId, quantity: 1 }], + metadata: { tier, billing_cycle: billingCycle }, + success_url: `${APP_URL}/api-keys?tier=${tier}&session_id={CHECKOUT_SESSION_ID}`, + cancel_url: `${APP_URL}/pricing`, }); - return NextResponse.json({ url: session.url || LINK_MAP[tier] }); + return NextResponse.json({ url: session.url }); } catch (err: any) { console.error('Stripe Checkout Error:', err); - return NextResponse.json({ url: LINK_MAP[tier] || 'https://polytranscript.com/pricing' }); + return NextResponse.json({ error: err.message || 'Checkout failed' }, { status: 502 }); } } diff --git a/frontend/src/app/api/keys/generate/route.ts b/frontend/src/app/api/keys/generate/route.ts index c94cefe..b938369 100644 --- a/frontend/src/app/api/keys/generate/route.ts +++ b/frontend/src/app/api/keys/generate/route.ts @@ -2,23 +2,30 @@ import { NextRequest, NextResponse } from 'next/server'; export async function POST(req: NextRequest) { const { searchParams } = new URL(req.url); - const tier = searchParams.get('tier') || 'starter'; - - const limits: Record = { - free: 50, - starter: 500, - pro: 3000, - scale: 15000, - }; + const tier = (searchParams.get('tier') || 'free').toLowerCase(); - const randomSuffix = Math.random().toString(36).substring(2, 14); - const key = `poly_${tier}_${randomSuffix}`; + if (tier !== 'free') { + return NextResponse.json( + { + detail: 'Paid API keys are issued only after a verified Stripe checkout webhook. Generate a free key or complete payment on /pricing.', + }, + { status: 403 } + ); + } - return NextResponse.json({ - key, - tier, - monthly_limit: limits[tier] || 500, - used_this_month: 0, - active: true, - }); + const backend = process.env.BACKEND_URL; + if (!backend) { + return NextResponse.json( + { detail: 'BACKEND_URL is not configured. Free keys must be minted by the FastAPI key store.' }, + { status: 503 } + ); + } + + try { + const resp = await fetch(`${backend.replace(/\/$/, '')}/api/v1/keys/generate?tier=free`, { method: 'POST' }); + const data = await resp.json().catch(() => ({ detail: 'Backend key mint failed' })); + return NextResponse.json(data, { status: resp.status }); + } catch (e: any) { + return NextResponse.json({ detail: `Key service unreachable: ${e.message}` }, { status: 503 }); + } } From f6404cd75e8fc21ab93171d5fbcccb6f29bc6fb0 Mon Sep 17 00:00:00 2001 From: Milbaxter Date: Wed, 2 Sep 2026 11:44:48 +0300 Subject: [PATCH 12/16] Proxy Stripe webhook, job poll, session keys, search, and chat to FastAPI. --- frontend/src/app/api/chat/route.ts | 20 ++++++++++++++++ frontend/src/app/api/jobs/[jobId]/route.ts | 16 +++++++++++++ .../src/app/api/keys/from-session/route.ts | 19 +++++++++++++++ frontend/src/app/api/search/route.ts | 20 ++++++++++++++++ frontend/src/app/api/webhooks/stripe/route.ts | 24 +++++++++++++++++++ 5 files changed, 99 insertions(+) create mode 100644 frontend/src/app/api/chat/route.ts create mode 100644 frontend/src/app/api/jobs/[jobId]/route.ts create mode 100644 frontend/src/app/api/keys/from-session/route.ts create mode 100644 frontend/src/app/api/search/route.ts create mode 100644 frontend/src/app/api/webhooks/stripe/route.ts diff --git a/frontend/src/app/api/chat/route.ts b/frontend/src/app/api/chat/route.ts new file mode 100644 index 0000000..b864212 --- /dev/null +++ b/frontend/src/app/api/chat/route.ts @@ -0,0 +1,20 @@ +import { NextRequest, NextResponse } from 'next/server'; + +export async function POST(req: NextRequest) { + const backend = process.env.BACKEND_URL; + if (!backend) { + return NextResponse.json({ detail: 'BACKEND_URL is not configured.' }, { status: 503 }); + } + try { + const body = await req.json(); + const resp = await fetch(`${backend.replace(/\/$/, '')}/api/v1/chat`, { + method: 'POST', + headers: { 'Content-Type': 'application/json' }, + body: JSON.stringify(body), + }); + const data = await resp.json().catch(() => ({ detail: 'Chat failed' })); + return NextResponse.json(data, { status: resp.status }); + } catch (e: any) { + return NextResponse.json({ detail: e.message || 'Backend unreachable' }, { status: 503 }); + } +} diff --git a/frontend/src/app/api/jobs/[jobId]/route.ts b/frontend/src/app/api/jobs/[jobId]/route.ts new file mode 100644 index 0000000..72ee862 --- /dev/null +++ b/frontend/src/app/api/jobs/[jobId]/route.ts @@ -0,0 +1,16 @@ +import { NextRequest, NextResponse } from 'next/server'; + +export async function GET(_req: NextRequest, ctx: { params: Promise<{ jobId: string }> | { jobId: string } }) { + const params = await Promise.resolve(ctx.params); + const backend = process.env.BACKEND_URL; + if (!backend) { + return NextResponse.json({ detail: 'BACKEND_URL is not configured.' }, { status: 503 }); + } + try { + const resp = await fetch(`${backend.replace(/\/$/, '')}/api/v1/jobs/${encodeURIComponent(params.jobId)}`); + const data = await resp.json().catch(() => ({ detail: 'Job lookup failed' })); + return NextResponse.json(data, { status: resp.status }); + } catch (e: any) { + return NextResponse.json({ detail: e.message || 'Backend unreachable' }, { status: 503 }); + } +} diff --git a/frontend/src/app/api/keys/from-session/route.ts b/frontend/src/app/api/keys/from-session/route.ts new file mode 100644 index 0000000..0d62e1f --- /dev/null +++ b/frontend/src/app/api/keys/from-session/route.ts @@ -0,0 +1,19 @@ +import { NextRequest, NextResponse } from 'next/server'; + +export async function GET(req: NextRequest) { + const sessionId = new URL(req.url).searchParams.get('session_id'); + if (!sessionId) { + return NextResponse.json({ detail: 'session_id is required' }, { status: 400 }); + } + const backend = process.env.BACKEND_URL; + if (!backend) { + return NextResponse.json({ detail: 'BACKEND_URL is not configured.' }, { status: 503 }); + } + try { + const resp = await fetch(`${backend.replace(/\/$/, '')}/api/v1/keys/by-session/${encodeURIComponent(sessionId)}`); + const data = await resp.json().catch(() => ({ detail: 'Lookup failed' })); + return NextResponse.json(data, { status: resp.status }); + } catch (e: any) { + return NextResponse.json({ detail: `Key service unreachable: ${e.message}` }, { status: 503 }); + } +} diff --git a/frontend/src/app/api/search/route.ts b/frontend/src/app/api/search/route.ts new file mode 100644 index 0000000..5d4767f --- /dev/null +++ b/frontend/src/app/api/search/route.ts @@ -0,0 +1,20 @@ +import { NextRequest, NextResponse } from 'next/server'; + +export async function POST(req: NextRequest) { + const backend = process.env.BACKEND_URL; + if (!backend) { + return NextResponse.json({ detail: 'BACKEND_URL is not configured.' }, { status: 503 }); + } + try { + const body = await req.json(); + const resp = await fetch(`${backend.replace(/\/$/, '')}/api/v1/search`, { + method: 'POST', + headers: { 'Content-Type': 'application/json' }, + body: JSON.stringify(body), + }); + const data = await resp.json().catch(() => ({ detail: 'Search failed' })); + return NextResponse.json(data, { status: resp.status }); + } catch (e: any) { + return NextResponse.json({ detail: e.message || 'Backend unreachable' }, { status: 503 }); + } +} diff --git a/frontend/src/app/api/webhooks/stripe/route.ts b/frontend/src/app/api/webhooks/stripe/route.ts new file mode 100644 index 0000000..b2801b1 --- /dev/null +++ b/frontend/src/app/api/webhooks/stripe/route.ts @@ -0,0 +1,24 @@ +import { NextRequest, NextResponse } from 'next/server'; + +export async function POST(req: NextRequest) { + const backend = process.env.BACKEND_URL; + if (!backend) { + return NextResponse.json({ error: 'BACKEND_URL is not configured.' }, { status: 503 }); + } + const signature = req.headers.get('stripe-signature') || ''; + const raw = await req.text(); + try { + const resp = await fetch(`${backend.replace(/\/$/, '')}/api/v1/webhooks/stripe`, { + method: 'POST', + headers: { + 'stripe-signature': signature, + 'content-type': 'application/json', + }, + body: raw, + }); + const data = await resp.json().catch(() => ({ received: false })); + return NextResponse.json(data, { status: resp.status }); + } catch (e: any) { + return NextResponse.json({ error: e.message || 'Webhook proxy failed' }, { status: 503 }); + } +} From 365ed94f5bb0371eec14c52a855611a0f741b40a Mon Sep 17 00:00:00 2001 From: Milbaxter Date: Wed, 2 Sep 2026 11:45:05 +0300 Subject: [PATCH 13/16] Fail closed on transcribe: no canned copy, require BACKEND_URL. --- frontend/src/app/api/transcribe/route.ts | 207 +++++++---------------- 1 file changed, 63 insertions(+), 144 deletions(-) diff --git a/frontend/src/app/api/transcribe/route.ts b/frontend/src/app/api/transcribe/route.ts index 62592f9..13d322f 100644 --- a/frontend/src/app/api/transcribe/route.ts +++ b/frontend/src/app/api/transcribe/route.ts @@ -17,6 +17,11 @@ function formatStart(seconds: number): string { return `${mins.toString().padStart(2, '0')}:${secs.toString().padStart(2, '0')}`; } +function failClosed(detail: string, status: number) { + return NextResponse.json({ detail }, { status }); +} + +/** Production never returns canned/demo transcript copy. */ export async function POST(req: NextRequest) { const startTime = Date.now(); try { @@ -25,55 +30,63 @@ export async function POST(req: NextRequest) { const language = body.language || 'en'; if (!url) { - return NextResponse.json({ detail: 'URL is required.' }, { status: 400 }); + return failClosed('URL is required.', 400); } - // 1. If external backend is explicitly configured in env, forward to it - const externalBackend = process.env.BACKEND_URL; - if (externalBackend) { + const backend = process.env.BACKEND_URL; + const apiKey = req.headers.get('x-api-key'); + + if (backend) { try { - const resp = await fetch(`${externalBackend}/api/v1/transcribe`, { + const resp = await fetch(`${backend.replace(/\/$/, '')}/api/v1/transcribe`, { method: 'POST', - headers: { 'Content-Type': 'application/json' }, + headers: { + 'Content-Type': 'application/json', + ...(apiKey ? { 'X-API-Key': apiKey } : {}), + }, body: JSON.stringify(body), }); - if (resp.ok) { - return NextResponse.json(await resp.json()); + const data = await resp.json().catch(() => ({ detail: 'Backend returned a non-JSON error.' })); + if (resp.status === 202) { + return NextResponse.json(data, { status: 202 }); + } + if (!resp.ok) { + const detail = data.detail || data.error || 'Transcription backend failed.'; + const status = resp.status >= 500 ? 503 : resp.status === 401 || resp.status === 403 || resp.status === 429 ? resp.status : 502; + return failClosed(typeof detail === 'string' ? detail : JSON.stringify(detail), status); } - } catch (e) { - console.warn('External backend unavailable, using serverless fallback:', e); + return NextResponse.json(data); + } catch (e: any) { + return failClosed( + `Transcription backend unreachable (${backend}). Set BACKEND_URL to a running FastAPI instance. ${e?.message || ''}`.trim(), + 503 + ); } } - // 2. Native Vercel Serverless Multi-Platform Ingestion - // A. YouTube URL Handler const ytMatch = url.match(/(?:youtube\.com\/(?:watch\?v=|embed\/|shorts\/)|youtu\.be\/)([\w-]{11})/); if (ytMatch) { - const videoId = ytMatch[1]; - const result = await handleYouTube(videoId, url, language); - result.processing_time_ms = Date.now() - startTime; - return NextResponse.json(result); - } - - // B. TikTok URL Handler - if (url.includes('tiktok.com')) { - const result = await handleTikTok(url); + const result = await handleYouTubeCaptions(ytMatch[1], url, language); + if (!result) { + return failClosed( + 'YouTube captions are unavailable and BACKEND_URL is not set. Whisper fallback requires the FastAPI backend.', + 503 + ); + } result.processing_time_ms = Date.now() - startTime; return NextResponse.json(result); } - // C. Podcast / RSS / Audio URL Handler - const result = await handlePodcast(url); - result.processing_time_ms = Date.now() - startTime; - return NextResponse.json(result); - + return failClosed( + 'BACKEND_URL is not configured. TikTok, podcast, and Whisper paths require the FastAPI backend. Set BACKEND_URL (docker-compose NEXT_PUBLIC_API_URL is unused by this route).', + 503 + ); } catch (err: any) { - return NextResponse.json({ detail: err.message || 'Failed to process media' }, { status: 500 }); + return failClosed(err.message || 'Failed to process media', 500); } } -async function handleYouTube(videoId: string, originalUrl: string, language: string) { - // 1. Fetch metadata via oEmbed +async function handleYouTubeCaptions(videoId: string, _originalUrl: string, language: string) { let title = `YouTube Video (${videoId})`; let author = 'YouTube Creator'; let thumbnail_url = `https://img.youtube.com/vi/${videoId}/maxresdefault.jpg`; @@ -88,50 +101,42 @@ async function handleYouTube(videoId: string, originalUrl: string, language: str } } catch {} - // 2. Extract captions from YouTube timedtext API - let segments: Segment[] = []; + const segments: Segment[] = []; try { const timedTextUrls = [ `https://www.youtube.com/api/timedtext?v=${videoId}&lang=${language}&fmt=json3`, `https://www.youtube.com/api/timedtext?v=${videoId}&lang=en&fmt=json3`, `https://www.youtube.com/api/timedtext?v=${videoId}&lang=en-US&fmt=json3`, - `https://www.youtube.com/api/timedtext?v=${videoId}&fmt=json3` + `https://www.youtube.com/api/timedtext?v=${videoId}&fmt=json3`, ]; for (const ttUrl of timedTextUrls) { const ttRes = await fetch(ttUrl, { headers: { 'User-Agent': 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7)' } }); - if (ttRes.ok) { - const data = await ttRes.json(); - if (data.events && Array.isArray(data.events)) { - for (const ev of data.events) { - if (!ev.segs) continue; - const text = ev.segs.map((s: any) => s.utf8 || '').join('').replace(/\n/g, ' ').trim(); - if (text && text !== '\n') { - const startSec = (ev.tStartMs || 0) / 1000; - const durSec = (ev.dDurationMs || 2000) / 1000; - segments.push({ - start: Math.round(startSec * 100) / 100, - end: Math.round((startSec + durSec) * 100) / 100, - text, - formatted_start: formatStart(startSec), - }); - } - } - if (segments.length > 0) break; + if (!ttRes.ok) continue; + const data = await ttRes.json().catch(() => null); + if (!data?.events || !Array.isArray(data.events)) continue; + for (const ev of data.events) { + if (!ev.segs) continue; + const text = ev.segs.map((s: any) => s.utf8 || '').join('').replace(/\n/g, ' ').trim(); + if (text && text !== '\n') { + const startSec = (ev.tStartMs || 0) / 1000; + const durSec = (ev.dDurationMs || 2000) / 1000; + segments.push({ + start: Math.round(startSec * 100) / 100, + end: Math.round((startSec + durSec) * 100) / 100, + text, + formatted_start: formatStart(startSec), + }); } } + if (segments.length > 0) break; } } catch (e) { console.warn('TimedText fetch failed:', e); } - // Fallback demo segments if captions are fully disabled on YouTube if (segments.length === 0) { - segments = [ - { start: 0.0, end: 12.0, text: `Welcome to "${title}". Spoken audio content is parsed and indexed natively.`, formatted_start: '00:00' }, - { start: 12.0, end: 32.0, text: 'This transcript was extracted and processed via PolyTranscript multi-platform intelligence.', formatted_start: '00:12' }, - { start: 32.0, end: 60.0, text: 'Use the export buttons or Model Context Protocol (MCP) server to query this transcript directly in Claude or Cursor.', formatted_start: '00:32' } - ]; + return null; } const fullText = segments.map((s) => s.text).join(' '); @@ -146,97 +151,11 @@ async function handleYouTube(videoId: string, originalUrl: string, language: str language, full_text: fullText, segments, + chapters: [], + summary: null, source_type: 'youtube_timedtext', word_count: fullText.split(/\s+/).filter(Boolean).length, processing_time_ms: 0, created_at: new Date().toISOString(), }; } - -async function handleTikTok(url: string) { - let title = 'TikTok Video'; - let author = 'TikTok Creator'; - let thumbnail_url = undefined; - - try { - const oembedRes = await fetch(`https://www.tiktok.com/oembed?url=${encodeURIComponent(url)}`); - if (oembedRes.ok) { - const oembed = await oembedRes.json(); - title = oembed.title || title; - author = oembed.author_name || author; - thumbnail_url = oembed.thumbnail_url; - } - } catch {} - - const segments: Segment[] = [ - { start: 0.0, end: 5.2, text: title || 'Welcome to this TikTok clip.', formatted_start: '00:00' }, - { start: 5.2, end: 15.0, text: 'Key insights and talking points extracted directly from the video audio.', formatted_start: '00:05' }, - { start: 15.0, end: 28.0, text: 'Transcribed with high accuracy and ready for AI agent integration.', formatted_start: '00:15' } - ]; - - const fullText = segments.map((s) => s.text).join(' '); - return { - metadata: { - title, - author, - thumbnail_url, - platform: 'tiktok' as const, - url, - }, - language: 'en', - full_text: fullText, - segments, - source_type: 'tiktok_audio_transcription', - word_count: fullText.split(/\s+/).filter(Boolean).length, - processing_time_ms: 0, - created_at: new Date().toISOString(), - }; -} - -async function handlePodcast(url: string) { - let title = 'Podcast Episode'; - let author = 'Podcast Host'; - let thumbnail_url = undefined; - - // Apple Podcasts lookup - const appleMatch = url.match(/podcasts\.apple\.com\/[\w-]+\/podcast\/[^/]+\/id(\d+)(?:\?i=(\d+))?/); - if (appleMatch) { - try { - const lookupId = appleMatch[2] || appleMatch[1]; - const lookupRes = await fetch(`https://itunes.apple.com/lookup?id=${lookupId}&entity=podcastEpisode`); - if (lookupRes.ok) { - const data = await lookupRes.json(); - if (data.results && data.results.length > 0) { - const item = data.results[0]; - title = item.trackName || item.collectionName || title; - author = item.artistName || author; - thumbnail_url = item.artworkUrl600 || item.artworkUrl100; - } - } - } catch {} - } - - const segments: Segment[] = [ - { start: 0.0, end: 18.0, text: `Welcome to ${title}. Today we dive deep into technology, software engineering, and artificial intelligence.`, formatted_start: '00:00' }, - { start: 18.0, end: 45.0, text: 'Discussing the fundamental shift toward agentic architectures and automated multi-modal pipelines.', formatted_start: '00:18' }, - { start: 45.0, end: 90.0, text: 'Why developers are moving from raw scrapers to unified API layers with Model Context Protocol (MCP).', formatted_start: '00:45' } - ]; - - const fullText = segments.map((s) => s.text).join(' '); - return { - metadata: { - title, - author, - thumbnail_url, - platform: 'podcast' as const, - url, - }, - language: 'en', - full_text: fullText, - segments, - source_type: 'podcast_transcription', - word_count: fullText.split(/\s+/).filter(Boolean).length, - processing_time_ms: 0, - created_at: new Date().toISOString(), - }; -} From 2dd383f5e789e35119fcb4d5d7bd6b6aec5a50d9 Mon Sep 17 00:00:00 2001 From: Milbaxter Date: Wed, 2 Sep 2026 11:48:58 +0300 Subject: [PATCH 14/16] Fix branding: PolyTranscript header, real sample MP3, OG image route. --- frontend/src/app/og-image.png/route.tsx | 31 +++++++++++++++++++++++++ frontend/src/components/Header.tsx | 4 +--- frontend/src/components/MediaInput.tsx | 6 ++--- 3 files changed, 34 insertions(+), 7 deletions(-) create mode 100644 frontend/src/app/og-image.png/route.tsx diff --git a/frontend/src/app/og-image.png/route.tsx b/frontend/src/app/og-image.png/route.tsx new file mode 100644 index 0000000..09123ca --- /dev/null +++ b/frontend/src/app/og-image.png/route.tsx @@ -0,0 +1,31 @@ +import { ImageResponse } from 'next/og'; + +export const runtime = 'edge'; + +export async function GET() { + return new ImageResponse( + ( +
+
+ MCP-ready transcripts +
+
PolyTranscript
+
+ YouTube • TikTok • Podcasts — timestamped text for Claude, Cursor, and your API. +
+
+ ), + { width: 1200, height: 630 } + ); +} diff --git a/frontend/src/components/Header.tsx b/frontend/src/components/Header.tsx index 3689ca2..541dbd3 100644 --- a/frontend/src/components/Header.tsx +++ b/frontend/src/components/Header.tsx @@ -28,7 +28,7 @@ export const Header: React.FC = () => {
- OmniTranscript + PolyTranscript MCP Agent @@ -37,7 +37,6 @@ export const Header: React.FC = () => {
- {/* Desktop Nav */} - {/* Action Buttons */}
= ({ onTranscribe, isLoading const sampleUrls = [ { label: '3Blue1Brown (YouTube)', url: 'https://www.youtube.com/watch?v=aircAruvnKk' }, { label: 'Lex Fridman (YouTube)', url: 'https://www.youtube.com/watch?v=jvqFAi7vkBc' }, - { label: 'Tech Podcast (RSS / MP3)', url: 'https://traffic.libsyn.com/show/episode1.mp3' }, + { label: 'CC0 sample MP3 (MDN)', url: 'https://interactive-examples.mdn.mozilla.net/media/cc0-audio/t-rex-roar.mp3' }, ]; return ( @@ -49,7 +49,7 @@ export const MediaInput: React.FC = ({ onTranscribe, isLoading type="text" value={url} onChange={(e) => setUrl(e.target.value)} - placeholder="Paste any YouTube, TikTok, Apple Podcast, RSS Feed, or Audio URL..." + placeholder="Paste any YouTube, TikTok, Apple Podcast, RSS episode, or Audio URL..." className="w-full bg-transparent text-white placeholder-slate-400 text-sm md:text-base focus:outline-none px-2 py-2.5 font-sans" disabled={isLoading} /> @@ -74,7 +74,6 @@ export const MediaInput: React.FC = ({ onTranscribe, isLoading
- {/* Quick Filters */}
Language: @@ -93,7 +92,6 @@ export const MediaInput: React.FC = ({ onTranscribe, isLoading
- {/* Fast Samples */}
Try: {sampleUrls.map((s, idx) => ( From c8c9bf438c7201fe6e9ba82c30e46daefeede33c Mon Sep 17 00:00:00 2001 From: Milbaxter Date: Wed, 2 Sep 2026 11:49:29 +0300 Subject: [PATCH 15/16] Add summary/chapters/Q&A panel and YouTube IFrame seekTo player. --- frontend/src/components/IntelligencePanel.tsx | 126 ++++++++++++++++++ frontend/src/components/InteractivePlayer.tsx | 73 ++++++++-- 2 files changed, 187 insertions(+), 12 deletions(-) create mode 100644 frontend/src/components/IntelligencePanel.tsx diff --git a/frontend/src/components/IntelligencePanel.tsx b/frontend/src/components/IntelligencePanel.tsx new file mode 100644 index 0000000..cac79e4 --- /dev/null +++ b/frontend/src/components/IntelligencePanel.tsx @@ -0,0 +1,126 @@ +'use client'; + +import React, { useState } from 'react'; +import { TranscriptResponse } from '../lib/types'; +import { askMedia } from '../lib/api'; +import { BookOpen, ListTree, MessageCircle, Loader2, Play } from 'lucide-react'; + +interface Props { + transcript: TranscriptResponse; + onSeek: (seconds: number) => void; +} + +export const IntelligencePanel: React.FC = ({ transcript, onSeek }) => { + const [question, setQuestion] = useState(''); + const [answer, setAnswer] = useState(null); + const [citations, setCitations] = useState>([]); + const [asking, setAsking] = useState(false); + const [askError, setAskError] = useState(null); + + const chapters = transcript.chapters || []; + const summary = transcript.summary; + + const handleAsk = async (e: React.FormEvent) => { + e.preventDefault(); + if (!question.trim() || asking) return; + setAsking(true); + setAskError(null); + try { + const res = await askMedia(question.trim(), transcript.full_text, transcript.segments); + setAnswer(res.answer); + setCitations(res.relevant_timestamps || []); + } catch (err: any) { + setAskError(err.message || 'Q&A failed'); + } finally { + setAsking(false); + } + }; + + return ( +
+ {summary && ( +
+

+ + Executive summary +

+

{summary.tldr}

+ {summary.key_takeaways?.length > 0 && ( +
    + {summary.key_takeaways.map((t, i) => ( +
  • • {t}
  • + ))} +
+ )} +
+ )} + + {chapters.length > 0 && ( +
+

+ + Chapters +

+
+ {chapters.map((c, i) => ( + + ))} +
+
+ )} + +
+

+ + Ask this transcript +

+

+ Keyword-overlap retrieval plus optional LLM. Not embedding-based semantic search. +

+
+ setQuestion(e.target.value)} + placeholder="What is the main argument?" + className="flex-1 bg-black/40 border border-white/10 rounded-lg px-3 py-2 text-xs text-white placeholder-slate-500 focus:outline-none" + /> + +
+ {askError &&

{askError}

} + {answer &&

{answer}

} + {citations.length > 0 && ( +
+ {citations.map((c, i) => ( + + ))} +
+ )} +
+
+ ); +}; diff --git a/frontend/src/components/InteractivePlayer.tsx b/frontend/src/components/InteractivePlayer.tsx index c22e426..1b204e1 100644 --- a/frontend/src/components/InteractivePlayer.tsx +++ b/frontend/src/components/InteractivePlayer.tsx @@ -2,9 +2,16 @@ import React, { useRef, useEffect } from 'react'; import { MediaMetadata } from '../lib/types'; -import { Music, Video } from 'lucide-react'; +import { Music } from 'lucide-react'; import { YoutubeIcon, TikTokIcon } from './Icons'; +declare global { + interface Window { + YT?: any; + onYouTubeIframeAPIReady?: () => void; + } +} + interface InteractivePlayerProps { metadata: MediaMetadata; seekTime: number | null; @@ -12,6 +19,8 @@ interface InteractivePlayerProps { export const InteractivePlayer: React.FC = ({ metadata, seekTime }) => { const audioRef = useRef(null); + const containerRef = useRef(null); + const playerRef = useRef(null); const getYouTubeId = (url: string) => { const match = url.match(/(?:youtu\.be\/|youtube\.com\/(?:watch\?v=|embed\/|shorts\/))([\w-]{11})/); @@ -21,11 +30,57 @@ export const InteractivePlayer: React.FC = ({ metadata, const ytId = getYouTubeId(metadata.url); useEffect(() => { - if (seekTime !== null) { - if (audioRef.current) { - audioRef.current.currentTime = seekTime; - audioRef.current.play().catch(() => {}); + if (!ytId || !containerRef.current) return; + let cancelled = false; + + const createPlayer = () => { + if (cancelled || !containerRef.current || !window.YT?.Player) return; + try { + playerRef.current?.destroy?.(); + } catch {} + playerRef.current = new window.YT.Player(containerRef.current, { + videoId: ytId, + width: '100%', + height: '100%', + playerVars: { enablejsapi: 1, rel: 0, modestbranding: 1 }, + }); + }; + + if (window.YT?.Player) { + createPlayer(); + } else { + const existing = document.querySelector('script[src="https://www.youtube.com/iframe_api"]'); + if (!existing) { + const tag = document.createElement('script'); + tag.src = 'https://www.youtube.com/iframe_api'; + document.head.appendChild(tag); } + const previous = window.onYouTubeIframeAPIReady; + window.onYouTubeIframeAPIReady = () => { + previous?.(); + createPlayer(); + }; + } + + return () => { + cancelled = true; + try { + playerRef.current?.destroy?.(); + } catch {} + playerRef.current = null; + }; + }, [ytId]); + + useEffect(() => { + if (seekTime === null) return; + if (audioRef.current) { + audioRef.current.currentTime = seekTime; + audioRef.current.play().catch(() => {}); + } + const player = playerRef.current; + if (player?.seekTo) { + player.seekTo(seekTime, true); + player.playVideo?.(); } }, [seekTime]); @@ -49,13 +104,7 @@ export const InteractivePlayer: React.FC = ({ metadata, {ytId ? (
-