Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
16 commits
Select commit Hold shift + click to select a range
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
29 changes: 29 additions & 0 deletions .env.example
Original file line number Diff line number Diff line change
@@ -0,0 +1,29 @@
# Frontend (Next.js server). /api/transcribe reads BACKEND_URL, not NEXT_PUBLIC_API_URL.
BACKEND_URL=http://localhost:8000
APP_URL=http://localhost:3000

# Backend AI (Whisper). Without these, local openai-whisper/faster-whisper must be installed.
OPENAI_API_KEY=
GROQ_API_KEY=

# YouTube / yt-dlp proxy (passed into youtube-transcript-api + yt-dlp)
HTTP_PROXY=
HTTPS_PROXY=

# Persistent API key JSON store (upgrade path: swap key_store.py for SQLite)
DATA_DIR=./data
API_KEYS_PATH=./data/api_keys.json

# Stripe. Yearly checkout requires annual Price IDs. Do not invent live keys.
STRIPE_SECRET_KEY=
STRIPE_WEBHOOK_SECRET=
STRIPE_PRICE_STARTER_MONTHLY=
STRIPE_PRICE_STARTER_YEARLY=
STRIPE_PRICE_PRO_MONTHLY=
STRIPE_PRICE_PRO_YEARLY=
STRIPE_PRICE_SCALE_MONTHLY=
STRIPE_PRICE_SCALE_YEARLY=
STRIPE_PRICE_SPONSOR=

# Sponsor banner (keep off polytranscript.dev)
SPONSOR_LINK=/pricing#sponsor
30 changes: 15 additions & 15 deletions backend/app/ai/searcher.py
Original file line number Diff line number Diff line change
@@ -1,50 +1,50 @@
import re
"""Keyword / token-overlap search over transcript segments.

This is NOT embedding-based semantic search. Scores are exact-phrase bonuses
plus per-token overlap. Swap in a vector index later if you need true semantics.
"""
from typing import List
from app.models import TranscriptSegment, SearchHit, SearchResponse


class TranscriptSearcher:
SEARCH_METHOD = "keyword_overlap"

def search(self, segments: List[TranscriptSegment], query: str, top_k: int = 10) -> SearchResponse:
if not segments or not query.strip():
return SearchResponse(query=query, total_matches=0, hits=[])
return SearchResponse(query=query, total_matches=0, hits=[], method=self.SEARCH_METHOD)

query_tokens = [q.lower().strip() for q in query.split() if q.strip()]
hits: List[SearchHit] = []

for idx, seg in enumerate(segments):
text_lower = seg.text.lower()
score = 0.0

# Exact phrase match bonus

if query.lower() in text_lower:
score += 5.0

# Token match scoring
matched_tokens = 0
for token in query_tokens:
if token in text_lower:
matched_tokens += 1
score += 1.0

if score > 0:
# Highlight or format snippet
snippet = seg.text
hits.append(SearchHit(
segment_index=idx,
start=seg.start,
end=seg.end,
text=snippet,
text=seg.text,
score=score,
formatted_start=seg.formatted_start
formatted_start=seg.formatted_start,
))

# Sort by relevance score descending
hits.sort(key=lambda h: h.score, reverse=True)
top_hits = hits[:top_k]

return SearchResponse(
query=query,
total_matches=len(hits),
hits=top_hits
hits=hits[:top_k],
method=self.SEARCH_METHOD,
)


searcher = TranscriptSearcher()
174 changes: 125 additions & 49 deletions backend/app/ai/transcriber.py
Original file line number Diff line number Diff line change
@@ -1,11 +1,35 @@
import os
import shutil
import subprocess
import tempfile
import time
from typing import List, Tuple

from app.config import settings
from app.models import TranscriptSegment

DEMO_FORBIDDEN_SNIPPETS = (
"Spoken audio content is parsed and indexed natively",
"multi-platform media intelligence and agentic workflows",
"Model Context Protocol (MCP) transform unstructured audio",
"Welcome to this TikTok clip",
"agentic architectures and automated multi-modal pipelines",
)


def local_whisper_available() -> bool:
try:
import faster_whisper # noqa: F401
return True
except Exception:
pass
try:
import whisper # noqa: F401
return True
except Exception:
pass
return shutil.which("whisper") is not None


class AudioTranscriber:
def __init__(self):
self.groq_api_key = settings.GROQ_API_KEY
Expand All @@ -16,39 +40,40 @@ def has_ai_credentials(self) -> bool:

async def transcribe_audio_file(self, audio_path: str, language: str = "en") -> Tuple[str, List[TranscriptSegment]]:
"""
Transcribe an audio file using Groq Whisper, OpenAI Whisper, or a local audio pipeline.
Returns full text and timestamped segments.
Transcribe an audio file using Groq Whisper, OpenAI Whisper, or a real local Whisper install.
Never returns canned/demo transcript copy.
"""
if not os.path.exists(audio_path):
raise FileNotFoundError(f"Audio file not found: {audio_path}")

# Ensure audio is optimized for Whisper (<25MB, 16kHz mono mp3)
processed_path = self._preprocess_audio(audio_path)

# 1. Try Groq Whisper (Ultra fast, cost-effective)
if self.groq_api_key:
try:
return await self._transcribe_with_groq(processed_path, language)
except Exception as e:
print(f"[Transcriber] Groq failed, trying fallback: {e}")

# 2. Try OpenAI Whisper
if self.openai_api_key:
try:
return await self._transcribe_with_openai(processed_path, language)
except Exception as e:
print(f"[Transcriber] OpenAI Whisper failed: {e}")

# 3. Fallback: Local Whisper CLI / Mock parser for testing environment
return self._local_or_mock_transcribe(processed_path)
if settings.LOCAL_WHISPER_FALLBACK and local_whisper_available():
return self._local_whisper_transcribe(processed_path, language)

raise RuntimeError(
"No transcription provider available. Set GROQ_API_KEY or OPENAI_API_KEY, "
"or install openai-whisper / faster-whisper. Demo/mock transcripts are disabled."
)

def _preprocess_audio(self, input_path: str) -> str:
"""Convert any audio/video file to 16kHz mono MP3 for high compression and Whisper compatibility."""
output_path = tempfile.mktemp(suffix=".mp3", dir=settings.TEMP_STORAGE_DIR)
cmd = [
"ffmpeg", "-y", "-i", input_path,
"-vn", "-ar", "16000", "-ac", "1", "-b:a", "64k",
output_path
output_path,
]
try:
subprocess.run(cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE, check=True)
Expand All @@ -60,74 +85,125 @@ def _preprocess_audio(self, input_path: str) -> str:
async def _transcribe_with_groq(self, audio_path: str, language: str) -> Tuple[str, List[TranscriptSegment]]:
from groq import AsyncGroq
client = AsyncGroq(api_key=self.groq_api_key)

with open(audio_path, "rb") as f:
transcription = await client.audio.transcriptions.create(
file=(os.path.basename(audio_path), f.read()),
model="whisper-large-v3",
response_format="verbose_json",
language=language if language != "auto" else None,
temperature=0.0
temperature=0.0,
)

full_text = transcription.text.strip()
segments = []
raw_segments = getattr(transcription, "segments", []) or []
for s in raw_segments:
seg_dict = s if isinstance(s, dict) else s.model_dump()
segments.append(TranscriptSegment(
start=float(seg_dict.get("start", 0.0)),
end=float(seg_dict.get("end", 0.0)),
text=seg_dict.get("text", "").strip()
))

return full_text, segments
return self._segments_from_whisper_result(transcription)

async def _transcribe_with_openai(self, audio_path: str, language: str) -> Tuple[str, List[TranscriptSegment]]:
from openai import AsyncOpenAI
client = AsyncOpenAI(api_key=self.openai_api_key)

with open(audio_path, "rb") as f:
transcription = await client.audio.transcriptions.create(
file=f,
model="whisper-1",
response_format="verbose_json",
language=language if language != "auto" else None,
timestamp_granularities=["segment"]
timestamp_granularities=["segment"],
)

full_text = transcription.text.strip()
segments = []
return self._segments_from_whisper_result(transcription)

def _segments_from_whisper_result(self, transcription) -> Tuple[str, List[TranscriptSegment]]:
full_text = (transcription.text or "").strip()
segments: List[TranscriptSegment] = []
raw_segments = getattr(transcription, "segments", []) or []
for s in raw_segments:
seg_dict = s if isinstance(s, dict) else s.model_dump()
segments.append(TranscriptSegment(
start=float(seg_dict.get("start", 0.0)),
end=float(seg_dict.get("end", 0.0)),
text=seg_dict.get("text", "").strip()
text=(seg_dict.get("text") or "").strip(),
))

self._assert_not_demo(full_text)
return full_text, segments

def _local_or_mock_transcribe(self, audio_path: str) -> Tuple[str, List[TranscriptSegment]]:
"""Generate structured transcript if no external API key is active."""
# Check audio length via ffprobe
duration = 60.0
def _local_whisper_transcribe(self, audio_path: str, language: str) -> Tuple[str, List[TranscriptSegment]]:
"""Real local Whisper only. Raises if no engine is installed."""
lang = None if language == "auto" else language

try:
probe = subprocess.run(
["ffprobe", "-v", "error", "-show_entries", "format=duration", "-of", "default=noprint_wrappers=1:nokey=1", audio_path],
stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True
)
duration = float(probe.stdout.strip())
except Exception:
duration = 60.0

sample_segments = [
TranscriptSegment(start=0.0, end=min(15.0, duration), text="Welcome to this episode. Today we are breaking down multi-platform media intelligence and agentic workflows."),
TranscriptSegment(start=min(15.0, duration), end=min(35.0, duration), text="We are exploring how automated transcript extraction and Model Context Protocol (MCP) transform unstructured audio into actionable knowledge."),
TranscriptSegment(start=min(35.0, duration), end=duration, text="By indexing YouTube, TikTok, and podcasts natively, autonomous agents can search soundbites and reason over rich media in real-time.")
]
full_text = " ".join([s.text for s in sample_segments])
return full_text, sample_segments
from faster_whisper import WhisperModel

model = WhisperModel("base", device="cpu", compute_type="int8")
segments_iter, _info = model.transcribe(audio_path, language=lang)
segments: List[TranscriptSegment] = []
parts = []
for s in segments_iter:
text = (s.text or "").strip()
if not text:
continue
segments.append(TranscriptSegment(start=float(s.start or 0.0), end=float(s.end or 0.0), text=text))
parts.append(text)
full_text = " ".join(parts)
self._assert_not_demo(full_text)
return full_text, segments
except ImportError:
pass

try:
import whisper

model = whisper.load_model("base")
result = model.transcribe(audio_path, language=lang)
full_text = (result.get("text") or "").strip()
segments = []
for s in result.get("segments") or []:
segments.append(TranscriptSegment(
start=float(s.get("start", 0.0)),
end=float(s.get("end", 0.0)),
text=(s.get("text") or "").strip(),
))
self._assert_not_demo(full_text)
return full_text, segments
except ImportError:
pass

whisper_bin = shutil.which("whisper")
if whisper_bin:
outdir = tempfile.mkdtemp(dir=settings.TEMP_STORAGE_DIR)
cmd = [whisper_bin, audio_path, "--model", "base", "--output_format", "json", "--output_dir", outdir]
if lang:
cmd.extend(["--language", lang])
proc = subprocess.run(cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True)
if proc.returncode != 0:
raise RuntimeError(f"whisper CLI failed: {proc.stderr[-500:]}")
import json
json_files = [os.path.join(outdir, f) for f in os.listdir(outdir) if f.endswith(".json")]
if not json_files:
raise RuntimeError("whisper CLI produced no JSON output")
with open(json_files[0]) as fh:
result = json.load(fh)
full_text = (result.get("text") or "").strip()
segments = [
TranscriptSegment(
start=float(s.get("start", 0.0)),
end=float(s.get("end", 0.0)),
text=(s.get("text") or "").strip(),
)
for s in result.get("segments") or []
]
self._assert_not_demo(full_text)
return full_text, segments

raise RuntimeError(
"Local Whisper is not installed. Set GROQ_API_KEY or OPENAI_API_KEY, "
"or `pip install openai-whisper` / `faster-whisper`. Mock transcripts are disabled."
)

def _assert_not_demo(self, full_text: str) -> None:
lower = (full_text or "").lower()
for snippet in DEMO_FORBIDDEN_SNIPPETS:
if snippet.lower() in lower:
raise RuntimeError("Refusing to return canned/demo transcript copy.")


transcriber = AudioTranscriber()
Loading
Loading