Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
40 changes: 0 additions & 40 deletions .github/workflows/docker_build_push.yml

This file was deleted.

1 change: 1 addition & 0 deletions .gitignore
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
# TranscriberBot-specific ignores
media/
.python-version

# Generic data-related ignores
*.csv
Expand Down
1 change: 0 additions & 1 deletion .python-version

This file was deleted.

4 changes: 4 additions & 0 deletions config/subscription.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,4 @@
{
"channel_id": "<id>",
"premium_join_link": "xxx"
}
3 changes: 2 additions & 1 deletion requirements.txt
Original file line number Diff line number Diff line change
Expand Up @@ -6,4 +6,5 @@ tesserocr
pydub
zbarlight
requests
sentry-sdk
sentry-sdk
audioread
10 changes: 9 additions & 1 deletion src/audiotools/speech.py
Original file line number Diff line number Diff line change
Expand Up @@ -97,7 +97,15 @@ async def transcribe_wit(path, api_key):


async def transcribe_whisper(path):
resp = requests.get(f"{config.get_config_prop('app')['whisper']['api_endpoint']}/transcribe?file_id={path}")
loop = asyncio.get_event_loop()
resp = await loop.run_in_executor(
None,
partial(requests.get,
url=f"{config.get_config_prop('app')['whisper']['api_endpoint']}/transcribe?file_id={path}")
)

if resp.status_code != 200:
raise ValueError(f"Error transcribing audio: {resp.text}")

# split the response into chunks of 4000 characters
chunks = textwrap.wrap(resp.text, 4000)
Expand Down
7 changes: 7 additions & 0 deletions src/config/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -68,3 +68,10 @@ def get_document_extensions():

def get_bot_admins():
return [int(id) for id in get_config_prop("telegram")["admins"]]


def get_premium_join_link():
return get_config_prop("subscription")["premium_join_link"]

def get_premium_chat_id():
return get_config_prop("subscription")["channel_id"]
2 changes: 1 addition & 1 deletion src/database/db.py
Original file line number Diff line number Diff line change
Expand Up @@ -95,7 +95,7 @@ def get_chat_voice_enabled(chat_id):
with TBDB._get_db() as db:
c = db.execute("SELECT voice_enabled FROM chats WHERE chat_id='{0}'".format(chat_id))
return c.fetchone()[0]
except TypeError as e:
except Exception as e:
logger.error("Error getting voice_enabled for chat %d: %s", chat_id, e)
raise e

Expand Down
2 changes: 1 addition & 1 deletion src/transcriberbot/blueprints/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -2,4 +2,4 @@
Author: Carlo Alberto Barbano <carlo.alberto.barbano@outlook.com>
Date: 15/02/25
"""
from . import commands, messages, voice, photos, chat_handlers
from . import commands, messages, voice, photos, chat_handlers, payments
35 changes: 35 additions & 0 deletions src/transcriberbot/blueprints/payments.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,35 @@
"""
Author: Carlo Alberto Barbano <carlo.barbano@unito.it>
Date: 20/02/25
"""
import config
import resources as R

from telegram import Update, InlineKeyboardMarkup, InlineKeyboardButton
from telegram.ext import ContextTypes
from database import TBDB
from transcriberbot.filters import is_premium_user


async def premium(update: Update, context: ContextTypes.DEFAULT_TYPE) -> None:
premium_join_link = config.get_premium_join_link()

keyboard = InlineKeyboardMarkup(
[[InlineKeyboardButton("Join", url=premium_join_link)]]
)

premium_join_message = R.get_string_resource(
'premium_join_message',
TBDB.get_chat_lang(update.effective_chat.id)
).replace('{invite_url}', premium_join_link)

current_plan = R.get_string_resource("current_plan_free")
if await is_premium_user(update, context):
current_plan = R.get_string_resource("current_plan_premium")

await update.effective_message.reply_text(
f"{premium_join_message}\n\n{current_plan}",
reply_markup=keyboard, parse_mode="html"
)


85 changes: 45 additions & 40 deletions src/transcriberbot/blueprints/voice.py
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,7 @@
import os
import traceback
import datetime
import audioread
from asyncio import CancelledError

import telegram
Expand All @@ -18,6 +19,7 @@
import config
import resources as R
from database import TBDB
from transcriberbot.filters import is_premium_user

logger = logging.getLogger(__name__)

Expand Down Expand Up @@ -94,6 +96,9 @@ async def run_voice_task(update: Update, context: ContextTypes.DEFAULT_TYPE, med

async def process_media_voice(update: Update, context: ContextTypes.DEFAULT_TYPE, media: [Voice | VideoNote | Document],
name: str) -> None:
print("Update:", update)
print("Effective user:", update.effective_user)

chat_id = update.effective_chat.id
file_size = media.file_size
max_size = config.get_config_prop("app").get("max_media_voice_file_size", 20 * 1024 * 1024)
Expand All @@ -119,7 +124,26 @@ async def process_media_voice(update: Update, context: ContextTypes.DEFAULT_TYPE
os.remove(file_path)


async def transcribe_audio_file(update: Update, context: ContextTypes.DEFAULT_TYPE, path: str):
def get_duration(update: Update, path: str):
media = (update.effective_message.voice or update.effective_message.audio or
update.effective_message.video_note or update.effective_message.video)
if media is not None:
return media.duration

with audioread.audio_open(path) as f:
return f.duration


async def get_backend(update: Update, context: ContextTypes.DEFAULT_TYPE, path):
backend = "wit"
if await is_premium_user(update, context):
logging.info("User is premium")
duration = get_duration(update, path)
if duration <= config.get_config_prop("app")["whisper"]["max_duration"]:
backend = "whisper"
return backend

async def run_transcription(update: Update, context: ContextTypes.DEFAULT_TYPE, path: str, backend: str):
chat_id = update.effective_chat.id
task_id = update.effective_message.message_id
lang = TBDB.get_chat_lang(chat_id)
Expand All @@ -137,7 +161,7 @@ async def transcribe_audio_file(update: Update, context: ContextTypes.DEFAULT_TY
logger.debug("Using key %s for lang %s", api_key, lang)

message = await context.bot.send_message(
chat_id, R.get_string_resource("transcribing", lang), parse_mode="html",
chat_id, f"{R.get_string_resource('transcribing', lang)} ({backend}, lang: {lang})", parse_mode="html",
reply_to_message_id=update.effective_message.message_id
)

Expand All @@ -151,7 +175,7 @@ async def transcribe_audio_file(update: Update, context: ContextTypes.DEFAULT_TY
text = R.get_string_resource("transcription_text", lang) + "\n"

try:
async for idx, speech, n_chunks in audiotools.transcribe(path, api_key):
async for idx, speech, n_chunks in audiotools.transcribe(path, api_key, backend=backend):
logging.debug(f"Transcription idx={idx} n_chunks={n_chunks}, text={speech}")
suffix = f" <b>[{idx + 1}/{n_chunks}]</b>" if idx < n_chunks - 1 else ""
reply_markup = keyboard if idx < n_chunks - 1 else None
Expand All @@ -172,43 +196,6 @@ async def transcribe_audio_file(update: Update, context: ContextTypes.DEFAULT_TY

text = f"{text} {speech}"

# retry_num = 0
# retry = True
# while retry: # Retry loop
# try:
# if len(text + " " + speech) >= 4000:
# text = R.get_string_resource("transcription_continues", lang) + "\n"
# message = await context.bot.send_message(
# chat_id, f"{text} {speech} {suffix}",
# reply_to_message_id=message.message_id, parse_mode="html",
# reply_markup=keyboard
# )
# else:
# message = await context.bot.edit_message_text(
# f"{text} {speech} {suffix}", chat_id=chat_id,
# message_id=message.message_id, parse_mode="html",
# reply_markup=keyboard
# )
#
# text += " " + speech
# retry = False
#
# except telegram.error.TimedOut as e:
# print(e)
# logger.error("Timeout error %s", traceback.format_exc())
# retry_num += 1
# if retry_num >= 3:
# retry = False
#
# except telegram.error.RetryAfter as r:
# logger.warning("Retrying after %d", r.retry_after)
# await asyncio.sleep(r.retry_after)
#
# except telegram.error.TelegramError:
# logger.error("Telegram error %s", traceback.format_exc())
# retry = False


except CancelledError:
logging.debug("Task cancelled")
await context.bot.edit_message_text(
Expand All @@ -226,3 +213,21 @@ async def transcribe_audio_file(update: Update, context: ContextTypes.DEFAULT_TY
)

raise e

async def transcribe_audio_file(update: Update, context: ContextTypes.DEFAULT_TYPE, path: str):
backend = await get_backend(update, context, path)

# try running transcription, if it fails with whisper, try wit
try:
await run_transcription(update, context, path, backend)
except Exception as e:
if backend == "whisper":
logger.error("Whisper transcription failed, falling back to wit", exc_info=True)
try:
await run_transcription(update, context, path, "wit")
except Exception as e2:
logger.error("Wit transcription also failed", exc_info=True)
raise e2
else:
raise e

20 changes: 11 additions & 9 deletions src/transcriberbot/bot.py
Original file line number Diff line number Diff line change
Expand Up @@ -2,23 +2,24 @@
Author: Carlo Alberto Barbano <carlo.alberto.barbano@outlook.com>
Date: 15/02/25
"""
from telegram import Update

import config
import logging

from telegram.ext import MessageHandler, ApplicationBuilder, CommandHandler, ContextTypes, CallbackQueryHandler, \
ChatMemberHandler
from functools import partial
from transcriberbot.blueprints import commands, messages, voice, photos, chat_handlers
from transcriberbot.blueprints.commands import set_language

from telegram import Update
from telegram.ext import MessageHandler, ApplicationBuilder, CommandHandler, CallbackQueryHandler, \
ChatMemberHandler
from telegram.ext.filters import VOICE, VIDEO_NOTE, AUDIO, PHOTO

import config
from transcriberbot.blueprints import commands, messages, voice, photos, chat_handlers, payments
from transcriberbot.blueprints.commands import set_language
from transcriberbot.filters import chat_admin, FromPrivate, AllowedDocument, BotAdmin


def run(bot_token: str):
application = (ApplicationBuilder()
# .base_url("https://api.telegram.org/bot{token}/test")
# .base_file_url("https://api.telegram.org/file/bot{token}/test")
.token(bot_token)
.concurrent_updates(True)
.build())
Expand All @@ -44,7 +45,8 @@ def run(bot_token: str):
'enable_qr': commands.enable_qr,
'translate': commands.translate,
'donate': commands.donate,
'privacy': commands.privacy
'privacy': commands.privacy,
'premium': payments.premium
}

for command, callback in chat_admin_handlers.items():
Expand Down
20 changes: 19 additions & 1 deletion src/transcriberbot/filters/filters.py
Original file line number Diff line number Diff line change
Expand Up @@ -5,7 +5,7 @@
import logging
import asyncio

from telegram.constants import ChatType
from telegram.constants import ChatType, ChatMemberStatus
from telegram.ext import ContextTypes
from telegram.ext.filters import UpdateFilter
from telegram import Update, ChatMember
Expand Down Expand Up @@ -86,6 +86,24 @@ async def chat_admin(update: Update, context: ContextTypes.DEFAULT_TYPE, callbac
return await callback(update, context)


async def is_premium_user(update: Update, context: ContextTypes.DEFAULT_TYPE):
user = (update.effective_user or update.channel_post.from_user)

# does not support anonymous channels
if user is None:
return False

user_id = user.id

premium_channel = await context.bot.get_chat(config.get_premium_chat_id())

try:
member = await premium_channel.get_member(user_id)
return member.status in (ChatMemberStatus.MEMBER, ChatMemberStatus.ADMINISTRATOR, ChatMemberStatus.OWNER)
except Exception as e:
return False


class BotAdmin(UpdateFilter):
"""
Checks if the message was sent by the bot admin.
Expand Down
16 changes: 13 additions & 3 deletions values/strings.xml
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,7 @@

<string name="message_welcome">
This bot transcribes audio and pictures into text. Add it to a group or forward audio messages and pictures to it.

{b}Note{/b}: Within groups, the bot will respond to commands by {b}non-anonymous admins only{/b}

Choose a language (for voice messages):
Expand Down Expand Up @@ -61,17 +62,26 @@ LTC: {code}LdsVPxqHR6PuKeMNvGYmMEkBRQ7M3AP3uY{/code}
<string name="flood_warning">{b}WARNING:{/b} Flood detected. Stop spamming please</string>

<string name="translate_language_not_found">Unknown language: {language}</string>
<string name="translate_language_missing">Please specify a language to translate to, e.g. "/translate english" </string>
<string name="translate_language_missing">Please specify a language to translate to, e.g. "/translate english"
</string>
<string name="translate_reply_to_message">You must reply to a message in order to translate it</string>

<string name="privacy_policy">
We do not store any personal data or media content (such as audio or pictures).
The only user data we store is the telegram chat id along with the chosen settings.
Voice/audio messages are sent to {a href="https://wit.ai/terms"}wit.ai{/a} for transcription, and are immediately deleted after the transcription is completed.
Voice/audio messages are sent to {a href="https://wit.ai/terms"}wit.ai{/a} for transcription, and are
immediately deleted after the transcription is completed.
Pictures are immediately deleted from our server after OCR or QR recognition is perfomed (offline).
We do NOT store ANY message content (text, multimedia or anything else).
</string>

<string name="unknown_api_key">{b}ERROR:{/b} Oops, no wit.ai API key could be found for the language {language}. :(</string>
<string name="unknown_api_key">{b}ERROR:{/b} Oops, no wit.ai API key could be found for the language {language}.
:(
</string>

<string name="premium_join_message">
To enable Transcriber Bot Premium, please join this channel: {invite_url}
</string>
<string name="current_plan_free">{b}Current plan:{/b} Free</string>
<string name="current_plan_premium">{b}Current plan:{/b} Premium</string>
</resources>