From 8678d8d4a083d84aa63613e9663e80cc2361cc54 Mon Sep 17 00:00:00 2001 From: Tony Giorgio Date: Wed, 2 Jul 2025 16:58:41 -0500 Subject: [PATCH 1/6] feat: replace rough token estimation with accurate gpt-tokenizer MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Added gpt-tokenizer (v3.0.1) for accurate token counting - Replaced character-based estimation (text.length/4) with proper tokenization - Maintains same token warning thresholds and UI behavior - Improves cost estimation accuracy for users 🤖 Generated with [Claude Code](https://claude.ai/code) Co-Authored-By: Claude --- frontend/bun.lock | 3 +++ frontend/package.json | 1 + frontend/src/components/ChatBox.tsx | 7 ++++--- 3 files changed, 8 insertions(+), 3 deletions(-) diff --git a/frontend/bun.lock b/frontend/bun.lock index ad1399f7d..354475982 100644 --- a/frontend/bun.lock +++ b/frontend/bun.lock @@ -23,6 +23,7 @@ "@tauri-apps/plugin-os": "^2.2.1", "class-variance-authority": "^0.7.0", "clsx": "^2.1.1", + "gpt-tokenizer": "^3.0.1", "lucide-react": "^0.436.0", "openai": "^4.56.1", "react": "^18.3.1", @@ -708,6 +709,8 @@ "gopd": ["gopd@1.2.0", "", {}, "sha512-ZUKRh6/kUFoAiTAtTYPZJ3hw9wNxx+BIBOijnlG9PnrJsCcSjs1wyyD6vJpaYtgnzDrKYRSqf3OO6Rfa93xsRg=="], + "gpt-tokenizer": ["gpt-tokenizer@3.0.1", "", {}, "sha512-5jdaspBq/w4sWw322SvQj1Fku+CN4OAfYZeeEg8U7CWtxBz+zkxZ3h0YOHD43ee+nZYZ5Ud70HRN0ANcdIj4qg=="], + "graphemer": ["graphemer@1.4.0", "", {}, "sha512-EtKwoO6kxCL9WO5xipiHTZlSzBm7WLT627TqC/uVRd0HKmq8NXyebnNYxDoBi7wt8eTWrUrKXCOVaFq9x1kgag=="], "has-flag": ["has-flag@4.0.0", "", {}, "sha512-EykJT/Q1KjTWctppgIAgfSO0tKVuZUjhgMr17kqTumMl6Afv3EISleU7qZUzoXDFTAHTDC4NOoG/ZxU3EvlMPQ=="], diff --git a/frontend/package.json b/frontend/package.json index a28e1d7d8..af4a9e6fc 100644 --- a/frontend/package.json +++ b/frontend/package.json @@ -35,6 +35,7 @@ "@tauri-apps/plugin-os": "^2.2.1", "class-variance-authority": "^0.7.0", "clsx": "^2.1.1", + "gpt-tokenizer": "^3.0.1", "lucide-react": "^0.436.0", "openai": "^4.56.1", "react": "^18.3.1", diff --git a/frontend/src/components/ChatBox.tsx b/frontend/src/components/ChatBox.tsx index d33872458..5d1236ce1 100644 --- a/frontend/src/components/ChatBox.tsx +++ b/frontend/src/components/ChatBox.tsx @@ -11,11 +11,12 @@ import { Route as ChatRoute } from "@/routes/_auth.chat.$chatId"; import { ChatMessage } from "@/state/LocalStateContext"; import { useNavigate, useRouter } from "@tanstack/react-router"; import { ModelSelector } from "@/components/ModelSelector"; +import { encode } from "gpt-tokenizer"; -// Rough token estimation function +// Accurate token counting using gpt-tokenizer function estimateTokenCount(text: string): number { - // A very rough estimation: ~4 characters per token on average - return Math.ceil(text.length / 4); + // Use gpt-tokenizer for accurate token counting + return encode(text).length; } function TokenWarning({ From 87dcd54cca1861572ab5482d0e471de059b79f0e Mon Sep 17 00:00:00 2001 From: Tony Giorgio Date: Wed, 2 Jul 2025 17:13:40 -0500 Subject: [PATCH 2/6] feat: add model-specific token limits with progressive warnings MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Added token limits to MODEL_CONFIG (Llama/Gemma: 70k, DeepSeek: 64k, default: 64k) - Implemented progressive warning thresholds: - 50%: "Tip" to compress chat - 95%: "Warning" with compress option available - 99%: "Error" with disabled submit button (textarea remains editable) - Updated UI styling for different severity levels - Users can delete text at 99% to get under the limit 🤖 Generated with [Claude Code](https://claude.ai/code) Co-Authored-By: Claude --- frontend/src/components/ChatBox.tsx | 99 ++++++++++++++++------- frontend/src/components/ModelSelector.tsx | 25 ++++-- 2 files changed, 91 insertions(+), 33 deletions(-) diff --git a/frontend/src/components/ChatBox.tsx b/frontend/src/components/ChatBox.tsx index 5d1236ce1..887fdea9b 100644 --- a/frontend/src/components/ChatBox.tsx +++ b/frontend/src/components/ChatBox.tsx @@ -6,11 +6,10 @@ import { useLocalState } from "@/state/useLocalState"; import { cn, useIsMobile } from "@/utils/utils"; import { useQuery } from "@tanstack/react-query"; import { getBillingService } from "@/billing/billingService"; -import { BillingStatus } from "@/billing/billingApi"; import { Route as ChatRoute } from "@/routes/_auth.chat.$chatId"; import { ChatMessage } from "@/state/LocalStateContext"; import { useNavigate, useRouter } from "@tanstack/react-router"; -import { ModelSelector } from "@/components/ModelSelector"; +import { ModelSelector, getModelTokenLimit } from "@/components/ModelSelector"; import { encode } from "gpt-tokenizer"; // Accurate token counting using gpt-tokenizer @@ -24,17 +23,17 @@ function TokenWarning({ currentInput, chatId, className, - billingStatus, onCompress, - isCompressing = false + isCompressing = false, + modelId }: { messages: ChatMessage[]; currentInput: string; chatId?: string; className?: string; - billingStatus?: BillingStatus; onCompress?: () => void; isCompressing?: boolean; + modelId: string; }) { const totalTokens = messages.reduce((acc, msg) => acc + estimateTokenCount(msg.content), 0) + @@ -42,18 +41,16 @@ function TokenWarning({ const navigate = useNavigate(); - // Check if user is on starter plan - const isStarter = billingStatus?.product_name?.toLowerCase().includes("starter") || false; + // Get model-specific token limit + const tokenLimit = getModelTokenLimit(modelId); + const tokenPercentage = (totalTokens / tokenLimit) * 100; - // Token thresholds for different plan types - const STARTER_WARNING_THRESHOLD = 4000; - const PRO_WARNING_THRESHOLD = 10000; + // Only show warning if above 50% + if (tokenPercentage < 50) return null; - // Different thresholds for starter vs pro users - const warningThreshold = isStarter ? STARTER_WARNING_THRESHOLD : PRO_WARNING_THRESHOLD; - - // Only show warning if above the threshold - if (totalTokens < warningThreshold) return null; + // Determine the severity and behavior based on percentage + const isAt95Percent = tokenPercentage >= 95; + const isAt99Percent = tokenPercentage >= 99; const handleNewChat = async (e: React.MouseEvent) => { e.preventDefault(); @@ -66,7 +63,17 @@ function TokenWarning({ } }; - // Determine button text based on compression state + // Get appropriate message and styling based on threshold + const getMessage = () => { + if (isAt99Percent) { + return "This chat is too long to continue."; + } else if (isAt95Percent) { + return "Chat is at capacity. Compress to continue."; + } else { + return "This chat is getting long. Compress it to save tokens."; + } + }; + const getButtonText = () => { if (isCompressing) { return { desktop: "Compressing...", mobile: "Compressing..." }; @@ -79,26 +86,46 @@ function TokenWarning({ const buttonText = getButtonText(); + // Determine background color based on severity + const bgClass = isAt99Percent + ? "bg-destructive/20 border border-destructive/30" + : isAt95Percent + ? "bg-warning/20 border border-warning/30" + : "bg-muted/50"; + return (
- Tip: - This chat is getting long. Compress it to save tokens. + + {isAt99Percent ? "Error:" : isAt95Percent ? "Warning:" : "Tip:"} + + {getMessage()}
- {chatId && ( + {chatId && !isAt99Percent && (