diff --git a/.gitignore b/.gitignore index 91b19f6d..564687d4 100644 --- a/.gitignore +++ b/.gitignore @@ -93,3 +93,4 @@ typings/ out/ .specstory .specstory/ +nul diff --git a/package-lock.json b/package-lock.json index 0367a16a..7eaa412e 100644 --- a/package-lock.json +++ b/package-lock.json @@ -12,6 +12,7 @@ "@google/genai": "^1.2.0", "@huggingface/transformers": "^3.8.1", "electron-squirrel-startup": "^1.0.1", + "lru-cache": "^11.0.0", "ollama": "^0.6.3", "ws": "^8.19.0" }, @@ -2484,6 +2485,16 @@ "url": "https://github.com/sponsors/isaacs" } }, + "node_modules/cacache/node_modules/lru-cache": { + "version": "7.18.3", + "resolved": "https://registry.npmjs.org/lru-cache/-/lru-cache-7.18.3.tgz", + "integrity": "sha512-jumlc0BIUrS3qJGgIkWZsyfAM7NCWiBcCDhnd+3NNM5KbBmLTgHVfWBcg6W+rLUsIpzpERPsvwUP7CckAQSOoA==", + "dev": true, + "license": "ISC", + "engines": { + "node": ">=12" + } + }, "node_modules/cacache/node_modules/minimatch": { "version": "5.1.6", "resolved": "https://registry.npmjs.org/minimatch/-/minimatch-5.1.6.tgz", @@ -5791,13 +5802,12 @@ } }, "node_modules/lru-cache": { - "version": "7.18.3", - "resolved": "https://registry.npmjs.org/lru-cache/-/lru-cache-7.18.3.tgz", - "integrity": "sha512-jumlc0BIUrS3qJGgIkWZsyfAM7NCWiBcCDhnd+3NNM5KbBmLTgHVfWBcg6W+rLUsIpzpERPsvwUP7CckAQSOoA==", - "dev": true, - "license": "ISC", + "version": "11.3.3", + "resolved": "https://registry.npmjs.org/lru-cache/-/lru-cache-11.3.3.tgz", + "integrity": "sha512-JvNw9Y81y33E+BEYPr0U7omo+U9AySnsMsEiXgwT6yqd31VQWTLNQqmT4ou5eqPFUrTfIDFta2wKhB1hyohtAQ==", + "license": "BlueOak-1.0.0", "engines": { - "node": ">=12" + "node": "20 || >=22" } }, "node_modules/macos-alias": { @@ -5843,6 +5853,16 @@ "node": "^12.13.0 || ^14.15.0 || >=16.0.0" } }, + "node_modules/make-fetch-happen/node_modules/lru-cache": { + "version": "7.18.3", + "resolved": "https://registry.npmjs.org/lru-cache/-/lru-cache-7.18.3.tgz", + "integrity": "sha512-jumlc0BIUrS3qJGgIkWZsyfAM7NCWiBcCDhnd+3NNM5KbBmLTgHVfWBcg6W+rLUsIpzpERPsvwUP7CckAQSOoA==", + "dev": true, + "license": "ISC", + "engines": { + "node": ">=12" + } + }, "node_modules/map-age-cleaner": { "version": "0.1.3", "resolved": "https://registry.npmjs.org/map-age-cleaner/-/map-age-cleaner-0.1.3.tgz", diff --git a/package.json b/package.json index b2b61fad..ed92ab32 100644 --- a/package.json +++ b/package.json @@ -29,6 +29,7 @@ "@google/genai": "^1.2.0", "@huggingface/transformers": "^3.8.1", "electron-squirrel-startup": "^1.0.1", + "lru-cache": "^11.0.0", "ollama": "^0.6.3", "ws": "^8.19.0" }, diff --git a/src/components/app/AppHeader.js b/src/components/app/AppHeader.js index 0d74d646..d55a3275 100644 --- a/src/components/app/AppHeader.js +++ b/src/components/app/AppHeader.js @@ -120,11 +120,29 @@ export class AppHeader extends LitElement { .update-button:hover { background: rgba(241, 76, 76, 0.1); } + + .partial-question { + color: var(--accent-color, #22c55e); + font-style: italic; + font-size: var(--header-font-size-small); + animation: pulse 1.5s infinite; + } + + @keyframes pulse { + 0%, + 100% { + opacity: 1; + } + 50% { + opacity: 0.6; + } + } `; static properties = { currentView: { type: String }, statusText: { type: String }, + partialQuestion: { type: String }, startTime: { type: Number }, onCustomizeClick: { type: Function }, onHelpClick: { type: Function }, @@ -140,6 +158,7 @@ export class AppHeader extends LitElement { super(); this.currentView = 'main'; this.statusText = ''; + this.partialQuestion = ''; this.startTime = null; this.onCustomizeClick = () => {}; this.onHelpClick = () => {}; @@ -280,33 +299,56 @@ export class AppHeader extends LitElement { ${this.currentView === 'assistant' ? html` ${elapsedTime} - ${this.statusText} + ${this.partialQuestion + ? html` + Listening (collecting) + "${this.partialQuestion}" + ` + : html`${this.statusText}`} ${this.isClickThrough ? html`click-through` : ''} ` : ''} ${this.currentView === 'main' ? html` - ${this.updateAvailable ? html` - - ` : ''} + ${this.updateAvailable + ? html` + + ` + : ''} ` @@ -319,14 +361,18 @@ export class AppHeader extends LitElement { ` : html` `} diff --git a/src/components/app/CheatingDaddyApp.js b/src/components/app/CheatingDaddyApp.js index e1d859c8..35eb7ee9 100644 --- a/src/components/app/CheatingDaddyApp.js +++ b/src/components/app/CheatingDaddyApp.js @@ -89,15 +89,15 @@ export class CheatingDaddyApp extends LitElement { } .traffic-light.close { - background: #FF5F57; + background: #ff5f57; } .traffic-light.minimize { - background: #FEBC2E; + background: #febc2e; } .traffic-light.maximize { - background: #28C840; + background: #28c840; } .window-controls { @@ -115,7 +115,9 @@ export class CheatingDaddyApp extends LitElement { background: transparent; color: var(--text-secondary); font-size: 15px; - transition: background var(--transition), color var(--transition); + transition: + background var(--transition), + color var(--transition); } .window-control:hover { @@ -136,7 +138,10 @@ export class CheatingDaddyApp extends LitElement { display: flex; flex-direction: column; padding: 42px 0 var(--space-md) 0; - transition: width var(--transition), min-width var(--transition), opacity var(--transition); + transition: + width var(--transition), + min-width var(--transition), + opacity var(--transition); } .sidebar.hidden { @@ -180,7 +185,9 @@ export class CheatingDaddyApp extends LitElement { font-size: var(--font-size-sm); font-weight: var(--font-weight-medium); cursor: pointer; - transition: color var(--transition), background var(--transition); + transition: + color var(--transition), + background var(--transition); border: none; background: none; width: 100%; @@ -223,7 +230,9 @@ export class CheatingDaddyApp extends LitElement { font-weight: var(--font-weight-medium); cursor: pointer; text-align: left; - transition: background var(--transition), border-color var(--transition); + transition: + background var(--transition), + border-color var(--transition); animation: update-wobble 5s ease-in-out infinite; } @@ -233,11 +242,23 @@ export class CheatingDaddyApp extends LitElement { } @keyframes update-wobble { - 0%, 90%, 100% { transform: rotate(0deg); } - 92% { transform: rotate(-2deg); } - 94% { transform: rotate(2deg); } - 96% { transform: rotate(-1.5deg); } - 98% { transform: rotate(1.5deg); } + 0%, + 90%, + 100% { + transform: rotate(0deg); + } + 92% { + transform: rotate(-2deg); + } + 94% { + transform: rotate(2deg); + } + 96% { + transform: rotate(-1.5deg); + } + 98% { + transform: rotate(1.5deg); + } } .update-btn svg { @@ -340,6 +361,26 @@ export class CheatingDaddyApp extends LitElement { color: var(--text-primary); } + .live-bar-text.partial-question { + color: var(--accent); + font-style: italic; + animation: pulse 1.5s infinite; + max-width: 300px; + white-space: nowrap; + overflow: hidden; + text-overflow: ellipsis; + } + + @keyframes pulse { + 0%, + 100% { + opacity: 1; + } + 50% { + opacity: 0.6; + } + } + /* Content inner */ .content-inner { flex: 1; @@ -383,6 +424,7 @@ export class CheatingDaddyApp extends LitElement { static properties = { currentView: { type: String }, statusText: { type: String }, + partialQuestion: { type: String }, startTime: { type: Number }, isRecording: { type: Boolean }, sessionActive: { type: Boolean }, @@ -406,6 +448,7 @@ export class CheatingDaddyApp extends LitElement { super(); this.currentView = 'main'; this.statusText = ''; + this.partialQuestion = ''; this.startTime = null; this.isRecording = false; this.sessionActive = false; @@ -457,10 +500,7 @@ export class CheatingDaddyApp extends LitElement { async _loadFromStorage() { try { - const [config, prefs] = await Promise.all([ - cheatingDaddy.storage.getConfig(), - cheatingDaddy.storage.getPreferences() - ]); + const [config, prefs] = await Promise.all([cheatingDaddy.storage.getConfig(), cheatingDaddy.storage.getPreferences()]); this.currentView = config.onboarded ? 'main' : 'onboarding'; this.selectedProfile = prefs.selectedProfile || 'interview'; @@ -484,6 +524,10 @@ export class CheatingDaddyApp extends LitElement { cheatingDaddy.ipc.on('new-response', response => this.addNewResponse(response)), cheatingDaddy.ipc.on('update-response', response => this.updateCurrentResponse(response)), cheatingDaddy.ipc.on('update-status', status => this.setStatus(status)), + cheatingDaddy.ipc.on('partial-question', question => { + this.partialQuestion = question; + this.requestUpdate(); + }), cheatingDaddy.ipc.on('click-through-toggled', isEnabled => { this._isClickThrough = isEnabled; }), @@ -738,10 +782,7 @@ export class CheatingDaddyApp extends LitElement { switch (this.currentView) { case 'onboarding': return html` - this.handleOnboardingComplete()} - .onClose=${() => this.handleClose()} - > + this.handleOnboardingComplete()} .onClose=${() => this.handleClose()}> `; case 'main': @@ -812,12 +853,76 @@ export class CheatingDaddyApp extends LitElement { renderSidebar() { const items = [ - { id: 'main', label: 'Home', icon: html`` }, - { id: 'ai-customize', label: 'AI Customization', icon: html`` }, - { id: 'history', label: 'History', icon: html`` }, - { id: 'customize', label: 'Settings', icon: html`` }, - { id: 'feedback', label: 'Feedback', icon: html`` }, - { id: 'help', label: 'Help', icon: html`` }, + { + id: 'main', + label: 'Home', + icon: html` + + + + + `, + }, + { + id: 'ai-customize', + label: 'AI Customization', + icon: html` + + `, + }, + { + id: 'history', + label: 'History', + icon: html` + + + + + `, + }, + { + id: 'customize', + label: 'Settings', + icon: html` + + + + + `, + }, + { + id: 'feedback', + label: 'Feedback', + icon: html` + + + + + `, + }, + { + id: 'help', + label: 'Help', + icon: html` + + + + + `, + }, ]; return html` @@ -826,26 +931,36 @@ export class CheatingDaddyApp extends LitElement {

Cheating Daddy

`; @@ -868,15 +983,22 @@ export class CheatingDaddyApp extends LitElement {
-
- ${profileLabels[this.selectedProfile] || 'Session'} -
+
${profileLabels[this.selectedProfile] || 'Session'}
- ${this.statusText ? html`${this.statusText}` : ''} + ${this.partialQuestion + ? html` + Listening (collecting) + "${this.partialQuestion}" + ` + : html`${this.statusText ? html`${this.statusText}` : ''}`} ${this.getElapsedTime()} ${this._isClickThrough ? html`[click through]` : ''} this.handleHideToggle()}>[hide] @@ -888,11 +1010,7 @@ export class CheatingDaddyApp extends LitElement { render() { // Onboarding is fullscreen, no sidebar if (this.currentView === 'onboarding') { - return html` -
- ${this.renderCurrentView()} -
- `; + return html`
${this.renderCurrentView()}
`; } const isLive = this._isLiveMode(); @@ -906,7 +1024,9 @@ export class CheatingDaddyApp extends LitElement {
- +
` @@ -922,9 +1042,7 @@ export class CheatingDaddyApp extends LitElement { ${this.renderSidebar()}
${isLive ? this.renderLiveBar() : ''} -
- ${this.renderCurrentView()} -
+
${this.renderCurrentView()}
`; diff --git a/src/preload.js b/src/preload.js index 6a531e25..1502b990 100644 --- a/src/preload.js +++ b/src/preload.js @@ -50,6 +50,7 @@ const validEventChannels = new Set([ 'new-response', 'update-response', 'update-status', + 'partial-question', 'click-through-toggled', 'reconnect-failed', 'whisper-downloading', diff --git a/src/utils/cache.js b/src/utils/cache.js new file mode 100644 index 00000000..d822ab4e --- /dev/null +++ b/src/utils/cache.js @@ -0,0 +1,106 @@ +// Simple Question Cache - Hash-based exact match caching +// Uses Map with time-to-live for simple caching + +class SimpleCache { + constructor(maxSize = 500, ttlMs = 3600000) { + this.cache = new Map(); + this.maxSize = maxSize; + this.ttlMs = ttlMs; + this.hits = 0; + this.misses = 0; + } + + hashQuestion(text) { + const normalized = text.toLowerCase().replace(/[^\w\s]/g, '').replace(/\s+/g, ' ').trim(); + let hash = 0; + for (let i = 0; i < normalized.length; i++) { + const char = normalized.charCodeAt(i); + hash = ((hash << 5) - hash) + char; + hash = hash & hash; + } + return hash.toString(36); + } + + get(question) { + const key = this.hashQuestion(question); + const entry = this.cache.get(key); + + if (!entry) { + this.misses++; + return null; + } + + // Check if expired + if (Date.now() - entry.timestamp > this.ttlMs) { + this.cache.delete(key); + this.misses++; + return null; + } + + this.hits++; + return entry.response; + } + + set(question, response) { + // Remove oldest if at capacity + if (this.cache.size >= this.maxSize) { + const firstKey = this.cache.keys().next().value; + this.cache.delete(firstKey); + } + + const key = this.hashQuestion(question); + this.cache.set(key, { + question: question, + response: response, + timestamp: Date.now() + }); + } + + getStats() { + const total = this.hits + this.misses; + return { + size: this.cache.size, + hits: this.hits, + misses: this.misses, + hitRate: total > 0 ? (this.hits / total) * 100 : 0 + }; + } + + clear() { + this.cache.clear(); + this.hits = 0; + this.misses = 0; + } +} + +const cache = new SimpleCache(500, 3600000); + +function getCachedResponse(question) { + const response = cache.get(question); + if (response) { + console.log('[Cache] HIT for question:', question.substring(0, 50) + '...'); + return { response }; + } + return null; +} + +function setCachedResponse(question, response) { + cache.set(question, response); + console.log('[Cache] SET for question:', question.substring(0, 50) + '...'); +} + +function getCacheStats() { + return cache.getStats(); +} + +function clearCache() { + cache.clear(); + console.log('[Cache] Cleared'); +} + +module.exports = { + getCachedResponse, + setCachedResponse, + getCacheStats, + clearCache +}; \ No newline at end of file diff --git a/src/utils/gemini.js b/src/utils/gemini.js index 51ff30df..3394ae27 100644 --- a/src/utils/gemini.js +++ b/src/utils/gemini.js @@ -5,6 +5,8 @@ const { saveDebugAudio } = require('../audioUtils'); const { getSystemPrompt } = require('./prompts'); const { getAvailableModel, incrementLimitCount, getApiKey, getGroqApiKey, incrementCharUsage, getModelForToday } = require('../storage'); const { connectCloud, sendCloudAudio, sendCloudText, sendCloudImage, closeCloud, isCloudActive, setOnTurnComplete } = require('./cloud'); +const { getCachedResponse, setCachedResponse, getCacheStats, clearCache } = require('./cache'); +const { shouldUseFastModel } = require('./router'); // Lazy-loaded to avoid circular dependency (localai.js imports from gemini.js) let _localai = null; @@ -29,6 +31,34 @@ let currentCustomPrompt = null; let isInitializingSession = false; let currentSystemPrompt = null; +// Fast response tracking +let isProcessingResponse = false; +let lastProcessedTranscription = ''; +let pendingTranscription = ''; +let pendingTranscriptionTimestamp = 0; +let pauseTimer = null; +let responseProcessedForCurrentTurn = false; + +// Check if question is complete (ends with ? or has complete sentences) +function isQuestionComplete(text) { + const trimmed = text.trim(); + return trimmed.endsWith('?') || (trimmed.match(/[.!?]+\s/g) || []).length >= 1; +} + +// Check if new input looks like continuation of previous (not a new question) +function isQuestionContinuation(newText, previousText) { + if (!previousText || previousText.trim().length === 0) return false; + + // If new text is very short (< 10 chars) and doesn't start with capital, it's likely a fragment + if (newText.trim().length < 10 && !/^[A-Z]/.test(newText.trim())) { + return true; + } + + // If previous text ends with a word (not punctuation), new text is likely continuation + const endsWithWord = /[a-zA-Z]$/.test(previousText.trim()); + return endsWithWord; +} + function formatSpeakerResults(results) { let text = ''; for (const result of results) { @@ -46,7 +76,6 @@ module.exports.formatSpeakerResults = formatSpeakerResults; let systemAudioProc = null; let messageBuffer = ''; - // Reconnection variables let isUserClosing = false; let sessionParams = null; @@ -68,9 +97,7 @@ function buildContextMessage() { if (validTurns.length === 0) return null; - const contextLines = validTurns.map(turn => - `[Interviewer]: ${turn.transcription.trim()}\n[Your answer]: ${turn.ai_response.trim()}` - ); + const contextLines = validTurns.map(turn => `[Interviewer]: ${turn.transcription.trim()}\n[Your answer]: ${turn.ai_response.trim()}`); return `Session reconnected. Here's the conversation so far:\n\n${contextLines.join('\n\n')}\n\nContinue from here.`; } @@ -91,7 +118,7 @@ function initializeNewSession(profile = null, customPrompt = null) { sendToRenderer('save-session-context', { sessionId: currentSessionId, profile: profile, - customPrompt: customPrompt || '' + customPrompt: customPrompt || '', }); } } @@ -127,7 +154,7 @@ function saveScreenAnalysis(prompt, response, model) { timestamp: Date.now(), prompt: prompt, response: response.trim(), - model: model + model: model, }; screenAnalysisHistory.push(analysisEntry); @@ -139,7 +166,7 @@ function saveScreenAnalysis(prompt, response, model) { analysis: analysisEntry, fullHistory: screenAnalysisHistory, profile: currentProfile, - customPrompt: currentCustomPrompt + customPrompt: currentCustomPrompt, }); } @@ -221,7 +248,7 @@ function summarizeLiveServerMessage(message) { } if (serverContent.modelTurn?.parts) { - return `modelTurn parts=${serverContent.modelTurn.parts.length}`; + // Log periodically - there are many chunks } return null; @@ -230,19 +257,19 @@ function summarizeLiveServerMessage(message) { // helper to check if groq has been configured function hasGroqKey() { const key = getGroqApiKey(); - return key && key.trim() != '' + return key && key.trim() != ''; } -function trimConversationHistoryForGemma(history, maxChars=42000) { - if(!history || history.length === 0) return []; +function trimConversationHistoryForGemma(history, maxChars = 42000) { + if (!history || history.length === 0) return []; let totalChars = 0; const trimmed = []; - for(let i = history.length - 1; i >= 0; i--) { + for (let i = history.length - 1; i >= 0; i--) { const turn = history[i]; const turnChars = (turn.content || '').length; - if(totalChars + turnChars > maxChars) break; + if (totalChars + turnChars > maxChars) break; totalChars += turnChars; trimmed.unshift(turn); } @@ -250,10 +277,17 @@ function trimConversationHistoryForGemma(history, maxChars=42000) { } function stripThinkingTags(text) { - return text.replace(/[\s\S]*?<\/think>/g, '').trim(); + if (!text) return ''; + let result = text; + // Remove ALL instances of thinking tags (using global flag) + result = result.replace(/[\s\S]*?<\/think>/gi, ''); + // Also handle cases where tags might be split across streamed chunks (incomplete) + result = result.replace(//gi, ''); + result = result.replace(/<\/think>/gi, ''); + return result.trim(); } -async function sendToGroq(transcription) { +async function sendToGroq(transcription, isPartial = false, preferFastModel = null) { const groqApiKey = getGroqApiKey(); if (!groqApiKey) { console.log('No Groq API key configured, skipping Groq response'); @@ -265,18 +299,73 @@ async function sendToGroq(transcription) { return; } - const modelToUse = getModelForToday(); - if (!modelToUse) { + // Check cache for exact or similar questions (skip for partial transcriptions) + if (!isPartial) { + const cachedResponse = getCachedResponse(transcription); + if (cachedResponse) { + const { response } = cachedResponse; + console.log('[Cache] Using cached response'); + sendToRenderer('new-response', response); + + groqConversationHistory.push({ + role: 'user', + content: transcription.trim(), + }); + groqConversationHistory.push({ + role: 'assistant', + content: response, + }); + + saveConversationTurn(transcription, response); + isProcessingResponse = false; + sendToRenderer('update-status', 'Listening... (cached)'); + return; + } + } + + // Determine which model to use based on query complexity + // preferFastModel can be: true (force fast), false (force complex), null (auto-detect) + let useFastModel = preferFastModel; + if (useFastModel === null) { + useFastModel = shouldUseFastModel(transcription); + } + + // Select model based on complexity + // Fast models: llama-3.1-70b-versatile (default), mixtral-8x7b-32768 + // Complex models: llama-3.1-405b-reasoning-ultra, deepseek-r1-distill-llama-70b + let modelToUse; + if (useFastModel) { + modelToUse = getModelForToday() || 'llama-3.1-70b-versatile'; + } else { + // Use a more capable model for complex queries + modelToUse = 'llama-3.1-405b-reasoning-ultra'; + } + + // Check if we have the model available + const availableModel = getModelForToday(); + if (!availableModel) { console.log('All Groq daily limits exhausted'); sendToRenderer('update-status', 'Groq limits reached for today'); return; } - console.log(`Sending to Groq (${modelToUse}):`, transcription.substring(0, 100) + '...'); + // If the complex model isn't available in today's pool, fall back to available model + if (!useFastModel && availableModel !== modelToUse) { + console.log('[Router] Complex model not in daily pool, using available:', availableModel); + modelToUse = availableModel; + } + + console.log(`Sending to Groq (${modelToUse}, fast=${useFastModel}):`, transcription.substring(0, 100) + '...'); + + // Mark as processing to prevent turnComplete from clearing state prematurely + isProcessingResponse = true; + + // Clear previous response and show processing status + sendToRenderer('new-response', 'Processing...'); groqConversationHistory.push({ role: 'user', - content: transcription.trim() + content: transcription.trim(), }); if (groqConversationHistory.length > 20) { @@ -287,24 +376,34 @@ async function sendToGroq(transcription) { const response = await fetch('https://api.groq.com/openai/v1/chat/completions', { method: 'POST', headers: { - 'Authorization': `Bearer ${groqApiKey}`, - 'Content-Type': 'application/json' + Authorization: `Bearer ${groqApiKey}`, + 'Content-Type': 'application/json', }, body: JSON.stringify({ model: modelToUse, - messages: [ - { role: 'system', content: currentSystemPrompt || 'You are a helpful assistant.' }, - ...groqConversationHistory - ], + messages: [{ role: 'system', content: currentSystemPrompt || 'You are a helpful assistant.' }, ...groqConversationHistory], stream: true, - temperature: 0.7, - max_tokens: 1024 - }) + temperature: useFastModel ? 0.7 : 0.3, // Lower temp for complex reasoning + max_tokens: useFastModel ? 1024 : 2048, // More tokens for complex + }), }); if (!response.ok) { const errorText = await response.text(); console.error('Groq API error:', response.status, errorText); + + // Fallback to Gemini on rate limit (429) + if (response.status === 429) { + console.log('[Fast Mode] Groq rate limited, falling back to Gemini...'); + sendToRenderer('update-status', 'Rate limited, using Gemini...'); + // Retry with Gemini + const apiKey = getApiKey(); + if (apiKey) { + sendToGemma(transcription); + return; + } + } + sendToRenderer('update-status', `Groq error: ${response.status}`); return; } @@ -333,6 +432,10 @@ async function sendToGroq(transcription) { fullText += token; const displayText = stripThinkingTags(fullText); if (displayText) { + // Only log occasionally to avoid spam + if (displayText.length % 200 === 0) { + console.log('[Fast Mode Debug] Response chunk:', displayText.substring(0, 20) + '...'); + } sendToRenderer(isFirst ? 'new-response' : 'update-response', displayText); isFirst = false; } @@ -357,17 +460,24 @@ async function sendToGroq(transcription) { if (cleanedResponse) { groqConversationHistory.push({ role: 'assistant', - content: cleanedResponse + content: cleanedResponse, }); + // Cache the response (skip partial transcriptions) + if (!isPartial) { + setCachedResponse(transcription, cleanedResponse); + } + saveConversationTurn(transcription, cleanedResponse); } console.log(`Groq response completed (${modelToUse})`); + // Allow next question to be processed immediately + isProcessingResponse = false; sendToRenderer('update-status', 'Listening...'); - } catch (error) { console.error('Error calling Groq API:', error); + isProcessingResponse = false; sendToRenderer('update-status', 'Groq error: ' + error.message); } } @@ -386,9 +496,15 @@ async function sendToGemma(transcription) { console.log('Sending to Gemma:', transcription.substring(0, 100) + '...'); + // Mark as processing to prevent turnComplete from clearing state prematurely + isProcessingResponse = true; + + // Clear previous response and show processing status + sendToRenderer('new-response', 'Processing...'); + groqConversationHistory.push({ role: 'user', - content: transcription.trim() + content: transcription.trim(), }); const trimmedHistory = trimConversationHistoryForGemma(groqConversationHistory, 42000); @@ -398,14 +514,14 @@ async function sendToGemma(transcription) { const messages = trimmedHistory.map(msg => ({ role: msg.role === 'assistant' ? 'model' : 'user', - parts: [{ text: msg.content }] + parts: [{ text: msg.content }], })); const systemPrompt = currentSystemPrompt || 'You are a helpful assistant.'; const messagesWithSystem = [ { role: 'user', parts: [{ text: systemPrompt }] }, { role: 'model', parts: [{ text: 'Understood. I will follow these instructions.' }] }, - ...messages + ...messages, ]; const response = await ai.models.generateContentStream({ @@ -435,7 +551,7 @@ async function sendToGemma(transcription) { if (fullText.trim()) { groqConversationHistory.push({ role: 'assistant', - content: fullText.trim() + content: fullText.trim(), }); if (groqConversationHistory.length > 40) { @@ -446,10 +562,11 @@ async function sendToGemma(transcription) { } console.log('Gemma response completed'); + isProcessingResponse = false; sendToRenderer('update-status', 'Listening...'); - } catch (error) { console.error('Error calling Gemma API:', error); + isProcessingResponse = false; sendToRenderer('update-status', 'Gemma error: ' + error.message); } } @@ -495,6 +612,16 @@ async function initializeGeminiSession(apiKey, customPrompt = '', profile = 'int callbacks: { onopen: function () { sendToRenderer('update-status', 'Live session connected'); + // Reset state for new session + responseProcessedForCurrentTurn = false; + lastProcessedTranscription = ''; + pendingTranscription = ''; + pendingTranscriptionTimestamp = 0; + isProcessingResponse = false; + if (pauseTimer) { + clearTimeout(pauseTimer); + pauseTimer = null; + } }, onmessage: function (message) { const summary = summarizeLiveServerMessage(message); @@ -503,31 +630,220 @@ async function initializeGeminiSession(apiKey, customPrompt = '', profile = 'int } // Handle input transcription (what was spoken) + let newInputText = ''; if (message.serverContent?.inputTranscription?.results) { currentTranscription += formatSpeakerResults(message.serverContent.inputTranscription.results); + newInputText = formatSpeakerResults(message.serverContent.inputTranscription.results); } else if (message.serverContent?.inputTranscription?.text) { const text = message.serverContent.inputTranscription.text; if (text.trim() !== '') { currentTranscription += text; + newInputText = text; + } + } + + // FAST MODE: Wait for complete question before processing + // A question is complete when: + // 1. Ends with "?" OR + // 2. Has complete sentence (ends with .!) OR + // 3. No new speech for 2 seconds (pause detection) + // 4. turnComplete fires + const inputText = message.serverContent?.inputTranscription?.text; + + if (inputText && inputText.trim().length > 0) { + // Check if this is a continuation of previous incomplete question + const isContinuation = isQuestionContinuation(inputText, pendingTranscription); + + // DEBUG: Log full question state (only first few chars to reduce spam) + console.log('[Fast Mode Debug]', { + input: inputText.substring(0, 20), + isContinuation, + processing: isProcessingResponse, + }); + + if (isContinuation) { + // Update pending transcription with new content + pendingTranscription = inputText; + pendingTranscriptionTimestamp = Date.now(); + console.log('[Fast Mode] Continuing question:', inputText.substring(0, 30) + '...'); + + // If we have enough content now (25+ chars), process it + if (inputText.trim().length >= 25) { + console.log('[Fast Mode] Question now complete:', inputText.substring(0, 50) + '...'); + console.log('[Fast Mode Debug] Calling Groq/Gemma', { hasGroqKey: hasGroqKey() }); + sendToRenderer('update-status', 'Processing...'); + + if (hasGroqKey()) { + sendToGroq(inputText.trim(), false); + } else { + sendToGemma(inputText.trim()); + } + + lastProcessedTranscription = inputText; + pendingTranscription = ''; + pendingTranscriptionTimestamp = 0; + if (pauseTimer) { + clearTimeout(pauseTimer); + pauseTimer = null; + } + } else { + // Reset the pause timer - wait for more speech + if (pauseTimer) { + clearTimeout(pauseTimer); + } + sendToRenderer('update-status', 'Listening... (collecting)'); + sendToRenderer('partial-question', pendingTranscription); + + // Wait 2 seconds of silence to process incomplete question + pauseTimer = setTimeout(() => { + const timeSinceLastUpdate = Date.now() - pendingTranscriptionTimestamp; + + // Use whichever has content: pendingTranscription OR currentTranscription + const questionToProcess = + pendingTranscription.trim().length > 10 + ? pendingTranscription + : currentTranscription.trim().length > 10 + ? currentTranscription + : ''; + + if (questionToProcess.trim().length > 10 && timeSinceLastUpdate >= 1800) { + console.log('[Fast Mode] Processing after pause:', questionToProcess.substring(0, 50) + '...'); + sendToRenderer('update-status', 'Processing...'); + + if (hasGroqKey()) { + sendToGroq(questionToProcess.trim(), false); + } else { + sendToGemma(questionToProcess.trim()); + } + + lastProcessedTranscription = questionToProcess; + pendingTranscription = ''; + pendingTranscriptionTimestamp = 0; + currentTranscription = ''; + pauseTimer = null; + } + }, 2000); + } + } else { + // New question (not continuation) - check if complete + const isPunctuationComplete = isQuestionComplete(inputText); + const isLongEnough = inputText.trim().length >= 25; + + if (isPunctuationComplete || isLongEnough) { + console.log('[Fast Mode] Complete question detected:', inputText.substring(0, 50) + '...'); + sendToRenderer('update-status', 'Processing...'); + + if (hasGroqKey()) { + sendToGroq(inputText.trim(), false); + } else { + sendToGemma(inputText.trim()); + } + + lastProcessedTranscription = inputText; + pendingTranscription = ''; + pendingTranscriptionTimestamp = 0; + } else { + // Start new incomplete question + pendingTranscription = inputText; + pendingTranscriptionTimestamp = Date.now(); + console.log('[Fast Mode] New question started:', inputText.substring(0, 30) + '...'); + sendToRenderer('update-status', 'Listening... (collecting)'); + + // Wait 2 seconds of silence to process + if (pauseTimer) clearTimeout(pauseTimer); + pauseTimer = setTimeout(() => { + const timeSinceLastUpdate = Date.now() - pendingTranscriptionTimestamp; + + // Use whichever has content: pendingTranscription OR currentTranscription + const questionToProcess = + pendingTranscription.trim().length > 10 + ? pendingTranscription + : currentTranscription.trim().length > 10 + ? currentTranscription + : ''; + + if (questionToProcess.trim().length > 10 && timeSinceLastUpdate >= 1800) { + console.log('[Fast Mode] Processing after pause:', questionToProcess.substring(0, 50) + '...'); + sendToRenderer('update-status', 'Processing...'); + + if (hasGroqKey()) { + sendToGroq(questionToProcess.trim(), false); + } else { + sendToGemma(questionToProcess.trim()); + } + + lastProcessedTranscription = questionToProcess; + pendingTranscription = ''; + pendingTranscriptionTimestamp = 0; + currentTranscription = ''; + pauseTimer = null; + } + }, 2000); + } } } - // DISABLED: Gemini's outputTranscription - using Groq for faster responses instead - // if (message.serverContent?.outputTranscription?.text) { ... } + // Show Gemini's response as backup when Groq fails or is unavailable + if (message.serverContent?.outputTranscription?.text) { + const liveText = message.serverContent.outputTranscription.text.trim(); + if (liveText.length > 10) { + // Only show if we haven't already shown a response + // This serves as backup when Groq is rate limited + } + } + // Legacy: generationComplete handler (now rarely triggered due to fast mode) if (message.serverContent?.generationComplete) { - if (currentTranscription.trim() !== '') { + console.log('[Gemini] Generation complete'); + } + + // When turn is complete, check if there's a pending question to process + if (message.serverContent?.turnComplete) { + console.log('[Fast Mode Debug] turnComplete fired', { + pendingTranscription: pendingTranscription ? pendingTranscription.substring(0, 30) : '(empty)', + pendingLen: pendingTranscription.trim().length, + currentTranscription: currentTranscription ? currentTranscription.substring(0, 30) : '(empty)', + currentLen: currentTranscription.trim().length, + isProcessingResponse, + }); + console.log('[Fast Mode] Turn complete, ready for next question'); + + // Use pendingTranscription OR currentTranscription (whichever has content) + const questionToProcess = + pendingTranscription.trim().length > 10 + ? pendingTranscription + : currentTranscription.trim().length > 10 + ? currentTranscription + : ''; + + // If there's a question that wasn't processed, process it now + if (questionToProcess.trim().length > 10 && !isProcessingResponse) { + console.log('[Fast Mode] Processing pending question on turnComplete:', questionToProcess.substring(0, 50) + '...'); + sendToRenderer('update-status', 'Processing...'); + if (hasGroqKey()) { - sendToGroq(currentTranscription); + sendToGroq(questionToProcess.trim(), false); } else { - sendToGemma(currentTranscription); + sendToGemma(questionToProcess.trim()); } + + lastProcessedTranscription = questionToProcess; + pendingTranscription = ''; + pendingTranscriptionTimestamp = 0; currentTranscription = ''; + } else { + // Normal reset - no question to process + isProcessingResponse = false; + currentTranscription = ''; + lastProcessedTranscription = ''; + pendingTranscription = ''; + pendingTranscriptionTimestamp = 0; } - messageBuffer = ''; - } - if (message.serverContent?.turnComplete) { + if (pauseTimer) { + clearTimeout(pauseTimer); + pauseTimer = null; + } sendToRenderer('update-status', 'Listening...'); } }, @@ -1142,6 +1458,57 @@ function setupGeminiIpcHandlers(geminiSessionRef) { return { success: false, error: error.message }; } }); + + // Cache management IPC handlers + ipcMain.handle('get-cache-stats', async event => { + try { + return { success: true, data: getCacheStats() }; + } catch (error) { + console.error('Error getting cache stats:', error); + return { success: false, error: error.message }; + } + }); + + ipcMain.handle('clear-cache', async event => { + try { + clearCache(); + return { success: true }; + } catch (error) { + console.error('Error clearing cache:', error); + return { success: false, error: error.message }; + } + }); + + // Fast Response Mode - directly process text without Gemini Live + ipcMain.handle('fast-response', async (event, text) => { + try { + if (!text || typeof text !== 'string' || text.trim().length === 0) { + return { success: false, error: 'Invalid text' }; + } + + sendToRenderer('update-status', 'Processing...'); + + // Check cache first + const cachedResponse = getCachedResponse(text); + if (cachedResponse) { + const { response } = cachedResponse; + sendToRenderer('new-response', response); + sendToRenderer('update-status', 'Listening... (cached)'); + return { success: true, cached: true }; + } + + // Send to Groq directly + if (hasGroqKey()) { + await sendToGroq(text.trim(), false); + return { success: true, cached: false }; + } else { + return { success: false, error: 'No Groq API key configured' }; + } + } catch (error) { + console.error('Error in fast response:', error); + return { success: false, error: error.message }; + } + }); } module.exports = { diff --git a/src/utils/localai.js b/src/utils/localai.js index cc6583fb..76c6c765 100644 --- a/src/utils/localai.js +++ b/src/utils/localai.js @@ -119,27 +119,87 @@ async function loadWhisperPipeline(modelName) { isWhisperLoading = true; console.log('[LocalAI] Loading Whisper model:', modelName); sendToRenderer('whisper-downloading', true); - sendToRenderer('update-status', 'Loading Whisper model (first time may take a while)...'); + sendToRenderer('update-status', 'Loading Whisper...'); try { - // Dynamic import for ESM module - const { pipeline, env } = await import('@huggingface/transformers'); - // Cache models outside the asar archive so ONNX runtime can load them + const os = require('os'); + const freeMem = os.freemem() / (1024 * 1024 * 1024); + console.log('[LocalAI] Free memory:', freeMem.toFixed(2), 'GB'); + + if (freeMem < 2) { + throw new Error('Need at least 2GB free RAM'); + } + + // Import with timeout + let pipeline, env; + try { + const imported = await Promise.race([ + import('@huggingface/transformers'), + new Promise((_, reject) => setTimeout(() => reject(new Error('Import timeout (30s)')), 30000)), + ]); + pipeline = imported.pipeline; + env = imported.env; + } catch (importError) { + throw new Error('ML library failed. Try Cloud mode.'); + } + const { app } = require('electron'); const path = require('path'); - env.cacheDir = path.join(app.getPath('userData'), 'whisper-models'); - whisperPipeline = await pipeline('automatic-speech-recognition', modelName, { + const cacheDir = path.join(app.getPath('userData'), 'whisper-models'); + + env.cacheDir = cacheDir; + env.allowLocal = true; + + // Use tiny model - it's smaller and more reliable + const actualModel = 'Xenova/whisper-tiny'; + console.log('[LocalAI] Loading:', actualModel); + + whisperPipeline = await pipeline('automatic-speech-recognition', actualModel, { dtype: 'q8', - device: 'auto', + device: 'cpu', + parallel: 1, }); - console.log('[LocalAI] Whisper model loaded successfully'); + + console.log('[LocalAI] Whisper loaded:', actualModel); sendToRenderer('whisper-downloading', false); isWhisperLoading = false; return whisperPipeline; } catch (error) { - console.error('[LocalAI] Failed to load Whisper model:', error); + console.error('[LocalAI] Whisper error:', error.message); + + // Try clearing cache and retry once + if (!error.message.includes('cache') && !error.message.includes('timeout')) { + console.log('[LocalAI] Retrying with fresh download...'); + try { + const { pipeline, env } = await import('@huggingface/transformers'); + const { app } = require('electron'); + const path = require('path'); + + // Delete cache dir + const cacheDir = path.join(app.getPath('userData'), 'whisper-models'); + try { + require('fs').rmSync(cacheDir, { recursive: true, force: true }); + } catch (e) {} + + env.cacheDir = cacheDir; + env.allowLocal = false; + + whisperPipeline = await pipeline('automatic-speech-recognition', 'Xenova/whisper-tiny', { + dtype: 'q8', + device: 'cpu', + }); + + console.log('[LocalAI] Whisper loaded after cache clear'); + sendToRenderer('whisper-downloading', false); + isWhisperLoading = false; + return whisperPipeline; + } catch (retryError) { + console.error('[LocalAI] Retry failed:', retryError.message); + } + } + sendToRenderer('whisper-downloading', false); - sendToRenderer('update-status', 'Failed to load Whisper model: ' + error.message); + sendToRenderer('update-status', 'Whisper failed. Try Cloud mode.'); isWhisperLoading = false; return null; } @@ -224,10 +284,7 @@ async function sendToOllama(transcription) { } try { - const messages = [ - { role: 'system', content: currentSystemPrompt || 'You are a helpful assistant.' }, - ...localConversationHistory, - ]; + const messages = [{ role: 'system', content: currentSystemPrompt || 'You are a helpful assistant.' }, ...localConversationHistory]; const response = await ollamaClient.chat({ model: ollamaModel, diff --git a/src/utils/router.js b/src/utils/router.js new file mode 100644 index 00000000..6b1eb4b9 --- /dev/null +++ b/src/utils/router.js @@ -0,0 +1,109 @@ +// Smart Model Router - Route queries based on complexity +// Fast models: Groq/Llama (50-150ms TTFT) +// Complex models: GPT-4/Claude (200-400ms TTFT, better reasoning) + +const SIMPLE_KEYWORDS = [ + 'what', 'when', 'where', 'who', 'how', 'is', 'are', 'was', 'were', + 'do', 'does', 'did', 'can', 'could', 'will', 'would', 'should', + 'tell', 'explain', 'define', 'describe', 'list', 'name', 'give', + 'hi', 'hello', 'hey', 'thanks', 'thank' +]; + +const COMPLEX_KEYWORDS = [ + 'why', 'because', 'design', 'architecture', 'system', 'optimize', + 'compare', 'difference', 'between', 'tradeoff', 'advantages', + 'disadvantages', 'implement', 'debug', 'fix', 'complex', 'algorithm', + 'code', 'function', 'class', 'api', 'database', 'performance', + 'scalability', 'security', 'authentication', 'authorization' +]; + +const TECHNICAL_PATTERNS = [ + /\b\d+\s*=\s*\d+/, // variable assignment + /\bfunction\s+\w+/, // function definition + /\bclass\s+\w+/, // class definition + /\bimport\s+from/, // import statement + /\bexport\s+(default\s+)?/, // export statement + /\bif\s*\(/, // if statement + /\bfor\s*\(/, // for loop + /\bwhile\s*\(/, // while loop + /\breturn\s+/, // return statement + /\basync\s+/, // async keyword + /\bawait\s+/, // await keyword + /\binterface\s+/, // interface (TypeScript) + /\btype\s+\w+\s*=/, // type alias + /\bconsole\.(log|error)/, // console methods + /\bJSON\.(stringify|parse)/, // JSON methods + /\bArray\.(map|filter|reduce)/, // array methods + /\bPromise/, // Promise + /=>\s*{/, // arrow function + /\(\s*\)\s*=>/, // arrow function shorthand + /\btry\s*{/, // try block + /\bcatch\s*\(/, // catch block + /\bswitch\s*\(/, // switch statement + /\bcase\s+/, // case statement +]; + +function analyzeQueryComplexity(text) { + const lowerText = text.toLowerCase(); + const words = lowerText.split(/\s+/); + + let simpleScore = 0; + let complexScore = 0; + + // Check for simple keywords + for (const word of words) { + if (SIMPLE_KEYWORDS.includes(word)) { + simpleScore += 1; + } + } + + // Check for complex keywords + for (const word of words) { + if (COMPLEX_KEYWORDS.includes(word)) { + complexScore += 2; + } + } + + // Check for technical patterns (indicates coding question) + for (const pattern of TECHNICAL_PATTERNS) { + if (pattern.test(text)) { + complexScore += 3; + } + } + + // Check question length - longer questions tend to be more complex + const wordCount = words.length; + if (wordCount < 10) { + simpleScore += 1; + } else if (wordCount > 25) { + complexScore += 2; + } + + // Check for multiple question marks (multiple sub-questions) + const questionCount = (text.match(/\?/g) || []).length; + if (questionCount > 1) { + complexScore += questionCount; + } + + console.log('[Router] Complexity analysis:', { text: text.substring(0, 30) + '...', simpleScore, complexScore }); + + return { simpleScore, complexScore }; +} + +function shouldUseFastModel(text) { + const { simpleScore, complexScore } = analyzeQueryComplexity(text); + + // Use fast model if simple score > complex score or complex score is low + const useFast = simpleScore > complexScore || complexScore < 3; + + console.log('[Router] Using fast model:', useFast, '(simple:', simpleScore, 'complex:', complexScore, ')'); + return useFast; +} + +module.exports = { + analyzeQueryComplexity, + shouldUseFastModel, + SIMPLE_KEYWORDS, + COMPLEX_KEYWORDS, + TECHNICAL_PATTERNS +}; \ No newline at end of file