Skip to content

Commit b90e8ba

Browse files
committed
Fix: Use correct function name semantic_similarity (not cosine_similarity)
Fixed NameError in ask_pancake_enhanced. Issue: - Code called: cosine_similarity(query_emb, bite_emb) - But function is named: semantic_similarity() - Result: NameError: name 'cosine_similarity' is not defined Fix: ✅ Changed all occurrences to semantic_similarity() This is the correct function that was defined in Part 5 (Multi-Pronged Similarity Index) which computes cosine similarity between two embeddings.
1 parent 2a9cb5f commit b90e8ba

1 file changed

Lines changed: 1 addition & 1 deletion

File tree

‎POC_Nov20_BITE_PANCAKE.ipynb‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -4244,7 +4244,7 @@
42444244
"metadata": {},
42454245
"outputs": [],
42464246
"source": [
4247-
"# Enhanced conversational AI with reasoning and timing\ndef print_enhanced_response(query: str, answer: str, timing: Dict, top_bites: List[Dict], scores: List[Dict]):\n \"\"\"Pretty print conversational AI response with reasoning\"\"\"\n \n print(\"\\n\" + \"\u2554\" + \"=\"*98 + \"\u2557\")\n print(f\"\u2551 \ud83e\udd16 CONVERSATIONAL AI QUERY{' '*70}\u2551\")\n print(\"\u2560\" + \"=\"*98 + \"\u2563\")\n print(f\"\u2551 \u2753 {query[:92]:<92} \u2551\")\n print(\"\u255a\" + \"=\"*98 + \"\u255d\")\n \n # Timing breakdown\n print(f\"\\n\u23f1\ufe0f TIMING BREAKDOWN:\")\n print(f\" \u251c\u2500 Retrieval: {timing['retrieval']:.3f}s\")\n print(f\" \u251c\u2500 LLM Generation: {timing['llm']:.3f}s\")\n print(f\" \u2514\u2500 Total: {timing['total']:.3f}s\")\n \n # Token usage and cost estimate\n if 'tokens' in timing:\n print(f\"\\n\ud83d\udcb0 COST ESTIMATE (GPT-4):\")\n print(f\" \u251c\u2500 Input tokens: {timing['tokens']['prompt']:,}\")\n print(f\" \u251c\u2500 Output tokens: {timing['tokens']['completion']:,}\")\n print(f\" \u2514\u2500 Est. cost: ${timing['cost']:.4f}\")\n \n # Top BITEs with scores\n print(f\"\\n\ud83c\udfaf TOP RETRIEVED BITEs (by multi-pronged similarity):\")\n for i, (bite, score_breakdown) in enumerate(zip(top_bites[:5], scores[:5]), 1):\n print(f\"\\n {i}. {bite['Header']['type'].upper()} | {bite['Header']['timestamp'][:10]}\")\n print(f\" \u251c\u2500 Semantic: {score_breakdown['semantic']:.3f}\")\n print(f\" \u251c\u2500 Spatial: {score_breakdown['spatial']:.3f}\")\n print(f\" \u251c\u2500 Temporal: {score_breakdown['temporal']:.3f}\")\n print(f\" \u2514\u2500 COMBINED: {score_breakdown['combined']:.3f}\")\n \n # Show snippet of body\n body_str = json.dumps(bite['Body'], indent=2)[:150]\n print(f\" \ud83d\udcc4 {body_str.replace(chr(10), ' ')[:80]}...\")\n \n # LLM Answer\n print(f\"\\n\ud83d\udca1 ANSWER:\")\n print(\" \" + \"\u2500\"*96)\n for line in answer.split('\\n'):\n if line.strip():\n print(f\" {line}\")\n print(\" \" + \"\u2500\"*96)\n\n\ndef ask_pancake_enhanced(query: str, days_back: int = 30, top_k: int = 5):\n \"\"\"\n Enhanced conversational AI with reasoning chain and timing\n \"\"\"\n import time\n \n timing = {}\n retrieval_start = time.time()\n \n # Step 1: RAG retrieval\n # Convert days_back to time_filter for rag_query\n from datetime import datetime, timedelta\n cutoff_time = (datetime.utcnow() - timedelta(days=days_back)).isoformat() + 'Z'\n time_filter = f\">= '{cutoff_time}'\"\n \n results = rag_query(query, top_k=top_k, time_filter=time_filter)\n \n timing['retrieval'] = time.time() - retrieval_start\n \n if not results:\n return \"No relevant data found.\", timing, [], []\n \n # Extract top BITEs and score breakdowns\n top_bites = results # rag_query returns list of bite dicts\n score_breakdowns = []\n \n for bite in results:\n combined_score = bite.get('semantic_distance', 0.0)\n # Recompute individual scores for display\n query_emb = get_embedding(query)\n bite_text = f\"{bite['Header']['type']}: {json.dumps(bite['Body'])}\"\n bite_emb = get_embedding(bite_text)\n \n sem_sim = cosine_similarity(query_emb, bite_emb) if query_emb and bite_emb else 0.0\n spat_sim = spatial_similarity(bite['Header']['geoid'], bite['Header']['geoid'])\n temp_sim = temporal_similarity(bite['Header']['timestamp'])\n \n score_breakdowns.append({\n 'semantic': sem_sim,\n 'spatial': spat_sim,\n 'temporal': temp_sim,\n 'combined': combined_score\n })\n \n # Step 2: Build context for LLM\n context = \"Here is the relevant PANCAKE data:\\n\\n\"\n for i, (bite, score) in enumerate(results, 1):\n context += f\"{i}. {bite['Header']['type']} ({bite['Header']['timestamp'][:10]}):\\n\"\n context += f\"{json.dumps(bite['Body'], indent=2)}\\n\\n\"\n \n # Step 3: LLM generation with timing\n llm_start = time.time()\n \n messages = [\n {\"role\": \"system\", \"content\": \"You are an agricultural AI assistant. Answer questions based on the provided PANCAKE data (BITEs). Be specific, actionable, and reference the data.\"},\n {\"role\": \"user\", \"content\": f\"Question: {query}\\n\\n{context}\"}\n ]\n \n response = client.chat.completions.create(\n model=\"gpt-4\",\n messages=messages,\n temperature=0.7,\n max_tokens=500\n )\n \n timing['llm'] = time.time() - llm_start\n timing['total'] = time.time() - retrieval_start\n \n # Token usage and cost\n timing['tokens'] = {\n 'prompt': response.usage.prompt_tokens,\n 'completion': response.usage.completion_tokens,\n 'total': response.usage.total_tokens\n }\n \n # GPT-4 pricing (as of 2024): $0.03/1K input, $0.06/1K output\n timing['cost'] = (timing['tokens']['prompt'] / 1000 * 0.03) + \\\n (timing['tokens']['completion'] / 1000 * 0.06)\n \n answer = response.choices[0].message.content\n \n return answer, timing, top_bites, score_breakdowns\n\n\nprint(\"\u2713 Enhanced conversational AI functions defined\")\n"
4247+
"# Enhanced conversational AI with reasoning and timing\ndef print_enhanced_response(query: str, answer: str, timing: Dict, top_bites: List[Dict], scores: List[Dict]):\n \"\"\"Pretty print conversational AI response with reasoning\"\"\"\n \n print(\"\\n\" + \"\u2554\" + \"=\"*98 + \"\u2557\")\n print(f\"\u2551 \ud83e\udd16 CONVERSATIONAL AI QUERY{' '*70}\u2551\")\n print(\"\u2560\" + \"=\"*98 + \"\u2563\")\n print(f\"\u2551 \u2753 {query[:92]:<92} \u2551\")\n print(\"\u255a\" + \"=\"*98 + \"\u255d\")\n \n # Timing breakdown\n print(f\"\\n\u23f1\ufe0f TIMING BREAKDOWN:\")\n print(f\" \u251c\u2500 Retrieval: {timing['retrieval']:.3f}s\")\n print(f\" \u251c\u2500 LLM Generation: {timing['llm']:.3f}s\")\n print(f\" \u2514\u2500 Total: {timing['total']:.3f}s\")\n \n # Token usage and cost estimate\n if 'tokens' in timing:\n print(f\"\\n\ud83d\udcb0 COST ESTIMATE (GPT-4):\")\n print(f\" \u251c\u2500 Input tokens: {timing['tokens']['prompt']:,}\")\n print(f\" \u251c\u2500 Output tokens: {timing['tokens']['completion']:,}\")\n print(f\" \u2514\u2500 Est. cost: ${timing['cost']:.4f}\")\n \n # Top BITEs with scores\n print(f\"\\n\ud83c\udfaf TOP RETRIEVED BITEs (by multi-pronged similarity):\")\n for i, (bite, score_breakdown) in enumerate(zip(top_bites[:5], scores[:5]), 1):\n print(f\"\\n {i}. {bite['Header']['type'].upper()} | {bite['Header']['timestamp'][:10]}\")\n print(f\" \u251c\u2500 Semantic: {score_breakdown['semantic']:.3f}\")\n print(f\" \u251c\u2500 Spatial: {score_breakdown['spatial']:.3f}\")\n print(f\" \u251c\u2500 Temporal: {score_breakdown['temporal']:.3f}\")\n print(f\" \u2514\u2500 COMBINED: {score_breakdown['combined']:.3f}\")\n \n # Show snippet of body\n body_str = json.dumps(bite['Body'], indent=2)[:150]\n print(f\" \ud83d\udcc4 {body_str.replace(chr(10), ' ')[:80]}...\")\n \n # LLM Answer\n print(f\"\\n\ud83d\udca1 ANSWER:\")\n print(\" \" + \"\u2500\"*96)\n for line in answer.split('\\n'):\n if line.strip():\n print(f\" {line}\")\n print(\" \" + \"\u2500\"*96)\n\n\ndef ask_pancake_enhanced(query: str, days_back: int = 30, top_k: int = 5):\n \"\"\"\n Enhanced conversational AI with reasoning chain and timing\n \"\"\"\n import time\n \n timing = {}\n retrieval_start = time.time()\n \n # Step 1: RAG retrieval\n # Convert days_back to time_filter for rag_query\n from datetime import datetime, timedelta\n cutoff_time = (datetime.utcnow() - timedelta(days=days_back)).isoformat() + 'Z'\n time_filter = f\">= '{cutoff_time}'\"\n \n results = rag_query(query, top_k=top_k, time_filter=time_filter)\n \n timing['retrieval'] = time.time() - retrieval_start\n \n if not results:\n return \"No relevant data found.\", timing, [], []\n \n # Extract top BITEs and score breakdowns\n top_bites = results # rag_query returns list of bite dicts\n score_breakdowns = []\n \n for bite in results:\n combined_score = bite.get('semantic_distance', 0.0)\n # Recompute individual scores for display\n query_emb = get_embedding(query)\n bite_text = f\"{bite['Header']['type']}: {json.dumps(bite['Body'])}\"\n bite_emb = get_embedding(bite_text)\n \n sem_sim = semantic_similarity(query_emb, bite_emb) if query_emb and bite_emb else 0.0\n spat_sim = spatial_similarity(bite['Header']['geoid'], bite['Header']['geoid'])\n temp_sim = temporal_similarity(bite['Header']['timestamp'])\n \n score_breakdowns.append({\n 'semantic': sem_sim,\n 'spatial': spat_sim,\n 'temporal': temp_sim,\n 'combined': combined_score\n })\n \n # Step 2: Build context for LLM\n context = \"Here is the relevant PANCAKE data:\\n\\n\"\n for i, (bite, score) in enumerate(results, 1):\n context += f\"{i}. {bite['Header']['type']} ({bite['Header']['timestamp'][:10]}):\\n\"\n context += f\"{json.dumps(bite['Body'], indent=2)}\\n\\n\"\n \n # Step 3: LLM generation with timing\n llm_start = time.time()\n \n messages = [\n {\"role\": \"system\", \"content\": \"You are an agricultural AI assistant. Answer questions based on the provided PANCAKE data (BITEs). Be specific, actionable, and reference the data.\"},\n {\"role\": \"user\", \"content\": f\"Question: {query}\\n\\n{context}\"}\n ]\n \n response = client.chat.completions.create(\n model=\"gpt-4\",\n messages=messages,\n temperature=0.7,\n max_tokens=500\n )\n \n timing['llm'] = time.time() - llm_start\n timing['total'] = time.time() - retrieval_start\n \n # Token usage and cost\n timing['tokens'] = {\n 'prompt': response.usage.prompt_tokens,\n 'completion': response.usage.completion_tokens,\n 'total': response.usage.total_tokens\n }\n \n # GPT-4 pricing (as of 2024): $0.03/1K input, $0.06/1K output\n timing['cost'] = (timing['tokens']['prompt'] / 1000 * 0.03) + \\\n (timing['tokens']['completion'] / 1000 * 0.06)\n \n answer = response.choices[0].message.content\n \n return answer, timing, top_bites, score_breakdowns\n\n\nprint(\"\u2713 Enhanced conversational AI functions defined\")\n"
42484248
]
42494249
},
42504250
{

0 commit comments

Comments
 (0)