Axcel1/icd_10_coding_assistant
1
1from qdrant_client import QdrantClient2from qdrant_client.models import VectorParams, Distance, PointStruct3import numpy as np4from typing import List, Dict, Optional, Tuple, Set5from collections import Counter, defaultdict6from sentence_transformers import SentenceTransformer7from concurrent.futures import ThreadPoolExecutor, as_completed8import time9import re10import pprint11import os12from dotenv import load_dotenv13 14# Load environment variables15load_dotenv()16 17 18class MultiCollectionChapterRetrieval:19 def __init__(self, use_cloud: bool = True):20 """21 Initialize with Qdrant Cloud or local connection22 23 Args:24 use_cloud: If True, connects to Qdrant Cloud using environment variables25 """26 if use_cloud:27 self.client = self._create_cloud_client()28 else:29 self.client = QdrantClient("http://localhost:6333")30 31 self.encoder = None32 33 # ICD-10 Chapter mapping (all 22 chapters)34 self.chapter_info = {35 "chapter_1_I": "Certain infectious and parasitic diseases",36 "chapter_2_II": "Neoplasms", 37 "chapter_3_III": "Diseases of the blood and blood-forming organs and certain disorders involving the immune mechanism",38 "chapter_4_IV": "Endocrine, nutritional and metabolic diseases",39 "chapter_5_V": "Mental and behavioural disorders",40 "chapter_6_VI": "Diseases of the nervous system",41 "chapter_7_VII": "Diseases of the eye and adnexa",42 "chapter_8_VIII": "Diseases of the ear and mastoid process",43 "chapter_9_IX": "Diseases of the circulatory system",44 "chapter_10_X": "Diseases of the respiratory system",45 "chapter_11_XI": "Diseases of the digestive system",46 "chapter_12_XII": "Diseases of the skin and subcutaneous tissue",47 "chapter_13_XIII": "Diseases of the musculoskeletal system and connective tissue",48 "chapter_14_XIV": "Diseases of the genitourinary system",49 "chapter_15_XV": "Pregnancy, childbirth and the puerperium",50 "chapter_16_XVI": "Certain conditions originating in the perinatal period",51 "chapter_17_XVII": "Congenital malformations, deformations and chromosomal abnormalities",52 "chapter_18_XVIII": "Symptoms, signs and abnormal clinical and laboratory findings, not elsewhere classified",53 "chapter_19_XIX": "Injury, poisoning and certain other consequences of external causes",54 "chapter_20_XX": "External causes of morbidity and mortality",55 "chapter_21_XXI": "Factors influencing health status and contact with health services",56 "chapter_22_XXII": "Codes for special purposes"57 }58 59 # Cache for collection names60 self._chapter_collections = None61 62 def _create_cloud_client(self) -> QdrantClient:63 """Create Qdrant Cloud client with authentication"""64 qdrant_url = os.getenv('QDRANT_URL')65 qdrant_api_key = os.getenv('QDRANT_API_KEY')66 67 if not qdrant_url or not qdrant_api_key:68 raise ValueError(69 "Qdrant Cloud credentials not found in environment variables.\n"70 "Please set QDRANT_URL and QDRANT_API_KEY in your .env file:\n"71 "QDRANT_URL=https://your-cluster-id.region.aws.cloud.qdrant.io:6333\n"72 "QDRANT_API_KEY=your-api-key-here"73 )74 75 print(f"๐ Connecting to Qdrant Cloud: {qdrant_url}")76 77 try:78 client = QdrantClient(79 url=qdrant_url,80 api_key=qdrant_api_key,81 timeout=60, # Increased timeout for cloud82 # Optional: Add additional cloud-specific settings83 prefer_grpc=True, # Use gRPC for better performance84 )85 86 # Test connection87 collections = client.get_collections()88 print(f"โ
Connected successfully! Found {len(collections.collections)} collections")89 90 91 return client92 93 except Exception as e:94 print(f"โ Failed to connect to Qdrant Cloud: {e}")95 print("Please check your QDRANT_URL and QDRANT_API_KEY in the .env file")96 raise97 98 def split_into_sentences(self, text: str) -> List[str]:99 """Split text into sentences using simple rules"""100 import re101 102 # Simple sentence splitting - you can enhance this with nltk or spacy if needed103 sentences = re.split(r'[.!?]+', text)104 sentences = [s.strip() for s in sentences if s.strip()]105 return sentences106 107 def load_encoder(self, model_name: str = "all-MiniLM-L6-v2"):108 """Load the sentence transformer model"""109 if self.encoder is None:110 print(f"๐ฅ Loading encoder: {model_name}")111 self.encoder = SentenceTransformer(model_name)112 print(f"โ
Encoder loaded successfully")113 114 def encode_query(self, query: str) -> List[float]:115 """Encode diagnostic string to vector"""116 if self.encoder is None:117 self.load_encoder()118 return self.encoder.encode([query])[0].tolist()119 120 def get_chapter_collections(self) -> Dict[str, str]:121 """122 Get mapping of chapter_id -> collection_name123 Discovers collections automatically based on naming patterns124 """125 if self._chapter_collections is not None:126 return self._chapter_collections127 128 try:129 collections = self.client.get_collections()130 chapter_collections = {}131 132 print("๐ Discovering chapter collections...")133 134 for collection in collections.collections:135 collection_name = collection.name136 137 # Try to match collection names to chapters138 chapter_match = None139 140 # Pattern 1: icd10_chapter_X_Y or chapter_X_Y141 pattern1 = re.search(r'chapter[_-]?(\d+)[_-]?([IVX]+)', collection_name, re.IGNORECASE)142 if pattern1:143 chapter_num = pattern1.group(1)144 roman = pattern1.group(2)145 chapter_match = f"chapter_{chapter_num}_{roman}"146 147 # Pattern 2: Single collection with all chapters (e.g., icd10_codes_all_chapters)148 elif 'all' in collection_name.lower() and ('chapter' in collection_name.lower() or 'icd' in collection_name.lower()):149 print(f" ๐ Found unified collection: {collection_name}")150 # For unified collections, we'll handle this differently151 chapter_collections['unified_collection'] = collection_name152 continue153 154 # Pattern 3: Just the chapter part (chapter1, chapterI, etc.)155 elif 'chapter' in collection_name.lower():156 numbers = re.findall(r'\d+', collection_name)157 romans = re.findall(r'[IVX]+', collection_name)158 159 if numbers and romans:160 chapter_match = f"chapter_{numbers[0]}_{romans[0]}"161 elif numbers:162 # Try to convert number to roman numeral163 num = int(numbers[0])164 roman_map = {1: 'I', 2: 'II', 3: 'III', 4: 'IV', 5: 'V', 6: 'VI', 7: 'VII', 165 8: 'VIII', 9: 'IX', 10: 'X', 11: 'XI', 12: 'XII', 13: 'XIII', 166 14: 'XIV', 15: 'XV', 16: 'XVI', 17: 'XVII', 18: 'XVIII', 19: 'XIX', 167 20: 'XX', 21: 'XXI', 22: 'XXII'}168 if num in roman_map:169 chapter_match = f"chapter_{num}_{roman_map[num]}"170 171 if chapter_match:172 chapter_collections[chapter_match] = collection_name173 print(f" โ {chapter_match} -> {collection_name}")174 175 print(f"๐ Found {len(chapter_collections)} chapter collections")176 177 # If we only found a unified collection, we'll need to handle searches differently178 if len(chapter_collections) == 1 and 'unified_collection' in chapter_collections:179 print("โ ๏ธ Only unified collection found. Searches will use chapter filtering.")180 181 self._chapter_collections = chapter_collections182 return chapter_collections183 184 except Exception as e:185 print(f"โ Error discovering collections: {e}")186 return {}187 188 def search_single_collection(189 self, 190 collection_name: str, 191 query_vector: List[float], 192 limit: int = 20,193 score_threshold: float = 0.3,194 chapter_filter: Optional[str] = None195 ) -> List[Dict]:196 """Search a single collection and return formatted results"""197 try:198 # Build search parameters199 search_params = {200 "collection_name": collection_name,201 "query_vector": query_vector,202 "limit": limit,203 "score_threshold": score_threshold204 }205 206 results = self.client.search(**search_params)207 208 formatted_results = []209 for result in results:210 formatted_results.append({211 'collection': collection_name,212 'score': result.score,213 'id': result.id,214 'payload': result.payload215 })216 217 return formatted_results218 219 except Exception as e:220 print(f"โ Error searching {collection_name}: {e}")221 if "timeout" in str(e).lower():222 print(" This might be due to network issues. Retrying with lower limit...")223 try:224 # Retry with reduced parameters225 search_params["limit"] = min(limit, 10)226 search_params["score_threshold"] = max(score_threshold, 0.5)227 results = self.client.search(**search_params)228 229 formatted_results = []230 for result in results:231 formatted_results.append({232 'collection': collection_name,233 'score': result.score,234 'id': result.id,235 'payload': result.payload236 })237 return formatted_results238 except:239 pass240 return []241 242 def analyze_chapters_parallel(243 self, 244 diagnostic_string: str,245 sample_size_per_chapter: int = 15,246 score_threshold: float = 0.3,247 max_workers: int = 4 # Reduced for cloud stability248 ) -> Dict[str, Dict]:249 """250 Analyze all chapter collections in parallel to determine relevance251 Optimized for cloud performance252 """253 query_vector = self.encode_query(diagnostic_string)254 chapter_collections = self.get_chapter_collections()255 256 if not chapter_collections:257 print("โ No chapter collections found!")258 return {}259 260 print(f"\n๐ Analyzing diagnostic: '{diagnostic_string}'")261 262 # Handle unified collection differently263 # if 'unified_collection' in chapter_collections:264 # return self._analyze_unified_collection(265 # diagnostic_string, query_vector, 266 # chapter_collections['unified_collection'],267 # sample_size_per_chapter, score_threshold268 # )269 270 print(f"๐ Searching {len(chapter_collections)} collections in parallel...")271 272 chapter_analysis = {}273 274 def search_chapter(chapter_id: str, collection_name: str) -> Tuple[str, List[Dict]]:275 """Search function for parallel execution with retry logic"""276 max_retries = 2277 for attempt in range(max_retries):278 try:279 results = self.search_single_collection(280 collection_name, query_vector, sample_size_per_chapter, score_threshold281 )282 return chapter_id, results283 except Exception as e:284 if attempt < max_retries - 1:285 print(f" โ ๏ธ Retry {attempt + 1} for {chapter_id}: {e}")286 time.sleep(1) # Brief delay before retry287 else:288 print(f" โ Failed {chapter_id} after {max_retries} attempts: {e}")289 return chapter_id, []290 291 # Execute searches in parallel292 start_time = time.time()293 294 with ThreadPoolExecutor(max_workers=max_workers) as executor:295 # Submit all search tasks296 future_to_chapter = {297 executor.submit(search_chapter, chapter_id, collection_name): chapter_id298 for chapter_id, collection_name in chapter_collections.items()299 if chapter_id != 'unified_collection'300 }301 302 # Collect results as they complete303 for future in as_completed(future_to_chapter):304 chapter_id = future_to_chapter[future]305 try:306 chapter_id, results = future.result(timeout=30) # 30 second timeout per search307 308 if results:309 scores = [r['score'] for r in results]310 311 # Calculate chapter statistics312 chapter_analysis[chapter_id] = {313 'collection_name': chapter_collections[chapter_id],314 'match_count': len(results),315 'max_score': max(scores),316 'avg_score': np.mean(scores),317 'median_score': np.median(scores),318 'min_score': min(scores),319 'score_std': np.std(scores),320 'top_matches': sorted(results, key=lambda x: x['score'], reverse=True)[:5],321 'all_results': results322 }323 324 # Calculate relevance score (weighted combination of metrics)325 relevance = (326 chapter_analysis[chapter_id]['avg_score'] * 0.4 +327 chapter_analysis[chapter_id]['max_score'] * 0.3 +328 min(len(results) / sample_size_per_chapter, 1.0) * 0.2 +329 (1.0 / (1.0 + chapter_analysis[chapter_id]['score_std'])) * 0.1330 )331 332 chapter_analysis[chapter_id]['relevance_score'] = relevance333 334 # print(f" โ
{chapter_id}: {len(results)} matches, relevance: {relevance:.4f}")335 # else:336 # print(f" โ {chapter_id}: No matches above threshold")337 338 except Exception as e:339 print(f" โ {chapter_id}: Error - {e}")340 341 elapsed = time.time() - start_time342 print(f"โฑ๏ธ Parallel analysis completed in {elapsed:.2f} seconds")343 344 # Sort by relevance score345 sorted_analysis = dict(sorted(346 chapter_analysis.items(), 347 key=lambda x: x[1]['relevance_score'], 348 reverse=True349 ))350 351 return sorted_analysis352 353 def _analyze_unified_collection(354 self,355 diagnostic_string: str,356 query_vector: List[float],357 collection_name: str,358 sample_size_per_chapter: int,359 score_threshold: float360 ) -> Dict[str, Dict]:361 """Analyze unified collection by searching with chapter filters"""362 print(f"๐ Analyzing unified collection: {collection_name}")363 364 chapter_analysis = {}365 366 # Search each chapter in the unified collection367 for chapter_id in self.chapter_info.keys():368 try:369 results = self.search_single_collection(370 collection_name, query_vector, sample_size_per_chapter, 371 score_threshold, chapter_filter=chapter_id372 )373 374 if results:375 scores = [r['score'] for r in results]376 377 chapter_analysis[chapter_id] = {378 'collection_name': collection_name,379 'match_count': len(results),380 'max_score': max(scores),381 'avg_score': np.mean(scores),382 'median_score': np.median(scores),383 'min_score': min(scores),384 'score_std': np.std(scores),385 'top_matches': sorted(results, key=lambda x: x['score'], reverse=True)[:5],386 'all_results': results387 }388 389 # Calculate relevance score390 relevance = (391 chapter_analysis[chapter_id]['avg_score'] * 0.4 +392 chapter_analysis[chapter_id]['max_score'] * 0.3 +393 min(len(results) / sample_size_per_chapter, 1.0) * 0.2 +394 (1.0 / (1.0 + chapter_analysis[chapter_id]['score_std'])) * 0.1395 )396 397 chapter_analysis[chapter_id]['relevance_score'] = relevance398 print(f" โ
{chapter_id}: {len(results)} matches, relevance: {relevance:.4f}")399 else:400 print(f" โ {chapter_id}: No matches above threshold")401 402 # Small delay to avoid overwhelming the cloud service403 time.sleep(0.1)404 405 except Exception as e:406 print(f" โ {chapter_id}: Error - {e}")407 408 # Sort by relevance score409 return dict(sorted(410 chapter_analysis.items(), 411 key=lambda x: x[1]['relevance_score'], 412 reverse=True413 ))414 415 def get_top_chapters(416 self, 417 diagnostic_string: str, 418 top_n: int = 5,419 min_relevance: float = 0.1420 ) -> List[Tuple[str, float, str]]:421 """422 Get top N most relevant chapters for a diagnostic string423 Returns: [(chapter_id, relevance_score, description)]424 """425 analysis = self.analyze_chapters_parallel(diagnostic_string)426 427 top_chapters = []428 for chapter_id, stats in analysis.items():429 relevance = stats['relevance_score']430 431 if relevance >= min_relevance and len(top_chapters) < top_n:432 description = self.chapter_info.get(chapter_id, "Unknown chapter")433 top_chapters.append((chapter_id, relevance, description))434 435 return top_chapters436 437 def search_targeted_chapters(438 self, 439 diagnostic_string: str,440 target_chapters: List[str] = None,441 results_per_chapter: int = 10, # Keep for backward compatibility442 results_per_sentence: int = 3,443 chapters_per_sentence: int = 2 # New parameter: how many top chapters to search per sentence444 ) -> Dict[str, Dict[str, List[Dict]]]:445 """446 Search only specific chapters or auto-identify top chapters for each sentence individually.447 Now searches only the most relevant chapters for each specific sentence.448 """449 print(f"\n=== STARTING search_targeted_chapters ===")450 print(f"Input parameters:")451 print(f" diagnostic_string: '{diagnostic_string[:100]}{'...' if len(diagnostic_string) > 100 else ''}'")452 print(f" target_chapters: {target_chapters}")453 print(f" results_per_sentence: {results_per_sentence}")454 print(f" chapters_per_sentence: {chapters_per_sentence}")455 456 # Split input into sentences first457 print(f"\n--- SENTENCE SPLITTING ---")458 sentences = self.split_into_sentences(diagnostic_string)459 print(f"Split into {len(sentences)} sentences:")460 for i, sentence in enumerate(sentences):461 print(f" [{i+1}]: '{sentence}'")462 463 print(f"\n--- GETTING CHAPTER COLLECTIONS ---")464 chapter_collections = self.get_chapter_collections()465 print(f"Available chapter collections: {len(chapter_collections)} total")466 print(f"Chapter IDs: {list(chapter_collections.keys())}")467 468 results = {}469 470 if target_chapters is None:471 print(f"\n=== AUTO-IDENTIFICATION MODE ===")472 print("Auto-identifying most relevant chapters for each sentence individually...")473 474 for i, sentence in enumerate(sentences):475 if sentence.strip(): # Skip empty sentences476 sentence_key = f"sentence_{i+1}"477 print(f"\n--- Processing sentence {i+1} ---")478 print(f"Sentence: '{sentence}'")479 print(f"Sentence key: {sentence_key}")480 481 # Get top chapters specifically for THIS sentence482 print(f"Getting top {chapters_per_sentence} chapters for this sentence...")483 try:484 sentence_top_chapters = self.get_top_chapters(485 sentence, 486 top_n=chapters_per_sentence, 487 min_relevance=0.05488 )489 print(f"Found {len(sentence_top_chapters)} relevant chapters:")490 for j, (ch_id, rel, desc) in enumerate(sentence_top_chapters):491 print(f" [{j+1}] {ch_id}: {rel:.4f} - {desc}")492 except Exception as e:493 print(f"ERROR in get_top_chapters: {e}")494 sentence_top_chapters = []495 496 # Search only the relevant chapters for this specific sentence497 print(f"Searching in {len(sentence_top_chapters)} selected chapters...")498 for chapter_id, relevance, description in sentence_top_chapters:499 print(f"\n >> Searching chapter: {chapter_id} (relevance: {relevance:.4f})")500 501 if chapter_id in chapter_collections:502 collection_name = chapter_collections[chapter_id]503 print(f" Collection name: {collection_name}")504 505 # Initialize chapter in results if not exists506 if chapter_id not in results:507 results[chapter_id] = {}508 print(f" Initialized results dict for chapter {chapter_id}")509 510 # Search this sentence in this specific chapter511 try:512 print(f" Encoding query for sentence...")513 query_vector = self.encode_query(sentence)514 print(f" Query vector shape: {getattr(query_vector, 'shape', 'N/A')}")515 516 print(f" Searching collection '{collection_name}' for top {results_per_sentence} results...")517 sentence_results = self.search_single_collection(518 collection_name, query_vector, results_per_sentence519 )520 print(f" Raw search returned {len(sentence_results) if sentence_results else 0} results")521 522 except Exception as e:523 print(f" ERROR during search: {e}")524 sentence_results = []525 526 if sentence_results:527 results[chapter_id][sentence_key] = {528 'text': sentence,529 'chapter_relevance': relevance,530 'results': sentence_results531 }532 print(f" โ Stored {len(sentence_results)} results for {chapter_id}[{sentence_key}]")533 534 # Debug: show top result scores535 if sentence_results:536 top_scores = [r.get('score', 'N/A') for r in sentence_results[:3]]537 print(f" Top 3 scores: {top_scores}")538 else:539 print(f" โ No results above threshold for {chapter_id}")540 else:541 print(f" ERROR: Chapter {chapter_id} collection not found in available collections")542 else:543 print(f"\n--- Skipping empty sentence {i+1} ---")544 545 else:546 print(f"\n=== PRE-SPECIFIED CHAPTERS MODE ===")547 print(f"Using pre-specified chapters: {target_chapters}")548 549 # Validate chapters exist550 valid_chapters = []551 invalid_chapters = []552 for chapter_id in target_chapters:553 if chapter_id in chapter_collections:554 valid_chapters.append(chapter_id)555 else:556 invalid_chapters.append(chapter_id)557 558 print(f"Valid chapters: {valid_chapters}")559 if invalid_chapters:560 print(f"WARNING: Invalid chapters (will be skipped): {invalid_chapters}")561 562 for chapter_id in valid_chapters:563 collection_name = chapter_collections[chapter_id]564 print(f"\n--- Searching chapter: {chapter_id} ---")565 print(f"Collection name: {collection_name}")566 567 chapter_results = {}568 569 # Search each sentence in this chapter570 for i, sentence in enumerate(sentences):571 if sentence.strip(): # Skip empty sentences572 sentence_key = f"sentence_{i+1}"573 print(f"\n >> Processing sentence {i+1} in {chapter_id}")574 print(f" Sentence: '{sentence}'")575 576 try:577 print(f" Encoding query...")578 query_vector = self.encode_query(sentence)579 print(f" Query vector shape: {getattr(query_vector, 'shape', 'N/A')}")580 581 print(f" Searching for top {results_per_sentence} results...")582 sentence_results = self.search_single_collection(583 collection_name, query_vector, results_per_sentence584 )585 print(f" Found {len(sentence_results) if sentence_results else 0} results")586 587 except Exception as e:588 print(f" ERROR during search: {e}")589 sentence_results = []590 591 if sentence_results:592 chapter_results[sentence_key] = {593 'text': sentence,594 'chapter_relevance': None, # Not calculated for pre-specified chapters595 'results': sentence_results596 }597 print(f" โ Stored results for sentence {i+1}")598 599 # Debug: show top result scores600 top_scores = [r.get('score', 'N/A') for r in sentence_results[:3]]601 print(f" Top 3 scores: {top_scores}")602 else:603 print(f" โ No results found for sentence {i+1}")604 else:605 print(f" >> Skipping empty sentence {i+1}")606 607 if chapter_results:608 results[chapter_id] = chapter_results609 print(f"\n โ Chapter {chapter_id}: Stored results for {len(chapter_results)} sentences")610 else:611 print(f"\n โ Chapter {chapter_id}: No results found")612 613 # Final summary614 print(f"\n=== SEARCH COMPLETE ===")615 print(f"Results summary:")616 total_results = 0617 for chapter_id, chapter_data in results.items():618 sentence_count = len(chapter_data)619 result_count = sum(len(sent_data.get('results', [])) for sent_data in chapter_data.values())620 total_results += result_count621 print(f" {chapter_id}: {sentence_count} sentences, {result_count} total results")622 623 print(f"Grand total: {len(results)} chapters, {total_results} results")624 print(f"=== END search_targeted_chapters ===\n")625 626 return results627 628 def format_chapter_analysis(self, diagnostic_string: str, detailed: bool = True) -> str:629 """Format comprehensive chapter analysis"""630 analysis = self.analyze_chapters_parallel(diagnostic_string)631 632 if not analysis:633 return "โ No relevant chapters found."634 635 output = []636 output.append(f"\n{'='*90}")637 output.append(f"๐ CHAPTER RELEVANCE ANALYSIS")638 output.append(f"๐ Diagnostic: '{diagnostic_string}'")639 output.append(f"{'='*90}")640 641 for i, (chapter_id, stats) in enumerate(analysis.items(), 1):642 if stats['relevance_score'] < 0.05: # Skip very low relevance643 continue644 645 description = self.chapter_info.get(chapter_id, "Unknown chapter")646 647 output.append(f"\n{i}. ๐ {chapter_id.upper()}")648 output.append(f" ๐ท๏ธ Collection: {stats['collection_name']}")649 output.append(f" ๐ Description: {description}")650 output.append(f" โญ Relevance Score: {stats['relevance_score']:.4f}")651 output.append(f" ๐ Statistics:")652 output.append(f" โข Matches: {stats['match_count']}")653 output.append(f" โข Max Score: {stats['max_score']:.4f}")654 output.append(f" โข Avg Score: {stats['avg_score']:.4f}")655 output.append(f" โข Score Range: {stats['min_score']:.4f} - {stats['max_score']:.4f}")656 657 if detailed:658 output.append(f"\n ๐ฏ Top Matches:")659 for j, match in enumerate(stats['top_matches'][:3], 1):660 code = match['payload'].get('code', 'N/A')661 title = match['payload'].get('title', 'N/A')662 score = match['score']663 output.append(f" {j}. {code} - {title}")664 output.append(f" ๐ฏ Similarity: {score:.4f}")665 666 output.append("-" * 90)667 668 return "\n".join(output)669 670 671# Convenience functions for multi-collection setup672def analyze_diagnostic_chapters(diagnostic_string: str, detailed: bool = True, use_cloud: bool = True) -> str:673 """674 Main function to analyze which chapters are most relevant for a diagnostic675 """676 retriever = MultiCollectionChapterRetrieval(use_cloud=use_cloud)677 return retriever.format_chapter_analysis(diagnostic_string, detailed)678 679def get_relevant_chapters(diagnostic_string: str, top_n: int = 5, use_cloud: bool = True) -> List[str]:680 """681 Get list of most relevant chapter IDs for a diagnostic string682 Returns: ['chapter_9_IX', 'chapter_10_X', ...]683 """684 retriever = MultiCollectionChapterRetrieval(use_cloud=use_cloud)685 top_chapters = retriever.get_top_chapters(diagnostic_string, top_n)686 return [chapter_id for chapter_id, _, _ in top_chapters]687 688def smart_diagnostic_search(689 diagnostic_string: str, 690 auto_select_chapters: bool = True,691 target_chapters: List[str] = None,692 results_per_sentence: int = 3, # Updated parameter name693 use_cloud: bool = True694) -> Dict[str, Dict[str, List[Dict]]]: # Updated return type695 """696 Intelligent diagnostic search that processes each sentence separately697 Optimized for Qdrant Cloud698 """699 retriever = MultiCollectionChapterRetrieval(use_cloud=use_cloud)700 701 if auto_select_chapters:702 return retriever.search_targeted_chapters(703 diagnostic_string, target_chapters, results_per_sentence=results_per_sentence704 )705 else:706 return retriever.search_targeted_chapters(707 diagnostic_string, target_chapters, results_per_sentence=results_per_sentence708 )709 710def format_smart_search_results(711 diagnostic_string: str,712 search_results: Dict[str, Dict[str, List[Dict]]], # Updated parameter type713 use_cloud: bool = True714) -> str:715 """Format the results from sentence-based smart_diagnostic_search"""716 717 if not search_results:718 return "โ No results found."719 720 retriever = MultiCollectionChapterRetrieval(use_cloud=use_cloud)721 722 output = []723 output.append(f"\n{'='*90}")724 output.append(f"๐ SENTENCE-BASED DIAGNOSTIC SEARCH RESULTS")725 output.append(f"๐ฏ Query: '{diagnostic_string}'")726 output.append(f"{'='*90}")727 728 # Count total results729 total_results = 0730 total_sentences = 0731 for chapter_results in search_results.values():732 total_sentences += len(chapter_results)733 for sentence_data in chapter_results.values():734 total_results += len(sentence_data['results'])735 736 output.append(f"๐ Total results: {total_results} across {len(search_results)} chapters and {total_sentences} sentences")737 738 for chapter_id, chapter_data in search_results.items():739 description = retriever.chapter_info.get(chapter_id, "Unknown chapter")740 741 output.append(f"\n๐ {chapter_id.upper()}")742 output.append(f" ๐ {description}")743 output.append(f" ๐ {len(chapter_data)} sentences processed")744 output.append("-" * 60)745 746 for sentence_key, sentence_data in chapter_data.items():747 sentence_text = sentence_data['text']748 results = sentence_data['results']749 750 output.append(f"\n ๐ {sentence_key.replace('_', ' ').title()}: \"{sentence_text}\"")751 output.append(f" ๐ฏ Top {len(results)} matches:")752 output.append("")753 754 for i, result in enumerate(results, 1):755 payload = result['payload']756 code = payload.get('code', 'N/A')757 title = payload.get('title', 'N/A')758 score = result['score']759 760 output.append(f" {i}. {code} - {title}")761 output.append(f" ๐ฏ Score: {score:.4f}")762 763 # Show description if available764 desc = payload.get('description', '')765 if desc:766 desc_preview = desc[:100] + "..." if len(desc) > 100 else desc767 output.append(f" ๐ {desc_preview}")768 769 output.append("")770 771 output.append("=" * 90)772 773 return "\n".join(output)774 775# Example usage776def example_multi_collection_analysis(use_cloud: bool = True):777 """Example of using the multi-collection chapter analysis"""778 779 test_cases = [780 "severe chest pain with shortness of breath",781 "type 2 diabetes with kidney complications", 782 "depression and anxiety disorder",783 "broken wrist from falling",784 "acute appendicitis with fever",785 "skin cancer melanoma",786 "pregnancy complications in third trimester"787 ]788 789 for diagnostic in test_cases:790 print(f"\n{'='*100}")791 print(f"๐ ANALYZING: {diagnostic}")792 print(f"{'='*100}")793 794 try:795 # Step 1: Analyze chapter relevance796 analysis = analyze_diagnostic_chapters(diagnostic, detailed=False, use_cloud=use_cloud)797 print(analysis)798 799 # Step 2: Get top relevant chapters800 top_chapters = get_relevant_chapters(diagnostic, top_n=3, use_cloud=use_cloud)801 print(f"\n๐ Top 3 relevant chapters: {top_chapters}")802 803 # Step 3: Smart search in those chapters804 search_results = smart_diagnostic_search(805 diagnostic, 806 results_per_sentence=5, 807 use_cloud=use_cloud808 )809 formatted_results = format_smart_search_results(810 diagnostic, 811 search_results, 812 use_cloud=use_cloud813 )814 print(formatted_results)815 816 except Exception as e:817 print(f"โ Error processing '{diagnostic}': {e}")818 continue819 820def test_cloud_connection():821 """Test Qdrant Cloud connection and basic functionality"""822 print("๐งช Testing Qdrant Cloud Connection...")823 824 try:825 retriever = MultiCollectionChapterRetrieval(use_cloud=True)826 827 # Test basic search828 test_query = "heart disease"829 print(f"\n๐ฌ Testing with query: '{test_query}'")830 831 # Get collections832 collections = retriever.get_chapter_collections()833 print(f"๐ Available collections: {len(collections)}")834 835 if collections:836 # Test search837 top_chapters = retriever.get_top_chapters(test_query, top_n=3)838 print(f"๐ฏ Top chapters for '{test_query}': {[ch[0] for ch in top_chapters]}")839 840 print("โ
Cloud connection test successful!")841 return True842 else:843 print("โ ๏ธ No collections found")844 return False845 846 except Exception as e:847 print(f"โ Cloud connection test failed: {e}")848 return False849 850if __name__ == "__main__":851 # Test cloud connection first852 if test_cloud_connection():853 print("\n" + "="*100)854 print("๐ Running example analysis with Qdrant Cloud...")855 print("="*100)856 857 # Run examples with cloud858 example_multi_collection_analysis(use_cloud=True)859 else:860 print("โ Skipping examples due to connection issues")861 862 # Or use directly:863 # chapters = get_relevant_chapters("heart attack symptoms", use_cloud=True)864 # results = smart_diagnostic_search("heart attack symptoms", use_cloud=True) 865 # print(format_smart_search_results("heart attack symptoms", results, use_cloud=True))