Team Ai
Apppublic

Axcel1/icd_10_coding_assistant

sourceHugging Facemitupdated 1y agoView on Hugging Face
1likes
chapter_retrieval_system_v2.py865 linesDownload Raw Back to root
1from qdrant_client import QdrantClient2from qdrant_client.models import VectorParams, Distance, PointStruct3import numpy as np4from typing import List, Dict, Optional, Tuple, Set5from collections import Counter, defaultdict6from sentence_transformers import SentenceTransformer7from concurrent.futures import ThreadPoolExecutor, as_completed8import time9import re10import pprint11import os12from dotenv import load_dotenv13 14# Load environment variables15load_dotenv()16 17 18class MultiCollectionChapterRetrieval:19    def __init__(self, use_cloud: bool = True):20        """21        Initialize with Qdrant Cloud or local connection22        23        Args:24            use_cloud: If True, connects to Qdrant Cloud using environment variables25        """26        if use_cloud:27            self.client = self._create_cloud_client()28        else:29            self.client = QdrantClient("http://localhost:6333")30        31        self.encoder = None32        33        # ICD-10 Chapter mapping (all 22 chapters)34        self.chapter_info = {35            "chapter_1_I": "Certain infectious and parasitic diseases",36            "chapter_2_II": "Neoplasms", 37            "chapter_3_III": "Diseases of the blood and blood-forming organs and certain disorders involving the immune mechanism",38            "chapter_4_IV": "Endocrine, nutritional and metabolic diseases",39            "chapter_5_V": "Mental and behavioural disorders",40            "chapter_6_VI": "Diseases of the nervous system",41            "chapter_7_VII": "Diseases of the eye and adnexa",42            "chapter_8_VIII": "Diseases of the ear and mastoid process",43            "chapter_9_IX": "Diseases of the circulatory system",44            "chapter_10_X": "Diseases of the respiratory system",45            "chapter_11_XI": "Diseases of the digestive system",46            "chapter_12_XII": "Diseases of the skin and subcutaneous tissue",47            "chapter_13_XIII": "Diseases of the musculoskeletal system and connective tissue",48            "chapter_14_XIV": "Diseases of the genitourinary system",49            "chapter_15_XV": "Pregnancy, childbirth and the puerperium",50            "chapter_16_XVI": "Certain conditions originating in the perinatal period",51            "chapter_17_XVII": "Congenital malformations, deformations and chromosomal abnormalities",52            "chapter_18_XVIII": "Symptoms, signs and abnormal clinical and laboratory findings, not elsewhere classified",53            "chapter_19_XIX": "Injury, poisoning and certain other consequences of external causes",54            "chapter_20_XX": "External causes of morbidity and mortality",55            "chapter_21_XXI": "Factors influencing health status and contact with health services",56            "chapter_22_XXII": "Codes for special purposes"57        }58        59        # Cache for collection names60        self._chapter_collections = None61    62    def _create_cloud_client(self) -> QdrantClient:63        """Create Qdrant Cloud client with authentication"""64        qdrant_url = os.getenv('QDRANT_URL')65        qdrant_api_key = os.getenv('QDRANT_API_KEY')66        67        if not qdrant_url or not qdrant_api_key:68            raise ValueError(69                "Qdrant Cloud credentials not found in environment variables.\n"70                "Please set QDRANT_URL and QDRANT_API_KEY in your .env file:\n"71                "QDRANT_URL=https://your-cluster-id.region.aws.cloud.qdrant.io:6333\n"72                "QDRANT_API_KEY=your-api-key-here"73            )74        75        print(f"๐Ÿ”— Connecting to Qdrant Cloud: {qdrant_url}")76        77        try:78            client = QdrantClient(79                url=qdrant_url,80                api_key=qdrant_api_key,81                timeout=60,  # Increased timeout for cloud82                # Optional: Add additional cloud-specific settings83                prefer_grpc=True,  # Use gRPC for better performance84            )85            86            # Test connection87            collections = client.get_collections()88            print(f"โœ… Connected successfully! Found {len(collections.collections)} collections")89            90            91            return client92            93        except Exception as e:94            print(f"โŒ Failed to connect to Qdrant Cloud: {e}")95            print("Please check your QDRANT_URL and QDRANT_API_KEY in the .env file")96            raise97 98    def split_into_sentences(self, text: str) -> List[str]:99        """Split text into sentences using simple rules"""100        import re101        102        # Simple sentence splitting - you can enhance this with nltk or spacy if needed103        sentences = re.split(r'[.!?]+', text)104        sentences = [s.strip() for s in sentences if s.strip()]105        return sentences106        107    def load_encoder(self, model_name: str = "all-MiniLM-L6-v2"):108        """Load the sentence transformer model"""109        if self.encoder is None:110            print(f"๐Ÿ“ฅ Loading encoder: {model_name}")111            self.encoder = SentenceTransformer(model_name)112            print(f"โœ… Encoder loaded successfully")113    114    def encode_query(self, query: str) -> List[float]:115        """Encode diagnostic string to vector"""116        if self.encoder is None:117            self.load_encoder()118        return self.encoder.encode([query])[0].tolist()119    120    def get_chapter_collections(self) -> Dict[str, str]:121        """122        Get mapping of chapter_id -> collection_name123        Discovers collections automatically based on naming patterns124        """125        if self._chapter_collections is not None:126            return self._chapter_collections127        128        try:129            collections = self.client.get_collections()130            chapter_collections = {}131            132            print("๐Ÿ” Discovering chapter collections...")133            134            for collection in collections.collections:135                collection_name = collection.name136                137                # Try to match collection names to chapters138                chapter_match = None139                140                # Pattern 1: icd10_chapter_X_Y or chapter_X_Y141                pattern1 = re.search(r'chapter[_-]?(\d+)[_-]?([IVX]+)', collection_name, re.IGNORECASE)142                if pattern1:143                    chapter_num = pattern1.group(1)144                    roman = pattern1.group(2)145                    chapter_match = f"chapter_{chapter_num}_{roman}"146                147                # Pattern 2: Single collection with all chapters (e.g., icd10_codes_all_chapters)148                elif 'all' in collection_name.lower() and ('chapter' in collection_name.lower() or 'icd' in collection_name.lower()):149                    print(f"  ๐Ÿ“š Found unified collection: {collection_name}")150                    # For unified collections, we'll handle this differently151                    chapter_collections['unified_collection'] = collection_name152                    continue153                154                # Pattern 3: Just the chapter part (chapter1, chapterI, etc.)155                elif 'chapter' in collection_name.lower():156                    numbers = re.findall(r'\d+', collection_name)157                    romans = re.findall(r'[IVX]+', collection_name)158                    159                    if numbers and romans:160                        chapter_match = f"chapter_{numbers[0]}_{romans[0]}"161                    elif numbers:162                        # Try to convert number to roman numeral163                        num = int(numbers[0])164                        roman_map = {1: 'I', 2: 'II', 3: 'III', 4: 'IV', 5: 'V', 6: 'VI', 7: 'VII', 165                                   8: 'VIII', 9: 'IX', 10: 'X', 11: 'XI', 12: 'XII', 13: 'XIII', 166                                   14: 'XIV', 15: 'XV', 16: 'XVI', 17: 'XVII', 18: 'XVIII', 19: 'XIX', 167                                   20: 'XX', 21: 'XXI', 22: 'XXII'}168                        if num in roman_map:169                            chapter_match = f"chapter_{num}_{roman_map[num]}"170                171                if chapter_match:172                    chapter_collections[chapter_match] = collection_name173                    print(f"  โœ“ {chapter_match} -> {collection_name}")174            175            print(f"๐Ÿ“Š Found {len(chapter_collections)} chapter collections")176            177            # If we only found a unified collection, we'll need to handle searches differently178            if len(chapter_collections) == 1 and 'unified_collection' in chapter_collections:179                print("โš ๏ธ  Only unified collection found. Searches will use chapter filtering.")180            181            self._chapter_collections = chapter_collections182            return chapter_collections183            184        except Exception as e:185            print(f"โŒ Error discovering collections: {e}")186            return {}187    188    def search_single_collection(189        self, 190        collection_name: str, 191        query_vector: List[float], 192        limit: int = 20,193        score_threshold: float = 0.3,194        chapter_filter: Optional[str] = None195    ) -> List[Dict]:196        """Search a single collection and return formatted results"""197        try:198            # Build search parameters199            search_params = {200                "collection_name": collection_name,201                "query_vector": query_vector,202                "limit": limit,203                "score_threshold": score_threshold204            }205            206            results = self.client.search(**search_params)207            208            formatted_results = []209            for result in results:210                formatted_results.append({211                    'collection': collection_name,212                    'score': result.score,213                    'id': result.id,214                    'payload': result.payload215                })216            217            return formatted_results218            219        except Exception as e:220            print(f"โŒ Error searching {collection_name}: {e}")221            if "timeout" in str(e).lower():222                print("   This might be due to network issues. Retrying with lower limit...")223                try:224                    # Retry with reduced parameters225                    search_params["limit"] = min(limit, 10)226                    search_params["score_threshold"] = max(score_threshold, 0.5)227                    results = self.client.search(**search_params)228                    229                    formatted_results = []230                    for result in results:231                        formatted_results.append({232                            'collection': collection_name,233                            'score': result.score,234                            'id': result.id,235                            'payload': result.payload236                        })237                    return formatted_results238                except:239                    pass240            return []241    242    def analyze_chapters_parallel(243        self, 244        diagnostic_string: str,245        sample_size_per_chapter: int = 15,246        score_threshold: float = 0.3,247        max_workers: int = 4  # Reduced for cloud stability248    ) -> Dict[str, Dict]:249        """250        Analyze all chapter collections in parallel to determine relevance251        Optimized for cloud performance252        """253        query_vector = self.encode_query(diagnostic_string)254        chapter_collections = self.get_chapter_collections()255        256        if not chapter_collections:257            print("โŒ No chapter collections found!")258            return {}259 260        print(f"\n๐Ÿ” Analyzing diagnostic: '{diagnostic_string}'")261        262        # Handle unified collection differently263        # if 'unified_collection' in chapter_collections:264        #     return self._analyze_unified_collection(265        #         diagnostic_string, query_vector, 266        #         chapter_collections['unified_collection'],267        #         sample_size_per_chapter, score_threshold268        #     )269        270        print(f"๐Ÿ”„ Searching {len(chapter_collections)} collections in parallel...")271        272        chapter_analysis = {}273        274        def search_chapter(chapter_id: str, collection_name: str) -> Tuple[str, List[Dict]]:275            """Search function for parallel execution with retry logic"""276            max_retries = 2277            for attempt in range(max_retries):278                try:279                    results = self.search_single_collection(280                        collection_name, query_vector, sample_size_per_chapter, score_threshold281                    )282                    return chapter_id, results283                except Exception as e:284                    if attempt < max_retries - 1:285                        print(f"  โš ๏ธ Retry {attempt + 1} for {chapter_id}: {e}")286                        time.sleep(1)  # Brief delay before retry287                    else:288                        print(f"  โŒ Failed {chapter_id} after {max_retries} attempts: {e}")289                        return chapter_id, []290        291        # Execute searches in parallel292        start_time = time.time()293        294        with ThreadPoolExecutor(max_workers=max_workers) as executor:295            # Submit all search tasks296            future_to_chapter = {297                executor.submit(search_chapter, chapter_id, collection_name): chapter_id298                for chapter_id, collection_name in chapter_collections.items()299                if chapter_id != 'unified_collection'300            }301            302            # Collect results as they complete303            for future in as_completed(future_to_chapter):304                chapter_id = future_to_chapter[future]305                try:306                    chapter_id, results = future.result(timeout=30)  # 30 second timeout per search307                    308                    if results:309                        scores = [r['score'] for r in results]310                        311                        # Calculate chapter statistics312                        chapter_analysis[chapter_id] = {313                            'collection_name': chapter_collections[chapter_id],314                            'match_count': len(results),315                            'max_score': max(scores),316                            'avg_score': np.mean(scores),317                            'median_score': np.median(scores),318                            'min_score': min(scores),319                            'score_std': np.std(scores),320                            'top_matches': sorted(results, key=lambda x: x['score'], reverse=True)[:5],321                            'all_results': results322                        }323                        324                        # Calculate relevance score (weighted combination of metrics)325                        relevance = (326                            chapter_analysis[chapter_id]['avg_score'] * 0.4 +327                            chapter_analysis[chapter_id]['max_score'] * 0.3 +328                            min(len(results) / sample_size_per_chapter, 1.0) * 0.2 +329                            (1.0 / (1.0 + chapter_analysis[chapter_id]['score_std'])) * 0.1330                        )331                        332                        chapter_analysis[chapter_id]['relevance_score'] = relevance333                        334                        # print(f"  โœ… {chapter_id}: {len(results)} matches, relevance: {relevance:.4f}")335                    # else:336                        # print(f"  โž– {chapter_id}: No matches above threshold")337                        338                except Exception as e:339                    print(f"  โŒ {chapter_id}: Error - {e}")340        341        elapsed = time.time() - start_time342        print(f"โฑ๏ธ Parallel analysis completed in {elapsed:.2f} seconds")343        344        # Sort by relevance score345        sorted_analysis = dict(sorted(346            chapter_analysis.items(), 347            key=lambda x: x[1]['relevance_score'], 348            reverse=True349        ))350        351        return sorted_analysis352    353    def _analyze_unified_collection(354        self,355        diagnostic_string: str,356        query_vector: List[float],357        collection_name: str,358        sample_size_per_chapter: int,359        score_threshold: float360    ) -> Dict[str, Dict]:361        """Analyze unified collection by searching with chapter filters"""362        print(f"๐Ÿ”„ Analyzing unified collection: {collection_name}")363        364        chapter_analysis = {}365        366        # Search each chapter in the unified collection367        for chapter_id in self.chapter_info.keys():368            try:369                results = self.search_single_collection(370                    collection_name, query_vector, sample_size_per_chapter, 371                    score_threshold, chapter_filter=chapter_id372                )373                374                if results:375                    scores = [r['score'] for r in results]376                    377                    chapter_analysis[chapter_id] = {378                        'collection_name': collection_name,379                        'match_count': len(results),380                        'max_score': max(scores),381                        'avg_score': np.mean(scores),382                        'median_score': np.median(scores),383                        'min_score': min(scores),384                        'score_std': np.std(scores),385                        'top_matches': sorted(results, key=lambda x: x['score'], reverse=True)[:5],386                        'all_results': results387                    }388                    389                    # Calculate relevance score390                    relevance = (391                        chapter_analysis[chapter_id]['avg_score'] * 0.4 +392                        chapter_analysis[chapter_id]['max_score'] * 0.3 +393                        min(len(results) / sample_size_per_chapter, 1.0) * 0.2 +394                        (1.0 / (1.0 + chapter_analysis[chapter_id]['score_std'])) * 0.1395                    )396                    397                    chapter_analysis[chapter_id]['relevance_score'] = relevance398                    print(f"  โœ… {chapter_id}: {len(results)} matches, relevance: {relevance:.4f}")399                else:400                    print(f"  โž– {chapter_id}: No matches above threshold")401                    402                # Small delay to avoid overwhelming the cloud service403                time.sleep(0.1)404                    405            except Exception as e:406                print(f"  โŒ {chapter_id}: Error - {e}")407        408        # Sort by relevance score409        return dict(sorted(410            chapter_analysis.items(), 411            key=lambda x: x[1]['relevance_score'], 412            reverse=True413        ))414    415    def get_top_chapters(416        self, 417        diagnostic_string: str, 418        top_n: int = 5,419        min_relevance: float = 0.1420    ) -> List[Tuple[str, float, str]]:421        """422        Get top N most relevant chapters for a diagnostic string423        Returns: [(chapter_id, relevance_score, description)]424        """425        analysis = self.analyze_chapters_parallel(diagnostic_string)426        427        top_chapters = []428        for chapter_id, stats in analysis.items():429            relevance = stats['relevance_score']430            431            if relevance >= min_relevance and len(top_chapters) < top_n:432                description = self.chapter_info.get(chapter_id, "Unknown chapter")433                top_chapters.append((chapter_id, relevance, description))434        435        return top_chapters436    437    def search_targeted_chapters(438        self, 439        diagnostic_string: str,440        target_chapters: List[str] = None,441        results_per_chapter: int = 10,  # Keep for backward compatibility442        results_per_sentence: int = 3,443        chapters_per_sentence: int = 2  # New parameter: how many top chapters to search per sentence444    ) -> Dict[str, Dict[str, List[Dict]]]:445        """446        Search only specific chapters or auto-identify top chapters for each sentence individually.447        Now searches only the most relevant chapters for each specific sentence.448        """449        print(f"\n=== STARTING search_targeted_chapters ===")450        print(f"Input parameters:")451        print(f"  diagnostic_string: '{diagnostic_string[:100]}{'...' if len(diagnostic_string) > 100 else ''}'")452        print(f"  target_chapters: {target_chapters}")453        print(f"  results_per_sentence: {results_per_sentence}")454        print(f"  chapters_per_sentence: {chapters_per_sentence}")455        456        # Split input into sentences first457        print(f"\n--- SENTENCE SPLITTING ---")458        sentences = self.split_into_sentences(diagnostic_string)459        print(f"Split into {len(sentences)} sentences:")460        for i, sentence in enumerate(sentences):461            print(f"  [{i+1}]: '{sentence}'")462        463        print(f"\n--- GETTING CHAPTER COLLECTIONS ---")464        chapter_collections = self.get_chapter_collections()465        print(f"Available chapter collections: {len(chapter_collections)} total")466        print(f"Chapter IDs: {list(chapter_collections.keys())}")467        468        results = {}469        470        if target_chapters is None:471            print(f"\n=== AUTO-IDENTIFICATION MODE ===")472            print("Auto-identifying most relevant chapters for each sentence individually...")473            474            for i, sentence in enumerate(sentences):475                if sentence.strip():  # Skip empty sentences476                    sentence_key = f"sentence_{i+1}"477                    print(f"\n--- Processing sentence {i+1} ---")478                    print(f"Sentence: '{sentence}'")479                    print(f"Sentence key: {sentence_key}")480                    481                    # Get top chapters specifically for THIS sentence482                    print(f"Getting top {chapters_per_sentence} chapters for this sentence...")483                    try:484                        sentence_top_chapters = self.get_top_chapters(485                            sentence, 486                            top_n=chapters_per_sentence, 487                            min_relevance=0.05488                        )489                        print(f"Found {len(sentence_top_chapters)} relevant chapters:")490                        for j, (ch_id, rel, desc) in enumerate(sentence_top_chapters):491                            print(f"  [{j+1}] {ch_id}: {rel:.4f} - {desc}")492                    except Exception as e:493                        print(f"ERROR in get_top_chapters: {e}")494                        sentence_top_chapters = []495                    496                    # Search only the relevant chapters for this specific sentence497                    print(f"Searching in {len(sentence_top_chapters)} selected chapters...")498                    for chapter_id, relevance, description in sentence_top_chapters:499                        print(f"\n  >> Searching chapter: {chapter_id} (relevance: {relevance:.4f})")500                        501                        if chapter_id in chapter_collections:502                            collection_name = chapter_collections[chapter_id]503                            print(f"     Collection name: {collection_name}")504                            505                            # Initialize chapter in results if not exists506                            if chapter_id not in results:507                                results[chapter_id] = {}508                                print(f"     Initialized results dict for chapter {chapter_id}")509                            510                            # Search this sentence in this specific chapter511                            try:512                                print(f"     Encoding query for sentence...")513                                query_vector = self.encode_query(sentence)514                                print(f"     Query vector shape: {getattr(query_vector, 'shape', 'N/A')}")515                                516                                print(f"     Searching collection '{collection_name}' for top {results_per_sentence} results...")517                                sentence_results = self.search_single_collection(518                                    collection_name, query_vector, results_per_sentence519                                )520                                print(f"     Raw search returned {len(sentence_results) if sentence_results else 0} results")521                                522                            except Exception as e:523                                print(f"     ERROR during search: {e}")524                                sentence_results = []525                            526                            if sentence_results:527                                results[chapter_id][sentence_key] = {528                                    'text': sentence,529                                    'chapter_relevance': relevance,530                                    'results': sentence_results531                                }532                                print(f"     โœ“ Stored {len(sentence_results)} results for {chapter_id}[{sentence_key}]")533                                534                                # Debug: show top result scores535                                if sentence_results:536                                    top_scores = [r.get('score', 'N/A') for r in sentence_results[:3]]537                                    print(f"     Top 3 scores: {top_scores}")538                            else:539                                print(f"     โœ— No results above threshold for {chapter_id}")540                        else:541                            print(f"     ERROR: Chapter {chapter_id} collection not found in available collections")542                else:543                    print(f"\n--- Skipping empty sentence {i+1} ---")544        545        else:546            print(f"\n=== PRE-SPECIFIED CHAPTERS MODE ===")547            print(f"Using pre-specified chapters: {target_chapters}")548            549            # Validate chapters exist550            valid_chapters = []551            invalid_chapters = []552            for chapter_id in target_chapters:553                if chapter_id in chapter_collections:554                    valid_chapters.append(chapter_id)555                else:556                    invalid_chapters.append(chapter_id)557            558            print(f"Valid chapters: {valid_chapters}")559            if invalid_chapters:560                print(f"WARNING: Invalid chapters (will be skipped): {invalid_chapters}")561            562            for chapter_id in valid_chapters:563                collection_name = chapter_collections[chapter_id]564                print(f"\n--- Searching chapter: {chapter_id} ---")565                print(f"Collection name: {collection_name}")566                567                chapter_results = {}568                569                # Search each sentence in this chapter570                for i, sentence in enumerate(sentences):571                    if sentence.strip():  # Skip empty sentences572                        sentence_key = f"sentence_{i+1}"573                        print(f"\n  >> Processing sentence {i+1} in {chapter_id}")574                        print(f"     Sentence: '{sentence}'")575                        576                        try:577                            print(f"     Encoding query...")578                            query_vector = self.encode_query(sentence)579                            print(f"     Query vector shape: {getattr(query_vector, 'shape', 'N/A')}")580                            581                            print(f"     Searching for top {results_per_sentence} results...")582                            sentence_results = self.search_single_collection(583                                collection_name, query_vector, results_per_sentence584                            )585                            print(f"     Found {len(sentence_results) if sentence_results else 0} results")586                            587                        except Exception as e:588                            print(f"     ERROR during search: {e}")589                            sentence_results = []590                        591                        if sentence_results:592                            chapter_results[sentence_key] = {593                                'text': sentence,594                                'chapter_relevance': None,  # Not calculated for pre-specified chapters595                                'results': sentence_results596                            }597                            print(f"     โœ“ Stored results for sentence {i+1}")598                            599                            # Debug: show top result scores600                            top_scores = [r.get('score', 'N/A') for r in sentence_results[:3]]601                            print(f"     Top 3 scores: {top_scores}")602                        else:603                            print(f"     โœ— No results found for sentence {i+1}")604                    else:605                        print(f"  >> Skipping empty sentence {i+1}")606                607                if chapter_results:608                    results[chapter_id] = chapter_results609                    print(f"\n  โœ“ Chapter {chapter_id}: Stored results for {len(chapter_results)} sentences")610                else:611                    print(f"\n  โœ— Chapter {chapter_id}: No results found")612        613        # Final summary614        print(f"\n=== SEARCH COMPLETE ===")615        print(f"Results summary:")616        total_results = 0617        for chapter_id, chapter_data in results.items():618            sentence_count = len(chapter_data)619            result_count = sum(len(sent_data.get('results', [])) for sent_data in chapter_data.values())620            total_results += result_count621            print(f"  {chapter_id}: {sentence_count} sentences, {result_count} total results")622        623        print(f"Grand total: {len(results)} chapters, {total_results} results")624        print(f"=== END search_targeted_chapters ===\n")625        626        return results627    628    def format_chapter_analysis(self, diagnostic_string: str, detailed: bool = True) -> str:629        """Format comprehensive chapter analysis"""630        analysis = self.analyze_chapters_parallel(diagnostic_string)631        632        if not analysis:633            return "โŒ No relevant chapters found."634        635        output = []636        output.append(f"\n{'='*90}")637        output.append(f"๐Ÿ“Š CHAPTER RELEVANCE ANALYSIS")638        output.append(f"๐Ÿ” Diagnostic: '{diagnostic_string}'")639        output.append(f"{'='*90}")640        641        for i, (chapter_id, stats) in enumerate(analysis.items(), 1):642            if stats['relevance_score'] < 0.05:  # Skip very low relevance643                continue644                645            description = self.chapter_info.get(chapter_id, "Unknown chapter")646            647            output.append(f"\n{i}. ๐Ÿ“š {chapter_id.upper()}")648            output.append(f"   ๐Ÿท๏ธ  Collection: {stats['collection_name']}")649            output.append(f"   ๐Ÿ“– Description: {description}")650            output.append(f"   โญ Relevance Score: {stats['relevance_score']:.4f}")651            output.append(f"   ๐Ÿ“Š Statistics:")652            output.append(f"      โ€ข Matches: {stats['match_count']}")653            output.append(f"      โ€ข Max Score: {stats['max_score']:.4f}")654            output.append(f"      โ€ข Avg Score: {stats['avg_score']:.4f}")655            output.append(f"      โ€ข Score Range: {stats['min_score']:.4f} - {stats['max_score']:.4f}")656            657            if detailed:658                output.append(f"\n   ๐ŸŽฏ Top Matches:")659                for j, match in enumerate(stats['top_matches'][:3], 1):660                    code = match['payload'].get('code', 'N/A')661                    title = match['payload'].get('title', 'N/A')662                    score = match['score']663                    output.append(f"      {j}. {code} - {title}")664                    output.append(f"         ๐Ÿ’ฏ Similarity: {score:.4f}")665            666            output.append("-" * 90)667        668        return "\n".join(output)669 670 671# Convenience functions for multi-collection setup672def analyze_diagnostic_chapters(diagnostic_string: str, detailed: bool = True, use_cloud: bool = True) -> str:673    """674    Main function to analyze which chapters are most relevant for a diagnostic675    """676    retriever = MultiCollectionChapterRetrieval(use_cloud=use_cloud)677    return retriever.format_chapter_analysis(diagnostic_string, detailed)678 679def get_relevant_chapters(diagnostic_string: str, top_n: int = 5, use_cloud: bool = True) -> List[str]:680    """681    Get list of most relevant chapter IDs for a diagnostic string682    Returns: ['chapter_9_IX', 'chapter_10_X', ...]683    """684    retriever = MultiCollectionChapterRetrieval(use_cloud=use_cloud)685    top_chapters = retriever.get_top_chapters(diagnostic_string, top_n)686    return [chapter_id for chapter_id, _, _ in top_chapters]687 688def smart_diagnostic_search(689    diagnostic_string: str, 690    auto_select_chapters: bool = True,691    target_chapters: List[str] = None,692    results_per_sentence: int = 3,  # Updated parameter name693    use_cloud: bool = True694) -> Dict[str, Dict[str, List[Dict]]]:  # Updated return type695    """696    Intelligent diagnostic search that processes each sentence separately697    Optimized for Qdrant Cloud698    """699    retriever = MultiCollectionChapterRetrieval(use_cloud=use_cloud)700    701    if auto_select_chapters:702        return retriever.search_targeted_chapters(703            diagnostic_string, target_chapters, results_per_sentence=results_per_sentence704        )705    else:706        return retriever.search_targeted_chapters(707            diagnostic_string, target_chapters, results_per_sentence=results_per_sentence708        )709 710def format_smart_search_results(711    diagnostic_string: str,712    search_results: Dict[str, Dict[str, List[Dict]]],  # Updated parameter type713    use_cloud: bool = True714) -> str:715    """Format the results from sentence-based smart_diagnostic_search"""716    717    if not search_results:718        return "โŒ No results found."719    720    retriever = MultiCollectionChapterRetrieval(use_cloud=use_cloud)721    722    output = []723    output.append(f"\n{'='*90}")724    output.append(f"๐Ÿ” SENTENCE-BASED DIAGNOSTIC SEARCH RESULTS")725    output.append(f"๐ŸŽฏ Query: '{diagnostic_string}'")726    output.append(f"{'='*90}")727    728    # Count total results729    total_results = 0730    total_sentences = 0731    for chapter_results in search_results.values():732        total_sentences += len(chapter_results)733        for sentence_data in chapter_results.values():734            total_results += len(sentence_data['results'])735    736    output.append(f"๐Ÿ“Š Total results: {total_results} across {len(search_results)} chapters and {total_sentences} sentences")737    738    for chapter_id, chapter_data in search_results.items():739        description = retriever.chapter_info.get(chapter_id, "Unknown chapter")740        741        output.append(f"\n๐Ÿ“š {chapter_id.upper()}")742        output.append(f"   ๐Ÿ“– {description}")743        output.append(f"   ๐Ÿ“ {len(chapter_data)} sentences processed")744        output.append("-" * 60)745        746        for sentence_key, sentence_data in chapter_data.items():747            sentence_text = sentence_data['text']748            results = sentence_data['results']749            750            output.append(f"\n   ๐Ÿ” {sentence_key.replace('_', ' ').title()}: \"{sentence_text}\"")751            output.append(f"   ๐ŸŽฏ Top {len(results)} matches:")752            output.append("")753            754            for i, result in enumerate(results, 1):755                payload = result['payload']756                code = payload.get('code', 'N/A')757                title = payload.get('title', 'N/A')758                score = result['score']759                760                output.append(f"      {i}. {code} - {title}")761                output.append(f"         ๐Ÿ’ฏ Score: {score:.4f}")762                763                # Show description if available764                desc = payload.get('description', '')765                if desc:766                    desc_preview = desc[:100] + "..." if len(desc) > 100 else desc767                    output.append(f"         ๐Ÿ“„ {desc_preview}")768                769                output.append("")770        771        output.append("=" * 90)772    773    return "\n".join(output)774 775# Example usage776def example_multi_collection_analysis(use_cloud: bool = True):777    """Example of using the multi-collection chapter analysis"""778    779    test_cases = [780        "severe chest pain with shortness of breath",781        "type 2 diabetes with kidney complications", 782        "depression and anxiety disorder",783        "broken wrist from falling",784        "acute appendicitis with fever",785        "skin cancer melanoma",786        "pregnancy complications in third trimester"787    ]788    789    for diagnostic in test_cases:790        print(f"\n{'='*100}")791        print(f"๐Ÿ” ANALYZING: {diagnostic}")792        print(f"{'='*100}")793        794        try:795            # Step 1: Analyze chapter relevance796            analysis = analyze_diagnostic_chapters(diagnostic, detailed=False, use_cloud=use_cloud)797            print(analysis)798            799            # Step 2: Get top relevant chapters800            top_chapters = get_relevant_chapters(diagnostic, top_n=3, use_cloud=use_cloud)801            print(f"\n๐Ÿ† Top 3 relevant chapters: {top_chapters}")802            803            # Step 3: Smart search in those chapters804            search_results = smart_diagnostic_search(805                diagnostic, 806                results_per_sentence=5, 807                use_cloud=use_cloud808            )809            formatted_results = format_smart_search_results(810                diagnostic, 811                search_results, 812                use_cloud=use_cloud813            )814            print(formatted_results)815            816        except Exception as e:817            print(f"โŒ Error processing '{diagnostic}': {e}")818            continue819 820def test_cloud_connection():821    """Test Qdrant Cloud connection and basic functionality"""822    print("๐Ÿงช Testing Qdrant Cloud Connection...")823    824    try:825        retriever = MultiCollectionChapterRetrieval(use_cloud=True)826        827        # Test basic search828        test_query = "heart disease"829        print(f"\n๐Ÿ”ฌ Testing with query: '{test_query}'")830        831        # Get collections832        collections = retriever.get_chapter_collections()833        print(f"๐Ÿ“Š Available collections: {len(collections)}")834        835        if collections:836            # Test search837            top_chapters = retriever.get_top_chapters(test_query, top_n=3)838            print(f"๐ŸŽฏ Top chapters for '{test_query}': {[ch[0] for ch in top_chapters]}")839            840            print("โœ… Cloud connection test successful!")841            return True842        else:843            print("โš ๏ธ  No collections found")844            return False845            846    except Exception as e:847        print(f"โŒ Cloud connection test failed: {e}")848        return False849 850if __name__ == "__main__":851    # Test cloud connection first852    if test_cloud_connection():853        print("\n" + "="*100)854        print("๐Ÿš€ Running example analysis with Qdrant Cloud...")855        print("="*100)856        857        # Run examples with cloud858        example_multi_collection_analysis(use_cloud=True)859    else:860        print("โŒ Skipping examples due to connection issues")861    862    # Or use directly:863    # chapters = get_relevant_chapters("heart attack symptoms", use_cloud=True)864    # results = smart_diagnostic_search("heart attack symptoms", use_cloud=True) 865    # print(format_smart_search_results("heart attack symptoms", results, use_cloud=True))