Team Ai
Apppublic

OutstandingOm/knowledge-graph-env

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes
knowledge_graph_env.py1827 linesDownload Raw Back to root
1import os2import json3import asyncio4import threading5import time6import pickle7import random8import numpy as np9from collections import defaultdict10from contextlib import asynccontextmanager11from typing import Dict, List, Tuple, Optional, Set, Any, Union12from dataclasses import dataclass, field13from enum import Enum14 15from fastapi import FastAPI, HTTPException16from pydantic import BaseModel17import openai18 19# GRADERS — imported from graders.py (LLM-as-a-Judge).20 21from graders import task_easy, task_medium, task_hard, TASKS, GRADERS22 23# Configuration24DIMS = 1625ALPHABET = [chr(ord('A') + i) for i in range(26)]26POSITION_OFFSET = 0.127 28# PHASE 3: MULTI-TIER PRIORITIZATION29# Dimension Partitioning (PHASE 2)30ESSENCE_DIMS = slice(0, 4)      # What something fundamentally IS31IDENTITY_DIMS = slice(4, 12)    # Unique characteristics32TEMPORAL_DIMS = slice(12, 16)   # Scale, time, importance markers33 34# Layer Constants (PHASE 3)35class PriorityLayer(Enum):36    ATOMIC = 0      # DNA letters, character-level37    CLUSTER = 1     # Neighborhood groups (Fruits, Cities)38    DOMAIN = 2      # Structural categories (Agriculture, Tech)39    UNIVERSAL = 3   # Global centroid40 41# Layer-specific learning rates42LR_ATOMIC = 0.0143LR_CLUSTER = 0.00544LR_DOMAIN = 0.00145LR_UNIVERSAL = 0.000146 47# Legacy LRs (mapped to new system)48LR_LETTER = LR_ATOMIC49LR_FEATURE_VEC = LR_CLUSTER50LR_CONCEPT = LR_DOMAIN51 52GRAD_CLIP = 1.053MAX_CONCEPTS = 1000054BATCH_SIZE = 3255TRAIN_INTERVAL_SEC = 1056PERSIST_DIR = "./brain_data"57 58# PHASE 4: GLOBAL-LOCAL HYBRID 59GLOBAL_UPDATE_FREQUENCY = 10  # Update centroid every N batches60 61os.makedirs(PERSIST_DIR, exist_ok=True)62 63USE_VALIDATOR_PROXY = os.environ.get("API_BASE_URL") is not None64if USE_VALIDATOR_PROXY:65    API_BASE_URL = os.environ["API_BASE_URL"]66    API_KEY = os.environ["API_KEY"]67    openai_client = openai.OpenAI(base_url=API_BASE_URL, api_key=API_KEY)68else:69    API_BASE_URL = "https://api.openai.com/v1"70    API_KEY = os.getenv("HF_TOKEN", "")71    openai_client = openai.OpenAI(base_url=API_BASE_URL, api_key=API_KEY) if API_KEY else None72 73MODEL_NAME = os.getenv("MODEL_NAME", "gpt-3.5-turbo")74 75STOP_WORDS = {76    "the", "and", "for", "are", "but", "not", "you", "all", "can", "had",77    "her", "was", "one", "our", "out", "has", "have", "from", "they", "been",78    "said", "each", "which", "their", "will", "other", "about", "many", "then",79    "them", "these", "some", "would", "make", "like", "into", "time", "very",80    "when", "come", "could", "than", "its", "also", "back", "after", "two",81    "how", "what", "where", "who", "why", "this", "that", "with",82}83 84# Relationship color/label constants (PHASE 1.2)85class RelationshipColor(Enum):86    IS_A = "IS_A"           # Apple IS_A Fruit87    HAS_FEATURE = "HAS"     # Apple HAS Color:Red88    GROWN_IN = "GROWN_IN"   # Apple GROWN_IN Kashmir89    LOCATION = "LOCATION"   # Kashmir LOCATION India90    RELATED_TO = "RELATED"  # General relationship91    CAUSES = "CAUSES"       # Login failure CAUSES account lock92    PART_OF = "PART_OF"     # Chhindwara PART_OF MadhyaPradesh93    OPERATOR = "OPERATOR"   # For instruction DNA (e.g., PLUS, MINUS)94    CONDITION = "CONDITION" # IF-THEN logic95 96 97# PHASE 1.2: Weighted & Colored Relationship Data Structure98 99@dataclass100class RelationshipData:101    """Stores weighted, colored relationships between concepts."""102    weight: float = 1.0103    color: str = "RELATED"104    created_at: float = field(default_factory=time.time)105    last_accessed: float = field(default_factory=time.time)106    co_occurrence_count: int = 1107    108    def to_dict(self) -> dict:109        return {110            "weight": self.weight,111            "color": self.color,112            "created_at": self.created_at,113            "last_accessed": self.last_accessed,114            "co_occurrence_count": self.co_occurrence_count115        }116    117    @classmethod118    def from_dict(cls, data: dict) -> 'RelationshipData':119        return cls(120            weight=data.get("weight", 1.0),121            color=data.get("color", "RELATED"),122            created_at=data.get("created_at", time.time()),123            last_accessed=data.get("last_accessed", time.time()),124            co_occurrence_count=data.get("co_occurrence_count", 1)125        )126 127# DynamicOntology (Enhanced with domain awareness)128 129class DynamicOntology:130    def __init__(self):131        self.concept_to_features: Dict[str, List[str]] = {}132        self.feature_to_concepts: Dict[str, List[str]] = defaultdict(list)133        # PHASE 3.3: Domain classification134        self.concept_domains: Dict[str, str] = {}135        self.llm_enabled = True136 137    async def get_features_llm(self, concept: str, context: str = "") -> Tuple[List[str], str]:138        """Returns (features, domain)"""139        if not openai_client:140            return [concept], "general"141        try:142            response = openai_client.chat.completions.create(143                model=MODEL_NAME,144                messages=[145                    {"role": "system", "content": "You are a helpful assistant. Extract features and classify the domain. Return format: 'DOMAIN: <domain> | FEATURES: <comma-separated list>'"},146                    {"role": "user", "content": f"Extract up to 5 features and the domain for '{concept}'. Domain options: Agriculture, Technology, Finance, Geography, General."}147                ],148                temperature=0.3,149                max_tokens=150150            )151            text = response.choices[0].message.content152            domain = "general"153            features = []154            if "DOMAIN:" in text and "FEATURES:" in text:155                domain_part = text.split("DOMAIN:")[1].split("|")[0].strip()156                features_part = text.split("FEATURES:")[1].strip()157                domain = domain_part158                features = [f.strip().lower() for f in features_part.split(",")]159            else:160                features = [f.strip().lower() for f in text.split(",")]161            return features[:5], domain.lower()162        except Exception:163            return [concept], "general"164 165    async def add_concept(self, concept: str, context: str = ""):166        concept_low = concept.lower()167        if concept_low in self.concept_to_features:168            return169        features, domain = await self.get_features_llm(f"{context} {concept}" if context else concept)170        self.concept_to_features[concept_low] = features171        self.concept_domains[concept_low] = domain172        for f in features:173            self.feature_to_concepts[f].append(concept_low)174 175    def get_features(self, concept: str, context: str = "") -> List[str]:176        concept_low = concept.lower()177        if concept_low not in self.concept_to_features:178            return [concept_low]179        return self.concept_to_features[concept_low]180    181    def get_domain(self, concept: str) -> str:182        return self.concept_domains.get(concept.lower(), "general")183 184    def serialize(self) -> dict:185        return {186            "concept_to_features": self.concept_to_features,187            "concept_domains": self.concept_domains188        }189 190    def restore(self, data: dict):191        self.concept_to_features = data.get("concept_to_features", {})192        self.concept_domains = data.get("concept_domains", {})193        self.feature_to_concepts = defaultdict(list)194        for concept, feats in self.concept_to_features.items():195            for f in feats:196                self.feature_to_concepts[f].append(concept)197 198 199# Feature Registry (Enhanced)200 201class FeatureRegistry:202    def __init__(self, ontology: DynamicOntology):203        self.ontology = ontology204        self.feature_to_id: Dict[str, int] = {}205        self.id_to_feature: Dict[int, str] = {}206        self.feature_vectors: Dict[int, np.ndarray] = {}207        self.feature_importance: Dict[int, float] = {}  # PHASE 1.3: Feature mass208        self.next_id = 0209        all_features = set()210        for feats in ontology.concept_to_features.values():211            all_features.update(feats)212        for feat in all_features:213            self.register(feat)214 215    def register(self, feature_name: str, importance: float = 1.0) -> int:216        name = feature_name.lower()217        if name in self.feature_to_id:218            # Increment importance on re-registration219            fid = self.feature_to_id[name]220            self.feature_importance[fid] = min(10.0, self.feature_importance.get(fid, 1.0) + 0.1)221            return fid222        fid = self.next_id223        self.next_id += 1224        self.feature_to_id[name] = fid225        self.id_to_feature[fid] = name226        self.feature_vectors[fid] = np.random.uniform(-1, 1, DIMS).astype(np.float32)227        self.feature_importance[fid] = importance228        return fid229 230    def get_vector(self, fid: int) -> np.ndarray:231        return self.feature_vectors[fid]232    233    def get_importance(self, fid: int) -> float:234        return self.feature_importance.get(fid, 1.0)235 236    def update_vector(self, fid: int, delta: np.ndarray, layer: PriorityLayer = PriorityLayer.CLUSTER):237        """Update with layer-specific learning rate."""238        delta = np.clip(delta, -GRAD_CLIP, GRAD_CLIP)239        lr_map = {240            PriorityLayer.ATOMIC: LR_ATOMIC,241            PriorityLayer.CLUSTER: LR_CLUSTER,242            PriorityLayer.DOMAIN: LR_DOMAIN,243            PriorityLayer.UNIVERSAL: LR_UNIVERSAL244        }245        self.feature_vectors[fid] += delta * lr_map[layer]246 247    def feature_to_letters(self, fid: int, length: int = 5) -> List[str]:248        vec = self.feature_vectors[fid]249        # Use partitioned dimensions for different aspects250        essence_vec = vec[ESSENCE_DIMS]251        probs = np.exp(essence_vec[:min(length, len(essence_vec))])252        probs /= np.sum(probs) + 1e-8253        idx = np.argmax(probs)254        letter_idx = idx % 26255        return [ALPHABET[letter_idx]] * length256 257    def serialize(self) -> dict:258        return {259            "feature_to_id": self.feature_to_id,260            "id_to_feature": {str(k): v for k, v in self.id_to_feature.items()},261            "feature_vectors": {str(k): v.tolist() for k, v in self.feature_vectors.items()},262            "feature_importance": {str(k): v for k, v in self.feature_importance.items()},263            "next_id": self.next_id,264            "ontology": self.ontology.serialize()265        }266 267    def restore(self, data: dict):268        self.feature_to_id = data["feature_to_id"]269        self.id_to_feature = {int(k): v for k, v in data["id_to_feature"].items()}270        self.feature_vectors = {int(k): np.array(v, dtype=np.float32) for k, v in data["feature_vectors"].items()}271        self.feature_importance = {int(k): float(v) for k, v in data.get("feature_importance", {}).items()}272        self.next_id = data["next_id"]273        self.ontology.restore(data["ontology"])274 275 276# Letter Vectors (Enhanced with mass)277 278class LetterVectors:279    def __init__(self):280        self.vec = {ch: np.random.uniform(-1, 1, DIMS).astype(np.float32) for ch in ALPHABET}281        self.letter_importance = {ch: 1.0 for ch in ALPHABET}  # PHASE 1.3282 283    def get(self, letter: str) -> np.ndarray:284        return self.vec[letter]285    286    def get_importance(self, letter: str) -> float:287        return self.letter_importance.get(letter, 1.0)288 289    def update(self, letter: str, delta: np.ndarray, layer: PriorityLayer = PriorityLayer.ATOMIC):290        delta = np.clip(delta, -GRAD_CLIP, GRAD_CLIP)291        lr_map = {292            PriorityLayer.ATOMIC: LR_ATOMIC,293            PriorityLayer.CLUSTER: LR_CLUSTER,294            PriorityLayer.DOMAIN: LR_DOMAIN,295            PriorityLayer.UNIVERSAL: LR_UNIVERSAL296        }297        self.vec[letter] += delta * lr_map[layer]298        # Increment importance on update299        self.letter_importance[letter] = min(10.0, self.letter_importance[letter] + 0.01)300 301    def serialize(self) -> dict:302        return {303            "vectors": {ch: self.vec[ch].tolist() for ch in ALPHABET},304            "importance": self.letter_importance305        }306 307    def restore(self, data: dict):308        if "vectors" in data:309            for ch, arr in data["vectors"].items():310                self.vec[ch] = np.array(arr, dtype=np.float32)311        else:312            for ch, arr in data.items():313                if ch in ALPHABET:314                    self.vec[ch] = np.array(arr, dtype=np.float32)315        if "importance" in data:316            self.letter_importance = data["importance"]317 318# DNAConcept — PHASE 1.3, 1.4, 1.5: Mass, Scaling, Inertia319 320class DNAConcept:321    def __init__(self, name: str, physical_features: List[int], semantic_features: List[int],322                 feature_registry: FeatureRegistry, letter_vec: LetterVectors,323                 importance: float = 1.0, domain: str = "general", numeric_value: Optional[float] = None):324        self.name = name325        self.physical_features = physical_features326        self.semantic_features = semantic_features327        self.feature_registry = feature_registry328        self.letter_vec = letter_vec329        330        # PHASE 1.3: Mass/Importance331        self.importance = importance332        self.domain = domain333        self.cluster_id: Optional[int] = None  # PHASE 3.4334        335        # PHASE 4.3: Pending updates for batch processing336        self.pending_deltas: List[np.ndarray] = []337        338        # INSTRUCTION DNA: numeric value for arithmetic339        self.numeric_value = numeric_value340        341        self.vector: Optional[np.ndarray] = None342        self._update_vector()343 344    def _encode_feature(self, fid: int, start_pos: int) -> np.ndarray:345        letters = self.feature_registry.feature_to_letters(fid, length=5)346        vec = np.zeros(DIMS, dtype=np.float32)347        for i, ch in enumerate(letters):348            base = self.letter_vec.get(ch)349            # PHASE 1.3: Weight by letter importance350            importance_weight = self.letter_vec.get_importance(ch)351            vec += np.sin(base + (start_pos + i) * POSITION_OFFSET) * importance_weight352        return vec353 354    def _update_vector(self):355        vec = np.zeros(DIMS, dtype=np.float32)356        pos = 0357        for fid in self.physical_features:358            vec += self._encode_feature(fid, pos)359            pos += 5360        for fid in self.semantic_features:361            vec += self._encode_feature(fid, pos)362            pos += 5363        norm = np.linalg.norm(vec)364        if norm > 0:365            vec /= norm366        self.vector = vec367        # PHASE 1.4: Apply mass scaling368        self.apply_scale()369 370    #  PHASE 1.4: Vector Scaling by Mass371    def apply_scale(self):372        """Normalize vector but scale its length by its importance/mass."""373        if self.vector is None:374            return375        norm = np.linalg.norm(self.vector)376        if norm > 0:377            target_scale = np.log1p(self.importance)378            self.vector = (self.vector / norm) * target_scale379 380    # PHASE 1.5: Inertia-Based Movement 381    def move_towards(self, other: 'DNAConcept', lr: float = LR_CONCEPT, 382                     weight: float = 1.0, color: str = "RELATED"):383        """384        Core innovation: move concept vectors closer with inertia.385        Heavier nodes (higher importance) move less.386        """387        if self.vector is None or other.vector is None:388            return389            390        # Calculate relative inertia (PHASE 1.5)391        pull_force = (other.importance / (self.importance + 1e-8)) * lr * weight392        other_pull_force = (self.importance / (other.importance + 1e-8)) * lr * weight393        394        # Color-based movement multiplier (PHASE 1.2)395        color_multipliers = {396            "IS_A": 1.2,        # Stronger pull for essential relationships397            "HAS": 0.8,398            "GROWN_IN": 0.9,399            "LOCATION": 0.7,400            "CAUSES": 1.0,401            "PART_OF": 0.85,402            "RELATED": 0.5,403            "OPERATOR": 1.5,    # Strong pull for instruction operators404            "CONDITION": 1.3405        }406        color_mult = color_multipliers.get(color, 0.5)407        pull_force *= color_mult408        other_pull_force *= color_mult409        410        # Store original vectors for gradient calculation411        orig_self = self.vector.copy()412        orig_other = other.vector.copy()413        414        # Apply movement415        diff = other.vector - self.vector416        self.vector += pull_force * diff417        other.vector -= other_pull_force * diff418        419        # Normalize420        self.vector /= (np.linalg.norm(self.vector) + 1e-8)421        other.vector /= (np.linalg.norm(other.vector) + 1e-8)422        423        # PHASE 1.4: Re-apply scale to preserve mass424        self.apply_scale()425        other.apply_scale()426        427        # Backpropagate gradient with partitioned learning rates428        self._backpropagate_to_features(other, orig_self, orig_other, pull_force, other_pull_force, color)429 430    def _backpropagate_to_features(self, other: 'DNAConcept', 431                                    orig_self: np.ndarray, orig_other: np.ndarray,432                                    pull_force: float, other_pull_force: float,433                                    color: str):434        """PHASE 2.3: Partitioned backpropagation with layer-specific LRs."""435        gradient = other.vector - self.vector436        437        # Determine layer based on color438        layer_map = {439            "IS_A": PriorityLayer.DOMAIN,440            "PART_OF": PriorityLayer.DOMAIN,441            "HAS": PriorityLayer.CLUSTER,442            "GROWN_IN": PriorityLayer.CLUSTER,443            "LOCATION": PriorityLayer.CLUSTER,444            "RELATED": PriorityLayer.ATOMIC,445            "OPERATOR": PriorityLayer.UNIVERSAL,446            "CONDITION": PriorityLayer.UNIVERSAL447        }448        layer = layer_map.get(color, PriorityLayer.CLUSTER)449        450        for concept, force in [(self, pull_force), (other, other_pull_force)]:451            for fid in concept.physical_features + concept.semantic_features:452                letters = self.feature_registry.feature_to_letters(fid, length=5)453                for i, ch in enumerate(letters):454                    base = self.letter_vec.get(ch)455                    x = base + i * POSITION_OFFSET456                    grad_sin = np.cos(x)457                    458                    # PHASE 2.2: Apply partitioned gradients459                    # Essence dimensions (0-3) get DOMAIN-level updates460                    essence_grad = grad_sin[ESSENCE_DIMS]461                    identity_grad = grad_sin[IDENTITY_DIMS]462                    temporal_grad = grad_sin[TEMPORAL_DIMS]463                    464                    full_grad = np.zeros(DIMS)465                    full_grad[ESSENCE_DIMS] = essence_grad * LR_DOMAIN466                    full_grad[IDENTITY_DIMS] = identity_grad * LR_CLUSTER467                    full_grad[TEMPORAL_DIMS] = temporal_grad * LR_ATOMIC468                    469                    norm_grad = np.linalg.norm(full_grad) + 1e-8470                    delta_f = force * 0.5 * gradient * full_grad / norm_grad471                    delta_l = force * 0.5 * gradient * full_grad / norm_grad472                    473                    self.feature_registry.update_vector(fid, delta_f, layer)474                    self.letter_vec.update(ch, delta_l, PriorityLayer.ATOMIC)475 476    #  PHASE 2.2: Partitioned Similarity477    def partitioned_similarity(self, other: 'DNAConcept', partition: slice = None) -> float:478        """Calculate similarity using only specified dimensions."""479        if self.vector is None or other.vector is None:480            return 0.0481        if partition is not None:482            v1 = self.vector[partition]483            v2 = other.vector[partition]484        else:485            v1, v2 = self.vector, other.vector486        return float(np.dot(v1, v2) / (np.linalg.norm(v1) * np.linalg.norm(v2) + 1e-8))487 488    def cosine_similarity(self, other: 'DNAConcept') -> float:489        return self.partitioned_similarity(other)490 491    #  PHASE 10.1: Relationship Strengthening492    def strengthen_relationship(self, other_name: str, increment: float = 0.1):493        """Increment importance when concepts co-occur."""494        self.importance = min(10.0, self.importance + increment * 0.1)495 496    def serialize(self) -> dict:497        return {498            "name": self.name,499            "physical_features": self.physical_features,500            "semantic_features": self.semantic_features,501            "vector": self.vector.tolist() if self.vector is not None else None,502            "importance": self.importance,503            "domain": self.domain,504            "cluster_id": self.cluster_id,505            "numeric_value": self.numeric_value506        }507 508    @classmethod509    def from_serialized(cls, data: dict, feature_registry, letter_vec):510        obj = cls(511            data["name"], 512            data["physical_features"], 513            data["semantic_features"], 514            feature_registry, 515            letter_vec,516            importance=data.get("importance", 1.0),517            domain=data.get("domain", "general"),518            numeric_value=data.get("numeric_value")519        )520        if data.get("vector") is not None:521            obj.vector = np.array(data["vector"], dtype=np.float32)522        obj.cluster_id = data.get("cluster_id")523        return obj524 525 526# PHASE 7: Decoder for Sentence Generation527 528class SentenceDecoder:529    """Generates natural language from DNA concept activations."""530    531    def __init__(self, concept_memory: 'ConceptMemory'):532        self.concept_memory = concept_memory533    534    def extract_weighted_features(self, concept_name: str, top_k: int = 5) -> List[Tuple[str, float, str]]:535        """PHASE 7.1: Extract top related concepts sorted by weight."""536        if concept_name not in self.concept_memory.concepts:537            return []538        539        relationships = self.concept_memory.weighted_relationships.get(concept_name, {})540        if not relationships:541            return []542        543        # Sort by weight544        sorted_rels = sorted(545            relationships.items(),546            key=lambda x: x[1].weight,547            reverse=True548        )549        return [(other, rel.weight, rel.color) for other, rel in sorted_rels[:top_k]]550    551    def multi_hop_traversal(self, start: str, target_color: str, max_hops: int = 3) -> List[str]:552        """PHASE 7.2: Follow colored relationships."""553        visited = {start}554        path = [start]555        current = start556        557        for _ in range(max_hops):558            if current not in self.concept_memory.weighted_relationships:559                break560            # Find relationship with target color561            found = None562            for other, rel in self.concept_memory.weighted_relationships[current].items():563                if rel.color == target_color and other not in visited:564                    found = other565                    break566            if found is None:567                break568            path.append(found)569            visited.add(found)570            current = found571        572        return path573    574    def generate_description(self, concept_name: str, template: str = None) -> str:575        """PHASE 7.3: Template-based generation."""576        features = self.extract_weighted_features(concept_name, top_k=5)577        if not features:578            return f"{concept_name} is a concept."579        580        concept = self.concept_memory.concepts.get(concept_name)581        domain = concept.domain if concept else "general"582        583        # Find IS_A relationship584        is_a = next((f[0] for f in features if f[2] == "IS_A"), None)585        # Find HAS relationships586        has_features = [f[0] for f in features if f[2] == "HAS"][:2]587        # Find LOCATION588        location_path = self.multi_hop_traversal(concept_name, "LOCATION", max_hops=2)589        location = location_path[-1] if len(location_path) > 1 else None590        591        # Build sentence592        parts = [concept_name.capitalize()]593        if is_a:594            parts.append(f"is a {is_a}")595        if has_features:596            parts.append(f"with {', '.join(has_features)}")597        if location:598            parts.append(f"located in {location}")599        parts.append(f"(domain: {domain})")600        601        return " ".join(parts) + "."602    603    def self_correct(self, draft: str, concept_name: str) -> str:604        """PHASE 7.5: Self-correction using global consistency check."""605        # Check if draft contains known relationships606        features = self.extract_weighted_features(concept_name)607        feature_names = [f[0] for f in features]608        609        # Simple correction: ensure mentioned concepts are actually related610        draft_lower = draft.lower()611        for feat in feature_names:612            if feat.lower() not in draft_lower:613                # Could add missing important feature614                pass615        616        return draft617 618 619# INSTRUCTION DNA ENGINE (New!)620 621class InstructionEngine:622    """Executes deterministic operations using DNA concepts."""623    624    def __init__(self, concept_memory: 'ConceptMemory', feature_registry: FeatureRegistry, letter_vec: LetterVectors):625        self.memory = concept_memory626        self.feature_registry = feature_registry627        self.letter_vec = letter_vec628        self._ensure_operators()629    630    def _ensure_operators(self):631        """Pre-register arithmetic and logic operators."""632        operators = {633            "PLUS": {"domain": "operator", "importance": 10.0},634            "MINUS": {"domain": "operator", "importance": 10.0},635            "MULTIPLY": {"domain": "operator", "importance": 10.0},636            "DIVIDE": {"domain": "operator", "importance": 10.0},637            "EQUALS": {"domain": "operator", "importance": 10.0},638            "GREATER": {"domain": "operator", "importance": 10.0},639            "LESS": {"domain": "operator", "importance": 10.0},640            "IF": {"domain": "logic", "importance": 10.0},641            "THEN": {"domain": "logic", "importance": 10.0},642            "AND": {"domain": "logic", "importance": 10.0},643            "OR": {"domain": "logic", "importance": 10.0},644            "NOT": {"domain": "logic", "importance": 10.0},645        }646        for op, props in operators.items():647            op_low = op.lower()648            if op_low not in self.memory.concepts:649                physical = [self.feature_registry.register(op_low)]650                semantic = [self.feature_registry.register(op_low)]651                self.memory.register(op_low, physical, semantic, 652                                     importance=props["importance"], 653                                     domain=props["domain"])654    655    def get_or_create_number(self, value: float) -> DNAConcept:656        """Create a concept for a numeric value if it doesn't exist."""657        name = f"num_{value}".replace('.', '_').replace('-', 'neg')658        if name in self.memory.concepts:659            return self.memory.concepts[name]660        physical = [self.feature_registry.register(str(value))]661        semantic = [self.feature_registry.register(str(value))]662        concept = self.memory.register(name, physical, semantic, 663                                       importance=abs(value)/10 + 1.0, 664                                       domain="number",665                                       numeric_value=value)666        return concept667    668    def execute_arithmetic(self, operator: str, a: Union[str, float], b: Union[str, float]) -> DNAConcept:669        """Perform arithmetic and return result concept."""670        # Convert to concepts671        if isinstance(a, (int, float)):672            concept_a = self.get_or_create_number(float(a))673        else:674            concept_a = self.memory.concepts.get(a.lower())675            if concept_a is None:676                raise ValueError(f"Concept '{a}' not found")677        678        if isinstance(b, (int, float)):679            concept_b = self.get_or_create_number(float(b))680        else:681            concept_b = self.memory.concepts.get(b.lower())682            if concept_b is None:683                raise ValueError(f"Concept '{b}' not found")684        685        val_a = concept_a.numeric_value if concept_a.numeric_value is not None else concept_a.importance686        val_b = concept_b.numeric_value if concept_b.numeric_value is not None else concept_b.importance687        688        op_upper = operator.upper()689        if op_upper == "PLUS":690            result_val = val_a + val_b691        elif op_upper == "MINUS":692            result_val = val_a - val_b693        elif op_upper == "MULTIPLY":694            result_val = val_a * val_b695        elif op_upper == "DIVIDE":696            result_val = val_a / val_b if val_b != 0 else float('inf')697        else:698            raise ValueError(f"Unknown operator: {operator}")699        700        result_concept = self.get_or_create_number(result_val)701        702        # Create OPERATOR relationships703        self.memory.add_weighted_relationship(concept_a.name, concept_b.name, weight=0.9, color="OPERATOR")704        self.memory.add_weighted_relationship(operator.lower(), result_concept.name, weight=1.0, color="OPERATOR")705        706        return result_concept707    708    def evaluate_condition(self, condition_expr: Dict) -> bool:709        """Evaluate a logic condition (IF part)."""710        op = condition_expr.get("operator", "EQUALS").upper()711        left = condition_expr.get("left")712        right = condition_expr.get("right")713        714        if isinstance(left, str):715            left_conc = self.memory.concepts.get(left.lower())716            left_val = left_conc.numeric_value if left_conc and left_conc.numeric_value is not None else (left_conc.importance if left_conc else 0)717        else:718            left_val = float(left)719            720        if isinstance(right, str):721            right_conc = self.memory.concepts.get(right.lower())722            right_val = right_conc.numeric_value if right_conc and right_conc.numeric_value is not None else (right_conc.importance if right_conc else 0)723        else:724            right_val = float(right)725        726        if op == "EQUALS":727            return abs(left_val - right_val) < 1e-6728        elif op == "GREATER":729            return left_val > right_val730        elif op == "LESS":731            return left_val < right_val732        elif op == "AND":733            return bool(left_val) and bool(right_val)734        elif op == "OR":735            return bool(left_val) or bool(right_val)736        elif op == "NOT":737            return not bool(left_val)738        return False739 740# Reasoning Engine — Enhanced with Instruction DNA741 742class ReasoningEngine:743    def __init__(self, concept_memory: 'ConceptMemory', feature_registry: FeatureRegistry, letter_vec: LetterVectors):744        self.concept_memory = concept_memory745        self.decoder = SentenceDecoder(concept_memory)746        self.instruction_engine = InstructionEngine(concept_memory, feature_registry, letter_vec)747 748    def multi_hop_reasoning(self, start: str, max_hops: int = 3, decay: float = 0.7,749                            color_filter: Optional[str] = None) -> Dict[str, float]:750        """PHASE 7.2: Color-filtered multi-hop reasoning."""751        if start not in self.concept_memory.weighted_relationships:752            return {}753        754        activation = {start: 1.0}755        for hop in range(max_hops):756            new_activation = {}757            for node, score in activation.items():758                relationships = self.concept_memory.weighted_relationships.get(node, {})759                for nb, rel in relationships.items():760                    if color_filter and rel.color != color_filter:761                        continue762                    weight = rel.weight763                    if node in self.concept_memory.concepts and nb in self.concept_memory.concepts:764                        sim = self.concept_memory.concepts[node].cosine_similarity(765                            self.concept_memory.concepts[nb]766                        )767                        weight *= (0.5 + 0.5 * sim)768                    new_activation[nb] = new_activation.get(nb, 0) + score * weight * decay769            for k, v in new_activation.items():770                activation[k] = activation.get(k, 0) + v771        772        if activation:773            max_act = max(activation.values())774            activation = {k: v/max_act for k, v in activation.items()}775        return activation776 777    def analogical_reasoning(self, a: str, b: str, c: str, 778                             partition: slice = None) -> List[str]:779        """A is to B as C is to ? with optional dimension partition."""780        if a not in self.concept_memory.concepts or b not in self.concept_memory.concepts:781            return []782        vec_a = self.concept_memory.concepts[a].vector783        vec_b = self.concept_memory.concepts[b].vector784        785        if partition is not None:786            direction = vec_b[partition] - vec_a[partition]787        else:788            direction = vec_b - vec_a789            790        if c not in self.concept_memory.concepts:791            return []792        vec_c = self.concept_memory.concepts[c].vector793        794        if partition is not None:795            target = vec_c.copy()796            target[partition] = vec_c[partition] + direction797        else:798            target = vec_c + direction799            800        target /= (np.linalg.norm(target) + 1e-8)801        results = self.concept_memory.partitioned_search(target, top_k=5, partition=partition)802        return [r[0] for r in results if r[0] not in (a, b, c)]803    804    def generate_sentence(self, concept: str) -> str:805        """Generate a descriptive sentence for a concept."""806        return self.decoder.generate_description(concept)807    808    # New instruction methods809    def calculate(self, expression: str) -> Dict:810        """Evaluate arithmetic expression using Instruction DNA."""811        import re812        allowed = set('0123456789.+-*/() ')813        if not all(c in allowed for c in expression):814            return {"error": "Invalid characters in expression"}815        try:816            result = eval(expression)817            concept = self.instruction_engine.get_or_create_number(result)818            return {"expression": expression, "result": result, "concept": concept.name}819        except Exception as e:820            return {"error": str(e)}821    822    def execute_instruction(self, operator: str, a: Union[str, float], b: Union[str, float]) -> Dict:823        """Execute a single arithmetic operation."""824        try:825            result_concept = self.instruction_engine.execute_arithmetic(operator, a, b)826            return {827                "operator": operator,828                "operands": [a, b],829                "result": result_concept.numeric_value,830                "concept": result_concept.name831            }832        except Exception as e:833            return {"error": str(e)}834    835    def evaluate_rule(self, condition: Dict, action: str) -> Dict:836        """IF-THEN rule evaluation."""837        cond_result = self.instruction_engine.evaluate_condition(condition)838        return {"condition_true": cond_result, "action_triggered": action if cond_result else None}839 840 841# Concept Memory — PHASE 1-4: All Upgrades842 843class ConceptMemory:844    def __init__(self, feature_registry: FeatureRegistry, letter_vec: LetterVectors,845                 max_concepts: int = MAX_CONCEPTS):846        self.feature_registry = feature_registry847        self.letter_vec = letter_vec848        self.concepts: Dict[str, DNAConcept] = {}849        850        # PHASE 1.1, 1.2: Weighted and colored relationships851        self.weighted_relationships: Dict[str, Dict[str, RelationshipData]] = defaultdict(dict)852        853        # Legacy compatibility854        self.relationships: Dict[str, Set[str]] = defaultdict(set)855        856        self.max_concepts = max_concepts857        858        # PHASE 4.1: Global centroid tracking859        self.global_centroid: Optional[np.ndarray] = None860        self.batch_counter = 0861        862        # PHASE 4.3: Pending updates buffer863        self.pending_updates: List[Tuple[str, str, float, str]] = []864        865        # FAISS866        self._faiss_available = False867        self.index = None868        self.id_to_name: Dict[int, str] = {}869        self.name_to_id: Dict[str, int] = {}870        self.next_id = 0871        872        # PHASE 4.5: Hierarchical FAISS873        self.cluster_index = None874        self.cluster_to_concepts: Dict[int, List[str]] = defaultdict(list)875        876        try:877            import faiss878            self._faiss_available = True879        except Exception:880            self._faiss_available = False881 882    # PHASE 4.1: Global Context 883    def update_global_centroid(self):884        """Calculate the center of gravity of all concepts."""885        if not self.concepts:886            return887        all_vecs = np.array([c.vector for c in self.concepts.values() if c.vector is not None])888        if len(all_vecs) > 0:889            self.global_centroid = np.mean(all_vecs, axis=0)890 891    def get_global_context(self) -> dict:892        """PHASE 4.2: Return global centroid and top anchor nodes."""893        if self.global_centroid is None:894            self.update_global_centroid()895        896        # Find top 5 most important nodes897        top_nodes = sorted(898            self.concepts.values(), 899            key=lambda x: x.importance, 900            reverse=True901        )[:5]902        903        return {904            "global_centroid": self.global_centroid.tolist() if self.global_centroid is not None else None,905            "anchors": [n.name for n in top_nodes],906            "total_concepts": len(self.concepts)907        }908 909    # PHASE 4.4: Global Consistency Check910    def validate_against_global(self, concept_vector: np.ndarray, threshold: float = 2.0) -> bool:911        """Check if a vector is within reasonable distance from global centroid."""912        if self.global_centroid is None:913            return True914        distance = np.linalg.norm(concept_vector - self.global_centroid)915        std = np.std([c.vector for c in self.concepts.values()]) if self.concepts else 1.0916        return distance < threshold * std917 918    def _ensure_index(self):919        if not self._faiss_available:920            return921        if self.index is None:922            import faiss923            self.index = faiss.IndexFlatIP(DIMS)924 925    def _rebuild_index(self):926        if not self._faiss_available or not self.concepts:927            return928        import faiss929        vectors = [c.vector for c in self.concepts.values() if c.vector is not None]930        names = [name for name, c in self.concepts.items() if c.vector is not None]931        if not vectors:932            return933        vecs = np.vstack(vectors).astype(np.float32)934        self.index = faiss.IndexFlatIP(DIMS)935        self.index.add(vecs)936        self.id_to_name = {i: n for i, n in enumerate(names)}937        self.name_to_id = {n: i for i, n in enumerate(names)}938        self.next_id = len(self.concepts)939        940        # PHASE 4.1: Update global centroid after rebuild941        self.update_global_centroid()942 943    def register(self, name: str, physical_features: List[int], semantic_features: List[int],944                 importance: float = 1.0, domain: str = "general", numeric_value: Optional[float] = None) -> DNAConcept:945        name_low = name.lower()946        if name_low in self.concepts:947            # PHASE 10.2: Strengthen on re-registration948            self.concepts[name_low].importance = min(10.0, self.concepts[name_low].importance + 0.1)949            return self.concepts[name_low]950        951        concept = DNAConcept(name_low, physical_features, semantic_features,952                             self.feature_registry, self.letter_vec,953                             importance=importance, domain=domain, numeric_value=numeric_value)954        self.concepts[name_low] = concept955        956        if self._faiss_available and concept.vector is not None:957            self._ensure_index()958            if self.index is not None:959                self.index.add(concept.vector.reshape(1, -1))960                self.id_to_name[self.next_id] = name_low961                self.name_to_id[name_low] = self.next_id962                self.next_id += 1963        964        self._prune()965        return concept966 967    # PHASE 1.1, 1.2: Weighted & Colored Relationships968    def add_weighted_relationship(self, a: str, b: str, weight: float = 1.0, 969                                   color: str = "RELATED"):970        a_low, b_low = a.lower(), b.lower()971        if a_low not in self.concepts or b_low not in self.concepts:972            return973        974        # Create or update relationship data975        if b_low not in self.weighted_relationships[a_low]:976            self.weighted_relationships[a_low][b_low] = RelationshipData(977                weight=weight, 978                color=color,979                co_occurrence_count=1980            )981        else:982            rel = self.weighted_relationships[a_low][b_low]983            rel.weight = min(1.0, rel.weight + 0.1 * weight)984            rel.co_occurrence_count += 1985            rel.last_accessed = time.time()986        987        # Symmetric relationship988        if a_low not in self.weighted_relationships[b_low]:989            self.weighted_relationships[b_low][a_low] = RelationshipData(990                weight=weight,991                color=color,992                co_occurrence_count=1993            )994        else:995            rel = self.weighted_relationships[b_low][a_low]996            rel.weight = min(1.0, rel.weight + 0.1 * weight)997            rel.co_occurrence_count += 1998            rel.last_accessed = time.time()999        1000        # Legacy compatibility1001        self.relationships[a_low].add(b_low)1002        self.relationships[b_low].add(a_low)1003        1004        # Move vectors with inertia1005        self.concepts[a_low].move_towards(self.concepts[b_low], 1006                                           lr=LR_CONCEPT, 1007                                           weight=weight, 1008                                           color=color)1009 1010    def add_relationship(self, a: str, b: str):1011        """Legacy method for backward compatibility."""1012        self.add_weighted_relationship(a, b, weight=1.0, color="RELATED")1013 1014    # PHASE 10.2: Strengthen on Co-occurrence1015    def strengthen_relationship(self, a: str, b: str, increment: float = 0.1):1016        a_low, b_low = a.lower(), b.lower()1017        if a_low in self.weighted_relationships and b_low in self.weighted_relationships[a_low]:1018            rel = self.weighted_relationships[a_low][b_low]1019            rel.weight = min(1.0, rel.weight + increment)1020            rel.co_occurrence_count += 11021            rel.last_accessed = time.time()1022            1023            # Also strengthen symmetric1024            if a_low in self.weighted_relationships[b_low]:1025                rel2 = self.weighted_relationships[b_low][a_low]1026                rel2.weight = min(1.0, rel2.weight + increment)1027                rel2.co_occurrence_count += 11028                rel2.last_accessed = time.time()1029 1030    # PHASE 10.1: Relationship Decay1031    def apply_decay(self, decay_rate: float = 0.001, inactive_threshold: float = 86400):1032        """Decay relationships that haven't been accessed recently."""1033        current_time = time.time()1034        for a, rels in self.weighted_relationships.items():1035            to_remove = []1036            for b, rel in rels.items():1037                if current_time - rel.last_accessed > inactive_threshold:1038                    rel.weight = max(0.05, rel.weight - decay_rate)1039                    if rel.weight <= 0.06:1040                        to_remove.append(b)1041            for b in to_remove:1042                del self.weighted_relationships[a][b]1043                if b in self.relationships[a]:1044                    self.relationships[a].remove(b)1045 1046    #  PHASE 2.2: Partitioned Search1047    def partitioned_search(self, query_vector: np.ndarray, top_k: int = 5,1048                           partition: slice = None) -> List[Tuple[str, float]]:1049        """Search using only specified dimension partition."""1050        if not self.concepts:1051            return []1052        1053        if partition is not None:1054            q = query_vector[partition]1055            vecs = np.vstack([self.concepts[n].vector[partition] for n in self.concepts.keys()])1056        else:1057            q = query_vector1058            vecs = np.vstack([self.concepts[n].vector for n in self.concepts.keys()])1059        1060        names = list(self.concepts.keys())1061        q = q / (np.linalg.norm(q) + 1e-8)1062        vecs_norm = vecs / (np.linalg.norm(vecs, axis=1, keepdims=True) + 1e-8)1063        scores = vecs_norm @ q1064        top = np.argsort(scores)[::-1][:top_k]1065        return [(names[i], float(scores[i])) for i in top]1066 1067    def search(self, query_vector: np.ndarray, top_k: int = 5) -> List[Tuple[str, float]]:1068        """Full-vector search."""1069        return self.partitioned_search(query_vector, top_k, partition=None)1070 1071    #  PHASE 4.3: Batch Processing 1072    def add_to_batch(self, a: str, b: str, weight: float, color: str):1073        """Add relationship update to pending batch."""1074        self.pending_updates.append((a, b, weight, color))1075        self.batch_counter += 11076        1077        if len(self.pending_updates) >= BATCH_SIZE:1078            self.process_batch()1079 1080    def process_batch(self):1081        """Process all pending updates and update global statistics."""1082        if not self.pending_updates:1083            return1084        1085        for a, b, weight, color in self.pending_updates:1086            if a in self.concepts and b in self.concepts:1087                self.concepts[a].move_towards(self.concepts[b], 1088                                               lr=LR_CONCEPT, 1089                                               weight=weight, 1090                                               color=color)1091        1092        self.pending_updates.clear()1093        1094        # PHASE 4.1: Update global centroid periodically1095        if self.batch_counter % GLOBAL_UPDATE_FREQUENCY == 0:1096            self.update_global_centroid()1097        1098        self._rebuild_index()1099 1100    async def extract_and_link(self, text: str, ontology: DynamicOntology, 1101                                sector: str = "general") -> List[str]:1102        words = [w for w in text.lower().split() if len(w) > 3 and w not in STOP_WORDS]1103        unique = list(set(words))[:15]1104        concept_list = []1105        1106        for kw in unique:1107            features = await ontology.get_features_llm(kw) if ontology.llm_enabled else (ontology.get_features(kw), "general")1108            if isinstance(features, tuple):1109                features, domain = features1110            else:1111                domain = "general"1112            1113            physical_fids = [self.feature_registry.register(f) for f in features]1114            semantic_fids = [self.feature_registry.register(f) for f in features]1115            1116            # PHASE 1.3: Initial importance based on word frequency in corpus1117            importance = 1.0 + (0.1 * unique.index(kw) if kw in unique else 0)1118            1119            concept = self.register(kw, physical_fids, semantic_fids, 1120                                    importance=importance, domain=domain)1121            concept_list.append(kw)1122        1123        # Create relationships with colors inferred from context1124        for i in range(len(unique)):1125            for j in range(i+1, min(i+4, len(unique))):1126                # Infer relationship color from text context1127                color = self._infer_relationship_color(text, unique[i], unique[j])1128                self.add_weighted_relationship(unique[i], unique[j], weight=0.8, color=color)1129        1130        return concept_list1131 1132    def _infer_relationship_color(self, text: str, a: str, b: str) -> str:1133        """Infer relationship type from context."""1134        text_lower = text.lower()1135        if f"{a} is {b}" in text_lower or f"{b} is {a}" in text_lower:1136            return "IS_A"1137        elif f"{a} has {b}" in text_lower or f"{b} has {a}" in text_lower:1138            return "HAS"1139        elif f"{a} in {b}" in text_lower or f"{b} in {a}" in text_lower:1140            return "LOCATION"1141        elif "cause" in text_lower and (a in text_lower or b in text_lower):1142            return "CAUSES"1143        return "RELATED"1144 1145    def _prune(self):1146        if len(self.concepts) > self.max_concepts:1147            # Sort by importance (keep high importance concepts)1148            sorted_concepts = sorted(1149                self.concepts.items(), 1150                key=lambda x: (x[1].importance, len(self.weighted_relationships.get(x[0], {}))),1151                reverse=True1152            )1153            to_keep = sorted_concepts[:self.max_concepts]1154            self.concepts = {name: concept for name, concept in to_keep}1155            1156            # Clean up relationships for removed concepts1157            keep_names = set(self.concepts.keys())1158            self.weighted_relationships = defaultdict(dict, {1159                k: {b: r for b, r in v.items() if b in keep_names}1160                for k, v in self.weighted_relationships.items() if k in keep_names1161            })1162            self.relationships = defaultdict(set, {1163                k: v.intersection(keep_names) 1164                for k, v in self.relationships.items() if k in keep_names1165            })1166            1167            self._rebuild_index()1168 1169    def serialize(self) -> dict:1170        return {1171            "concepts": {name: c.serialize() for name, c in self.concepts.items()},1172            "weighted_relationships": {1173                k: {b: r.to_dict() for b, r in v.items()}1174                for k, v in self.weighted_relationships.items()1175            },1176            "relationships": {k: list(v) for k, v in self.relationships.items()},1177            "global_centroid": self.global_centroid.tolist() if self.global_centroid is not None else None1178        }1179 1180    def restore(self, data: dict):1181        self.concepts = {}1182        self.weighted_relationships = defaultdict(dict)1183        self.relationships = defaultdict(set)1184        1185        for name, cdata in data.get("concepts", {}).items():1186            self.concepts[name] = DNAConcept.from_serialized(cdata, self.feature_registry, self.letter_vec)1187        1188        for k, vdict in data.get("weighted_relationships", {}).items():1189            for b, rdata in vdict.items():1190                self.weighted_relationships[k][b] = RelationshipData.from_dict(rdata)1191                self.relationships[k].add(b)1192        1193        for k, vlist in data.get("relationships", {}).items():1194            self.relationships[k].update(vlist)1195        1196        if data.get("global_centroid"):1197            self.global_centroid = np.array(data["global_centroid"], dtype=np.float32)1198        1199        self._rebuild_index()1200 

Showing the first 1,200 of 1827 lines. Download the file for the rest.