OutstandingOm/knowledge-graph-env
1
1import os2import json3import asyncio4import threading5import time6import pickle7import random8import numpy as np9from collections import defaultdict10from contextlib import asynccontextmanager11from typing import Dict, List, Tuple, Optional, Set, Any, Union12from dataclasses import dataclass, field13from enum import Enum14 15from fastapi import FastAPI, HTTPException16from pydantic import BaseModel17import openai18 19# GRADERS — imported from graders.py (LLM-as-a-Judge).20 21from graders import task_easy, task_medium, task_hard, TASKS, GRADERS22 23# Configuration24DIMS = 1625ALPHABET = [chr(ord('A') + i) for i in range(26)]26POSITION_OFFSET = 0.127 28# PHASE 3: MULTI-TIER PRIORITIZATION29# Dimension Partitioning (PHASE 2)30ESSENCE_DIMS = slice(0, 4) # What something fundamentally IS31IDENTITY_DIMS = slice(4, 12) # Unique characteristics32TEMPORAL_DIMS = slice(12, 16) # Scale, time, importance markers33 34# Layer Constants (PHASE 3)35class PriorityLayer(Enum):36 ATOMIC = 0 # DNA letters, character-level37 CLUSTER = 1 # Neighborhood groups (Fruits, Cities)38 DOMAIN = 2 # Structural categories (Agriculture, Tech)39 UNIVERSAL = 3 # Global centroid40 41# Layer-specific learning rates42LR_ATOMIC = 0.0143LR_CLUSTER = 0.00544LR_DOMAIN = 0.00145LR_UNIVERSAL = 0.000146 47# Legacy LRs (mapped to new system)48LR_LETTER = LR_ATOMIC49LR_FEATURE_VEC = LR_CLUSTER50LR_CONCEPT = LR_DOMAIN51 52GRAD_CLIP = 1.053MAX_CONCEPTS = 1000054BATCH_SIZE = 3255TRAIN_INTERVAL_SEC = 1056PERSIST_DIR = "./brain_data"57 58# PHASE 4: GLOBAL-LOCAL HYBRID 59GLOBAL_UPDATE_FREQUENCY = 10 # Update centroid every N batches60 61os.makedirs(PERSIST_DIR, exist_ok=True)62 63USE_VALIDATOR_PROXY = os.environ.get("API_BASE_URL") is not None64if USE_VALIDATOR_PROXY:65 API_BASE_URL = os.environ["API_BASE_URL"]66 API_KEY = os.environ["API_KEY"]67 openai_client = openai.OpenAI(base_url=API_BASE_URL, api_key=API_KEY)68else:69 API_BASE_URL = "https://api.openai.com/v1"70 API_KEY = os.getenv("HF_TOKEN", "")71 openai_client = openai.OpenAI(base_url=API_BASE_URL, api_key=API_KEY) if API_KEY else None72 73MODEL_NAME = os.getenv("MODEL_NAME", "gpt-3.5-turbo")74 75STOP_WORDS = {76 "the", "and", "for", "are", "but", "not", "you", "all", "can", "had",77 "her", "was", "one", "our", "out", "has", "have", "from", "they", "been",78 "said", "each", "which", "their", "will", "other", "about", "many", "then",79 "them", "these", "some", "would", "make", "like", "into", "time", "very",80 "when", "come", "could", "than", "its", "also", "back", "after", "two",81 "how", "what", "where", "who", "why", "this", "that", "with",82}83 84# Relationship color/label constants (PHASE 1.2)85class RelationshipColor(Enum):86 IS_A = "IS_A" # Apple IS_A Fruit87 HAS_FEATURE = "HAS" # Apple HAS Color:Red88 GROWN_IN = "GROWN_IN" # Apple GROWN_IN Kashmir89 LOCATION = "LOCATION" # Kashmir LOCATION India90 RELATED_TO = "RELATED" # General relationship91 CAUSES = "CAUSES" # Login failure CAUSES account lock92 PART_OF = "PART_OF" # Chhindwara PART_OF MadhyaPradesh93 OPERATOR = "OPERATOR" # For instruction DNA (e.g., PLUS, MINUS)94 CONDITION = "CONDITION" # IF-THEN logic95 96 97# PHASE 1.2: Weighted & Colored Relationship Data Structure98 99@dataclass100class RelationshipData:101 """Stores weighted, colored relationships between concepts."""102 weight: float = 1.0103 color: str = "RELATED"104 created_at: float = field(default_factory=time.time)105 last_accessed: float = field(default_factory=time.time)106 co_occurrence_count: int = 1107 108 def to_dict(self) -> dict:109 return {110 "weight": self.weight,111 "color": self.color,112 "created_at": self.created_at,113 "last_accessed": self.last_accessed,114 "co_occurrence_count": self.co_occurrence_count115 }116 117 @classmethod118 def from_dict(cls, data: dict) -> 'RelationshipData':119 return cls(120 weight=data.get("weight", 1.0),121 color=data.get("color", "RELATED"),122 created_at=data.get("created_at", time.time()),123 last_accessed=data.get("last_accessed", time.time()),124 co_occurrence_count=data.get("co_occurrence_count", 1)125 )126 127# DynamicOntology (Enhanced with domain awareness)128 129class DynamicOntology:130 def __init__(self):131 self.concept_to_features: Dict[str, List[str]] = {}132 self.feature_to_concepts: Dict[str, List[str]] = defaultdict(list)133 # PHASE 3.3: Domain classification134 self.concept_domains: Dict[str, str] = {}135 self.llm_enabled = True136 137 async def get_features_llm(self, concept: str, context: str = "") -> Tuple[List[str], str]:138 """Returns (features, domain)"""139 if not openai_client:140 return [concept], "general"141 try:142 response = openai_client.chat.completions.create(143 model=MODEL_NAME,144 messages=[145 {"role": "system", "content": "You are a helpful assistant. Extract features and classify the domain. Return format: 'DOMAIN: <domain> | FEATURES: <comma-separated list>'"},146 {"role": "user", "content": f"Extract up to 5 features and the domain for '{concept}'. Domain options: Agriculture, Technology, Finance, Geography, General."}147 ],148 temperature=0.3,149 max_tokens=150150 )151 text = response.choices[0].message.content152 domain = "general"153 features = []154 if "DOMAIN:" in text and "FEATURES:" in text:155 domain_part = text.split("DOMAIN:")[1].split("|")[0].strip()156 features_part = text.split("FEATURES:")[1].strip()157 domain = domain_part158 features = [f.strip().lower() for f in features_part.split(",")]159 else:160 features = [f.strip().lower() for f in text.split(",")]161 return features[:5], domain.lower()162 except Exception:163 return [concept], "general"164 165 async def add_concept(self, concept: str, context: str = ""):166 concept_low = concept.lower()167 if concept_low in self.concept_to_features:168 return169 features, domain = await self.get_features_llm(f"{context} {concept}" if context else concept)170 self.concept_to_features[concept_low] = features171 self.concept_domains[concept_low] = domain172 for f in features:173 self.feature_to_concepts[f].append(concept_low)174 175 def get_features(self, concept: str, context: str = "") -> List[str]:176 concept_low = concept.lower()177 if concept_low not in self.concept_to_features:178 return [concept_low]179 return self.concept_to_features[concept_low]180 181 def get_domain(self, concept: str) -> str:182 return self.concept_domains.get(concept.lower(), "general")183 184 def serialize(self) -> dict:185 return {186 "concept_to_features": self.concept_to_features,187 "concept_domains": self.concept_domains188 }189 190 def restore(self, data: dict):191 self.concept_to_features = data.get("concept_to_features", {})192 self.concept_domains = data.get("concept_domains", {})193 self.feature_to_concepts = defaultdict(list)194 for concept, feats in self.concept_to_features.items():195 for f in feats:196 self.feature_to_concepts[f].append(concept)197 198 199# Feature Registry (Enhanced)200 201class FeatureRegistry:202 def __init__(self, ontology: DynamicOntology):203 self.ontology = ontology204 self.feature_to_id: Dict[str, int] = {}205 self.id_to_feature: Dict[int, str] = {}206 self.feature_vectors: Dict[int, np.ndarray] = {}207 self.feature_importance: Dict[int, float] = {} # PHASE 1.3: Feature mass208 self.next_id = 0209 all_features = set()210 for feats in ontology.concept_to_features.values():211 all_features.update(feats)212 for feat in all_features:213 self.register(feat)214 215 def register(self, feature_name: str, importance: float = 1.0) -> int:216 name = feature_name.lower()217 if name in self.feature_to_id:218 # Increment importance on re-registration219 fid = self.feature_to_id[name]220 self.feature_importance[fid] = min(10.0, self.feature_importance.get(fid, 1.0) + 0.1)221 return fid222 fid = self.next_id223 self.next_id += 1224 self.feature_to_id[name] = fid225 self.id_to_feature[fid] = name226 self.feature_vectors[fid] = np.random.uniform(-1, 1, DIMS).astype(np.float32)227 self.feature_importance[fid] = importance228 return fid229 230 def get_vector(self, fid: int) -> np.ndarray:231 return self.feature_vectors[fid]232 233 def get_importance(self, fid: int) -> float:234 return self.feature_importance.get(fid, 1.0)235 236 def update_vector(self, fid: int, delta: np.ndarray, layer: PriorityLayer = PriorityLayer.CLUSTER):237 """Update with layer-specific learning rate."""238 delta = np.clip(delta, -GRAD_CLIP, GRAD_CLIP)239 lr_map = {240 PriorityLayer.ATOMIC: LR_ATOMIC,241 PriorityLayer.CLUSTER: LR_CLUSTER,242 PriorityLayer.DOMAIN: LR_DOMAIN,243 PriorityLayer.UNIVERSAL: LR_UNIVERSAL244 }245 self.feature_vectors[fid] += delta * lr_map[layer]246 247 def feature_to_letters(self, fid: int, length: int = 5) -> List[str]:248 vec = self.feature_vectors[fid]249 # Use partitioned dimensions for different aspects250 essence_vec = vec[ESSENCE_DIMS]251 probs = np.exp(essence_vec[:min(length, len(essence_vec))])252 probs /= np.sum(probs) + 1e-8253 idx = np.argmax(probs)254 letter_idx = idx % 26255 return [ALPHABET[letter_idx]] * length256 257 def serialize(self) -> dict:258 return {259 "feature_to_id": self.feature_to_id,260 "id_to_feature": {str(k): v for k, v in self.id_to_feature.items()},261 "feature_vectors": {str(k): v.tolist() for k, v in self.feature_vectors.items()},262 "feature_importance": {str(k): v for k, v in self.feature_importance.items()},263 "next_id": self.next_id,264 "ontology": self.ontology.serialize()265 }266 267 def restore(self, data: dict):268 self.feature_to_id = data["feature_to_id"]269 self.id_to_feature = {int(k): v for k, v in data["id_to_feature"].items()}270 self.feature_vectors = {int(k): np.array(v, dtype=np.float32) for k, v in data["feature_vectors"].items()}271 self.feature_importance = {int(k): float(v) for k, v in data.get("feature_importance", {}).items()}272 self.next_id = data["next_id"]273 self.ontology.restore(data["ontology"])274 275 276# Letter Vectors (Enhanced with mass)277 278class LetterVectors:279 def __init__(self):280 self.vec = {ch: np.random.uniform(-1, 1, DIMS).astype(np.float32) for ch in ALPHABET}281 self.letter_importance = {ch: 1.0 for ch in ALPHABET} # PHASE 1.3282 283 def get(self, letter: str) -> np.ndarray:284 return self.vec[letter]285 286 def get_importance(self, letter: str) -> float:287 return self.letter_importance.get(letter, 1.0)288 289 def update(self, letter: str, delta: np.ndarray, layer: PriorityLayer = PriorityLayer.ATOMIC):290 delta = np.clip(delta, -GRAD_CLIP, GRAD_CLIP)291 lr_map = {292 PriorityLayer.ATOMIC: LR_ATOMIC,293 PriorityLayer.CLUSTER: LR_CLUSTER,294 PriorityLayer.DOMAIN: LR_DOMAIN,295 PriorityLayer.UNIVERSAL: LR_UNIVERSAL296 }297 self.vec[letter] += delta * lr_map[layer]298 # Increment importance on update299 self.letter_importance[letter] = min(10.0, self.letter_importance[letter] + 0.01)300 301 def serialize(self) -> dict:302 return {303 "vectors": {ch: self.vec[ch].tolist() for ch in ALPHABET},304 "importance": self.letter_importance305 }306 307 def restore(self, data: dict):308 if "vectors" in data:309 for ch, arr in data["vectors"].items():310 self.vec[ch] = np.array(arr, dtype=np.float32)311 else:312 for ch, arr in data.items():313 if ch in ALPHABET:314 self.vec[ch] = np.array(arr, dtype=np.float32)315 if "importance" in data:316 self.letter_importance = data["importance"]317 318# DNAConcept — PHASE 1.3, 1.4, 1.5: Mass, Scaling, Inertia319 320class DNAConcept:321 def __init__(self, name: str, physical_features: List[int], semantic_features: List[int],322 feature_registry: FeatureRegistry, letter_vec: LetterVectors,323 importance: float = 1.0, domain: str = "general", numeric_value: Optional[float] = None):324 self.name = name325 self.physical_features = physical_features326 self.semantic_features = semantic_features327 self.feature_registry = feature_registry328 self.letter_vec = letter_vec329 330 # PHASE 1.3: Mass/Importance331 self.importance = importance332 self.domain = domain333 self.cluster_id: Optional[int] = None # PHASE 3.4334 335 # PHASE 4.3: Pending updates for batch processing336 self.pending_deltas: List[np.ndarray] = []337 338 # INSTRUCTION DNA: numeric value for arithmetic339 self.numeric_value = numeric_value340 341 self.vector: Optional[np.ndarray] = None342 self._update_vector()343 344 def _encode_feature(self, fid: int, start_pos: int) -> np.ndarray:345 letters = self.feature_registry.feature_to_letters(fid, length=5)346 vec = np.zeros(DIMS, dtype=np.float32)347 for i, ch in enumerate(letters):348 base = self.letter_vec.get(ch)349 # PHASE 1.3: Weight by letter importance350 importance_weight = self.letter_vec.get_importance(ch)351 vec += np.sin(base + (start_pos + i) * POSITION_OFFSET) * importance_weight352 return vec353 354 def _update_vector(self):355 vec = np.zeros(DIMS, dtype=np.float32)356 pos = 0357 for fid in self.physical_features:358 vec += self._encode_feature(fid, pos)359 pos += 5360 for fid in self.semantic_features:361 vec += self._encode_feature(fid, pos)362 pos += 5363 norm = np.linalg.norm(vec)364 if norm > 0:365 vec /= norm366 self.vector = vec367 # PHASE 1.4: Apply mass scaling368 self.apply_scale()369 370 # PHASE 1.4: Vector Scaling by Mass371 def apply_scale(self):372 """Normalize vector but scale its length by its importance/mass."""373 if self.vector is None:374 return375 norm = np.linalg.norm(self.vector)376 if norm > 0:377 target_scale = np.log1p(self.importance)378 self.vector = (self.vector / norm) * target_scale379 380 # PHASE 1.5: Inertia-Based Movement 381 def move_towards(self, other: 'DNAConcept', lr: float = LR_CONCEPT, 382 weight: float = 1.0, color: str = "RELATED"):383 """384 Core innovation: move concept vectors closer with inertia.385 Heavier nodes (higher importance) move less.386 """387 if self.vector is None or other.vector is None:388 return389 390 # Calculate relative inertia (PHASE 1.5)391 pull_force = (other.importance / (self.importance + 1e-8)) * lr * weight392 other_pull_force = (self.importance / (other.importance + 1e-8)) * lr * weight393 394 # Color-based movement multiplier (PHASE 1.2)395 color_multipliers = {396 "IS_A": 1.2, # Stronger pull for essential relationships397 "HAS": 0.8,398 "GROWN_IN": 0.9,399 "LOCATION": 0.7,400 "CAUSES": 1.0,401 "PART_OF": 0.85,402 "RELATED": 0.5,403 "OPERATOR": 1.5, # Strong pull for instruction operators404 "CONDITION": 1.3405 }406 color_mult = color_multipliers.get(color, 0.5)407 pull_force *= color_mult408 other_pull_force *= color_mult409 410 # Store original vectors for gradient calculation411 orig_self = self.vector.copy()412 orig_other = other.vector.copy()413 414 # Apply movement415 diff = other.vector - self.vector416 self.vector += pull_force * diff417 other.vector -= other_pull_force * diff418 419 # Normalize420 self.vector /= (np.linalg.norm(self.vector) + 1e-8)421 other.vector /= (np.linalg.norm(other.vector) + 1e-8)422 423 # PHASE 1.4: Re-apply scale to preserve mass424 self.apply_scale()425 other.apply_scale()426 427 # Backpropagate gradient with partitioned learning rates428 self._backpropagate_to_features(other, orig_self, orig_other, pull_force, other_pull_force, color)429 430 def _backpropagate_to_features(self, other: 'DNAConcept', 431 orig_self: np.ndarray, orig_other: np.ndarray,432 pull_force: float, other_pull_force: float,433 color: str):434 """PHASE 2.3: Partitioned backpropagation with layer-specific LRs."""435 gradient = other.vector - self.vector436 437 # Determine layer based on color438 layer_map = {439 "IS_A": PriorityLayer.DOMAIN,440 "PART_OF": PriorityLayer.DOMAIN,441 "HAS": PriorityLayer.CLUSTER,442 "GROWN_IN": PriorityLayer.CLUSTER,443 "LOCATION": PriorityLayer.CLUSTER,444 "RELATED": PriorityLayer.ATOMIC,445 "OPERATOR": PriorityLayer.UNIVERSAL,446 "CONDITION": PriorityLayer.UNIVERSAL447 }448 layer = layer_map.get(color, PriorityLayer.CLUSTER)449 450 for concept, force in [(self, pull_force), (other, other_pull_force)]:451 for fid in concept.physical_features + concept.semantic_features:452 letters = self.feature_registry.feature_to_letters(fid, length=5)453 for i, ch in enumerate(letters):454 base = self.letter_vec.get(ch)455 x = base + i * POSITION_OFFSET456 grad_sin = np.cos(x)457 458 # PHASE 2.2: Apply partitioned gradients459 # Essence dimensions (0-3) get DOMAIN-level updates460 essence_grad = grad_sin[ESSENCE_DIMS]461 identity_grad = grad_sin[IDENTITY_DIMS]462 temporal_grad = grad_sin[TEMPORAL_DIMS]463 464 full_grad = np.zeros(DIMS)465 full_grad[ESSENCE_DIMS] = essence_grad * LR_DOMAIN466 full_grad[IDENTITY_DIMS] = identity_grad * LR_CLUSTER467 full_grad[TEMPORAL_DIMS] = temporal_grad * LR_ATOMIC468 469 norm_grad = np.linalg.norm(full_grad) + 1e-8470 delta_f = force * 0.5 * gradient * full_grad / norm_grad471 delta_l = force * 0.5 * gradient * full_grad / norm_grad472 473 self.feature_registry.update_vector(fid, delta_f, layer)474 self.letter_vec.update(ch, delta_l, PriorityLayer.ATOMIC)475 476 # PHASE 2.2: Partitioned Similarity477 def partitioned_similarity(self, other: 'DNAConcept', partition: slice = None) -> float:478 """Calculate similarity using only specified dimensions."""479 if self.vector is None or other.vector is None:480 return 0.0481 if partition is not None:482 v1 = self.vector[partition]483 v2 = other.vector[partition]484 else:485 v1, v2 = self.vector, other.vector486 return float(np.dot(v1, v2) / (np.linalg.norm(v1) * np.linalg.norm(v2) + 1e-8))487 488 def cosine_similarity(self, other: 'DNAConcept') -> float:489 return self.partitioned_similarity(other)490 491 # PHASE 10.1: Relationship Strengthening492 def strengthen_relationship(self, other_name: str, increment: float = 0.1):493 """Increment importance when concepts co-occur."""494 self.importance = min(10.0, self.importance + increment * 0.1)495 496 def serialize(self) -> dict:497 return {498 "name": self.name,499 "physical_features": self.physical_features,500 "semantic_features": self.semantic_features,501 "vector": self.vector.tolist() if self.vector is not None else None,502 "importance": self.importance,503 "domain": self.domain,504 "cluster_id": self.cluster_id,505 "numeric_value": self.numeric_value506 }507 508 @classmethod509 def from_serialized(cls, data: dict, feature_registry, letter_vec):510 obj = cls(511 data["name"], 512 data["physical_features"], 513 data["semantic_features"], 514 feature_registry, 515 letter_vec,516 importance=data.get("importance", 1.0),517 domain=data.get("domain", "general"),518 numeric_value=data.get("numeric_value")519 )520 if data.get("vector") is not None:521 obj.vector = np.array(data["vector"], dtype=np.float32)522 obj.cluster_id = data.get("cluster_id")523 return obj524 525 526# PHASE 7: Decoder for Sentence Generation527 528class SentenceDecoder:529 """Generates natural language from DNA concept activations."""530 531 def __init__(self, concept_memory: 'ConceptMemory'):532 self.concept_memory = concept_memory533 534 def extract_weighted_features(self, concept_name: str, top_k: int = 5) -> List[Tuple[str, float, str]]:535 """PHASE 7.1: Extract top related concepts sorted by weight."""536 if concept_name not in self.concept_memory.concepts:537 return []538 539 relationships = self.concept_memory.weighted_relationships.get(concept_name, {})540 if not relationships:541 return []542 543 # Sort by weight544 sorted_rels = sorted(545 relationships.items(),546 key=lambda x: x[1].weight,547 reverse=True548 )549 return [(other, rel.weight, rel.color) for other, rel in sorted_rels[:top_k]]550 551 def multi_hop_traversal(self, start: str, target_color: str, max_hops: int = 3) -> List[str]:552 """PHASE 7.2: Follow colored relationships."""553 visited = {start}554 path = [start]555 current = start556 557 for _ in range(max_hops):558 if current not in self.concept_memory.weighted_relationships:559 break560 # Find relationship with target color561 found = None562 for other, rel in self.concept_memory.weighted_relationships[current].items():563 if rel.color == target_color and other not in visited:564 found = other565 break566 if found is None:567 break568 path.append(found)569 visited.add(found)570 current = found571 572 return path573 574 def generate_description(self, concept_name: str, template: str = None) -> str:575 """PHASE 7.3: Template-based generation."""576 features = self.extract_weighted_features(concept_name, top_k=5)577 if not features:578 return f"{concept_name} is a concept."579 580 concept = self.concept_memory.concepts.get(concept_name)581 domain = concept.domain if concept else "general"582 583 # Find IS_A relationship584 is_a = next((f[0] for f in features if f[2] == "IS_A"), None)585 # Find HAS relationships586 has_features = [f[0] for f in features if f[2] == "HAS"][:2]587 # Find LOCATION588 location_path = self.multi_hop_traversal(concept_name, "LOCATION", max_hops=2)589 location = location_path[-1] if len(location_path) > 1 else None590 591 # Build sentence592 parts = [concept_name.capitalize()]593 if is_a:594 parts.append(f"is a {is_a}")595 if has_features:596 parts.append(f"with {', '.join(has_features)}")597 if location:598 parts.append(f"located in {location}")599 parts.append(f"(domain: {domain})")600 601 return " ".join(parts) + "."602 603 def self_correct(self, draft: str, concept_name: str) -> str:604 """PHASE 7.5: Self-correction using global consistency check."""605 # Check if draft contains known relationships606 features = self.extract_weighted_features(concept_name)607 feature_names = [f[0] for f in features]608 609 # Simple correction: ensure mentioned concepts are actually related610 draft_lower = draft.lower()611 for feat in feature_names:612 if feat.lower() not in draft_lower:613 # Could add missing important feature614 pass615 616 return draft617 618 619# INSTRUCTION DNA ENGINE (New!)620 621class InstructionEngine:622 """Executes deterministic operations using DNA concepts."""623 624 def __init__(self, concept_memory: 'ConceptMemory', feature_registry: FeatureRegistry, letter_vec: LetterVectors):625 self.memory = concept_memory626 self.feature_registry = feature_registry627 self.letter_vec = letter_vec628 self._ensure_operators()629 630 def _ensure_operators(self):631 """Pre-register arithmetic and logic operators."""632 operators = {633 "PLUS": {"domain": "operator", "importance": 10.0},634 "MINUS": {"domain": "operator", "importance": 10.0},635 "MULTIPLY": {"domain": "operator", "importance": 10.0},636 "DIVIDE": {"domain": "operator", "importance": 10.0},637 "EQUALS": {"domain": "operator", "importance": 10.0},638 "GREATER": {"domain": "operator", "importance": 10.0},639 "LESS": {"domain": "operator", "importance": 10.0},640 "IF": {"domain": "logic", "importance": 10.0},641 "THEN": {"domain": "logic", "importance": 10.0},642 "AND": {"domain": "logic", "importance": 10.0},643 "OR": {"domain": "logic", "importance": 10.0},644 "NOT": {"domain": "logic", "importance": 10.0},645 }646 for op, props in operators.items():647 op_low = op.lower()648 if op_low not in self.memory.concepts:649 physical = [self.feature_registry.register(op_low)]650 semantic = [self.feature_registry.register(op_low)]651 self.memory.register(op_low, physical, semantic, 652 importance=props["importance"], 653 domain=props["domain"])654 655 def get_or_create_number(self, value: float) -> DNAConcept:656 """Create a concept for a numeric value if it doesn't exist."""657 name = f"num_{value}".replace('.', '_').replace('-', 'neg')658 if name in self.memory.concepts:659 return self.memory.concepts[name]660 physical = [self.feature_registry.register(str(value))]661 semantic = [self.feature_registry.register(str(value))]662 concept = self.memory.register(name, physical, semantic, 663 importance=abs(value)/10 + 1.0, 664 domain="number",665 numeric_value=value)666 return concept667 668 def execute_arithmetic(self, operator: str, a: Union[str, float], b: Union[str, float]) -> DNAConcept:669 """Perform arithmetic and return result concept."""670 # Convert to concepts671 if isinstance(a, (int, float)):672 concept_a = self.get_or_create_number(float(a))673 else:674 concept_a = self.memory.concepts.get(a.lower())675 if concept_a is None:676 raise ValueError(f"Concept '{a}' not found")677 678 if isinstance(b, (int, float)):679 concept_b = self.get_or_create_number(float(b))680 else:681 concept_b = self.memory.concepts.get(b.lower())682 if concept_b is None:683 raise ValueError(f"Concept '{b}' not found")684 685 val_a = concept_a.numeric_value if concept_a.numeric_value is not None else concept_a.importance686 val_b = concept_b.numeric_value if concept_b.numeric_value is not None else concept_b.importance687 688 op_upper = operator.upper()689 if op_upper == "PLUS":690 result_val = val_a + val_b691 elif op_upper == "MINUS":692 result_val = val_a - val_b693 elif op_upper == "MULTIPLY":694 result_val = val_a * val_b695 elif op_upper == "DIVIDE":696 result_val = val_a / val_b if val_b != 0 else float('inf')697 else:698 raise ValueError(f"Unknown operator: {operator}")699 700 result_concept = self.get_or_create_number(result_val)701 702 # Create OPERATOR relationships703 self.memory.add_weighted_relationship(concept_a.name, concept_b.name, weight=0.9, color="OPERATOR")704 self.memory.add_weighted_relationship(operator.lower(), result_concept.name, weight=1.0, color="OPERATOR")705 706 return result_concept707 708 def evaluate_condition(self, condition_expr: Dict) -> bool:709 """Evaluate a logic condition (IF part)."""710 op = condition_expr.get("operator", "EQUALS").upper()711 left = condition_expr.get("left")712 right = condition_expr.get("right")713 714 if isinstance(left, str):715 left_conc = self.memory.concepts.get(left.lower())716 left_val = left_conc.numeric_value if left_conc and left_conc.numeric_value is not None else (left_conc.importance if left_conc else 0)717 else:718 left_val = float(left)719 720 if isinstance(right, str):721 right_conc = self.memory.concepts.get(right.lower())722 right_val = right_conc.numeric_value if right_conc and right_conc.numeric_value is not None else (right_conc.importance if right_conc else 0)723 else:724 right_val = float(right)725 726 if op == "EQUALS":727 return abs(left_val - right_val) < 1e-6728 elif op == "GREATER":729 return left_val > right_val730 elif op == "LESS":731 return left_val < right_val732 elif op == "AND":733 return bool(left_val) and bool(right_val)734 elif op == "OR":735 return bool(left_val) or bool(right_val)736 elif op == "NOT":737 return not bool(left_val)738 return False739 740# Reasoning Engine — Enhanced with Instruction DNA741 742class ReasoningEngine:743 def __init__(self, concept_memory: 'ConceptMemory', feature_registry: FeatureRegistry, letter_vec: LetterVectors):744 self.concept_memory = concept_memory745 self.decoder = SentenceDecoder(concept_memory)746 self.instruction_engine = InstructionEngine(concept_memory, feature_registry, letter_vec)747 748 def multi_hop_reasoning(self, start: str, max_hops: int = 3, decay: float = 0.7,749 color_filter: Optional[str] = None) -> Dict[str, float]:750 """PHASE 7.2: Color-filtered multi-hop reasoning."""751 if start not in self.concept_memory.weighted_relationships:752 return {}753 754 activation = {start: 1.0}755 for hop in range(max_hops):756 new_activation = {}757 for node, score in activation.items():758 relationships = self.concept_memory.weighted_relationships.get(node, {})759 for nb, rel in relationships.items():760 if color_filter and rel.color != color_filter:761 continue762 weight = rel.weight763 if node in self.concept_memory.concepts and nb in self.concept_memory.concepts:764 sim = self.concept_memory.concepts[node].cosine_similarity(765 self.concept_memory.concepts[nb]766 )767 weight *= (0.5 + 0.5 * sim)768 new_activation[nb] = new_activation.get(nb, 0) + score * weight * decay769 for k, v in new_activation.items():770 activation[k] = activation.get(k, 0) + v771 772 if activation:773 max_act = max(activation.values())774 activation = {k: v/max_act for k, v in activation.items()}775 return activation776 777 def analogical_reasoning(self, a: str, b: str, c: str, 778 partition: slice = None) -> List[str]:779 """A is to B as C is to ? with optional dimension partition."""780 if a not in self.concept_memory.concepts or b not in self.concept_memory.concepts:781 return []782 vec_a = self.concept_memory.concepts[a].vector783 vec_b = self.concept_memory.concepts[b].vector784 785 if partition is not None:786 direction = vec_b[partition] - vec_a[partition]787 else:788 direction = vec_b - vec_a789 790 if c not in self.concept_memory.concepts:791 return []792 vec_c = self.concept_memory.concepts[c].vector793 794 if partition is not None:795 target = vec_c.copy()796 target[partition] = vec_c[partition] + direction797 else:798 target = vec_c + direction799 800 target /= (np.linalg.norm(target) + 1e-8)801 results = self.concept_memory.partitioned_search(target, top_k=5, partition=partition)802 return [r[0] for r in results if r[0] not in (a, b, c)]803 804 def generate_sentence(self, concept: str) -> str:805 """Generate a descriptive sentence for a concept."""806 return self.decoder.generate_description(concept)807 808 # New instruction methods809 def calculate(self, expression: str) -> Dict:810 """Evaluate arithmetic expression using Instruction DNA."""811 import re812 allowed = set('0123456789.+-*/() ')813 if not all(c in allowed for c in expression):814 return {"error": "Invalid characters in expression"}815 try:816 result = eval(expression)817 concept = self.instruction_engine.get_or_create_number(result)818 return {"expression": expression, "result": result, "concept": concept.name}819 except Exception as e:820 return {"error": str(e)}821 822 def execute_instruction(self, operator: str, a: Union[str, float], b: Union[str, float]) -> Dict:823 """Execute a single arithmetic operation."""824 try:825 result_concept = self.instruction_engine.execute_arithmetic(operator, a, b)826 return {827 "operator": operator,828 "operands": [a, b],829 "result": result_concept.numeric_value,830 "concept": result_concept.name831 }832 except Exception as e:833 return {"error": str(e)}834 835 def evaluate_rule(self, condition: Dict, action: str) -> Dict:836 """IF-THEN rule evaluation."""837 cond_result = self.instruction_engine.evaluate_condition(condition)838 return {"condition_true": cond_result, "action_triggered": action if cond_result else None}839 840 841# Concept Memory — PHASE 1-4: All Upgrades842 843class ConceptMemory:844 def __init__(self, feature_registry: FeatureRegistry, letter_vec: LetterVectors,845 max_concepts: int = MAX_CONCEPTS):846 self.feature_registry = feature_registry847 self.letter_vec = letter_vec848 self.concepts: Dict[str, DNAConcept] = {}849 850 # PHASE 1.1, 1.2: Weighted and colored relationships851 self.weighted_relationships: Dict[str, Dict[str, RelationshipData]] = defaultdict(dict)852 853 # Legacy compatibility854 self.relationships: Dict[str, Set[str]] = defaultdict(set)855 856 self.max_concepts = max_concepts857 858 # PHASE 4.1: Global centroid tracking859 self.global_centroid: Optional[np.ndarray] = None860 self.batch_counter = 0861 862 # PHASE 4.3: Pending updates buffer863 self.pending_updates: List[Tuple[str, str, float, str]] = []864 865 # FAISS866 self._faiss_available = False867 self.index = None868 self.id_to_name: Dict[int, str] = {}869 self.name_to_id: Dict[str, int] = {}870 self.next_id = 0871 872 # PHASE 4.5: Hierarchical FAISS873 self.cluster_index = None874 self.cluster_to_concepts: Dict[int, List[str]] = defaultdict(list)875 876 try:877 import faiss878 self._faiss_available = True879 except Exception:880 self._faiss_available = False881 882 # PHASE 4.1: Global Context 883 def update_global_centroid(self):884 """Calculate the center of gravity of all concepts."""885 if not self.concepts:886 return887 all_vecs = np.array([c.vector for c in self.concepts.values() if c.vector is not None])888 if len(all_vecs) > 0:889 self.global_centroid = np.mean(all_vecs, axis=0)890 891 def get_global_context(self) -> dict:892 """PHASE 4.2: Return global centroid and top anchor nodes."""893 if self.global_centroid is None:894 self.update_global_centroid()895 896 # Find top 5 most important nodes897 top_nodes = sorted(898 self.concepts.values(), 899 key=lambda x: x.importance, 900 reverse=True901 )[:5]902 903 return {904 "global_centroid": self.global_centroid.tolist() if self.global_centroid is not None else None,905 "anchors": [n.name for n in top_nodes],906 "total_concepts": len(self.concepts)907 }908 909 # PHASE 4.4: Global Consistency Check910 def validate_against_global(self, concept_vector: np.ndarray, threshold: float = 2.0) -> bool:911 """Check if a vector is within reasonable distance from global centroid."""912 if self.global_centroid is None:913 return True914 distance = np.linalg.norm(concept_vector - self.global_centroid)915 std = np.std([c.vector for c in self.concepts.values()]) if self.concepts else 1.0916 return distance < threshold * std917 918 def _ensure_index(self):919 if not self._faiss_available:920 return921 if self.index is None:922 import faiss923 self.index = faiss.IndexFlatIP(DIMS)924 925 def _rebuild_index(self):926 if not self._faiss_available or not self.concepts:927 return928 import faiss929 vectors = [c.vector for c in self.concepts.values() if c.vector is not None]930 names = [name for name, c in self.concepts.items() if c.vector is not None]931 if not vectors:932 return933 vecs = np.vstack(vectors).astype(np.float32)934 self.index = faiss.IndexFlatIP(DIMS)935 self.index.add(vecs)936 self.id_to_name = {i: n for i, n in enumerate(names)}937 self.name_to_id = {n: i for i, n in enumerate(names)}938 self.next_id = len(self.concepts)939 940 # PHASE 4.1: Update global centroid after rebuild941 self.update_global_centroid()942 943 def register(self, name: str, physical_features: List[int], semantic_features: List[int],944 importance: float = 1.0, domain: str = "general", numeric_value: Optional[float] = None) -> DNAConcept:945 name_low = name.lower()946 if name_low in self.concepts:947 # PHASE 10.2: Strengthen on re-registration948 self.concepts[name_low].importance = min(10.0, self.concepts[name_low].importance + 0.1)949 return self.concepts[name_low]950 951 concept = DNAConcept(name_low, physical_features, semantic_features,952 self.feature_registry, self.letter_vec,953 importance=importance, domain=domain, numeric_value=numeric_value)954 self.concepts[name_low] = concept955 956 if self._faiss_available and concept.vector is not None:957 self._ensure_index()958 if self.index is not None:959 self.index.add(concept.vector.reshape(1, -1))960 self.id_to_name[self.next_id] = name_low961 self.name_to_id[name_low] = self.next_id962 self.next_id += 1963 964 self._prune()965 return concept966 967 # PHASE 1.1, 1.2: Weighted & Colored Relationships968 def add_weighted_relationship(self, a: str, b: str, weight: float = 1.0, 969 color: str = "RELATED"):970 a_low, b_low = a.lower(), b.lower()971 if a_low not in self.concepts or b_low not in self.concepts:972 return973 974 # Create or update relationship data975 if b_low not in self.weighted_relationships[a_low]:976 self.weighted_relationships[a_low][b_low] = RelationshipData(977 weight=weight, 978 color=color,979 co_occurrence_count=1980 )981 else:982 rel = self.weighted_relationships[a_low][b_low]983 rel.weight = min(1.0, rel.weight + 0.1 * weight)984 rel.co_occurrence_count += 1985 rel.last_accessed = time.time()986 987 # Symmetric relationship988 if a_low not in self.weighted_relationships[b_low]:989 self.weighted_relationships[b_low][a_low] = RelationshipData(990 weight=weight,991 color=color,992 co_occurrence_count=1993 )994 else:995 rel = self.weighted_relationships[b_low][a_low]996 rel.weight = min(1.0, rel.weight + 0.1 * weight)997 rel.co_occurrence_count += 1998 rel.last_accessed = time.time()999 1000 # Legacy compatibility1001 self.relationships[a_low].add(b_low)1002 self.relationships[b_low].add(a_low)1003 1004 # Move vectors with inertia1005 self.concepts[a_low].move_towards(self.concepts[b_low], 1006 lr=LR_CONCEPT, 1007 weight=weight, 1008 color=color)1009 1010 def add_relationship(self, a: str, b: str):1011 """Legacy method for backward compatibility."""1012 self.add_weighted_relationship(a, b, weight=1.0, color="RELATED")1013 1014 # PHASE 10.2: Strengthen on Co-occurrence1015 def strengthen_relationship(self, a: str, b: str, increment: float = 0.1):1016 a_low, b_low = a.lower(), b.lower()1017 if a_low in self.weighted_relationships and b_low in self.weighted_relationships[a_low]:1018 rel = self.weighted_relationships[a_low][b_low]1019 rel.weight = min(1.0, rel.weight + increment)1020 rel.co_occurrence_count += 11021 rel.last_accessed = time.time()1022 1023 # Also strengthen symmetric1024 if a_low in self.weighted_relationships[b_low]:1025 rel2 = self.weighted_relationships[b_low][a_low]1026 rel2.weight = min(1.0, rel2.weight + increment)1027 rel2.co_occurrence_count += 11028 rel2.last_accessed = time.time()1029 1030 # PHASE 10.1: Relationship Decay1031 def apply_decay(self, decay_rate: float = 0.001, inactive_threshold: float = 86400):1032 """Decay relationships that haven't been accessed recently."""1033 current_time = time.time()1034 for a, rels in self.weighted_relationships.items():1035 to_remove = []1036 for b, rel in rels.items():1037 if current_time - rel.last_accessed > inactive_threshold:1038 rel.weight = max(0.05, rel.weight - decay_rate)1039 if rel.weight <= 0.06:1040 to_remove.append(b)1041 for b in to_remove:1042 del self.weighted_relationships[a][b]1043 if b in self.relationships[a]:1044 self.relationships[a].remove(b)1045 1046 # PHASE 2.2: Partitioned Search1047 def partitioned_search(self, query_vector: np.ndarray, top_k: int = 5,1048 partition: slice = None) -> List[Tuple[str, float]]:1049 """Search using only specified dimension partition."""1050 if not self.concepts:1051 return []1052 1053 if partition is not None:1054 q = query_vector[partition]1055 vecs = np.vstack([self.concepts[n].vector[partition] for n in self.concepts.keys()])1056 else:1057 q = query_vector1058 vecs = np.vstack([self.concepts[n].vector for n in self.concepts.keys()])1059 1060 names = list(self.concepts.keys())1061 q = q / (np.linalg.norm(q) + 1e-8)1062 vecs_norm = vecs / (np.linalg.norm(vecs, axis=1, keepdims=True) + 1e-8)1063 scores = vecs_norm @ q1064 top = np.argsort(scores)[::-1][:top_k]1065 return [(names[i], float(scores[i])) for i in top]1066 1067 def search(self, query_vector: np.ndarray, top_k: int = 5) -> List[Tuple[str, float]]:1068 """Full-vector search."""1069 return self.partitioned_search(query_vector, top_k, partition=None)1070 1071 # PHASE 4.3: Batch Processing 1072 def add_to_batch(self, a: str, b: str, weight: float, color: str):1073 """Add relationship update to pending batch."""1074 self.pending_updates.append((a, b, weight, color))1075 self.batch_counter += 11076 1077 if len(self.pending_updates) >= BATCH_SIZE:1078 self.process_batch()1079 1080 def process_batch(self):1081 """Process all pending updates and update global statistics."""1082 if not self.pending_updates:1083 return1084 1085 for a, b, weight, color in self.pending_updates:1086 if a in self.concepts and b in self.concepts:1087 self.concepts[a].move_towards(self.concepts[b], 1088 lr=LR_CONCEPT, 1089 weight=weight, 1090 color=color)1091 1092 self.pending_updates.clear()1093 1094 # PHASE 4.1: Update global centroid periodically1095 if self.batch_counter % GLOBAL_UPDATE_FREQUENCY == 0:1096 self.update_global_centroid()1097 1098 self._rebuild_index()1099 1100 async def extract_and_link(self, text: str, ontology: DynamicOntology, 1101 sector: str = "general") -> List[str]:1102 words = [w for w in text.lower().split() if len(w) > 3 and w not in STOP_WORDS]1103 unique = list(set(words))[:15]1104 concept_list = []1105 1106 for kw in unique:1107 features = await ontology.get_features_llm(kw) if ontology.llm_enabled else (ontology.get_features(kw), "general")1108 if isinstance(features, tuple):1109 features, domain = features1110 else:1111 domain = "general"1112 1113 physical_fids = [self.feature_registry.register(f) for f in features]1114 semantic_fids = [self.feature_registry.register(f) for f in features]1115 1116 # PHASE 1.3: Initial importance based on word frequency in corpus1117 importance = 1.0 + (0.1 * unique.index(kw) if kw in unique else 0)1118 1119 concept = self.register(kw, physical_fids, semantic_fids, 1120 importance=importance, domain=domain)1121 concept_list.append(kw)1122 1123 # Create relationships with colors inferred from context1124 for i in range(len(unique)):1125 for j in range(i+1, min(i+4, len(unique))):1126 # Infer relationship color from text context1127 color = self._infer_relationship_color(text, unique[i], unique[j])1128 self.add_weighted_relationship(unique[i], unique[j], weight=0.8, color=color)1129 1130 return concept_list1131 1132 def _infer_relationship_color(self, text: str, a: str, b: str) -> str:1133 """Infer relationship type from context."""1134 text_lower = text.lower()1135 if f"{a} is {b}" in text_lower or f"{b} is {a}" in text_lower:1136 return "IS_A"1137 elif f"{a} has {b}" in text_lower or f"{b} has {a}" in text_lower:1138 return "HAS"1139 elif f"{a} in {b}" in text_lower or f"{b} in {a}" in text_lower:1140 return "LOCATION"1141 elif "cause" in text_lower and (a in text_lower or b in text_lower):1142 return "CAUSES"1143 return "RELATED"1144 1145 def _prune(self):1146 if len(self.concepts) > self.max_concepts:1147 # Sort by importance (keep high importance concepts)1148 sorted_concepts = sorted(1149 self.concepts.items(), 1150 key=lambda x: (x[1].importance, len(self.weighted_relationships.get(x[0], {}))),1151 reverse=True1152 )1153 to_keep = sorted_concepts[:self.max_concepts]1154 self.concepts = {name: concept for name, concept in to_keep}1155 1156 # Clean up relationships for removed concepts1157 keep_names = set(self.concepts.keys())1158 self.weighted_relationships = defaultdict(dict, {1159 k: {b: r for b, r in v.items() if b in keep_names}1160 for k, v in self.weighted_relationships.items() if k in keep_names1161 })1162 self.relationships = defaultdict(set, {1163 k: v.intersection(keep_names) 1164 for k, v in self.relationships.items() if k in keep_names1165 })1166 1167 self._rebuild_index()1168 1169 def serialize(self) -> dict:1170 return {1171 "concepts": {name: c.serialize() for name, c in self.concepts.items()},1172 "weighted_relationships": {1173 k: {b: r.to_dict() for b, r in v.items()}1174 for k, v in self.weighted_relationships.items()1175 },1176 "relationships": {k: list(v) for k, v in self.relationships.items()},1177 "global_centroid": self.global_centroid.tolist() if self.global_centroid is not None else None1178 }1179 1180 def restore(self, data: dict):1181 self.concepts = {}1182 self.weighted_relationships = defaultdict(dict)1183 self.relationships = defaultdict(set)1184 1185 for name, cdata in data.get("concepts", {}).items():1186 self.concepts[name] = DNAConcept.from_serialized(cdata, self.feature_registry, self.letter_vec)1187 1188 for k, vdict in data.get("weighted_relationships", {}).items():1189 for b, rdata in vdict.items():1190 self.weighted_relationships[k][b] = RelationshipData.from_dict(rdata)1191 self.relationships[k].add(b)1192 1193 for k, vlist in data.get("relationships", {}).items():1194 self.relationships[k].update(vlist)1195 1196 if data.get("global_centroid"):1197 self.global_centroid = np.array(data["global_centroid"], dtype=np.float32)1198 1199 self._rebuild_index()1200 