diff --git a/4-Infrastructure/infra/adaptive_delta_gcl.py b/4-Infrastructure/infra/adaptive_delta_gcl.py deleted file mode 100644 index 7aa70ce5..00000000 --- a/4-Infrastructure/infra/adaptive_delta_gcl.py +++ /dev/null @@ -1,350 +0,0 @@ -#!/usr/bin/env python3 -""" -Adaptive Delta GCL Compression - -Adaptive compression system that analyzes data patterns and selects optimal -compression strategies based on historical performance. - -Features: -- Pattern detection in metadata -- Adaptive strategy selection -- Performance tracking -- Automatic strategy optimization - -AUDIT-ONLY SHIM: -This script is not a source-of-truth implementation under docs/AGENTS.md. -It may generate exploratory JSON/audit evidence only. Any curvature, -compression strategy, invariant, or branching logic here must be ported to -Lean before being used by the core model or release claims. -""" - -import sys -from pathlib import Path -# Add project root to Python path -project_root = Path(__file__).parent.parent -sys.path.insert(0, str(project_root)) - -import json -import hashlib -from typing import Dict, Any, List, Optional, Tuple -from dataclasses import dataclass, field -from collections import defaultdict, deque -from datetime import datetime -from enum import Enum - -from infra.delta_gcl_compression_service import DeltaGCLCompressionService, CompressionResult - - -class CompressionStrategy(Enum): - """Compression strategy types.""" - DELTA_ONLY = "delta_only" - PTOS_ONLY = "ptos_only" - GCL_ONLY = "gcl_only" - DELTA_PTOS = "delta_ptos" - DELTA_GCL = "delta_gcl" - FULL_STACK = "full_stack" - ADAPTIVE = "adaptive" - - -@dataclass -class PatternFeatures: - """Features extracted from data patterns.""" - field_change_rate: float # Rate at which fields change - value_variance: float # Variance in field values - sequence_length: int # Length of data sequence - entropy: float # Shannon entropy of data - temporal_correlation: float # Correlation with previous data - - -@dataclass -class StrategyPerformance: - """Performance metrics for a compression strategy.""" - strategy: CompressionStrategy - avg_compression_ratio: float - avg_compression_time: float - success_rate: float - total_uses: int - last_used: float - - -class PatternAnalyzer: - """Analyzes data patterns to guide compression strategy selection.""" - - def __init__(self, history_size: int = 100): - self.history_size = history_size - self.pattern_history: deque = deque(maxlen=history_size) - - def extract_features(self, manifest: Dict[str, Any], - previous: Optional[Dict[str, Any]] = None) -> PatternFeatures: - """Extract pattern features from manifest.""" - # Field change rate - if previous: - changed_fields = sum(1 for k in manifest.keys() if k in previous and manifest[k] != previous[k]) - field_change_rate = changed_fields / len(manifest) - else: - field_change_rate = 1.0 # All fields "changed" (no previous) - - # Value variance (simplified) - values = list(manifest.values()) - if values and isinstance(values[0], (int, float)): - value_variance = (max(values) - min(values)) / (len(values) or 1) - else: - value_variance = 0.5 - - # Sequence length - sequence_length = len(json.dumps(manifest)) - - # Entropy (simplified character distribution) - manifest_str = json.dumps(manifest) - char_counts = defaultdict(int) - for char in manifest_str: - char_counts[char] += 1 - if manifest_str: - # Shannon entropy: -sum(p * log2(p)) - import math - entropy = -sum((count / len(manifest_str)) * math.log2(count / len(manifest_str)) - for count in char_counts.values()) - else: - entropy = 0 - - # Temporal correlation (simplified) - if previous: - prev_str = json.dumps(previous) - common_chars = sum(1 for c in set(manifest_str + prev_str) if c in manifest_str and c in prev_str) - temporal_correlation = common_chars / len(set(manifest_str + prev_str)) - else: - temporal_correlation = 0.0 - - return PatternFeatures( - field_change_rate=field_change_rate, - value_variance=value_variance, - sequence_length=sequence_length, - entropy=entropy, - temporal_correlation=temporal_correlation - ) - - def predict_best_strategy(self, features: PatternFeatures) -> CompressionStrategy: - """Predict best compression strategy based on patterns.""" - # Simple heuristic-based strategy selection - if features.field_change_rate < 0.2: - # Low change rate: delta encoding very effective - return CompressionStrategy.DELTA_PTOS - elif features.temporal_correlation > 0.8: - # High correlation: delta encoding effective - return CompressionStrategy.DELTA_GCL - elif features.entropy < 0.5: - # Low entropy: GCL compression effective - return CompressionStrategy.GCL_ONLY - else: - # Default: full stack - return CompressionStrategy.FULL_STACK - - -class AdaptiveDeltaGCLCompressor: - """ - Adaptive compression system that learns optimal strategies from data patterns. - """ - - def __init__(self): - self.compression_service = DeltaGCLCompressionService() - self.pattern_analyzer = PatternAnalyzer() - self.strategy_performance: Dict[CompressionStrategy, StrategyPerformance] = {} - self.previous_manifests: Dict[str, Dict[str, Any]] = {} - self.compression_history: List[Tuple[PatternFeatures, CompressionStrategy, CompressionResult]] = [] - - def compress_adaptive(self, manifest: Dict[str, Any], - manifest_id: str) -> CompressionResult: - """ - Compress manifest using adaptive strategy selection. - """ - # Get previous manifest for pattern analysis - previous = self.previous_manifests.get(manifest_id) - - # Extract pattern features - features = self.pattern_analyzer.extract_features(manifest, previous) - - # Predict best strategy - strategy = self.pattern_analyzer.predict_best_strategy(features) - - # Apply strategy - result = self._apply_strategy(manifest, manifest_id, strategy, previous) - - # Track performance - self._track_performance(features, strategy, result) - - # Store manifest for future delta encoding - self.previous_manifests[manifest_id] = manifest - - return result - - def _apply_strategy(self, manifest: Dict[str, Any], manifest_id: str, - strategy: CompressionStrategy, - previous: Optional[Dict[str, Any]] = None) -> CompressionResult: - """Apply specific compression strategy.""" - use_delta = previous is not None - - match strategy: - case CompressionStrategy.DELTA_ONLY: - # Delta encoding only - return self.compression_service.compress_manifest(manifest, manifest_id, use_delta=use_delta) - case CompressionStrategy.PTOS_ONLY: - # PTOS dictionary only (no delta) - return self.compression_service.compress_manifest(manifest, manifest_id, use_delta=False) - case CompressionStrategy.GCL_ONLY: - # GCL only (no delta, no PTOS) - # For now, use standard compression - return self.compression_service.compress_manifest(manifest, manifest_id, use_delta=False) - case CompressionStrategy.DELTA_PTOS: - # Delta + PTOS - return self.compression_service.compress_manifest(manifest, manifest_id, use_delta=use_delta) - case CompressionStrategy.DELTA_GCL: - # Delta + GCL - return self.compression_service.compress_manifest(manifest, manifest_id, use_delta=use_delta) - case CompressionStrategy.FULL_STACK: - # Full stack (delta + PTOS + GCL) - return self.compression_service.compress_manifest(manifest, manifest_id, use_delta=use_delta) - case CompressionStrategy.ADAPTIVE: - # Let the service decide - return self.compression_service.compress_manifest(manifest, manifest_id, use_delta=use_delta) - - def _track_performance(self, features: PatternFeatures, - strategy: CompressionStrategy, - result: CompressionResult): - """Track performance of compression strategies.""" - # Update strategy performance metrics - if strategy not in self.strategy_performance: - self.strategy_performance[strategy] = StrategyPerformance( - strategy=strategy, - avg_compression_ratio=0.0, - avg_compression_time=0.0, - success_rate=1.0, - total_uses=0, - last_used=0.0 - ) - - perf = self.strategy_performance[strategy] - perf.total_uses += 1 - perf.last_used = datetime.now().timestamp() - - # Update average compression ratio - current_ratio = result.stats["reduction_percent"] / 100 - perf.avg_compression_ratio = ( - (perf.avg_compression_ratio * (perf.total_uses - 1) + current_ratio) - / perf.total_uses - ) - - # Store in history - self.compression_history.append((features, strategy, result)) - - # Limit history size - if len(self.compression_history) > 1000: - self.compression_history = self.compression_history[-1000:] - - def get_performance_report(self) -> Dict[str, Any]: - """Get performance report for all strategies.""" - report = { - "total_compressions": len(self.compression_history), - "strategies": {} - } - - for strategy, perf in self.strategy_performance.items(): - report["strategies"][strategy.value] = { - "avg_compression_ratio": perf.avg_compression_ratio, - "avg_compression_time": perf.avg_compression_time, - "success_rate": perf.success_rate, - "total_uses": perf.total_uses, - "last_used": perf.last_used - } - - return report - - def optimize_strategies(self): - """Optimize strategy selection based on historical performance.""" - if not self.compression_history: - return - - # Find best performing strategy overall - best_strategy = max( - self.strategy_performance.items(), - key=lambda x: x[1].avg_compression_ratio - ) - - print(f"[Adaptive] Best performing strategy: {best_strategy[0].value}") - print(f"[Adaptive] Average compression ratio: {best_strategy[1].avg_compression_ratio:.2%}") - - -# Singleton instance -_adaptive_instance: Optional[AdaptiveDeltaGCLCompressor] = None - - -def get_adaptive_compressor() -> AdaptiveDeltaGCLCompressor: - """Get singleton adaptive compressor instance.""" - global _adaptive_instance - if _adaptive_instance is None: - _adaptive_instance = AdaptiveDeltaGCLCompressor() - return _adaptive_instance - - -def compress_adaptive_manifest(manifest: Dict[str, Any], manifest_id: str) -> CompressionResult: - """Convenience function to compress manifest with adaptive strategy.""" - compressor = get_adaptive_compressor() - return compressor.compress_adaptive(manifest, manifest_id) - - -if __name__ == "__main__": - # Test adaptive compression - print("=" * 70) - print("ADAPTIVE DELTA GCL COMPRESSION") - print("=" * 70) - - compressor = AdaptiveDeltaGCLCompressor() - - # Create test manifests - manifest1 = { - "layer": "CORE", - "domain": "COMPUTE", - "tier": "FOAM", - "condition": "STABLE", - "data": {"value": 100, "status": "active"} - } - - print("\n[1] Compressing first manifest...") - result1 = compressor.compress_adaptive(manifest1, "adaptive_test_1") - print(f" Compression: {result1.stats['reduction_percent']:.2f}% reduction") - print(f" Strategy: Adaptive") - - # Second manifest (similar to first - should use delta) - manifest2 = manifest1.copy() - manifest2["data"]["value"] = 105 - - print("\n[2] Compressing second manifest (similar to first)...") - result2 = compressor.compress_adaptive(manifest2, "adaptive_test_1") - print(f" Compression: {result2.stats['reduction_percent']:.2f}% reduction") - - # Third manifest (different - should use different strategy) - manifest3 = { - "layer": "CARRY", - "domain": "TOKEN", - "tier": "PLASMA", - "condition": "EXPERIMENTAL", - "data": {"value": 200, "status": "pending", "priority": "high"} - } - - print("\n[3] Compressing third manifest (different)...") - result3 = compressor.compress_adaptive(manifest3, "adaptive_test_2") - print(f" Compression: {result3.stats['reduction_percent']:.2f}% reduction") - - # Optimize strategies - print("\n[4] Optimizing strategies...") - compressor.optimize_strategies() - - # Performance report - print("\n[5] Performance report...") - report = compressor.get_performance_report() - print(f" Total compressions: {report['total_compressions']}") - for strategy, metrics in report['strategies'].items(): - print(f" {strategy}: {metrics['total_uses']} uses, {metrics['avg_compression_ratio']:.2%} avg ratio") - - print("\n" + "=" * 70) - print("ADAPTIVE COMPRESSION OPERATIONAL") - print("=" * 70) diff --git a/4-Infrastructure/infra/ascii_art_competition.py b/4-Infrastructure/infra/ascii_art_competition.py deleted file mode 100644 index 4836f653..00000000 --- a/4-Infrastructure/infra/ascii_art_competition.py +++ /dev/null @@ -1,369 +0,0 @@ -#!/usr/bin/env python3 -""" -Swarm Competition Integration for ASCII Art - -Integrates ASCII art store with the swarm competition system to spur -competition for better ASCII art generation, style classification, -semantic matching, and ranking. - -Competition Areas: -1. ASCII Art Generation - agents compete to generate higher-quality art -2. Style Classification - agents compete for better style detection -3. Semantic Matching - agents compete for better vector representations -4. Ranking and Curation - agents compete to rank art by quality -""" - -import sys -import json -import sqlite3 -from pathlib import Path -from typing import Dict, List, Optional, Any -from dataclasses import dataclass -from enum import Enum -import hashlib - -sys.path.insert(0, str(Path(__file__).parent.parent)) - -from infra.ascii_art_store import AsciiArtStore, AsciiArtEntry -from infra.ene_api import ENEAPIHook, AccessLevel - - -class CompetitionType(Enum): - """Types of ASCII art competitions""" - GENERATION = "generation" - STYLE_CLASSIFICATION = "style_classification" - SEMANTIC_MATCHING = "semantic_matching" - RANKING = "ranking" - - -@dataclass -class CompetitionEntry: - """Entry in ASCII art competition""" - agent_id: str - competition_type: CompetitionType - ascii_art_id: str - score: float - metrics: Dict[str, float] - timestamp: int - proposal: str - - -class AsciiArtCompetition: - """Swarm competition manager for ASCII art""" - - def __init__(self, db_path: str = "/home/allaun/Documents/Research Stack/data/substrate_index.db"): - self.db_path = db_path - self.ascii_store = AsciiArtStore() - self.ene_api = ENEAPIHook() - self._init_competition_tables() - - def _init_competition_tables(self): - """Initialize competition-specific tables""" - conn = sqlite3.connect(self.db_path, timeout=30.0) - cursor = conn.cursor() - cursor.execute("PRAGMA journal_mode=WAL") - cursor.execute("PRAGMA synchronous=NORMAL") - - # Competition entries table - cursor.execute(""" - CREATE TABLE IF NOT EXISTS ascii_art_competition ( - id TEXT PRIMARY KEY, - agent_id TEXT NOT NULL, - competition_type TEXT NOT NULL, - ascii_art_id TEXT, - score REAL NOT NULL, - metrics TEXT NOT NULL, - proposal TEXT, - timestamp INTEGER NOT NULL, - approved BOOLEAN DEFAULT FALSE - ) - """) - - # Leaderboard table - cursor.execute(""" - CREATE TABLE IF NOT EXISTS ascii_art_leaderboard ( - agent_id TEXT PRIMARY KEY, - total_score REAL NOT NULL, - competitions_won INTEGER DEFAULT 0, - entries_count INTEGER DEFAULT 0, - last_updated INTEGER NOT NULL - ) - """) - - # Competition metrics tracking - cursor.execute(""" - CREATE TABLE IF NOT EXISTS ascii_art_metrics ( - metric_name TEXT PRIMARY KEY, - current_best REAL, - best_agent_id TEXT, - timestamp INTEGER NOT NULL - ) - """) - - conn.commit() - conn.close() - - def submit_competition_entry(self, entry: CompetitionEntry) -> bool: - """Submit an entry to the competition""" - try: - entry_id = f"comp_{entry.competition_type.value}_{entry.agent_id}_{int(hashlib.sha256(entry.agent_id.encode()).hexdigest()[:8], 16)}" - - conn = sqlite3.connect(self.db_path, timeout=30.0) - cursor = conn.cursor() - - cursor.execute(""" - INSERT INTO ascii_art_competition - (id, agent_id, competition_type, ascii_art_id, score, metrics, proposal, timestamp) - VALUES (?, ?, ?, ?, ?, ?, ?, ?) - """, ( - entry_id, - entry.agent_id, - entry.competition_type.value, - entry.ascii_art_id, - entry.score, - json.dumps(entry.metrics), - entry.proposal, - entry.timestamp - )) - - # Update leaderboard - self._update_leaderboard(entry.agent_id, entry.score) - - # Update metrics tracking - for metric_name, metric_value in entry.metrics.items(): - self._update_metric(metric_name, metric_value, entry.agent_id) - - conn.commit() - conn.close() - - return True - except Exception as e: - print(f"Error submitting competition entry: {e}") - return False - - def _update_leaderboard(self, agent_id: str, score: float): - """Update agent leaderboard""" - conn = sqlite3.connect(self.db_path, timeout=30.0) - cursor = conn.cursor() - - cursor.execute(""" - INSERT INTO ascii_art_leaderboard (agent_id, total_score, competitions_won, entries_count, last_updated) - VALUES (?, 0, 0, 1, ?) - ON CONFLICT(agent_id) DO UPDATE SET - total_score = total_score + ?, - entries_count = entries_count + 1, - last_updated = ? - """, (agent_id, int(__import__('time').time()), score, int(__import__('time').time()))) - - conn.commit() - conn.close() - - def _update_metric(self, metric_name: str, value: float, agent_id: str): - """Update competition metrics tracking""" - conn = sqlite3.connect(self.db_path, timeout=30.0) - cursor = conn.cursor() - - cursor.execute(""" - INSERT INTO ascii_art_metrics (metric_name, current_best, best_agent_id, timestamp) - VALUES (?, ?, ?, ?) - ON CONFLICT(metric_name) DO UPDATE SET - current_best = CASE WHEN ? > current_best THEN ? ELSE current_best END, - best_agent_id = CASE WHEN ? > current_best THEN ? ELSE best_agent_id END, - timestamp = CASE WHEN ? > current_best THEN ? ELSE timestamp END - """, (metric_name, value, agent_id, int(__import__('time').time()), - value, value, value, agent_id, value, int(__import__('time').time()))) - - conn.commit() - conn.close() - - def evaluate_generation_quality(self, ascii_art: str) -> Dict[str, float]: - """Evaluate ASCII art generation quality metrics""" - lines = ascii_art.split('\n') - width = max(len(line) for line in lines) if lines else 0 - height = len(lines) - - # Quality metrics - aspect_ratio_score = 1.0 - abs(1.0 - (width / height)) if height > 0 else 0.0 - line_consistency = 1.0 if len(set(len(line) for line in lines)) == 1 else 0.5 - character_diversity = len(set(ascii_art)) / 95.0 # Normalized by printable ASCII - - return { - "aspect_ratio": aspect_ratio_score, - "line_consistency": line_consistency, - "character_diversity": min(character_diversity, 1.0), - "overall_quality": (aspect_ratio_score + line_consistency + character_diversity) / 3.0 - } - - def evaluate_style_classification(self, ascii_art: str, predicted_style: str, actual_style: str) -> float: - """Evaluate style classification accuracy""" - return 1.0 if predicted_style == actual_style else 0.0 - - def evaluate_semantic_similarity(self, text1: str, text2: str) -> float: - """Evaluate semantic similarity between two texts""" - # Simple character-level similarity - set1 = set(text1.lower()) - set2 = set(text2.lower()) - intersection = len(set1 & set2) - union = len(set1 | set2) - return intersection / union if union > 0 else 0.0 - - def get_leaderboard(self, limit: int = 10) -> List[Dict]: - """Get current competition leaderboard""" - try: - conn = sqlite3.connect(self.db_path, timeout=30.0) - cursor = conn.cursor() - - cursor.execute(""" - SELECT agent_id, total_score, competitions_won, entries_count - FROM ascii_art_leaderboard - ORDER BY total_score DESC - LIMIT ? - """, (limit,)) - - rows = cursor.fetchall() - conn.close() - - return [ - { - "agent_id": row[0], - "total_score": row[1], - "competitions_won": row[2], - "entries_count": row[3] - } - for row in rows - ] - except Exception as e: - print(f"Error getting leaderboard: {e}") - return [] - - def get_best_metrics(self) -> Dict[str, Any]: - """Get best metrics across all competitions""" - try: - conn = sqlite3.connect(self.db_path, timeout=30.0) - cursor = conn.cursor() - - cursor.execute("SELECT * FROM ascii_art_metrics") - rows = cursor.fetchall() - conn.close() - - return { - row[0]: { - "current_best": row[1], - "best_agent_id": row[2], - "timestamp": row[3] - } - for row in rows - } - except Exception as e: - print(f"Error getting best metrics: {e}") - return {} - - def approve_winner(self, competition_type: CompetitionType, agent_id: str) -> bool: - """Approve a competition winner and integrate their work""" - try: - conn = sqlite3.connect(self.db_path, timeout=30.0) - cursor = conn.cursor() - - # Get highest scoring entry for this agent and competition type - cursor.execute(""" - SELECT id FROM ascii_art_competition - WHERE competition_type = ? AND agent_id = ? - ORDER BY score DESC - LIMIT 1 - """, (competition_type.value, agent_id)) - - row = cursor.fetchone() - if row: - entry_id = row[0] - # Mark entry as approved - cursor.execute(""" - UPDATE ascii_art_competition - SET approved = TRUE - WHERE id = ? - """, (entry_id,)) - - # Increment competitions won - cursor.execute(""" - UPDATE ascii_art_leaderboard - SET competitions_won = competitions_won + 1 - WHERE agent_id = ? - """, (agent_id)) - - conn.commit() - conn.close() - - return True - except Exception as e: - print(f"Error approving winner: {e}") - return False - - -# Example usage -if __name__ == "__main__": - print("=" * 70) - print("ASCII ART SWARM COMPETITION TEST") - print("=" * 70) - - competition = AsciiArtCompetition() - - # Test 1: Submit competition entries - print("\n[Test 1] Submitting competition entries...") - - # Agent 1 entry - entry1 = CompetitionEntry( - agent_id="agent_alpha", - competition_type=CompetitionType.GENERATION, - ascii_art_id="test_smiley_001", - score=0.85, - metrics={ - "aspect_ratio": 0.9, - "line_consistency": 0.8, - "character_diversity": 0.85, - "overall_quality": 0.85 - }, - timestamp=int(__import__('time').time()), - proposal="Improved aspect ratio detection" - ) - - competition.submit_competition_entry(entry1) - - # Agent 2 entry - entry2 = CompetitionEntry( - agent_id="agent_beta", - competition_type=CompetitionType.GENERATION, - ascii_art_id="test_smiley_001", - score=0.78, - metrics={ - "aspect_ratio": 0.85, - "line_consistency": 0.75, - "character_diversity": 0.74, - "overall_quality": 0.78 - }, - timestamp=int(__import__('time').time()), - proposal="Optimized character selection" - ) - - competition.submit_competition_entry(entry2) - - # Test 2: Get leaderboard - print("\n[Test 2] Getting leaderboard...") - leaderboard = competition.get_leaderboard(limit=5) - print(json.dumps(leaderboard, indent=2)) - - # Test 3: Get best metrics - print("\n[Test 3] Getting best metrics...") - best_metrics = competition.get_best_metrics() - print(json.dumps(best_metrics, indent=2)) - - # Test 4: Approve winner - print("\n[Test 4] Approving winner...") - competition.approve_winner(CompetitionType.GENERATION, "agent_alpha") - - # Test 5: Updated leaderboard - print("\n[Test 5] Updated leaderboard...") - leaderboard = competition.get_leaderboard(limit=5) - print(json.dumps(leaderboard, indent=2)) - - print("\n" + "=" * 70) - print("ASCII ART SWARM COMPETITION INTEGRATION COMPLETE") - print("=" * 70) diff --git a/4-Infrastructure/infra/ascii_art_store.py b/4-Infrastructure/infra/ascii_art_store.py deleted file mode 100644 index 39779fd2..00000000 --- a/4-Infrastructure/infra/ascii_art_store.py +++ /dev/null @@ -1,411 +0,0 @@ -#!/usr/bin/env python3 -""" -ASCII Art Store Integration with ENE - -Integrates the Hugging Face dataset "Csplk/THE.ASCII.ART.EMPORIUM" -with the ENE ecosystem for secure storage and retrieval of ASCII art. - -Dataset: 30,969 ASCII art entries -- Task: Text generation -- Format: Text -- Language: English -- Size: 1M - 10M -- Category: Art - -Integration: -- ASCII art stored securely in ENE database -- Semantic vector encoding for similarity search -- Hyperbolic encoding for hierarchical concept matching -- Access control via ENE security system -""" - -import sys -import json -import sqlite3 -from pathlib import Path -from typing import Dict, List, Optional, Any -from dataclasses import dataclass -import hashlib - -sys.path.insert(0, str(Path(__file__).parent.parent)) - -from infra.ene_api import ENEAPIHook, AccessLevel -from infra.lean_unified_shim import OmnidirectionalInterface - - -@dataclass -class AsciiArtEntry: - """ASCII art entry from the dataset""" - id: str - text: str - category: Optional[str] = None - style: Optional[str] = None - width: Optional[int] = None - height: Optional[int] = None - metadata: Dict = None - - -class AsciiArtStore: - """ASCII Art Store integrated with ENE""" - - def __init__(self, db_path: str = "/home/allaun/Documents/Research Stack/data/substrate_index.db"): - self.db_path = db_path - self.ene_api = ENEAPIHook() - self.omni_interface = OmnidirectionalInterface() - self._init_ascii_tables() - - def _init_ascii_tables(self): - """Initialize ASCII art specific tables in ENE database""" - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - # ASCII art catalog table - cursor.execute(""" - CREATE TABLE IF NOT EXISTS ascii_art_catalog ( - id TEXT PRIMARY KEY, - category TEXT, - style TEXT, - width INTEGER, - height INTEGER, - line_count INTEGER, - char_count INTEGER, - semantic_vector TEXT, - hyperbolic_coords TEXT, - created_at INTEGER NOT NULL, - access_count INTEGER DEFAULT 0 - ) - """) - - # ASCII art style index - cursor.execute(""" - CREATE TABLE IF NOT EXISTS ascii_art_styles ( - style TEXT PRIMARY KEY, - count INTEGER DEFAULT 0, - avg_width REAL, - avg_height REAL, - last_updated INTEGER NOT NULL - ) - """) - - conn.commit() - conn.close() - - def _analyze_ascii_art(self, text: str) -> Dict[str, Any]: - """Analyze ASCII art properties""" - lines = text.split('\n') - width = max(len(line) for line in lines) if lines else 0 - height = len(lines) - line_count = len(lines) - char_count = sum(len(line) for line in lines) - - # Detect style based on patterns - style = self._detect_style(text) - - return { - "width": width, - "height": height, - "line_count": line_count, - "char_count": char_count, - "style": style - } - - def _detect_style(self, text: str) -> str: - """Detect ASCII art style from patterns""" - # Simple heuristic style detection - if '█' in text or '▓' in text or '▒' in text: - return "block" - elif any(c in text for c in '/\\|()_'): - return "line" - elif any(c in text for c in '@#%*+=-:.'): - return "ascii" - else: - return "mixed" - - def store_ascii_art(self, entry: AsciiArtEntry) -> bool: - """Store ASCII art entry in ENE database""" - try: - # Analyze the art - analysis = self._analyze_ascii_art(entry.text) - - # Generate semantic vector - semantic_vector = self.omni_interface._derive_semantic_vector( - entry.text[:500], # Use first 500 chars for semantic vector - {"category": entry.category, "style": analysis["style"]} - ) - - # Generate hyperbolic encoding - import numpy as np - vector_array = np.array(semantic_vector) - hyperbolic = self.omni_interface.hyperbolic_cache.get_or_encode(vector_array) - - # Store in ENE secure storage - self.ene_api.store_sensitive_data( - pkg=f"ascii_art/{entry.id}", - payload=entry.text, - classification=AccessLevel.PUBLIC, # ASCII art is public - semantic_vector=semantic_vector - ) - - # Store metadata in catalog - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - cursor.execute(""" - INSERT OR REPLACE INTO ascii_art_catalog - (id, category, style, width, height, line_count, char_count, - semantic_vector, hyperbolic_coords, created_at) - VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?) - """, ( - entry.id, - entry.category or "general", - analysis["style"], - analysis["width"], - analysis["height"], - analysis["line_count"], - analysis["char_count"], - json.dumps(semantic_vector), - json.dumps(hyperbolic.coordinates.tolist()), - int(__import__('time').time()) - )) - - # Update style index - cursor.execute(""" - INSERT OR REPLACE INTO ascii_art_styles - (style, count, avg_width, avg_height, last_updated) - VALUES (?, 1, ?, ?, ?) - """, ( - analysis["style"], - analysis["width"], - analysis["height"], - int(__import__('time').time()) - )) - - conn.commit() - conn.close() - - return True - except Exception as e: - print(f"Error storing ASCII art: {e}") - return False - - def retrieve_ascii_art(self, art_id: str) -> Optional[AsciiArtEntry]: - """Retrieve ASCII art entry by ID""" - try: - # Retrieve from ENE secure storage - result = self.ene_api.retrieve_sensitive_data( - f"ascii_art/{art_id}", - AccessLevel.PUBLIC - ) - - if result.get("success"): - # Get metadata from catalog - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - cursor.execute(""" - SELECT category, style, width, height, line_count, char_count - FROM ascii_art_catalog - WHERE id = ? - """, (art_id,)) - - row = cursor.fetchone() - conn.close() - - if row: - # Increment access count - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - cursor.execute( - "UPDATE ascii_art_catalog SET access_count = access_count + 1 WHERE id = ?", - (art_id,) - ) - conn.commit() - conn.close() - - return AsciiArtEntry( - id=art_id, - text=result["payload"], - category=row[0], - style=row[1], - width=row[2], - height=row[3], - metadata={"line_count": row[4], "char_count": row[5]} - ) - - return None - except Exception as e: - print(f"Error retrieving ASCII art: {e}") - return None - - def search_by_style(self, style: str, limit: int = 20) -> List[Dict]: - """Search ASCII art by style""" - try: - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - cursor.execute(""" - SELECT id, category, style, width, height, line_count, char_count - FROM ascii_art_catalog - WHERE style = ? - ORDER BY access_count DESC - LIMIT ? - """, (style, limit)) - - rows = cursor.fetchall() - conn.close() - - return [ - { - "id": row[0], - "category": row[1], - "style": row[2], - "width": row[3], - "height": row[4], - "line_count": row[5], - "char_count": row[6] - } - for row in rows - ] - except Exception as e: - print(f"Error searching by style: {e}") - return [] - - def semantic_search(self, query: str, limit: int = 20) -> List[Dict]: - """Semantic search for ASCII art using hyperbolic encoding""" - try: - # Derive semantic vector for query - semantic_vector = self.omni_interface._derive_semantic_vector(query) - - # Use hyperbolic similarity search - result = self.omni_interface.hyperbolic_similarity_search(query, top_k=limit) - - # Get matching ASCII art entries - similar_ids = [str(sim[0]) for sim in result["similar_results"]] - - if not similar_ids: - return [] - - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - placeholders = ",".join(["?" for _ in similar_ids]) - cursor.execute(f""" - SELECT id, category, style, width, height, line_count, char_count - FROM ascii_art_catalog - WHERE id IN ({placeholders}) - ORDER BY access_count DESC - """, similar_ids) - - rows = cursor.fetchall() - conn.close() - - return [ - { - "id": row[0], - "category": row[1], - "style": row[2], - "width": row[3], - "height": row[4], - "line_count": row[5], - "char_count": row[6], - "similarity": result["similar_results"][i][1] if i < len(result["similar_results"]) else 0.0 - } - for i, row in enumerate(rows) - ] - except Exception as e: - print(f"Error in semantic search: {e}") - return [] - - def get_statistics(self) -> Dict[str, Any]: - """Get ASCII art store statistics""" - try: - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - # Total entries - cursor.execute("SELECT COUNT(*) FROM ascii_art_catalog") - total = cursor.fetchone()[0] - - # Style distribution - cursor.execute("SELECT style, COUNT(*) FROM ascii_art_catalog GROUP BY style") - style_dist = dict(cursor.fetchall()) - - # Average dimensions - cursor.execute("SELECT AVG(width), AVG(height) FROM ascii_art_catalog") - avg_dims = cursor.fetchone() - - # Total access count - cursor.execute("SELECT SUM(access_count) FROM ascii_art_catalog") - total_access = cursor.fetchone()[0] or 0 - - conn.close() - - return { - "total_entries": total, - "style_distribution": style_dist, - "average_dimensions": { - "width": round(avg_dims[0] or 0, 2), - "height": round(avg_dims[1] or 0, 2) - }, - "total_access_count": total_access - } - except Exception as e: - print(f"Error getting statistics: {e}") - return {} - - -# Example usage and testing -if __name__ == "__main__": - print("=" * 70) - print("ASCII ART STORE - ENE INTEGRATION TEST") - print("=" * 70) - - store = AsciiArtStore() - - # Test 1: Store sample ASCII art - print("\n[Test 1] Storing sample ASCII art...") - sample_art = """ - _____ - / \\ - | O O | - | ^ | - | \\_/ | - \\_____/ -""" - entry = AsciiArtEntry( - id="test_smiley_001", - text=sample_art, - category="faces", - style="line" - ) - - success = store.store_ascii_art(entry) - print(f"Store result: {success}") - - # Test 2: Retrieve ASCII art - print("\n[Test 2] Retrieving ASCII art...") - retrieved = store.retrieve_ascii_art("test_smiley_001") - if retrieved: - print(f"Retrieved: {retrieved.text}") - print(f"Style: {retrieved.style}, Width: {retrieved.width}, Height: {retrieved.height}") - else: - print("Failed to retrieve") - - # Test 3: Search by style - print("\n[Test 3] Searching by style...") - line_art = store.search_by_style("line", limit=5) - print(f"Found {len(line_art)} line-style entries") - - # Test 4: Semantic search - print("\n[Test 4] Semantic search...") - semantic_results = store.semantic_search("smiley face happy", limit=5) - print(f"Found {len(semantic_results)} semantic matches") - - # Test 5: Statistics - print("\n[Test 5] Store statistics...") - stats = store.get_statistics() - print(json.dumps(stats, indent=2)) - - print("\n" + "=" * 70) - print("ASCII ART STORE INTEGRATION COMPLETE") - print("=" * 70) diff --git a/4-Infrastructure/infra/cloud_runtime.py b/4-Infrastructure/infra/cloud_runtime.py deleted file mode 100644 index eb4504af..00000000 --- a/4-Infrastructure/infra/cloud_runtime.py +++ /dev/null @@ -1,550 +0,0 @@ -#!/usr/bin/env python3 -""" -Cloud Runtime System — Central Orchestration Layer - -Implements the cloud architecture from the diagram: -- Agent Runtime: Central execution environment -- Session State: Session management and persistence -- File System: Unified storage interface -- MCP Integration: Model Context Protocol orchestration -- Tools: Unified tool registry and execution - -This ties together existing infrastructure components into a coherent cloud system. -""" - -import asyncio -import json -import sqlite3 -import time -import uuid -import hashlib -from pathlib import Path -from typing import Dict, List, Optional, Any, Callable -from dataclasses import dataclass, field, asdict -from datetime import datetime, timedelta -from enum import Enum -import subprocess -import sys - -# Add paths to existing infrastructure -ROOT = Path(__file__).parent.parent.parent -sys.path.insert(0, str(ROOT / "4-Infrastructure" / "infra")) -sys.path.insert(0, str(ROOT / "5-Applications" / "scripts")) - - -class SessionState(Enum): - """Session lifecycle states.""" - INITIALIZING = "initializing" - ACTIVE = "active" - SUSPENDED = "suspended" - COMPLETED = "completed" - FAILED = "failed" - - -@dataclass -class Session: - """Cloud session representation.""" - session_id: str - created_at: float - updated_at: float - state: SessionState - workspace: str - agent_id: Optional[str] = None - context: Dict[str, Any] = field(default_factory=dict) - file_system_state: Dict[str, Any] = field(default_factory=dict) - mcp_connections: Dict[str, bool] = field(default_factory=dict) - tool_registry: List[str] = field(default_factory=list) - - -@dataclass -class AgentRuntime: - """Agent runtime configuration and state.""" - runtime_id: str - agent_type: str - model: str - capabilities: List[str] - status: str = "idle" - current_session: Optional[str] = None - metrics: Dict[str, Any] = field(default_factory=dict) - - -class CloudFileSystem: - """ - Unified file system interface for cloud runtime. - - Provides consistent file operations across local storage, - cloud storage, and MCP-mounted file systems. - """ - - def __init__(self, base_path: str = "/home/allaun/Documents/Research Stack"): - self.base_path = Path(base_path) - self.shared_data = self.base_path / "shared-data" - self.mount_points: Dict[str, Path] = {} - - def register_mount(self, name: str, path: str): - """Register a file system mount point.""" - self.mount_points[name] = Path(path) - - def read_file(self, path: str, mount: str = "local") -> str: - """Read file from specified mount.""" - if mount == "local": - file_path = self.base_path / path - elif mount in self.mount_points: - file_path = self.mount_points[mount] / path - else: - raise ValueError(f"Unknown mount: {mount}") - - return file_path.read_text() - - def write_file(self, path: str, content: str, mount: str = "local"): - """Write file to specified mount.""" - if mount == "local": - file_path = self.base_path / path - elif mount in self.mount_points: - file_path = self.mount_points[mount] / path - else: - raise ValueError(f"Unknown mount: {mount}") - - file_path.parent.mkdir(parents=True, exist_ok=True) - file_path.write_text(content) - - def list_files(self, path: str = "", mount: str = "local") -> List[str]: - """List files in directory.""" - if mount == "local": - dir_path = self.base_path / path - elif mount in self.mount_points: - dir_path = self.mount_points[mount] / path - else: - raise ValueError(f"Unknown mount: {mount}") - - if not dir_path.exists(): - return [] - - return [f.name for f in dir_path.iterdir() if f.is_file()] - - -class MCPOrchestrator: - """ - MCP (Model Context Protocol) orchestrator. - - Manages connections to multiple MCP servers and provides - unified interface for tool execution across all servers. - """ - - def __init__(self): - self.servers: Dict[str, Any] = {} - self.server_status: Dict[str, bool] = {} - self.tool_registry: Dict[str, Dict[str, Any]] = {} - - async def initialize_server(self, server_name: str, config: Dict[str, Any]): - """Initialize an MCP server connection.""" - try: - # This would connect to actual MCP servers - # For now, we'll simulate the connection - self.servers[server_name] = config - self.server_status[server_name] = True - print(f"[MCP] Initialized server: {server_name}") - return True - except Exception as e: - print(f"[MCP] Failed to initialize {server_name}: {e}") - self.server_status[server_name] = False - return False - - async def call_tool(self, server_name: str, tool_name: str, arguments: Dict[str, Any]) -> Any: - """Execute a tool on a specific MCP server.""" - if server_name not in self.servers or not self.server_status[server_name]: - raise ValueError(f"Server {server_name} not available") - - # Simulate tool execution - print(f"[MCP] Calling {tool_name} on {server_name} with args: {arguments}") - return {"ok": True, "result": f"Executed {tool_name}"} - - def get_available_tools(self, server_name: str) -> List[str]: - """Get available tools from a server.""" - if server_name not in self.servers: - return [] - return ["tool1", "tool2", "tool3"] # Placeholder - - -class ToolRegistry: - """ - Unified tool registry. - - Manages all available tools from MCP servers, local scripts, - and cloud functions in a single registry. - """ - - def __init__(self): - self.tools: Dict[str, Dict[str, Any]] = {} - self.categories: Dict[str, List[str]] = { - "file": [], - "mcp": [], - "cloud": [], - "local": [] - } - - def register_tool(self, tool_id: str, tool_def: Dict[str, Any], category: str = "local"): - """Register a tool in the registry.""" - self.tools[tool_id] = tool_def - if category not in self.categories: - self.categories[category] = [] - self.categories[category].append(tool_id) - - def get_tool(self, tool_id: str) -> Optional[Dict[str, Any]]: - """Get tool definition.""" - return self.tools.get(tool_id) - - def list_tools(self, category: Optional[str] = None) -> List[str]: - """List tools by category.""" - if category: - return self.categories.get(category, []) - return list(self.tools.keys()) - - def execute_tool(self, tool_id: str, arguments: Dict[str, Any]) -> Any: - """Execute a tool by ID.""" - tool = self.get_tool(tool_id) - if not tool: - raise ValueError(f"Tool {tool_id} not found") - - # Execute based on tool type - if tool.get("type") == "mcp": - # Would call MCP orchestrator - return {"ok": True, "result": f"Executed MCP tool {tool_id}"} - elif tool.get("type") == "local": - # Execute local script - return self._execute_local_tool(tool, arguments) - else: - return {"ok": False, "error": f"Unknown tool type: {tool.get('type')}"} - - def _execute_local_tool(self, tool: Dict[str, Any], arguments: Dict[str, Any]) -> Any: - """Execute a local tool script.""" - script_path = tool.get("path") - if not script_path: - return {"ok": False, "error": "No script path"} - - try: - # Execute script - result = subprocess.run( - ["python3", script_path], - capture_output=True, - text=True, - timeout=30 - ) - return {"ok": True, "output": result.stdout, "error": result.stderr} - except Exception as e: - return {"ok": False, "error": str(e)} - - -class CloudRuntime: - """ - Central Cloud Runtime — Main Orchestrator - - Implements the complete cloud architecture from the diagram: - - Manages sessions and their state - - Coordinates agent runtimes - - Orchestrates MCP connections - - Provides unified tool interface - - Manages file system operations - """ - - def __init__(self, db_path: str = "/home/allaun/Documents/Research Stack/shared-data/data/cloud_runtime.db"): - self.db_path = db_path - self.sessions: Dict[str, Session] = {} - self.agent_runtimes: Dict[str, AgentRuntime] = {} - self.file_system = CloudFileSystem() - self.mcp_orchestrator = MCPOrchestrator() - self.tool_registry = ToolRegistry() - - self._init_database() - self._load_sessions() - self._init_agent_runtimes() - self._register_tools() - - def _init_database(self): - """Initialize runtime database.""" - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - cursor.execute(""" - CREATE TABLE IF NOT EXISTS sessions ( - session_id TEXT PRIMARY KEY, - created_at REAL, - updated_at REAL, - state TEXT, - workspace TEXT, - agent_id TEXT, - context TEXT, - file_system_state TEXT, - mcp_connections TEXT, - tool_registry TEXT - ) - """) - - cursor.execute(""" - CREATE TABLE IF NOT EXISTS agent_runtimes ( - runtime_id TEXT PRIMARY KEY, - agent_type TEXT, - model TEXT, - capabilities TEXT, - status TEXT, - current_session TEXT, - metrics TEXT - ) - """) - - conn.commit() - conn.close() - - def _load_sessions(self): - """Load sessions from database.""" - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - cursor.execute("SELECT * FROM sessions") - for row in cursor.fetchall(): - session = Session( - session_id=row[0], - created_at=row[1], - updated_at=row[2], - state=SessionState(row[3]), - workspace=row[4], - agent_id=row[5], - context=json.loads(row[6]) if row[6] else {}, - file_system_state=json.loads(row[7]) if row[7] else {}, - mcp_connections=json.loads(row[8]) if row[8] else {}, - tool_registry=json.loads(row[9]) if row[9] else [] - ) - self.sessions[session.session_id] = session - - conn.close() - - def _init_agent_runtimes(self): - """Initialize default agent runtimes.""" - # Devin CLI agent - self.agent_runtimes["devin-cli"] = AgentRuntime( - runtime_id="devin-cli", - agent_type="devin-cli", - model="swe-1-6", - capabilities=["code_editing", "file_operations", "terminal"], - status="idle" - ) - - # Devin Cloud agent - self.agent_runtimes["devin-cloud"] = AgentRuntime( - runtime_id="devin-cloud", - agent_type="devin-cloud", - model="adaptive", - capabilities=["cloud_execution", "mcp_tools", "session_management"], - status="idle" - ) - - async def _init_mcp_servers(self): - """Initialize MCP servers from configuration.""" - # Initialize known MCP servers - mcp_configs = { - "ene": {"type": "stdio", "command": "python3", "args": ["mcp_ene_atlas.py"]}, - "sovereign-research-stack": {"type": "stdio", "command": "python3", "args": ["mcp_server.py"]}, - "sqlite": {"type": "stdio", "command": "uvx", "args": ["mcp-server-sqlite"]}, - } - - for server_name, config in mcp_configs.items(): - await self.mcp_orchestrator.initialize_server(server_name, config) - - def _register_tools(self): - """Register available tools from various sources.""" - # Register file system tools - self.tool_registry.register_tool("fs.read", { - "name": "Read File", - "description": "Read file from file system", - "type": "local", - "parameters": {"path": "string", "mount": "string"} - }, "file") - - self.tool_registry.register_tool("fs.write", { - "name": "Write File", - "description": "Write file to file system", - "type": "local", - "parameters": {"path": "string", "content": "string", "mount": "string"} - }, "file") - - # Register MCP tools - self.tool_registry.register_tool("mcp.ene.lookup", { - "name": "ENE Atlas Lookup", - "description": "Lookup memory atoms in ENE Atlas", - "type": "mcp", - "server": "ene" - }, "mcp") - - async def create_session(self, workspace: str, agent_type: str = "devin-cli") -> str: - """Create a new cloud session.""" - session_id = str(uuid.uuid4()) - now = time.time() - - session = Session( - session_id=session_id, - created_at=now, - updated_at=now, - state=SessionState.INITIALIZING, - workspace=workspace, - agent_id=agent_type - ) - - self.sessions[session_id] = session - self._save_session(session) - - # Initialize session - await self._initialize_session(session_id) - - return session_id - - async def _initialize_session(self, session_id: str): - """Initialize a session with required resources.""" - session = self.sessions[session_id] - - # Set up MCP connections - for server_name in self.mcp_orchestrator.servers: - session.mcp_connections[server_name] = self.mcp_orchestrator.server_status[server_name] - - # Initialize file system state - session.file_system_state = { - "mounts": list(self.file_system.mount_points.keys()), - "workspace_files": self.file_system.list_files(mount="local") - } - - # Register available tools - session.tool_registry = self.tool_registry.list_tools() - - # Update session state - session.state = SessionState.ACTIVE - session.updated_at = time.time() - self._save_session(session) - - def _save_session(self, session: Session): - """Save session to database.""" - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - cursor.execute(""" - INSERT OR REPLACE INTO sessions - (session_id, created_at, updated_at, state, workspace, agent_id, - context, file_system_state, mcp_connections, tool_registry) - VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?) - """, ( - session.session_id, - session.created_at, - session.updated_at, - session.state.value, - session.workspace, - session.agent_id, - json.dumps(session.context), - json.dumps(session.file_system_state), - json.dumps(session.mcp_connections), - json.dumps(session.tool_registry) - )) - - conn.commit() - conn.close() - - async def execute_tool(self, session_id: str, tool_id: str, arguments: Dict[str, Any]) -> Any: - """Execute a tool within a session.""" - if session_id not in self.sessions: - raise ValueError(f"Session {session_id} not found") - - session = self.sessions[session_id] - session.updated_at = time.time() - - # Execute tool - result = self.tool_registry.execute_tool(tool_id, arguments) - - # Update session context - session.context[f"last_tool_{tool_id}"] = { - "executed_at": time.time(), - "arguments": arguments, - "result": result - } - - self._save_session(session) - return result - - def get_session_status(self, session_id: str) -> Dict[str, Any]: - """Get session status and state.""" - if session_id not in self.sessions: - raise ValueError(f"Session {session_id} not found") - - session = self.sessions[session_id] - return { - "session_id": session.session_id, - "state": session.state.value, - "workspace": session.workspace, - "agent_id": session.agent_id, - "created_at": session.created_at, - "updated_at": session.updated_at, - "mcp_connections": session.mcp_connections, - "available_tools": len(session.tool_registry) - } - - def get_runtime_status(self) -> Dict[str, Any]: - """Get overall runtime status.""" - return { - "sessions": { - "total": len(self.sessions), - "active": sum(1 for s in self.sessions.values() if s.state == SessionState.ACTIVE), - "by_state": {state.value: sum(1 for s in self.sessions.values() if s.state == state) - for state in SessionState} - }, - "agent_runtimes": { - runtime_id: { - "type": runtime.agent_type, - "model": runtime.model, - "status": runtime.status, - "current_session": runtime.current_session - } - for runtime_id, runtime in self.agent_runtimes.items() - }, - "mcp_servers": { - "total": len(self.mcp_orchestrator.servers), - "connected": sum(1 for s in self.mcp_orchestrator.server_status.values() if s), - "status": self.mcp_orchestrator.server_status - }, - "tools": { - "total": len(self.tool_registry.tools), - "by_category": {cat: len(tools) for cat, tools in self.tool_registry.categories.items()} - }, - "file_system": { - "base_path": str(self.file_system.base_path), - "mounts": list(self.file_system.mount_points.keys()) - } - } - - -async def main(): - """Main entry point for cloud runtime.""" - runtime = CloudRuntime() - - print("=== Cloud Runtime System ===") - print("Initializing cloud infrastructure...") - - # Initialize MCP servers - print("\nInitializing MCP servers...") - await runtime._init_mcp_servers() - - # Create a test session - print("\nCreating test session...") - session_id = await runtime.create_session("/home/allaun/Documents/Research Stack", "devin-cli") - print(f"Created session: {session_id}") - - # Get session status - print("\nSession status:") - status = runtime.get_session_status(session_id) - print(json.dumps(status, indent=2)) - - # Get runtime status - print("\nRuntime status:") - runtime_status = runtime.get_runtime_status() - print(json.dumps(runtime_status, indent=2)) - - print("\n=== Cloud Runtime Initialized Successfully ===") - - -if __name__ == "__main__": - asyncio.run(main()) \ No newline at end of file diff --git a/4-Infrastructure/infra/credential_provider.py b/4-Infrastructure/infra/credential_provider.py deleted file mode 100644 index 4bb90c88..00000000 --- a/4-Infrastructure/infra/credential_provider.py +++ /dev/null @@ -1,305 +0,0 @@ -from __future__ import annotations - -import json -import os -import socket -import urllib.parse -from dataclasses import dataclass, field -from pathlib import Path -from typing import Any - -CREDENTIAL_PROVIDER_VERSION = "0.4" - -# Remote credential server (microVM) — all nodes query this as primary -_cred_server_url = os.environ.get("RS_CREDENTIAL_SERVER", "http://100.101.247.127:8444") - -# Config file path — local fallback if remote is unreachable -_cred_config_path = os.environ.get( - "RS_CREDENTIAL_CONFIG", - "/etc/rs-surface/credentials.json", -) - -RDS_HOST = os.environ.get( - "RDS_HOST", - "database-1-instance-1.cghu8yqogqwo.us-east-1.rds.amazonaws.com", -) -RDS_PORT = int(os.environ.get("RDS_PORT", "5432")) -RDS_USER = os.environ.get("RDS_USER", "postgres") -RDS_IAM_TOKEN = os.environ.get("RDS_IAM_TOKEN", "") - - -@dataclass -class Credential: - provider: str - key_name: str - value: str - metadata: dict[str, Any] = field(default_factory=dict) - - -PROVIDER_ENV_MAP: dict[str, dict[str, str]] = { - "deepseek": { - "env_var": "DEEPSEEK_API_KEY", - "description": "DeepSeek LLM API", - }, - "quandela": { - "env_var": "QUANDELA_API_KEY", - "description": "Quandela quantum cloud", - }, - "wolfram_alpha": { - "env_var": "WOLFRAM_ALPHA_APPID", - "description": "Wolfram Alpha API", - }, - "notion": { - "env_var": "NOTION_API_KEY", - "description": "Notion API", - }, - "linear": { - "env_var": "LINEAR_API_KEY", - "description": "Linear API", - }, - "gemini": { - "env_var": "GEMINI_API_KEY", - "description": "Google Gemini API", - }, - "ollama": { - "env_var": "OLLAMA_API_KEY", - "description": "Ollama local API", - }, - "brave_search": { - "env_var": "BRAVE_API_KEY", - "description": "Brave Search API", - }, - "neural_endeavor": { - "env_var": "ENE_ENCRYPTION_KEY", - "description": "ENE encryption master key", - }, - "bedrock": { - "env_var": "AWS_BEARER_TOKEN_BEDROCK", - "description": "Amazon Bedrock API", - }, - "venice": { - "env_var": "VENICE_API_KEY", - "description": "Venice AI API", - }, -} - - -def _rds_connect(): - import psycopg2 - import os - - # Primary check: RDS_IAM_TOKEN env var - password = RDS_IAM_TOKEN - - if not password: - # Generate dynamic IAM authentication token using boto3 - try: - import boto3 - region = os.environ.get("AWS_DEFAULT_REGION", "us-east-1") - client = boto3.client('rds', region_name=region) - password = client.generate_db_auth_token( - DBHostname=RDS_HOST, - Port=RDS_PORT, - DBUsername=RDS_USER, - Region=region - ) - except Exception as e: - # Fallback to standard environment password or log warning - pass - - if not password: - password = os.environ.get("RDS_PASSWORD", "") - - return psycopg2.connect( - host=RDS_HOST, - port=RDS_PORT, - user=RDS_USER, - password=password, - dbname="postgres", - connect_timeout=10, - sslmode="require", - ) - - -def _load_from_rds() -> list[Credential]: - try: - conn = _rds_connect() - cur = conn.cursor() - cur.execute(""" - SELECT pkg, provider, encode(encrypted_payload, 'escape'), classification - FROM credential_store.credentials - WHERE is_active = TRUE - ORDER BY provider - """) - creds = [] - for pkg, provider, payload_raw, classification in cur.fetchall(): - creds.append(Credential( - provider=provider, - key_name=pkg, - value=payload_raw, - metadata={ - "source": "rds", - "pkg": pkg, - "classification": classification, - }, - )) - conn.close() - return creds - except Exception as e: - return [] - - -def _load_from_remote() -> list[Credential]: - """Fetch credentials from the remote credential server (microVM).""" - import urllib.request - import urllib.error - - # Skip self-query: don't try to connect to ourselves - self_hosts = {"localhost", "127.0.0.1", "0.0.0.0", "", socket.gethostname()} - netloc = urllib.parse.urlparse(_cred_server_url).hostname or "" - if netloc in self_hosts: - return [] - - try: - req = urllib.request.Request(_cred_server_url + "/credentials") - with urllib.request.urlopen(req, timeout=5) as resp: - manifest = json.loads(resp.read().decode("utf-8")) - except (urllib.error.URLError, json.JSONDecodeError, OSError): - return [] - - providers = manifest.get("providers", []) - creds: list[Credential] = [] - for p in providers: - name = p.get("name") - if not name: - continue - try: - req = urllib.request.Request(f"{_cred_server_url}/credentials/{name}") - with urllib.request.urlopen(req, timeout=5) as resp: - data = json.loads(resp.read().decode("utf-8")) - except (urllib.error.URLError, json.JSONDecodeError, OSError): - continue - if data.get("ok"): - creds.append(Credential( - provider=name, - key_name=data.get("key_name", f"remote/{name}"), - value=data["key"], - metadata={"source": "remote", "server": _cred_server_url}, - )) - return creds - - -def _load_from_config() -> list[Credential]: - """Read credentials from a JSON config file. - - Format: - { - "deepseek": "sk-...", - "bedrock": "ABSK...", - "ollama": "..." - } - """ - path = Path(_cred_config_path) - if not path.exists(): - return [] - try: - with path.open("r", encoding="utf-8") as f: - data = json.load(f) - except (json.JSONDecodeError, OSError): - return [] - if not isinstance(data, dict): - return [] - creds: list[Credential] = [] - for provider_name, value in data.items(): - if not isinstance(value, str) or not value.strip(): - continue - cfg = PROVIDER_ENV_MAP.get(provider_name, {}) - creds.append(Credential( - provider=provider_name, - key_name=f"config/{provider_name}", - value=value.strip(), - metadata={ - "description": cfg.get("description", provider_name), - "source": "config_file", - "file": str(path), - }, - )) - return creds - - -def _load_from_env() -> list[Credential]: - creds: list[Credential] = [] - for provider, cfg in PROVIDER_ENV_MAP.items(): - value = os.getenv(cfg["env_var"]) - if not value: - continue - creds.append(Credential( - provider=provider, - key_name=cfg["env_var"], - value=value, - metadata={"description": cfg["description"], "source": "env"}, - )) - return creds - - -def load_credentials() -> list[Credential]: - creds = _load_from_rds() - if creds: - return creds - creds = _load_from_remote() - if creds: - return creds - creds = _load_from_config() - if creds: - return creds - return _load_from_env() - - -def _detect_backend() -> str: - if _load_from_rds(): - return "rds" - if _load_from_remote(): - return "remote" - if _load_from_config(): - return "config_file" - return "env" - - -def credential_status() -> dict[str, Any]: - creds = load_credentials() - return { - "ok": True, - "node": os.environ.get("RS_SURFACE_NODE_ID", "unknown"), - "provider_version": CREDENTIAL_PROVIDER_VERSION, - "backend": _detect_backend(), - "available_providers": [c.provider for c in creds], - "count": len(creds), - "remote_server": _cred_server_url, - } - - -def resolve_credential(provider: str) -> Credential | None: - creds = load_credentials() - for c in creds: - if c.provider == provider: - return c - return None - - -def provider_manifest() -> dict[str, Any]: - creds = load_credentials() - return { - "ok": True, - "service_kind": "apiProvider", - "backend": _detect_backend(), - "providers": [ - { - "name": c.provider, - "description": c.metadata.get("description", ""), - "key_name": c.key_name, - "available": True, - } - for c in creds - ], - "node_id": os.environ.get("RS_SURFACE_NODE_ID", "unknown"), - } diff --git a/4-Infrastructure/infra/credential_server.py b/4-Infrastructure/infra/credential_server.py deleted file mode 100644 index 7bdb3144..00000000 --- a/4-Infrastructure/infra/credential_server.py +++ /dev/null @@ -1,319 +0,0 @@ -"""Credential server — REST + OpenAPI access point for API keys. - -Reads credentials from credential_provider (env vars + config file + RDS) -and serves them via HTTP so no script needs direct env var access. - -Usage: - # Standalone - python3 credential_server.py [--port 8444] [--bind 0.0.0.0] - - # Via existing embedded surface (if RS_CREDENTIAL_SERVER_PORT is set) -""" - -from __future__ import annotations - -import json -import os -import re -import sys -import urllib.request -import urllib.parse -from http import HTTPStatus -from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer -from pathlib import Path -from typing import Any - -from credential_provider import ( - credential_status, - provider_manifest, - resolve_credential, - _rds_connect, - _cred_server_url -) - -VERSION = "0.4" -OPENAPI_VERSION = "3.0.3" - -OPENAPI_SPEC: dict[str, Any] = { - "openapi": OPENAPI_VERSION, - "info": { - "title": "Research Stack Credential Provider", - "description": "Central API key access point for all stack services. " - "Set keys once on the microVM, consume from anywhere.", - "version": VERSION, - }, - "servers": [ - {"url": "http://localhost:8444", "description": "Local / Tailnet default"}, - ], - "paths": { - "/": { - "get": { - "summary": "Service root", - "responses": {"200": {"description": "Service info + links"}}, - } - }, - "/health": { - "get": { - "summary": "Health check", - "responses": {"200": {"description": "OK"}}, - } - }, - "/openapi.json": { - "get": { - "summary": "OpenAPI 3.0 specification", - "responses": {"200": {"description": "This document"}}, - } - }, - "/credentials": { - "get": { - "summary": "List all available credential providers", - "responses": { - "200": { - "description": "Provider manifest", - "content": { - "application/json": { - "schema": { - "type": "object", - "properties": { - "providers": { - "type": "array", - "items": { - "type": "object", - "properties": { - "name": {"type": "string"}, - "description": {"type": "string"}, - "available": {"type": "boolean"}, - }, - }, - } - }, - } - } - }, - } - }, - } - }, - "/credentials/{provider}": { - "get": { - "summary": "Resolve a specific provider credential", - "parameters": [ - { - "name": "provider", - "in": "path", - "required": True, - "schema": {"type": "string"}, - "description": "Provider name (deepseek, bedrock, etc.)", - } - ], - "responses": { - "200": {"description": "Credential value"}, - "404": {"description": "Provider not found"}, - }, - } - }, - "/status": { - "get": { - "summary": "Credential provider status", - "responses": {"200": {"description": "Backend info + counts"}}, - } - }, - }, -} - - -SERVICE_INFO: dict[str, Any] = { - "service": "credential-provider", - "version": VERSION, - "docs": "/openapi.json", - "credentials": "/credentials", - "status": "/status", - "health": "/health", -} - - -class CredentialHTTPHandler(BaseHTTPRequestHandler): - - def log_message(self, fmt, *args): - sys.stderr.write(f"[credential-server] {args[0]} {args[1]} {args[2]}\n") - - def _send_json(self, data: Any, status: int = 200): - body = json.dumps(data, sort_keys=True, indent=2).encode("utf-8") + b"\n" - self.send_response(status) - self.send_header("Content-Type", "application/json") - self.send_header("Content-Length", str(len(body))) - self.send_header("Access-Control-Allow-Origin", "*") - self.end_headers() - self.wfile.write(body) - - def _send_error(self, status: int, message: str): - self._send_json({"ok": False, "error": message}, status) - - def do_GET(self): - path = self.path.rstrip("/") or "/" - - if path in ("/", "/api"): - return self._send_json(SERVICE_INFO) - - if path in ("/health", "/api/health"): - return self._send_json({"status": "ok", "version": VERSION}) - - if path == "/openapi.json": - return self._send_json(OPENAPI_SPEC) - - if path in ("/status", "/api/status"): - return self._send_json(credential_status()) - - if path == "/credentials" or path == "/api/credentials": - return self._send_json(provider_manifest()) - - m = re.match(r"^(?:/api)?/credentials/([a-zA-Z0-9_-]+)$", path) - if m: - provider = m.group(1).lower() - cred = resolve_credential(provider) - if cred is None: - return self._send_error(404, f"provider '{provider}' not found") - return self._send_json({ - "ok": True, - "provider": cred.provider, - "key": cred.value, - "key_name": cred.key_name, - }) - - return self._send_error(404, f"not found: {path}") - - def do_POST(self): - path = self.path.rstrip("/") or "/" - - if path in ("/api/webhooks/linear", "/webhooks/linear"): - # Read raw body - content_length = int(self.headers.get("Content-Length", 0)) - raw_body = self.rfile.read(content_length) - - # Node identification logic: proxy or process - node_id = os.environ.get("RS_SURFACE_NODE_ID", "") - if node_id != "MicroVM-Racknerd": - # Act as Proxy: Forward POST request to the central microVM - target_url = f"{_cred_server_url}/api/webhooks/linear" - req = urllib.request.Request(target_url, data=raw_body, method="POST") - for k, v in self.headers.items(): - if k.lower() in ("content-type", "linear-signature"): - req.add_header(k, v) - try: - with urllib.request.urlopen(req, timeout=10) as resp: - resp_data = json.loads(resp.read().decode("utf-8")) - return self._send_json(resp_data, resp.status) - except urllib.error.HTTPError as e: - try: - err_data = json.loads(e.read().decode("utf-8")) - return self._send_json(err_data, e.code) - except Exception: - return self._send_error(e.code, f"Forwarding failed: {e}") - except Exception as e: - return self._send_error(500, f"Forwarding connection error: {e}") - - # Act as Server: Process Linear webhook body and upsert into Aurora RDS ENE packages table - try: - data = json.loads(raw_body.decode("utf-8")) - except Exception as e: - return self._send_error(400, f"Invalid JSON payload: {e}") - - action = data.get("action") - issue_data = data.get("data", {}) - payload_type = data.get("type") - - if payload_type != "Issue": - return self._send_json({"ok": True, "message": f"Ignored non-Issue payload: {payload_type}"}) - - identifier = issue_data.get("identifier") - if not identifier: - return self._send_error(400, "Missing issue identifier") - - pkg_id = f"linear/{identifier}" - - if action == "remove": - try: - conn = _rds_connect() - cur = conn.cursor() - cur.execute("DELETE FROM ene.packages WHERE pkg = %s", (pkg_id,)) - conn.commit() - conn.close() - return self._send_json({"ok": True, "action": "delete", "pkg": pkg_id}) - except Exception as e: - return self._send_error(500, f"Database delete error: {e}") - - # Extract title and description - title = issue_data.get("title", "") - description = issue_data.get("description", "") - full_text = f"{title}\n\n{description}" if description else title - url = issue_data.get("url", "") - - # Extract labels - labels_data = issue_data.get("labels", []) - labels_list = [] - if isinstance(labels_data, list): - for l in labels_data: - if isinstance(l, dict): - labels_list.append(l.get("name")) - elif isinstance(l, str): - labels_list.append(l) - tags_json = json.dumps(labels_list) - - import datetime - now_iso = datetime.datetime.now(datetime.timezone.utc).isoformat() - - try: - conn = _rds_connect() - cur = conn.cursor() - cur.execute(""" - INSERT INTO ene.packages ( - pkg, version, domain, tier, archetype, - tags, description, source, indexed_utc - ) VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s) - ON CONFLICT (pkg) DO UPDATE SET - description = EXCLUDED.description, - tags = EXCLUDED.tags, - source = EXCLUDED.source, - indexed_utc = EXCLUDED.indexed_utc - """, ( - pkg_id, - "1.0.0", - "LINEAR", - "INTENT", - "issue", - tags_json, - full_text, - url, - now_iso - )) - conn.commit() - conn.close() - return self._send_json({"ok": True, "action": action, "pkg": pkg_id}) - except Exception as e: - return self._send_error(500, f"Database upsert error: {e}") - - return self._send_error(404, f"not found: {path}") - - do_HEAD = do_GET - - -def main(): - import argparse - - parser = argparse.ArgumentParser(description="Research Stack Credential Server") - parser.add_argument("--port", type=int, default=int(os.environ.get("RS_CREDENTIAL_PORT", "8444"))) - parser.add_argument("--bind", default=os.environ.get("RS_CREDENTIAL_BIND", "0.0.0.0")) - args = parser.parse_args() - - server = ThreadingHTTPServer((args.bind, args.port), CredentialHTTPHandler) - print(f"[credential-server] listening on {args.bind}:{args.port}", flush=True) - print(f"[credential-server] OpenAPI spec at http://{args.bind}:{args.port}/openapi.json", flush=True) - try: - server.serve_forever() - except KeyboardInterrupt: - print("\n[credential-server] shutting down", flush=True) - server.server_close() - - -if __name__ == "__main__": - main() diff --git a/4-Infrastructure/infra/deepseek_adapter.py b/4-Infrastructure/infra/deepseek_adapter.py deleted file mode 100644 index 14293c07..00000000 --- a/4-Infrastructure/infra/deepseek_adapter.py +++ /dev/null @@ -1,113 +0,0 @@ -#!/usr/bin/env python3 -""" -deepseek_adapter.py — DeepSeek-V4 Integration - -Provides seamless access to DeepSeek-V4-Pro (Reasoning) and -DeepSeek-V4-Flash (Speed) for the Sovereign Research Stack. -Supports both local (Ollama) and API modes. -""" - -import os -import json -from typing import Dict, Any, List, Optional, Union -import requests -from dotenv import load_dotenv - -# Load environment variables from .env -load_dotenv() - -class DeepSeekV4: - """ - Unified client for DeepSeek-V4 models. - """ - - def __init__( - self, - api_key: Optional[str] = None, - local_url: str = "http://localhost:11434", - use_local: bool = True - ): - self.api_key = api_key or os.getenv("DEEPSEEK_API_KEY") - self.local_url = local_url - self.use_local = use_local - self.api_base = "https://api.deepseek.com/v1" - - def chat( - self, - messages: List[Dict[str, str]], - model: str = "deepseek-v4-pro", - stream: bool = False, - **kwargs - ) -> Union[Dict[str, Any], Any]: - """ - Perform a chat completion. - """ - if self.use_local: - return self._chat_local(messages, model, stream, **kwargs) - else: - return self._chat_api(messages, model, stream, **kwargs) - - def _chat_api(self, messages, model, stream, **kwargs): - if not self.api_key: - raise ValueError("DEEPSEEK_API_KEY not found in environment or constructor") - - url = f"{self.api_base}/chat/completions" - headers = { - "Authorization": f"Bearer {self.api_key}", - "Content-Type": "application/json" - } - payload = { - "model": model, - "messages": messages, - "stream": stream, - **kwargs - } - - resp = requests.post(url, headers=headers, json=payload) - resp.raise_for_status() - return resp.json() - - def _chat_local(self, messages, model, stream, **kwargs): - url = f"{self.local_url}/api/chat" - payload = { - "model": model, - "messages": messages, - "stream": stream, - "options": kwargs - } - - resp = requests.post(url, json=payload) - resp.raise_for_status() - if stream: - return resp # User handles generator - return resp.json() - -# ═══════════════════════════════════════════════════════════════════════════ -# Formalization Specialist -# ═══════════════════════════════════════════════════════════════════════════ - -class DeepSeekProver: - """ - Specialized agent for Lean 4 formalization using DeepSeek-V4-Pro. - """ - def __init__(self, client: DeepSeekV4): - self.client = client - - def formalize(self, mathematical_statement: str) -> str: - prompt = f""" -You are a Lean 4 formalization expert. -Convert the following mathematical statement into valid Lean 4 code. -Provide only the code in a fenced code block. - -Statement: -{mathematical_statement} -""" - messages = [{"role": "user", "content": prompt}] - # Using Qwen 2.5 Coder for local speed and Lean proficiency - model = "qwen2.5-coder:14b" if self.client.use_local else "deepseek-v4-pro" - res = self.client.chat(messages, model=model) - - if self.client.use_local: - return res["message"]["content"] - else: - return res["choices"][0]["message"]["content"] diff --git a/4-Infrastructure/infra/delta_gcl_compression_service.py b/4-Infrastructure/infra/delta_gcl_compression_service.py deleted file mode 100644 index fa3b0add..00000000 --- a/4-Infrastructure/infra/delta_gcl_compression_service.py +++ /dev/null @@ -1,343 +0,0 @@ -#!/usr/bin/env python3 -""" -Delta GCL Compression Service - -Real-time compression service for metadata using Delta GCL algorithm. -Integrates with Lean DeltaGCLCompression module via shim. - -Features: -- Real-time metadata compression -- Delta encoding with previous state tracking -- PTOS dictionary compression -- Variable-length GCL encoding -- Compression statistics tracking - -Per AGENTS.md: This is a shim layer - logic in Lean, only JSON marshaling here. -""" - -import sys -from pathlib import Path -# Add project root to Python path -project_root = Path(__file__).parent.parent -sys.path.insert(0, str(project_root)) - -import json -import hashlib -from typing import Dict, Any, List, Optional -from dataclasses import dataclass, asdict -from collections import defaultdict -import threading -import time - - -@dataclass -class CompressionRequest: - """Compression request from client""" - manifest: Dict[str, Any] - manifest_id: str - timestamp: float - - -@dataclass -class CompressionResult: - """Compression result""" - delta_gcl: str - stats: Dict[str, Any] - compressed_at: float - verified: bool = False - verification_error: Optional[str] = None - - -@dataclass -class CompressionStats: - """Aggregate compression statistics""" - total_compressions: int - total_original_size: int - total_compressed_size: int - avg_reduction_percent: float - - -class DeltaGCLCompressionService: - """Real-time Delta GCL compression service""" - - def __init__(self): - self.previous_manifests: Dict[str, Dict[str, Any]] = {} - self.compression_history: List[CompressionResult] = [] - self.stats = CompressionStats(0, 0, 0, 0.0) - self.lock = threading.Lock() - - # Import Lean shim - try: - from infra.lean_unified_shim import LeanUnifiedShim - - # Check if Lean binary exists - lean_bin_path = Path("/home/allaun/Documents/Research Stack/0-Core-Formalism/lean/Semantics/.lake/build/bin/SemanticsCli") - if lean_bin_path.exists(): - self.lean_shim = LeanUnifiedShim() - self.use_lean = True - print("[INFO] Using Lean DeltaGCLCompression module") - else: - print(f"[WARNING] Lean binary not found at {lean_bin_path}, using Python fallback") - self.use_lean = False - from scripts.delta_gcl_encoder import DeltaGCLEncoder - self.python_encoder = DeltaGCLEncoder() - except ImportError as e: - print(f"[WARNING] Lean shim import failed ({e}), using Python fallback") - self.use_lean = False - # Fallback to Python implementation - try: - from scripts.delta_gcl_encoder import DeltaGCLEncoder - self.python_encoder = DeltaGCLEncoder() - except ImportError: - print("[ERROR] Python fallback also not available") - self.python_encoder = None - - def compress_manifest(self, manifest: Dict[str, Any], - manifest_id: str, - use_delta: bool = True, - verify: bool = True) -> CompressionResult: - """Compress manifest to Delta GCL with optional verification""" - with self.lock: - timestamp = time.time() - - if self.use_lean: - # Use Lean implementation via shim - delta_gcl = self._compress_with_lean(manifest, manifest_id, use_delta) - else: - # Fallback to Python implementation - delta_gcl = self._compress_with_python(manifest, manifest_id, use_delta) - - # Calculate statistics - original_size = json.dumps(manifest).__sizeof__() - compressed_size = len(delta_gcl) - reduction = original_size - compressed_size - reduction_percent = (reduction / original_size * 100) if original_size > 0 else 0 - - stats = { - "original_size": original_size, - "compressed_size": compressed_size, - "reduction": reduction, - "reduction_percent": reduction_percent, - "use_delta": use_delta - } - - # Verify compression if requested - verified = False - verification_error = None - if verify: - verified, verification_error = self.verify_compression(delta_gcl, manifest) - - # Update aggregate statistics - self.stats.total_compressions += 1 - self.stats.total_original_size += original_size - self.stats.total_compressed_size += compressed_size - self.stats.avg_reduction_percent = ( - (self.stats.avg_reduction_percent * (self.stats.total_compressions - 1) + reduction_percent) - / self.stats.total_compressions - ) - - # Store result - result = CompressionResult(delta_gcl, stats, timestamp, verified, verification_error) - self.compression_history.append(result) - - # Update previous manifest for delta encoding - self.previous_manifests[manifest_id] = manifest - - return result - - def _compress_with_lean(self, manifest: Dict[str, Any], - manifest_id: str, use_delta: bool) -> str: - """Compress using Lean DeltaGCLCompression module""" - previous = None - if use_delta and manifest_id in self.previous_manifests: - previous = self.previous_manifests[manifest_id] - - # Convert manifest to PTOS format for Lean - ptos_manifest = self._to_ptos_manifest(manifest) - - # Call Lean shim - if previous: - ptos_previous = self._to_ptos_manifest(previous) - result = self.lean_shim.delta_gcl_encode(ptos_manifest, ptos_previous) - else: - result = self.lean_shim.delta_gcl_encode(ptos_manifest) - - # Extract Delta GCL sequence - delta_marker = result.get("deltaMarker", "F") - ptos_bytes = result.get("ptosBytes", "") - field_codes = result.get("fieldCodes", "") - - return f"{delta_marker}{ptos_bytes}{field_codes}" - - def _compress_with_python(self, manifest: Dict[str, Any], - manifest_id: str, use_delta: bool) -> str: - """Compress using Python fallback implementation""" - previous = None - if use_delta and manifest_id in self.previous_manifests: - previous = self.previous_manifests[manifest_id] - - return self.python_encoder.encode_to_delta_gcl(manifest, previous) - - def _to_ptos_manifest(self, manifest: Dict[str, Any]) -> Dict[str, Any]: - """Convert generic manifest to PTOS format for Lean""" - return { - "layer": manifest.get("layer", "CORE"), - "domain": manifest.get("domain", "COMPUTE"), - "tier": manifest.get("tier", "FOAM"), - "condition": manifest.get("condition", "STABLE") - } - - def verify_compression(self, compressed: str, original: Dict[str, Any]) -> tuple[bool, Optional[str]]: - """ - Verify that compressed Delta GCL can be decoded to original. - - Per canonical lock-in: Delta GCL is lawful base codec. - Verification is semantic authority - must check invariant preservation. - """ - try: - # Decode compressed Delta GCL - if self.use_lean: - # Use Lean shim for verification - decoded = self._decode_with_lean(compressed) - else: - # Use Python fallback for verification - decoded = self._decode_with_python(compressed) - - # Check structural equivalence - if not self._structural_match(decoded, original): - return False, "Structural mismatch between decoded and original" - - # Check invariant preservation - if not self._verify_invariants(decoded, original): - return False, "Invariant violation detected" - - return True, None - except Exception as e: - return False, f"Verification failed: {str(e)}" - - def _decode_with_lean(self, compressed: str) -> Dict[str, Any]: - """Decode Delta GCL using Lean shim""" - # Parse compressed format - delta_marker = compressed[0] if compressed else "F" - ptos_bytes = compressed[1:-len(compressed)+len(compressed)-len(compressed)+len(compressed)] if len(compressed) > 1 else "" - # Simplified - actual implementation would call Lean shim - return {"layer": "CORE", "domain": "COMPUTE", "tier": "FOAM", "condition": "STABLE"} - - def _decode_with_python(self, compressed: str) -> Dict[str, Any]: - """Decode Delta GCL using Python fallback""" - if self.python_encoder: - return self.python_encoder.decode_from_delta_gcl(compressed) - return {} - - def _structural_match(self, decoded: Dict[str, Any], original: Dict[str, Any]) -> bool: - """Check if decoded structurally matches original (critical fields only)""" - # Only check critical PTOS fields for structural match - # Additional metadata fields may be compressed differently - critical_fields = ["layer", "domain", "tier", "condition"] - for field in critical_fields: - if field in original and decoded.get(field) != original.get(field): - return False - return True - - def _verify_invariants(self, decoded: Dict[str, Any], original: Dict[str, Any]) -> bool: - """Verify that invariants are preserved""" - # Check that layer, domain, tier, condition are preserved - critical_fields = ["layer", "domain", "tier", "condition"] - for field in critical_fields: - if field in original and decoded.get(field) != original.get(field): - return False - return True - - def get_stats(self) -> CompressionStats: - """Get aggregate compression statistics""" - with self.lock: - return CompressionStats( - self.stats.total_compressions, - self.stats.total_original_size, - self.stats.total_compressed_size, - self.stats.avg_reduction_percent - ) - - def get_compression_history(self, limit: int = 100) -> List[CompressionResult]: - """Get recent compression history""" - with self.lock: - return self.compression_history[-limit:] - - def clear_history(self): - """Clear compression history""" - with self.lock: - self.compression_history.clear() - - def reset_stats(self): - """Reset aggregate statistics""" - with self.lock: - self.stats = CompressionStats(0, 0, 0, 0.0) - - -# Singleton instance -_service_instance: Optional[DeltaGCLCompressionService] = None -_service_lock = threading.Lock() - - -def get_compression_service() -> DeltaGCLCompressionService: - """Get singleton compression service instance""" - global _service_instance - if _service_instance is None: - with _service_lock: - if _service_instance is None: - _service_instance = DeltaGCLCompressionService() - return _service_instance - - -def compress_metadata(manifest: Dict[str, Any], manifest_id: str, - use_delta: bool = True, verify: bool = True) -> CompressionResult: - """Convenience function to compress metadata with verification""" - service = get_compression_service() - return service.compress_manifest(manifest, manifest_id, use_delta, verify) - - -if __name__ == "__main__": - # Test the service with verification - test_manifest = { - "layer": "CORE", - "domain": "COMPUTE", - "tier": "FOAM", - "condition": "STABLE", - "metadata": {"compression_ratio": 0.92, "field_phi": 0.85} - } - - service = DeltaGCLCompressionService() - - # First compression (full with verification) - print("=" * 70) - print("DELTA GCL COMPRESSION WITH VERIFICATION") - print("=" * 70) - - result1 = service.compress_manifest(test_manifest, "test_1", verify=True) - print(f"\n[1] First compression (full):") - print(f" Delta GCL: {result1.delta_gcl[:50]}...") - print(f" Stats: {result1.stats}") - print(f" Verified: {result1.verified}") - if result1.verification_error: - print(f" Verification Error: {result1.verification_error}") - - # Second compression (delta with verification) - test_manifest2 = test_manifest.copy() - test_manifest2["tier"] = "PLASMA" - result2 = service.compress_manifest(test_manifest2, "test_1", verify=True) - print(f"\n[2] Second compression (delta):") - print(f" Delta GCL: {result2.delta_gcl[:50]}...") - print(f" Stats: {result2.stats}") - print(f" Verified: {result2.verified}") - if result2.verification_error: - print(f" Verification Error: {result2.verification_error}") - - # Aggregate stats - stats = service.get_stats() - print(f"\n[3] Aggregate stats:") - print(f" Total compressions: {stats.total_compressions}") - print(f" Avg reduction: {stats.avg_reduction_percent:.2f}%") - - print("\n" + "=" * 70) - print("VERIFICATION LAYER OPERATIONAL") - print("=" * 70) diff --git a/4-Infrastructure/infra/embedded_surface/gcl_edge_in_place_upgrade.sh b/4-Infrastructure/infra/embedded_surface/gcl_edge_in_place_upgrade.sh index 48ea75ab..bee137fc 100755 --- a/4-Infrastructure/infra/embedded_surface/gcl_edge_in_place_upgrade.sh +++ b/4-Infrastructure/infra/embedded_surface/gcl_edge_in_place_upgrade.sh @@ -2,7 +2,7 @@ set -euo pipefail SERVICE_NAME="${SERVICE_NAME:-rs-surface-api.service}" -CURRENT_SERVER="${CURRENT_SERVER:-/opt/rs-surface/current/server.py}" +CURRENT_SERVER="${CURRENT_SERVER:-/opt/rs-surface/current/rs-surface}" CURRENT_PROFILE="${CURRENT_PROFILE:-/etc/rs-surface/node.json}" STATE_DIR="${STATE_DIR:-/var/lib/rs-surface}" MOUNT_DIR="${MOUNT_DIR:-/mnt/topological-storage}" @@ -14,12 +14,12 @@ HEALTH_URL="${HEALTH_URL:-http://127.0.0.1:8080/health}" usage() { cat <<'USAGE' Usage: - gcl_edge_in_place_upgrade.sh upgrade + gcl_edge_in_place_upgrade.sh upgrade gcl_edge_in_place_upgrade.sh rollback Environment overrides: SERVICE_NAME systemd service to restart, default rs-surface-api.service - CURRENT_SERVER installed server path, default /opt/rs-surface/current/server.py + CURRENT_SERVER installed server path, default /opt/rs-surface/current/rs-surface CURRENT_PROFILE installed profile path, default /etc/rs-surface/node.json STATE_DIR surface state dir, default /var/lib/rs-surface MOUNT_DIR topology mount dir, default /mnt/topological-storage @@ -60,8 +60,8 @@ validate_inputs() { local profile_src="$2" test -f "$server_src" + test -x "$server_src" test -f "$profile_src" - python3 -m py_compile "$server_src" python3 -m json.tool "$profile_src" >/dev/null } @@ -87,7 +87,7 @@ smoke_test_candidate() { RS_SURFACE_MOUNT="$MOUNT_DIR" \ RS_SURFACE_HOST="$TEMP_HOST" \ RS_SURFACE_PORT="$TEMP_PORT" \ - python3 "$server_src" >/tmp/rs-surface-upgrade-smoke.log 2>&1 & + "$server_src" >/tmp/rs-surface-upgrade-smoke.log 2>&1 & pid="$!" for _ in 1 2 3 4 5; do @@ -106,7 +106,7 @@ save_current() { local rollback_dir="$1" mkdir -p "$rollback_dir" if [ -f "$CURRENT_SERVER" ]; then - install -m 0755 "$CURRENT_SERVER" "$rollback_dir/server.py" + install -m 0755 "$CURRENT_SERVER" "$rollback_dir/rs-surface" fi if [ -f "$CURRENT_PROFILE" ]; then install -m 0644 "$CURRENT_PROFILE" "$rollback_dir/node.json" @@ -126,12 +126,12 @@ EOF restore_from() { local rollback_dir="$1" test -d "$rollback_dir" - test -f "$rollback_dir/server.py" + test -f "$rollback_dir/rs-surface" test -f "$rollback_dir/node.json" install -d -m 0755 "$(dirname "$CURRENT_SERVER")" install -d -m 0755 "$(dirname "$CURRENT_PROFILE")" - install -m 0755 "$rollback_dir/server.py" "$CURRENT_SERVER" + install -m 0755 "$rollback_dir/rs-surface" "$CURRENT_SERVER" install -m 0644 "$rollback_dir/node.json" "$CURRENT_PROFILE" systemctl daemon-reload systemctl restart "$SERVICE_NAME" diff --git a/4-Infrastructure/infra/embedded_surface/server.py b/4-Infrastructure/infra/embedded_surface/server.py deleted file mode 100644 index 78bb7fb6..00000000 --- a/4-Infrastructure/infra/embedded_surface/server.py +++ /dev/null @@ -1,545 +0,0 @@ -#!/usr/bin/env python3 -""" -Minimal embedded node surface. - -This is a Docker-testable implementation of the surface contract in -docs/specs/EMBEDDED_NODE_SURFACE_SPEC.md. It intentionally uses only Python -stdlib so tiny nodes can run it before the final embedded image exists. -""" - -from __future__ import annotations - -import base64 -import hashlib -import json -import os -import socket -import struct -import time -import zlib -from http import HTTPStatus -from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer -from pathlib import Path -from typing import Any - - -GUID = "258EAFA5-E914-47DA-95CA-C5AB0DC85B11" - -OP_HEALTH = 0 -OP_STATUS = 1 -OP_METRICS = 2 -OP_ATTEST = 3 -OP_COMPRESS = 4 -OP_RGFLOW = 5 -OP_ROUTE = 6 -OP_MOUNT_STATUS = 7 -OP_SNAPSHOT = 8 -OP_ENTER_RECOVERY = 9 -OP_PRIMITIVES = 10 -OP_PLAN_ROUTE = 11 -OP_WIKI = 12 -OP_FRACTAL_FOLD = 13 -OP_META_AUTOTYPE = 14 -OP_CREDENTIALS = 15 - -CODEC_NONE = 0 -CODEC_ZLIB_TEST = 1 - - -def env_path(name: str, default: str) -> Path: - return Path(os.environ.get(name, default)) - - -PROFILE_PATH = env_path("RS_SURFACE_PROFILE", "/etc/rs-surface/node.json") -STATE_DIR = env_path("RS_SURFACE_STATE", "/var/lib/rs-surface") -MOUNT_DIR = env_path("RS_SURFACE_MOUNT", "/mnt/topological-storage") -STARTED_AT = time.time() - - -def load_profile() -> dict[str, Any]: - with PROFILE_PATH.open("r", encoding="utf-8") as f: - return json.load(f) - - -PROFILE = load_profile() - - -def ensure_state() -> None: - STATE_DIR.mkdir(parents=True, exist_ok=True) - MOUNT_DIR.mkdir(parents=True, exist_ok=True) - node_id = STATE_DIR / "node-id" - if not node_id.exists(): - node_id.write_text(PROFILE["node_id"] + "\n", encoding="utf-8") - last_good = STATE_DIR / "last-good.json" - if not last_good.exists(): - last_good.write_text( - json.dumps({"ok": True, "created_at": time.time()}, sort_keys=True) + "\n", - encoding="utf-8", - ) - - -def storage_status() -> str: - marker = MOUNT_DIR / ".rs-surface-mounted" - if marker.exists(): - return "mounted" - if MOUNT_DIR.exists(): - return "degraded" - return "absent" - - -def health_payload() -> dict[str, Any]: - return { - "ok": True, - "node": PROFILE["node_id"], - "role": PROFILE["role"], - "mode": PROFILE.get("mode_default", "normal"), - "surface_version": PROFILE["surface_version"], - "storage": storage_status(), - "last_good": (STATE_DIR / "last-good.json").exists(), - "uptime_seconds": round(time.time() - STARTED_AT, 3), - } - - -def status_payload() -> dict[str, Any]: - return { - "profile": PROFILE, - "state_dir": str(STATE_DIR), - "mount_dir": str(MOUNT_DIR), - "hostname": socket.gethostname(), - } - - -def metrics_payload() -> dict[str, Any]: - state_bytes = 0 - for path in STATE_DIR.rglob("*"): - if path.is_file(): - state_bytes += path.stat().st_size - return { - "uptime_seconds": round(time.time() - STARTED_AT, 3), - "state_bytes": state_bytes, - "state_budget_mb": PROFILE.get("local_state_budget_mb"), - } - - -def primitive_payload() -> dict[str, Any]: - substrate = PROFILE.get("topological_substrate", {}) - primitives = substrate.get("primitives") - if primitives is None: - primitives = [ - "health", - "status", - "metrics", - "attest", - "compress", - "rgflow", - "route", - "mount_status", - "snapshot", - "recovery", - "plan_route", - "wiki", - "fractal_fold", - "meta_autotype", - "credentials", - ] - return { - "node": PROFILE["node_id"], - "role": PROFILE["role"], - "substrate": substrate, - "primitives": primitives, - } - - -def resolve_bind_host() -> str: - override = os.environ.get("RS_SURFACE_HOST") - if override: - return override - - api = PROFILE.get("api", {}) - bind = api.get("bind", "public") - if bind == "localhost": - return "127.0.0.1" - if bind == "public": - return "0.0.0.0" - if bind == "tailscale": - host = api.get("tailscale_ip") or PROFILE.get("tailscale_ip") - if not host: - raise RuntimeError("api.bind=tailscale requires api.tailscale_ip or top-level tailscale_ip") - return str(host) - raise RuntimeError(f"unsupported api.bind {bind!r}") - - -def rgflow_score(data: bytes) -> dict[str, Any]: - if not data: - return {"lawful": True, "score": 1.0, "reason": "empty-control-frame"} - unique = len(set(data)) - density = unique / 256.0 - compressed = zlib.compress(data, level=6) - ratio = len(compressed) / max(1, len(data)) - lawful = ratio < 0.98 or len(data) < 512 - return { - "lawful": lawful, - "score": round(max(0.0, 1.0 - ratio + density), 6), - "compression_ratio": round(ratio, 6), - "byte_diversity": round(density, 6), - "reason": "test-rgflow-heuristic", - } - - -def canonical_json(obj: Any) -> bytes: - return json.dumps(obj, sort_keys=True, separators=(",", ":")).encode("utf-8") - - -def decode_payload(codec: int, payload: bytes) -> bytes: - if codec == CODEC_NONE: - return payload - if codec == CODEC_ZLIB_TEST: - return zlib.decompress(payload) - raise ValueError(f"unsupported codec {codec}") - - -def encode_payload(codec: int, payload: bytes) -> bytes: - if codec == CODEC_NONE: - return payload - if codec == CODEC_ZLIB_TEST: - return zlib.compress(payload, level=6) - raise ValueError(f"unsupported codec {codec}") - - -def handle_surface_op(op: int, payload: bytes) -> dict[str, Any]: - if op == OP_HEALTH: - return health_payload() - if op == OP_STATUS: - return status_payload() - if op == OP_METRICS: - return metrics_payload() - if op == OP_ATTEST: - return { - "sha256": hashlib.sha256(payload).hexdigest(), - "bytes": len(payload), - "node": PROFILE["node_id"], - } - if op == OP_COMPRESS: - compressed = zlib.compress(payload, level=6) - return { - "codec": "zlib-test", - "raw_bytes": len(payload), - "compressed_bytes": len(compressed), - "ratio": round(len(compressed) / max(1, len(payload)), 6), - "payload_b64": base64.b64encode(compressed).decode("ascii"), - } - if op == OP_RGFLOW: - return rgflow_score(payload) - if op == OP_ROUTE: - return { - "accepted": rgflow_score(payload)["lawful"], - "route": "local" if len(payload) < 4096 else "atlas", - } - if op == OP_MOUNT_STATUS: - return { - "storage": storage_status(), - "mount_point": str(MOUNT_DIR), - "provider": PROFILE["storage"]["provider"], - "required_for_boot": PROFILE["storage"]["required_for_boot"], - } - if op == OP_SNAPSHOT: - digest = hashlib.sha256(payload).hexdigest() - snapshot = STATE_DIR / "snapshot-last.json" - snapshot.write_text( - json.dumps({"sha256": digest, "bytes": len(payload), "t": time.time()}, sort_keys=True) + "\n", - encoding="utf-8", - ) - return {"snapshotted": True, "sha256": digest} - if op == OP_ENTER_RECOVERY: - return {"accepted": False, "reason": "recovery transition disabled in test image"} - if op == OP_PRIMITIVES: - return primitive_payload() - if op == OP_PLAN_ROUTE: - try: - from omni_lut.unified_compression_route import choose_route - except ModuleNotFoundError: - import sys - - sys.path.insert(0, str(Path(__file__).resolve().parent)) - from omni_lut.unified_compression_route import choose_route - - return choose_route(payload) - if op == OP_WIKI: - try: - request = json.loads(payload.decode("utf-8")) if payload else {"op": "recent"} - if not isinstance(request, dict): - raise ValueError("wiki payload must be a JSON object") - use_rds = os.environ.get("RS_USE_RDS", "").lower() in ("1", "true", "yes") - if use_rds: - try: - from infra.ene_rds_wiki_layer import ENERDSWikiLayer - except ModuleNotFoundError: - import sys - sys.path.insert(0, str(Path(__file__).resolve().parents[2])) - from infra.ene_rds_wiki_layer import ENERDSWikiLayer - - wiki = ENERDSWikiLayer() - else: - try: - from infra.ene_wiki_layer import ENEWikiLayer - except ModuleNotFoundError: - import sys - sys.path.insert(0, str(Path(__file__).resolve().parents[2])) - from infra.ene_wiki_layer import ENEWikiLayer - - wiki = ENEWikiLayer(STATE_DIR / "ene-wiki.db") - return wiki.handle_request(request) - except Exception as exc: - return {"ok": False, "op": "wiki", "error": str(exc)} - if op == OP_FRACTAL_FOLD: - try: - request = json.loads(payload.decode("utf-8")) if payload else {"op": "manifest"} - if not isinstance(request, dict): - raise ValueError("fractal_fold payload must be a JSON object") - use_rds = os.environ.get("RS_USE_RDS", "").lower() in ("1", "true", "yes") - if use_rds: - try: - from infra.ene_rds_fractal_fold import ENERDSFractalFold - except ModuleNotFoundError: - import sys - sys.path.insert(0, str(Path(__file__).resolve().parents[2])) - from infra.ene_rds_fractal_fold import ENERDSFractalFold - - fold = ENERDSFractalFold() - else: - try: - from infra.ene_fractal_fold import ENEFractalFold - except ModuleNotFoundError: - import sys - sys.path.insert(0, str(Path(__file__).resolve().parents[2])) - from infra.ene_fractal_fold import ENEFractalFold - - fold = ENEFractalFold(STATE_DIR / "ene-fractal-fold.db") - return fold.handle_request(request) - except Exception as exc: - return {"ok": False, "op": "fractal_fold", "error": str(exc)} - if op == OP_META_AUTOTYPE: - try: - request = json.loads(payload.decode("utf-8")) if payload else {"text": ""} - if not isinstance(request, dict): - raise ValueError("meta_autotype payload must be a JSON object") - try: - from infra.ene_meta_autotype import handle_request - except ModuleNotFoundError: - import sys - - sys.path.insert(0, str(Path(__file__).resolve().parents[2])) - from infra.ene_meta_autotype import handle_request - - return handle_request(request) - except Exception as exc: - return {"ok": False, "op": "meta_autotype", "error": str(exc)} - if op == OP_CREDENTIALS: - try: - request = json.loads(payload.decode("utf-8")) if payload else {} - if not isinstance(request, dict): - raise ValueError("credentials payload must be a JSON object") - try: - from credential_provider import ( - credential_status, - provider_manifest, - resolve_credential, - ) - except ModuleNotFoundError: - try: - from infra.credential_provider import ( - credential_status, - provider_manifest, - resolve_credential, - ) - except ModuleNotFoundError: - import sys - - sys.path.insert(0, str(Path(__file__).resolve().parent)) - from credential_provider import ( - credential_status, - provider_manifest, - resolve_credential, - ) - - action = request.get("action", "status") - if action == "status": - return credential_status() - if action == "manifest": - return provider_manifest() - if action == "resolve": - provider = request.get("provider", "") - if not provider: - return {"ok": False, "error": "missing provider name"} - cred = resolve_credential(provider) - if cred is None: - return {"ok": False, "error": f"provider {provider!r} not available"} - return {"ok": True, "provider": cred.provider, "value": cred.value} - return {"ok": False, "error": f"unknown credentials action {action!r}"} - except Exception as exc: - return {"ok": False, "op": "credentials", "error": str(exc)} - return {"error": "unknown-op", "op": op} - - -def parse_surface_frame(data: bytes) -> tuple[int, int, int, bytes]: - if len(data) < 16: - raise ValueError("surface frame too short") - version, _flags, codec, op = data[:4] - if version != 1: - raise ValueError(f"unsupported version {version}") - request_id, payload_len, crc_expected = struct.unpack(" bytes: - raw_payload = canonical_json(payload_obj) - payload = encode_payload(codec, raw_payload) - header = struct.pack( - " bytes: - chunks = [] - remaining = n - while remaining: - chunk = sock.recv(remaining) - if not chunk: - raise EOFError("socket closed") - chunks.append(chunk) - remaining -= len(chunk) - return b"".join(chunks) - - -def read_ws_message(sock: socket.socket) -> bytes: - b1, b2 = read_exact(sock, 2) - opcode = b1 & 0x0F - masked = bool(b2 & 0x80) - length = b2 & 0x7F - if opcode == 0x8: - raise EOFError("websocket close") - if length == 126: - length = struct.unpack("!H", read_exact(sock, 2))[0] - elif length == 127: - length = struct.unpack("!Q", read_exact(sock, 8))[0] - mask = read_exact(sock, 4) if masked else b"" - payload = bytearray(read_exact(sock, length)) - if masked: - for i in range(length): - payload[i] ^= mask[i % 4] - return bytes(payload) - - -def write_ws_message(sock: socket.socket, payload: bytes) -> None: - header = bytearray([0x82]) - length = len(payload) - if length < 126: - header.append(length) - elif length <= 0xFFFF: - header.append(126) - header.extend(struct.pack("!H", length)) - else: - header.append(127) - header.extend(struct.pack("!Q", length)) - sock.sendall(bytes(header) + payload) - - -class SurfaceHandler(BaseHTTPRequestHandler): - server_version = "rs-surface/0.1" - - def log_message(self, fmt: str, *args: Any) -> None: - print(f"{self.address_string()} - {fmt % args}", flush=True) - - def send_json(self, obj: Any, status: int = 200) -> None: - payload = json.dumps(obj, sort_keys=True).encode("utf-8") - self.send_response(status) - self.send_header("content-type", "application/json") - self.send_header("content-length", str(len(payload))) - self.end_headers() - self.wfile.write(payload) - - def do_GET(self) -> None: - if self.path == "/health": - self.send_json(health_payload()) - return - if self.path == "/status": - self.send_json(status_payload()) - return - if self.path == "/metrics": - self.send_json(metrics_payload()) - return - if self.path == "/primitives": - self.send_json(primitive_payload()) - return - if self.path == "/ws": - self.handle_ws() - return - if self.path == "/credentials": - try: - from credential_provider import credential_status - except ModuleNotFoundError: - try: - from infra.credential_provider import credential_status - except ModuleNotFoundError: - import sys - - sys.path.insert(0, str(Path(__file__).resolve().parent)) - from credential_provider import credential_status - - self.send_json(credential_status()) - return - self.send_json({"error": "not-found"}, HTTPStatus.NOT_FOUND) - - def handle_ws(self) -> None: - key = self.headers.get("Sec-WebSocket-Key") - if not key: - self.send_json({"error": "missing-websocket-key"}, HTTPStatus.BAD_REQUEST) - return - accept = base64.b64encode(hashlib.sha1((key + GUID).encode("ascii")).digest()).decode("ascii") - self.send_response(HTTPStatus.SWITCHING_PROTOCOLS) - self.send_header("Upgrade", "websocket") - self.send_header("Connection", "Upgrade") - self.send_header("Sec-WebSocket-Accept", accept) - self.end_headers() - - sock = self.connection - while True: - try: - data = read_ws_message(sock) - request_id, codec, op, payload = parse_surface_frame(data) - result = handle_surface_op(op, payload) - write_ws_message(sock, build_surface_frame(request_id, op, result, codec=codec)) - except EOFError: - return - except Exception as exc: - frame = build_surface_frame(0, OP_STATUS, {"error": str(exc)}, codec=CODEC_NONE) - write_ws_message(sock, frame) - return - - -def main() -> int: - ensure_state() - host = resolve_bind_host() - port = int(os.environ.get("RS_SURFACE_PORT", str(PROFILE.get("api", {}).get("plain_health_port", 8080)))) - httpd = ThreadingHTTPServer((host, port), SurfaceHandler) - print(f"rs-surface listening on {host}:{port} node={PROFILE['node_id']}", flush=True) - httpd.serve_forever() - return 0 - - -if __name__ == "__main__": - raise SystemExit(main()) diff --git a/4-Infrastructure/infra/embedded_surface/xen_alpine/install_rs_surface_openrc.sh b/4-Infrastructure/infra/embedded_surface/xen_alpine/install_rs_surface_openrc.sh index 27cf4e63..cc03143e 100755 --- a/4-Infrastructure/infra/embedded_surface/xen_alpine/install_rs_surface_openrc.sh +++ b/4-Infrastructure/infra/embedded_surface/xen_alpine/install_rs_surface_openrc.sh @@ -11,7 +11,7 @@ RUN_DIR="${RUN_DIR:-/run/rs-surface}" MOUNT_DIR="${MOUNT_DIR:-/mnt/topological-storage}" SOURCE_ROOT="${SOURCE_ROOT:-$(cd "$(dirname "$0")/.." && pwd)}" PROFILE_SRC="${PROFILE_SRC:-$SOURCE_ROOT/profiles/xen-alpine-surface.json}" -SERVER_SRC="${SERVER_SRC:-$SOURCE_ROOT/server.py}" +SERVER_SRC="${SERVER_SRC:-$SOURCE_ROOT/rs-surface/target/x86_64-unknown-linux-musl/release/rs-surface}" SERVICE_SRC="${SERVICE_SRC:-$(cd "$(dirname "$0")" && pwd)/rs-surface.openrc}" need_root() { @@ -23,7 +23,7 @@ need_root() { install_packages() { if command -v apk >/dev/null 2>&1; then - apk add --no-cache python3 ca-certificates >/dev/null + apk add --no-cache ca-certificates >/dev/null fi } @@ -38,11 +38,12 @@ ensure_user() { install_surface() { test -f "$SERVER_SRC" + test -x "$SERVER_SRC" test -f "$PROFILE_SRC" test -f "$SERVICE_SRC" mkdir -p "$INSTALL_ROOT" "$CONFIG_DIR" "$STATE_DIR" "$LOG_DIR" "$RUN_DIR" "$MOUNT_DIR" - install -m 0755 "$SERVER_SRC" "$INSTALL_ROOT/server.py" + install -m 0755 "$SERVER_SRC" "$INSTALL_ROOT/rs-surface" install -m 0644 "$PROFILE_SRC" "$CONFIG_DIR/node.json" install -m 0755 "$SERVICE_SRC" /etc/init.d/rs-surface chown -R "$SERVICE_USER:$SERVICE_GROUP" "$STATE_DIR" "$LOG_DIR" "$RUN_DIR" "$MOUNT_DIR" @@ -62,7 +63,7 @@ enable_service cat < "$APKOVL_DIR/root/opt/rs-surface/boot-rs-surface.sh" <<'EOF' #!/bin/sh set -eu -echo "rs-surface qemu bootstrap starting" >&2 -if ! command -v python3 >/dev/null 2>&1; then - apk add --no-cache python3 ca-certificates >&2 -fi +echo "rs-surface (Rust) qemu bootstrap starting" >&2 mkdir -p /var/lib/rs-surface /var/log/rs-surface /run/rs-surface /mnt/topological-storage export RS_SURFACE_PROFILE="${RS_SURFACE_PROFILE:-/etc/rs-surface/node.json}" export RS_SURFACE_STATE="${RS_SURFACE_STATE:-/var/lib/rs-surface}" export RS_SURFACE_MOUNT="${RS_SURFACE_MOUNT:-/mnt/topological-storage}" export RS_SURFACE_HOST="${RS_SURFACE_HOST:-0.0.0.0}" export RS_SURFACE_PORT="${RS_SURFACE_PORT:-8080}" -exec python3 /opt/rs-surface/server.py +exec /opt/rs-surface/rs-surface EOF fi diff --git a/4-Infrastructure/infra/ene-session-sync/src/deepseek_adapter.rs b/4-Infrastructure/infra/ene-session-sync/src/deepseek_adapter.rs new file mode 100644 index 00000000..1bd62dc7 --- /dev/null +++ b/4-Infrastructure/infra/ene-session-sync/src/deepseek_adapter.rs @@ -0,0 +1,237 @@ +#![allow(dead_code)] +//! deepseek_adapter.rs — DeepSeek / Ollama chat adapter. +//! +//! Port of deepseek_adapter.py (113 lines). +//! Provides a unified client that routes to either a local Ollama instance or +//! the DeepSeek remote API, plus a `DeepSeekProver` shim for Lean 4 +//! formalization requests. + +use anyhow::{Context, Result}; +use serde_json::{json, Value}; + +// ───────────────────────────────────────────────────────────────────────────── +// §1 DeepSeekV4 client +// ───────────────────────────────────────────────────────────────────────────── + +/// Unified chat client for DeepSeek / Ollama. +/// +/// Routing: +/// - `use_local = true` → Ollama REST API at `local_url` +/// - `use_local = false` → DeepSeek remote API at `api_base` with bearer auth +pub struct DeepSeekV4 { + api_key: Option, + local_url: String, + use_local: bool, + api_base: String, + http: reqwest::Client, +} + +impl DeepSeekV4 { + /// Explicit constructor. + pub fn new(api_key: Option, local_url: impl Into, use_local: bool) -> Self { + Self { + api_key, + local_url: local_url.into(), + use_local, + api_base: "https://api.deepseek.com/v1".to_string(), + http: reqwest::Client::new(), + } + } + + /// Build from environment. + /// + /// Reads `DEEPSEEK_API_KEY`. Defaults to `localhost:11434` and + /// `use_local = true` so the crate works offline without any key. + pub fn from_env() -> Self { + let api_key = std::env::var("DEEPSEEK_API_KEY").ok(); + let local_url = std::env::var("OLLAMA_URL") + .unwrap_or_else(|_| "http://localhost:11434".to_string()); + // If an API key was provided, prefer the remote endpoint. + let use_local = api_key.is_none(); + Self::new(api_key, local_url, use_local) + } + + /// Send a chat request and return the raw JSON response body. + /// + /// `stream` is accepted for API symmetry but always forced to `false` — + /// streaming responses require a different response-parsing path that is + /// not needed by any current caller. + pub async fn chat( + &self, + messages: &[Value], + model: &str, + _stream: bool, + ) -> Result { + if self.use_local { + self.chat_local(messages, model).await + } else { + self.chat_remote(messages, model).await + } + } + + // ── Private helpers ─────────────────────────────────────────────────────── + + async fn chat_local(&self, messages: &[Value], model: &str) -> Result { + let url = format!("{}/api/chat", self.local_url); + let payload = json!({ + "model": model, + "messages": messages, + "stream": false, + }); + let resp = self + .http + .post(&url) + .json(&payload) + .send() + .await + .with_context(|| format!("POST {}", url))?; + let status = resp.status(); + let body: Value = resp.json().await.context("parse ollama response JSON")?; + if !status.is_success() { + anyhow::bail!("ollama returned {}: {:?}", status, body); + } + Ok(body) + } + + async fn chat_remote(&self, messages: &[Value], model: &str) -> Result { + let url = format!("{}/chat/completions", self.api_base); + let payload = json!({ + "model": model, + "messages": messages, + "stream": false, + }); + let mut req = self.http.post(&url).json(&payload); + if let Some(ref key) = self.api_key { + req = req.bearer_auth(key); + } + let resp = req + .send() + .await + .with_context(|| format!("POST {}", url))?; + let status = resp.status(); + let body: Value = resp.json().await.context("parse deepseek response JSON")?; + if !status.is_success() { + anyhow::bail!("deepseek API returned {}: {:?}", status, body); + } + Ok(body) + } + + /// Extract the assistant text content from a chat response. + /// + /// Handles both Ollama format (`response.message.content`) and OpenAI + /// format (`choices[0].message.content`). + pub fn extract_content(resp: &Value) -> String { + // Ollama: { "message": { "content": "..." } } + if let Some(content) = resp + .get("message") + .and_then(|m| m.get("content")) + .and_then(|c| c.as_str()) + { + return content.to_string(); + } + // OpenAI / DeepSeek: { "choices": [{ "message": { "content": "..." } }] } + if let Some(content) = resp + .get("choices") + .and_then(|c| c.get(0)) + .and_then(|c| c.get("message")) + .and_then(|m| m.get("content")) + .and_then(|c| c.as_str()) + { + return content.to_string(); + } + String::new() + } +} + +// ───────────────────────────────────────────────────────────────────────────── +// §2 DeepSeekProver — Lean 4 formalization wrapper +// ───────────────────────────────────────────────────────────────────────────── + +/// High-level prover shim that wraps `DeepSeekV4` for Lean 4 formalization. +pub struct DeepSeekProver { + client: DeepSeekV4, +} + +impl DeepSeekProver { + pub fn new(client: DeepSeekV4) -> Self { + Self { client } + } + + /// Build from environment (delegates to `DeepSeekV4::from_env`). + pub fn from_env() -> Self { + Self::new(DeepSeekV4::from_env()) + } + + /// Ask the model to produce a Lean 4 formalization of `statement`. + /// + /// Returns the raw string content from the assistant turn. + pub async fn formalize(&self, statement: &str) -> Result { + let system_prompt = "You are an expert in formal mathematics and Lean 4. \ + Given an informal mathematical statement, produce a complete and \ + correct Lean 4 formalization. Output only valid Lean 4 code, \ + no explanations."; + + let user_prompt = format!( + "Formalize the following mathematical statement in Lean 4:\n\n{}", + statement + ); + + let messages = vec![ + json!({ "role": "system", "content": system_prompt }), + json!({ "role": "user", "content": user_prompt }), + ]; + + // Model selection: local Ollama vs. remote DeepSeek. + let model = if self.client.use_local { + "qwen2.5-coder:14b" + } else { + "deepseek-v4-pro" + }; + + let resp = self + .client + .chat(&messages, model, false) + .await + .context("formalize chat call")?; + + Ok(DeepSeekV4::extract_content(&resp)) + } +} + +// ───────────────────────────────────────────────────────────────────────────── +// §3 Tests +// ───────────────────────────────────────────────────────────────────────────── + +#[cfg(test)] +mod tests { + use super::*; + use serde_json::json; + + #[test] + fn extract_content_ollama_format() { + let resp = json!({ "message": { "content": "hello" } }); + assert_eq!(DeepSeekV4::extract_content(&resp), "hello"); + } + + #[test] + fn extract_content_openai_format() { + let resp = json!({ + "choices": [{ "message": { "content": "world" } }] + }); + assert_eq!(DeepSeekV4::extract_content(&resp), "world"); + } + + #[test] + fn extract_content_empty() { + let resp = json!({}); + assert_eq!(DeepSeekV4::extract_content(&resp), ""); + } + + #[test] + fn from_env_defaults() { + // Should not panic even without env vars set. + let client = DeepSeekV4::from_env(); + assert!(client.use_local); + assert!(client.api_key.is_none() || client.api_key.is_some()); + } +} diff --git a/4-Infrastructure/infra/ene-session-sync/src/ene_cloud_credential_manager.rs b/4-Infrastructure/infra/ene-session-sync/src/ene_cloud_credential_manager.rs new file mode 100644 index 00000000..6f8a7f1b --- /dev/null +++ b/4-Infrastructure/infra/ene-session-sync/src/ene_cloud_credential_manager.rs @@ -0,0 +1,647 @@ +#![allow(dead_code)] +//! ene_cloud_credential_manager.rs — ENE node balancer and cloud credential +//! manager with SQLite persistence. +//! +//! Port of ene_cloud_credential_manager.py (656 lines). + +use anyhow::{Context, Result}; +use rusqlite::{params, Connection}; +use serde_json::Value; +use sha2::{Digest, Sha256}; +use std::collections::HashMap; +use std::time::{SystemTime, UNIX_EPOCH}; + +// ───────────────────────────────────────────────────────────────────────────── +// §0 Shared helpers +// ───────────────────────────────────────────────────────────────────────────── + +/// Current wall-clock time in whole seconds since the UNIX epoch. +fn now_secs() -> i64 { + SystemTime::now() + .duration_since(UNIX_EPOCH) + .unwrap_or_default() + .as_secs() as i64 +} + +/// SHA-256 of `data`; returns the first 16 lowercase hex characters. +fn sha256_hex16(data: &[u8]) -> String { + let mut h = Sha256::new(); + h.update(data); + hex::encode(&h.finalize()[..8]) // 8 bytes → 16 hex chars +} + +// ───────────────────────────────────────────────────────────────────────────── +// §1 NodeStats / NodeConnection +// ───────────────────────────────────────────────────────────────────────────── + +/// Aggregated statistics for a registered ENE node. +#[derive(Debug, Clone)] +pub struct NodeStats { + pub node_id: String, + pub total_connections: i64, + pub total_bytes: i64, + pub avg_latency: f64, + pub error_rate: f64, + /// In [0.0, 1.0]. Higher is healthier. + pub health_score: f64, +} + +impl NodeStats { + fn new(node_id: &str) -> Self { + Self { + node_id: node_id.to_owned(), + total_connections: 0, + total_bytes: 0, + avg_latency: 0.0, + error_rate: 0.0, + health_score: 1.0, + } + } +} + +/// A live (or recently closed) connection to an ENE node. +#[derive(Debug, Clone)] +pub struct NodeConnection { + pub node_id: String, + pub credential_id: String, + pub connected_at: f64, + pub last_activity: f64, + pub bytes_transferred: i64, + pub error_count: i64, + pub latency_ms: f64, + pub status: String, +} + +// ───────────────────────────────────────────────────────────────────────────── +// §2 ENENodeBalancer +// ───────────────────────────────────────────────────────────────────────────── + +/// Load-balancing dispatcher for ENE nodes with SQLite-backed statistics. +pub struct ENENodeBalancer { + db_path: String, + /// In-memory shadow of the `nodes` table (node_id → stats). + pub nodes: HashMap, + /// Connections that have been recorded but not yet closed. + pub active_connections: HashMap, +} + +impl ENENodeBalancer { + // ── §2.1 Constructor ──────────────────────────────────────────────────── + + /// Open (or create) the SQLite database at `db_path`, ensure the schema + /// exists, and load all previously-registered nodes into memory. + pub fn new(db_path: &str) -> Result { + let mut balancer = Self { + db_path: db_path.to_owned(), + nodes: HashMap::new(), + active_connections: HashMap::new(), + }; + balancer.init_tables()?; + balancer.load_nodes()?; + Ok(balancer) + } + + fn open(&self) -> Result { + Connection::open(&self.db_path) + .with_context(|| format!("ENENodeBalancer: open db {:?}", self.db_path)) + } + + fn init_tables(&self) -> Result<()> { + let conn = self.open()?; + conn.execute_batch( + "CREATE TABLE IF NOT EXISTS nodes ( + node_id TEXT PRIMARY KEY, + credential_id TEXT NOT NULL, + total_connections INTEGER NOT NULL DEFAULT 0, + total_bytes INTEGER NOT NULL DEFAULT 0, + avg_latency REAL NOT NULL DEFAULT 0.0, + error_rate REAL NOT NULL DEFAULT 0.0, + health_score REAL NOT NULL DEFAULT 1.0, + registered_at INTEGER NOT NULL + ); + CREATE TABLE IF NOT EXISTS connections ( + connection_id TEXT PRIMARY KEY, + node_id TEXT NOT NULL, + credential_id TEXT NOT NULL, + connected_at REAL NOT NULL, + last_activity REAL NOT NULL, + bytes_transferred INTEGER NOT NULL DEFAULT 0, + error_count INTEGER NOT NULL DEFAULT 0, + latency_ms REAL NOT NULL DEFAULT 0.0, + status TEXT NOT NULL DEFAULT 'active' + );", + ) + .context("ENENodeBalancer: init_tables")?; + Ok(()) + } + + fn load_nodes(&mut self) -> Result<()> { + let conn = self.open()?; + let mut stmt = conn.prepare( + "SELECT node_id, total_connections, total_bytes, avg_latency, error_rate, health_score + FROM nodes", + )?; + let rows = stmt.query_map([], |row| { + Ok(NodeStats { + node_id: row.get(0)?, + total_connections: row.get(1)?, + total_bytes: row.get(2)?, + avg_latency: row.get(3)?, + error_rate: row.get(4)?, + health_score: row.get(5)?, + }) + })?; + for row in rows { + let stats = row?; + self.nodes.insert(stats.node_id.clone(), stats); + } + Ok(()) + } + + // ── §2.2 Node management ──────────────────────────────────────────────── + + /// Register a new node; returns `true` if inserted, `false` if it already + /// existed. + pub fn register_node(&mut self, node_id: &str, credential_id: &str) -> Result { + if self.nodes.contains_key(node_id) { + return Ok(false); + } + let conn = self.open()?; + conn.execute( + "INSERT OR IGNORE INTO nodes + (node_id, credential_id, registered_at) + VALUES (?1, ?2, ?3)", + params![node_id, credential_id, now_secs()], + ) + .context("ENENodeBalancer: register_node")?; + self.nodes.insert(node_id.to_owned(), NodeStats::new(node_id)); + Ok(true) + } + + // ── §2.3 Node selection ───────────────────────────────────────────────── + + /// Pick a node according to the named strategy. + /// + /// | Strategy | Description | + /// |--------------------|----------------------------------------------| + /// | `health_weighted` | Probabilistic, weighted by `health_score`. | + /// | `round_robin` | First node in sorted order. | + /// | `latency` | Node with the lowest `avg_latency`. | + /// | `least_connections`| Node with the fewest `total_connections`. | + /// + /// Returns `None` when no nodes are registered. + pub fn select_node(&self, strategy: &str) -> Option { + if self.nodes.is_empty() { + return None; + } + + // Collect a stable-ordered snapshot. + let mut nodes_vec: Vec<&NodeStats> = self.nodes.values().collect(); + nodes_vec.sort_by(|a, b| a.node_id.cmp(&b.node_id)); + + match strategy { + "health_weighted" => self.select_health_weighted(&nodes_vec), + "round_robin" => nodes_vec.first().map(|n| n.node_id.clone()), + "latency" => nodes_vec + .iter() + .min_by(|a, b| { + a.avg_latency + .partial_cmp(&b.avg_latency) + .unwrap_or(std::cmp::Ordering::Equal) + }) + .map(|n| n.node_id.clone()), + "least_connections" => nodes_vec + .iter() + .min_by_key(|n| n.total_connections) + .map(|n| n.node_id.clone()), + _ => nodes_vec.first().map(|n| n.node_id.clone()), + } + } + + /// Health-weighted selection using a hash-based pseudo-random threshold. + /// + /// 1. Sort nodes by health_score descending. + /// 2. Scale health scores to integers (×1000) to get total_weight. + /// 3. Hash `(sorted_ids joined by "+") + now_secs` → u64 seed. + /// 4. `threshold = seed % total_weight`. + /// 5. Walk cumulative sums; return first node where cumulative ≥ threshold. + fn select_health_weighted(&self, nodes: &[&NodeStats]) -> Option { + // Sort descending by health so healthiest nodes are visited first. + let mut sorted: Vec<&NodeStats> = nodes.to_vec(); + sorted.sort_by(|a, b| { + b.health_score + .partial_cmp(&a.health_score) + .unwrap_or(std::cmp::Ordering::Equal) + }); + + let weights: Vec = sorted + .iter() + .map(|n| (n.health_score.max(0.0) * 1000.0) as u64 + 1) + .collect(); + let total_weight: u64 = weights.iter().sum(); + if total_weight == 0 { + return sorted.first().map(|n| n.node_id.clone()); + } + + // Derive a pseudo-random threshold from the node IDs + current time. + let ids_str = sorted + .iter() + .map(|n| n.node_id.as_str()) + .collect::>() + .join("+"); + let seed_input = format!("{}{}", ids_str, now_secs()); + let mut h = Sha256::new(); + h.update(seed_input.as_bytes()); + let digest = h.finalize(); + let seed_bytes: [u8; 8] = digest[..8].try_into().unwrap_or([0u8; 8]); + let seed = u64::from_le_bytes(seed_bytes); + let threshold = seed % total_weight; + + let mut cumulative = 0u64; + for (node, &w) in sorted.iter().zip(weights.iter()) { + cumulative += w; + if cumulative >= threshold { + return Some(node.node_id.clone()); + } + } + // Fallback (should not be reached). + sorted.first().map(|n| n.node_id.clone()) + } + + // ── §2.4 Connection lifecycle ─────────────────────────────────────────── + + /// Record a new connection to `node_id`, returning a unique `connection_id`. + /// + /// The `connection_id` is the first 16 hex chars of `SHA-256(node_id + + /// timestamp_secs_as_str)`. + pub fn record_connection( + &mut self, + node_id: &str, + credential_id: &str, + ) -> Result { + let now = now_secs() as f64; + let connection_id = + sha256_hex16(format!("{}{}", node_id, now as i64).as_bytes()); + + let conn_rec = NodeConnection { + node_id: node_id.to_owned(), + credential_id: credential_id.to_owned(), + connected_at: now, + last_activity: now, + bytes_transferred: 0, + error_count: 0, + latency_ms: 0.0, + status: "active".to_owned(), + }; + + let db = self.open()?; + db.execute( + "INSERT OR REPLACE INTO connections + (connection_id, node_id, credential_id, connected_at, last_activity, status) + VALUES (?1, ?2, ?3, ?4, ?5, 'active')", + params![connection_id, node_id, credential_id, now, now], + ) + .context("ENENodeBalancer: record_connection")?; + + // Update in-memory node stats. + if let Some(stats) = self.nodes.get_mut(node_id) { + stats.total_connections += 1; + } + + self.active_connections + .insert(connection_id.clone(), conn_rec); + Ok(connection_id) + } + + /// Update bytes-transferred and latency for an active connection. + pub fn update_connection_stats( + &mut self, + connection_id: &str, + bytes_delta: i64, + latency_ms: f64, + error: bool, + ) { + let now = now_secs() as f64; + if let Some(c) = self.active_connections.get_mut(connection_id) { + c.bytes_transferred += bytes_delta; + c.last_activity = now; + // Rolling average latency (simple mean). + c.latency_ms = (c.latency_ms + latency_ms) / 2.0; + if error { + c.error_count += 1; + } + + // Mirror to node stats. + if let Some(stats) = self.nodes.get_mut(&c.node_id.clone()) { + stats.total_bytes += bytes_delta; + stats.avg_latency = (stats.avg_latency + latency_ms) / 2.0; + if error && c.error_count > 0 { + stats.error_rate = c.error_count as f64 + / (c.error_count + stats.total_connections) as f64; + } + } + } + } + + /// Mark a connection as closed and persist final statistics. + pub fn close_connection(&mut self, connection_id: &str) -> Result<()> { + let db = self.open()?; + db.execute( + "UPDATE connections SET status = 'closed' WHERE connection_id = ?1", + params![connection_id], + ) + .context("ENENodeBalancer: close_connection")?; + self.active_connections.remove(connection_id); + Ok(()) + } + + // ── §2.5 Health check ─────────────────────────────────────────────────── + + /// Perform a (synthetic) health check for `node_id`. + /// + /// - If `error_rate < 0.1` → `health_score += 0.05` (clamp 1.0). + /// - Otherwise → `health_score -= 0.1` (clamp 0.0). + /// + /// Returns `true` when the node is considered healthy after the update. + pub fn health_check(&mut self, node_id: &str) -> Result { + let healthy = { + if let Some(stats) = self.nodes.get_mut(node_id) { + if stats.error_rate < 0.1 { + stats.health_score = (stats.health_score + 0.05).min(1.0); + true + } else { + stats.health_score = (stats.health_score - 0.1).max(0.0); + false + } + } else { + anyhow::bail!("health_check: unknown node '{}'", node_id); + } + }; + + // Persist the updated health score. + let score = self.nodes[node_id].health_score; + let db = self.open()?; + db.execute( + "UPDATE nodes SET health_score = ?1 WHERE node_id = ?2", + params![score, node_id], + ) + .context("ENENodeBalancer: health_check persist")?; + + Ok(healthy) + } + + // ── §2.6 Introspection ────────────────────────────────────────────────── + + /// Look up current statistics for a single node. + pub fn get_node_stats(&self, node_id: &str) -> Option<&NodeStats> { + self.nodes.get(node_id) + } + + /// All registered nodes. + pub fn get_all_nodes(&self) -> &HashMap { + &self.nodes + } +} + +// ───────────────────────────────────────────────────────────────────────────── +// §3 CloudCredential / ENECredentialManager +// ───────────────────────────────────────────────────────────────────────────── + +/// A cloud provider credential with an encrypted payload and node assignments. +#[derive(Debug, Clone)] +pub struct CloudCredential { + pub credential_id: String, + pub provider: String, + /// Raw payload bytes (encryption delegated to `ene_core`). + pub encrypted_payload: Vec, + pub access_level: String, + /// Node IDs to which this credential is assigned. + pub node_assignments: Vec, + pub usage_count: i64, + pub health_score: f64, + pub is_active: bool, +} + +/// In-memory credential store backed by an `ENENodeBalancer`. +pub struct ENECredentialManager { + credentials: HashMap, + pub balancer: ENENodeBalancer, +} + +impl ENECredentialManager { + // ── §3.1 Constructor ──────────────────────────────────────────────────── + + pub fn new(db_path: &str) -> Result { + Ok(Self { + credentials: HashMap::new(), + balancer: ENENodeBalancer::new(db_path)?, + }) + } + + // ── §3.2 Credential CRUD ───────────────────────────────────────────────── + + /// Create a new `CloudCredential`. + /// + /// `credential_id` = first 16 hex chars of SHA-256(canonical JSON of payload). + /// The `encrypted_payload` field stores the raw JSON bytes; actual encryption + /// is handled by the `ene_core` module. + pub fn create_credential( + &mut self, + provider: &str, + payload: &Value, + access_level: &str, + node_assignments: Vec, + ) -> CloudCredential { + let canonical = payload.to_string(); + let credential_id = sha256_hex16(canonical.as_bytes()); + + let cred = CloudCredential { + credential_id: credential_id.clone(), + provider: provider.to_owned(), + encrypted_payload: canonical.into_bytes(), + access_level: access_level.to_owned(), + node_assignments, + usage_count: 0, + health_score: 1.0, + is_active: true, + }; + self.credentials.insert(credential_id, cred.clone()); + cred + } + + /// Retrieve a credential by ID; returns the payload as a JSON `Value`. + pub fn get_credential(&self, credential_id: &str) -> Option { + let cred = self.credentials.get(credential_id)?; + if !cred.is_active { + return None; + } + serde_json::from_slice(&cred.encrypted_payload).ok() + } + + /// All currently active credentials. + pub fn list_credentials(&self) -> Vec<&CloudCredential> { + self.credentials + .values() + .filter(|c| c.is_active) + .collect() + } + + /// Deactivate a credential; returns `true` when it existed and was active. + pub fn revoke_credential(&mut self, credential_id: &str) -> bool { + if let Some(cred) = self.credentials.get_mut(credential_id) { + if cred.is_active { + cred.is_active = false; + return true; + } + } + false + } +} + +// ───────────────────────────────────────────────────────────────────────────── +// §4 ENECloudBridge +// ───────────────────────────────────────────────────────────────────────────── + +/// High-level bridge: combines credential management with the node balancer to +/// simulate cloud storage connections. +pub struct ENECloudBridge { + credential_manager: ENECredentialManager, +} + +impl ENECloudBridge { + pub fn new(db_path: &str) -> Result { + Ok(Self { + credential_manager: ENECredentialManager::new(db_path)?, + }) + } + + /// Select a healthy node and open a connection for `provider`. + /// + /// Returns the `connection_id` on success, or `None` if no suitable node + /// is available. + pub fn connect_to_storage(&mut self, provider: &str, node_id: &str) -> Option { + // Register the node if not already known. + let cred_id = sha256_hex16(provider.as_bytes()); + let _ = self + .credential_manager + .balancer + .register_node(node_id, &cred_id); + + // Use health_weighted selection; fall back to the supplied node_id. + let selected = self + .credential_manager + .balancer + .select_node("health_weighted") + .unwrap_or_else(|| node_id.to_owned()); + + self.credential_manager + .balancer + .record_connection(&selected, &cred_id) + .ok() + } + + /// Simulate a data transfer on an open connection. + /// + /// Returns `true` when the connection exists and the update succeeds. + pub fn transfer_data(&mut self, connection_id: &str, data: &[u8]) -> bool { + if !self + .credential_manager + .balancer + .active_connections + .contains_key(connection_id) + { + return false; + } + self.credential_manager.balancer.update_connection_stats( + connection_id, + data.len() as i64, + 1.0, // synthetic 1 ms latency per transfer + false, + ); + true + } + + /// Close an open connection. + pub fn close_connection(&mut self, connection_id: &str) -> Result<()> { + self.credential_manager + .balancer + .close_connection(connection_id) + } +} + +// ───────────────────────────────────────────────────────────────────────────── +// §5 Tests +// ───────────────────────────────────────────────────────────────────────────── + +#[cfg(test)] +mod tests { + use super::*; + use tempfile::NamedTempFile; + + fn temp_db() -> String { + NamedTempFile::new().unwrap().path().to_string_lossy().to_string() + } + + #[test] + fn test_register_and_select_node() { + let db = temp_db(); + let mut balancer = ENENodeBalancer::new(&db).unwrap(); + balancer.register_node("node-A", "cred-1").unwrap(); + balancer.register_node("node-B", "cred-2").unwrap(); + + // round_robin always returns the first sorted node. + let selected = balancer.select_node("round_robin").unwrap(); + assert_eq!(selected, "node-A"); + } + + #[test] + fn test_record_and_close_connection() { + let db = temp_db(); + let mut balancer = ENENodeBalancer::new(&db).unwrap(); + balancer.register_node("n1", "c1").unwrap(); + let conn_id = balancer.record_connection("n1", "c1").unwrap(); + assert_eq!(conn_id.len(), 16); + balancer.close_connection(&conn_id).unwrap(); + assert!(!balancer.active_connections.contains_key(&conn_id)); + } + + #[test] + fn test_health_check_improves_score() { + let db = temp_db(); + let mut balancer = ENENodeBalancer::new(&db).unwrap(); + balancer.register_node("n2", "c2").unwrap(); + // Initial error_rate is 0.0, so health should improve. + let healthy = balancer.health_check("n2").unwrap(); + assert!(healthy); + let score = balancer.get_node_stats("n2").unwrap().health_score; + assert!(score > 1.0 - 1e-9, "score should be at ceiling 1.0; got {}", score); + } + + #[test] + fn test_credential_lifecycle() { + let db = temp_db(); + let mut mgr = ENECredentialManager::new(&db).unwrap(); + let payload = serde_json::json!({ "key": "secret123", "region": "us-east-1" }); + let cred = mgr.create_credential("aws", &payload, "restricted", vec!["node-1".into()]); + + assert!(!cred.credential_id.is_empty()); + assert_eq!(cred.provider, "aws"); + + let retrieved = mgr.get_credential(&cred.credential_id).unwrap(); + assert_eq!(retrieved["key"], "secret123"); + + let revoked = mgr.revoke_credential(&cred.credential_id); + assert!(revoked); + assert!(mgr.get_credential(&cred.credential_id).is_none()); + } + + #[test] + fn test_cloud_bridge_connect_transfer_close() { + let db = temp_db(); + let mut bridge = ENECloudBridge::new(&db).unwrap(); + let conn_id = bridge.connect_to_storage("s3", "node-x").unwrap(); + let ok = bridge.transfer_data(&conn_id, b"hello world"); + assert!(ok); + bridge.close_connection(&conn_id).unwrap(); + } +} diff --git a/4-Infrastructure/infra/ene-session-sync/src/ene_core.rs b/4-Infrastructure/infra/ene-session-sync/src/ene_core.rs index a290055e..ab66a506 100644 --- a/4-Infrastructure/infra/ene-session-sync/src/ene_core.rs +++ b/4-Infrastructure/infra/ene-session-sync/src/ene_core.rs @@ -448,25 +448,26 @@ impl ENEAPIHook { ) -> anyhow::Result> { let conn = self.connect()?; - let result: Option<(String, String, i64, String, Option)> = { + let result: Option<(String, String, String, i64, String, Option)> = { let mut stmt = conn.prepare( - "SELECT encrypted_payload, nonce, classification, integrity_hash, access_log + "SELECT encrypted_payload, nonce, pkg, classification, integrity_hash, access_log FROM sensitive_data WHERE id = ?1", )?; stmt.query_row(params![data_id], |row| { Ok(( row.get::<_, String>(0)?, row.get::<_, String>(1)?, - row.get::<_, i64>(2)?, - row.get::<_, String>(3)?, - row.get::<_, Option>(4)?, + row.get::<_, String>(2)?, + row.get::<_, i64>(3)?, + row.get::<_, String>(4)?, + row.get::<_, Option>(5)?, )) }) .optional() .context("query sensitive_data by id")? }; - let (enc_payload, nonce, class_int, stored_hash, access_log_raw) = match result { + let (enc_payload, nonce, pkg, class_int, stored_hash, access_log_raw) = match result { Some(row) => row, None => return Ok(None), }; @@ -486,12 +487,8 @@ impl ENEAPIHook { "nonce": nonce, }); - // We stored `pkg` as AAD but we only have data_id here; use empty aad - // as the fallback (the stored `aad` field in the envelope will be used - // if present, but we wrote `pkg` as AAD without embedding it). To - // keep decrypt deterministic, pass empty bytes — the stored envelope - // aad field (absent in the old path) takes precedence in decrypt_with_key. - let plaintext_bytes = self.security.decrypt(&envelope, b"")?; + // Use pkg as AAD — matches what store_sensitive_data passed to encrypt. + let plaintext_bytes = self.security.decrypt(&envelope, pkg.as_bytes())?; let plaintext = String::from_utf8(plaintext_bytes) .context("decrypted data is not valid UTF-8")?; diff --git a/4-Infrastructure/infra/ene-session-sync/src/enhanced_swarm.rs b/4-Infrastructure/infra/ene-session-sync/src/enhanced_swarm.rs new file mode 100644 index 00000000..3f6625d7 --- /dev/null +++ b/4-Infrastructure/infra/ene-session-sync/src/enhanced_swarm.rs @@ -0,0 +1,139 @@ +#![allow(dead_code)] +//! enhanced_swarm.rs — Enhanced integrated swarm stub. +//! +//! Port of enhanced_integrated_swarm.py (109 lines). +//! +//! The original Python module depended on `lean_unified_shim`, which has been +//! removed from the repository. This module provides the same public surface +//! but returns a structured manifest that callers can use to trigger the actual +//! `lake build` pipeline in `0-Core-Formalism/lean/Semantics/`. + +use serde_json::{json, Value}; + +// ───────────────────────────────────────────────────────────────────────────── +// §1 EnhancedIntegratedSwarm +// ───────────────────────────────────────────────────────────────────────────── + +/// Swarm orchestrator stub. +/// +/// `lean_unified_shim` has been removed; swarm analysis now delegates to the +/// Lean lake build pipeline directly. Call [`perform_deep_analysis`] to +/// obtain a manifest that downstream tooling can use to trigger `lake build` +/// in `0-Core-Formalism/lean/Semantics/`. +pub struct EnhancedIntegratedSwarm { + lean_path: String, +} + +impl EnhancedIntegratedSwarm { + /// Construct a new swarm stub pointing at the given Lean source path. + /// + /// `lean_path` should be the directory containing the Lean `lakefile.lean` + /// (e.g. `"0-Core-Formalism/lean/Semantics"`). + pub fn new(lean_path: &str) -> Self { + Self { + lean_path: lean_path.to_string(), + } + } + + // ── §1.1 Deep analysis ────────────────────────────────────────────────── + + /// Return a stub analysis manifest. + /// + /// The manifest carries `status: "stub"` and a `note` explaining that + /// `lean_unified_shim` has been removed, together with a `lean_path` field + /// that downstream tools can use to invoke `lake build` directly. + /// + /// When the shim is eventually replaced by a live Lean bridge, this method + /// will return real `domains`, `subdomains`, `tensor_types`, and `manifold` + /// data. + pub fn perform_deep_analysis(&self) -> Value { + // lean_unified_shim has been removed; swarm analysis now delegates + // to the Lean lake build pipeline directly. This stub returns a + // manifest that callers can use to trigger the actual lake build. + json!({ + "status": "stub", + "note": "enhanced_swarm: lean_unified_shim removed; run `lake build` in 0-Core-Formalism/lean/Semantics for live analysis", + "lean_path": self.lean_path, + "domains": [], + "subdomains": [], + "tensor_types": [], + "manifold": { + "nodes": [], + "edges": [], + "topology": {} + }, + "metadata": { + "total_domains": 0, + "total_subdomains": 0, + "total_tensor_types": 0, + "manifold_nodes": 0, + "manifold_edges": 0, + "manifold_dimension": 0 + } + }) + } + + // ── §1.2 Pretty-print helper ──────────────────────────────────────────── + + /// Print `analysis` to stdout (or an error to stderr when `"error"` is + /// present in the root object). + pub fn print_analysis(&self, analysis: &Value) { + if let Some(err) = analysis.get("error") { + eprintln!("ERROR: {}", err); + return; + } + println!( + "{}", + serde_json::to_string_pretty(analysis).unwrap_or_default() + ); + } +} + +// ───────────────────────────────────────────────────────────────────────────── +// §2 Tests +// ───────────────────────────────────────────────────────────────────────────── + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_stub_status() { + let swarm = EnhancedIntegratedSwarm::new("0-Core-Formalism/lean/Semantics"); + let analysis = swarm.perform_deep_analysis(); + assert_eq!(analysis["status"], "stub"); + } + + #[test] + fn test_lean_path_reflected() { + let path = "some/lean/path"; + let swarm = EnhancedIntegratedSwarm::new(path); + let analysis = swarm.perform_deep_analysis(); + assert_eq!(analysis["lean_path"], path); + } + + #[test] + fn test_empty_collections() { + let swarm = EnhancedIntegratedSwarm::new("."); + let analysis = swarm.perform_deep_analysis(); + assert!(analysis["domains"].as_array().unwrap().is_empty()); + assert!(analysis["subdomains"].as_array().unwrap().is_empty()); + assert!(analysis["tensor_types"].as_array().unwrap().is_empty()); + assert_eq!(analysis["metadata"]["total_domains"], 0); + } + + #[test] + fn test_print_analysis_error_branch() { + // Smoke-test: should not panic. + let swarm = EnhancedIntegratedSwarm::new("."); + let err_val = serde_json::json!({ "error": "something went wrong" }); + swarm.print_analysis(&err_val); // prints to stderr, no panic + } + + #[test] + fn test_print_analysis_ok_branch() { + let swarm = EnhancedIntegratedSwarm::new("."); + let analysis = swarm.perform_deep_analysis(); + swarm.print_analysis(&analysis); // prints to stdout, no panic + } +} diff --git a/4-Infrastructure/infra/ene-session-sync/src/gemma_integration.rs b/4-Infrastructure/infra/ene-session-sync/src/gemma_integration.rs new file mode 100644 index 00000000..932afbe8 --- /dev/null +++ b/4-Infrastructure/infra/ene-session-sync/src/gemma_integration.rs @@ -0,0 +1,518 @@ +#![allow(dead_code)] +//! gemma_integration.rs — Gemma-4 task dispatcher with SQLite persistence. +//! +//! Port of gemma_4_integration.py (397 lines). A per-operation SQLite +//! connection is opened for each method call, mirroring the Python pattern of +//! constructing a new connection per operation. + +use anyhow::{Context, Result}; +use rusqlite::{params, Connection, OptionalExtension}; +use serde_json::{json, Value}; +use std::time::{SystemTime, UNIX_EPOCH}; + +// ───────────────────────────────────────────────────────────────────────────── +// §0 Shared helpers +// ───────────────────────────────────────────────────────────────────────────── + +/// Current wall-clock seconds since the UNIX epoch. +fn now_secs() -> i64 { + SystemTime::now() + .duration_since(UNIX_EPOCH) + .unwrap_or_default() + .as_secs() as i64 +} + +// ───────────────────────────────────────────────────────────────────────────── +// §1 Variant and task-type enumerations +// ───────────────────────────────────────────────────────────────────────────── + +/// Supported Gemma-4 model variants. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum GemmaVariant { + /// `google/gemma-4-2b-it` — 2B instruction-tuned. + E2B, + /// `google/gemma-4-4b-it` — 4B instruction-tuned (default). + E4B, + /// `google/gemma-4-31b-it` — 31B instruction-tuned. + E31B, + /// `google/gemma-4-26b-a4b-it` — 26B MoE (active 4B) instruction-tuned. + E26BA4B, +} + +impl GemmaVariant { + /// Hugging Face model identifier string. + pub fn as_str(&self) -> &'static str { + match self { + Self::E2B => "google/gemma-4-2b-it", + Self::E4B => "google/gemma-4-4b-it", + Self::E31B => "google/gemma-4-31b-it", + Self::E26BA4B => "google/gemma-4-26b-a4b-it", + } + } + + /// Parse a model-identifier string; falls back to `E4B` on unknown input. + pub fn from_str(s: &str) -> Self { + match s { + "google/gemma-4-2b-it" => Self::E2B, + "google/gemma-4-4b-it" => Self::E4B, + "google/gemma-4-31b-it" => Self::E31B, + "google/gemma-4-26b-a4b-it" => Self::E26BA4B, + _ => Self::E4B, + } + } +} + +/// High-level task categories supported by this integration. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum GemmaTask { + TextGeneration, + MultimodalProcessing, + AudioTranscription, + ImageUnderstanding, + Reasoning, + CodeGeneration, + FunctionCalling, +} + +impl GemmaTask { + /// Canonical lowercase identifier used for storage and metrics. + pub fn as_str(&self) -> &'static str { + match self { + Self::TextGeneration => "text_generation", + Self::MultimodalProcessing => "multimodal_processing", + Self::AudioTranscription => "audio_transcription", + Self::ImageUnderstanding => "image_understanding", + Self::Reasoning => "reasoning", + Self::CodeGeneration => "code_generation", + Self::FunctionCalling => "function_calling", + } + } + + /// Parse the canonical identifier; falls back to `TextGeneration`. + pub fn from_str(s: &str) -> Self { + match s { + "text_generation" => Self::TextGeneration, + "multimodal_processing" => Self::MultimodalProcessing, + "audio_transcription" => Self::AudioTranscription, + "image_understanding" => Self::ImageUnderstanding, + "reasoning" => Self::Reasoning, + "code_generation" => Self::CodeGeneration, + "function_calling" => Self::FunctionCalling, + _ => Self::TextGeneration, + } + } +} + +// ───────────────────────────────────────────────────────────────────────────── +// §2 GemmaTaskRequest +// ───────────────────────────────────────────────────────────────────────────── + +/// A fully-specified Gemma-4 task ready for submission. +pub struct GemmaTaskRequest { + /// Caller-supplied or generated unique identifier. + pub task_id: String, + pub task_type: GemmaTask, + pub variant: GemmaVariant, + /// Arbitrary input payload (prompt, code, image bytes, …). + pub input_data: Value, + /// Whether the "thinking" scratchpad should be enabled in the model. + pub enable_thinking: bool, + /// Maximum tokens to generate. + pub max_tokens: i64, + /// Scheduling priority; higher is more urgent. + pub priority: i64, +} + +// ───────────────────────────────────────────────────────────────────────────── +// §3 Gemma4Integration +// ───────────────────────────────────────────────────────────────────────────── + +/// SQLite-backed dispatcher for Gemma-4 inference tasks. +/// +/// Each public method opens its own `rusqlite::Connection` (stateless per-op +/// pattern, matching the Python port). +pub struct Gemma4Integration { + db_path: String, + /// Default variant used when no variant is explicitly specified. + _default_variant: GemmaVariant, +} + +impl Gemma4Integration { + // ── §3.1 Constructor ──────────────────────────────────────────────────── + + /// Open (or create) the SQLite database at `db_path` and initialise all + /// required tables. + pub fn new(db_path: &str, default_variant: GemmaVariant) -> Result { + let integration = Self { + db_path: db_path.to_owned(), + _default_variant: default_variant, + }; + integration.init_tables()?; + Ok(integration) + } + + fn open(&self) -> Result { + Connection::open(&self.db_path) + .with_context(|| format!("Gemma4Integration: open db {:?}", self.db_path)) + } + + fn init_tables(&self) -> Result<()> { + let conn = self.open()?; + conn.execute_batch( + "CREATE TABLE IF NOT EXISTS gemma_task_requests ( + task_id TEXT PRIMARY KEY, + task_type TEXT NOT NULL, + variant TEXT NOT NULL, + input_data TEXT NOT NULL, + enable_thinking INTEGER NOT NULL DEFAULT 0, + max_tokens INTEGER NOT NULL DEFAULT 512, + priority INTEGER NOT NULL DEFAULT 0, + status TEXT NOT NULL DEFAULT 'pending', + created_at INTEGER NOT NULL, + started_at INTEGER, + completed_at INTEGER, + result TEXT, + error TEXT + ); + CREATE TABLE IF NOT EXISTS gemma_performance_metrics ( + variant TEXT PRIMARY KEY, + total_tasks INTEGER NOT NULL DEFAULT 0, + avg_latency REAL NOT NULL DEFAULT 0.0, + avg_tokens_per_second REAL NOT NULL DEFAULT 0.0, + last_updated INTEGER NOT NULL + );", + ) + .context("Gemma4Integration: init_tables")?; + Ok(()) + } + + // ── §3.2 Task submission ──────────────────────────────────────────────── + + /// Insert a task into the database and bump the per-variant metrics row. + /// + /// Returns the `task_id` on success. + pub fn submit_task(&self, task: &GemmaTaskRequest) -> Result { + let conn = self.open()?; + + let input_json = task.input_data.to_string(); + let now = now_secs(); + + conn.execute( + "INSERT OR REPLACE INTO gemma_task_requests + (task_id, task_type, variant, input_data, enable_thinking, + max_tokens, priority, status, created_at) + VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7, 'pending', ?8)", + params![ + task.task_id, + task.task_type.as_str(), + task.variant.as_str(), + input_json, + task.enable_thinking as i64, + task.max_tokens, + task.priority, + now, + ], + ) + .context("Gemma4Integration: submit_task insert")?; + + // Upsert performance-metrics row. + conn.execute( + "INSERT INTO gemma_performance_metrics + (variant, total_tasks, avg_latency, avg_tokens_per_second, last_updated) + VALUES (?1, 1, 0.0, 0.0, ?2) + ON CONFLICT(variant) DO UPDATE SET + total_tasks = total_tasks + 1, + last_updated = excluded.last_updated", + params![task.variant.as_str(), now], + ) + .context("Gemma4Integration: submit_task metrics upsert")?; + + Ok(task.task_id.clone()) + } + + // ── §3.3 Task execution ───────────────────────────────────────────────── + + /// Transition the task from `pending` → `in_progress` → `completed`, + /// calling the simulated execution kernel in between. + pub fn execute_task(&self, task_id: &str) -> Result { + let conn = self.open()?; + let now = now_secs(); + + // Fetch task metadata. + let row: Option<(String, String, i64)> = conn + .query_row( + "SELECT variant, task_type, max_tokens FROM gemma_task_requests + WHERE task_id = ?1", + params![task_id], + |r| Ok((r.get(0)?, r.get(1)?, r.get(2)?)), + ) + .optional() + .context("Gemma4Integration: execute_task fetch")?; + + let (variant, task_type, max_tokens) = row + .ok_or_else(|| anyhow::anyhow!("execute_task: unknown task_id '{}'", task_id))?; + + // Mark in-progress. + conn.execute( + "UPDATE gemma_task_requests SET status = 'in_progress', started_at = ?1 + WHERE task_id = ?2", + params![now, task_id], + ) + .context("Gemma4Integration: execute_task mark in_progress")?; + + // Simulate execution. + let result = self._simulate_gemma_execution(task_id, &variant, &task_type, max_tokens); + let result_json = result.to_string(); + + let completed_at = now_secs(); + // Mark completed. + conn.execute( + "UPDATE gemma_task_requests + SET status = 'completed', completed_at = ?1, result = ?2 + WHERE task_id = ?3", + params![completed_at, result_json, task_id], + ) + .context("Gemma4Integration: execute_task mark completed")?; + + // Update metrics with synthetic latency. + let latency_ms = ((completed_at - now) * 1000) as f64 + 100.0; + let tokens_per_sec = if latency_ms > 0.0 { + (max_tokens as f64 / 2.0) / (latency_ms / 1000.0) + } else { + 0.0 + }; + conn.execute( + "UPDATE gemma_performance_metrics + SET avg_latency = (avg_latency + ?1) / 2.0, + avg_tokens_per_second = (avg_tokens_per_second + ?2) / 2.0, + last_updated = ?3 + WHERE variant = ?4", + params![latency_ms, tokens_per_sec, completed_at, variant], + ) + .context("Gemma4Integration: execute_task update metrics")?; + + Ok(result) + } + + /// Simulate Gemma inference — returns a structured `serde_json::Value`. + fn _simulate_gemma_execution( + &self, + task_id: &str, + variant: &str, + task_type: &str, + max_tokens: i64, + ) -> Value { + json!({ + "task_id": task_id, + "variant": variant, + "output": format!("Simulated output for {}", task_type), + "tokens_generated": max_tokens / 2, + "latency_ms": 100, + }) + } + + // ── §3.4 Queries ──────────────────────────────────────────────────────── + + /// Retrieve the result of a completed task, or `None` when still pending. + pub fn get_task_result(&self, task_id: &str) -> Result> { + let conn = self.open()?; + let result: Option> = conn + .query_row( + "SELECT result FROM gemma_task_requests WHERE task_id = ?1", + params![task_id], + |r| r.get(0), + ) + .optional() + .context("Gemma4Integration: get_task_result")?; + + match result { + None => Ok(None), // task_id not found + Some(None) => Ok(None), // task found but result not yet set + Some(Some(json_str)) => { + let v: Value = serde_json::from_str(&json_str) + .context("Gemma4Integration: get_task_result parse")?; + Ok(Some(v)) + } + } + } + + /// Retrieve aggregated performance metrics for a specific variant. + pub fn get_performance_metrics(&self, variant: &str) -> Result> { + let conn = self.open()?; + let row: Option<(i64, f64, f64, i64)> = conn + .query_row( + "SELECT total_tasks, avg_latency, avg_tokens_per_second, last_updated + FROM gemma_performance_metrics WHERE variant = ?1", + params![variant], + |r| Ok((r.get(0)?, r.get(1)?, r.get(2)?, r.get(3)?)), + ) + .optional() + .context("Gemma4Integration: get_performance_metrics")?; + + Ok(row.map(|(total, latency, tps, updated)| { + json!({ + "variant": variant, + "total_tasks": total, + "avg_latency_ms": latency, + "avg_tokens_per_second": tps, + "last_updated": updated, + }) + })) + } + + /// List all tasks, optionally filtered by `status`. + pub fn list_tasks(&self, status: Option<&str>) -> Result> { + let conn = self.open()?; + + let (sql, filter): (&str, Option<&str>) = match status { + Some(s) => ( + "SELECT task_id, task_type, variant, status, priority, created_at, completed_at + FROM gemma_task_requests WHERE status = ?1 ORDER BY priority DESC, created_at ASC", + Some(s), + ), + None => ( + "SELECT task_id, task_type, variant, status, priority, created_at, completed_at + FROM gemma_task_requests ORDER BY priority DESC, created_at ASC", + None, + ), + }; + + let mut stmt = conn + .prepare(sql) + .context("Gemma4Integration: list_tasks prepare")?; + + let map_row = |r: &rusqlite::Row<'_>| -> rusqlite::Result { + let task_id: String = r.get(0)?; + let task_type: String = r.get(1)?; + let variant: String = r.get(2)?; + let row_status: String = r.get(3)?; + let priority: i64 = r.get(4)?; + let created_at: i64 = r.get(5)?; + let completed_at: Option = r.get(6)?; + Ok(json!({ + "task_id": task_id, + "task_type": task_type, + "variant": variant, + "status": row_status, + "priority": priority, + "created_at": created_at, + "completed_at": completed_at, + })) + }; + + let rows: Vec = if let Some(s) = filter { + stmt.query_map(params![s], map_row) + .context("Gemma4Integration: list_tasks query (filtered)")? + .collect::>>() + .context("Gemma4Integration: list_tasks collect (filtered)")? + } else { + stmt.query_map([], map_row) + .context("Gemma4Integration: list_tasks query")? + .collect::>>() + .context("Gemma4Integration: list_tasks collect")? + }; + + Ok(rows) + } +} + +// ───────────────────────────────────────────────────────────────────────────── +// §4 Tests +// ───────────────────────────────────────────────────────────────────────────── + +#[cfg(test)] +mod tests { + use super::*; + use tempfile::NamedTempFile; + + fn temp_db() -> String { + NamedTempFile::new().unwrap().path().to_string_lossy().to_string() + } + + fn make_task(id: &str) -> GemmaTaskRequest { + GemmaTaskRequest { + task_id: id.to_owned(), + task_type: GemmaTask::Reasoning, + variant: GemmaVariant::E4B, + input_data: json!({ "prompt": "Explain braid theory." }), + enable_thinking: true, + max_tokens: 256, + priority: 5, + } + } + + #[test] + fn test_submit_and_execute() { + let db = temp_db(); + let g = Gemma4Integration::new(&db, GemmaVariant::E4B).unwrap(); + let task = make_task("t-001"); + g.submit_task(&task).unwrap(); + let result = g.execute_task("t-001").unwrap(); + assert_eq!(result["task_id"], "t-001"); + assert_eq!(result["variant"], GemmaVariant::E4B.as_str()); + assert!(result["tokens_generated"].as_i64().unwrap() == 128); + } + + #[test] + fn test_get_task_result_after_execute() { + let db = temp_db(); + let g = Gemma4Integration::new(&db, GemmaVariant::E2B).unwrap(); + let task = make_task("t-002"); + g.submit_task(&task).unwrap(); + g.execute_task("t-002").unwrap(); + let result = g.get_task_result("t-002").unwrap(); + assert!(result.is_some()); + } + + #[test] + fn test_list_tasks_filtered() { + let db = temp_db(); + let g = Gemma4Integration::new(&db, GemmaVariant::E4B).unwrap(); + g.submit_task(&make_task("t-003")).unwrap(); + g.submit_task(&make_task("t-004")).unwrap(); + g.execute_task("t-003").unwrap(); + + let completed = g.list_tasks(Some("completed")).unwrap(); + assert_eq!(completed.len(), 1); + assert_eq!(completed[0]["task_id"], "t-003"); + + let pending = g.list_tasks(Some("pending")).unwrap(); + assert_eq!(pending.len(), 1); + assert_eq!(pending[0]["task_id"], "t-004"); + } + + #[test] + fn test_performance_metrics() { + let db = temp_db(); + let g = Gemma4Integration::new(&db, GemmaVariant::E4B).unwrap(); + g.submit_task(&make_task("t-005")).unwrap(); + g.execute_task("t-005").unwrap(); + let metrics = g + .get_performance_metrics(GemmaVariant::E4B.as_str()) + .unwrap(); + assert!(metrics.is_some()); + assert!(metrics.unwrap()["total_tasks"].as_i64().unwrap() >= 1); + } + + #[test] + fn test_variant_round_trip() { + assert_eq!( + GemmaVariant::from_str(GemmaVariant::E26BA4B.as_str()), + GemmaVariant::E26BA4B + ); + // Unknown string → default E4B. + assert_eq!(GemmaVariant::from_str("unknown"), GemmaVariant::E4B); + } + + #[test] + fn test_task_type_round_trip() { + for t in &[ + GemmaTask::TextGeneration, + GemmaTask::Reasoning, + GemmaTask::CodeGeneration, + GemmaTask::FunctionCalling, + ] { + assert_eq!(GemmaTask::from_str(t.as_str()).as_str(), t.as_str()); + } + } +} diff --git a/4-Infrastructure/infra/ene-session-sync/src/hyperbolic_encoding.rs b/4-Infrastructure/infra/ene-session-sync/src/hyperbolic_encoding.rs new file mode 100644 index 00000000..6128e0bc --- /dev/null +++ b/4-Infrastructure/infra/ene-session-sync/src/hyperbolic_encoding.rs @@ -0,0 +1,467 @@ +#![allow(dead_code)] +//! hyperbolic_encoding.rs — Poincaré-disk manifold encoder / decoder. +//! +//! Port of hyperbolic_encoding.py (364 lines). All arithmetic is plain f64; +//! no external linear-algebra crate is required. + +use sha2::{Digest, Sha256}; +use std::collections::HashMap; + +// ───────────────────────────────────────────────────────────────────────────── +// §1 Core types +// ───────────────────────────────────────────────────────────────────────────── + +/// A point on the 2-D Poincaré disk together with optional original data. +/// +/// Invariant: `‖coordinates‖ < 1.0`. +#[derive(Debug, Clone)] +pub struct HyperbolicVector { + /// 2-D Poincaré-disk coordinates (x, y). + pub coordinates: [f64; 2], + /// Ambient dimension of the source vector (14 for this encoder). + pub dimension: usize, + /// The original high-dimensional vector, stored for lossless round-trips. + pub original: Option>, +} + +/// Projection weights used when mapping the 14-element source vector onto +/// the two disk axes. Mirrors the constants in `hyperbolic_encoding.py`. +const WEIGHTS: [f64; 14] = [ + 0.0019, 0.0020, 0.0024, 0.0025, 0.0023, 0.0016, 0.0019, 0.0018, 0.0020, + 0.0025, 0.0018, 0.0022, 0.0021, 0.0026, +]; + +/// Encoder that maps high-dimensional vectors into the Poincaré disk model of +/// hyperbolic space with a given (negative) curvature. +pub struct HyperbolicManifoldEncoder { + /// Riemannian curvature — must be negative (default −1.0). + pub curvature: f64, +} + +impl HyperbolicManifoldEncoder { + /// Create a new encoder with the specified curvature. + /// + /// ``` + /// # use ene_session_sync::hyperbolic_encoding::HyperbolicManifoldEncoder; + /// let enc = HyperbolicManifoldEncoder::new(-1.0); + /// ``` + pub fn new(curvature: f64) -> Self { + Self { curvature } + } + + // ── §1.1 Encode ────────────────────────────────────────────────────────── + + /// Project a 14-element Euclidean vector into the Poincaré disk. + /// + /// Even-indexed components contribute to x; odd-indexed to y. + /// The result is normalised so that `‖(x, y)‖ < 0.99`. + pub fn encode_to_poincare(&self, vector: &[f64]) -> anyhow::Result { + if vector.len() != 14 { + anyhow::bail!( + "hyperbolic_encoding: expected 14-element vector, got {}", + vector.len() + ); + } + + let mut x = 0.0_f64; + let mut y = 0.0_f64; + + for (i, (&v, &w)) in vector.iter().zip(WEIGHTS.iter()).enumerate() { + if i % 2 == 0 { + x += v * w; + } else { + y += v * w; + } + } + + // Normalise so that the point lies strictly inside the unit disk. + let norm = (x * x + y * y).sqrt(); + if norm >= 0.99 { + let scale = 0.98 / norm; + x *= scale; + y *= scale; + } + + Ok(HyperbolicVector { + coordinates: [x, y], + dimension: 14, + original: Some(vector.to_vec()), + }) + } + + // ── §1.2 Decode ────────────────────────────────────────────────────────── + + /// Recover the original vector from a `HyperbolicVector`. + /// + /// If the `original` field was stored during encoding it is returned + /// directly (lossless). Otherwise the disk coordinates are lifted back + /// to Euclidean space via an inverse exponential-map approximation. + pub fn decode_from_poincare(&self, hv: &HyperbolicVector) -> Vec { + if let Some(ref orig) = hv.original { + return orig.clone(); + } + + // Inverse projection: reconstruct a 14-element vector from (x, y). + // We reverse the weighted summation by distributing x back to even + // indices and y back to odd indices, weighted by the reciprocal of + // the per-index weight (clamped to avoid division by zero). + let [x, y] = hv.coordinates; + let mut out = vec![0.0_f64; 14]; + let weight_sum_even: f64 = WEIGHTS.iter().enumerate() + .filter(|(i, _)| i % 2 == 0) + .map(|(_, &w)| w) + .sum(); + let weight_sum_odd: f64 = WEIGHTS.iter().enumerate() + .filter(|(i, _)| i % 2 != 0) + .map(|(_, &w)| w) + .sum(); + + for (i, w) in WEIGHTS.iter().enumerate() { + let w_safe = if *w < 1e-12 { 1e-12 } else { *w }; + if i % 2 == 0 { + out[i] = if weight_sum_even > 1e-12 { + x * (w_safe / weight_sum_even) + } else { + 0.0 + }; + } else { + out[i] = if weight_sum_odd > 1e-12 { + y * (w_safe / weight_sum_odd) + } else { + 0.0 + }; + } + } + out + } + + // ── §1.3 Möbius transform ──────────────────────────────────────────────── + + /// Möbius transform on the Poincaré disk. + /// + /// Given translation point `a` (in the disk) and disk point `z`, computes + /// the standard gyrovector Möbius addition: + /// + /// ```text + /// T_a(z) = ((1 − ‖a‖²)z + (1 + ‖z‖² + 2⟨z,a⟩)a) + /// ────────────────────────────────────────── + /// (1 + 2⟨a,z⟩ + ‖a‖²‖z‖²) + /// ``` + /// + /// This satisfies `T_0(z) = z` (identity) and maps the open unit disk to + /// itself. Returns an error when the denominator is effectively zero. + pub fn mobius_transform( + &self, + a: [f64; 2], + z: [f64; 2], + ) -> anyhow::Result<[f64; 2]> { + let dot_az = a[0] * z[0] + a[1] * z[1]; + let norm_a_sq = a[0] * a[0] + a[1] * a[1]; + let norm_z_sq = z[0] * z[0] + z[1] * z[1]; + + let denom = 1.0 + 2.0 * dot_az + norm_a_sq * norm_z_sq; + + if denom.abs() < 1e-12 { + anyhow::bail!("mobius_transform: degenerate denominator ({:.2e})", denom); + } + + let scale_z = 1.0 - norm_a_sq; + let scale_a = 1.0 + norm_z_sq + 2.0 * dot_az; + + let rx = (scale_z * z[0] + scale_a * a[0]) / denom; + let ry = (scale_z * z[1] + scale_a * a[1]) / denom; + + Ok([rx, ry]) + } + + // ── §1.4 Hyperbolic distance ───────────────────────────────────────────── + + /// Poincaré-disk geodesic distance between two points `x` and `y`. + /// + /// Uses the formula: + /// ```text + /// d(x,y) = acosh(1 + 2‖x−y‖² / ((1−‖x‖²)(1−‖y‖²))) + /// ``` + /// The inner ratio is clamped to `1e10` to avoid numerical overflow near + /// the boundary. + pub fn hyperbolic_distance(&self, x: [f64; 2], y: [f64; 2]) -> f64 { + let dx = x[0] - y[0]; + let dy = x[1] - y[1]; + let diff_sq = dx * dx + dy * dy; + + let norm_x_sq = (x[0] * x[0] + x[1] * x[1]).min(1.0 - 1e-9); + let norm_y_sq = (y[0] * y[0] + y[1] * y[1]).min(1.0 - 1e-9); + + let denom = (1.0 - norm_x_sq) * (1.0 - norm_y_sq); + let ratio = if denom < 1e-12 { + 1e10 + } else { + (2.0 * diff_sq / denom).min(1e10) + }; + + (1.0 + ratio).acosh() + } + + // ── §1.5 Hierarchical similarity ──────────────────────────────────────── + + /// Compute a hierarchical similarity score for `parent` and `child`. + /// + /// Both vectors are encoded to the disk. If the child is further from the + /// origin than the parent (i.e. it sits deeper in the hierarchy), a score + /// combining angular and radial proximity is returned; otherwise `0.0`. + pub fn hierarchical_similarity( + &self, + parent: &[f64], + child: &[f64], + ) -> anyhow::Result { + let parent_hv = self.encode_to_poincare(parent)?; + let child_hv = self.encode_to_poincare(child)?; + + let origin = [0.0_f64; 2]; + let parent_dist = self.hyperbolic_distance(parent_hv.coordinates, origin); + let child_dist = self.hyperbolic_distance(child_hv.coordinates, origin); + + if child_dist <= parent_dist { + return Ok(0.0); + } + + // Angular similarity: cosine of the angle between the two disk vectors. + let [px, py] = parent_hv.coordinates; + let [cx, cy] = child_hv.coordinates; + let p_norm = (px * px + py * py).sqrt().max(1e-12); + let c_norm = (cx * cx + cy * cy).sqrt().max(1e-12); + let angular_sim = ((px * cx + py * cy) / (p_norm * c_norm)).clamp(-1.0, 1.0); + + // Radial similarity: how close the radii are. + let radial_sim = 1.0 - (child_dist - parent_dist).abs() / (child_dist + 1e-12); + + // Combined score (equal-weight average). + Ok(0.5 * (angular_sim + radial_sim)) + } + + // ── §1.6 Batch helpers ─────────────────────────────────────────────────── + + /// Encode a slice of vectors, returning one result per input. + pub fn encode_batch(&self, vectors: &[Vec]) -> Vec> { + vectors.iter().map(|v| self.encode_to_poincare(v)).collect() + } + + /// Decode a slice of `HyperbolicVector`s. + pub fn decode_batch(&self, hvs: &[HyperbolicVector]) -> Vec> { + hvs.iter().map(|hv| self.decode_from_poincare(hv)).collect() + } +} + +// ───────────────────────────────────────────────────────────────────────────── +// §2 HyperbolicCache +// ───────────────────────────────────────────────────────────────────────────── + +/// Memoised wrapper around `HyperbolicManifoldEncoder`. +/// +/// The cache key is the first 16 hex characters of the SHA-256 digest of the +/// raw f64 bytes — fast enough for session-sync workloads. +pub struct HyperbolicCache { + encoder: HyperbolicManifoldEncoder, + cache: HashMap, +} + +impl HyperbolicCache { + /// Create a new cache backed by an encoder with the given curvature. + pub fn new(curvature: f64) -> Self { + Self { + encoder: HyperbolicManifoldEncoder::new(curvature), + cache: HashMap::new(), + } + } + + /// Compute a stable cache key from a vector. + /// + /// The key is the first 16 lowercase hex characters of SHA-256(raw f64 LE bytes). + fn _hash_vector(v: &[f64]) -> String { + let mut hasher = Sha256::new(); + for &val in v { + hasher.update(val.to_le_bytes()); + } + let digest = hasher.finalize(); + hex::encode(&digest[..8]) // 8 bytes → 16 hex chars + } + + /// Return the cached `HyperbolicVector` for `vector`, encoding on first access. + pub fn get_or_encode(&mut self, vector: &[f64]) -> anyhow::Result<&HyperbolicVector> { + let key = Self::_hash_vector(vector); + if !self.cache.contains_key(&key) { + let hv = self.encoder.encode_to_poincare(vector)?; + self.cache.insert(key.clone(), hv); + } + Ok(self.cache.get(&key).expect("just inserted")) + } + + /// Evict all cached entries. + pub fn clear(&mut self) { + self.cache.clear(); + } + + /// Number of entries currently in the cache. + pub fn size(&self) -> usize { + self.cache.len() + } +} + +// ───────────────────────────────────────────────────────────────────────────── +// §3 HyperbolicSemanticSpace +// ───────────────────────────────────────────────────────────────────────────── + +/// A named collection of concepts mapped into the Poincaré disk. +/// +/// Supports nearest-neighbour retrieval by hyperbolic distance and basic +/// hierarchical relationship queries. +pub struct HyperbolicSemanticSpace { + encoder: HyperbolicManifoldEncoder, + concepts: HashMap, +} + +impl HyperbolicSemanticSpace { + /// Create an empty semantic space with the given curvature. + pub fn new(curvature: f64) -> Self { + Self { + encoder: HyperbolicManifoldEncoder::new(curvature), + concepts: HashMap::new(), + } + } + + /// Encode `vector` and store it under `name`. + pub fn add_concept(&mut self, name: &str, vector: &[f64]) -> anyhow::Result<()> { + let hv = self.encoder.encode_to_poincare(vector)?; + self.concepts.insert(name.to_owned(), hv); + Ok(()) + } + + /// Return the `top_k` concepts closest to `query` by hyperbolic distance. + /// + /// Results are ordered ascending by distance (nearest first). + pub fn find_similar( + &self, + query: &[f64], + top_k: usize, + ) -> anyhow::Result> { + let query_hv = self.encoder.encode_to_poincare(query)?; + + let mut distances: Vec<(String, f64)> = self + .concepts + .iter() + .map(|(name, hv)| { + let d = self.encoder.hyperbolic_distance(query_hv.coordinates, hv.coordinates); + (name.clone(), d) + }) + .collect(); + + distances.sort_by(|a, b| a.1.partial_cmp(&b.1).unwrap_or(std::cmp::Ordering::Equal)); + distances.truncate(top_k); + Ok(distances) + } + + /// Hierarchical similarity between two named concepts. + /// + /// Returns `0.0` when either concept is not registered. + pub fn get_hierarchy(&self, parent_name: &str, child_name: &str) -> f64 { + let (Some(p), Some(c)) = ( + self.concepts.get(parent_name), + self.concepts.get(child_name), + ) else { + return 0.0; + }; + + // Prefer lossless originals when available; fall back to disk coordinates. + let p_vec: Vec = p.original.clone().unwrap_or_else(|| { + self.encoder.decode_from_poincare(p) + }); + let c_vec: Vec = c.original.clone().unwrap_or_else(|| { + self.encoder.decode_from_poincare(c) + }); + + self.encoder + .hierarchical_similarity(&p_vec, &c_vec) + .unwrap_or(0.0) + } +} + +// ───────────────────────────────────────────────────────────────────────────── +// §4 Tests +// ───────────────────────────────────────────────────────────────────────────── + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_poincare_encode_decode() { + let enc = HyperbolicManifoldEncoder::new(-1.0); + let v = vec![0.1f64; 14]; + let hv = enc.encode_to_poincare(&v).unwrap(); + let norm = (hv.coordinates[0].powi(2) + hv.coordinates[1].powi(2)).sqrt(); + assert!(norm < 1.0, "must be inside unit disk; got norm={}", norm); + } + + #[test] + fn test_hyperbolic_distance_positive() { + let enc = HyperbolicManifoldEncoder::new(-1.0); + let d = enc.hyperbolic_distance([0.0, 0.0], [0.5, 0.0]); + assert!(d > 0.0, "distance must be positive; got {}", d); + } + + #[test] + fn test_encode_wrong_length() { + let enc = HyperbolicManifoldEncoder::new(-1.0); + assert!(enc.encode_to_poincare(&[0.0; 5]).is_err()); + } + + #[test] + fn test_decode_round_trip() { + let enc = HyperbolicManifoldEncoder::new(-1.0); + let orig: Vec = (0..14).map(|i| i as f64 * 0.05).collect(); + let hv = enc.encode_to_poincare(&orig).unwrap(); + let decoded = enc.decode_from_poincare(&hv); + // Original is stored, so round-trip must be exact. + assert_eq!(decoded, orig); + } + + #[test] + fn test_cache_size() { + let mut cache = HyperbolicCache::new(-1.0); + let v1 = vec![0.1f64; 14]; + let v2 = vec![0.2f64; 14]; + cache.get_or_encode(&v1).unwrap(); + cache.get_or_encode(&v2).unwrap(); + // Second access to v1 — no new entry. + cache.get_or_encode(&v1).unwrap(); + assert_eq!(cache.size(), 2); + cache.clear(); + assert_eq!(cache.size(), 0); + } + + #[test] + fn test_semantic_space_find_similar() { + let mut space = HyperbolicSemanticSpace::new(-1.0); + let base: Vec = vec![0.1; 14]; + let close: Vec = vec![0.11; 14]; + let far: Vec = (0..14).map(|i| if i % 2 == 0 { 5.0 } else { -5.0 }).collect(); + space.add_concept("base", &base).unwrap(); + space.add_concept("close", &close).unwrap(); + space.add_concept("far", &far).unwrap(); + let results = space.find_similar(&base, 2).unwrap(); + assert_eq!(results.len(), 2); + // "base" itself should be closest (distance ≈ 0). + assert_eq!(results[0].0, "base"); + } + + #[test] + fn test_mobius_transform_identity() { + // T_0(z) should equal z. + let enc = HyperbolicManifoldEncoder::new(-1.0); + let z = [0.3_f64, 0.4_f64]; + let result = enc.mobius_transform([0.0, 0.0], z).unwrap(); + let eps = 1e-10; + assert!((result[0] - z[0]).abs() < eps); + assert!((result[1] - z[1]).abs() < eps); + } +} diff --git a/4-Infrastructure/infra/ene-session-sync/src/knowledge_ingestion.rs b/4-Infrastructure/infra/ene-session-sync/src/knowledge_ingestion.rs new file mode 100644 index 00000000..0e3fa18a --- /dev/null +++ b/4-Infrastructure/infra/ene-session-sync/src/knowledge_ingestion.rs @@ -0,0 +1,502 @@ +#![allow(dead_code)] +//! knowledge_ingestion.rs — Knowledge ingestion adapters. +//! +//! Port of knowledge_ingestion.py (482 lines). +//! Three adapters: +//! - `WolframAlphaKnowledge` — Wolfram|Alpha v2 query API +//! - `OpenMathKnowledge` — OpenMath Content Dictionary (.ocd) fetcher +//! - `NLabWikiKnowledge` — nLab wiki page fetcher / crawler + +use serde_json::{json, Value}; +use std::collections::HashMap; + +// ───────────────────────────────────────────────────────────────────────────── +// §1 WolframAlphaKnowledge +// ───────────────────────────────────────────────────────────────────────────── + +/// Adapter for the Wolfram|Alpha v2 short-answer / query API. +pub struct WolframAlphaKnowledge { + api_key: String, + cache: HashMap, + rate_limit_remaining: u32, +} + +impl WolframAlphaKnowledge { + pub fn new(api_key: String) -> Self { + Self { + api_key, + cache: HashMap::new(), + rate_limit_remaining: 100, + } + } + + /// Query Wolfram|Alpha for a free-form `question`. + /// + /// Results are cached by question string. Returns `None` when the API + /// call fails, rate-limit is exhausted, or no pods are returned. + pub async fn query(&mut self, question: &str) -> Option { + // Serve from cache when available. + if let Some(cached) = self.cache.get(question) { + return Some(cached.clone()); + } + // Simple rate-limit guard. + if self.rate_limit_remaining == 0 { + return None; + } + self.rate_limit_remaining = self.rate_limit_remaining.saturating_sub(1); + + let client = reqwest::Client::new(); + let resp = client + .get("https://api.wolframalpha.com/v2/query") + .query(&[ + ("input", question), + ("format", "plaintext"), + ("output", "json"), + ("appid", &self.api_key), + ]) + .send() + .await + .ok()?; + + if !resp.status().is_success() { + return None; + } + + let body: Value = resp.json().await.ok()?; + + // Navigate into the queryresult pods. + let pods = body + .get("queryresult") + .and_then(|qr| qr.get("pods")) + .and_then(|p| p.as_array()) + .cloned() + .unwrap_or_default(); + + if pods.is_empty() { + return None; + } + + // Flatten all plaintext subpod values into a single results list. + let mut results: Vec = Vec::new(); + for pod in &pods { + let title = pod.get("title").and_then(|t| t.as_str()).unwrap_or(""); + if let Some(subpods) = pod.get("subpods").and_then(|s| s.as_array()) { + for sp in subpods { + let text = sp + .get("plaintext") + .and_then(|t| t.as_str()) + .unwrap_or("") + .trim() + .to_string(); + if !text.is_empty() { + results.push(json!({ "pod": title, "text": text })); + } + } + } + } + + let result = json!({ + "question": question, + "results": results, + }); + self.cache.insert(question.to_string(), result.clone()); + Some(result) + } + + /// Fetch domain knowledge for a mathematical domain. + /// + /// Runs three queries (core concepts, key theorems, applications) and + /// consolidates them into a single JSON object. + pub async fn get_domain_knowledge(&mut self, domain: &str) -> Value { + let q_concepts = format!("{} mathematics concepts", domain); + let q_theorems = format!("{} key theorems", domain); + let q_applications = format!("{} applications", domain); + + let concepts = self + .query(&q_concepts) + .await + .unwrap_or_else(|| json!({"results": []})); + let theorems = self + .query(&q_theorems) + .await + .unwrap_or_else(|| json!({"results": []})); + let applications = self + .query(&q_applications) + .await + .unwrap_or_else(|| json!({"results": []})); + + json!({ + "domain": domain, + "concepts": concepts.get("results").cloned().unwrap_or(json!([])), + "theorems": theorems.get("results").cloned().unwrap_or(json!([])), + "applications": applications.get("results").cloned().unwrap_or(json!([])), + }) + } +} + +// ───────────────────────────────────────────────────────────────────────────── +// §2 OpenMathKnowledge +// ───────────────────────────────────────────────────────────────────────────── + +/// Adapter for OpenMath Content Dictionary (.ocd) files hosted at +/// `www.openmath.org/cd/`. +pub struct OpenMathKnowledge { + base_url: String, + cache: HashMap, +} + +impl OpenMathKnowledge { + pub fn new() -> Self { + Self { + base_url: "https://www.openmath.org/cd".to_string(), + cache: HashMap::new(), + } + } + + /// Fetch and parse a single Content Dictionary by name. + /// + /// The .ocd format is XML; we extract `` / `` symbol + /// entries with simple string scanning — no external XML crate required. + pub async fn fetch_content_dictionary(&mut self, cd_name: &str) -> Option { + if let Some(cached) = self.cache.get(cd_name) { + return Some(cached.clone()); + } + + let url = format!("{}/{}.ocd", self.base_url, cd_name); + let client = reqwest::Client::new(); + let resp = client.get(&url).send().await.ok()?; + if !resp.status().is_success() { + return None; + } + let text = resp.text().await.ok()?; + + let symbols = parse_ocd_symbols(&text, cd_name); + + let result = json!({ + "name": cd_name, + "symbols": symbols, + }); + self.cache.insert(cd_name.to_string(), result.clone()); + Some(result) + } + + /// The canonical set of Content Dictionaries relevant to the research stack. + pub fn get_relevant_cds() -> Vec<&'static str> { + vec![ + "arith1", "alg1", "relation1", "set1", "logic1", "fns1", "nums1", + "calculus1", "complex1", "linalg1", "analysis1", "geometry", + "topology", + ] + } + + /// Fetch all relevant Content Dictionaries and return a combined summary. + pub async fn ingest_all_relevant_cds(&mut self) -> Value { + let cds = Self::get_relevant_cds(); + let mut all: Vec = Vec::with_capacity(cds.len()); + for cd in cds { + let entry = self + .fetch_content_dictionary(cd) + .await + .unwrap_or_else(|| json!({ "name": cd, "symbols": [] })); + all.push(entry); + } + json!({ "content_dictionaries": all }) + } +} + +/// Parse symbol definitions from raw .ocd XML text. +/// +/// Looks for `` blocks and extracts `` tags within them. +/// Falls back to scanning `name="..."` attribute style for alternate formats. +fn parse_ocd_symbols(xml: &str, cd_name: &str) -> Vec { + let mut symbols: Vec = Vec::new(); + + // Strategy 1: `...` inside `` blocks. + let mut search = xml; + while let Some(def_start) = search.find("") { + let after_open = &search[def_start + "".len()..]; + let block_end = after_open.find("").unwrap_or(after_open.len()); + let block = &after_open[..block_end]; + + let name = extract_xml_text(block, "Name").unwrap_or_default(); + let description = extract_xml_text(block, "Description").unwrap_or_default(); + let role = extract_xml_text(block, "Role").unwrap_or_default(); + + if !name.is_empty() { + symbols.push(json!({ + "name": name, + "cd": cd_name, + "role": role, + "description": description, + })); + } + + // Advance past this block. + let consumed = def_start + "".len() + block_end; + if consumed >= search.len() { + break; + } + search = &search[consumed..]; + } + + // Strategy 2 fallback: scan for `name="..."` attributes if no CDDefinition + // blocks were found (some CDs use a more compact format). + if symbols.is_empty() { + let mut s = xml; + while let Some(pos) = s.find("name=\"") { + let after = &s[pos + 6..]; + let end = after.find('"').unwrap_or(after.len()); + let name = after[..end].trim().to_string(); + if !name.is_empty() { + symbols.push(json!({ + "name": name, + "cd": cd_name, + "role": "", + })); + } + s = &s[pos + 6..]; + } + } + + symbols +} + +/// Extract the text content of the first occurrence of `...`. +fn extract_xml_text(xml: &str, tag: &str) -> Option { + let open = format!("<{}>", tag); + let close = format!("", tag); + let start = xml.find(&open)? + open.len(); + let end = xml[start..].find(&close)? + start; + Some(xml[start..end].trim().to_string()) +} + +// ───────────────────────────────────────────────────────────────────────────── +// §3 NLabWikiKnowledge +// ───────────────────────────────────────────────────────────────────────────── + +/// Adapter for the nLab wiki at `ncatlab.org`. +pub struct NLabWikiKnowledge { + base_url: String, + cache: HashMap, +} + +impl NLabWikiKnowledge { + pub fn new() -> Self { + Self { + base_url: "https://ncatlab.org/nlab/show".to_string(), + cache: HashMap::new(), + } + } + + /// Fetch and lightly parse an nLab page by name. + /// + /// Returns a `{"name", "url", "content", "links": [...]}` object. + /// HTML parsing is purely string-based — no external HTML parser crate. + pub async fn fetch_page(&mut self, page_name: &str) -> Option { + if let Some(cached) = self.cache.get(page_name) { + return Some(cached.clone()); + } + + // URL-encode the page name (replace spaces with +). + let encoded = page_name.replace(' ', "+"); + let url = format!("{}/{}", self.base_url, encoded); + + let client = reqwest::Client::new(); + let resp = client + .get(&url) + .header("User-Agent", "ene-session-sync/0.1 (research stack)") + .send() + .await + .ok()?; + + if !resp.status().is_success() { + return None; + } + + let html = resp.text().await.ok()?; + + // Extract main content: look for
...
+ // or the first substantial
block. Strip all HTML tags from the + // extracted slice to obtain plain text. + let content = extract_nlab_content(&html); + let links = extract_nlab_links(&html); + + let result = json!({ + "name": page_name, + "url": url, + "content": content, + "links": links, + }); + self.cache.insert(page_name.to_string(), result.clone()); + Some(result) + } + + /// BFS crawler: return page names reachable from `page_name` up to `depth` + /// hops (including the starting page itself). + pub async fn get_related_pages(&mut self, page_name: &str, depth: u32) -> Vec { + let mut visited: Vec = Vec::new(); + let mut frontier: Vec = vec![page_name.to_string()]; + + for _ in 0..depth { + let mut next: Vec = Vec::new(); + for page in frontier.drain(..) { + if visited.contains(&page) { + continue; + } + visited.push(page.clone()); + if let Some(data) = self.fetch_page(&page).await { + if let Some(links) = data.get("links").and_then(|l| l.as_array()) { + for link in links { + if let Some(s) = link.as_str() { + if !visited.contains(&s.to_string()) { + next.push(s.to_string()); + } + } + } + } + } + } + frontier = next; + if frontier.is_empty() { + break; + } + } + visited + } +} + +// ── HTML helpers ────────────────────────────────────────────────────────────── + +/// Extract a plain-text content summary from an nLab HTML page. +/// +/// Looks for `
` tags from the selected slice and collapses +/// whitespace. +fn extract_nlab_content(html: &str) -> String { + // Locate the start of the content div. + let start_marker = html + .find("class=\"entry_content\"") + .or_else(|| html.find("id=\"content\"")) + .and_then(|pos| { + // Walk backwards to the `').map(|p| p + 1).unwrap_or(0); + // Take up to 8 KB of content. + let end = (tag_end + 8192).min(after_open.len()); + &after_open[tag_end..end] + } else { + // No content div found — take the first 4 KB of body text. + &html[..html.len().min(4096)] + }; + + strip_html_tags(slice) +} + +/// Collect `href` values from `` tags that look like nLab +/// internal links (`/nlab/show/...`). +fn extract_nlab_links(html: &str) -> Vec { + let mut links: Vec = Vec::new(); + let mut search = html; + while let Some(pos) = search.find("href=\"/nlab/show/") { + let after = &search[pos + 6..]; // skip href=" + let end = after.find('"').unwrap_or(after.len()); + let href = &after[..end]; + // Turn `/nlab/show/foo+bar` → `foo bar` as a page name. + let page = href + .trim_start_matches("/nlab/show/") + .replace('+', " ") + .replace("%20", " "); + let val = Value::String(page); + if !links.contains(&val) { + links.push(val); + } + search = &search[pos + 6..]; + } + links +} + +/// Strip all HTML tags from `text`, returning collapsed plain text. +/// +/// Tags are identified as `<...>` spans. Consecutive whitespace is collapsed +/// to a single space. +fn strip_html_tags(text: &str) -> String { + let mut out = String::with_capacity(text.len()); + let mut in_tag = false; + for ch in text.chars() { + match ch { + '<' => in_tag = true, + '>' => in_tag = false, + _ if !in_tag => out.push(ch), + _ => {} + } + } + // Collapse runs of whitespace. + let mut collapsed = String::with_capacity(out.len()); + let mut prev_space = false; + for ch in out.chars() { + if ch.is_whitespace() { + if !prev_space { + collapsed.push(' '); + } + prev_space = true; + } else { + collapsed.push(ch); + prev_space = false; + } + } + collapsed.trim().to_string() +} + +// ───────────────────────────────────────────────────────────────────────────── +// §4 Tests +// ───────────────────────────────────────────────────────────────────────────── + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn strip_html_tags_basic() { + let html = "
Hello world!
"; + assert_eq!(strip_html_tags(html), "Hello world!"); + } + + #[test] + fn extract_xml_text_basic() { + let xml = "plusAddition"; + assert_eq!(extract_xml_text(xml, "Name"), Some("plus".to_string())); + assert_eq!( + extract_xml_text(xml, "Description"), + Some("Addition".to_string()) + ); + assert_eq!(extract_xml_text(xml, "Role"), None); + } + + #[test] + fn parse_ocd_symbols_name_attr_fallback() { + let xml = r#""#; + let syms = parse_ocd_symbols(xml, "arith1"); + assert_eq!(syms.len(), 2); + assert_eq!(syms[0]["name"], "plus"); + assert_eq!(syms[1]["name"], "times"); + } + + #[test] + fn relevant_cds_count() { + assert_eq!(OpenMathKnowledge::get_relevant_cds().len(), 13); + } + + #[test] + fn wolfram_rate_limit_guard() { + let mut wa = WolframAlphaKnowledge::new("dummy".to_string()); + wa.rate_limit_remaining = 0; + // Sync check — we can't await in a non-async test but we can verify + // the struct field directly. + assert_eq!(wa.rate_limit_remaining, 0); + } +} diff --git a/4-Infrastructure/infra/ene-session-sync/src/main.rs b/4-Infrastructure/infra/ene-session-sync/src/main.rs index 57d4305c..9c2cc3c0 100644 --- a/4-Infrastructure/infra/ene-session-sync/src/main.rs +++ b/4-Infrastructure/infra/ene-session-sync/src/main.rs @@ -1,15 +1,24 @@ mod bridge; mod compression; mod credential; +mod deepseek_adapter; mod embed; +mod ene_cloud_credential_manager; mod ene_core; +mod enhanced_swarm; mod fractal_fold; +mod gemma_integration; +mod hyperbolic_encoding; +mod knowledge_ingestion; +mod manifold_perception; mod math; mod meta_autotype; mod misc; mod models; mod normalize; mod s3c; +mod s3c_lean_review; +mod search_adapter; mod sink; mod source; mod swarm; diff --git a/4-Infrastructure/infra/ene-session-sync/src/manifold_perception.rs b/4-Infrastructure/infra/ene-session-sync/src/manifold_perception.rs new file mode 100644 index 00000000..2432e2e7 --- /dev/null +++ b/4-Infrastructure/infra/ene-session-sync/src/manifold_perception.rs @@ -0,0 +1,559 @@ +#![allow(dead_code)] +//! manifold_perception.rs — Topological manifest scanner for the research stack. +//! +//! Port of manifold_perception.py (390 lines). +//! Walks the filesystem under a given root, classifies every file, computes +//! per-file SHA-256 digests and line counts, parses Lean 4 metadata, identifies +//! structural "holes", and returns a `ManifoldReport`. +//! +//! Pure `std`-only directory walking; uses the `sha2` crate (already in Cargo.toml) +//! for file hashing. + +use serde_json::{json, Value}; +use sha2::{Digest, Sha256}; +use std::path::Path; + +// ───────────────────────────────────────────────────────────────────────────── +// §1 Core types +// ───────────────────────────────────────────────────────────────────────────── + +/// A single classified file in the research stack. +#[derive(Debug, Clone)] +pub struct ManifoldArtifact { + /// Absolute (or relative-to-root) path string. + pub path: String, + /// Coarse kind: "lean", "doc", "infra", "data", "other". + pub kind: String, + /// Number of lines in the file. + pub lines: usize, + /// First 16 hex characters of the SHA-256 digest. + pub hash: String, +} + +/// A structural gap detected in the manifold. +#[derive(Debug, Clone)] +pub struct ManifoldHole { + /// Human-readable centre / location of the gap. + pub center: String, + /// What kind of artifact was expected here. + pub expected_kind: String, + /// "low", "medium", or "high". + pub severity: String, + /// Free-text description. + pub description: String, + /// How many expected items are missing. + pub missing_count: usize, +} + +/// A topological boundary dimension of the manifold. +#[derive(Debug, Clone)] +pub struct ManifoldBoundary { + pub dimension: String, + pub present_count: usize, + pub description: String, +} + +/// The full manifest report produced by `build_manifold`. +#[derive(Debug, Clone)] +pub struct ManifoldReport { + /// ISO-8601 UTC timestamp of when the report was generated. + pub generated_at: String, + pub lean_count: usize, + pub doc_count: usize, + pub infra_count: usize, + pub total_lean_lines: usize, + pub total_doc_lines: usize, + pub total_theorems: usize, + pub total_sorry: usize, + pub holes: Vec, + pub boundaries: Vec, + pub artifacts: Vec, +} + +// ───────────────────────────────────────────────────────────────────────────── +// §2 File classification +// ───────────────────────────────────────────────────────────────────────────── + +/// Classify a file path into a coarse kind string. +/// +/// Priority order matches the Python port: +/// - `.lean` → "lean" +/// - `.md` / `.rst` / `.tex` → "doc" +/// - `.rs` / `.py` / `.sh` / `.toml` / `.json` / `.yaml` / `.yml` → "infra" +/// - `.db` / `.sqlite` / `.bin` / `.parquet` → "data" +/// - anything else → "other" +pub fn classify_artifact(path: &Path) -> &'static str { + let ext = path + .extension() + .and_then(|e| e.to_str()) + .unwrap_or("") + .to_ascii_lowercase(); + + match ext.as_str() { + "lean" => "lean", + "md" | "rst" | "tex" | "txt" => "doc", + "rs" | "py" | "sh" | "toml" | "json" | "yaml" | "yml" | "lock" => "infra", + "db" | "sqlite" | "sqlite3" | "bin" | "parquet" => "data", + _ => { + // Check filename fragments for known infrastructure files. + if let Some(name) = path.file_name().and_then(|n| n.to_str()) { + let lower = name.to_ascii_lowercase(); + if lower.contains("makefile") + || lower.contains("dockerfile") + || lower.contains("cargo") + || lower.contains("lakefile") + { + return "infra"; + } + if lower.ends_with(".olean") || lower.ends_with(".ilean") { + return "lean"; + } + } + "other" + } + } +} + +// ───────────────────────────────────────────────────────────────────────────── +// §3 File utilities +// ───────────────────────────────────────────────────────────────────────────── + +/// Compute the SHA-256 of a file and return the first 16 hex characters. +/// +/// Returns an all-zero string on read error rather than propagating. +pub fn sha256_file(path: &Path) -> String { + let data = match std::fs::read(path) { + Ok(d) => d, + Err(_) => return "0000000000000000".to_string(), + }; + let mut hasher = Sha256::new(); + hasher.update(&data); + let digest = hasher.finalize(); + // hex::encode gives 64 chars; take first 16. + let full = hex::encode(digest); + full[..16].to_string() +} + +/// Count the number of newline characters in a file. +/// +/// Returns 0 on read error. +pub fn count_lines(path: &Path) -> usize { + let data = match std::fs::read(path) { + Ok(d) => d, + Err(_) => return 0, + }; + data.iter().filter(|&&b| b == b'\n').count() +} + +// ───────────────────────────────────────────────────────────────────────────── +// §4 Lean 4 metadata +// ───────────────────────────────────────────────────────────────────────────── + +/// Count Lean 4 syntactic occurrences in a source file. +/// +/// Returns a JSON object: +/// ```json +/// { "theorems": 3, "defs": 7, "inductives": 1, "structures": 2, +/// "evals": 0, "sorries": 1 } +/// ``` +pub fn parse_lean_metadata(path: &Path) -> Value { + let text = match std::fs::read_to_string(path) { + Ok(t) => t, + Err(_) => return json!({ "theorems": 0, "defs": 0, "inductives": 0, + "structures": 0, "evals": 0, "sorries": 0 }), + }; + + let theorems = count_occurrences(&text, "theorem "); + let defs = count_occurrences(&text, "def "); + let inductives = count_occurrences(&text, "inductive "); + let structures = count_occurrences(&text, "structure "); + let evals = count_occurrences(&text, "#eval"); + let sorries = count_occurrences(&text, "sorry"); + + json!({ + "theorems": theorems, + "defs": defs, + "inductives": inductives, + "structures": structures, + "evals": evals, + "sorries": sorries, + }) +} + +/// Count non-overlapping occurrences of `needle` in `haystack`. +fn count_occurrences(haystack: &str, needle: &str) -> usize { + let mut count = 0; + let mut start = 0; + while let Some(pos) = haystack[start..].find(needle) { + count += 1; + start += pos + needle.len(); + } + count +} + +// ───────────────────────────────────────────────────────────────────────────── +// §5 Directory walker +// ───────────────────────────────────────────────────────────────────────────── + +/// Recursively collect all regular files under `dir`. +/// +/// Silently skips unreadable directories. Ignores hidden directories (those +/// whose name starts with `.`) to avoid traversing `.git`, `.lake`, etc. +fn walk_files(dir: &Path, out: &mut Vec) { + let rd = match std::fs::read_dir(dir) { + Ok(r) => r, + Err(_) => return, + }; + for entry in rd.flatten() { + let path = entry.path(); + let name = path + .file_name() + .and_then(|n| n.to_str()) + .unwrap_or(""); + // Skip hidden dirs / files. + if name.starts_with('.') { + continue; + } + let meta = match entry.metadata() { + Ok(m) => m, + Err(_) => continue, + }; + if meta.is_dir() { + walk_files(&path, out); + } else if meta.is_file() { + out.push(path); + } + } +} + +// ───────────────────────────────────────────────────────────────────────────── +// §6 Hole detection +// ───────────────────────────────────────────────────────────────────────────── + +/// Detect structural gaps from the parsed Lean metadata of a set of artifacts. +fn detect_holes(artifacts: &[ManifoldArtifact], lean_meta: &[(String, Value)]) -> Vec { + let mut holes: Vec = Vec::new(); + + for (path, meta) in lean_meta { + let sorry_count = meta["sorries"].as_u64().unwrap_or(0) as usize; + let theorem_count = meta["theorems"].as_u64().unwrap_or(0) as usize; + let def_count = meta["defs"].as_u64().unwrap_or(0) as usize; + let eval_count = meta["evals"].as_u64().unwrap_or(0) as usize; + + // Sorry without accompanying theorem — unfinished proof. + if sorry_count > 0 && theorem_count == 0 { + holes.push(ManifoldHole { + center: path.clone(), + expected_kind: "lean".to_string(), + severity: "high".to_string(), + description: format!( + "{} sorry(s) with no theorem statement — proof gap", + sorry_count + ), + missing_count: sorry_count, + }); + } + // Has defs but no #eval — possible dead code or untestable fragment. + if def_count > 0 && eval_count == 0 { + holes.push(ManifoldHole { + center: path.clone(), + expected_kind: "lean".to_string(), + severity: "low".to_string(), + description: format!( + "{} def(s) with no #eval — consider adding executable tests", + def_count + ), + missing_count: def_count, + }); + } + } + + // Structural gap: no Lean files at all. + if artifacts.iter().all(|a| a.kind != "lean") { + holes.push(ManifoldHole { + center: "stack root".to_string(), + expected_kind: "lean".to_string(), + severity: "high".to_string(), + description: "No Lean 4 source files found under the stack root".to_string(), + missing_count: 1, + }); + } + + // Gap: no documentation files. + if artifacts.iter().all(|a| a.kind != "doc") { + holes.push(ManifoldHole { + center: "stack root".to_string(), + expected_kind: "doc".to_string(), + severity: "medium".to_string(), + description: "No documentation files (.md / .rst / .tex) found".to_string(), + missing_count: 1, + }); + } + + holes +} + +// ───────────────────────────────────────────────────────────────────────────── +// §7 Public API +// ───────────────────────────────────────────────────────────────────────────── + +/// Walk `stack_root` and produce a `ManifoldReport`. +pub fn build_manifold(stack_root: &Path) -> ManifoldReport { + let mut paths: Vec = Vec::new(); + walk_files(stack_root, &mut paths); + paths.sort(); + + let mut artifacts: Vec = Vec::with_capacity(paths.len()); + let mut lean_meta: Vec<(String, Value)> = Vec::new(); + + let mut lean_count = 0usize; + let mut doc_count = 0usize; + let mut infra_count = 0usize; + let mut total_lean_lines = 0usize; + let mut total_doc_lines = 0usize; + let mut total_theorems = 0usize; + let mut total_sorry = 0usize; + + for path in &paths { + let kind = classify_artifact(path); + let lines = count_lines(path); + let hash = sha256_file(path); + + let path_str = path.to_string_lossy().into_owned(); + + match kind { + "lean" => { + lean_count += 1; + total_lean_lines += lines; + let meta = parse_lean_metadata(path); + total_theorems += meta["theorems"].as_u64().unwrap_or(0) as usize; + total_sorry += meta["sorries"].as_u64().unwrap_or(0) as usize; + lean_meta.push((path_str.clone(), meta)); + } + "doc" => { + doc_count += 1; + total_doc_lines += lines; + } + "infra" => { + infra_count += 1; + } + _ => {} + } + + artifacts.push(ManifoldArtifact { + path: path_str, + kind: kind.to_string(), + lines, + hash, + }); + } + + let holes = detect_holes(&artifacts, &lean_meta); + + // Boundary dimensions — coarse topological summary. + let boundaries = vec![ + ManifoldBoundary { + dimension: "lean".to_string(), + present_count: lean_count, + description: format!("{} Lean 4 source files, {} lines", lean_count, total_lean_lines), + }, + ManifoldBoundary { + dimension: "doc".to_string(), + present_count: doc_count, + description: format!("{} documentation files, {} lines", doc_count, total_doc_lines), + }, + ManifoldBoundary { + dimension: "infra".to_string(), + present_count: infra_count, + description: format!("{} infrastructure files", infra_count), + }, + ManifoldBoundary { + dimension: "theorems".to_string(), + present_count: total_theorems, + description: format!("{} theorem declarations found", total_theorems), + }, + ManifoldBoundary { + dimension: "sorry".to_string(), + present_count: total_sorry, + description: format!("{} sorry occurrences (proof gaps)", total_sorry), + }, + ]; + + let generated_at = chrono::Utc::now().to_rfc3339(); + + ManifoldReport { + generated_at, + lean_count, + doc_count, + infra_count, + total_lean_lines, + total_doc_lines, + total_theorems, + total_sorry, + holes, + boundaries, + artifacts, + } +} + +/// Build the manifold and serialize it to a `serde_json::Value`. +pub fn build_manifold_json(stack_root: &Path) -> Value { + let report = build_manifold(stack_root); + + let artifacts: Vec = report + .artifacts + .iter() + .map(|a| { + json!({ + "path": a.path, + "kind": a.kind, + "lines": a.lines, + "hash": a.hash, + }) + }) + .collect(); + + let holes: Vec = report + .holes + .iter() + .map(|h| { + json!({ + "center": h.center, + "expected_kind": h.expected_kind, + "severity": h.severity, + "description": h.description, + "missing_count": h.missing_count, + }) + }) + .collect(); + + let boundaries: Vec = report + .boundaries + .iter() + .map(|b| { + json!({ + "dimension": b.dimension, + "present_count": b.present_count, + "description": b.description, + }) + }) + .collect(); + + json!({ + "schema": "manifold_report_v1", + "generated_at": report.generated_at, + "lean_count": report.lean_count, + "doc_count": report.doc_count, + "infra_count": report.infra_count, + "total_lean_lines": report.total_lean_lines, + "total_doc_lines": report.total_doc_lines, + "total_theorems": report.total_theorems, + "total_sorry": report.total_sorry, + "holes": holes, + "boundaries": boundaries, + "artifacts": artifacts, + }) +} + +// ───────────────────────────────────────────────────────────────────────────── +// §8 Tests +// ───────────────────────────────────────────────────────────────────────────── + +#[cfg(test)] +mod tests { + use super::*; + use std::io::Write; + use tempfile::TempDir; + + fn make_temp_file(dir: &TempDir, name: &str, content: &str) -> std::path::PathBuf { + let path = dir.path().join(name); + let mut f = std::fs::File::create(&path).unwrap(); + f.write_all(content.as_bytes()).unwrap(); + path + } + + #[test] + fn classify_lean() { + assert_eq!(classify_artifact(Path::new("foo.lean")), "lean"); + assert_eq!(classify_artifact(Path::new("bar.olean")), "lean"); + } + + #[test] + fn classify_doc() { + assert_eq!(classify_artifact(Path::new("README.md")), "doc"); + assert_eq!(classify_artifact(Path::new("paper.tex")), "doc"); + } + + #[test] + fn classify_infra() { + assert_eq!(classify_artifact(Path::new("main.rs")), "infra"); + assert_eq!(classify_artifact(Path::new("Cargo.toml")), "infra"); + assert_eq!(classify_artifact(Path::new("script.py")), "infra"); + } + + #[test] + fn classify_other() { + assert_eq!(classify_artifact(Path::new("image.png")), "other"); + assert_eq!(classify_artifact(Path::new("video.mp4")), "other"); + } + + #[test] + fn count_lines_basic() { + let dir = TempDir::new().unwrap(); + let path = make_temp_file(&dir, "test.txt", "line1\nline2\nline3\n"); + assert_eq!(count_lines(&path), 3); + } + + #[test] + fn sha256_file_len() { + let dir = TempDir::new().unwrap(); + let path = make_temp_file(&dir, "test.txt", "hello world"); + let h = sha256_file(&path); + assert_eq!(h.len(), 16); + assert!(h.chars().all(|c| c.is_ascii_hexdigit())); + } + + #[test] + fn count_occurrences_basic() { + let text = "theorem foo : theorem bar : sorry sorry"; + assert_eq!(count_occurrences(text, "theorem "), 2); + assert_eq!(count_occurrences(text, "sorry"), 2); + } + + #[test] + fn parse_lean_metadata_counts() { + let dir = TempDir::new().unwrap(); + let src = "theorem foo : 1 = 1 := by rfl\nsorry\ndef bar := 42\n#eval bar\n"; + let path = make_temp_file(&dir, "test.lean", src); + let meta = parse_lean_metadata(&path); + assert_eq!(meta["theorems"], 1); + assert_eq!(meta["sorries"], 1); + assert_eq!(meta["defs"], 1); + assert_eq!(meta["evals"], 1); + } + + #[test] + fn build_manifold_smoke() { + let dir = TempDir::new().unwrap(); + make_temp_file(&dir, "a.lean", "theorem foo : 1 = 1 := by rfl\n"); + make_temp_file(&dir, "README.md", "# Hello\n"); + make_temp_file(&dir, "build.rs", "fn main() {}\n"); + + let report = build_manifold(dir.path()); + assert_eq!(report.lean_count, 1); + assert_eq!(report.doc_count, 1); + assert_eq!(report.infra_count, 1); + assert_eq!(report.total_theorems, 1); + assert!(!report.artifacts.is_empty()); + } + + #[test] + fn build_manifold_json_schema() { + let dir = TempDir::new().unwrap(); + make_temp_file(&dir, "x.lean", "sorry\n"); + let v = build_manifold_json(dir.path()); + assert_eq!(v["schema"], "manifold_report_v1"); + assert!(v["holes"].as_array().is_some()); + assert!(v["artifacts"].as_array().is_some()); + } +} diff --git a/4-Infrastructure/infra/ene-session-sync/src/math.rs b/4-Infrastructure/infra/ene-session-sync/src/math.rs index 88670c9c..276e4ed6 100644 --- a/4-Infrastructure/infra/ene-session-sync/src/math.rs +++ b/4-Infrastructure/infra/ene-session-sync/src/math.rs @@ -689,8 +689,8 @@ mod tests { let dist = all_pairs_distances(&g); let (_, avg) = diameter_and_avg(&dist); // Finite non-zero distances for chain A→B→C→D: - // A→B=1, A→C=2, A→D=3 → sum=6, count=3, avg=2.0 - assert!((avg - 2.0).abs() < 1e-9); + // A→B=1, A→C=2, A→D=3, B→C=1, B→D=2, C→D=1 → sum=10, count=6, avg=10/6 + assert!((avg - 10.0 / 6.0).abs() < 1e-9); } // ── §5 Curvature ────────────────────────────────────────────────────────── diff --git a/4-Infrastructure/infra/ene-session-sync/src/s3c.rs b/4-Infrastructure/infra/ene-session-sync/src/s3c.rs index 87698e45..83ab6ebe 100644 --- a/4-Infrastructure/infra/ene-session-sync/src/s3c.rs +++ b/4-Infrastructure/infra/ene-session-sync/src/s3c.rs @@ -29,7 +29,7 @@ pub struct ShellCoords { pub b: u32, /// Intersection form a · b. pub mass: u32, - /// Shell width a + b + 1 = 2k + 1. + /// Shell width a + b = 2k + 1. pub width: u32, } @@ -48,7 +48,7 @@ pub fn shell_decomposition(n: u32) -> ShellCoords { a, b, mass: a * b, - width: a + b + 1, + width: a + b, } } @@ -600,18 +600,18 @@ mod tests { #[test] fn test_shell_decomp_perfect_square() { - // n = 9 = 3² → k=3, a=0, b=1 + // n = 9 = 3²: k=3, a=0, b=(4²-9)=7, mass=0, width=a+b=7=2k+1 let s = shell_decomposition(9); assert_eq!(s.k, 3); assert_eq!(s.a, 0); - assert_eq!(s.b, 1); + assert_eq!(s.b, 7); assert_eq!(s.mass, 0); assert_eq!(s.width, 2 * 3 + 1); } #[test] fn test_shell_decomp_midpoint() { - // n = 6 = 2² + 2; k=2, a=2, b=(9-6)=3 + // n = 6 = 2² + 2: k=2, a=2, b=(9-6)=3, mass=6, width=5=2*2+1 let s = shell_decomposition(6); assert_eq!(s.k, 2); assert_eq!(s.a, 2); @@ -635,7 +635,7 @@ mod tests { for n in 0u32..=1000 { let s = shell_decomposition(n); assert_eq!(s.width, 2 * s.k + 1, "n={}", n); - assert_eq!(s.a + s.b, 2 * s.k, "n={}", n); + assert_eq!(s.a + s.b, 2 * s.k + 1, "n={}", n); } } diff --git a/4-Infrastructure/infra/ene-session-sync/src/s3c_lean_review.rs b/4-Infrastructure/infra/ene-session-sync/src/s3c_lean_review.rs new file mode 100644 index 00000000..8800656a --- /dev/null +++ b/4-Infrastructure/infra/ene-session-sync/src/s3c_lean_review.rs @@ -0,0 +1,123 @@ +#![allow(dead_code)] +//! s3c_lean_review.rs — Submit an S3C Lean source file for Gemma-4 review. +//! +//! Port of s3c_lean_review.py (86 lines). +//! +//! Reads the requested `.lean` file, constructs a high-priority `Reasoning` +//! task, submits it to [`Gemma4Integration`], executes it synchronously, and +//! returns the structured result. + +use crate::gemma_integration::{Gemma4Integration, GemmaTask, GemmaTaskRequest, GemmaVariant}; +use std::time::{SystemTime, UNIX_EPOCH}; + +// ───────────────────────────────────────────────────────────────────────────── +// §1 Public entry point +// ───────────────────────────────────────────────────────────────────────────── + +/// Read `s3c_lean_path`, submit a Gemma-4 code-review task, execute it, and +/// return the result JSON. +/// +/// # Errors +/// +/// Returns an error if the file cannot be read, the database cannot be opened, +/// or task submission / execution fails. +/// +/// # Example +/// +/// ```no_run +/// use ene_session_sync::s3c_lean_review::run_s3c_lean_review; +/// +/// let result = run_s3c_lean_review( +/// "0-Core-Formalism/lean/Semantics/S3C.lean", +/// "/tmp/gemma_review.db", +/// ).unwrap(); +/// println!("{}", result["output"]); +/// ``` +pub fn run_s3c_lean_review( + s3c_lean_path: &str, + db_path: &str, +) -> anyhow::Result { + // ── Read the Lean source file ──────────────────────────────────────────── + let code = std::fs::read_to_string(s3c_lean_path) + .map_err(|e| anyhow::anyhow!("cannot read {}: {}", s3c_lean_path, e))?; + + // ── Construct the task ─────────────────────────────────────────────────── + let gemma = Gemma4Integration::new(db_path, GemmaVariant::E4B)?; + + let now = SystemTime::now() + .duration_since(UNIX_EPOCH) + .unwrap_or_default() + .as_secs(); + let task_id = format!("s3c_review_{}", now); + + let task = GemmaTaskRequest { + task_id: task_id.clone(), + task_type: GemmaTask::Reasoning, + variant: GemmaVariant::E4B, + input_data: serde_json::json!({ + "prompt": "Review the following Lean 4 code for S3C manifold processing. \ + Identify any type errors, missing proofs, incomplete definitions, \ + and suggest improvements.", + "code": code, + "file": "S3C.lean", + "enable_thinking": true, + }), + enable_thinking: true, + max_tokens: 2048, + priority: 9, + }; + + // ── Submit and execute ─────────────────────────────────────────────────── + gemma.submit_task(&task)?; + let result = gemma.execute_task(&task_id)?; + Ok(result) +} + +// ───────────────────────────────────────────────────────────────────────────── +// §2 Tests +// ───────────────────────────────────────────────────────────────────────────── + +#[cfg(test)] +mod tests { + use super::*; + use std::io::Write; + use tempfile::NamedTempFile; + + /// Write a tiny fake `.lean` file and run the review pipeline against a + /// temp SQLite database. + #[test] + fn test_run_s3c_lean_review_with_stub_file() { + // Create a temporary Lean source file. + let mut lean_file = NamedTempFile::new().unwrap(); + writeln!( + lean_file, + "-- S3C stub\ndef hello : Nat := 42" + ) + .unwrap(); + + // Create a temporary database. + let db_file = NamedTempFile::new().unwrap(); + let db_path = db_file.path().to_str().unwrap().to_owned(); + + let lean_path = lean_file.path().to_str().unwrap().to_owned(); + let result = run_s3c_lean_review(&lean_path, &db_path).unwrap(); + + // The stub executor must produce these fields. + assert!(result.get("task_id").is_some(), "missing task_id"); + assert!(result.get("output").is_some(), "missing output"); + assert!(result.get("tokens_generated").is_some(), "missing tokens_generated"); + + // tokens_generated should be half of max_tokens (2048 / 2 = 1024). + assert_eq!(result["tokens_generated"].as_i64().unwrap(), 1024); + } + + #[test] + fn test_run_s3c_lean_review_missing_file() { + let db_file = NamedTempFile::new().unwrap(); + let db_path = db_file.path().to_str().unwrap().to_owned(); + let err = run_s3c_lean_review("/nonexistent/path/S3C.lean", &db_path); + assert!(err.is_err(), "expected error for missing file"); + let msg = format!("{}", err.unwrap_err()); + assert!(msg.contains("cannot read"), "unexpected error: {}", msg); + } +} diff --git a/4-Infrastructure/infra/ene-session-sync/src/search_adapter.rs b/4-Infrastructure/infra/ene-session-sync/src/search_adapter.rs new file mode 100644 index 00000000..e51f2c2d --- /dev/null +++ b/4-Infrastructure/infra/ene-session-sync/src/search_adapter.rs @@ -0,0 +1,206 @@ +#![allow(dead_code)] +//! search_adapter.rs — Web search provider adapters. +//! +//! Port of search_adapter.py (77 lines). +//! Two providers: +//! - `GoogleSurfProvider` — placeholder (real dispatch via MCP server) +//! - `BraveSearchProvider` — live search via Brave Search API + +use serde_json::{json, Value}; + +// ───────────────────────────────────────────────────────────────────────────── +// §1 SearchResult +// ───────────────────────────────────────────────────────────────────────────── + +/// A single web search result. +#[derive(Debug, Clone)] +pub struct SearchResult { + pub title: String, + pub url: String, + pub snippet: String, +} + +impl SearchResult { + pub fn new(title: impl Into, url: impl Into, snippet: impl Into) -> Self { + Self { + title: title.into(), + url: url.into(), + snippet: snippet.into(), + } + } + + /// Serialize to a JSON object with `title`, `url`, and `snippet` keys. + pub fn to_value(&self) -> Value { + json!({ + "title": self.title, + "url": self.url, + "snippet": self.snippet, + }) + } +} + +// ───────────────────────────────────────────────────────────────────────────── +// §2 GoogleSurfProvider +// ───────────────────────────────────────────────────────────────────────────── + +/// Placeholder Google search provider. +/// +/// Real integration is handled via an MCP server external to this binary. +/// Returns an empty result set so call-sites compile and type-check cleanly. +pub struct GoogleSurfProvider; + +impl GoogleSurfProvider { + pub fn new() -> Self { + Self + } + + /// Submit a search query. Always returns an empty vec — see module doc. + pub async fn search(&self, _query: &str, _limit: usize) -> Vec { + // Dispatch is handled by the MCP server layer. This shim exists so + // that code depending on `GoogleSurfProvider` compiles and can be + // wired up later without API changes. + Vec::new() + } +} + +impl Default for GoogleSurfProvider { + fn default() -> Self { + Self::new() + } +} + +// ───────────────────────────────────────────────────────────────────────────── +// §3 BraveSearchProvider +// ───────────────────────────────────────────────────────────────────────────── + +/// Live web search via the Brave Search REST API. +/// +/// Requires a `BRAVE_API_KEY` subscription token. Returns an empty result set +/// when no key is configured or when the API call fails. +pub struct BraveSearchProvider { + api_key: Option, + http: reqwest::Client, +} + +impl BraveSearchProvider { + pub fn new(api_key: Option) -> Self { + Self { + api_key, + http: reqwest::Client::new(), + } + } + + /// Build from the `BRAVE_API_KEY` environment variable. + pub fn from_env() -> Self { + Self::new(std::env::var("BRAVE_API_KEY").ok()) + } + + /// Execute a web search query and return up to `limit` results. + /// + /// Hits `https://api.search.brave.com/res/v1/web/search`. + /// Returns an empty vec when no API key is present or on any error. + pub async fn search(&self, query: &str, limit: usize) -> Vec { + let key = match &self.api_key { + Some(k) => k.clone(), + None => return Vec::new(), + }; + + let url = "https://api.search.brave.com/res/v1/web/search"; + let resp = match self + .http + .get(url) + .header("Accept", "application/json") + .header("X-Subscription-Token", &key) + .query(&[("q", query), ("count", &limit.to_string())]) + .send() + .await + { + Ok(r) => r, + Err(e) => { + tracing::warn!("brave search request failed: {}", e); + return Vec::new(); + } + }; + + if !resp.status().is_success() { + tracing::warn!("brave search returned {}", resp.status()); + return Vec::new(); + } + + let body: Value = match resp.json().await { + Ok(v) => v, + Err(e) => { + tracing::warn!("brave search response parse error: {}", e); + return Vec::new(); + } + }; + + // Response schema: { "web": { "results": [ { "title", "url", "description" } ] } } + let raw_results = body + .get("web") + .and_then(|w| w.get("results")) + .and_then(|r| r.as_array()) + .cloned() + .unwrap_or_default(); + + raw_results + .into_iter() + .map(|item| { + let title = item + .get("title") + .and_then(|t| t.as_str()) + .unwrap_or("") + .to_string(); + let url = item + .get("url") + .and_then(|u| u.as_str()) + .unwrap_or("") + .to_string(); + let snippet = item + .get("description") + .and_then(|d| d.as_str()) + .unwrap_or("") + .to_string(); + SearchResult { title, url, snippet } + }) + .collect() + } +} + +// ───────────────────────────────────────────────────────────────────────────── +// §4 Tests +// ───────────────────────────────────────────────────────────────────────────── + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn search_result_to_value() { + let r = SearchResult::new("My Title", "https://example.com", "A snippet."); + let v = r.to_value(); + assert_eq!(v["title"], "My Title"); + assert_eq!(v["url"], "https://example.com"); + assert_eq!(v["snippet"], "A snippet."); + } + + #[test] + fn brave_no_key_returns_empty() { + // Without an API key the provider is constructed correctly but will + // return empty — we verify the struct, not the async path. + let provider = BraveSearchProvider::new(None); + assert!(provider.api_key.is_none()); + } + + #[test] + fn google_surf_provider_default() { + let _ = GoogleSurfProvider::default(); + } + + #[tokio::test] + async fn google_surf_returns_empty() { + let p = GoogleSurfProvider::new(); + let results = p.search("test query", 10).await; + assert!(results.is_empty()); + } +} diff --git a/4-Infrastructure/infra/ene_api.py b/4-Infrastructure/infra/ene_api.py deleted file mode 100755 index f6a02473..00000000 --- a/4-Infrastructure/infra/ene_api.py +++ /dev/null @@ -1,350 +0,0 @@ -#!/usr/bin/env python3 -""" -ENE API Hook - Secure interface for Endless Node Edges - -Provides a REST API for ENE operations with secure sensitive data handling. -All sensitive data is encrypted at rest using AES-256-GCM with keys derived -from the semantic manifold coordinates. - -Security Features: -- AES-256-GCM encryption for all sensitive data -- Key derivation from semantic space (hyperbolic manifold) -- Access control with clearance levels -- Audit logging with cryptographic signatures -- Rate limiting and request validation -""" - -import hashlib -import hmac -import json -import os -import sqlite3 -import time -import sys -from base64 import b64encode, b64decode -from cryptography.hazmat.primitives.ciphers.aead import AESGCM -from cryptography.hazmat.primitives import hashes -from cryptography.hazmat.primitives.kdf.pbkdf2 import PBKDF2HMAC -from cryptography.hazmat.backends import default_backend -from dataclasses import dataclass -from enum import IntEnum -from typing import Optional, Dict, List, Any -from pathlib import Path - -# Import metafoam compression and GCL encoding -sys.path.insert(0, str(Path(__file__).parent.parent / "scripts")) -try: - from compression_metafoam_adapter import MetafoamCompressionAdapter - from genetic_codon_encoder import GeneticCodingLanguage - from delta_gcl_encoder import DeltaGCLEncoder -except ImportError: - MetafoamCompressionAdapter = None - GeneticCodingLanguage = None - DeltaGCLEncoder = None - -# Configuration -DB_PATH = "/home/allaun/Documents/Research Stack/data/substrate_index.db" -SECRET_KEY = os.getenv("ENE_SECRET_KEY", "default-secret-key-change-in-production").encode() -SALT = b"ene-semantic-salt-2024" - -class AccessLevel(IntEnum): - PUBLIC = 0 - INTERNAL = 1 - RESTRICTED = 2 - SECRET = 3 - -@dataclass -class SensitiveData: - payload: str - classification: AccessLevel - integrity_hash: str - timestamp: int - -@dataclass -class SecurityState: - encryption_key: bytes - access_level: AccessLevel - audit_log: List[str] - -class ENESecurityManager: - """Manages security operations for ENE sensitive data""" - - def __init__(self): - self.backend = default_backend() - self._load_or_generate_key() - - def _load_or_generate_key(self): - """Load encryption key from environment or generate from semantic space""" - key_material = os.getenv("ENE_ENCRYPTION_KEY") - if key_material: - try: - self.encryption_key = b64decode(key_material) - except Exception: - # If base64 decode fails, use raw key - self.encryption_key = key_material.encode()[:32].ljust(32, b'\0') - else: - # Derive key from semantic space (placeholder - should use actual semantic vector) - kdf = PBKDF2HMAC( - algorithm=hashes.SHA256(), - length=32, - salt=SALT, - iterations=100000, - backend=self.backend - ) - self.encryption_key = kdf.derive(SECRET_KEY[:32]) - - def derive_key_from_semantic(self, semantic_vector: List[float]) -> bytes: - """Derive encryption key from semantic manifold coordinates""" - # XOR all semantic axes with proper bounds - base_key = 0 - for v in semantic_vector: - # Clamp value to 0-1 range before scaling - clamped = max(0.0, min(1.0, v)) - base_key ^= int(clamped * 0xFFFFFFFF) & 0xFFFFFFFF - - # Apply golden ratio mixing with overflow handling - mixed = (base_key * 0x9E3779B9) & 0xFFFFFFFF - - # Derive final key - kdf = PBKDF2HMAC( - algorithm=hashes.SHA256(), - length=32, - salt=SALT, - iterations=100000, - backend=self.backend - ) - return kdf.derive(mixed.to_bytes(4, byteorder='big')) - - def encrypt_data(self, plaintext: str, associated_data: bytes = b"") -> Dict[str, Any]: - """Encrypt data using AES-256-GCM""" - aesgcm = AESGCM(self.encryption_key) - nonce = os.urandom(12) - ciphertext = aesgcm.encrypt(nonce, plaintext.encode(), associated_data) - - return { - "ciphertext": b64encode(ciphertext).decode(), - "nonce": b64encode(nonce).decode(), - "associated_data": b64encode(associated_data).decode() if associated_data else None - } - - def decrypt_data(self, encrypted: Dict[str, Any], associated_data: bytes = b"") -> str: - """Decrypt data using AES-256-GCM""" - aesgcm = AESGCM(self.encryption_key) - ciphertext = b64decode(encrypted["ciphertext"]) - nonce = b64decode(encrypted["nonce"]) - - if encrypted.get("associated_data"): - associated_data = b64decode(encrypted["associated_data"]) - - plaintext = aesgcm.decrypt(nonce, ciphertext, associated_data) - return plaintext.decode() - - def check_access(self, clearance: AccessLevel, classification: AccessLevel) -> bool: - """Check if clearance level is sufficient for data classification""" - return clearance >= classification - - def compute_integrity_hash(self, data: str) -> str: - """Compute SHA-256 integrity hash""" - return hashlib.sha256(data.encode()).hexdigest() - -class ENEAPIHook: - """REST API hook for ENE operations""" - - def __init__(self): - self.security = ENESecurityManager() - self.db_path = DB_PATH - self._init_database() - - def _init_database(self): - """Initialize database with sensitive_data table if needed""" - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - cursor.execute(""" - CREATE TABLE IF NOT EXISTS sensitive_data ( - id TEXT PRIMARY KEY, - pkg TEXT NOT NULL, - encrypted_payload TEXT NOT NULL, - nonce TEXT NOT NULL, - classification INTEGER NOT NULL, - integrity_hash TEXT NOT NULL, - created_at INTEGER NOT NULL, - access_log TEXT - ) - """) - - conn.commit() - conn.close() - - def store_sensitive_data(self, pkg: str, payload: str, classification: AccessLevel, - semantic_vector: Optional[List[float]] = None, - use_metafoam: bool = True, - use_delta_gcl: bool = True) -> Dict[str, Any]: - """Store sensitive data with encryption and optional metafoam compression (defaults to delta GCL)""" - try: - # Derive key from semantic vector if provided - if semantic_vector: - self.security.encryption_key = self.security.derive_key_from_semantic(semantic_vector) - - # Metafoam compression and GCL encoding (if available) - gcl_sequence = None - compression_stats = {} - - if use_metafoam and use_delta_gcl and MetafoamCompressionAdapter and DeltaGCLEncoder: - # Create temporary file for compression - import tempfile - with tempfile.NamedTemporaryFile(mode='wb', delete=False, suffix='.bin') as tmp_file: - tmp_file.write(payload.encode()) - tmp_path = tmp_file.name - - try: - # Compress with metafoam - adapter = MetafoamCompressionAdapter(use_genetic=True) - with tempfile.TemporaryDirectory() as tmp_dir: - compression_result = adapter.compress_with_metafoam_metadata(tmp_path, tmp_dir) - - # Encode manifest to Delta GCL (optimized) - delta_encoder = DeltaGCLEncoder() - with open(compression_result['manifest_path'], 'r') as f: - manifest = json.load(f) - gcl_sequence = delta_encoder.encode_to_delta_gcl(manifest) - - # Extract compression stats - comp_meta = manifest.get('compression_metadata', {}) - compression_stats = { - "compression_ratio": comp_meta.get('compression_ratio', 0.0), - "field_phi": comp_meta.get('field_phi', 0.0), - "foam_score": manifest.get('foam_score', 0.0), - "rgflow_lawful": comp_meta.get('rgflow_lawful', False), - "tags": manifest.get('tags', []), - "gcl_encoding": "delta_optimized" - } - finally: - # Cleanup temp file - import os - os.unlink(tmp_path) - - # Store GCL sequence instead of full payload if available - data_to_store = gcl_sequence if gcl_sequence else payload - - # Encrypt data - encrypted = self.security.encrypt_data(data_to_store, pkg.encode()) - - # Compute integrity hash - integrity_hash = self.security.compute_integrity_hash(data_to_store) - - # Store in database - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - # Check if we need to add gcl_sequence column - cursor.execute("PRAGMA table_info(sensitive_data)") - columns = [col[1] for col in cursor.fetchall()] - if 'gcl_sequence' not in columns: - cursor.execute("ALTER TABLE sensitive_data ADD COLUMN gcl_sequence TEXT") - if 'compression_stats' not in columns: - cursor.execute("ALTER TABLE sensitive_data ADD COLUMN compression_stats TEXT") - - data_id = hashlib.sha256(f"{pkg}{int(time.time())}".encode()).hexdigest() - - cursor.execute(""" - INSERT INTO sensitive_data - (id, pkg, encrypted_payload, nonce, classification, integrity_hash, created_at, access_log, gcl_sequence, compression_stats) - VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?) - """, ( - data_id, - pkg, - encrypted["ciphertext"], - encrypted["nonce"], - classification, - integrity_hash, - int(time.time()), - json.dumps({"action": "store", "timestamp": int(time.time())}), - gcl_sequence, - json.dumps(compression_stats) if compression_stats else None - )) - - conn.commit() - conn.close() - - return { - "success": True, - "id": data_id, - "classification": classification, - "stored_at": int(time.time()), - "gcl_encoded": gcl_sequence is not None, - "gcl_length": len(gcl_sequence) if gcl_sequence else 0, - "gcl_type": "delta_optimized" if gcl_sequence and len(gcl_sequence) < 50 else "standard", - "compression_stats": compression_stats - } - except Exception as e: - return {"success": False, "error": str(e)} - - def retrieve_sensitive_data(self, pkg: str, clearance: AccessLevel) -> Dict[str, Any]: - """Retrieve sensitive data with access control""" - try: - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - cursor.execute(""" - SELECT id, encrypted_payload, nonce, classification, integrity_hash, gcl_sequence, compression_stats - FROM sensitive_data WHERE pkg = ? - ORDER BY created_at DESC LIMIT 1 - """, (pkg,)) - - row = cursor.fetchone() - conn.close() - - if not row: - return {"success": False, "error": "Data not found"} - - data_id, encrypted_payload, nonce, classification, integrity_hash, gcl_sequence, compression_stats = row - - # Check access - if not self.security.check_access(clearance, AccessLevel(classification)): - return {"success": False, "error": "Access denied: insufficient clearance"} - - # Decrypt data - encrypted = { - "ciphertext": encrypted_payload, - "nonce": nonce - } - - decrypted = self.security.decrypt_data(encrypted, pkg.encode()) - - # Verify integrity (recompute and compare) - computed_hash = self.security.compute_integrity_hash(decrypted) - if computed_hash != integrity_hash: - return {"success": False, "error": "Integrity check failed"} - - return { - "success": True, - "id": data_id, - "payload": decrypted, - "classification": classification, - "gcl_sequence": gcl_sequence, - "compression_stats": json.loads(compression_stats) if compression_stats else None - } - except Exception as e: - return {"success": False, "error": str(e)} - -# Example usage -if __name__ == "__main__": - api = ENEAPIHook() - - # Store sensitive data - result = api.store_sensitive_data( - pkg="test/package", - payload="SECRET_INFORMATION", - classification=AccessLevel.SECRET, - semantic_vector=[0.5, 0.3, 0.7, 0.2] - ) - print("Store result:", result) - - # Retrieve with sufficient clearance - result = api.retrieve_sensitive_data("test/package", AccessLevel.SECRET) - print("Retrieve result (SECRET clearance):", result) - - # Retrieve with insufficient clearance - result = api.retrieve_sensitive_data("test/package", AccessLevel.PUBLIC) - print("Retrieve result (PUBLIC clearance):", result) diff --git a/4-Infrastructure/infra/ene_cloud_credential_manager.py b/4-Infrastructure/infra/ene_cloud_credential_manager.py deleted file mode 100644 index 6c27c455..00000000 --- a/4-Infrastructure/infra/ene_cloud_credential_manager.py +++ /dev/null @@ -1,656 +0,0 @@ -#!/usr/bin/env python3 -""" -ene_cloud_credential_manager.py — ENE Cloud Credential & Node Balancing System - -ENE (Endless Node Edges) manages API keys for cloud storage with: -- Secure credential storage via ENE encryption -- Node connection balancing (load distribution) -- Automatic failover and rotation -- Health monitoring per node - -Architecture: -- ENE: Central credential authority -- Nodes: Distributed connection endpoints -- Balancer: Round-robin with health checks -- Storage: Google Drive topological surface -""" - -import json -import time -import hashlib -import random -from dataclasses import dataclass, field -from typing import Dict, List, Optional, Any, Tuple -from datetime import datetime, timedelta -from pathlib import Path -import sqlite3 - -# Import ENE security -from ene_api import ENESecurityManager, AccessLevel - - -@dataclass -class CloudCredential: - """Encrypted cloud storage credential.""" - credential_id: str - provider: str # "gdrive", "s3", "azure", etc. - encrypted_payload: bytes - access_level: AccessLevel - node_assignments: List[str] # Which nodes can use this - usage_count: int = 0 - last_rotated: float = field(default_factory=time.time) - health_score: float = 1.0 # 0.0 to 1.0 - is_active: bool = True - - -@dataclass -class NodeConnection: - """Active node connection to cloud storage.""" - node_id: str - credential_id: str - connected_at: float - last_activity: float - bytes_transferred: int = 0 - error_count: int = 0 - latency_ms: float = 0.0 - status: str = "active" # active, draining, failed - - -@dataclass -class NodeStats: - """Statistics for a connection node.""" - node_id: str - total_connections: int = 0 - total_bytes: int = 0 - avg_latency: float = 0.0 - error_rate: float = 0.0 - health_score: float = 1.0 - - -class ENENodeBalancer: - """ - ENE Node Connection Balancer - - Distributes cloud storage connections across nodes with: - - Weighted round-robin (based on health scores) - - Health check monitoring - - Automatic failover - - Latency-aware routing - """ - - def __init__(self, db_path: str = "/home/allaun/Documents/Research Stack/shared-data/data/ene_cloud_nodes.db"): - self.db_path = db_path - self.nodes: Dict[str, NodeStats] = {} - self.active_connections: Dict[str, NodeConnection] = {} - self.credential_rotation_queue: List[str] = [] - self._init_database() - self._load_nodes() - - def _init_database(self): - """Initialize SQLite database for node tracking.""" - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - cursor.execute(""" - CREATE TABLE IF NOT EXISTS nodes ( - node_id TEXT PRIMARY KEY, - credential_id TEXT, - health_score REAL DEFAULT 1.0, - total_connections INTEGER DEFAULT 0, - total_bytes INTEGER DEFAULT 0, - avg_latency_ms REAL DEFAULT 0.0, - error_rate REAL DEFAULT 0.0, - last_seen REAL DEFAULT 0.0, - is_active INTEGER DEFAULT 1 - ) - """) - - cursor.execute(""" - CREATE TABLE IF NOT EXISTS connections ( - connection_id TEXT PRIMARY KEY, - node_id TEXT, - credential_id TEXT, - connected_at REAL, - last_activity REAL, - bytes_transferred INTEGER DEFAULT 0, - error_count INTEGER DEFAULT 0, - latency_ms REAL DEFAULT 0.0, - status TEXT DEFAULT 'active' - ) - """) - - conn.commit() - conn.close() - - def _load_nodes(self): - """Load nodes from database.""" - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - cursor.execute("SELECT node_id, health_score, total_connections, total_bytes, avg_latency_ms, error_rate FROM nodes WHERE is_active = 1") - for row in cursor.fetchall(): - node_id, health, connections, bytes_total, latency, error_rate = row - self.nodes[node_id] = NodeStats( - node_id=node_id, - total_connections=connections, - total_bytes=bytes_total, - avg_latency=latency, - error_rate=error_rate, - health_score=health - ) - - conn.close() - - def register_node(self, node_id: str, credential_id: str) -> bool: - """Register a new node for cloud storage connections.""" - if node_id in self.nodes: - return False - - self.nodes[node_id] = NodeStats(node_id=node_id) - - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - cursor.execute( - "INSERT OR REPLACE INTO nodes (node_id, credential_id, last_seen) VALUES (?, ?, ?)", - (node_id, credential_id, time.time()) - ) - conn.commit() - conn.close() - - return True - - def select_node(self, strategy: str = "health_weighted") -> Optional[str]: - """ - Select best node for new connection. - - Strategies: - - health_weighted: Weighted by health score - - round_robin: Simple rotation - - latency: Lowest latency - - least_connections: Fewest active connections - """ - active_nodes = [n for n in self.nodes.values() if n.health_score > 0.5] - - if not active_nodes: - return None - - if strategy == "health_weighted": - # Weighted random selection based on health scores - total_health = sum(n.health_score for n in active_nodes) - r = random.uniform(0, total_health) - cumulative = 0 - for node in active_nodes: - cumulative += node.health_score - if r <= cumulative: - return node.node_id - return active_nodes[-1].node_id - - elif strategy == "round_robin": - # Simple rotation (would need state persistence) - return active_nodes[0].node_id - - elif strategy == "latency": - # Select lowest latency node - return min(active_nodes, key=lambda n: n.avg_latency).node_id - - elif strategy == "least_connections": - # Select node with fewest connections - return min(active_nodes, key=lambda n: n.total_connections).node_id - - return active_nodes[0].node_id - - def record_connection(self, node_id: str, credential_id: str) -> str: - """Record new connection from node.""" - connection_id = f"conn_{hashlib.sha256(f'{node_id}{time.time()}'.encode()).hexdigest()[:16]}" - - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - cursor.execute( - """INSERT INTO connections - (connection_id, node_id, credential_id, connected_at, last_activity) - VALUES (?, ?, ?, ?, ?)""", - (connection_id, node_id, credential_id, time.time(), time.time()) - ) - - # Update node stats - cursor.execute( - "UPDATE nodes SET total_connections = total_connections + 1, last_seen = ? WHERE node_id = ?", - (time.time(), node_id) - ) - - conn.commit() - conn.close() - - # Update in-memory - self.active_connections[connection_id] = NodeConnection( - node_id=node_id, - credential_id=credential_id, - connected_at=time.time(), - last_activity=time.time() - ) - - if node_id in self.nodes: - self.nodes[node_id].total_connections += 1 - - return connection_id - - def update_connection_stats(self, connection_id: str, bytes_delta: int = 0, - latency_ms: float = 0.0, error: bool = False): - """Update connection statistics.""" - if connection_id not in self.active_connections: - return - - conn_info = self.active_connections[connection_id] - conn_info.last_activity = time.time() - conn_info.bytes_transferred += bytes_delta - - if error: - conn_info.error_count += 1 - - if latency_ms > 0: - conn_info.latency_ms = latency_ms - - # Update database - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - cursor.execute( - """UPDATE connections - SET last_activity = ?, bytes_transferred = bytes_transferred + ?, - error_count = error_count + ?, latency_ms = ? - WHERE connection_id = ?""", - (time.time(), bytes_delta, 1 if error else 0, latency_ms, connection_id) - ) - - conn.commit() - conn.close() - - def get_node_health(self, node_id: str) -> float: - """Calculate health score for a node (0.0 to 1.0).""" - if node_id not in self.nodes: - return 0.0 - - stats = self.nodes[node_id] - - # Factors: - # - Error rate (lower is better) - 40% weight - # - Latency (lower is better) - 30% weight - # - Connection count (moderate is good) - 30% weight - - error_factor = max(0, 1.0 - (stats.error_rate * 10)) # 10% errors = 0 health - latency_factor = max(0, 1.0 - (stats.avg_latency / 1000)) # 1000ms = 0 health - connection_factor = 1.0 if 10 <= stats.total_connections <= 100 else 0.7 - - health = (error_factor * 0.4) + (latency_factor * 0.3) + (connection_factor * 0.3) - return min(1.0, max(0.0, health)) - - def get_balancer_stats(self) -> Dict[str, Any]: - """Get overall balancer statistics.""" - total_nodes = len(self.nodes) - active_nodes = sum(1 for n in self.nodes.values() if n.health_score > 0.5) - total_connections = len(self.active_connections) - total_bytes = sum(n.total_bytes for n in self.nodes.values()) - - return { - "total_nodes": total_nodes, - "active_nodes": active_nodes, - "active_connections": total_connections, - "total_bytes_transferred": total_bytes, - "avg_node_health": sum(n.health_score for n in self.nodes.values()) / total_nodes if total_nodes > 0 else 0, - "nodes": {node_id: { - "health": stats.health_score, - "connections": stats.total_connections, - "bytes": stats.total_bytes, - "latency": stats.avg_latency - } for node_id, stats in self.nodes.items()} - } - - -class ENECloudCredentialManager: - """ - ENE Central Credential Authority - - Manages cloud storage API keys with: - - Secure encryption via ENE - - Credential rotation - - Node assignment - - Access control - """ - - def __init__(self, db_path: str = "/home/allaun/Documents/Research Stack/shared-data/data/ene_cloud_credentials.db"): - self.db_path = db_path - self.security = ENESecurityManager() - self.balancer = ENENodeBalancer() - self.credentials: Dict[str, CloudCredential] = {} - self._init_database() - self._load_credentials() - - def _init_database(self): - """Initialize credential database.""" - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - cursor.execute(""" - CREATE TABLE IF NOT EXISTS credentials ( - credential_id TEXT PRIMARY KEY, - provider TEXT, - encrypted_payload BLOB, - access_level INTEGER, - node_assignments TEXT, -- JSON list - usage_count INTEGER DEFAULT 0, - last_rotated REAL, - health_score REAL DEFAULT 1.0, - is_active INTEGER DEFAULT 1 - ) - """) - - conn.commit() - conn.close() - - def _load_credentials(self): - """Load credentials from database.""" - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - cursor.execute(""" - SELECT credential_id, provider, encrypted_payload, access_level, - node_assignments, usage_count, last_rotated, health_score - FROM credentials WHERE is_active = 1 - """) - - for row in cursor.fetchall(): - cred_id, provider, payload, access_level, nodes_json, usage, rotated, health = row - node_assignments = json.loads(nodes_json) if nodes_json else [] - - self.credentials[cred_id] = CloudCredential( - credential_id=cred_id, - provider=provider, - encrypted_payload=payload, - access_level=AccessLevel(access_level), - node_assignments=node_assignments, - usage_count=usage, - last_rotated=rotated, - health_score=health - ) - - conn.close() - - def store_credential(self, provider: str, api_key: str, secret: str, - node_assignments: List[str] = None, - access_level: AccessLevel = AccessLevel.RESTRICTED) -> str: - """Store new cloud credential (encrypted).""" - credential_id = f"cred_{provider}_{hashlib.sha256(str(time.time()).encode()).hexdigest()[:8]}" - - # Encrypt the credential using ENE - credential_data = json.dumps({"api_key": api_key, "secret": secret}) - encrypted_dict = self.security.encrypt_data(credential_data, provider.encode()) - encrypted = json.dumps(encrypted_dict).encode() - - nodes = node_assignments or [] - - # Store in database - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - cursor.execute( - """INSERT INTO credentials - (credential_id, provider, encrypted_payload, access_level, node_assignments, last_rotated) - VALUES (?, ?, ?, ?, ?, ?)""", - (credential_id, provider, encrypted, access_level.value, json.dumps(nodes), time.time()) - ) - - conn.commit() - conn.close() - - # Store in memory - self.credentials[credential_id] = CloudCredential( - credential_id=credential_id, - provider=provider, - encrypted_payload=encrypted, - access_level=access_level, - node_assignments=nodes - ) - - return credential_id - - def get_credential_for_node(self, node_id: str, provider: str) -> Optional[Dict[str, str]]: - """Get decrypted credential for a specific node.""" - # Find credential assigned to this node - eligible_creds = [ - c for c in self.credentials.values() - if c.provider == provider and (not c.node_assignments or node_id in c.node_assignments) - ] - - if not eligible_creds: - return None - - # Select healthiest credential - cred = max(eligible_creds, key=lambda c: c.health_score) - - # Decrypt using ENE - try: - encrypted_dict = json.loads(cred.encrypted_payload.decode()) - decrypted = self.security.decrypt_data(encrypted_dict, cred.provider.encode()) - data = json.loads(decrypted) - - # Update usage - cred.usage_count += 1 - self._update_credential_stats(cred.credential_id) - - return data - except Exception as e: - cred.health_score *= 0.9 # Degrade health on failure - return None - - def _update_credential_stats(self, credential_id: str): - """Update credential usage stats.""" - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - cursor.execute( - "UPDATE credentials SET usage_count = usage_count + 1 WHERE credential_id = ?", - (credential_id,) - ) - - conn.commit() - conn.close() - - def rotate_credential(self, credential_id: str, new_api_key: str, new_secret: str) -> bool: - """Rotate credential to new API key.""" - if credential_id not in self.credentials: - return False - - # Encrypt new credentials using ENE - credential_data = json.dumps({"api_key": new_api_key, "secret": new_secret}) - encrypted_dict = self.security.encrypt_data(credential_data, b"gdrive") - encrypted = json.dumps(encrypted_dict).encode() - - # Update database - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - cursor.execute( - "UPDATE credentials SET encrypted_payload = ?, last_rotated = ? WHERE credential_id = ?", - (encrypted, time.time(), credential_id) - ) - - conn.commit() - conn.close() - - # Update memory - self.credentials[credential_id].encrypted_payload = encrypted - self.credentials[credential_id].last_rotated = time.time() - - return True - - def get_connection(self, provider: str, node_id: str = None) -> Tuple[str, str, Dict[str, str]]: - """ - Get connection for cloud storage. - - Returns: (connection_id, node_id, credentials) - """ - # Select node if not specified - if node_id is None: - node_id = self.balancer.select_node(strategy="health_weighted") - - if not node_id: - raise Exception("No healthy nodes available") - - # Get credential for node - creds = self.get_credential_for_node(node_id, provider) - if not creds: - raise Exception(f"No credentials available for {provider} on node {node_id}") - - # Find credential ID - cred_id = None - for cid, c in self.credentials.items(): - if c.provider == provider and (not c.node_assignments or node_id in c.node_assignments): - cred_id = cid - break - - # Record connection - connection_id = self.balancer.record_connection(node_id, cred_id) - - return connection_id, node_id, creds - - def report_connection_result(self, connection_id: str, bytes_transferred: int = 0, - latency_ms: float = 0.0, error: bool = False): - """Report result of connection operation.""" - self.balancer.update_connection_stats(connection_id, bytes_transferred, latency_ms, error) - - -# ═══════════════════════════════════════════════════════════════════════════ -# Integration with Topological Storage -# ═══════════════════════════════════════════════════════════════════════════ - -class ENETopologicalStorage: - """ - ENE-managed topological storage surface. - - Combines credential management with node balancing for - automatic cloud storage operations. - """ - - def __init__(self): - self.credential_manager = ENECloudCredentialManager() - self.balancer = self.credential_manager.balancer - - def upload_waveprobe(self, local_path: str, remote_path: str) -> Dict[str, Any]: - """ - Upload file via ENE-managed connection. - - Automatically: - 1. Selects best node - 2. Retrieves encrypted credentials - 3. Executes upload - 4. Reports stats - """ - # Get connection (node + credentials) - connection_id, node_id, creds = self.credential_manager.get_connection("gdrive") - - # Simulate upload (would use actual Rclone with credentials) - start_time = time.time() - - # In real implementation, would call rclone with the credentials - # rclone copy local_path Gdrive:topological_storage/... - - # Simulate latency - import random - latency_ms = random.uniform(50, 200) - time.sleep(latency_ms / 1000) - - # Get file size - file_size = Path(local_path).stat().st_size - - # Report success - self.credential_manager.report_connection_result( - connection_id, - bytes_transferred=file_size, - latency_ms=latency_ms, - error=False - ) - - return { - "connection_id": connection_id, - "node_id": node_id, - "uploaded": True, - "bytes": file_size, - "latency_ms": latency_ms, - "duration": time.time() - start_time - } - - def get_storage_health(self) -> Dict[str, Any]: - """Get overall topological storage health.""" - balancer_stats = self.balancer.get_balancer_stats() - - return { - "topological_storage": "operational", - "provider": "gdrive", - "mount_point": "Gdrive:topological_storage", - "ene_managed": True, - "node_balancing": "active", - "credential_rotation": "enabled", - "balancer_stats": balancer_stats - } - - -# ═══════════════════════════════════════════════════════════════════════════ -# Example Usage -# ═══════════════════════════════════════════════════════════════════════════ - -if __name__ == "__main__": - print("=" * 70) - print("ENE CLOUD CREDENTIAL & NODE BALANCING SYSTEM") - print("=" * 70) - - # Initialize ENE system - ene = ENETopologicalStorage() - - # Register nodes - print("\n[1] Registering nodes...") - for i in range(3): - node_id = f"node_{i+1}" - ene.balancer.register_node(node_id, f"cred_gdrive_{i+1}") - print(f" Registered: {node_id}") - - # Store credential - print("\n[2] Storing Google Drive credential...") - cred_id = ene.credential_manager.store_credential( - provider="gdrive", - api_key="[REDACTED_GOOGLE_API_KEY]", - secret="[REDACTED_GOOGLE_SECRET]", - node_assignments=["node_1", "node_2", "node_3"] - ) - print(f" Credential ID: {cred_id}") - print(f" Encrypted: AES-256-GCM via ENE") - - # Get connection - print("\n[3] Getting connection via node balancing...") - conn_id, node_id, creds = ene.credential_manager.get_connection("gdrive") - print(f" Connection ID: {conn_id}") - print(f" Selected Node: {node_id}") - print(f" Strategy: health_weighted") - - # Simulate upload - print("\n[4] Simulating waveprobe upload...") - result = ene.upload_waveprobe( - local_path="/tmp/test_waveprobe.json", - remote_path="Gdrive:topological_storage/waveprobes/test.json" - ) - print(f" Uploaded via: {result['node_id']}") - print(f" Latency: {result['latency_ms']:.1f}ms") - print(f" Bytes: {result['bytes']}") - - # Get health - print("\n[5] Storage health...") - health = ene.get_storage_health() - print(f" Status: {health['topological_storage']}") - print(f" ENE Managed: {health['ene_managed']}") - print(f" Nodes: {health['balancer_stats']['total_nodes']}") - print(f" Active Connections: {health['balancer_stats']['active_connections']}") - - print("\n" + "=" * 70) - print("ENE SYSTEM OPERATIONAL") - print("API keys secured | Nodes balanced | Storage topological") - print("=" * 70) diff --git a/4-Infrastructure/infra/ene_fractal_fold.py b/4-Infrastructure/infra/ene_fractal_fold.py deleted file mode 100644 index bbb7714f..00000000 --- a/4-Infrastructure/infra/ene_fractal_fold.py +++ /dev/null @@ -1,895 +0,0 @@ -#!/usr/bin/env python3 -"""ENE fractal fold codec. - -This module gives ENE one canonical substrate for: - -- fractal encoding: recursive self-similar nodes with O(log n) path traversal; -- manifold folding: Gray-code folded addresses that preserve near neighbors; -- damage prevention: parent/child hash consistency checks; -- endless growth: no fixed maximum tree size, only local storage limits. - -It is a compact Python carrier for the ENE schema. It is not a compression -oracle; it is a lawful addressing and verification layer that other codecs can -use before or after Delta GCL, MS3C, or wiki/package binding. -""" - -from __future__ import annotations - -import argparse -import base64 -import hashlib -import json -import math -import sqlite3 -import time -import xml.etree.ElementTree as ET -from dataclasses import asdict, dataclass -from pathlib import Path -from typing import Any - - -VERSION = "ene_fractal_fold_v1" -SOURCE_TYPE = "json_catalog" -GOLDEN_ANGLE = math.pi * (3.0 - math.sqrt(5.0)) - - -def canonical_json(obj: Any) -> str: - return json.dumps(obj, sort_keys=True, separators=(",", ":"), ensure_ascii=False) - - -def sha256_text(text: str) -> str: - return hashlib.sha256(text.encode("utf-8")).hexdigest() - - -def sha256_bytes(data: bytes) -> str: - return hashlib.sha256(data).hexdigest() - - -def iso_utc() -> str: - return time.strftime("%Y-%m-%dT%H:%M:%S", time.gmtime()) - - -def gray_code(index: int) -> int: - """Fold a linear index into a Gray-code neighborhood address.""" - return index ^ (index >> 1) - - -def inverse_gray_code(code: int) -> int: - value = code - while code > 0: - code >>= 1 - value ^= code - return value - - -def golden_spiral_point(address: int, level: int = 0) -> dict[str, float]: - """Project a folded address into a deterministic golden-spiral chart.""" - radius = math.sqrt(address + 1.0) - theta = (address + 1.0) * GOLDEN_ANGLE - shell = float(level + 1) - return { - "x": round(radius * math.cos(theta), 9), - "y": round(radius * math.sin(theta), 9), - "r": round(radius, 9), - "theta": round(theta % (2.0 * math.pi), 9), - "shell": shell, - } - - -def manifold_distance(a: dict[str, float], b: dict[str, float]) -> float: - dx = a["x"] - b["x"] - dy = a["y"] - b["y"] - ds = a["shell"] - b["shell"] - return math.sqrt(dx * dx + dy * dy + ds * ds) - - -def tree_depth(leaves: int, branching_factor: int) -> int: - if leaves <= 1: - return 0 - return math.ceil(math.log(leaves, branching_factor)) - - -@dataclass(frozen=True) -class FractalNode: - node_hash: str - kind: str - level: int - ordinal: int - fold_address: int - start_leaf: int - end_leaf: int - size_bytes: int - children: list[str] - payload_b64: str | None = None - - -@dataclass(frozen=True) -class FractalManifest: - root_hash: str - name: str - byte_len: int - leaves_count: int - depth: int - chunk_size: int - branching_factor: int - created_at: str - receipt: str - - -def make_leaf(chunk: bytes, ordinal: int) -> FractalNode: - payload_hash = sha256_bytes(chunk) - body = { - "v": VERSION, - "kind": "leaf", - "level": 0, - "ordinal": ordinal, - "fold_address": gray_code(ordinal), - "size_bytes": len(chunk), - "payload_hash": payload_hash, - } - return FractalNode( - node_hash=sha256_text(canonical_json(body)), - kind="leaf", - level=0, - ordinal=ordinal, - fold_address=gray_code(ordinal), - start_leaf=ordinal, - end_leaf=ordinal, - size_bytes=len(chunk), - children=[], - payload_b64=base64.b64encode(chunk).decode("ascii"), - ) - - -def make_parent(children: list[FractalNode], level: int, ordinal: int) -> FractalNode: - body = { - "v": VERSION, - "kind": "parent", - "level": level, - "ordinal": ordinal, - "fold_address": gray_code(ordinal), - "start_leaf": children[0].start_leaf, - "end_leaf": children[-1].end_leaf, - "size_bytes": sum(child.size_bytes for child in children), - "children": [child.node_hash for child in children], - } - return FractalNode( - node_hash=sha256_text(canonical_json(body)), - kind="parent", - level=level, - ordinal=ordinal, - fold_address=gray_code(ordinal), - start_leaf=children[0].start_leaf, - end_leaf=children[-1].end_leaf, - size_bytes=sum(child.size_bytes for child in children), - children=[child.node_hash for child in children], - ) - - -def encode_fractal_chunks(chunks: list[bytes], name: str, chunk_size: int, branching_factor: int = 4) -> tuple[FractalManifest, list[FractalNode]]: - if branching_factor < 2: - raise ValueError("branching_factor must be at least 2") - chunks = chunks or [b""] - all_nodes: list[FractalNode] = [make_leaf(chunk, idx) for idx, chunk in enumerate(chunks)] - current = all_nodes[:] - level = 1 - while len(current) > 1: - parents: list[FractalNode] = [] - for idx in range(0, len(current), branching_factor): - parent = make_parent(current[idx : idx + branching_factor], level, len(parents)) - parents.append(parent) - all_nodes.extend(parents) - current = parents - level += 1 - - root = current[0] - created_at = iso_utc() - receipt_body = { - "v": VERSION, - "name": name, - "root_hash": root.node_hash, - "byte_len": sum(len(chunk) for chunk in chunks), - "leaves_count": len(chunks), - "depth": level - 1, - "chunk_size": chunk_size, - "branching_factor": branching_factor, - "created_at": created_at, - } - receipt = sha256_text(canonical_json(receipt_body)) - manifest = FractalManifest( - root_hash=root.node_hash, - name=name, - byte_len=sum(len(chunk) for chunk in chunks), - leaves_count=len(chunks), - depth=level - 1, - chunk_size=chunk_size, - branching_factor=branching_factor, - created_at=created_at, - receipt=receipt, - ) - return manifest, all_nodes - - -def encode_fractal(data: bytes, name: str, chunk_size: int = 4096, branching_factor: int = 4) -> tuple[FractalManifest, list[FractalNode]]: - if chunk_size <= 0: - raise ValueError("chunk_size must be positive") - chunks = [data[i : i + chunk_size] for i in range(0, len(data), chunk_size)] or [b""] - return encode_fractal_chunks(chunks, name, chunk_size, branching_factor) - - -def parse_graphml_concepts(graphml: bytes) -> tuple[list[dict[str, Any]], list[bytes]]: - """Convert GraphML nodes/edges into ENE concept records. - - Each graph node becomes one fractal leaf so concept retrieval maps directly - to the O(log n) navigation path. - """ - root = ET.fromstring(graphml) - ns = {"g": "http://graphml.graphdrawing.org/xmlns"} - key_names = { - key.attrib.get("id", ""): key.attrib.get("attr.name", key.attrib.get("id", "")) - for key in root.findall("g:key", ns) - } - nodes: dict[str, dict[str, Any]] = {} - adjacency: dict[str, list[str]] = {} - for node in root.findall(".//g:node", ns): - node_id = node.attrib["id"] - data = {} - for item in node.findall("g:data", ns): - data[key_names.get(item.attrib.get("key", ""), item.attrib.get("key", ""))] = item.text or "" - nodes[node_id] = { - "graph_node_id": node_id, - "name": data.get("name", node_id), - "family": data.get("family", ""), - "domain": data.get("domain", ""), - } - adjacency[node_id] = [] - for edge in root.findall(".//g:edge", ns): - source = edge.attrib.get("source") - target = edge.attrib.get("target") - if source in adjacency and target: - adjacency[source].append(target) - - records = [] - for idx, node_id in enumerate(sorted(nodes)): - record = { - "kind": "graphml_concept", - "leaf_index": idx, - "fold_address": gray_code(idx), - "golden_spiral": golden_spiral_point(gray_code(idx), 0), - **nodes[node_id], - "neighbors": sorted(adjacency.get(node_id, [])), - } - records.append(record) - chunks = [canonical_json(record).encode("utf-8") for record in records] - return records, chunks - - -def node_record(node: FractalNode) -> dict[str, Any]: - return asdict(node) - - -def archive_record(manifest: FractalManifest) -> dict[str, Any]: - raw_content = { - "kind": "ene_fractal_fold_manifest", - "version": VERSION, - "manifest": asdict(manifest), - "properties": { - "fractal_encoding": "self_similar_recursive_tree", - "traversal": "O(log_B(n)) by root-to-leaf proof path", - "entity_retrieval": "O(log_B(n)) via manifold_distance_pruning", - "manifold_folding": "gray_code_folded_leaf_addresses", - "navigation": "golden_spiral_address_chart", - "damage_prevention": "fractal_hash_parent_child_consistency", - "endless": "unbounded_recursion_subject_to_storage_policy", - }, - } - content_hash = sha256_text(canonical_json(raw_content)) - return { - "archive_id": f"json_catalog_ene_fractal_{manifest.root_hash[:16]}", - "source_type": SOURCE_TYPE, - "source_file": f"ene-fractal://{manifest.root_hash}", - "raw_content": raw_content, - "extracted_text": f"{manifest.name} {manifest.root_hash} fractal fold manifold compression", - "extracted_at": manifest.created_at, - "content_hash": content_hash, - "extraction_version": VERSION, - } - - -def jsonl_event(record: dict[str, Any], manifest: FractalManifest) -> dict[str, Any]: - pkg = f"ene/fractal/{manifest.root_hash[:16]}" - event = { - "src": "ene", - "op": "upsert", - "data": { - "pkg": pkg, - "version": VERSION, - "tier": "RESEARCH", - "domain": "topology", - "archetype": "fractal_fold", - "description": "ENE fractal fold manifest", - "tags": ["ene", "fractal", "fold", "manifold", "compression"], - "source": "ene_fractal_fold", - "sha256": record["content_hash"], - }, - "bind": { - "lawful": True, - "class": "geometric_bind", - "invariant": "parent_child_hash_consistency", - }, - "provenance": { - "archive_id": record["archive_id"], - "attestation_hash": f"sha256:{manifest.receipt}", - "node": "ene-fractal-fold", - }, - } - event["id"] = f"ene:{sha256_text(canonical_json(event))[:32]}" - return event - - -class ENEFractalFold: - def __init__(self, db_path: str | Path): - self.db_path = Path(db_path) - self.db_path.parent.mkdir(parents=True, exist_ok=True) - self._init_db() - - def _connect(self) -> sqlite3.Connection: - conn = sqlite3.connect(self.db_path) - conn.row_factory = sqlite3.Row - return conn - - def _init_db(self) -> None: - with self._connect() as conn: - conn.executescript( - """ - CREATE TABLE IF NOT EXISTS ene_fractal_manifolds ( - root_hash TEXT PRIMARY KEY, - name TEXT NOT NULL, - byte_len INTEGER NOT NULL, - leaves_count INTEGER NOT NULL, - depth INTEGER NOT NULL, - chunk_size INTEGER NOT NULL, - branching_factor INTEGER NOT NULL, - created_at TEXT NOT NULL, - receipt TEXT NOT NULL, - archive_record TEXT NOT NULL, - jsonl_event TEXT NOT NULL - ); - - CREATE TABLE IF NOT EXISTS ene_fractal_nodes ( - root_hash TEXT NOT NULL, - node_hash TEXT NOT NULL, - kind TEXT NOT NULL, - level INTEGER NOT NULL, - ordinal INTEGER NOT NULL, - fold_address INTEGER NOT NULL, - start_leaf INTEGER NOT NULL, - end_leaf INTEGER NOT NULL, - size_bytes INTEGER NOT NULL, - children TEXT NOT NULL, - payload_b64 TEXT, - PRIMARY KEY (root_hash, node_hash) - ); - - CREATE INDEX IF NOT EXISTS idx_ene_fractal_leaf - ON ene_fractal_nodes(root_hash, level, ordinal); - - CREATE TABLE IF NOT EXISTS ene_fractal_graph_entities ( - root_hash TEXT NOT NULL, - graph_node_id TEXT NOT NULL, - leaf_index INTEGER NOT NULL, - name TEXT NOT NULL, - family TEXT, - domain TEXT, - neighbors TEXT NOT NULL, - PRIMARY KEY (root_hash, graph_node_id) - ); - - CREATE INDEX IF NOT EXISTS idx_ene_fractal_graph_name - ON ene_fractal_graph_entities(root_hash, name); - """ - ) - - def _store(self, manifest: FractalManifest, nodes: list[FractalNode]) -> tuple[dict[str, Any], dict[str, Any]]: - record = archive_record(manifest) - event = jsonl_event(record, manifest) - with self._connect() as conn: - conn.execute( - """ - INSERT OR REPLACE INTO ene_fractal_manifolds - (root_hash, name, byte_len, leaves_count, depth, chunk_size, branching_factor, - created_at, receipt, archive_record, jsonl_event) - VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) - """, - ( - manifest.root_hash, - manifest.name, - manifest.byte_len, - manifest.leaves_count, - manifest.depth, - manifest.chunk_size, - manifest.branching_factor, - manifest.created_at, - manifest.receipt, - canonical_json(record), - canonical_json(event), - ), - ) - conn.execute("DELETE FROM ene_fractal_nodes WHERE root_hash = ?", (manifest.root_hash,)) - conn.executemany( - """ - INSERT INTO ene_fractal_nodes - (root_hash, node_hash, kind, level, ordinal, fold_address, start_leaf, - end_leaf, size_bytes, children, payload_b64) - VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) - """, - [ - ( - manifest.root_hash, - node.node_hash, - node.kind, - node.level, - node.ordinal, - node.fold_address, - node.start_leaf, - node.end_leaf, - node.size_bytes, - canonical_json(node.children), - node.payload_b64, - ) - for node in nodes - ], - ) - return record, event - - def put(self, data: bytes, name: str = "unnamed", chunk_size: int = 4096, branching_factor: int = 4) -> dict[str, Any]: - manifest, nodes = encode_fractal(data, name, chunk_size, branching_factor) - record, event = self._store(manifest, nodes) - return { - "ok": True, - "op": "fractal_put", - "manifest": asdict(manifest), - "archive_record": record, - "jsonl_event": event, - } - - def put_graphml(self, graphml: bytes, name: str = "graphml", branching_factor: int = 4) -> dict[str, Any]: - records, chunks = parse_graphml_concepts(graphml) - max_chunk = max((len(chunk) for chunk in chunks), default=0) - manifest, nodes = encode_fractal_chunks(chunks, name, max_chunk, branching_factor) - record, event = self._store(manifest, nodes) - with self._connect() as conn: - conn.execute("DELETE FROM ene_fractal_graph_entities WHERE root_hash = ?", (manifest.root_hash,)) - conn.executemany( - """ - INSERT INTO ene_fractal_graph_entities - (root_hash, graph_node_id, leaf_index, name, family, domain, neighbors) - VALUES (?, ?, ?, ?, ?, ?, ?) - """, - [ - ( - manifest.root_hash, - concept["graph_node_id"], - concept["leaf_index"], - concept["name"], - concept["family"], - concept["domain"], - canonical_json(concept["neighbors"]), - ) - for concept in records - ], - ) - return { - "ok": True, - "op": "fractal_graphml_put", - "manifest": asdict(manifest), - "graphml": {"concepts": len(records), "concept_leaf_mode": True}, - "archive_record": record, - "jsonl_event": event, - } - - def manifest(self, root_hash: str) -> dict[str, Any] | None: - with self._connect() as conn: - row = conn.execute( - "SELECT * FROM ene_fractal_manifolds WHERE root_hash = ?", - (root_hash,), - ).fetchone() - if row is None: - return None - return { - "root_hash": row["root_hash"], - "name": row["name"], - "byte_len": row["byte_len"], - "leaves_count": row["leaves_count"], - "depth": row["depth"], - "chunk_size": row["chunk_size"], - "branching_factor": row["branching_factor"], - "created_at": row["created_at"], - "receipt": row["receipt"], - "archive_record": json.loads(row["archive_record"]), - "jsonl_event": json.loads(row["jsonl_event"]), - } - - def _node(self, conn: sqlite3.Connection, root_hash: str, node_hash: str) -> sqlite3.Row: - row = conn.execute( - "SELECT * FROM ene_fractal_nodes WHERE root_hash = ? AND node_hash = ?", - (root_hash, node_hash), - ).fetchone() - if row is None: - raise KeyError(f"missing fractal node {node_hash}") - return row - - def proof(self, root_hash: str, leaf_index: int) -> dict[str, Any]: - meta = self.manifest(root_hash) - if meta is None: - raise KeyError(f"unknown root {root_hash}") - if leaf_index < 0 or leaf_index >= meta["leaves_count"]: - raise IndexError("leaf_index outside manifold") - - with self._connect() as conn: - frontier = root_hash - path = [] - target_point = golden_spiral_point(gray_code(leaf_index), 0) - while True: - row = self._node(conn, root_hash, frontier) - children = json.loads(row["children"]) - node_point = golden_spiral_point(row["fold_address"], row["level"]) - path.append( - {key: row[key] for key in row.keys() if key != "payload_b64"} - | { - "children": children, - "golden_spiral": node_point, - "distance_to_target": round(manifold_distance(node_point, target_point), 9), - } - ) - if row["kind"] == "leaf": - payload = base64.b64decode(row["payload_b64"] or "") - break - next_hash = None - pruned = [] - for child_hash in children: - child = self._node(conn, root_hash, child_hash) - child_point = golden_spiral_point(child["fold_address"], child["level"]) - pruned.append( - { - "node_hash": child_hash, - "covers_target": child["start_leaf"] <= leaf_index <= child["end_leaf"], - "distance": round(manifold_distance(child_point, target_point), 9), - } - ) - if child["start_leaf"] <= leaf_index <= child["end_leaf"]: - next_hash = child_hash - path[-1]["manifold_distance_pruning"] = sorted(pruned, key=lambda item: item["distance"]) - if next_hash is None: - raise ValueError("corrupt tree: no child covers requested leaf") - frontier = next_hash - - path_valid = self._verify_path_rows(path, payload) - return { - "ok": True, - "op": "fractal_proof", - "root_hash": root_hash, - "leaf_index": leaf_index, - "fold_address": gray_code(leaf_index), - "inverse_fold_address": inverse_gray_code(gray_code(leaf_index)), - "golden_spiral": golden_spiral_point(gray_code(leaf_index), 0), - "traversal_cost": len(path), - "expected_complexity": f"O(log_{meta['branching_factor']}(n))", - "path_hash_verified": path_valid, - "path": path, - "payload_b64": base64.b64encode(payload).decode("ascii"), - } - - def _verify_path_rows(self, path: list[dict[str, Any]], payload: bytes) -> bool: - if not path: - return False - leaf = path[-1] - if leaf["kind"] != "leaf": - return False - expected_leaf = make_leaf(payload, leaf["ordinal"]).node_hash - if expected_leaf != leaf["node_hash"]: - return False - child_hash = expected_leaf - for row in reversed(path[:-1]): - if child_hash not in row["children"]: - return False - child_hash = row["node_hash"] - return child_hash == path[0]["node_hash"] - - def verify(self, root_hash: str) -> dict[str, Any]: - meta = self.manifest(root_hash) - if meta is None: - raise KeyError(f"unknown root {root_hash}") - errors = [] - with self._connect() as conn: - rows = conn.execute( - "SELECT * FROM ene_fractal_nodes WHERE root_hash = ?", - (root_hash,), - ).fetchall() - by_hash = {row["node_hash"]: row for row in rows} - for row in rows: - children = json.loads(row["children"]) - if row["kind"] == "leaf": - payload = base64.b64decode(row["payload_b64"] or "") - expected = make_leaf(payload, row["ordinal"]).node_hash - if expected != row["node_hash"]: - errors.append({"node": row["node_hash"], "error": "leaf_hash_mismatch", "expected": expected}) - continue - child_nodes = [] - for child_hash in children: - child = by_hash.get(child_hash) - if child is None: - errors.append({"node": row["node_hash"], "error": "missing_child", "child": child_hash}) - continue - child_nodes.append( - FractalNode( - node_hash=child["node_hash"], - kind=child["kind"], - level=child["level"], - ordinal=child["ordinal"], - fold_address=child["fold_address"], - start_leaf=child["start_leaf"], - end_leaf=child["end_leaf"], - size_bytes=child["size_bytes"], - children=json.loads(child["children"]), - payload_b64=child["payload_b64"], - ) - ) - if len(child_nodes) == len(children): - expected = make_parent(child_nodes, row["level"], row["ordinal"]).node_hash - if expected != row["node_hash"]: - errors.append({"node": row["node_hash"], "error": "parent_hash_mismatch", "expected": expected}) - - return { - "ok": not errors, - "op": "fractal_verify", - "root_hash": root_hash, - "checked_nodes": len(rows), - "errors": errors, - "damage_detected": bool(errors), - } - - def neighbors(self, root_hash: str, leaf_index: int) -> dict[str, Any]: - meta = self.manifest(root_hash) - if meta is None: - raise KeyError(f"unknown root {root_hash}") - candidates = sorted(set(idx for idx in (leaf_index - 1, leaf_index, leaf_index + 1) if 0 <= idx < meta["leaves_count"])) - return { - "ok": True, - "op": "fractal_neighbors", - "root_hash": root_hash, - "leaf_index": leaf_index, - "fold_address": gray_code(leaf_index), - "neighbors": [ - { - "leaf_index": idx, - "fold_address": gray_code(idx), - "golden_spiral": golden_spiral_point(gray_code(idx), 0), - "fold_distance": bin(gray_code(idx) ^ gray_code(leaf_index)).count("1"), - } - for idx in candidates - ], - } - - def graph_entity(self, root_hash: str, graph_node_id: str = "", name: str = "") -> dict[str, Any]: - with self._connect() as conn: - if graph_node_id: - row = conn.execute( - "SELECT * FROM ene_fractal_graph_entities WHERE root_hash = ? AND graph_node_id = ?", - (root_hash, graph_node_id), - ).fetchone() - else: - row = conn.execute( - "SELECT * FROM ene_fractal_graph_entities WHERE root_hash = ? AND lower(name) = lower(?)", - (root_hash, name), - ).fetchone() - if row is None: - return {"ok": False, "op": "fractal_graph_entity", "error": "graph entity not found"} - proof = self.navigate(root_hash, row["leaf_index"]) - return { - "ok": True, - "op": "fractal_graph_entity", - "root_hash": root_hash, - "entity": { - "graph_node_id": row["graph_node_id"], - "leaf_index": row["leaf_index"], - "name": row["name"], - "family": row["family"], - "domain": row["domain"], - "neighbors": json.loads(row["neighbors"]), - }, - "retrieval": proof, - } - - def graph_neighbors(self, root_hash: str, graph_node_id: str) -> dict[str, Any]: - entity = self.graph_entity(root_hash, graph_node_id=graph_node_id) - if not entity.get("ok"): - return entity - neighbor_ids = entity["entity"]["neighbors"] - with self._connect() as conn: - rows = [ - conn.execute( - "SELECT * FROM ene_fractal_graph_entities WHERE root_hash = ? AND graph_node_id = ?", - (root_hash, neighbor_id), - ).fetchone() - for neighbor_id in neighbor_ids - ] - return { - "ok": True, - "op": "fractal_graph_neighbors", - "root_hash": root_hash, - "graph_node_id": graph_node_id, - "neighbors": [ - { - "graph_node_id": row["graph_node_id"], - "leaf_index": row["leaf_index"], - "name": row["name"], - "family": row["family"], - "domain": row["domain"], - "fold_address": gray_code(row["leaf_index"]), - "golden_spiral": golden_spiral_point(gray_code(row["leaf_index"]), 0), - } - for row in rows - if row is not None - ], - } - - def navigate(self, root_hash: str, leaf_index: int) -> dict[str, Any]: - """Retrieve an entity path by fixed-factor manifold-distance pruning.""" - meta = self.manifest(root_hash) - if meta is None: - raise KeyError(f"unknown root {root_hash}") - if leaf_index < 0 or leaf_index >= meta["leaves_count"]: - raise IndexError("leaf_index outside manifold") - - target_point = golden_spiral_point(gray_code(leaf_index), 0) - with self._connect() as conn: - frontier = root_hash - path = [] - while True: - row = self._node(conn, root_hash, frontier) - children = json.loads(row["children"]) - path.append( - { - "node_hash": row["node_hash"], - "kind": row["kind"], - "level": row["level"], - "ordinal": row["ordinal"], - "start_leaf": row["start_leaf"], - "end_leaf": row["end_leaf"], - "fold_address": row["fold_address"], - "children": children, - "golden_spiral": golden_spiral_point(row["fold_address"], row["level"]), - } - ) - if row["kind"] == "leaf": - payload = base64.b64decode(row["payload_b64"] or "") - break - - ranked = [] - for child_hash in children: - child = self._node(conn, root_hash, child_hash) - point = golden_spiral_point(child["fold_address"], child["level"]) - ranked.append( - { - "node_hash": child_hash, - "distance": manifold_distance(point, target_point), - "covers_target": child["start_leaf"] <= leaf_index <= child["end_leaf"], - } - ) - ranked.sort(key=lambda item: (not item["covers_target"], item["distance"])) - path[-1]["pruned_candidates"] = [ - { - "node_hash": item["node_hash"], - "distance": round(item["distance"], 9), - "covers_target": item["covers_target"], - } - for item in ranked - ] - frontier = ranked[0]["node_hash"] - - return { - "ok": True, - "op": "fractal_navigate", - "root_hash": root_hash, - "leaf_index": leaf_index, - "target": { - "fold_address": gray_code(leaf_index), - "golden_spiral": target_point, - }, - "retrieval_complexity": f"O(log_{meta['branching_factor']}(n))", - "navigation": "golden_spiral_manifold_distance_pruning", - "path_hash_verified": self._verify_path_rows(path, payload), - "path": path, - "payload_b64": base64.b64encode(payload).decode("ascii"), - } - - def handle_request(self, request: dict[str, Any]) -> dict[str, Any]: - op = str(request.get("op", "manifest")) - if op in {"put", "encode"}: - if "data_b64" in request: - data = base64.b64decode(str(request["data_b64"])) - else: - data = str(request.get("text", "")).encode("utf-8") - return self.put( - data=data, - name=str(request.get("name", "unnamed")), - chunk_size=int(request.get("chunk_size", 4096)), - branching_factor=int(request.get("branching_factor", 4)), - ) - if op in {"put_graphml", "graphml"}: - if "data_b64" in request: - data = base64.b64decode(str(request["data_b64"])) - else: - data = str(request.get("text", "")).encode("utf-8") - return self.put_graphml( - graphml=data, - name=str(request.get("name", "graphml")), - branching_factor=int(request.get("branching_factor", 4)), - ) - root_hash = str(request.get("root_hash", "")) - if not root_hash: - raise ValueError("root_hash is required for this operation") - if op == "manifest": - meta = self.manifest(root_hash) - return {"ok": meta is not None, "op": "fractal_manifest", "manifest": meta} - if op == "proof": - return self.proof(root_hash, int(request.get("leaf_index", 0))) - if op in {"navigate", "get"}: - return self.navigate(root_hash, int(request.get("leaf_index", 0))) - if op == "verify": - return self.verify(root_hash) - if op == "neighbors": - return self.neighbors(root_hash, int(request.get("leaf_index", 0))) - if op in {"graph_entity", "concept"}: - return self.graph_entity( - root_hash, - graph_node_id=str(request.get("graph_node_id", "")), - name=str(request.get("name", "")), - ) - if op in {"graph_neighbors", "concept_neighbors"}: - return self.graph_neighbors(root_hash, str(request.get("graph_node_id", ""))) - raise ValueError(f"unsupported fractal op {op!r}") - - -def main() -> int: - parser = argparse.ArgumentParser(description="ENE fractal fold codec") - parser.add_argument("--db", default="data/ene-fractal-fold.db") - parser.add_argument("--op", default="put") - parser.add_argument("--name", default="cli") - parser.add_argument("--text", default="") - parser.add_argument("--file", type=argparse.FileType("rb")) - parser.add_argument("--graph-node-id") - parser.add_argument("--root-hash") - parser.add_argument("--leaf-index", type=int, default=0) - parser.add_argument("--chunk-size", type=int, default=4096) - parser.add_argument("--branching-factor", type=int, default=4) - args = parser.parse_args() - - layer = ENEFractalFold(args.db) - if args.op in {"put", "encode"}: - data = args.file.read() if args.file else args.text.encode("utf-8") - result = layer.put(data, args.name, args.chunk_size, args.branching_factor) - elif args.op in {"put_graphml", "graphml"}: - data = args.file.read() if args.file else args.text.encode("utf-8") - result = layer.put_graphml(data, args.name, args.branching_factor) - elif args.op in {"graph_entity", "concept", "graph_neighbors"}: - result = layer.handle_request( - { - "op": args.op, - "root_hash": args.root_hash, - "leaf_index": args.leaf_index, - "graph_node_id": args.graph_node_id or "", - "name": args.name, - } - ) - else: - result = layer.handle_request( - {"op": args.op, "root_hash": args.root_hash, "leaf_index": args.leaf_index} - ) - print(json.dumps(result, indent=2, sort_keys=True)) - return 0 - - -if __name__ == "__main__": - raise SystemExit(main()) diff --git a/4-Infrastructure/infra/ene_meta_autotype.py b/4-Infrastructure/infra/ene_meta_autotype.py deleted file mode 100644 index 9f539062..00000000 --- a/4-Infrastructure/infra/ene_meta_autotype.py +++ /dev/null @@ -1,219 +0,0 @@ -#!/usr/bin/env python3 -"""ENE meta-autotype shim. - -When ENE sees data without a defined ingestion surface, this module creates -contingent fields instead of pretending the schema is known. It is intentionally -deterministic: a tiny classifier/autotyper with receipts, not an external LLM. -""" - -from __future__ import annotations - -import argparse -import base64 -import hashlib -import json -import re -import time -from dataclasses import asdict, dataclass -from typing import Any - - -VERSION = "ene_meta_autotype_v1" - - -def canonical_json(obj: Any) -> str: - return json.dumps(obj, sort_keys=True, separators=(",", ":"), ensure_ascii=False) - - -def sha256_text(text: str) -> str: - return hashlib.sha256(text.encode("utf-8")).hexdigest() - - -def iso_utc() -> str: - return time.strftime("%Y-%m-%dT%H:%M:%S", time.gmtime()) - - -@dataclass(frozen=True) -class ContingentField: - name: str - inferred_type: str - confidence: float - extraction_rule: str - bind_class: str - status: str = "contingent" - - -def scalar_type(value: Any) -> str: - if isinstance(value, bool): - return "boolean" - if isinstance(value, int) and not isinstance(value, bool): - return "integer" - if isinstance(value, float): - return "float" - if isinstance(value, list): - return "array" - if isinstance(value, dict): - return "object" - if value is None: - return "null" - text = str(value) - if re.fullmatch(r"[0-9a-f]{64}", text): - return "sha256_hex" - if re.fullmatch(r"-?\d+", text): - return "integer_string" - if re.fullmatch(r"-?\d+\.\d+", text): - return "float_string" - if text.startswith(("http://", "https://")): - return "url" - return "string" - - -def bind_class_for(name: str, inferred_type: str) -> str: - lower = name.lower() - if inferred_type in {"sha256_hex"} or "hash" in lower or "receipt" in lower: - return "attestation_bind" - if any(token in lower for token in ("x", "y", "z", "coord", "manifold", "topology", "graph")): - return "geometric_bind" - if any(token in lower for token in ("policy", "allow", "deny", "risk", "security")): - return "control_bind" - return "informational_bind" - - -def surface_hint(text: str, parsed: Any | None) -> str: - lower = text.lower() - if isinstance(parsed, dict): - keys = {str(key).lower() for key in parsed.keys()} - if {"nodes", "edges"} & keys or "graphml" in lower: - return "graph_concept_surface" - if {"archive_id", "source_type", "raw_content"} <= keys: - return "ene_archive_surface" - if {"pkg", "tier", "domain"} <= keys: - return "ene_package_surface" - if "= 8: - return "sequence_surface" - return "unknown_surface" - - -def autotype_payload(data: bytes, name: str = "payload") -> dict[str, Any]: - text = data.decode("utf-8", errors="replace") - parsed: Any | None = None - fields: list[ContingentField] = [] - try: - parsed = json.loads(text) - except json.JSONDecodeError: - parsed = None - - if isinstance(parsed, dict): - for key, value in sorted(parsed.items()): - inferred = scalar_type(value) - fields.append( - ContingentField( - name=str(key), - inferred_type=inferred, - confidence=0.85, - extraction_rule=f"json_pointer:/{key}", - bind_class=bind_class_for(str(key), inferred), - ) - ) - elif isinstance(parsed, list): - fields.append( - ContingentField( - name="items", - inferred_type="array", - confidence=0.8, - extraction_rule="json_root_array", - bind_class="informational_bind", - ) - ) - else: - tokens = re.findall(r"[A-Za-z_][A-Za-z0-9_]{2,}", text) - for token in sorted(set(tokens[:64]))[:16]: - fields.append( - ContingentField( - name=token, - inferred_type="token", - confidence=0.45, - extraction_rule="regex_identifier_token", - bind_class=bind_class_for(token, "string"), - ) - ) - - hint = surface_hint(text, parsed) - raw_content = { - "kind": "ene_meta_autotype", - "version": VERSION, - "name": name, - "surface_hint": hint, - "byte_len": len(data), - "contingent_fields": [asdict(field) for field in fields], - "policy": { - "defined_ingestion_surface": hint != "unknown_surface", - "authority": "contingent_until_bound_by_ingestion_surface", - "required_gate": ["OBSERVE", "BIND", "ROUTE", "POLICY_CHECK", "VERIFY", "RECEIPT"], - }, - } - content_hash = sha256_text(canonical_json(raw_content)) - receipt = sha256_text(canonical_json({"v": VERSION, "content_hash": content_hash, "name": name})) - return { - "ok": True, - "op": "meta_autotype", - "surface_hint": hint, - "field_count": len(fields), - "archive_record": { - "archive_id": f"json_catalog_ene_meta_autotype_{content_hash[:16]}", - "source_type": "json_catalog", - "source_file": f"ene-meta-autotype://{content_hash[:16]}", - "raw_content": raw_content, - "extracted_text": text[:10000], - "extracted_at": iso_utc(), - "content_hash": content_hash, - "extraction_version": VERSION, - }, - "jsonl_event": { - "src": "ene", - "op": "upsert", - "data": { - "pkg": f"ene/meta-autotype/{content_hash[:16]}", - "version": VERSION, - "tier": "RESEARCH", - "domain": "semantic", - "archetype": "contingent_schema", - "tags": ["ene", "meta_autotype", hint], - "sha256": content_hash, - }, - "bind": { - "lawful": True, - "class": "informational_bind", - "invariant": "contingent_fields_are_not_authoritative", - }, - "provenance": {"attestation_hash": f"sha256:{receipt}"}, - }, - "receipt": receipt, - } - - -def handle_request(request: dict[str, Any]) -> dict[str, Any]: - if "data_b64" in request: - data = base64.b64decode(str(request["data_b64"])) - else: - data = str(request.get("text", "")).encode("utf-8") - return autotype_payload(data, str(request.get("name", "payload"))) - - -def main() -> int: - parser = argparse.ArgumentParser(description="ENE meta-autotype shim") - parser.add_argument("--name", default="cli") - parser.add_argument("--text", default="") - parser.add_argument("--file", type=argparse.FileType("rb")) - args = parser.parse_args() - data = args.file.read() if args.file else args.text.encode("utf-8") - print(json.dumps(autotype_payload(data, args.name), indent=2, sort_keys=True)) - return 0 - - -if __name__ == "__main__": - raise SystemExit(main()) diff --git a/4-Infrastructure/infra/ene_rds_fractal_fold.py b/4-Infrastructure/infra/ene_rds_fractal_fold.py deleted file mode 100644 index 8a179425..00000000 --- a/4-Infrastructure/infra/ene_rds_fractal_fold.py +++ /dev/null @@ -1,592 +0,0 @@ -"""RDS-backed ENEFractalFold — drop-in replacement for SQLite ENEFractalFold. - -API-compatible with ene_fractal_fold.ENEFractalFold: same dataclasses, same -handle_request() protocol, but backed by PostgreSQL via psycopg2. - -Constructor: - ENERDSFractalFold(dsn="postgresql://user:pass@host:5432/dbname") - -The DSN defaults to the RDS_HOST / RDS_PORT / RDS_USER / RDS_PASSWORD / RDS_DB -environment variables. -""" - -from __future__ import annotations - -import base64 -import json -import math -import time -from dataclasses import asdict, dataclass -from pathlib import Path -from typing import Any - -import psycopg2 -import psycopg2.extras - -from infra.ene_fractal_fold import ( - FractalNode, - FractalManifest, - VERSION, - GOLDEN_ANGLE, - canonical_json, - sha256_text, - sha256_bytes, - gray_code, - inverse_gray_code, - golden_spiral_point, - manifold_distance, - tree_depth, - make_leaf, - make_parent, - encode_fractal, - encode_fractal_chunks, - parse_graphml_concepts, - node_record, - archive_record, - jsonl_event, -) - - -import os - - -def _default_dsn() -> str: - host = os.environ.get("RDS_HOST", "database-1-instance-1.cghu8yqogqwo.us-east-1.rds.amazonaws.com") - port = os.environ.get("RDS_PORT", "5432") - user = os.environ.get("RDS_USER", "postgres") - password = os.environ.get("RDS_PASSWORD") or os.environ.get("RDS_IAM_TOKEN", "") - dbname = os.environ.get("RDS_DB", "postgres") - return f"host={host} port={port} dbname={dbname} user={user} password={password} sslmode=require" - - -class ENERDSFractalFold: - def __init__(self, dsn: str | None = None): - self.dsn = dsn or _default_dsn() - self._init_db() - - def _get_conn(self): - return psycopg2.connect(self.dsn) - - def _init_db(self) -> None: - with self._get_conn() as conn: - with conn.cursor() as cur: - cur.execute("CREATE SCHEMA IF NOT EXISTS ene") - cur.execute(""" - CREATE TABLE IF NOT EXISTS ene.fractal_manifolds ( - root_hash TEXT PRIMARY KEY, - name TEXT NOT NULL, - byte_len INTEGER NOT NULL, - leaves_count INTEGER NOT NULL, - depth INTEGER NOT NULL, - chunk_size INTEGER NOT NULL, - branching_factor INTEGER NOT NULL, - created_at TIMESTAMPTZ NOT NULL DEFAULT now(), - receipt TEXT NOT NULL, - archive_record JSONB NOT NULL DEFAULT '{}', - jsonl_event JSONB NOT NULL DEFAULT '{}' - ) - """) - cur.execute(""" - CREATE TABLE IF NOT EXISTS ene.fractal_nodes ( - root_hash TEXT NOT NULL, - node_hash TEXT NOT NULL, - kind TEXT NOT NULL, - level INTEGER NOT NULL, - ordinal INTEGER NOT NULL, - fold_address INTEGER NOT NULL, - start_leaf INTEGER NOT NULL, - end_leaf INTEGER NOT NULL, - size_bytes INTEGER NOT NULL, - children TEXT NOT NULL, - payload_b64 TEXT, - PRIMARY KEY (root_hash, node_hash) - ) - """) - cur.execute(""" - CREATE INDEX IF NOT EXISTS idx_rds_fractal_leaf - ON ene.fractal_nodes (root_hash, level, ordinal) - """) - cur.execute(""" - CREATE TABLE IF NOT EXISTS ene.fractal_graph_entities ( - root_hash TEXT NOT NULL, - graph_node_id TEXT NOT NULL, - leaf_index INTEGER NOT NULL, - name TEXT NOT NULL, - family TEXT, - domain TEXT, - neighbors TEXT NOT NULL, - PRIMARY KEY (root_hash, graph_node_id) - ) - """) - cur.execute(""" - CREATE INDEX IF NOT EXISTS idx_rds_fractal_graph_name - ON ene.fractal_graph_entities (root_hash, name) - """) - conn.commit() - - def _store(self, manifest: FractalManifest, nodes: list[FractalNode]) -> tuple[dict[str, Any], dict[str, Any]]: - record = archive_record(manifest) - event = jsonl_event(record, manifest) - with self._get_conn() as conn: - with conn.cursor() as cur: - cur.execute(""" - INSERT INTO ene.fractal_manifolds - (root_hash, name, byte_len, leaves_count, depth, chunk_size, branching_factor, - created_at, receipt, archive_record, jsonl_event) - VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s) - ON CONFLICT (root_hash) DO UPDATE SET - name = EXCLUDED.name, - byte_len = EXCLUDED.byte_len, - leaves_count = EXCLUDED.leaves_count, - depth = EXCLUDED.depth, - chunk_size = EXCLUDED.chunk_size, - branching_factor = EXCLUDED.branching_factor, - receipt = EXCLUDED.receipt, - archive_record = EXCLUDED.archive_record, - jsonl_event = EXCLUDED.jsonl_event - """, ( - manifest.root_hash, manifest.name, manifest.byte_len, - manifest.leaves_count, manifest.depth, manifest.chunk_size, - manifest.branching_factor, manifest.created_at, manifest.receipt, - json.dumps(record, sort_keys=True), json.dumps(event, sort_keys=True), - )) - - cur.execute("DELETE FROM ene.fractal_nodes WHERE root_hash = %s", (manifest.root_hash,)) - psycopg2.extras.execute_values( - cur, - """ - INSERT INTO ene.fractal_nodes - (root_hash, node_hash, kind, level, ordinal, fold_address, start_leaf, - end_leaf, size_bytes, children, payload_b64) - VALUES %s - """, - [ - ( - manifest.root_hash, node.node_hash, node.kind, node.level, - node.ordinal, node.fold_address, node.start_leaf, node.end_leaf, - node.size_bytes, canonical_json(node.children), node.payload_b64, - ) - for node in nodes - ], - ) - conn.commit() - return record, event - - def put(self, data: bytes, name: str = "unnamed", chunk_size: int = 4096, branching_factor: int = 4) -> dict[str, Any]: - manifest, nodes = encode_fractal(data, name, chunk_size, branching_factor) - record, event = self._store(manifest, nodes) - return { - "ok": True, "op": "fractal_put", - "manifest": asdict(manifest), - "archive_record": record, - "jsonl_event": event, - } - - def put_graphml(self, graphml: bytes, name: str = "graphml", branching_factor: int = 4) -> dict[str, Any]: - records, chunks = parse_graphml_concepts(graphml) - max_chunk = max((len(chunk) for chunk in chunks), default=0) - manifest, nodes = encode_fractal_chunks(chunks, name, max_chunk, branching_factor) - record, event = self._store(manifest, nodes) - with self._get_conn() as conn: - with conn.cursor() as cur: - cur.execute("DELETE FROM ene.fractal_graph_entities WHERE root_hash = %s", (manifest.root_hash,)) - psycopg2.extras.execute_values( - cur, - """ - INSERT INTO ene.fractal_graph_entities - (root_hash, graph_node_id, leaf_index, name, family, domain, neighbors) - VALUES %s - """, - [ - ( - manifest.root_hash, concept["graph_node_id"], - concept["leaf_index"], concept["name"], - concept["family"], concept["domain"], - canonical_json(concept["neighbors"]), - ) - for concept in records - ], - ) - conn.commit() - return { - "ok": True, "op": "fractal_graphml_put", - "manifest": asdict(manifest), - "graphml": {"concepts": len(records), "concept_leaf_mode": True}, - "archive_record": record, - "jsonl_event": event, - } - - def manifest(self, root_hash: str) -> dict[str, Any] | None: - with self._get_conn() as conn: - with conn.cursor(cursor_factory=psycopg2.extras.RealDictCursor) as cur: - cur.execute("SELECT * FROM ene.fractal_manifolds WHERE root_hash = %s", (root_hash,)) - row = cur.fetchone() - if row is None: - return None - return { - "root_hash": row["root_hash"], - "name": row["name"], - "byte_len": row["byte_len"], - "leaves_count": row["leaves_count"], - "depth": row["depth"], - "chunk_size": row["chunk_size"], - "branching_factor": row["branching_factor"], - "created_at": row["created_at"].isoformat() if hasattr(row["created_at"], "isoformat") else str(row["created_at"]), - "receipt": row["receipt"], - "archive_record": row["archive_record"] if isinstance(row["archive_record"], dict) else json.loads(row["archive_record"] or "{}"), - "jsonl_event": row["jsonl_event"] if isinstance(row["jsonl_event"], dict) else json.loads(row["jsonl_event"] or "{}"), - } - - def _node(self, cur, root_hash: str, node_hash: str) -> dict[str, Any]: - cur.execute( - "SELECT * FROM ene.fractal_nodes WHERE root_hash = %s AND node_hash = %s", - (root_hash, node_hash), - ) - row = cur.fetchone() - if row is None: - raise KeyError(f"missing fractal node {node_hash}") - return dict(row) - - def proof(self, root_hash: str, leaf_index: int) -> dict[str, Any]: - meta = self.manifest(root_hash) - if meta is None: - raise KeyError(f"unknown root {root_hash}") - if leaf_index < 0 or leaf_index >= meta["leaves_count"]: - raise IndexError("leaf_index outside manifold") - - with self._get_conn() as conn: - with conn.cursor(cursor_factory=psycopg2.extras.RealDictCursor) as cur: - frontier = root_hash - path = [] - target_point = golden_spiral_point(gray_code(leaf_index), 0) - while True: - row = self._node(cur, root_hash, frontier) - children = json.loads(row["children"]) - node_point = golden_spiral_point(row["fold_address"], row["level"]) - entry = {k: v for k, v in dict(row).items() if k != "payload_b64"} - entry["children"] = children - entry["golden_spiral"] = node_point - entry["distance_to_target"] = round(manifold_distance(node_point, target_point), 9) - path.append(entry) - - if row["kind"] == "leaf": - payload = base64.b64decode(row["payload_b64"] or "") - break - - next_hash = None - pruned = [] - for child_hash in children: - child = self._node(cur, root_hash, child_hash) - child_point = golden_spiral_point(child["fold_address"], child["level"]) - pruned.append({ - "node_hash": child_hash, - "covers_target": child["start_leaf"] <= leaf_index <= child["end_leaf"], - "distance": round(manifold_distance(child_point, target_point), 9), - }) - if child["start_leaf"] <= leaf_index <= child["end_leaf"]: - next_hash = child_hash - path[-1]["manifold_distance_pruning"] = sorted(pruned, key=lambda item: item["distance"]) - if next_hash is None: - raise ValueError("corrupt tree: no child covers requested leaf") - frontier = next_hash - - path_valid = self._verify_path_rows(path, payload) - return { - "ok": True, "op": "fractal_proof", - "root_hash": root_hash, "leaf_index": leaf_index, - "fold_address": gray_code(leaf_index), - "inverse_fold_address": inverse_gray_code(gray_code(leaf_index)), - "golden_spiral": golden_spiral_point(gray_code(leaf_index), 0), - "traversal_cost": len(path), - "expected_complexity": f"O(log_{meta['branching_factor']}(n))", - "path_hash_verified": path_valid, - "path": path, - "payload_b64": base64.b64encode(payload).decode("ascii"), - } - - def _verify_path_rows(self, path: list[dict[str, Any]], payload: bytes) -> bool: - if not path: - return False - leaf = path[-1] - if leaf["kind"] != "leaf": - return False - expected_leaf = make_leaf(payload, leaf["ordinal"]).node_hash - if expected_leaf != leaf["node_hash"]: - return False - child_hash = expected_leaf - for row in reversed(path[:-1]): - if child_hash not in row["children"]: - return False - child_hash = row["node_hash"] - return child_hash == path[0]["node_hash"] - - def verify(self, root_hash: str) -> dict[str, Any]: - meta = self.manifest(root_hash) - if meta is None: - raise KeyError(f"unknown root {root_hash}") - errors = [] - with self._get_conn() as conn: - with conn.cursor(cursor_factory=psycopg2.extras.RealDictCursor) as cur: - cur.execute("SELECT * FROM ene.fractal_nodes WHERE root_hash = %s", (root_hash,)) - rows = cur.fetchall() - by_hash = {r["node_hash"]: r for r in rows} - for row in rows: - children = json.loads(row["children"]) - if row["kind"] == "leaf": - payload = base64.b64decode(row["payload_b64"] or "") - expected = make_leaf(payload, row["ordinal"]).node_hash - if expected != row["node_hash"]: - errors.append({"node": row["node_hash"], "error": "leaf_hash_mismatch", "expected": expected}) - continue - child_nodes = [] - for child_hash in children: - child = by_hash.get(child_hash) - if child is None: - errors.append({"node": row["node_hash"], "error": "missing_child", "child": child_hash}) - continue - child_nodes.append(FractalNode( - node_hash=child["node_hash"], kind=child["kind"], - level=child["level"], ordinal=child["ordinal"], - fold_address=child["fold_address"], - start_leaf=child["start_leaf"], end_leaf=child["end_leaf"], - size_bytes=child["size_bytes"], - children=json.loads(child["children"]), - payload_b64=child["payload_b64"], - )) - if len(child_nodes) == len(children): - expected = make_parent(child_nodes, row["level"], row["ordinal"]).node_hash - if expected != row["node_hash"]: - errors.append({"node": row["node_hash"], "error": "parent_hash_mismatch", "expected": expected}) - return { - "ok": not errors, "op": "fractal_verify", - "root_hash": root_hash, "checked_nodes": len(rows), - "errors": errors, "damage_detected": bool(errors), - } - - def neighbors(self, root_hash: str, leaf_index: int) -> dict[str, Any]: - meta = self.manifest(root_hash) - if meta is None: - raise KeyError(f"unknown root {root_hash}") - candidates = sorted(set(idx for idx in (leaf_index - 1, leaf_index, leaf_index + 1) if 0 <= idx < meta["leaves_count"])) - return { - "ok": True, "op": "fractal_neighbors", - "root_hash": root_hash, "leaf_index": leaf_index, - "fold_address": gray_code(leaf_index), - "neighbors": [ - { - "leaf_index": idx, "fold_address": gray_code(idx), - "golden_spiral": golden_spiral_point(gray_code(idx), 0), - "fold_distance": bin(gray_code(idx) ^ gray_code(leaf_index)).count("1"), - } - for idx in candidates - ], - } - - def graph_entity(self, root_hash: str, graph_node_id: str = "", name: str = "") -> dict[str, Any]: - with self._get_conn() as conn: - with conn.cursor(cursor_factory=psycopg2.extras.RealDictCursor) as cur: - if graph_node_id: - cur.execute( - "SELECT * FROM ene.fractal_graph_entities WHERE root_hash = %s AND graph_node_id = %s", - (root_hash, graph_node_id), - ) - else: - cur.execute( - "SELECT * FROM ene.fractal_graph_entities WHERE root_hash = %s AND lower(name) = lower(%s)", - (root_hash, name), - ) - row = cur.fetchone() - if row is None: - return {"ok": False, "op": "fractal_graph_entity", "error": "graph entity not found"} - proof = self.navigate(root_hash, row["leaf_index"]) - return { - "ok": True, "op": "fractal_graph_entity", - "root_hash": root_hash, - "entity": { - "graph_node_id": row["graph_node_id"], - "leaf_index": row["leaf_index"], - "name": row["name"], - "family": row["family"], - "domain": row["domain"], - "neighbors": json.loads(row["neighbors"]), - }, - "retrieval": proof, - } - - def graph_neighbors(self, root_hash: str, graph_node_id: str) -> dict[str, Any]: - entity = self.graph_entity(root_hash, graph_node_id=graph_node_id) - if not entity.get("ok"): - return entity - neighbor_ids = entity["entity"]["neighbors"] - with self._get_conn() as conn: - with conn.cursor(cursor_factory=psycopg2.extras.RealDictCursor) as cur: - rows = [] - for nid in neighbor_ids: - cur.execute( - "SELECT * FROM ene.fractal_graph_entities WHERE root_hash = %s AND graph_node_id = %s", - (root_hash, nid), - ) - row = cur.fetchone() - if row is not None: - rows.append(row) - return { - "ok": True, "op": "fractal_graph_neighbors", - "root_hash": root_hash, "graph_node_id": graph_node_id, - "neighbors": [ - { - "graph_node_id": r["graph_node_id"], - "leaf_index": r["leaf_index"], - "name": r["name"], - "family": r["family"], - "domain": r["domain"], - "fold_address": gray_code(r["leaf_index"]), - "golden_spiral": golden_spiral_point(gray_code(r["leaf_index"]), 0), - } - for r in rows - ], - } - - def navigate(self, root_hash: str, leaf_index: int) -> dict[str, Any]: - meta = self.manifest(root_hash) - if meta is None: - raise KeyError(f"unknown root {root_hash}") - if leaf_index < 0 or leaf_index >= meta["leaves_count"]: - raise IndexError("leaf_index outside manifold") - - target_point = golden_spiral_point(gray_code(leaf_index), 0) - with self._get_conn() as conn: - with conn.cursor(cursor_factory=psycopg2.extras.RealDictCursor) as cur: - frontier = root_hash - path = [] - while True: - row = self._node(cur, root_hash, frontier) - children = json.loads(row["children"]) - entry = { - "node_hash": row["node_hash"], "kind": row["kind"], - "level": row["level"], "ordinal": row["ordinal"], - "start_leaf": row["start_leaf"], "end_leaf": row["end_leaf"], - "fold_address": row["fold_address"], - "children": children, - "golden_spiral": golden_spiral_point(row["fold_address"], row["level"]), - } - path.append(entry) - - if row["kind"] == "leaf": - payload = base64.b64decode(row["payload_b64"] or "") - break - - ranked = [] - for child_hash in children: - child = self._node(cur, root_hash, child_hash) - point = golden_spiral_point(child["fold_address"], child["level"]) - ranked.append({ - "node_hash": child_hash, - "distance": manifold_distance(point, target_point), - "covers_target": child["start_leaf"] <= leaf_index <= child["end_leaf"], - }) - ranked.sort(key=lambda item: (not item["covers_target"], item["distance"])) - path[-1]["pruned_candidates"] = [ - {"node_hash": item["node_hash"], "distance": round(item["distance"], 9), "covers_target": item["covers_target"]} - for item in ranked - ] - frontier = ranked[0]["node_hash"] - - return { - "ok": True, "op": "fractal_navigate", - "root_hash": root_hash, "leaf_index": leaf_index, - "target": {"fold_address": gray_code(leaf_index), "golden_spiral": target_point}, - "retrieval_complexity": f"O(log_{meta['branching_factor']}(n))", - "navigation": "golden_spiral_manifold_distance_pruning", - "path_hash_verified": self._verify_path_rows(path, payload), - "path": path, - "payload_b64": base64.b64encode(payload).decode("ascii"), - } - - def handle_request(self, request: dict[str, Any]) -> dict[str, Any]: - op = str(request.get("op", "manifest")) - if op in {"put", "encode"}: - if "data_b64" in request: - data = base64.b64decode(str(request["data_b64"])) - else: - data = str(request.get("text", "")).encode("utf-8") - return self.put( - data=data, - name=str(request.get("name", "unnamed")), - chunk_size=int(request.get("chunk_size", 4096)), - branching_factor=int(request.get("branching_factor", 4)), - ) - if op in {"put_graphml", "graphml"}: - if "data_b64" in request: - data = base64.b64decode(str(request["data_b64"])) - else: - data = str(request.get("text", "")).encode("utf-8") - return self.put_graphml( - graphml=data, - name=str(request.get("name", "graphml")), - branching_factor=int(request.get("branching_factor", 4)), - ) - root_hash = str(request.get("root_hash", "")) - if not root_hash: - raise ValueError("root_hash is required for this operation") - if op == "manifest": - meta = self.manifest(root_hash) - return {"ok": meta is not None, "op": "fractal_manifest", "manifest": meta} - if op == "proof": - return self.proof(root_hash, int(request.get("leaf_index", 0))) - if op in {"navigate", "get"}: - return self.navigate(root_hash, int(request.get("leaf_index", 0))) - if op == "verify": - return self.verify(root_hash) - if op == "neighbors": - return self.neighbors(root_hash, int(request.get("leaf_index", 0))) - if op in {"graph_entity", "concept"}: - return self.graph_entity( - root_hash, - graph_node_id=str(request.get("graph_node_id", "")), - name=str(request.get("name", "")), - ) - if op in {"graph_neighbors", "concept_neighbors"}: - return self.graph_neighbors(root_hash, str(request.get("graph_node_id", ""))) - raise ValueError(f"unsupported fractal op {op!r}") - - -def main() -> int: - import argparse - parser = argparse.ArgumentParser(description="ENE RDS fractal fold codec") - parser.add_argument("--dsn", help="PostgreSQL DSN") - parser.add_argument("--op", default="put") - parser.add_argument("--name", default="cli") - parser.add_argument("--text", default="") - parser.add_argument("--file", type=argparse.FileType("rb")) - parser.add_argument("--graph-node-id") - parser.add_argument("--root-hash") - parser.add_argument("--leaf-index", type=int, default=0) - parser.add_argument("--chunk-size", type=int, default=4096) - parser.add_argument("--branching-factor", type=int, default=4) - args = parser.parse_args() - - layer = ENERDSFractalFold(args.dsn) - if args.op in {"put", "encode"}: - data = args.file.read() if args.file else args.text.encode("utf-8") - result = layer.put(data, args.name, args.chunk_size, args.branching_factor) - elif args.op in {"put_graphml", "graphml"}: - data = args.file.read() if args.file else args.text.encode("utf-8") - result = layer.put_graphml(data, args.name, args.branching_factor) - elif args.op in {"graph_entity", "concept", "graph_neighbors"}: - result = layer.handle_request({ - "op": args.op, "root_hash": args.root_hash, - "leaf_index": args.leaf_index, - "graph_node_id": args.graph_node_id or "", - "name": args.name, - }) - else: - result = layer.handle_request({ - "op": args.op, "root_hash": args.root_hash, - "leaf_index": args.leaf_index, - }) - print(json.dumps(result, indent=2, sort_keys=True)) - return 0 - - -if __name__ == "__main__": - raise SystemExit(main()) diff --git a/4-Infrastructure/infra/ene_rds_wiki_layer.py b/4-Infrastructure/infra/ene_rds_wiki_layer.py deleted file mode 100644 index 076abebc..00000000 --- a/4-Infrastructure/infra/ene_rds_wiki_layer.py +++ /dev/null @@ -1,438 +0,0 @@ -"""RDS-backed ENEWikiLayer — drop-in replacement for SQLite ENEWikiLayer. - -API-compatible with ene_wiki_layer.ENEWikiLayer: same dataclasses, same -handle_request() protocol, but backed by PostgreSQL via psycopg2. - -Constructor: - ENERDSWikiLayer(dsn="postgresql://user:pass@host:5432/dbname") - -The DSN defaults to the RDS_HOST / RDS_PORT / RDS_USER / RDS_PASSWORD / RDS_DB -environment variables (or postgres as dbname, research_stack as fallback). -""" - -from __future__ import annotations - -import hashlib -import json -import re -import time -from dataclasses import asdict, dataclass -from datetime import datetime, timezone -from pathlib import Path -from typing import Any - -import psycopg2 -import psycopg2.extras -import psycopg2.pool - -from infra.ene_wiki_layer import ( - WikiPage, - WikiRevision, - normalize_title, - title_slug, - extract_links, - extract_categories, - write_receipt, - canonical_json, - sha256_text, - iso_utc, - concept_vector_for_wiki, - genome_from_vector, - make_archive_record, - make_jsonl_event, -) - - -DEFAULT_DSN = None - - -def _default_dsn() -> str: - host = os.environ.get("RDS_HOST", "database-1-instance-1.cghu8yqogqwo.us-east-1.rds.amazonaws.com") - port = os.environ.get("RDS_PORT", "5432") - user = os.environ.get("RDS_USER", "postgres") - password = os.environ.get("RDS_PASSWORD") or os.environ.get("RDS_IAM_TOKEN", "") - dbname = os.environ.get("RDS_DB", "postgres") - return f"host={host} port={port} dbname={dbname} user={user} password={password} sslmode=require" - - -import os - - -class ENERDSWikiLayer: - def __init__(self, dsn: str | None = None): - self.dsn = dsn or _default_dsn() - self._pool: psycopg2.pool.ThreadedConnectionPool | None = None - self._init_tables() - - def _get_conn(self): - return psycopg2.connect(self.dsn) - - def _init_tables(self) -> None: - with self._get_conn() as conn: - with conn.cursor() as cur: - cur.execute("CREATE SCHEMA IF NOT EXISTS ene") - cur.execute(""" - CREATE TABLE IF NOT EXISTS ene.wiki_pages ( - slug TEXT PRIMARY KEY, - title TEXT NOT NULL, - latest_revision INTEGER NOT NULL, - updated_at TIMESTAMPTZ NOT NULL DEFAULT now(), - receipt TEXT NOT NULL - ) - """) - cur.execute(""" - CREATE TABLE IF NOT EXISTS ene.wiki_revisions ( - slug TEXT NOT NULL, - revision INTEGER NOT NULL, - title TEXT NOT NULL, - text TEXT NOT NULL, - author TEXT NOT NULL DEFAULT 'ene', - summary TEXT NOT NULL DEFAULT '', - created_at TIMESTAMPTZ NOT NULL DEFAULT now(), - receipt TEXT NOT NULL, - archive_id TEXT, - content_hash TEXT, - archive_record JSONB DEFAULT '{}', - jsonl_event JSONB DEFAULT '{}', - PRIMARY KEY (slug, revision) - ) - """) - cur.execute(""" - CREATE TABLE IF NOT EXISTS ene.wiki_links ( - slug TEXT NOT NULL, - target_slug TEXT NOT NULL, - target_title TEXT NOT NULL, - PRIMARY KEY (slug, target_slug) - ) - """) - cur.execute(""" - CREATE TABLE IF NOT EXISTS ene.wiki_categories ( - slug TEXT NOT NULL, - category TEXT NOT NULL, - PRIMARY KEY (slug, category) - ) - """) - cur.execute(""" - CREATE TABLE IF NOT EXISTS ene.packages ( - pkg TEXT PRIMARY KEY, - version TEXT NOT NULL, - tier TEXT, - domain TEXT, - archetype TEXT, - description TEXT, - tags JSONB DEFAULT '[]', - source TEXT, - sha256 TEXT, - indexed_utc TEXT, - concept_anchor JSONB DEFAULT '{}', - concept_vector JSONB DEFAULT '[]', - idea_weights JSONB DEFAULT '{}', - analog_map JSONB DEFAULT '{}' - ) - """) - self._ensure_columns(conn, "ene.wiki_revisions", cur) - conn.commit() - - def _ensure_columns(self, conn, table: str, cur) -> None: - cur.execute(""" - SELECT column_name FROM information_schema.columns - WHERE table_schema = 'ene' AND table_name = %s - """, (table.split('.')[1],)) - existing = {r[0] for r in cur.fetchall()} - additions = { - "archive_id": "TEXT", - "content_hash": "TEXT", - "archive_record": "JSONB DEFAULT '{}'", - "jsonl_event": "JSONB DEFAULT '{}'", - } - for name, decl in additions.items(): - if name not in existing: - cur.execute(f"ALTER TABLE {table} ADD COLUMN IF NOT EXISTS {name} {decl}") - - def _upsert_package(self, conn, cur, event: dict[str, Any]) -> None: - data = event["data"] - cur.execute(""" - INSERT INTO ene.packages ( - pkg, version, tier, domain, archetype, description, - tags, source, sha256, indexed_utc, - concept_anchor, concept_vector, idea_weights, analog_map - ) VALUES ( - %(pkg)s, %(version)s, %(tier)s, %(domain)s, %(archetype)s, %(description)s, - %(tags)s, %(source)s, %(sha256)s, %(indexed_utc)s, - %(concept_anchor)s, %(concept_vector)s, %(idea_weights)s, %(analog_map)s - ) ON CONFLICT (pkg) DO UPDATE SET - version = EXCLUDED.version, - tier = EXCLUDED.tier, - domain = EXCLUDED.domain, - archetype = EXCLUDED.archetype, - description = EXCLUDED.description, - tags = EXCLUDED.tags, - source = EXCLUDED.source, - sha256 = EXCLUDED.sha256, - indexed_utc = EXCLUDED.indexed_utc, - concept_anchor = EXCLUDED.concept_anchor, - concept_vector = EXCLUDED.concept_vector, - idea_weights = EXCLUDED.idea_weights, - analog_map = EXCLUDED.analog_map - """, { - "pkg": data["pkg"], - "version": data["version"], - "tier": data["tier"], - "domain": data["domain"], - "archetype": data["archetype"], - "description": data["description"], - "tags": json.dumps(data["tags"], sort_keys=True), - "source": data["source"], - "sha256": data["sha256"], - "indexed_utc": data["indexed_utc"], - "concept_anchor": json.dumps(data["concept_anchor"], sort_keys=True), - "concept_vector": json.dumps(data["concept_vector"], sort_keys=True), - "idea_weights": json.dumps(data["idea_weights"], sort_keys=True), - "analog_map": json.dumps(data["analog_map"], sort_keys=True), - }) - - def admit_write(self, title: str, text: str) -> tuple[bool, str]: - try: - normalize_title(title) - except ValueError as exc: - return False, str(exc) - if len(text.encode("utf-8")) > 256_000: - return False, "wiki text too large" - lowered = text.lower() - if " WikiRevision: - admitted, reason = self.admit_write(title, text) - if not admitted: - raise ValueError(reason) - - normalized = normalize_title(title) - slug = title_slug(normalized) - now_ts = int(time.time()) - now_dt = datetime.fromtimestamp(now_ts, tz=timezone.utc) - links = extract_links(text) - categories = extract_categories(text) - - with self._get_conn() as conn: - with conn.cursor() as cur: - cur.execute( - "SELECT latest_revision FROM ene.wiki_pages WHERE slug = %s", - (slug,), - ) - row = cur.fetchone() - revision = int(row[0]) + 1 if row else 1 - receipt = write_receipt(slug, revision, text, author, now_ts) - archive_record = make_archive_record( - normalized, slug, revision, text, author, summary, now_ts, links, categories - ) - concept_vector = concept_vector_for_wiki(normalized, text, links, categories) - jsonl_event = make_jsonl_event(archive_record, concept_vector, receipt) - - cur.execute(""" - INSERT INTO ene.wiki_revisions - (slug, revision, title, text, author, summary, created_at, receipt, - archive_id, content_hash, archive_record, jsonl_event) - VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s) - """, ( - slug, revision, normalized, text, author, summary, now_dt, receipt, - archive_record["archive_id"], archive_record["content_hash"], - json.dumps(archive_record, sort_keys=True), - json.dumps(jsonl_event, sort_keys=True), - )) - - cur.execute(""" - INSERT INTO ene.wiki_pages (slug, title, latest_revision, updated_at, receipt) - VALUES (%s, %s, %s, %s, %s) - ON CONFLICT (slug) DO UPDATE SET - title = EXCLUDED.title, - latest_revision = EXCLUDED.latest_revision, - updated_at = EXCLUDED.updated_at, - receipt = EXCLUDED.receipt - """, (slug, normalized, revision, now_dt, receipt)) - - cur.execute("DELETE FROM ene.wiki_links WHERE slug = %s", (slug,)) - cur.execute("DELETE FROM ene.wiki_categories WHERE slug = %s", (slug,)) - - if links: - psycopg2.extras.execute_values( - cur, - "INSERT INTO ene.wiki_links (slug, target_slug, target_title) VALUES %s ON CONFLICT DO NOTHING", - [(slug, title_slug(link), link) for link in links], - ) - if categories: - psycopg2.extras.execute_values( - cur, - "INSERT INTO ene.wiki_categories (slug, category) VALUES %s ON CONFLICT DO NOTHING", - [(slug, cat) for cat in categories], - ) - - self._upsert_package(conn, cur, jsonl_event) - conn.commit() - - return WikiRevision( - normalized, slug, revision, text, author, summary, now_ts, receipt, - links, categories, archive_record, jsonl_event, - ) - - def get_page(self, title: str, revision: int | None = None) -> WikiRevision | None: - slug = title_slug(title) - with self._get_conn() as conn: - with conn.cursor(cursor_factory=psycopg2.extras.RealDictCursor) as cur: - if revision is None: - cur.execute( - "SELECT latest_revision FROM ene.wiki_pages WHERE slug = %s", - (slug,), - ) - page = cur.fetchone() - if not page: - return None - revision = int(page["latest_revision"]) - else: - revision = int(revision) - - cur.execute(""" - SELECT title, slug, revision, text, author, summary, created_at, receipt, - archive_record, jsonl_event - FROM ene.wiki_revisions - WHERE slug = %s AND revision = %s - """, (slug, revision)) - row = cur.fetchone() - if not row: - return None - - cur.execute( - "SELECT target_title FROM ene.wiki_links WHERE slug = %s ORDER BY lower(target_title)", - (slug,), - ) - links = [r["target_title"] for r in cur.fetchall()] - - cur.execute( - "SELECT category FROM ene.wiki_categories WHERE slug = %s ORDER BY lower(category)", - (slug,), - ) - categories = [r["category"] for r in cur.fetchall()] - - archive_record = row["archive_record"] if isinstance(row["archive_record"], dict) else json.loads(row["archive_record"] or "{}") - jsonl_event = row["jsonl_event"] if isinstance(row["jsonl_event"], dict) else json.loads(row["jsonl_event"] or "{}") - created_ts = int(row["created_at"].timestamp()) if hasattr(row["created_at"], "timestamp") else 0 - - return WikiRevision( - row["title"], row["slug"], int(row["revision"]), row["text"], - row["author"], row["summary"], created_ts, row["receipt"], - links, categories, archive_record, jsonl_event, - ) - - def search(self, query: str, limit: int = 20) -> list[WikiPage]: - term = f"%{query.strip()}%" - limit = max(1, min(limit, 100)) - with self._get_conn() as conn: - with conn.cursor(cursor_factory=psycopg2.extras.RealDictCursor) as cur: - cur.execute(""" - SELECT p.title, p.slug, p.latest_revision, p.updated_at, p.receipt - FROM ene.wiki_pages p - JOIN ene.wiki_revisions r - ON r.slug = p.slug AND r.revision = p.latest_revision - WHERE p.title ILIKE %s OR r.text ILIKE %s - ORDER BY p.updated_at DESC - LIMIT %s - """, (term, term, limit)) - rows = cur.fetchall() - return [ - WikiPage( - r["title"], r["slug"], int(r["latest_revision"]), - int(r["updated_at"].timestamp()) if hasattr(r["updated_at"], "timestamp") else 0, - r["receipt"], - ) - for r in rows - ] - - def backlinks(self, title: str) -> list[WikiPage]: - target = title_slug(title) - with self._get_conn() as conn: - with conn.cursor(cursor_factory=psycopg2.extras.RealDictCursor) as cur: - cur.execute(""" - SELECT p.title, p.slug, p.latest_revision, p.updated_at, p.receipt - FROM ene.wiki_links l - JOIN ene.wiki_pages p ON p.slug = l.slug - WHERE l.target_slug = %s - ORDER BY lower(p.title) - """, (target,)) - rows = cur.fetchall() - return [ - WikiPage( - r["title"], r["slug"], int(r["latest_revision"]), - int(r["updated_at"].timestamp()) if hasattr(r["updated_at"], "timestamp") else 0, - r["receipt"], - ) - for r in rows - ] - - def recent_changes(self, limit: int = 20) -> list[WikiPage]: - limit = max(1, min(limit, 100)) - with self._get_conn() as conn: - with conn.cursor(cursor_factory=psycopg2.extras.RealDictCursor) as cur: - cur.execute(""" - SELECT title, slug, latest_revision, updated_at, receipt - FROM ene.wiki_pages - ORDER BY updated_at DESC - LIMIT %s - """, (limit,)) - rows = cur.fetchall() - return [ - WikiPage( - r["title"], r["slug"], int(r["latest_revision"]), - int(r["updated_at"].timestamp()) if hasattr(r["updated_at"], "timestamp") else 0, - r["receipt"], - ) - for r in rows - ] - - def handle_request(self, request: dict[str, Any]) -> dict[str, Any]: - op = str(request.get("op", "recent")) - if op in {"put", "edit"}: - revision = self.put_page( - title=str(request.get("title", "")), - text=str(request.get("text", "")), - author=str(request.get("author", "ene")), - summary=str(request.get("summary", "")), - ) - return {"ok": True, "op": op, "revision": asdict(revision)} - if op == "get": - page = self.get_page(str(request.get("title", "")), request.get("revision")) - return {"ok": page is not None, "op": op, "page": asdict(page) if page else None} - if op == "search": - pages = self.search(str(request.get("query", "")), int(request.get("limit", 20))) - return {"ok": True, "op": op, "pages": [asdict(page) for page in pages]} - if op == "backlinks": - pages = self.backlinks(str(request.get("title", ""))) - return {"ok": True, "op": op, "pages": [asdict(page) for page in pages]} - if op == "recent": - pages = self.recent_changes(int(request.get("limit", 20))) - return {"ok": True, "op": op, "pages": [asdict(page) for page in pages]} - raise ValueError(f"unsupported wiki op {op!r}") - - -def main() -> int: - import argparse - parser = argparse.ArgumentParser(description="ENE RDS wiki layer") - parser.add_argument("--dsn", help="PostgreSQL DSN") - parser.add_argument("--op", choices=["put", "get", "search", "backlinks", "recent"], default="recent") - parser.add_argument("--title", default="") - parser.add_argument("--text", default="") - parser.add_argument("--query", default="") - parser.add_argument("--author", default="ene") - args = parser.parse_args() - - wiki = ENERDSWikiLayer(args.dsn) - request: dict[str, Any] = { - "op": args.op, "title": args.title, "text": args.text, - "query": args.query, "author": args.author, - } - print(json.dumps(wiki.handle_request(request), sort_keys=True, indent=2)) - return 0 - - -if __name__ == "__main__": - raise SystemExit(main()) diff --git a/4-Infrastructure/infra/ene_wiki_layer.py b/4-Infrastructure/infra/ene_wiki_layer.py deleted file mode 100644 index 52623f5f..00000000 --- a/4-Infrastructure/infra/ene_wiki_layer.py +++ /dev/null @@ -1,607 +0,0 @@ -#!/usr/bin/env python3 -"""ENE MediaWiki-like layer. - -This is a compact wiki surface for ENE rather than a full MediaWiki install. -It provides page revisions, wiki links, categories, recent changes, backlinks, -and deterministic write receipts using only SQLite and the Python stdlib. -""" - -from __future__ import annotations - -import hashlib -import json -import re -import sqlite3 -import time -from dataclasses import asdict, dataclass -from datetime import datetime, timezone -from pathlib import Path -from typing import Any - - -DEFAULT_DB_PATH = Path("/home/allaun/Documents/Research Stack/data/substrate_index.db") -LINK_RE = re.compile(r"\[\[([^\[\]\n|#]+)(?:[#|][^\[\]\n]*)?\]\]") -CATEGORY_RE = re.compile(r"\[\[\s*Category\s*:\s*([^\[\]\n|#]+)(?:[#|][^\[\]\n]*)?\]\]", re.I) - - -@dataclass(frozen=True) -class WikiPage: - title: str - slug: str - latest_revision: int - updated_at: int - receipt: str - - -@dataclass(frozen=True) -class WikiRevision: - title: str - slug: str - revision: int - text: str - author: str - summary: str - created_at: int - receipt: str - links: list[str] - categories: list[str] - archive_record: dict[str, Any] - jsonl_event: dict[str, Any] - - -def normalize_title(title: str) -> str: - cleaned = " ".join(title.strip().split()) - if not cleaned: - raise ValueError("wiki title is required") - if len(cleaned) > 160: - raise ValueError("wiki title too long") - return cleaned - - -def title_slug(title: str) -> str: - normalized = normalize_title(title) - slug = normalized.lower() - slug = re.sub(r"[^a-z0-9._ -]+", "", slug) - slug = re.sub(r"\s+", "_", slug).strip("_") - if not slug: - slug = hashlib.sha256(normalized.encode("utf-8")).hexdigest()[:16] - return slug - - -def extract_links(text: str) -> list[str]: - links = [] - for match in LINK_RE.finditer(text): - target = normalize_title(match.group(1)) - if not target.lower().startswith("category:"): - links.append(target) - return sorted(set(links), key=str.lower) - - -def extract_categories(text: str) -> list[str]: - categories = [normalize_title(match.group(1)) for match in CATEGORY_RE.finditer(text)] - return sorted(set(categories), key=str.lower) - - -def write_receipt(slug: str, revision: int, text: str, author: str, created_at: int) -> str: - payload = json.dumps( - { - "author": author, - "created_at": created_at, - "revision": revision, - "slug": slug, - "text_sha256": hashlib.sha256(text.encode("utf-8")).hexdigest(), - }, - sort_keys=True, - separators=(",", ":"), - ) - return hashlib.sha256(payload.encode("utf-8")).hexdigest() - - -def canonical_json(obj: Any) -> str: - return json.dumps(obj, sort_keys=True, separators=(",", ":"), ensure_ascii=False) - - -def sha256_text(text: str) -> str: - return hashlib.sha256(text.encode("utf-8")).hexdigest() - - -def iso_utc(ts: int | None = None) -> str: - dt = datetime.fromtimestamp(ts or int(time.time()), tz=timezone.utc) - return dt.replace(microsecond=0).isoformat().replace("+00:00", "Z") - - -def concept_vector_for_wiki(title: str, text: str, links: list[str], categories: list[str]) -> list[float]: - """Derive a deterministic ConceptVector14-ish vector for wiki pages. - - This follows the ENE schema's fixed 14-axis shape without pretending to be - a learned embedding. It is a bounded route/index vector. - """ - lowered = f"{title}\n{text}".lower() - axes = [0.0] * 14 - axes[2] = min(1.0, (lowered.count("topology") + lowered.count("manifold") + len(links)) / 12.0) - axes[5] = min(1.0, (lowered.count("hash") + lowered.count("receipt") + lowered.count("verify")) / 8.0) - axes[6] = min(1.0, (lowered.count("sqlite") + lowered.count("schema") + lowered.count("index")) / 8.0) - axes[7] = min(1.0, (len(set(re.findall(r"[a-zA-Z][a-zA-Z0-9_]{2,}", lowered))) / 500.0)) - axes[11] = min(1.0, (lowered.count("proof") + lowered.count("lean") + lowered.count("theorem")) / 8.0) - axes[12] = min(1.0, (len(categories) + lowered.count("archive") + lowered.count("history")) / 8.0) - axes[13] = min(1.0, (lowered.count("author") + lowered.count("provenance") + lowered.count("attest")) / 8.0) - if not any(axes): - axes[7] = 1.0 - norm = sum(x * x for x in axes) ** 0.5 - return [round(x / norm, 6) if norm else 0.0 for x in axes] - - -def genome_from_vector(vector: list[float]) -> dict[str, int]: - bins = [min(7, max(0, int(v * 8))) for v in vector] - return { - "mu": bins[1] if len(bins) > 1 else 0, - "rho": bins[7] if len(bins) > 7 else 0, - "c": bins[6] if len(bins) > 6 else 0, - "m": bins[2] if len(bins) > 2 else 0, - "ne": bins[12] if len(bins) > 12 else 0, - "sig": bins[13] if len(bins) > 13 else 0, - } - - -def archive_id_for(slug: str, content_hash: str) -> str: - return f"json_catalog_ene_wiki_{slug}_{content_hash[:16]}" - - -def make_archive_record( - title: str, - slug: str, - revision: int, - text: str, - author: str, - summary: str, - created_at: int, - links: list[str], - categories: list[str], -) -> dict[str, Any]: - raw_content = { - "kind": "ene_wiki_page", - "title": title, - "slug": slug, - "revision": revision, - "text": text, - "author": author, - "summary": summary, - "links": links, - "categories": categories, - } - content_hash = sha256_text(canonical_json(raw_content)) - return { - "archive_id": archive_id_for(slug, content_hash), - "source_type": "json_catalog", - "source_file": f"ene-wiki://{slug}/{revision}", - "raw_content": raw_content, - "extracted_text": text[:10_000], - "extracted_at": iso_utc(created_at), - "content_hash": content_hash, - "extraction_version": "ene_wiki_layer_v1", - } - - -def make_jsonl_event(record: dict[str, Any], concept_vector: list[float], receipt: str) -> dict[str, Any]: - now = time.time() - pkg = f"ene/wiki/{record['raw_content']['slug']}" - data = { - "pkg": pkg, - "version": record["extracted_at"], - "tier": "RESEARCH", - "domain": "semantic", - "archetype": "wiki_page", - "description": record["raw_content"]["title"], - "tags": ["ene", "wiki", *record["raw_content"]["categories"]], - "source": "ene_wiki_layer", - "sha256": record["content_hash"], - "indexed_utc": record["extracted_at"], - "concept_anchor": { - "domain": "semantic", - "concept": record["raw_content"]["slug"], - "resolution": "FORMING", - }, - "concept_vector": concept_vector, - "idea_weights": {"wiki_link_count": min(1.0, len(record["raw_content"]["links"]) / 16.0)}, - "analog_map": {"mediawiki": "ene_wiki_layer", "archive_id": record["archive_id"]}, - } - event_id = f"ene:{record['archive_id']}" - event = { - "t": now, - "src": "ene", - "id": event_id, - "op": "upsert", - "data": data, - "genome": genome_from_vector(concept_vector), - "bind": { - "lawful": True, - "cost": max(1, len(record["extracted_text"].encode("utf-8"))) << 16, - "invariant": "ene_wiki_revision_is_append_only", - "class": "informational_bind", - }, - "provenance": { - "node": "ene-wiki-layer", - "lake_seed": "local", - "tailscale_ip": "", - "attestation_hash": f"sha256:{receipt}", - "prev_id": None, - }, - } - return event - - -class ENEWikiLayer: - """Small revisioned wiki surface for ENE.""" - - def __init__(self, db_path: str | Path = DEFAULT_DB_PATH): - self.db_path = Path(db_path) - self.db_path.parent.mkdir(parents=True, exist_ok=True) - self._init_tables() - - def _connect(self) -> sqlite3.Connection: - conn = sqlite3.connect(self.db_path) - conn.row_factory = sqlite3.Row - return conn - - def _init_tables(self) -> None: - with self._connect() as conn: - conn.executescript( - """ - CREATE TABLE IF NOT EXISTS ene_wiki_pages ( - slug TEXT PRIMARY KEY, - title TEXT NOT NULL, - latest_revision INTEGER NOT NULL, - updated_at INTEGER NOT NULL, - receipt TEXT NOT NULL - ); - - CREATE TABLE IF NOT EXISTS ene_wiki_revisions ( - slug TEXT NOT NULL, - revision INTEGER NOT NULL, - title TEXT NOT NULL, - text TEXT NOT NULL, - author TEXT NOT NULL, - summary TEXT NOT NULL, - created_at INTEGER NOT NULL, - receipt TEXT NOT NULL, - archive_id TEXT, - content_hash TEXT, - archive_record TEXT, - jsonl_event TEXT, - PRIMARY KEY (slug, revision) - ); - - CREATE TABLE IF NOT EXISTS ene_wiki_links ( - slug TEXT NOT NULL, - target_slug TEXT NOT NULL, - target_title TEXT NOT NULL, - PRIMARY KEY (slug, target_slug) - ); - - CREATE TABLE IF NOT EXISTS ene_wiki_categories ( - slug TEXT NOT NULL, - category TEXT NOT NULL, - PRIMARY KEY (slug, category) - ); - """ - ) - self._ensure_columns( - conn, - "ene_wiki_revisions", - { - "archive_id": "TEXT", - "content_hash": "TEXT", - "archive_record": "TEXT", - "jsonl_event": "TEXT", - }, - ) - self._ensure_packages_table(conn) - - def _ensure_columns(self, conn: sqlite3.Connection, table: str, columns: dict[str, str]) -> None: - existing = {row["name"] for row in conn.execute(f"PRAGMA table_info({table})")} - for name, decl in columns.items(): - if name not in existing: - conn.execute(f"ALTER TABLE {table} ADD COLUMN {name} {decl}") - - def _ensure_packages_table(self, conn: sqlite3.Connection) -> None: - conn.execute( - """ - CREATE TABLE IF NOT EXISTS packages ( - pkg TEXT PRIMARY KEY, - version TEXT NOT NULL, - tier TEXT, - domain TEXT, - archetype TEXT, - description TEXT, - tags TEXT, - source TEXT, - sha256 TEXT, - indexed_utc TEXT, - concept_anchor TEXT, - concept_vector TEXT, - idea_weights TEXT, - analog_map TEXT - ) - """ - ) - - def _upsert_package(self, conn: sqlite3.Connection, event: dict[str, Any]) -> None: - data = event["data"] - columns = {row["name"] for row in conn.execute("PRAGMA table_info(packages)")} - candidate = { - "pkg": data["pkg"], - "version": data["version"], - "tier": data["tier"], - "domain": data["domain"], - "archetype": data["archetype"], - "description": data["description"], - "tags": json.dumps(data["tags"], sort_keys=True), - "source": data["source"], - "sha256": data["sha256"], - "indexed_utc": data["indexed_utc"], - "concept_anchor": json.dumps(data["concept_anchor"], sort_keys=True), - "concept_vector": json.dumps(data["concept_vector"], sort_keys=True), - "idea_weights": json.dumps(data["idea_weights"], sort_keys=True), - "analog_map": json.dumps(data["analog_map"], sort_keys=True), - } - record = {key: value for key, value in candidate.items() if key in columns} - keys = list(record) - placeholders = ", ".join(["?"] * len(keys)) - assignments = ", ".join(f"{key}=excluded.{key}" for key in keys if key != "pkg") - conn.execute( - f""" - INSERT INTO packages ({", ".join(keys)}) - VALUES ({placeholders}) - ON CONFLICT(pkg) DO UPDATE SET {assignments} - """, - [record[key] for key in keys], - ) - - def admit_write(self, title: str, text: str) -> tuple[bool, str]: - try: - normalize_title(title) - except ValueError as exc: - return False, str(exc) - if len(text.encode("utf-8")) > 256_000: - return False, "wiki text too large" - lowered = text.lower() - if " WikiRevision: - admitted, reason = self.admit_write(title, text) - if not admitted: - raise ValueError(reason) - - normalized = normalize_title(title) - slug = title_slug(normalized) - now = int(time.time()) - links = extract_links(text) - categories = extract_categories(text) - - with self._connect() as conn: - row = conn.execute( - "SELECT latest_revision FROM ene_wiki_pages WHERE slug = ?", - (slug,), - ).fetchone() - revision = int(row["latest_revision"]) + 1 if row else 1 - receipt = write_receipt(slug, revision, text, author, now) - archive_record = make_archive_record(normalized, slug, revision, text, author, summary, now, links, categories) - concept_vector = concept_vector_for_wiki(normalized, text, links, categories) - jsonl_event = make_jsonl_event(archive_record, concept_vector, receipt) - - conn.execute( - """ - INSERT INTO ene_wiki_revisions - (slug, revision, title, text, author, summary, created_at, receipt, - archive_id, content_hash, archive_record, jsonl_event) - VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) - """, - ( - slug, - revision, - normalized, - text, - author, - summary, - now, - receipt, - archive_record["archive_id"], - archive_record["content_hash"], - canonical_json(archive_record), - canonical_json(jsonl_event), - ), - ) - conn.execute( - """ - INSERT INTO ene_wiki_pages (slug, title, latest_revision, updated_at, receipt) - VALUES (?, ?, ?, ?, ?) - ON CONFLICT(slug) DO UPDATE SET - title = excluded.title, - latest_revision = excluded.latest_revision, - updated_at = excluded.updated_at, - receipt = excluded.receipt - """, - (slug, normalized, revision, now, receipt), - ) - conn.execute("DELETE FROM ene_wiki_links WHERE slug = ?", (slug,)) - conn.execute("DELETE FROM ene_wiki_categories WHERE slug = ?", (slug,)) - conn.executemany( - "INSERT OR REPLACE INTO ene_wiki_links (slug, target_slug, target_title) VALUES (?, ?, ?)", - [(slug, title_slug(link), link) for link in links], - ) - conn.executemany( - "INSERT OR REPLACE INTO ene_wiki_categories (slug, category) VALUES (?, ?)", - [(slug, category) for category in categories], - ) - self._upsert_package(conn, jsonl_event) - - return WikiRevision( - normalized, - slug, - revision, - text, - author, - summary, - now, - receipt, - links, - categories, - archive_record, - jsonl_event, - ) - - def get_page(self, title: str, revision: int | None = None) -> WikiRevision | None: - slug = title_slug(title) - with self._connect() as conn: - if revision is None: - page = conn.execute("SELECT latest_revision FROM ene_wiki_pages WHERE slug = ?", (slug,)).fetchone() - if not page: - return None - revision = int(page["latest_revision"]) - else: - revision = int(revision) - row = conn.execute( - """ - SELECT title, slug, revision, text, author, summary, created_at, receipt, - archive_record, jsonl_event - FROM ene_wiki_revisions - WHERE slug = ? AND revision = ? - """, - (slug, revision), - ).fetchone() - if not row: - return None - links = [ - link["target_title"] - for link in conn.execute( - "SELECT target_title FROM ene_wiki_links WHERE slug = ? ORDER BY lower(target_title)", - (slug,), - ) - ] - categories = [ - cat["category"] - for cat in conn.execute( - "SELECT category FROM ene_wiki_categories WHERE slug = ? ORDER BY lower(category)", - (slug,), - ) - ] - archive_record = json.loads(row["archive_record"]) if row["archive_record"] else {} - jsonl_event = json.loads(row["jsonl_event"]) if row["jsonl_event"] else {} - return WikiRevision( - row["title"], - row["slug"], - int(row["revision"]), - row["text"], - row["author"], - row["summary"], - int(row["created_at"]), - row["receipt"], - links, - categories, - archive_record, - jsonl_event, - ) - - def search(self, query: str, limit: int = 20) -> list[WikiPage]: - term = f"%{query.strip()}%" - limit = max(1, min(limit, 100)) - with self._connect() as conn: - rows = conn.execute( - """ - SELECT p.title, p.slug, p.latest_revision, p.updated_at, p.receipt - FROM ene_wiki_pages p - JOIN ene_wiki_revisions r - ON r.slug = p.slug AND r.revision = p.latest_revision - WHERE p.title LIKE ? OR r.text LIKE ? - ORDER BY p.updated_at DESC - LIMIT ? - """, - (term, term, limit), - ).fetchall() - return [ - WikiPage(row["title"], row["slug"], int(row["latest_revision"]), int(row["updated_at"]), row["receipt"]) - for row in rows - ] - - def backlinks(self, title: str) -> list[WikiPage]: - target = title_slug(title) - with self._connect() as conn: - rows = conn.execute( - """ - SELECT p.title, p.slug, p.latest_revision, p.updated_at, p.receipt - FROM ene_wiki_links l - JOIN ene_wiki_pages p ON p.slug = l.slug - WHERE l.target_slug = ? - ORDER BY lower(p.title) - """, - (target,), - ).fetchall() - return [ - WikiPage(row["title"], row["slug"], int(row["latest_revision"]), int(row["updated_at"]), row["receipt"]) - for row in rows - ] - - def recent_changes(self, limit: int = 20) -> list[WikiPage]: - limit = max(1, min(limit, 100)) - with self._connect() as conn: - rows = conn.execute( - """ - SELECT title, slug, latest_revision, updated_at, receipt - FROM ene_wiki_pages - ORDER BY updated_at DESC - LIMIT ? - """, - (limit,), - ).fetchall() - return [ - WikiPage(row["title"], row["slug"], int(row["latest_revision"]), int(row["updated_at"]), row["receipt"]) - for row in rows - ] - - def handle_request(self, request: dict[str, Any]) -> dict[str, Any]: - op = str(request.get("op", "recent")) - if op in {"put", "edit"}: - revision = self.put_page( - title=str(request.get("title", "")), - text=str(request.get("text", "")), - author=str(request.get("author", "ene")), - summary=str(request.get("summary", "")), - ) - return {"ok": True, "op": op, "revision": asdict(revision)} - if op == "get": - page = self.get_page(str(request.get("title", "")), request.get("revision")) - return {"ok": page is not None, "op": op, "page": asdict(page) if page else None} - if op == "search": - pages = self.search(str(request.get("query", "")), int(request.get("limit", 20))) - return {"ok": True, "op": op, "pages": [asdict(page) for page in pages]} - if op == "backlinks": - pages = self.backlinks(str(request.get("title", ""))) - return {"ok": True, "op": op, "pages": [asdict(page) for page in pages]} - if op == "recent": - pages = self.recent_changes(int(request.get("limit", 20))) - return {"ok": True, "op": op, "pages": [asdict(page) for page in pages]} - raise ValueError(f"unsupported wiki op {op!r}") - - -def main() -> int: - import argparse - - parser = argparse.ArgumentParser(description="ENE wiki layer") - parser.add_argument("--db", default=str(DEFAULT_DB_PATH)) - parser.add_argument("--op", choices=["put", "get", "search", "backlinks", "recent"], default="recent") - parser.add_argument("--title", default="") - parser.add_argument("--text", default="") - parser.add_argument("--query", default="") - parser.add_argument("--author", default="ene") - args = parser.parse_args() - - wiki = ENEWikiLayer(args.db) - request: dict[str, Any] = {"op": args.op, "title": args.title, "text": args.text, "query": args.query, "author": args.author} - print(json.dumps(wiki.handle_request(request), sort_keys=True, indent=2)) - return 0 - - -if __name__ == "__main__": - raise SystemExit(main()) diff --git a/4-Infrastructure/infra/enhanced_integrated_swarm.py b/4-Infrastructure/infra/enhanced_integrated_swarm.py deleted file mode 100644 index 3ada5e0b..00000000 --- a/4-Infrastructure/infra/enhanced_integrated_swarm.py +++ /dev/null @@ -1,109 +0,0 @@ -import sys -import os -import json -import logging - -# Ensure project root is in path -sys.path.append(os.path.abspath(os.path.join(os.path.dirname(__file__), '..'))) - -from infra.lean_unified_shim import LeanUnifiedShim - -logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s') -logger = logging.getLogger("EnhancedIntegratedSwarm") - -class EnhancedIntegratedSwarm: - """ - Enhanced Integrated Swarm for deep codebase analysis. - Proper shim: calls Lean bindserver for all logic, no Python logic. - """ - - def __init__(self, lean_path="0-Core-Formalism/lean/Semantics"): - self.shim = LeanUnifiedShim(lean_path) - - def perform_deep_analysis(self) -> Dict[str, Any]: - """ - Perform deep codebase analysis via Lean bindserver. - All logic is in Lean, this is just a shim for JSON serialization/subprocess. - """ - logger.info("Enhanced Integrated Swarm: Initiating deep codebase analysis via Lean bindserver...") - - result = self.shim.swarm_manifold_analysis() - - if "error" in result: - logger.error(f"Swarm analysis failed: {result['error']}") - return {"error": result['error']} - - logger.info("Swarm analysis complete. Parsing results from Lean...") - - try: - # Parse the JSON result from Lean - if isinstance(result, str): - analysis = json.loads(result) - else: - analysis = result - - logger.info(f"Analysis complete: {len(analysis.get('domains', []))} domains, {len(analysis.get('subdomains', []))} subdomains, {len(analysis.get('tensor_types', []))} tensor types, manifold with {len(analysis.get('manifold', {}).get('nodes', []))} nodes") - return analysis - except json.JSONDecodeError as e: - logger.error(f"Failed to parse swarm analysis result: {e}") - return {"error": f"JSON parse error: {e}"} - - def print_analysis(self, analysis: Dict[str, Any]): - """Print formatted analysis results.""" - if "error" in analysis: - print(f"ERROR: {analysis['error']}") - return - - print("\n" + "=" * 60) - print("ENHANCED INTEGRATED SWARM - DEEP CODEBASE ANALYSIS") - print("=" * 60) - - metadata = analysis.get("metadata", {}) - print(f"\nTotal Domains: {metadata.get('total_domains', 'N/A')}") - print(f"Total Subdomains: {metadata.get('total_subdomains', 'N/A')}") - print(f"Total Tensor Types: {metadata.get('total_tensor_types', 'N/A')}") - print(f"Manifold Nodes: {metadata.get('manifold_nodes', 'N/A')}") - print(f"Manifold Edges: {metadata.get('manifold_edges', 'N/A')}") - print(f"Manifold Dimension: {metadata.get('manifold_dimension', 'N/A')}") - print(f"Analysis Timestamp: {metadata.get('analysis_timestamp', 'N/A')}") - - print("\n" + "-" * 60) - print("DOMAINS") - print("-" * 60) - for domain in analysis.get("domains", []): - dim = domain.get("dimensionality", "N/A") - print(f" {domain['name']}: {dim}-dim") - print(f" {domain['description']}") - - print("\n" + "-" * 60) - print("SUBDOMAINS") - print("-" * 60) - for subdomain in analysis.get("subdomains", []): - print(f" {subdomain['name']}:") - categories = subdomain.get("categories", []) - for cat in categories: - print(f" - {cat}") - - print("\n" + "-" * 60) - print("TENSOR TYPES") - print("-" * 60) - for tensor in analysis.get("tensor_types", []): - print(f" {tensor['name']}: {tensor['description']}") - - print("\n" + "-" * 60) - print("MANIFOLD STRUCTURE") - print("-" * 60) - manifold = analysis.get("manifold", {}) - topology = manifold.get("topology", {}) - print(f" Dimension: {topology.get('dimension', 'N/A')}") - print(f" Nodes: {len(manifold.get('nodes', []))}") - print(f" Edges: {len(manifold.get('edges', []))}") - print(f" Connected Components: {topology.get('connectedComponents', 'N/A')}") - print(f" Euler Characteristic: {topology.get('eulerCharacteristic', 'N/A')}") - - print("\n" + "=" * 60) - -if __name__ == "__main__": - swarm = EnhancedIntegratedSwarm() - analysis = swarm.perform_deep_analysis() - swarm.print_analysis(analysis) diff --git a/4-Infrastructure/infra/gemma_4_integration.py b/4-Infrastructure/infra/gemma_4_integration.py deleted file mode 100644 index 304548f3..00000000 --- a/4-Infrastructure/infra/gemma_4_integration.py +++ /dev/null @@ -1,397 +0,0 @@ -#!/usr/bin/env python3 -""" -Gemma 4 Integration with Topological State Machine - -Integrates Gemma 4 (low-load, flexible multimodal model) with the -GPU duty assignment system and omnidirectional interface. - -Gemma 4 Variants: -- E2B: 2B effective parameters (smallest, supports audio) -- E4B: 4B effective parameters (supports audio) - RECOMMENDED for low load -- 31B: 31B parameters (dense model) -- 26B-A4B: 26B total, 4B active (MoE model, runs like 4B) -""" - -import sys -import json -import sqlite3 -from pathlib import Path -from typing import Dict, List, Optional, Any -from dataclasses import dataclass -from enum import Enum -import hashlib - -sys.path.insert(0, str(Path(__file__).parent.parent)) - -from infra.ene_api import ENEAPIHook, AccessLevel -from infra.lean_unified_shim import GPUDutyAssignmentSystem, DutyType - - -class GemmaVariant(Enum): - """Gemma 4 model variants""" - E2B = "google/gemma-4-E2B-it" # 2B effective, audio support - E4B = "google/gemma-4-E4B-it" # 4B effective, audio support - RECOMMENDED - E31B = "google/gemma-4-31B-it" # 31B dense - E26B_A4B = "google/gemma-4-26B-A4B-it" # 26B total, 4B active MoE - - @classmethod - def from_string(cls, value: str): - """Convert string to GemmaVariant""" - value_map = { - "E2B": cls.E2B, - "E4B": cls.E4B, - "E31B": cls.E31B, - "26B-A4B": cls.E26B_A4B - } - return value_map.get(value, cls.E4B) # Default to E4B - - -class GemmaTask(Enum): - """Types of tasks Gemma 4 can perform""" - TEXT_GENERATION = "text_generation" - MULTIMODAL_PROCESSING = "multimodal_processing" - AUDIO_TRANSCRIPTION = "audio_transcription" - IMAGE_UNDERSTANDING = "image_understanding" - REASONING = "reasoning" - CODE_GENERATION = "code_generation" - FUNCTION_CALLING = "function_calling" - - -@dataclass -class GemmaTaskRequest: - """Task request for Gemma 4""" - task_id: str - task_type: GemmaTask - variant: GemmaVariant - input_data: Dict[str, Any] - enable_thinking: bool = False - max_tokens: int = 1024 - priority: int = 5 - - -class Gemma4Integration: - """Gemma 4 integration with TSM""" - - def __init__(self, db_path: str = "/home/allaun/Documents/Research Stack/data/substrate_index.db", - default_variant: GemmaVariant = GemmaVariant.E4B): - self.db_path = db_path - self.ene_api = ENEAPIHook() - self.default_variant = default_variant - self.gpu_duty_system = GPUDutyAssignmentSystem() - self._init_gemma_tables() - - def _init_gemma_tables(self): - """Initialize Gemma 4 integration tables""" - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - # Gemma task requests table - cursor.execute(""" - CREATE TABLE IF NOT EXISTS gemma_task_requests ( - task_id TEXT PRIMARY KEY, - task_type TEXT NOT NULL, - variant TEXT NOT NULL, - input_data TEXT NOT NULL, - enable_thinking BOOLEAN DEFAULT FALSE, - max_tokens INTEGER DEFAULT 1024, - priority INTEGER DEFAULT 5, - status TEXT NOT NULL, - created_at INTEGER NOT NULL, - started_at INTEGER, - completed_at INTEGER, - result TEXT, - error TEXT - ) - """) - - # Gemma performance metrics - cursor.execute(""" - CREATE TABLE IF NOT EXISTS gemma_performance_metrics ( - variant TEXT PRIMARY KEY, - total_tasks INTEGER DEFAULT 0, - avg_latency REAL, - avg_tokens_per_second REAL, - last_updated INTEGER NOT NULL - ) - """) - - conn.commit() - conn.close() - - def submit_task(self, task: GemmaTaskRequest) -> str: - """Submit a task to Gemma 4""" - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - cursor.execute(""" - INSERT INTO gemma_task_requests - (task_id, task_type, variant, input_data, enable_thinking, max_tokens, priority, status, created_at) - VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?) - """, ( - task.task_id, - task.task_type.value, - task.variant.value, - json.dumps(task.input_data), - task.enable_thinking, - task.max_tokens, - task.priority, - "pending", - int(__import__('time').time()) - )) - - # Update performance metrics - cursor.execute(""" - INSERT INTO gemma_performance_metrics (variant, total_tasks, last_updated) - VALUES (?, 1, ?) - ON CONFLICT(variant) DO UPDATE SET - total_tasks = total_tasks + 1, - last_updated = ? - """, (task.variant.value, int(__import__('time').time()), int(__import__('time').time()))) - - conn.commit() - conn.close() - - return task.task_id - - def execute_task(self, task_id: str) -> Dict[str, Any]: - """Execute a Gemma 4 task""" - task = self._get_task(task_id) - if not task: - return {"success": False, "error": "Task not found"} - - # Mark as in progress - self._update_task_status(task_id, "in_progress") - - try: - # Assign to GPU duty system for execution - gpu_duty_id = self.gpu_duty_system.assign_duty( - DutyType.GENERAL_COMPUTE, - "gemma_4_integration", - { - "task_id": task_id, - "task_type": task["task_type"], - "variant": task["variant"], - "input_data": json.loads(task["input_data"]), - "enable_thinking": task["enable_thinking"], - "max_tokens": task["max_tokens"] - }, - priority=task["priority"] - ) - - # Execute the duty - duty_result = self.gpu_duty_system.execute_duty(gpu_duty_id) - - if duty_result.get("success"): - result = self._simulate_gemma_execution(task) - self._update_task_status(task_id, "completed", result) - return {"success": True, "result": result} - else: - self._update_task_status(task_id, "failed", error=duty_result.get("error")) - return {"success": False, "error": duty_result.get("error")} - - except Exception as e: - self._update_task_status(task_id, "failed", error=str(e)) - return {"success": False, "error": str(e)} - - def _get_task(self, task_id: str) -> Optional[Dict]: - """Get task by ID""" - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - cursor.execute(""" - SELECT task_id, task_type, variant, input_data, enable_thinking, max_tokens, priority, status, created_at - FROM gemma_task_requests - WHERE task_id = ? - """, (task_id,)) - - row = cursor.fetchone() - conn.close() - - if row: - return { - "task_id": row[0], - "task_type": row[1], - "variant": row[2], - "input_data": row[3], - "enable_thinking": row[4], - "max_tokens": row[5], - "priority": row[6], - "status": row[7], - "created_at": row[8] - } - return None - - def _update_task_status(self, task_id: str, status: str, result: Optional[Dict] = None, error: Optional[str] = None): - """Update task status""" - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - update_fields = ["status = ?"] - update_values = [status] - - if status == "in_progress": - update_fields.append("started_at = ?") - update_values.append(int(__import__('time').time())) - elif status in ["completed", "failed"]: - update_fields.append("completed_at = ?") - update_values.append(int(__import__('time').time())) - - if result: - update_fields.append("result = ?") - update_values.append(json.dumps(result)) - if error: - update_fields.append("error = ?") - update_values.append(error) - - update_values.append(task_id) - - cursor.execute(f""" - UPDATE gemma_task_requests - SET {', '.join(update_fields)} - WHERE task_id = ? - """, update_values) - - conn.commit() - conn.close() - - def _simulate_gemma_execution(self, task: Dict) -> Dict: - """Simulate Gemma 4 execution (placeholder for actual model loading)""" - task_type = task["task_type"] - input_data = json.loads(task["input_data"]) - - # Simulate execution based on task type - if task_type == "text_generation": - result = { - "generated_text": "This is a simulated response from Gemma 4", - "tokens_generated": 50, - "reasoning_enabled": task["enable_thinking"] - } - elif task_type == "multimodal_processing": - result = { - "processed_modalities": ["text", "image"], - "understanding": "Multimodal content processed successfully" - } - elif task_type == "audio_transcription": - result = { - "transcription": "Audio transcribed successfully", - "language": "en" - } - elif task_type == "reasoning": - result = { - "reasoning_steps": ["Step 1: Analyze", "Step 2: Deduce", "Step 3: Conclude"], - "final_answer": "Reasoning completed" - } - else: - result = { - "status": "completed", - "task_type": task_type - } - - return result - - def get_task_queue(self) -> List[Dict]: - """Get current Gemma task queue""" - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - cursor.execute(""" - SELECT task_id, task_type, variant, priority, status, created_at - FROM gemma_task_requests - WHERE status IN ('pending', 'in_progress') - ORDER BY priority DESC, created_at ASC - """) - - rows = cursor.fetchall() - conn.close() - - return [ - { - "task_id": row[0], - "task_type": row[1], - "variant": row[2], - "priority": row[3], - "status": row[4], - "created_at": row[5] - } - for row in rows - ] - - def get_performance_metrics(self) -> Dict[str, Any]: - """Get performance metrics by variant""" - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - cursor.execute("SELECT * FROM gemma_performance_metrics") - rows = cursor.fetchall() - conn.close() - - metrics = {} - for row in rows: - metrics[row[0]] = { - "total_tasks": row[1], - "avg_latency": row[2], - "avg_tokens_per_second": row[3], - "last_updated": row[4] - } - - return metrics - - -# Example usage -if __name__ == "__main__": - print("=" * 70) - print("GEMMA 4 INTEGRATION TEST") - print("=" * 70) - - gemma = Gemma4Integration(default_variant=GemmaVariant.E4B) - - # Test 1: Submit text generation task - print("\n[Test 1] Submitting text generation task...") - task1 = GemmaTaskRequest( - task_id="gemma_task_001", - task_type=GemmaTask.TEXT_GENERATION, - variant=GemmaVariant.E4B, - input_data={"prompt": "Explain the concept of hyperbolic manifolds"}, - enable_thinking=True, - max_tokens=512, - priority=8 - ) - - task_id = gemma.submit_task(task1) - print(f"Task submitted: {task_id}") - - # Test 2: Submit multimodal task - print("\n[Test 2] Submitting multimodal task...") - task2 = GemmaTaskRequest( - task_id="gemma_task_002", - task_type=GemmaTask.MULTIMODAL_PROCESSING, - variant=GemmaVariant.E4B, - input_data={"text": "Describe this image", "image_url": "placeholder"}, - enable_thinking=False, - max_tokens=256, - priority=6 - ) - - task_id = gemma.submit_task(task2) - print(f"Task submitted: {task_id}") - - # Test 3: Get task queue - print("\n[Test 3] Getting task queue...") - queue = gemma.get_task_queue() - print(f"Queue length: {len(queue)}") - for task in queue: - print(f" - {task['task_id']}: {task['task_type']} (priority: {task['priority']})") - - # Test 4: Execute task - print("\n[Test 4] Executing task...") - result = gemma.execute_task("gemma_task_001") - print(f"Result: {result}") - - # Test 5: Get performance metrics - print("\n[Test 5] Getting performance metrics...") - metrics = gemma.get_performance_metrics() - print(json.dumps(metrics, indent=2)) - - print("\n" + "=" * 70) - print("GEMMA 4 INTEGRATION TEST COMPLETE") - print("=" * 70) diff --git a/4-Infrastructure/infra/hyperbolic_encoding.py b/4-Infrastructure/infra/hyperbolic_encoding.py deleted file mode 100644 index c36c625f..00000000 --- a/4-Infrastructure/infra/hyperbolic_encoding.py +++ /dev/null @@ -1,364 +0,0 @@ -#!/usr/bin/env python3 -""" -Hyperbolic Manifold Coordinate Encoding (ENC-004) - -Implements Poincaré disk coordinates with Möbius transformations for -semantic vector encoding in hyperbolic space. - -Benefits: -- Improves hierarchical concept representation by ~35% -- Better semantic similarity for hierarchical relationships -- Natural tree-like structure in hyperbolic space -- Exponential growth of space with distance from origin - -Mathematical Foundation: -- Poincaré disk model of hyperbolic geometry -- Möbius transformations for distance-preserving operations -- Hyperbolic distance: d(x, y) = acosh(1 + 2||x-y||² / ((1-||x||²)(1-||y||²))) -""" - -import numpy as np -import json -from dataclasses import dataclass -from typing import List, Tuple, Optional -import math - - -@dataclass -class HyperbolicVector: - """Vector in Poincaré disk coordinates""" - coordinates: np.ndarray # 2D coordinates in Poincaré disk - dimension: int # Original dimension (for reconstruction) - metadata: dict = None - - -class HyperbolicManifoldEncoder: - """Encoder/Decoder for hyperbolic manifold coordinates""" - - def __init__(self, curvature: float = -1.0): - """ - Initialize hyperbolic encoder - - Args: - curvature: Curvature of hyperbolic space (default -1.0 for Poincaré disk) - """ - self.curvature = curvature - self.embedding_dim = 2 # Poincaré disk is 2D - - def encode_to_poincare(self, vector: np.ndarray) -> HyperbolicVector: - """ - Encode Euclidean vector to Poincaré disk coordinates - - Args: - vector: Input vector (14D semantic vector) - - Returns: - HyperbolicVector with Poincaré disk coordinates - """ - # Project high-dimensional vector to 2D using PCA-like projection - # For semantic vectors, we use a weighted projection based on dimension importance - if len(vector) != 14: - raise ValueError(f"Expected 14D vector, got {len(vector)}D") - - # Dimension weights based on swarm analysis (dominant dimensions: 13, 9, 3, 2, 4) - weights = np.array([ - 0.0019, 0.0020, 0.0024, 0.0025, # 0-3 - 0.0023, 0.0016, 0.0019, 0.0018, # 4-7 - 0.0020, 0.0025, 0.0018, 0.0022, # 8-11 - 0.0021, 0.0026 # 12-13 (13 is dominant) - ]) - - # Weighted projection to 2D - # x coordinate: weighted sum of even indices - x = np.sum(vector[::2] * weights[::2]) - # y coordinate: weighted sum of odd indices - y = np.sum(vector[1::2] * weights[1::2]) - - # Normalize to unit disk (||coord|| < 1) - coord = np.array([x, y]) - norm = np.linalg.norm(coord) - - if norm >= 0.99: # Ensure strictly inside disk - coord = coord / norm * 0.99 - - return HyperbolicVector( - coordinates=coord, - dimension=len(vector), - metadata={"norm": norm, "original_vector": vector.tolist()} - ) - - def decode_from_poincare(self, hyperbolic: HyperbolicVector) -> np.ndarray: - """ - Decode from Poincaré disk back to original vector space - - Args: - hyperbolic: HyperbolicVector with Poincaré coordinates - - Returns: - Reconstructed 14D vector - """ - if hyperbolic.metadata and "original_vector" in hyperbolic.metadata: - # If we stored the original, return it - return np.array(hyperbolic.metadata["original_vector"]) - - # Otherwise, reconstruct from 2D coordinates - # This is a lossy reconstruction - in production, would use learned decoder - coord = hyperbolic.coordinates - - # Expand back to 14D using inverse projection - weights = np.array([ - 0.0019, 0.0020, 0.0024, 0.0025, - 0.0023, 0.0016, 0.0019, 0.0018, - 0.0020, 0.0025, 0.0018, 0.0022, - 0.0021, 0.0026 - ]) - - reconstructed = np.zeros(14) - reconstructed[::2] = coord[0] * weights[::2] / np.sum(weights[::2]) - reconstructed[1::2] = coord[1] * weights[1::2] / np.sum(weights[1::2]) - - # Normalize to original range [0, 1] - reconstructed = np.clip(reconstructed, 0, 1) - - return reconstructed - - def mobius_transform(self, a: np.ndarray, z: np.ndarray) -> np.ndarray: - """ - Apply Möbius transformation to point z in Poincaré disk - - Args: - a: Transformation parameter (point in Poincaré disk) - z: Point to transform - - Returns: - Transformed point - """ - if np.linalg.norm(a) >= 1: - raise ValueError("Transformation parameter must be inside unit disk") - - # Möbius transformation formula: - # M_a(z) = ((1 + 2 + ||a||²)z + (1 + ||z||²)a) / - # (1 + 2 + ||a||² + ||z||²) - - a_norm_sq = np.dot(a, a) - z_norm_sq = np.dot(z, z) - az = np.dot(a, z) - - numerator = ((1 + 2*az + a_norm_sq) * z + (1 + z_norm_sq) * a) - denominator = (1 + 2*az + a_norm_sq + z_norm_sq) - - return numerator / denominator - - def hyperbolic_distance(self, x: np.ndarray, y: np.ndarray) -> float: - """ - Compute hyperbolic distance between two points in Poincaré disk - - Args: - x, y: Points in Poincaré disk - - Returns: - Hyperbolic distance - """ - x_norm_sq = np.dot(x, x) - y_norm_sq = np.dot(y, y) - diff_norm_sq = np.sum((x - y) ** 2) - - # Poincaré disk distance formula - numerator = 2 * diff_norm_sq - denominator = (1 - x_norm_sq) * (1 - y_norm_sq) - - # Clamp to avoid numerical issues - ratio = min(numerator / denominator, 1e10) - - return np.arccosh(1 + ratio) - - def encode_batch(self, vectors: List[np.ndarray]) -> List[HyperbolicVector]: - """Encode multiple vectors""" - return [self.encode_to_poincare(v) for v in vectors] - - def decode_batch(self, hyperbolic_vectors: List[HyperbolicVector]) -> List[np.ndarray]: - """Decode multiple vectors""" - return [self.decode_from_poincare(hv) for hv in hyperbolic_vectors] - - def hierarchical_similarity(self, parent: np.ndarray, child: np.ndarray) -> float: - """ - Compute hierarchical similarity between parent and child concepts - - In hyperbolic space, hierarchical relationships are naturally encoded - through radial distance from origin (root concepts near center) - - Args: - parent: Parent concept vector - child: Child concept vector - - Returns: - Hierarchical similarity score (0-1) - """ - parent_hyperbolic = self.encode_to_poincare(parent) - child_hyperbolic = self.encode_to_poincare(child) - - parent_dist = np.linalg.norm(parent_hyperbolic.coordinates) - child_dist = np.linalg.norm(child_hyperbolic.coordinates) - - # In hyperbolic space, parent should be closer to origin than child - if child_dist > parent_dist: - # Valid hierarchical relationship - # Similarity decreases with angular separation - angle = np.arctan2( - child_hyperbolic.coordinates[1], child_hyperbolic.coordinates[0] - ) - np.arctan2( - parent_hyperbolic.coordinates[1], parent_hyperbolic.coordinates[0] - ) - angular_similarity = np.cos(angle) - - # Combine radial and angular similarity - radial_similarity = 1 - (child_dist - parent_dist) - return 0.5 * angular_similarity + 0.5 * radial_similarity - else: - # Invalid hierarchy (child closer to origin than parent) - return 0.0 - - -class HyperbolicCache: - """Cache for hyperbolic encoded vectors""" - - def __init__(self): - self.encoder = HyperbolicManifoldEncoder() - self.cache = {} # Maps vector hash to HyperbolicVector - - def _hash_vector(self, vector: np.ndarray) -> str: - """Compute hash of vector for cache key""" - return hash(tuple(v for v in vector)) - - def get_or_encode(self, vector: np.ndarray) -> HyperbolicVector: - """Get encoded vector from cache or encode it""" - key = self._hash_vector(vector) - - if key not in self.cache: - self.cache[key] = self.encoder.encode_to_poincare(vector) - - return self.cache[key] - - def get_or_decode(self, hyperbolic: HyperbolicVector) -> np.ndarray: - """Get decoded vector from cache or decode it""" - # For decoding, we just use the encoder's decode method - # In production, could cache decodings too - return self.encoder.decode_from_poincare(hyperbolic) - - def similarity_search(self, query: np.ndarray, top_k: int = 5) -> List[Tuple[str, float]]: - """ - Find most similar vectors using hyperbolic distance - - Args: - query: Query vector - top_k: Number of results to return - - Returns: - List of (hash, similarity) tuples - """ - query_hyperbolic = self.encoder.encode_to_poincare(query) - - similarities = [] - for key, hyperbolic in self.cache.items(): - distance = self.encoder.hyperbolic_distance( - query_hyperbolic.coordinates, - hyperbolic.coordinates - ) - # Convert distance to similarity (closer = more similar) - similarity = 1 / (1 + distance) - similarities.append((key, similarity)) - - # Sort by similarity descending - similarities.sort(key=lambda x: x[1], reverse=True) - - return similarities[:top_k] - - -# Integration with omnidirectional interface -def integrate_hyperbolic_encoding(): - """Integration function to enable hyperbolic encoding in omnidirectional interface""" - - # This would be called to patch the omnidirectional interface - # For now, we provide the encoder instance - return HyperbolicManifoldEncoder() - - -# Example usage and testing -if __name__ == "__main__": - print("=" * 70) - print("HYPERBOLIC MANIFOLD COORDINATE ENCODING TEST") - print("=" * 70) - - encoder = HyperbolicManifoldEncoder() - cache = HyperbolicCache() - - # Test 1: Encode a semantic vector - print("\n[Test 1] Encoding 14D semantic vector to Poincaré disk...") - test_vector = np.array([ - 0.0019, 0.0020, 0.0024, 0.0025, - 0.0023, 0.0016, 0.0019, 0.0018, - 0.0020, 0.0025, 0.0018, 0.0022, - 0.0021, 0.0026 - ]) - - hyperbolic = encoder.encode_to_poincare(test_vector) - print(f"Original vector: {test_vector[:5]}... (14D)") - print(f"Encoded coordinates: {hyperbolic.coordinates}") - print(f"Norm from origin: {hyperbolic.metadata['norm']:.6f}") - - # Test 2: Decode back - print("\n[Test 2] Decoding from Poincaré disk...") - reconstructed = encoder.decode_from_poincare(hyperbolic) - print(f"Reconstructed vector: {reconstructed[:5]}... (14D)") - reconstruction_error = np.linalg.norm(test_vector - reconstructed) - print(f"Reconstruction error: {reconstruction_error:.6f}") - - # Test 3: Möbius transformation - print("\n[Test 3] Möbius transformation...") - a = np.array([0.3, 0.2]) - z = np.array([0.5, 0.4]) - transformed = encoder.mobius_transform(a, z) - print(f"Original point: {z}") - print(f"Transformed point: {transformed}") - print(f"Distance preserved: {encoder.hyperbolic_distance(z, transformed):.6f}") - - # Test 4: Hyperbolic distance - print("\n[Test 4] Hyperbolic distance computation...") - x = np.array([0.1, 0.1]) - y = np.array([0.2, 0.2]) - euclidean_dist = np.linalg.norm(x - y) - hyperbolic_dist = encoder.hyperbolic_distance(x, y) - print(f"Euclidean distance: {euclidean_dist:.6f}") - print(f"Hyperbolic distance: {hyperbolic_dist:.6f}") - - # Test 5: Hierarchical similarity - print("\n[Test 5] Hierarchical similarity...") - parent = np.array([0.001, 0.001, 0.001, 0.001, 0.001, 0.001, 0.001, 0.001, - 0.001, 0.001, 0.001, 0.001, 0.001, 0.001]) - child = np.array([0.003, 0.003, 0.003, 0.003, 0.003, 0.003, 0.003, 0.003, - 0.003, 0.003, 0.003, 0.003, 0.003, 0.003]) - similarity = encoder.hierarchical_similarity(parent, child) - print(f"Parent-child hierarchical similarity: {similarity:.6f}") - - # Test 6: Cache performance - print("\n[Test 6] Cache performance...") - vectors = [np.random.rand(14) * 0.004 for _ in range(100)] - - import time - start = time.time() - for v in vectors: - cache.get_or_encode(v) - encode_time = time.time() - start - - start = time.time() - for v in vectors: - cache.get_or_encode(v) # Should hit cache - cache_time = time.time() - start - - print(f"First pass (encode): {encode_time:.6f}s") - print(f"Second pass (cache): {cache_time:.6f}s") - print(f"Speedup: {encode_time / cache_time:.2f}x") - - print("\n" + "=" * 70) - print("HYPERBOLIC ENCODING ENABLED SUCCESSFULLY") - print("=" * 70) diff --git a/4-Infrastructure/infra/knowledge_ingestion.py b/4-Infrastructure/infra/knowledge_ingestion.py deleted file mode 100644 index 6d771cea..00000000 --- a/4-Infrastructure/infra/knowledge_ingestion.py +++ /dev/null @@ -1,482 +0,0 @@ -""" -Knowledge Ingestion Module for Swarm -Integrates multiple public domain knowledge sources: -- Wolfram Alpha API (computational knowledge) -- OpenMath Content Dictionaries (mathematical symbols) -- nLab wiki (research-level mathematics/physics) -""" - -import requests -import json -import re -import xml.etree.ElementTree as ET -from bs4 import BeautifulSoup -from typing import Dict, List, Optional, Any -import logging -from pathlib import Path - -logger = logging.getLogger(__name__) - - -class WolframAlphaKnowledge: - """Wolfram Alpha API integration for computational knowledge retrieval""" - - def __init__(self, api_key: str): - self.api_key = api_key - self.base_url = "https://api.wolframalpha.com/v2/query" - self.rate_limit_remaining = 2000 # Free tier: 2000 calls/month - self.cache = {} # Simple in-memory cache - - def query(self, question: str, format: str = "json") -> Optional[Dict[str, Any]]: - """ - Query Wolfram Alpha API with caching and rate limiting - - Args: - question: The question to ask - format: Response format (json, xml, html) - - Returns: - Parsed response or None if failed - """ - # Check cache first - cache_key = f"{question}_{format}" - if cache_key in self.cache: - logger.info(f"Cache hit for: {question[:50]}...") - return self.cache[cache_key] - - # Rate limit check - if self.rate_limit_remaining <= 0: - logger.warning("Wolfram Alpha API rate limit reached") - return None - - try: - params = { - "input": question, - "format": "plaintext", - "output": "JSON", - "appid": self.api_key, - "includepodid": "Result" - } - - response = requests.get(self.base_url, params=params, timeout=10) - response.raise_for_status() - - self.rate_limit_remaining -= 1 - - if format == "json": - result = response.json() - else: - result = {"raw": response.text} - - # Cache the result - self.cache[cache_key] = result - logger.info(f"Wolfram Alpha query successful: {question[:50]}...") - return result - - except requests.exceptions.RequestException as e: - logger.error(f"Wolfram Alpha API error: {e}") - return None - - def get_domain_knowledge(self, domain: str) -> Dict[str, Any]: - """ - Get comprehensive knowledge about a specific domain - - Args: - domain: Domain name (mathematics, physics, geometry, topology, etc.) - - Returns: - Structured domain knowledge - """ - queries = [ - f"What are the main concepts in {domain}", - f"List important theorems in {domain}", - f"Key applications of {domain}" - ] - - knowledge = { - "domain": domain, - "concepts": [], - "theorems": [], - "applications": [] - } - - for query in queries: - result = self.query(query) - if result and "queryresult" in result: - pods = result["queryresult"].get("pods", []) - for pod in pods: - if "subpods" in pod: - for subpod in pod["subpods"]: - if "plaintext" in subpod: - text = subpod["plaintext"] - if "concepts" in query: - knowledge["concepts"].append(text) - elif "theorems" in query: - knowledge["theorems"].append(text) - elif "applications" in query: - knowledge["applications"].append(text) - - return knowledge - - -class OpenMathKnowledge: - """OpenMath Content Dictionary ingestion""" - - def __init__(self): - self.base_url = "https://openmath.org/cd" - self.cache = {} - - def fetch_content_dictionary(self, cd_name: str) -> Optional[Dict[str, Any]]: - """ - Fetch an OpenMath Content Dictionary - - Args: - cd_name: Name of the content dictionary (e.g., "arith1", "alg1") - - Returns: - Parsed content dictionary or None if failed - """ - cache_key = f"openmath_{cd_name}" - if cache_key in self.cache: - return self.cache[cache_key] - - try: - url = f"{self.base_url}/{cd_name}.ocd" - response = requests.get(url, timeout=10) - response.raise_for_status() - - # Parse XML - remove namespace for easier parsing - root = ET.fromstring(response.text) - - cd_data = { - "name": cd_name, - "symbols": [] - } - - # Try multiple namespace approaches - namespaces = { - 'om': 'http://www.openmath.org/OpenMathCD', - 'm': 'http://www.w3.org/1998/Math/MathML', - '': '' - } - - for ns_prefix, ns_uri in namespaces.items(): - for symbol in root.findall(".//Symbol"): - if symbol.tag.endswith("Symbol"): - symbol_data = { - "name": symbol.get("name", ""), - "cd": symbol.get("cd", ""), - "role": symbol.get("role", ""), - "description": "" - } - - # Get description - for desc in symbol.findall(".//math"): - if desc.text: - symbol_data["description"] = desc.text - - if symbol_data["name"]: - cd_data["symbols"].append(symbol_data) - - # If no symbols found with namespace, try without - if not cd_data["symbols"]: - for elem in root.iter(): - if elem.tag.endswith("Symbol"): - symbol_data = { - "name": elem.get("name", ""), - "cd": elem.get("cd", ""), - "role": elem.get("role", ""), - "description": "" - } - if symbol_data["name"]: - cd_data["symbols"].append(symbol_data) - - self.cache[cache_key] = cd_data - logger.info(f"OpenMath CD fetched: {cd_name}") - return cd_data - - except (requests.exceptions.RequestException, ET.ParseError) as e: - logger.error(f"OpenMath CD fetch error for {cd_name}: {e}") - return None - - def get_relevant_cds(self) -> List[str]: - """ - Get list of relevant Content Dictionaries for this codebase - - Returns: - List of CD names - """ - # Relevant CDs for mathlib, physics, geometry, topology - relevant_cds = [ - "arith1", # Arithmetic - "alg1", # Algebra - "relation1", # Relations - "set1", # Sets - "logic1", # Logic - "fns1", # Functions - "nums1", # Numbers - "calculus1", # Calculus - "complex1", # Complex numbers - "linalg1", # Linear algebra - "analysis1", # Analysis - "geometry", # Geometry - "topology", # Topology - ] - return relevant_cds - - def ingest_all_relevant_cds(self) -> Dict[str, Dict[str, Any]]: - """ - Ingest all relevant Content Dictionaries - - Returns: - Dictionary mapping CD names to their data - """ - all_cds = {} - for cd_name in self.get_relevant_cds(): - cd_data = self.fetch_content_dictionary(cd_name) - if cd_data: - all_cds[cd_name] = cd_data - - logger.info(f"Ingested {len(all_cds)} OpenMath Content Dictionaries") - return all_cds - - -class NLabKnowledge: - """nLab local Git mirror ingestion for research-level mathematics/physics""" - - def __init__(self, nlab_path="docs/nlab"): - self.nlab_path = nlab_path - self.cache = {} - - def read_local_page(self, page_name: str) -> Optional[Dict[str, Any]]: - """ - Read an nLab page from local Git mirror - - Args: - page_name: Name of the nLab page (e.g., "topological_space", "category_theory") - - Returns: - Parsed page data or None if failed - """ - cache_key = f"nlab_{page_name}" - if cache_key in self.cache: - return self.cache[cache_key] - - try: - # Try multiple file paths - possible_paths = [ - os.path.join(self.nlab_path, page_name), - os.path.join(self.nlab_path, f"{page_name}.md"), - os.path.join(self.nlab_path, f"{page_name}.html"), - os.path.join(self.nlab_path, page_name, "index.md"), - os.path.join(self.nlab_path, page_name, "index.html") - ] - - content = None - file_path = None - - for path in possible_paths: - if os.path.exists(path): - file_path = path - with open(path, 'r', encoding='utf-8', errors='ignore') as f: - content = f.read() - break - - if not content: - # Try searching for files that contain the page name - for root, dirs, files in os.walk(self.nlab_path): - for file in files: - if page_name.lower() in file.lower(): - full_path = os.path.join(root, file) - try: - with open(full_path, 'r', encoding='utf-8', errors='ignore') as f: - content = f.read() - file_path = full_path - break - except: - continue - if content: - break - - if not content: - logger.warning(f"nLab page not found: {page_name}") - return None - - page_data = { - "name": page_name, - "title": page_name, - "content": content[:5000], # Limit content for efficiency - "file_path": file_path, - "categories": [], - "links": [] - } - - # Extract title from content - if "# " in content: - lines = content.split('\n') - for line in lines: - if line.startswith("# "): - page_data["title"] = line[2:].strip() - break - - # Extract links (markdown format) - import re - links = re.findall(r'\[([^\]]+)\]\([^\)]+\)', content) - page_data["links"] = links[:20] # Limit links - - self.cache[cache_key] = page_data - logger.info(f"nLab page read from local: {page_name} ({file_path})") - return page_data - - except Exception as e: - logger.error(f"nLab local read error for {page_name}: {e}") - return None - - def get_relevant_pages(self) -> List[str]: - """ - Get list of relevant nLab pages for this codebase - - Returns: - List of page names - """ - relevant_pages = [ - "topological_space", - "manifold", - "category_theory", - "homotopy_type_theory", - "higher_category_theory", - "simplicial_set", - "cohomology", - "homology", - "fiber_bundle", - "vector_bundle", - "symplectic_manifold", - "Riemannian_manifold", - "Lie_group", - "Lie_algebra", - "sheaf", - "topos" - ] - return relevant_pages - - def ingest_all_relevant_pages(self) -> Dict[str, Dict[str, Any]]: - """ - Ingest all relevant nLab pages from local Git mirror - - Returns: - Dictionary mapping page names to their data - """ - all_pages = {} - for page_name in self.get_relevant_pages(): - page_data = self.read_local_page(page_name) - if page_data: - all_pages[page_name] = page_data - - logger.info(f"Ingested {len(all_pages)} nLab pages from local Git mirror") - return all_pages - - -class KnowledgeIngestion: - """Main knowledge ingestion orchestrator""" - - def __init__(self, wolfram_api_key: Optional[str] = None): - self.wolfram = WolframAlphaKnowledge(wolfram_api_key) if wolfram_api_key else None - self.openmath = OpenMathKnowledge() - self.nlab = NLabKnowledge() - self.knowledge_base = { - "wolfram": {}, - "openmath": {}, - "nlab": {} - } - - def ingest_all(self) -> Dict[str, Any]: - """ - Ingest knowledge from all sources - - Returns: - Combined knowledge base - """ - logger.info("Starting knowledge ingestion from all sources") - - # Ingest from Wolfram Alpha if API key provided - if self.wolfram: - domains = ["mathematics", "physics", "geometry", "topology", "category_theory"] - for domain in domains: - knowledge = self.wolfram.get_domain_knowledge(domain) - self.knowledge_base["wolfram"][domain] = knowledge - logger.info("Wolfram Alpha ingestion complete") - - # Ingest from OpenMath - self.knowledge_base["openmath"] = self.openmath.ingest_all_relevant_cds() - - # Ingest from nLab - self.knowledge_base["nlab"] = self.nlab.ingest_all_relevant_pages() - - logger.info("Knowledge ingestion complete") - return self.knowledge_base - - def query_knowledge_base(self, question: str) -> Dict[str, Any]: - """ - Query the knowledge base with a question - - Args: - question: The question to answer - - Returns: - Relevant knowledge from all sources - """ - results = { - "question": question, - "sources": [] - } - - # Try Wolfram Alpha first if available - if self.wolfram: - wolfram_result = self.wolfram.query(question) - if wolfram_result: - results["sources"].append({ - "name": "Wolfram Alpha", - "data": wolfram_result - }) - - # Search OpenMath for relevant symbols - question_lower = question.lower() - for cd_name, cd_data in self.knowledge_base["openmath"].items(): - for symbol in cd_data.get("symbols", []): - if any(keyword in symbol.get("name", "").lower() for keyword in question_lower.split()): - results["sources"].append({ - "name": f"OpenMath: {cd_name}", - "data": symbol - }) - - # Search nLab for relevant pages - for page_name, page_data in self.knowledge_base["nlab"].items(): - if any(keyword in page_data.get("title", "").lower() for keyword in question_lower.split()): - results["sources"].append({ - "name": f"nLab: {page_name}", - "data": page_data - }) - - return results - - def export_knowledge_base(self, output_path: str): - """ - Export knowledge base to JSON file - - Args: - output_path: Path to output JSON file - """ - with open(output_path, 'w') as f: - json.dump(self.knowledge_base, f, indent=2) - logger.info(f"Knowledge base exported to {output_path}") - - def load_knowledge_base(self, input_path: str): - """ - Load knowledge base from JSON file - - Args: - input_path: Path to input JSON file - """ - with open(input_path, 'r') as f: - self.knowledge_base = json.load(f) - logger.info(f"Knowledge base loaded from {input_path}") diff --git a/4-Infrastructure/infra/manifold_geometry.py b/4-Infrastructure/infra/manifold_geometry.py deleted file mode 100644 index 6e9c46f9..00000000 --- a/4-Infrastructure/infra/manifold_geometry.py +++ /dev/null @@ -1,350 +0,0 @@ -#!/usr/bin/env python3 -""" -Manifold Intrinsic Geometry — Computing the Actual Shape of the Research Stack - -This script extracts the dependency graph from Lean imports and computes: -- Geodesic distances (shortest import path) -- Curvature (information flow divergence/convergence at each module) -- Ricci curvature (neighborhood flow density) -- Central hubs (high betweenness centrality) -- Boundary modules (low in-degree, high out-degree = sources; high in, low out = sinks) -- Cyclic dependencies (genus / non-trivial topology) -- Module clustering (connected components) - -The output is a JSON file with the full geometric characterization. -""" - -import os -import re -import json -import sys -from pathlib import Path -from collections import defaultdict, deque -from itertools import combinations - -SEMANTICS_DIR = Path("/home/allaun/Documents/Research Stack/0-Core-Formalism/lean/Semantics/Semantics") -OUTPUT_PATH = Path("/home/allaun/Documents/Research Stack/data/manifold_intrinsic_geometry.json") - - -def extract_imports(filepath: Path) -> list: - """Extract Semantics.* imports from a Lean file.""" - imports = [] - try: - text = filepath.read_text() - for line in text.splitlines(): - m = re.match(r'^\s*import\s+Semantics\.(\S+)', line) - if m: - imports.append(m.group(1)) - except Exception: - pass - return imports - - -def build_graph(): - """Build adjacency list of the import graph.""" - nodes = set() - edges = defaultdict(set) # a -> {b, c} means a imports b and c - reverse_edges = defaultdict(set) # b -> {a, c} means b is imported by a and c - - if not SEMANTICS_DIR.exists(): - print(f"ERROR: {SEMANTICS_DIR} not found") - sys.exit(1) - - # Find all .lean files in Semantics directory (excluding .lake) - lean_files = [] - for root, dirs, files in os.walk(SEMANTICS_DIR): - if '.lake' in root: - continue - dirs[:] = [d for d in dirs if d != '.lake'] - for f in files: - if f.endswith('.lean'): - lean_files.append(Path(root) / f) - - for fpath in lean_files: - rel = fpath.relative_to(SEMANTICS_DIR) - # Module name: path with / replaced by . and .lean stripped - mod_name = str(rel).replace('/', '.').replace('.lean', '') - nodes.add(mod_name) - imports = extract_imports(fpath) - for imp in imports: - edges[mod_name].add(imp) - reverse_edges[imp].add(mod_name) - nodes.add(imp) - - return nodes, edges, reverse_edges - - -def bfs_distance(start, edges, all_nodes): - """Compute shortest path distances from start to all reachable nodes.""" - dist = {n: float('inf') for n in all_nodes} - dist[start] = 0 - q = deque([start]) - while q: - u = q.popleft() - for v in edges.get(u, []): - if dist[v] == float('inf'): - dist[v] = dist[u] + 1 - q.append(v) - return dist - - -def compute_all_pairs_distances(nodes, edges): - """Compute all-pairs shortest path distances.""" - distances = {} - for n in nodes: - distances[n] = bfs_distance(n, edges, nodes) - return distances - - -def compute_curvature(node, edges, reverse_edges): - """ - Compute Ollivier-Ricci curvature approximation for a node. - Roughly: if many modules import this node (converge), positive curvature. - If this node imports many and is imported by few (diverge), negative curvature. - """ - out_deg = len(edges.get(node, set())) - in_deg = len(reverse_edges.get(node, set())) - if out_deg + in_deg == 0: - return 0.0 - # Simple approximation: curvature = (in - out) / (in + out) - # Positive = sink (information converges here) - # Negative = source (information diverges from here) - return (in_deg - out_deg) / (in_deg + out_deg) - - -def compute_betweenness_centrality(nodes, edges): - """Compute betweenness centrality (approximate, for connected pairs).""" - # For each pair (s, t), count how many shortest paths go through each node - centrality = {n: 0 for n in nodes} - node_list = list(nodes) - - for s in node_list: - # BFS from s - dist = {n: float('inf') for n in nodes} - dist[s] = 0 - pred = {n: [] for n in nodes} - q = deque([s]) - while q: - u = q.popleft() - for v in edges.get(u, set()): - if dist[v] == float('inf'): - dist[v] = dist[u] + 1 - q.append(v) - pred[v].append(u) - elif dist[v] == dist[u] + 1: - pred[v].append(u) - - # Count paths (simplified: assume each edge contributes 1 path) - for t in node_list: - if t == s or dist[t] == float('inf'): - continue - # Mark all nodes on any shortest path from s to t - visited = set() - stack = [t] - while stack: - u = stack.pop() - if u in visited or u == s: - continue - visited.add(u) - for p in pred.get(u, []): - if p not in visited: - stack.append(p) - for u in visited: - if u != t: - centrality[u] += 1 - - # Normalize - max_c = max(centrality.values()) if centrality else 1 - if max_c > 0: - centrality = {k: v / max_c for k, v in centrality.items()} - return centrality - - -def find_cycles(nodes, edges): - """Find all simple cycles in the graph (up to length 5 for performance).""" - cycles = [] - visited = set() - - def dfs(node, path, depth): - if depth > 5: - return - for neighbor in edges.get(node, set()): - if neighbor == path[0] and len(path) >= 2: - cycles.append(path + [neighbor]) - elif neighbor not in path and neighbor not in visited: - visited.add(neighbor) - dfs(neighbor, path + [neighbor], depth + 1) - visited.remove(neighbor) - - for n in nodes: - visited.clear() - visited.add(n) - dfs(n, [n], 1) - - # Deduplicate (same cycle, different start points) - unique_cycles = [] - seen = set() - for c in cycles: - # Normalize: start from smallest element, keep direction - start_idx = c.index(min(c[:-1])) - normalized = tuple(c[start_idx:-1] + c[:start_idx] + [c[start_idx]]) - if normalized not in seen: - seen.add(normalized) - unique_cycles.append(c) - - return unique_cycles - - -def find_connected_components(nodes, edges): - """Find weakly connected components (treating edges as undirected).""" - visited = set() - components = [] - - # Build undirected adjacency - undirected = defaultdict(set) - for u, vs in edges.items(): - for v in vs: - undirected[u].add(v) - undirected[v].add(u) - - def bfs(start): - comp = [] - q = deque([start]) - visited.add(start) - while q: - u = q.popleft() - comp.append(u) - for v in undirected[u]: - if v not in visited: - visited.add(v) - q.append(v) - return comp - - for n in nodes: - if n not in visited: - components.append(bfs(n)) - - return components - - -def main(): - print("[ManifoldGeometry] Building dependency graph...") - nodes, edges, reverse_edges = build_graph() - print(f" Nodes: {len(nodes)}") - print(f" Edges: {sum(len(v) for v in edges.values())}") - - print("[ManifoldGeometry] Computing geodesic distances...") - distances = compute_all_pairs_distances(nodes, edges) - - # Compute diameter (longest shortest path) - finite_dists = [d for dd in distances.values() for d in dd.values() if d != float('inf') and d > 0] - diameter = max(finite_dists) if finite_dists else 0 - avg_dist = sum(finite_dists) / len(finite_dists) if finite_dists else 0 - - print(f" Diameter: {diameter}") - print(f" Average distance: {avg_dist:.2f}") - - print("[ManifoldGeometry] Computing curvature...") - curvature = {n: compute_curvature(n, edges, reverse_edges) for n in nodes} - - print("[ManifoldGeometry] Computing centrality...") - centrality = compute_betweenness_centrality(nodes, edges) - - print("[ManifoldGeometry] Finding cycles...") - cycles = find_cycles(nodes, edges) - print(f" Cycles found: {len(cycles)}") - - print("[ManifoldGeometry] Finding connected components...") - components = find_connected_components(nodes, edges) - print(f" Components: {len(components)}") - - # Identify key geometric features - hubs = sorted(centrality.items(), key=lambda x: x[1], reverse=True)[:15] - high_curv = sorted(curvature.items(), key=lambda x: x[1], reverse=True)[:10] - low_curv = sorted(curvature.items(), key=lambda x: x[1])[:10] - - # Boundary detection - sources = [(n, len(edges.get(n, set())), len(reverse_edges.get(n, set()))) - for n in nodes if len(reverse_edges.get(n, set())) == 0 and len(edges.get(n, set())) > 0] - sinks = [(n, len(edges.get(n, set())), len(reverse_edges.get(n, set()))) - for n in nodes if len(edges.get(n, set())) == 0 and len(reverse_edges.get(n, set())) > 0] - - # Modules with no imports and no importers (isolated points) - isolated = [n for n in nodes if len(edges.get(n, set())) == 0 and len(reverse_edges.get(n, set())) == 0] - - report = { - "meta": { - "node_count": len(nodes), - "edge_count": sum(len(v) for v in edges.values()), - "diameter": diameter, - "average_distance": avg_dist, - "cycle_count": len(cycles), - "component_count": len(components), - }, - "hubs": [{"module": n, "centrality": round(c, 4)} for n, c in hubs], - "positive_curvature": [{"module": n, "curvature": round(c, 4)} for n, c in high_curv], - "negative_curvature": [{"module": n, "curvature": round(c, 4)} for n, c in low_curv], - "sources": [{"module": n, "out_degree": o, "in_degree": i} for n, o, i in sources], - "sinks": [{"module": n, "out_degree": o, "in_degree": i} for n, o, i in sinks], - "isolated": isolated, - "cycles": [c for c in cycles[:20]], # Limit output size - "components": [ - {"size": len(comp), "modules": comp[:50]} # Truncate large components - for comp in sorted(components, key=len, reverse=True) - ], - "full_graph": { - "nodes": list(nodes), - "edges": {k: list(v) for k, v in edges.items()}, - }, - } - - OUTPUT_PATH.parent.mkdir(parents=True, exist_ok=True) - with open(OUTPUT_PATH, 'w') as f: - json.dump(report, f, indent=2) - print(f"[ManifoldGeometry] Wrote: {OUTPUT_PATH}") - - # Print summary - print("\n" + "=" * 70) - print("INTRINSIC GEOMETRY OF THE RESEARCH STACK") - print("=" * 70) - print(f"\nScale: {len(nodes)} modules, {sum(len(v) for v in edges.values())} import edges") - print(f"Diameter (longest shortest path): {diameter}") - print(f"Average geodesic distance: {avg_dist:.2f}") - print(f"Cycles (non-trivial topology): {len(cycles)}") - print(f"Connected components: {len(components)}") - - print(f"\n--- HUBS (High Betweenness Centrality) ---") - for n, c in hubs[:10]: - print(f" {n:40s} centrality={c:.4f}") - - print(f"\n--- POSITIVE CURVATURE (Information Converges Here) ---") - for n, c in high_curv[:5]: - print(f" {n:40s} curvature={c:+.4f}") - - print(f"\n--- NEGATIVE CURVATURE (Information Diverges From Here) ---") - for n, c in low_curv[:5]: - print(f" {n:40s} curvature={c:+.4f}") - - print(f"\n--- SOURCES (No imports, pure origin) ---") - for n, o, i in sources[:5]: - print(f" {n:40s} out={o}") - - print(f"\n--- SINKS (No exports, dead ends) ---") - for n, o, i in sinks[:5]: - print(f" {n:40s} in={i}") - - if isolated: - print(f"\n--- ISOLATED POINTS (No connections) ---") - for n in isolated[:10]: - print(f" {n}") - - if cycles: - print(f"\n--- CYCLES (Non-contractible loops) ---") - for c in cycles[:5]: - print(f" {' -> '.join(c)}") - - print("\n" + "=" * 70) - - -if __name__ == "__main__": - main() diff --git a/4-Infrastructure/infra/manifold_perception.py b/4-Infrastructure/infra/manifold_perception.py deleted file mode 100644 index 85f09656..00000000 --- a/4-Infrastructure/infra/manifold_perception.py +++ /dev/null @@ -1,390 +0,0 @@ -#!/usr/bin/env python3 -""" -Manifold Perception Engine — Topological Analysis of the Research Stack - -This script implements the extraction engine for ManifoldTopology.lean. -It scans the entire Research Stack, classifies artifacts, maps them onto -manifold dimensions, identifies boundaries and holes, and generates a -structured topological report. - -Per AGENTS.md §0: Lean is the source of truth. This script is an extraction -engine only — all invariants are formally defined in ManifoldTopology.lean. - -Usage: - cd /home/allaun/Research\ Stack && python3 infra/manifold_perception.py - -Output: - data/manifold_topology_report.json — structured topological analysis - data/manifold_holes.json — detected gaps requiring attention - data/manifold_boundaries.json — boundary analysis per dimension -""" - -import os -import re -import json -import hashlib -from pathlib import Path -from collections import defaultdict -from datetime import datetime - - -RESEARCH_STACK = Path("/home/allaun/Documents/Research Stack") -DATA_DIR = RESEARCH_STACK / "data" -LEAN_DIR = RESEARCH_STACK / "tools" / "lean" / "Semantics" / "Semantics" -DOCS_DIR = RESEARCH_STACK / "docs" -INFRA_DIR = RESEARCH_STACK / "4-Infrastructure" / "infra" - - -def sha256_file(path: Path) -> str: - """Compute SHA-256 hash of file contents.""" - h = hashlib.sha256() - h.update(path.read_bytes()) - return h.hexdigest()[:16] - - -def count_lines(path: Path) -> int: - """Count lines in a file.""" - try: - return len(path.read_text().splitlines()) - except Exception: - return 0 - - -def classify_artifact(path: Path) -> str: - """Classify an artifact into a PointKind-equivalent category.""" - suffix = path.suffix.lower() - rel = str(path.relative_to(RESEARCH_STACK)) - - if suffix == ".lean": - if "ReceiptCore" in rel or "ManifoldTopology" in rel or "GeometricCompression" in rel: - return "core_lean" - return "lean_module" - elif suffix == ".md": - if "MATH_MODEL_MAP" in rel: - return "math_model_registry" - return "markdown_doc" - elif suffix == ".py": - if "manifold_perception" in rel: - return "extraction_engine" - return "python_script" - elif suffix == ".toml": - return "toml_config" - elif suffix in (".json", ".jsonl"): - return "json_data" - elif suffix == ".tsv": - return "tsv_data" - elif suffix == ".parquet": - return "parquet_data" - elif suffix == ".db": - return "sqlite_database" - elif suffix in (".rs", ".rsx"): - return "rust_source" - elif suffix in (".c", ".cpp", ".h", ".hpp"): - return "c_source" - elif suffix in (".v", ".sv", ".vhd", ".vhdl"): - return "hardware_source" - else: - return "other" - - -def parse_lean_file(path: Path) -> dict: - """Extract metadata from a Lean file.""" - text = path.read_text() - return { - "theorem_count": len(re.findall(r'^\s*theorem\s+\w+', text, re.MULTILINE)), - "def_count": len(re.findall(r'^\s*def\s+\w+', text, re.MULTILINE)), - "inductive_count": len(re.findall(r'^\s*inductive\s+\w+', text, re.MULTILINE)), - "structure_count": len(re.findall(r'^\s*structure\s+\w+', text, re.MULTILINE)), - "eval_count": len(re.findall(r'^\s*#eval', text, re.MULTILINE)), - "sorry_count": len(re.findall(r'\bsorry\b', text)), - "namespace_count": len(re.findall(r'^\s*namespace\s+', text, re.MULTILINE)), - "imports": re.findall(r'^\s*import\s+(.+)', text, re.MULTILINE), - } - - -def parse_markdown_file(path: Path) -> dict: - """Extract metadata from a Markdown file.""" - text = path.read_text() - return { - "heading_count": len(re.findall(r'^#{1,6}\s+', text, re.MULTILINE)), - "equation_count": len(re.findall(r'\$\$.+?\$\$', text, re.DOTALL)), - "inline_equation_count": len(re.findall(r'\$(?!\$).+?\$', text)), - "table_count": len(re.findall(r'^\|.*\|.*\|', text, re.MULTILINE)), - "code_block_count": len(re.findall(r'^```', text, re.MULTILINE)), - "reference_count": len(re.findall(r'\[.*?\]\(.*?\)', text)), - } - - -def parse_math_model_map(path: Path) -> dict: - """Extract MATH_MODEL_MAP entries.""" - text = path.read_text() - entries = [] - for line in text.splitlines(): - if line.startswith("|") and not line.startswith("| #") and not line.startswith("|---"): - parts = [p.strip() for p in line.split("|")] - if len(parts) >= 4 and parts[1].isdigit(): - entries.append({ - "id": int(parts[1]), - "name": parts[2], - "equation": parts[3], - "description": parts[4] if len(parts) > 4 else "", - }) - return {"entries": entries, "count": len(entries)} - - -def scan_directory(directory: Path, ignore_patterns=None) -> list: - """Recursively scan a directory and return all file paths.""" - if ignore_patterns is None: - ignore_patterns = ['.git', '.lake', 'build', '__pycache__', '.mypy_cache', '.pytest_cache', 'node_modules', '.rclone'] - - files = [] - for root, dirs, filenames in os.walk(directory): - # Filter out ignored directories - dirs[:] = [d for d in dirs if d not in ignore_patterns and not d.startswith('.')] - for fname in filenames: - files.append(Path(root) / fname) - return files - - -def build_manifold(): - """Build the complete topological manifold of the Research Stack.""" - print("[ManifoldPerception] Scanning Research Stack...") - - # --- Scan all directories --- - lean_files = list(LEAN_DIR.rglob("*.lean")) if LEAN_DIR.exists() else [] - docs_files = list(DOCS_DIR.rglob("*.md")) if DOCS_DIR.exists() else [] - infra_files = list(INFRA_DIR.rglob("*.py")) if INFRA_DIR.exists() else [] - data_files = list(DATA_DIR.rglob("*")) if DATA_DIR.exists() else [] - - # Remove hidden/lake/build artifacts - lean_files = [f for f in lean_files if '.lake' not in str(f) and 'build' not in str(f)] - - print(f" Found {len(lean_files)} Lean files") - print(f" Found {len(docs_files)} Markdown docs") - print(f" Found {len(infra_files)} Python infra files") - print(f" Found {len(data_files)} data artifacts") - - # --- Classify Lean files --- - lean_metadata = [] - for f in lean_files: - meta = parse_lean_file(f) - meta.update({ - "path": str(f.relative_to(RESEARCH_STACK)), - "lines": count_lines(f), - "hash": sha256_file(f), - "kind": classify_artifact(f), - }) - lean_metadata.append(meta) - - # --- Classify docs --- - docs_metadata = [] - for f in docs_files: - meta = parse_markdown_file(f) - meta.update({ - "path": str(f.relative_to(RESEARCH_STACK)), - "lines": count_lines(f), - "hash": sha256_file(f), - "kind": classify_artifact(f), - }) - docs_metadata.append(meta) - - # --- MATH_MODEL_MAP analysis --- - math_map_path = DOCS_DIR / "MATH_MODEL_MAP.md" - math_map = parse_math_model_map(math_map_path) if math_map_path.exists() else {"entries": [], "count": 0} - - # --- Compute aggregate statistics --- - total_lean_lines = sum(m["lines"] for m in lean_metadata) - total_doc_lines = sum(m["lines"] for m in docs_metadata) - total_theorems = sum(m["theorem_count"] for m in lean_metadata) - total_defs = sum(m["def_count"] for m in lean_metadata) - total_sorry = sum(m["sorry_count"] for m in lean_metadata) - total_evals = sum(m["eval_count"] for m in lean_metadata) - total_structures = sum(m["structure_count"] for m in lean_metadata) - total_inductives = sum(m["inductive_count"] for m in lean_metadata) - - # --- Identify holes (gaps) --- - holes = [] - - # Hole 1: Lean files with sorry but no proven theorems - for m in lean_metadata: - if m["sorry_count"] > 0 and m["theorem_count"] == 0: - holes.append({ - "center": m["path"], - "expected_kind": "theorem_or_def", - "severity": "critical", - "description": f"File has {m['sorry_count']} sorry() but 0 theorems — blocked formalization", - "missing_count": m["sorry_count"], - }) - - # Hole 2: Lean files with no #eval witnesses - for m in lean_metadata: - if m["eval_count"] == 0 and m["def_count"] > 0: - holes.append({ - "center": m["path"], - "expected_kind": "eval_witness", - "severity": "structural", - "description": f"File has {m['def_count']} definitions but 0 #eval witnesses", - "missing_count": m["def_count"], - }) - - # Hole 3: MATH_MODEL_MAP entries without Lean implementations - lean_modules = {m["path"].replace("0-Core-Formalism/lean/Semantics/Semantics/", "").replace(".lean", "") - for m in lean_metadata} - for entry in math_map["entries"]: - name = entry["name"].replace(" ", "").replace("-", "") - # Heuristic: does any Lean module name match? - if not any(name.lower() in mod.lower() for mod in lean_modules): - holes.append({ - "center": f"MATH_MODEL_MAP#{entry['id']}", - "expected_kind": "lean_module", - "severity": "structural", - "description": f"Model '{entry['name']}' has no corresponding Lean module", - "missing_count": 1, - }) - - # Hole 4: Missing receipt infrastructure for new domains - receipt_kinds = ["leanBuild", "benchmark", "sourceAudit", "reverseCollapse", - "deltaPhiAudit", "adversarialTrial", "humanReview", - "wardenEmission", "externalProof"] - # Already all present in ReceiptCore — this is a boundary, not a hole - - # --- Identify boundaries --- - boundaries = [] - - # Boundary 1: Total Lean code size - boundaries.append({ - "dimension": "lineCount", - "position": total_lean_lines, - "is_terminal": total_lean_lines > 120000, - "description": f"Lean corpus at {total_lean_lines} lines (capacity ~130K)", - }) - - # Boundary 2: TTM Layer M concentration - layer_m_files = [m for m in lean_metadata if "ReceiptCore" in m["path"] or - "ManifoldTopology" in m["path"] or - "GeometricCompression" in m["path"] or - "FixedPoint" in m["path"]] - boundaries.append({ - "dimension": "ttmLayer", - "position": len(layer_m_files), - "is_terminal": len(layer_m_files) > 10, - "description": f"Layer M (Lean Semantics): {len(layer_m_files)} core modules", - }) - - # Boundary 3: Proof completeness - proof_ratio = (total_theorems / (total_theorems + total_sorry)) if (total_theorems + total_sorry) > 0 else 1.0 - boundaries.append({ - "dimension": "proofCompleteness", - "position": int(proof_ratio * 100), - "is_terminal": proof_ratio >= 0.95, - "description": f"Proof completeness: {proof_ratio:.1%} ({total_theorems} theorems, {total_sorry} sorry)", - }) - - # Boundary 4: Documentation coverage - doc_coverage = len(docs_metadata) / max(len(lean_metadata), 1) - boundaries.append({ - "dimension": "documentationCoverage", - "position": int(doc_coverage * 100), - "is_terminal": doc_coverage >= 1.0, - "description": f"Doc coverage: {doc_coverage:.1%} ({len(docs_metadata)} docs / {len(lean_metadata)} Lean files)", - }) - - # --- Compute cross-reference density --- - all_imports = set() - for m in lean_metadata: - for imp in m.get("imports", []): - all_imports.add(imp.strip()) - boundaries.append({ - "dimension": "crossReferenceDensity", - "position": len(all_imports), - "is_terminal": len(all_imports) > 50, - "description": f"Cross-reference density: {len(all_imports)} unique imports", - }) - - # --- Build the manifold report --- - report = { - "generated_at": datetime.utcnow().isoformat(), - "observer": "aiFull", - "dimensions": { - "ttmLayer": 13, - "formalizationDepth": 5, - "fileCount": len(lean_files) + len(docs_files), - "lineCount": total_lean_lines + total_doc_lines, - "crossReferenceDensity": len(all_imports), - "documentationCoverage": int(doc_coverage * 100), - "proofCompleteness": int(proof_ratio * 100), - }, - "points": { - "lean_files": len(lean_metadata), - "doc_files": len(docs_metadata), - "infra_files": len(infra_files), - "data_files": len(data_files), - "math_models": math_map["count"], - "total_lines": total_lean_lines + total_doc_lines, - }, - "structures": { - "theorems": total_theorems, - "definitions": total_defs, - "inductives": total_inductives, - "structures": total_structures, - "eval_witnesses": total_evals, - "sorry_markers": total_sorry, - }, - "boundaries": boundaries, - "holes": holes, - "top_files_by_theorems": sorted(lean_metadata, key=lambda x: x["theorem_count"], reverse=True)[:10], - "top_files_by_sorry": sorted(lean_metadata, key=lambda x: x["sorry_count"], reverse=True)[:10], - } - - # --- Write outputs --- - DATA_DIR.mkdir(parents=True, exist_ok=True) - - report_path = DATA_DIR / "manifold_topology_report.json" - with open(report_path, 'w') as f: - json.dump(report, f, indent=2, default=str) - print(f"[ManifoldPerception] Wrote: {report_path}") - - holes_path = DATA_DIR / "manifold_holes.json" - with open(holes_path, 'w') as f: - json.dump({"holes": holes, "count": len(holes), "severity_counts": { - "cosmetic": len([h for h in holes if h["severity"] == "cosmetic"]), - "structural": len([h for h in holes if h["severity"] == "structural"]), - "critical": len([h for h in holes if h["severity"] == "critical"]), - "existential": len([h for h in holes if h["severity"] == "existential"]), - }}, f, indent=2, default=str) - print(f"[ManifoldPerception] Wrote: {holes_path}") - - boundaries_path = DATA_DIR / "manifold_boundaries.json" - with open(boundaries_path, 'w') as f: - json.dump({"boundaries": boundaries, "count": len(boundaries)}, f, indent=2, default=str) - print(f"[ManifoldPerception] Wrote: {boundaries_path}") - - # --- Print summary to console --- - print("\n" + "=" * 70) - print("MANIFOLD TOPOLOGY REPORT") - print("=" * 70) - print(f"\nDimensions:") - for k, v in report["dimensions"].items(): - print(f" {k:25s}: {v}") - print(f"\nArtifacts:") - for k, v in report["points"].items(): - print(f" {k:25s}: {v}") - print(f"\nFormal Structures:") - for k, v in report["structures"].items(): - print(f" {k:25s}: {v}") - print(f"\nBoundaries: {len(boundaries)}") - for b in boundaries: - term = "TERMINAL" if b["is_terminal"] else "soft" - print(f" [{term}] {b['dimension']:25s} @ {b['position']:6d} — {b['description']}") - print(f"\nHoles: {len(holes)}") - for h in holes[:10]: # Show top 10 - print(f" [{h['severity']:10s}] {h['center']:40s} — {h['description']}") - if len(holes) > 10: - print(f" ... and {len(holes) - 10} more holes") - print("\n" + "=" * 70) - - return report - - -if __name__ == "__main__": - build_manifold() diff --git a/4-Infrastructure/infra/moe_ene_cache.py b/4-Infrastructure/infra/moe_ene_cache.py deleted file mode 100644 index 8d9d3330..00000000 --- a/4-Infrastructure/infra/moe_ene_cache.py +++ /dev/null @@ -1,370 +0,0 @@ -#!/usr/bin/env python3 -""" -MoE-ENE Cache Integration - -Caches Mixture-of-Experts (MoE) configurations and outputs in the ENE database -for fast retrieval and persistent storage. Integrates with EtaMoE.lean and -SwarmMoERewiring.lean for expert management. - -Cache Strategy: -- Expert configurations stored as sensitive data with RESTRICTED classification -- Performance metrics cached with semantic vectors for retrieval -- Gating weights tracked with version history -- Swarm-driven rewiring proposals stored with audit trail -""" - -import json -import sqlite3 -import sys -import time -from dataclasses import dataclass, asdict -from pathlib import Path -from typing import Optional, Dict, List, Any - -# Add parent directory to path for imports -sys.path.insert(0, str(Path(__file__).parent.parent)) - -from infra.ene_api import ENEAPIHook, AccessLevel - - -@dataclass -class ExpertConfiguration: - """MoE Expert Configuration""" - expert_id: int - gating_weight: float # g - quality_weight: float # w - coherence: float # h - penalty_weight: float # v - distortion: float # p - arity: float # N - cost_coefficient: float # a - overhead: float # c - semantic_vector: List[float] # 14D semantic space - domain: str - version: str - - -@dataclass -class MoECacheEntry: - """Cached MoE computation result""" - cache_key: str - expert_ids: List[int] - eta_moe_result: float - i_discarded: float - timestamp: int - semantic_vector: List[float] - confidence: float - - -class MoEENECache: - """MoE Cache Manager using ENE database""" - - def __init__(self, db_path: str = "/home/allaun/Documents/Research Stack/data/substrate_index.db"): - self.db_path = db_path - self.ene_api = ENEAPIHook() - self._init_cache_tables() - - def _init_cache_tables(self): - """Initialize cache-specific tables in ENE database""" - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - # Expert configurations table - cursor.execute(""" - CREATE TABLE IF NOT EXISTS moe_expert_cache ( - expert_id INTEGER PRIMARY KEY, - domain TEXT NOT NULL, - config_json TEXT NOT NULL, - semantic_vector TEXT NOT NULL, - version TEXT NOT NULL, - created_at INTEGER NOT NULL, - updated_at INTEGER NOT NULL, - cache_hit_count INTEGER DEFAULT 0 - ) - """) - - # Computation results cache - cursor.execute(""" - CREATE TABLE IF NOT EXISTS moe_computation_cache ( - cache_key TEXT PRIMARY KEY, - expert_ids TEXT NOT NULL, - eta_moe_result REAL NOT NULL, - i_discarded REAL NOT NULL, - semantic_vector TEXT NOT NULL, - confidence REAL NOT NULL, - created_at INTEGER NOT NULL, - hit_count INTEGER DEFAULT 0 - ) - """) - - # Rewiring proposals audit log - cursor.execute(""" - CREATE TABLE IF NOT EXISTS moe_rewiring_audit ( - id TEXT PRIMARY KEY, - expert_id INTEGER NOT NULL, - proposal_json TEXT NOT NULL, - swarm_consensus REAL NOT NULL, - proposing_agent TEXT NOT NULL, - approved BOOLEAN NOT NULL, - created_at INTEGER NOT NULL - ) - """) - - conn.commit() - conn.close() - - def cache_expert_config(self, config: ExpertConfiguration) -> bool: - """Cache expert configuration in ENE database""" - try: - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - config_json = json.dumps(asdict(config)) - semantic_vector_json = json.dumps(config.semantic_vector) - now = int(time.time()) - - cursor.execute(""" - INSERT OR REPLACE INTO moe_expert_cache - (expert_id, domain, config_json, semantic_vector, version, created_at, updated_at) - VALUES (?, ?, ?, ?, ?, ?, ?) - """, ( - config.expert_id, - config.domain, - config_json, - semantic_vector_json, - config.version, - now, - now - )) - - # Also store in ENE sensitive_data for security - self.ene_api.store_sensitive_data( - pkg=f"moe/expert/{config.expert_id}", - payload=config_json, - classification=AccessLevel.RESTRICTED, - semantic_vector=config.semantic_vector - ) - - conn.commit() - conn.close() - return True - except Exception as e: - print(f"Error caching expert config: {e}") - return False - - def retrieve_expert_config(self, expert_id: int) -> Optional[ExpertConfiguration]: - """Retrieve expert configuration from cache""" - try: - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - cursor.execute(""" - SELECT config_json, cache_hit_count - FROM moe_expert_cache - WHERE expert_id = ? - """, (expert_id,)) - - row = cursor.fetchone() - - if row: - # Increment hit count - cursor.execute(""" - UPDATE moe_expert_cache - SET cache_hit_count = cache_hit_count + 1 - WHERE expert_id = ? - """, (expert_id,)) - conn.commit() - conn.close() - - config_dict = json.loads(row[0]) - return ExpertConfiguration(**config_dict) - - conn.close() - return None - except Exception as e: - print(f"Error retrieving expert config: {e}") - return None - - def cache_computation_result(self, entry: MoECacheEntry) -> bool: - """Cache MoE computation result""" - try: - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - expert_ids_json = json.dumps(entry.expert_ids) - semantic_vector_json = json.dumps(entry.semantic_vector) - - cursor.execute(""" - INSERT OR REPLACE INTO moe_computation_cache - (cache_key, expert_ids, eta_moe_result, i_discarded, semantic_vector, confidence, created_at) - VALUES (?, ?, ?, ?, ?, ?, ?) - """, ( - entry.cache_key, - expert_ids_json, - entry.eta_moe_result, - entry.i_discarded, - semantic_vector_json, - entry.confidence, - entry.timestamp - )) - - conn.commit() - conn.close() - return True - except Exception as e: - print(f"Error caching computation result: {e}") - return False - - def retrieve_computation_result(self, cache_key: str) -> Optional[MoECacheEntry]: - """Retrieve cached computation result""" - try: - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - cursor.execute(""" - SELECT expert_ids, eta_moe_result, i_discarded, semantic_vector, confidence, created_at - FROM moe_computation_cache - WHERE cache_key = ? - """, (cache_key,)) - - row = cursor.fetchone() - - if row: - # Increment hit count - cursor.execute(""" - UPDATE moe_computation_cache - SET hit_count = hit_count + 1 - WHERE cache_key = ? - """, (cache_key,)) - conn.commit() - conn.close() - - return MoECacheEntry( - cache_key=cache_key, - expert_ids=json.loads(row[0]), - eta_moe_result=row[1], - i_discarded=row[2], - semantic_vector=json.loads(row[3]), - confidence=row[4], - timestamp=row[5] - ) - - conn.close() - return None - except Exception as e: - print(f"Error retrieving computation result: {e}") - return None - - def log_rewiring_proposal(self, expert_id: int, proposal: Dict, swarm_consensus: float, - proposing_agent: str) -> str: - """Log rewiring proposal to audit trail""" - proposal_id = f"rewire_{expert_id}_{int(time.time())}" - - try: - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - cursor.execute(""" - INSERT INTO moe_rewiring_audit - (id, expert_id, proposal_json, swarm_consensus, proposing_agent, approved, created_at) - VALUES (?, ?, ?, ?, ?, ?, ?) - """, ( - proposal_id, - expert_id, - json.dumps(proposal), - swarm_consensus, - proposing_agent, - False, # Pending approval - int(time.time()) - )) - - conn.commit() - conn.close() - return proposal_id - except Exception as e: - print(f"Error logging rewiring proposal: {e}") - return "" - - def get_cache_statistics(self) -> Dict[str, Any]: - """Get cache statistics""" - try: - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - # Expert cache stats - cursor.execute("SELECT COUNT(*), SUM(cache_hit_count) FROM moe_expert_cache") - expert_count, expert_hits = cursor.fetchone() - - # Computation cache stats - cursor.execute("SELECT COUNT(*), SUM(hit_count) FROM moe_computation_cache") - comp_count, comp_hits = cursor.fetchone() - - # Rewiring audit stats - cursor.execute("SELECT COUNT(*) FROM moe_rewiring_audit") - audit_count = cursor.fetchone()[0] - - conn.close() - - return { - "expert_cache_entries": expert_count or 0, - "expert_cache_hits": expert_hits or 0, - "computation_cache_entries": comp_count or 0, - "computation_cache_hits": comp_hits or 0, - "rewiring_proposals": audit_count or 0 - } - except Exception as e: - print(f"Error getting cache statistics: {e}") - return {} - - -# Example usage -if __name__ == "__main__": - cache = MoEENECache() - - # Cache an expert configuration - config = ExpertConfiguration( - expert_id=1, - gating_weight=0.7, - quality_weight=0.8, - coherence=0.9, - penalty_weight=0.1, - distortion=0.05, - arity=5.0, - cost_coefficient=0.02, - overhead=0.01, - semantic_vector=[0.5, 0.3, 0.7, 0.2, 0.1, 0.4, 0.6, 0.8, 0.2, 0.3, 0.5, 0.7, 0.1, 0.4], - domain="neural_manifold", - version="1.0.0" - ) - - print("Caching expert configuration...") - cache.cache_expert_config(config) - - # Retrieve it - print("Retrieving expert configuration...") - retrieved = cache.retrieve_expert_config(1) - print(f"Retrieved: {retrieved}") - - # Cache a computation result - entry = MoECacheEntry( - cache_key="eta_moe_12345", - expert_ids=[1, 2, 3], - eta_moe_result=0.85, - i_discarded=0.1, - semantic_vector=[0.5, 0.3, 0.7, 0.2, 0.1, 0.4, 0.6, 0.8, 0.2, 0.3, 0.5, 0.7, 0.1, 0.4], - confidence=0.95, - timestamp=int(time.time()) - ) - - print("Caching computation result...") - cache.cache_computation_result(entry) - - # Retrieve it - print("Retrieving computation result...") - retrieved_entry = cache.retrieve_computation_result("eta_moe_12345") - print(f"Retrieved: {retrieved_entry}") - - # Get statistics - print("Cache statistics:") - stats = cache.get_cache_statistics() - print(json.dumps(stats, indent=2)) diff --git a/4-Infrastructure/infra/neural_delta_gcl_compressor.py b/4-Infrastructure/infra/neural_delta_gcl_compressor.py deleted file mode 100644 index 4269f6b3..00000000 --- a/4-Infrastructure/infra/neural_delta_gcl_compressor.py +++ /dev/null @@ -1,289 +0,0 @@ -#!/usr/bin/env python3 -""" -Neural Delta GCL Compressor (Optional Second Stage) - -VAE-style neural compression layered on top of Delta GCL. -Per canonical lock-in: Delta GCL is lawful base codec, neural layer is learned transport compressor. - -Canonical Field Equation: -q_θ(z | x) = N( μ_θ(x), diag(σ²_θ(x)) ) -z = μ_θ(x) + σ_θ(x) ⊙ ε, ε ~ N(0, I) -x̂ = g_φ(z) -L = D(x, x̂) + β · KL(q_θ(z | x) || N(0, I)) -R_total = R_ΔGCL · R_neural - -Architecture: -raw metadata m → DeltaGCL(m) = x → q_θ(z | x) → z → x̂ = g_φ(z) -→ verify x̂ ≈ x → verify DeltaGCLDecode(x̂) preserves invariant → commit or refuse - -Per AGENTS.md: This is a shim layer - logic in Lean, only JSON marshaling here. -""" - -import sys -from pathlib import Path -# Add project root to Python path -project_root = Path(__file__).parent.parent -sys.path.insert(0, str(project_root)) - -import json -import hashlib -from typing import Dict, Any, Optional, Tuple -from dataclasses import dataclass -import numpy as np - -from infra.delta_gcl_compression_service import DeltaGCLCompressionService, CompressionResult - - -@dataclass -class NeuralCompressionResult: - """Neural compression result""" - latent: np.ndarray # 64-dimensional latent representation - reconstructed_delta_gcl: str - neural_ratio: float - total_ratio: float - verified: bool - verification_error: Optional[str] - - -class NeuralDeltaGCLCompressor: - """ - VAE-style neural compressor for Delta GCL sequences. - - This is an optional second stage that compresses Delta GCL output further. - Per canonical lock-in: Delta GCL remains the lawful base codec. - """ - - def __init__(self, latent_dim: int = 64): - self.latent_dim = latent_dim - self.delta_gcl_service = DeltaGCLCompressionService() - - # Model parameters (placeholder - would be loaded from trained model) - self.encoder_mean = None # μ_θ - self.encoder_std = None # σ_θ - self.decoder_weights = None # g_φ - self.beta = 1e-3 # KL regularization weight - - # Model state - self.is_trained = False - self.model_version = "0.1.0-placeholder" - - def encode(self, delta_gcl: str) -> Tuple[np.ndarray, np.ndarray, np.ndarray]: - """ - Encode Delta GCL sequence to latent representation. - - q_θ(z | x) = N( μ_θ(x), diag(σ²_θ(x)) ) - z = μ_θ(x) + σ_θ(x) ⊙ ε, ε ~ N(0, I) - - Returns: (μ, σ, z) - """ - if not self.is_trained: - # Placeholder: use hash-based encoding until model is trained - hash_bytes = hashlib.sha256(delta_gcl.encode()).digest() - # Convert first 64 bytes to float array - mu = np.frombuffer(hash_bytes[:self.latent_dim * 4], dtype=np.float32) - if len(mu) < self.latent_dim: - mu = np.pad(mu, (0, self.latent_dim - len(mu))) - sigma = np.ones(self.latent_dim) * 0.1 # Fixed std for placeholder - else: - # Actual VAE encoder would be called here - mu, sigma = self._encode_with_model(delta_gcl) - - # Reparameterization trick - epsilon = np.random.randn(self.latent_dim) - z = mu + sigma * epsilon - - return mu, sigma, z - - def _encode_with_model(self, delta_gcl: str) -> Tuple[np.ndarray, np.ndarray]: - """Encode using actual trained VAE encoder (not implemented yet)""" - # This would call the actual neural network - # For now, return placeholder - return np.zeros(self.latent_dim), np.ones(self.latent_dim) - - def decode(self, z: np.ndarray) -> str: - """ - Decode latent representation back to Delta GCL sequence. - - x̂ = g_φ(z) - """ - if not self.is_trained: - # Placeholder: use deterministic hash-based decoding - z_bytes = z.tobytes()[:32] - hash_val = hashlib.sha256(z_bytes).hexdigest() - # Reconstruct Delta GCL format (simplified) - return f"F{hash_val[:8]}" - else: - # Actual VAE decoder would be called here - return self._decode_with_model(z) - - def _decode_with_model(self, z: np.ndarray) -> str: - """Decode using actual trained VAE decoder (not implemented yet)""" - # This would call the actual neural network - return "" - - def compress_with_neural(self, manifest: Dict[str, Any], - manifest_id: str, - verify: bool = True) -> NeuralCompressionResult: - """ - Compress manifest using two-stage process: - 1. Delta GCL (lawful base codec) - 2. Neural compression (learned transport compressor) - - Canonical stack: - raw metadata m → DeltaGCL(m) = x → q_θ(z | x) → z → x̂ = g_φ(z) - → verify x̂ ≈ x → verify DeltaGCLDecode(x̂) preserves invariant → commit or refuse - """ - # Stage 1: Delta GCL compression (lawful base codec) - delta_gcl_result = self.delta_gcl_service.compress_manifest( - manifest, manifest_id, use_delta=True, verify=True - ) - - if not delta_gcl_result.verified: - return NeuralCompressionResult( - latent=np.zeros(self.latent_dim), - reconstructed_delta_gcl="", - neural_ratio=1.0, - total_ratio=delta_gcl_result.stats["compressed_size"] / delta_gcl_result.stats["original_size"], - verified=False, - verification_error=f"Delta GCL verification failed: {delta_gcl_result.verification_error}" - ) - - delta_gcl = delta_gcl_result.delta_gcl - original_size = delta_gcl_result.stats["original_size"] - delta_gcl_size = delta_gcl_result.stats["compressed_size"] - - # Stage 2: Neural compression (learned transport compressor) - mu, sigma, z = self.encode(delta_gcl) - - # Calculate neural compression ratio - latent_size = self.latent_dim * 4 # 4 bytes per float32 - neural_ratio = latent_size / delta_gcl_size - - # Decode latent back to Delta GCL - reconstructed_delta_gcl = self.decode(z) - - # Verification: check x̂ ≈ x - reconstruction_match = self._verify_reconstruction(delta_gcl, reconstructed_delta_gcl) - - # Verification: check DeltaGCLDecode(x̂) preserves invariant - if verify and reconstruction_match: - # Decode reconstructed Delta GCL and check invariants - verified, error = self.delta_gcl_service.verify_compression( - reconstructed_delta_gcl, manifest - ) - else: - verified = False - error = "Reconstruction verification failed" if not reconstruction_match else None - - # Total compression ratio - total_ratio = latent_size / original_size - - return NeuralCompressionResult( - latent=z, - reconstructed_delta_gcl=reconstructed_delta_gcl, - neural_ratio=neural_ratio, - total_ratio=total_ratio, - verified=verified, - verification_error=error - ) - - def _verify_reconstruction(self, original: str, reconstructed: str) -> bool: - """Verify that reconstructed Delta GCL matches original (x̂ ≈ x)""" - if not self.is_trained: - # Placeholder: accept reconstruction for untrained model - return True - - # For trained model, check if reconstructed is close to original - # This would use a similarity metric (e.g., BLEU, edit distance) - # For now, simple equality check - return original == reconstructed - - def compute_loss(self, original: str, reconstructed: str, - mu: np.ndarray, sigma: np.ndarray) -> float: - """ - Compute VAE loss function. - - L = D(x, x̂) + β · KL(q_θ(z | x) || N(0, I)) - """ - # Reconstruction loss D(x, x̂) - reconstruction_loss = self._reconstruction_loss(original, reconstructed) - - # KL divergence: KL(q_θ(z | x) || N(0, I)) - kl_divergence = self._kl_divergence(mu, sigma) - - # Total loss - total_loss = reconstruction_loss + self.beta * kl_divergence - - return total_loss - - def _reconstruction_loss(self, original: str, reconstructed: str) -> float: - """Compute reconstruction loss D(x, x̂)""" - # Simple character-level cross-entropy placeholder - # In practice, would use token-level loss - if not original or not reconstructed: - return 0.0 - - # Simplified: edit distance normalized by length - max_len = max(len(original), len(reconstructed)) - if max_len == 0: - return 0.0 - - # Placeholder: use length difference as loss - return abs(len(original) - len(reconstructed)) / max_len - - def _kl_divergence(self, mu: np.ndarray, sigma: np.ndarray) -> float: - """ - Compute KL divergence: KL(q_θ(z | x) || N(0, I)) - - KL(N(μ, σ²) || N(0, I)) = -0.5 * Σ(1 + log(σ²) - μ² - σ²) - """ - return -0.5 * np.sum(1 + np.log(sigma**2) - mu**2 - sigma**2) - - -# Singleton instance -_neural_instance: Optional[NeuralDeltaGCLCompressor] = None - - -def get_neural_compressor() -> NeuralDeltaGCLCompressor: - """Get singleton neural compressor instance""" - global _neural_instance - if _neural_instance is None: - _neural_instance = NeuralDeltaGCLCompressor() - return _neural_instance - - -if __name__ == "__main__": - # Test neural compression (placeholder mode) - print("=" * 70) - print("NEURAL DELTA GCL COMPRESSION (Placeholder)") - print("=" * 70) - - compressor = NeuralDeltaGCLCompressor() - - test_manifest = { - "layer": "CORE", - "domain": "COMPUTE", - "tier": "FOAM", - "condition": "STABLE", - "metadata": {"compression_ratio": 0.92, "field_phi": 0.85} - } - - print(f"\n[1] Compressing with neural layer...") - result = compressor.compress_with_neural(test_manifest, "neural_test_1", verify=False) - - print(f" Neural ratio: {result.neural_ratio:.4f}") - print(f" Total ratio: {result.total_ratio:.4f}") - print(f" Total reduction: {(1 - result.total_ratio) * 100:.2f}%") - print(f" Verified: {result.verified}") - if result.verification_error: - print(f" Verification Error: {result.verification_error}") - - print(f"\n[2] Model state:") - print(f" Trained: {compressor.is_trained}") - print(f" Version: {compressor.model_version}") - print(f" Latent dim: {compressor.latent_dim}") - - print("\n" + "=" * 70) - print("NEURAL COMPRESSION PLACEHOLDER OPERATIONAL") - print("Note: Actual neural network training required for production use") - print("=" * 70) diff --git a/4-Infrastructure/infra/nixos-setup-cred-server.sh b/4-Infrastructure/infra/nixos-setup-cred-server.sh index 3af7d963..2b2b761e 100644 --- a/4-Infrastructure/infra/nixos-setup-cred-server.sh +++ b/4-Infrastructure/infra/nixos-setup-cred-server.sh @@ -12,10 +12,12 @@ Wants=network-online.target Type=simple User=root WorkingDirectory=/opt/rs-surface -ExecStart=/run/current-system/sw/bin/python3 /opt/rs-surface/credential_server.py --port 8444 --bind 0.0.0.0 +ExecStart=/opt/rs-surface/rs-surface Restart=always RestartSec=5 Environment=RS_CREDENTIAL_CONFIG=/etc/rs-surface/credentials.json +Environment=RS_SURFACE_PORT=8444 +Environment=RS_SURFACE_HOST=0.0.0.0 Environment=RS_CREDENTIAL_SERVER=http://100.101.247.127:8444 Environment=RS_SURFACE_NODE_ID=aws-nixos-node-1 diff --git a/4-Infrastructure/infra/recover_credential_server.sh b/4-Infrastructure/infra/recover_credential_server.sh index 650e4c94..df7e575b 100644 --- a/4-Infrastructure/infra/recover_credential_server.sh +++ b/4-Infrastructure/infra/recover_credential_server.sh @@ -25,12 +25,15 @@ echo "=== Deploying credential server to ${VM_IP} ===" # Create directories $SSH "mkdir -p /opt/rs-surface /etc/rs-surface" -# Copy Python modules -echo "Uploading credential_provider.py..." -$SCP "$REPO_ROOT/4-Infrastructure/infra/credential_provider.py" ${VM_USER}@${VM_IP}:/opt/rs-surface/credential_provider.py - -echo "Uploading credential_server.py..." -$SCP "$REPO_ROOT/4-Infrastructure/infra/credential_server.py" ${VM_USER}@${VM_IP}:/opt/rs-surface/credential_server.py +# Upload rs-surface binary (credential endpoint is served by rs-surface on /credentials) +RS_SURFACE_BIN="$REPO_ROOT/4-Infrastructure/infra/embedded_surface/rs-surface/target/x86_64-unknown-linux-musl/release/rs-surface" +if [ ! -x "$RS_SURFACE_BIN" ]; then + echo "ERROR: rs-surface binary not found at $RS_SURFACE_BIN — build it first with:" + echo " cd $REPO_ROOT/4-Infrastructure/infra/embedded_surface/rs-surface && cargo build --release --target x86_64-unknown-linux-musl" + exit 1 +fi +echo "Uploading rs-surface binary..." +$SCP "$RS_SURFACE_BIN" ${VM_USER}@${VM_IP}:/opt/rs-surface/rs-surface # Copy credentials config CRED_JSON="/tmp/rs-credentials.json" @@ -65,10 +68,12 @@ Wants=network-online.target Type=simple User=root WorkingDirectory=/opt/rs-surface -ExecStart=/usr/bin/python3 /opt/rs-surface/credential_server.py --port 8444 --bind 0.0.0.0 +ExecStart=/opt/rs-surface/rs-surface Restart=always RestartSec=5 Environment=RS_CREDENTIAL_CONFIG=/etc/rs-surface/credentials.json +Environment=RS_SURFACE_PORT=8444 +Environment=RS_SURFACE_HOST=0.0.0.0 [Install] WantedBy=multi-user.target diff --git a/4-Infrastructure/infra/research_engine.py b/4-Infrastructure/infra/research_engine.py deleted file mode 100644 index c420417f..00000000 --- a/4-Infrastructure/infra/research_engine.py +++ /dev/null @@ -1,92 +0,0 @@ -#!/usr/bin/env python3 -""" -research_engine.py — High-level research orchestration - -Combines search (Google/Brave) and deep extraction (Servo) -to provide a sovereign research experience. -""" - -import asyncio -from typing import List, Dict, Any, Optional -from infra.search_adapter import UnifiedSearcher -from infra.servo_fetch_adapter import ServoSwarmInterface -import json -from pathlib import Path - -class ResearchEngine: - def __init__(self, servo_binary: Optional[str] = None): - self.searcher = UnifiedSearcher() - self.servo = ServoSwarmInterface(servo_binary) - self.project_root = Path(__file__).parent.parent - self.lake_path = self.project_root / "data" / "web_lake.jsonl" - - async def deep_research(self, query: str, limit: int = 5) -> Dict[str, Any]: - """ - 1. Search for query - 2. Fetch results in parallel using Servo - 3. Consolidate findings - """ - print(f"Starting deep research for: {query}") - - # 1. Search - search_results = self.searcher.search(query, limit=limit) - if not search_results: - return {"error": "No search results found", "query": query} - - # 2. Fetch in parallel (simulated or real) - tasks = [] - for res in search_results: - print(f"Queueing fetch for: {res.url}") - # We'll use a thread pool or just synchronous loop for now if Servo is sync - # Actually, our Servo adapter is currently synchronous in its execute_task. - # We can run them in threads. - tasks.append(self._fetch_and_process(res)) - - # For now, let's just do them sequentially to avoid overloading the headless browser - # unless we have a robust pool. - results = [] - for task in tasks: - results.append(await task) - - return { - "query": query, - "results": results, - "total_fetched": len(results) - } - - async def _fetch_and_process(self, search_res) -> Dict[str, Any]: - """Fetch a single result and prepare it for the lake.""" - try: - fetch_res = self.servo.fetch(search_res.url, json=True) - content = fetch_res.get("result", {}).get("content", "") - - processed = { - "url": search_res.url, - "title": search_res.title, - "snippet": search_res.snippet, - "content_preview": content[:1000] if content else "", - "full_content": content, - "timestamp": asyncio.get_event_loop().time() - } - - # Auto-ingest to lake - self._ingest(processed) - - return processed - except Exception as e: - return {"url": search_res.url, "error": str(e)} - - def _ingest(self, data: Dict[str, Any]): - """Append to lake.""" - self.lake_path.parent.mkdir(parents=True, exist_ok=True) - with open(self.lake_path, "a", encoding="utf-8") as f: - f.write(json.dumps(data) + "\n") - -if __name__ == "__main__": - # Test script - async def test(): - engine = ResearchEngine() - res = await engine.deep_research("Servo browser performance", limit=2) - print(json.dumps(res, indent=2)) - - # asyncio.run(test()) diff --git a/4-Infrastructure/infra/s3c_iterative_improvement.py b/4-Infrastructure/infra/s3c_iterative_improvement.py deleted file mode 100644 index afa1ab25..00000000 --- a/4-Infrastructure/infra/s3c_iterative_improvement.py +++ /dev/null @@ -1,166 +0,0 @@ -#!/usr/bin/env python3 -""" -S3C Lean Iterative Improvement via Gemma 4 -Loops Gemma 4 to iteratively add theorems and proofs to S3C.lean -""" - -import sys -import json -import sqlite3 -from pathlib import Path -from datetime import datetime -import time - -sys.path.insert(0, str(Path(__file__).parent.parent)) - -# Direct Gemma 4 integration (bypassing GPU duty system for simplicity) -# We'll use a simple loop that generates theorem suggestions - -def read_s3c_lean(): - """Read current S3C.lean file""" - with open("/home/allaun/Documents/Research Stack/0-Core-Formalism/lean/Semantics/Semantics/S3C.lean", 'r') as f: - return f.read() - -def write_s3c_lean(content): - """Write updated S3C.lean file""" - with open("/home/allaun/Documents/Research Stack/0-Core-Formalism/lean/Semantics/Semantics/S3C.lean", 'w') as f: - f.write(content) - -def lake_build(): - """Run lake build to check compilation""" - import subprocess - result = subprocess.run( - ["lake", "build"], - cwd="/home/allaun/Documents/Research Stack/0-Core-Formalism/lean/Semantics", - capture_output=True, - text=True - ) - return result.returncode == 0, result.stdout, result.stderr - -def add_theorems_iteration(current_code, iteration): - """ - Add theorems to S3C.lean based on iteration number - This is a manual implementation since Gemma 4 integration failed - """ - - # Theorems to add (from manual review) - theorems_to_add = { - 1: '''/-- Shell decomposition correctness theorem -/ -theorem shellDecompositionCorrect (n : UInt32) : - let coords := shellDecomposition n - coords.k * coords.k + coords.a = n := by - simp [shellDecomposition] - -''', - 2: '''/-- Mass is intersection form theorem -/ -theorem massIsIntersectionForm (n : UInt32) : - let coords := shellDecomposition n - coords.mass = coords.a * coords.b := by - simp [shellDecomposition] - -''', - 3: '''/-- Emission gate requires contact theorem -/ -theorem emissionGateRequiresContact (sample : UInt32) : - let state := processAudioSample sample - state.emit → state.contact.kappaA ∧ state.contact.kappaC := by - cases state.emit <;> <;> <;> rfl - -''', - 4: '''/-- Bind lawful theorem -/ -theorem s3cAudioBindLawful (sample : UInt32) : - (s3cAudioBind sample).lawful = true := by - rfl - -''', - 5: '''/-- Progressive binding cost non-negative theorem -/ -theorem progressiveBindingCostNonNegative (n : UInt32) : - progressiveBindingCost n ≥ 0 := by - cases n <;> <;> <;> simp [progressiveBindingCost] - -''' - } - - if iteration in theorems_to_add: - # Find the end of the namespace (before "end Semantics.S3C") - end_marker = "end Semantics.S3C" - if end_marker in current_code: - # Insert theorems before the end marker - theorem_code = theorems_to_add[iteration] - updated_code = current_code.replace(end_marker, theorem_code + end_marker) - return updated_code, f"Added theorem iteration {iteration}" - else: - return current_code, f"Could not find end marker in iteration {iteration}" - else: - return current_code, f"No theorems for iteration {iteration}, done" - -def main(): - print("=" * 70) - print("S3C LEAN ITERATIVE IMPROVEMENT") - print("=" * 70) - - current_code = read_s3c_lean() - print(f"Read S3C.lean: {len(current_code)} characters") - - # Initial build check - print("\nInitial lake build check...") - success, stdout, stderr = lake_build() - print(f"Build status: {'PASS' if success else 'FAIL'}") - if not success: - print(f"Error: {stderr[:500]}") - - # Iterative improvement loop - max_iterations = 5 - for i in range(1, max_iterations + 1): - print(f"\n--- Iteration {i}/{max_iterations} ---") - - # Add theorems - updated_code, message = add_theorems_iteration(current_code, i) - print(f"{message}") - - if updated_code == current_code: - print("No changes made, stopping iteration") - break - - # Write updated code - write_s3c_lean(updated_code) - current_code = updated_code - - # Check build - print("Checking lake build...") - success, stdout, stderr = lake_build() - print(f"Build status: {'PASS' if success else 'FAIL'}") - - if success: - print("✅ Iteration {i} successful") - else: - print(f"❌ Iteration {i} failed: {stderr[:200]}") - # Revert on failure - write_s3c_lean(current_code) - print("Reverted changes") - break - - time.sleep(1) # Brief pause between iterations - - # Final build check - print("\n" + "=" * 70) - print("FINAL BUILD CHECK") - print("=" * 70) - success, stdout, stderr = lake_build() - print(f"Final build status: {'PASS' if success else 'FAIL'}") - - if success: - print("✅ All iterations successful, S3C.lean improved") - else: - print("❌ Build failed, check errors") - - # Summary - print(f"\nFinal code length: {len(current_code)} characters") - print("Improvements added:") - print(" - Shell decomposition correctness theorem") - print(" - Mass intersection form theorem") - print(" - Emission gate contact theorem") - print(" - Bind lawful theorem") - print(" - Progressive binding cost non-negative theorem") - -if __name__ == "__main__": - main() diff --git a/4-Infrastructure/infra/s3c_lean_review.py b/4-Infrastructure/infra/s3c_lean_review.py deleted file mode 100644 index eac6caa1..00000000 --- a/4-Infrastructure/infra/s3c_lean_review.py +++ /dev/null @@ -1,86 +0,0 @@ -#!/usr/bin/env python3 -""" -S3C Lean Code Review via Gemma 4 -Submits S3C.lean for code review by Gemma 4 -""" - -import sys -import json -import sqlite3 -from pathlib import Path -from datetime import datetime - -sys.path.insert(0, str(Path(__file__).parent.parent)) - -from infra.gemma_4_integration import GemmaVariant, GemmaTask, GemmaTaskRequest, Gemma4Integration - -def main(): - print("=" * 70) - print("S3C LEAN CODE REVIEW VIA GEMMA 4") - print("=" * 70) - - # Read S3C.lean file - s3c_lean_path = "/home/allaun/Documents/Research Stack/0-Core-Formalism/lean/Semantics/Semantics/S3C.lean" - with open(s3c_lean_path, 'r') as f: - s3c_code = f.read() - - print(f"\nRead S3C.lean: {len(s3c_code)} characters") - - # Initialize Gemma integration - gemma = Gemma4Integration(default_variant=GemmaVariant.E4B) - - # Create review task - task_id = f"s3c_review_{datetime.now().strftime('%Y%m%d_%H%M%S')}" - review_task = GemmaTaskRequest( - task_id=task_id, - task_type=GemmaTask.REASONING, - variant=GemmaVariant.E4B, - input_data={ - "prompt": """Review the following Lean 4 code for S3C manifold processing. Check for: -1. Adherence to AGENTS.md rules (naming conventions, no Float, bind primitive, etc.) -2. Mathematical correctness of shell decomposition n = k^2 + a -3. Correctness of J-score calculation: J(n) = ab*F_m + (a-b)*F_p + -4. Emission gate logic: kappa_A AND kappa_C AND J > 0 -5. Proper use of Q16_16 fixed-point arithmetic -6. Missing theorems or correctness proofs -7. Potential improvements or optimizations - -Provide specific feedback on any issues found.""", - "code": s3c_code, - "file": "S3C.lean", - "enable_thinking": True - }, - enable_thinking=True, - max_tokens=2048, - priority=9 - ) - - print(f"\nSubmitting review task: {task_id}") - submitted_id = gemma.submit_task(review_task) - print(f"Task submitted: {submitted_id}") - - # Execute the task - print(f"\nExecuting review task...") - result = gemma.execute_task(submitted_id) - - print(f"\nReview Result:") - print(json.dumps(result, indent=2)) - - # Save review to file - review_path = f"/home/allaun/Documents/Research Stack/data/s3c_lean_review_{datetime.now().strftime('%Y%m%d_%H%M%S')}.json" - with open(review_path, 'w') as f: - json.dump({ - "task_id": task_id, - "submitted_id": submitted_id, - "result": result, - "timestamp": datetime.now().isoformat() - }, f, indent=2) - - print(f"\nReview saved to: {review_path}") - - print("\n" + "=" * 70) - print("S3C LEAN CODE REVIEW COMPLETE") - print("=" * 70) - -if __name__ == "__main__": - main() diff --git a/4-Infrastructure/infra/search_adapter.py b/4-Infrastructure/infra/search_adapter.py deleted file mode 100644 index 5b3eb620..00000000 --- a/4-Infrastructure/infra/search_adapter.py +++ /dev/null @@ -1,77 +0,0 @@ -#!/usr/bin/env python3 -""" -search_adapter.py — Search implementation for Research Stack - -Integrates multiple search providers (Google Surf, Brave Search) -into a unified interface. -""" - -import os -import json -import subprocess -from typing import List, Dict, Any, Optional - -class SearchResult: - def __init__(self, title: str, url: str, snippet: str): - self.title = title - self.url = url - self.snippet = snippet - - def to_dict(self): - return {"title": self.title, "url": self.url, "snippet": self.snippet} - -class SearchProvider: - def search(self, query: str, limit: int = 5) -> List[SearchResult]: - raise NotImplementedError - -class GoogleSurfProvider(SearchProvider): - def search(self, query: str, limit: int = 5) -> List[SearchResult]: - # Using npx to call the google-surf-mcp tool if possible - # Actually, since it's an MCP server, we might prefer a direct search if it has a CLI - # But for now, we'll simulate or use a fallback if not configured. - print(f"Searching Google Surf for: {query}") - # Placeholder for real integration - return [] - -class BraveSearchProvider(SearchProvider): - def __init__(self, api_key: Optional[str] = None): - self.api_key = api_key or os.getenv("BRAVE_API_KEY") - - def search(self, query: str, limit: int = 5) -> List[SearchResult]: - if not self.api_key or "YOUR_BRAVE_API_KEY" in self.api_key: - return [] - - import requests - url = "https://api.search.brave.com/res/v1/web/search" - headers = { - "Accept": "application/json", - "Accept-Encoding": "gzip", - "X-Subscription-Token": self.api_key - } - params = {"q": query, "count": limit} - - try: - resp = requests.get(url, headers=headers, params=params, timeout=10) - resp.raise_for_status() - data = resp.json() - results = [] - for item in data.get("web", {}).get("results", []): - results.append(SearchResult(item["title"], item["url"], item["description"])) - return results - except Exception as e: - print(f"Brave Search Error: {e}") - return [] - -class UnifiedSearcher: - def __init__(self): - self.providers = [ - BraveSearchProvider(), - GoogleSurfProvider() - ] - - def search(self, query: str, limit: int = 5) -> List[SearchResult]: - for provider in self.providers: - results = provider.search(query, limit) - if results: - return results - return [] diff --git a/4-Infrastructure/infra/servo_fetch_adapter.py b/4-Infrastructure/infra/servo_fetch_adapter.py deleted file mode 100644 index 182864bd..00000000 --- a/4-Infrastructure/infra/servo_fetch_adapter.py +++ /dev/null @@ -1,159 +0,0 @@ -#!/usr/bin/env python3 -""" -servo_fetch_adapter.py — Servo-Fetch implementation for SwarmWebSurface - -Integrates the high-performance, GPU-less Servo browser engine into the -Sovereign Stack's web interaction surface. -""" - -import subprocess -import json -import os -from pathlib import Path -from typing import Dict, Any, Optional, List -import time - -from infra.web_interaction_surface import WebInteractionSurface, WebTask, DutyType - -class ServoFetchSurface(WebInteractionSurface): - """ - Real implementation of WebInteractionSurface using servo-fetch. - """ - - def __init__(self, binary_path: Optional[str] = None): - super().__init__() - # Priority order: explicitly passed path -> env var -> standard tool path -> PATH - self.binary_path = ( - binary_path or - os.getenv("SERVO_FETCH_PATH") or - str(Path(__file__).parent.parent / "tools" / "bin" / "servo-fetch") - ) - - # Verify binary exists or is in PATH - if not Path(self.binary_path).exists() and subprocess.run(["which", "servo-fetch"], capture_output=True).returncode != 0: - print(f"Warning: servo-fetch binary not found at {self.binary_path} or in PATH", file=os.sys.stderr) - - def _execute_servo(self, args: List[str], timeout: int = 60) -> subprocess.CompletedProcess: - """Run the servo-fetch binary.""" - cmd = [self.binary_path] + args - - # Add xvfb-run if requested or if we detect we're headless and it's available - # On some systems, servo-fetch might need a dummy display - if os.getenv("USE_XVFB") == "1" or (not os.getenv("DISPLAY") and subprocess.run(["which", "xvfb-run"], capture_output=True).returncode == 0): - cmd = ["xvfb-run", "--auto-servernum"] + cmd - - try: - return subprocess.run( - cmd, - capture_output=True, - text=True, - timeout=timeout - ) - except subprocess.TimeoutExpired: - raise RuntimeError(f"servo-fetch timed out after {timeout}s") - except Exception as e: - raise RuntimeError(f"Failed to execute servo-fetch: {e}") - - def _simulate_execution(self, task: WebTask) -> Dict[str, Any]: - """ - Override the simulation with real execution. - Note: The base class calls this from execute_task. - """ - url = task.url - options = task.options or {} - - # Map DutyType to servo-fetch CLI arguments - if task.duty_type == DutyType.WEB_NAVIGATION or task.duty_type == DutyType.CONTENT_EXTRACTION: - args = [url] - if options.get("json"): - args.append("--json") - if options.get("selector"): - args.extend(["--selector", options["selector"]]) - if options.get("timeout"): - args.extend(["--timeout", str(options["timeout"])]) - if options.get("settle"): - args.extend(["--settle", str(options["settle"])]) - - proc = self._execute_servo(args) - - if proc.returncode != 0: - return {"error": proc.stderr, "exit_code": proc.returncode} - - # If JSON mode, parse it - if "--json" in args: - try: - return json.loads(proc.stdout) - except: - return {"content": proc.stdout} - return {"content": proc.stdout} - - elif task.duty_type == DutyType.SCREENSHOT_CAPTURE: - temp_path = options.get("path") or f"/tmp/screenshot_{task.task_id}.png" - args = [url, "--screenshot", temp_path] - if options.get("full_page"): - args.append("--full-page") - - proc = self._execute_servo(args) - if proc.returncode != 0: - return {"error": proc.stderr, "exit_code": proc.returncode} - - return { - "screenshot_path": temp_path, - "url": url, - "timestamp": time.time() - } - - elif task.duty_type == DutyType.DISTRIBUTED_CRAWL: - # For crawling, we use the 'crawl' subcommand - args = ["crawl", url] - if options.get("limit"): - args.extend(["--limit", str(options["limit"])]) - if options.get("depth"): - args.extend(["--max-depth", str(options["depth"])]) - if options.get("json"): - args.append("--json") - - proc = self._execute_servo(args) - if proc.returncode != 0: - return {"error": proc.stderr, "exit_code": proc.returncode} - - # servo-fetch crawl returns NDJSON - results = [] - for line in proc.stdout.splitlines(): - if line.strip(): - try: - results.append(json.loads(line)) - except: - results.append({"raw": line}) - - return { - "pages_crawled": len(results), - "results": results - } - - return {"error": f"DutyType {task.duty_type} not implemented in ServoFetchSurface"} - -# ═══════════════════════════════════════════════════════════════════════════ -# Simplified Swarm Interface -# ═══════════════════════════════════════════════════════════════════════════ - -class ServoSwarmInterface: - def __init__(self, binary_path: Optional[str] = None): - self.surface = ServoFetchSurface(binary_path) - - def fetch(self, url: str, **kwargs) -> Dict[str, Any]: - # Ensure url is not in kwargs to avoid double passing - kwargs.pop("url", None) - task_info = self.surface.submit_task(DutyType.CONTENT_EXTRACTION, url, options=kwargs) - return self.surface.execute_task(task_info["task_id"]) - - def screenshot(self, url: str, path: str, **kwargs) -> Dict[str, Any]: - kwargs.pop("url", None) - kwargs["path"] = path - task_info = self.surface.submit_task(DutyType.SCREENSHOT_CAPTURE, url, options=kwargs) - return self.surface.execute_task(task_info["task_id"]) - - def crawl(self, url: str, **kwargs) -> Dict[str, Any]: - kwargs.pop("url", None) - task_info = self.surface.submit_task(DutyType.DISTRIBUTED_CRAWL, url, options=kwargs) - return self.surface.execute_task(task_info["task_id"]) diff --git a/4-Infrastructure/infra/swarm_ene_middleware.py b/4-Infrastructure/infra/swarm_ene_middleware.py deleted file mode 100644 index 0028c40d..00000000 --- a/4-Infrastructure/infra/swarm_ene_middleware.py +++ /dev/null @@ -1,427 +0,0 @@ -#!/usr/bin/env python3 -""" -Swarm API-ENE Middleware - -Middleware layer that hooks the Research/Swarm API into the ENE database -for caching, audit logging, and secure data storage. Intercepts API calls, -checks ENE cache first, and stores results for future use. - -Features: -- Query result caching in ENE database -- Semantic vector-based retrieval -- Access control integration -- Audit logging for all operations -- Automatic cache invalidation on updates -""" - -import hashlib -import json -import sqlite3 -import sys -import time -from dataclasses import dataclass -from pathlib import Path -from typing import Dict, List, Optional, Any - -# Add parent directory to path for imports -sys.path.insert(0, str(Path(__file__).parent.parent)) - -from infra.ene_api import ENEAPIHook, AccessLevel - - -@dataclass -class SwarmQueryCache: - """Cached swarm query result""" - query_hash: str - subjects: List[str] - keywords: Optional[str] - formal_status: Optional[str] - results: List[Dict[str, Any]] - count: int - confidence: float - timestamp: int - semantic_vector: List[float] - ttl: int = 3600 # 1 hour default - - -class SwarmENEMiddleware: - """Middleware for Swarm API integration with ENE""" - - def __init__(self, db_path: str = "/home/allaun/Documents/Research Stack/data/substrate_index.db"): - self.db_path = db_path - self.ene_api = ENEAPIHook() - self._init_middleware_tables() - - def _init_middleware_tables(self): - """Initialize middleware tables in ENE database""" - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - # Query cache table - cursor.execute(""" - CREATE TABLE IF NOT EXISTS swarm_query_cache ( - query_hash TEXT PRIMARY KEY, - subjects TEXT NOT NULL, - keywords TEXT, - formal_status TEXT, - results TEXT NOT NULL, - count INTEGER NOT NULL, - confidence REAL NOT NULL, - semantic_vector TEXT NOT NULL, - created_at INTEGER NOT NULL, - ttl INTEGER NOT NULL, - hit_count INTEGER DEFAULT 0 - ) - """) - - # API operation audit log - cursor.execute(""" - CREATE TABLE IF NOT EXISTS swarm_api_audit ( - id TEXT PRIMARY KEY, - operation TEXT NOT NULL, - query_hash TEXT, - parameters TEXT NOT NULL, - result_cached BOOLEAN NOT NULL, - result_count INTEGER, - execution_time_ms REAL NOT NULL, - created_at INTEGER NOT NULL - ) - """) - - # Semantic index for queries - cursor.execute(""" - CREATE TABLE IF NOT EXISTS swarm_semantic_index ( - id TEXT PRIMARY KEY, - query_hash TEXT NOT NULL, - semantic_vector TEXT NOT NULL, - domain TEXT NOT NULL, - created_at INTEGER NOT NULL, - FOREIGN KEY (query_hash) REFERENCES swarm_query_cache(query_hash) - ) - """) - - conn.commit() - conn.close() - - def _compute_query_hash(self, subjects: List[str], keywords: Optional[str], - formal_status: Optional[str]) -> str: - """Compute hash for query cache key""" - query_str = json.dumps({ - "subjects": sorted(subjects), - "keywords": keywords, - "formal_status": formal_status - }, sort_keys=True) - return hashlib.sha256(query_str.encode()).hexdigest() - - def _derive_semantic_vector(self, subjects: List[str], keywords: Optional[str]) -> List[float]: - """Derive 14D semantic vector from query parameters""" - # Simple heuristic: distribute subjects across semantic axes - # In production, this would use actual semantic embeddings - vector = [0.0] * 14 - - for i, subject in enumerate(subjects): - axis = i % 14 - # Hash subject to get a value in [0,1] - h = hashlib.md5(subject.encode()).hexdigest() - value = int(h[:8], 16) / 0xFFFFFFFF - vector[axis] = value - - if keywords: - # Add keyword influence to first axis - h = hashlib.md5(keywords.encode()).hexdigest() - value = int(h[:8], 16) / 0xFFFFFFFF - vector[0] = (vector[0] + value) / 2 - - return vector - - def check_cache(self, subjects: List[str], keywords: Optional[str], - formal_status: Optional[str]) -> Optional[SwarmQueryCache]: - """Check if query result is cached""" - try: - query_hash = self._compute_query_hash(subjects, keywords, formal_status) - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - cursor.execute(""" - SELECT subjects, keywords, formal_status, results, count, confidence, - semantic_vector, created_at, ttl, hit_count - FROM swarm_query_cache - WHERE query_hash = ? - """, (query_hash,)) - - row = cursor.fetchone() - - if row: - # Check if cache entry is still valid - now = int(time.time()) - created_at = row[7] - ttl = row[8] - - if now - created_at < ttl: - # Increment hit count - cursor.execute(""" - UPDATE swarm_query_cache - SET hit_count = hit_count + 1 - WHERE query_hash = ? - """, (query_hash,)) - conn.commit() - conn.close() - - return SwarmQueryCache( - query_hash=query_hash, - subjects=json.loads(row[0]), - keywords=row[1], - formal_status=row[2], - results=json.loads(row[3]), - count=row[4], - confidence=row[5], - semantic_vector=json.loads(row[6]), - timestamp=created_at, - ttl=ttl - ) - else: - # Cache expired, delete it - cursor.execute("DELETE FROM swarm_query_cache WHERE query_hash = ?", (query_hash,)) - conn.commit() - conn.close() - return None - - conn.close() - return None - except Exception as e: - print(f"Error checking cache: {e}") - return None - - def store_cache(self, subjects: List[str], keywords: Optional[str], formal_status: Optional[str], - results: List[Dict[str, Any]], count: int, confidence: float, - ttl: int = 3600) -> bool: - """Store query result in cache""" - try: - query_hash = self._compute_query_hash(subjects, keywords, formal_status) - semantic_vector = self._derive_semantic_vector(subjects, keywords) - now = int(time.time()) - - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - cursor.execute(""" - INSERT OR REPLACE INTO swarm_query_cache - (query_hash, subjects, keywords, formal_status, results, count, confidence, - semantic_vector, created_at, ttl) - VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?) - """, ( - query_hash, - json.dumps(subjects), - keywords, - formal_status, - json.dumps(results), - count, - confidence, - json.dumps(semantic_vector), - now, - ttl - )) - - # Also store in semantic index - domain = subjects[0] if subjects else "general" - cursor.execute(""" - INSERT OR REPLACE INTO swarm_semantic_index - (id, query_hash, semantic_vector, domain, created_at) - VALUES (?, ?, ?, ?, ?) - """, ( - f"semantic_{query_hash}", - query_hash, - json.dumps(semantic_vector), - domain, - now - )) - - conn.commit() - conn.close() - - # Store sensitive results in ENE secure storage - if results: - self.ene_api.store_sensitive_data( - pkg=f"swarm/query/{query_hash}", - payload=json.dumps(results), - classification=AccessLevel.INTERNAL, - semantic_vector=semantic_vector - ) - - return True - except Exception as e: - print(f"Error storing cache: {e}") - return False - - def log_operation(self, operation: str, parameters: Dict, query_hash: Optional[str], - result_cached: bool, result_count: Optional[int], execution_time_ms: float): - """Log API operation to audit trail""" - try: - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - operation_id = f"op_{operation}_{int(time.time() * 1000000)}" - - cursor.execute(""" - INSERT INTO swarm_api_audit - (id, operation, query_hash, parameters, result_cached, result_count, execution_time_ms, created_at) - VALUES (?, ?, ?, ?, ?, ?, ?, ?) - """, ( - operation_id, - operation, - query_hash, - json.dumps(parameters), - result_cached, - result_count, - execution_time_ms, - int(time.time()) - )) - - conn.commit() - conn.close() - except Exception as e: - print(f"Error logging operation: {e}") - - def invalidate_cache(self, subjects: Optional[List[str]] = None): - """Invalidate cache entries matching criteria""" - try: - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - if subjects: - # Invalidate specific subject queries - for subject in subjects: - cursor.execute(""" - DELETE FROM swarm_query_cache - WHERE subjects LIKE ? - """, (f"%{subject}%",)) - else: - # Invalidate all cache - cursor.execute("DELETE FROM swarm_query_cache") - - conn.commit() - conn.close() - return True - except Exception as e: - print(f"Error invalidating cache: {e}") - return False - - def get_cache_statistics(self) -> Dict[str, Any]: - """Get middleware cache statistics""" - try: - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - # Query cache stats - cursor.execute("SELECT COUNT(*), SUM(hit_count) FROM swarm_query_cache") - cache_count, cache_hits = cursor.fetchone() - - # Audit log stats - cursor.execute("SELECT COUNT(*) FROM swarm_api_audit") - audit_count = cursor.fetchone()[0] - - # Cache hit rate - cursor.execute(""" - SELECT - COUNT(CASE WHEN hit_count > 0 THEN 1 END) * 100.0 / COUNT(*) - FROM swarm_query_cache - """) - hit_rate = cursor.fetchone()[0] or 0 - - conn.close() - - return { - "cached_queries": cache_count or 0, - "cache_hits": cache_hits or 0, - "audit_log_entries": audit_count or 0, - "cache_hit_rate": round(hit_rate, 2) - } - except Exception as e: - print(f"Error getting statistics: {e}") - return {} - - def semantic_search(self, query_vector: List[float], threshold: float = 0.7) -> List[str]: - """Find similar queries using semantic vector similarity""" - try: - conn = sqlite3.connect(self.db_path) - cursor = conn.cursor() - - cursor.execute("SELECT query_hash, semantic_vector FROM swarm_semantic_index") - rows = cursor.fetchall() - - similar_queries = [] - for query_hash, vector_json in rows: - cached_vector = json.loads(vector_json) - # Compute cosine similarity - similarity = self._cosine_similarity(query_vector, cached_vector) - if similarity >= threshold: - similar_queries.append((query_hash, similarity)) - - conn.close() - - # Sort by similarity descending - similar_queries.sort(key=lambda x: x[1], reverse=True) - return [q[0] for q in similar_queries] - except Exception as e: - print(f"Error in semantic search: {e}") - return [] - - def _cosine_similarity(self, v1: List[float], v2: List[float]) -> float: - """Compute cosine similarity between two vectors""" - dot_product = sum(a * b for a, b in zip(v1, v2)) - norm1 = sum(a * a for a in v1) ** 0.5 - norm2 = sum(b * b for b in v2) ** 0.5 - if norm1 == 0 or norm2 == 0: - return 0 - return dot_product / (norm1 * norm2) - - -# Example usage -if __name__ == "__main__": - middleware = SwarmENEMiddleware() - - # Test cache check (should miss) - print("Checking cache (should miss)...") - cached = middleware.check_cache( - subjects=["geometry", "topology"], - keywords="manifold", - formal_status="proven" - ) - print(f"Cached result: {cached}") - - # Store a cache entry - print("Storing cache entry...") - middleware.store_cache( - subjects=["geometry", "topology"], - keywords="manifold", - formal_status="proven", - results=[{"id": 1, "name": "Test Result"}], - count=1, - confidence=0.95, - ttl=3600 - ) - - # Check cache again (should hit) - print("Checking cache again (should hit)...") - cached = middleware.check_cache( - subjects=["geometry", "topology"], - keywords="manifold", - formal_status="proven" - ) - print(f"Cached result: {cached}") - - # Log an operation - print("Logging operation...") - middleware.log_operation( - operation="query", - parameters={"subjects": ["geometry"], "limit": 10}, - query_hash=None, - result_cached=False, - result_count=5, - execution_time_ms=123.45 - ) - - # Get statistics - print("Cache statistics:") - stats = middleware.get_cache_statistics() - print(json.dumps(stats, indent=2)) diff --git a/4-Infrastructure/infra/swarm_execution_layer.py b/4-Infrastructure/infra/swarm_execution_layer.py deleted file mode 100644 index b2cb30fb..00000000 --- a/4-Infrastructure/infra/swarm_execution_layer.py +++ /dev/null @@ -1,321 +0,0 @@ -#!/usr/bin/env python3 -""" -Swarm Execution Layer -Receives swarm recommendations and executes them with GPU acceleration support. -Integrates with EnhancedIntegratedSwarm to get prioritized tasks and execute them. -""" - -import json -import logging -import subprocess -import time -from typing import Dict, Any, List, Optional -from dataclasses import dataclass -from enum import Enum -import os - -logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s') -logger = logging.getLogger(__name__) - - -class TaskStatus(Enum): - PENDING = "pending" - IN_PROGRESS = "in_progress" - COMPLETED = "completed" - FAILED = "failed" - SKIPPED = "skipped" - - -@dataclass -class Task: - id: str - description: str - priority: float - status: TaskStatus - gpu_accelerated: bool - estimated_cycles: int - actual_cycles: int = 0 - result: Optional[Dict[str, Any]] = None - error: Optional[str] = None - - -class SwarmExecutionLayer: - """ - Execution layer that receives swarm recommendations and executes them. - Supports GPU acceleration for suitable tasks and tracks progress. - """ - - def __init__(self): - self.tasks: List[Task] = [] - self.current_cycle = 0 - self.total_cycles = 0 - self.gpu_available = False - self.gpu_type = None - self.execution_log: List[Dict[str, Any]] = [] - - # Check for GPU availability - self._detect_gpu() - - def _detect_gpu(self): - """Detect GPU availability and type.""" - try: - result = subprocess.run( - ["nvidia-smi", "--query-gpu=name", "--format=csv,noheader"], - capture_output=True, - text=True, - timeout=5 - ) - if result.returncode == 0: - self.gpu_available = True - self.gpu_type = result.stdout.strip() - logger.info(f"GPU detected: {self.gpu_type}") - else: - logger.info("No NVIDIA GPU detected") - except (subprocess.TimeoutExpired, FileNotFoundError): - logger.info("GPU detection failed - assuming no GPU available") - self.gpu_available = False - - def receive_swarm_recommendations(self, swarm_analysis: Dict[str, Any]) -> List[Task]: - """ - Receive prioritized tasks from swarm analysis and convert to execution tasks. - - Args: - swarm_analysis: Swarm analysis output containing prioritized tasks - - Returns: - List of Task objects ready for execution - """ - logger.info("Receiving swarm recommendations for execution...") - - # Extract prioritized tasks from swarm analysis - prioritization = swarm_analysis.get("priority_state_tracking", {}) - current_priorities = prioritization.get("current_priorities", []) - - tasks = [] - for idx, priority_item in enumerate(current_priorities): - task = Task( - id=f"task_{idx}", - description=priority_item.get("task", "Unknown task"), - priority=priority_item.get("current_priority", 0.0), - status=TaskStatus.PENDING, - gpu_accelerated=False, - estimated_cycles=5 - ) - tasks.append(task) - - self.tasks = tasks - self.total_cycles = sum(task.estimated_cycles for task in tasks) - - logger.info(f"Received {len(tasks)} tasks, estimated {self.total_cycles} cycles") - self._log_execution_event("recommendations_received", {"task_count": len(tasks), "total_cycles": self.total_cycles}) - - return tasks - - def _should_use_gpu(self, _task_description: str) -> bool: - """GPU suitability decision placeholder. - Per AGENTS.md §6: branching decisions belong in Lean. - TODO(lean-port): Route through bindserver once SwarmExecution endpoint exists. - """ - return False - - def _estimate_task_cycles(self, _priority_item: Dict[str, Any]) -> int: - """Cycle estimation placeholder. - Per AGENTS.md §6: cost computation belongs in Lean. - TODO(lean-port): Route through bindserver once SwarmExecution endpoint exists. - """ - return 5 - - def execute_task(self, task: Task) -> Dict[str, Any]: - """ - Execute a single task with optional GPU acceleration. - - Args: - task: Task to execute - - Returns: - Execution result - """ - logger.info(f"Executing task: {task.description} (GPU: {task.gpu_accelerated})") - task.status = TaskStatus.IN_PROGRESS - - start_time = time.time() - - try: - # Simulate task execution (in real implementation, this would execute actual work) - result = self._execute_task_impl(task) - - task.status = TaskStatus.COMPLETED - task.result = result - task.actual_cycles = task.estimated_cycles # In real implementation, track actual cycles - - elapsed_time = time.time() - start_time - logger.info(f"Task completed in {elapsed_time:.2f}s") - - self._log_execution_event("task_completed", { - "task_id": task.id, - "description": task.description, - "elapsed_time": elapsed_time, - "gpu_used": task.gpu_accelerated - }) - - return result - - except Exception as e: - task.status = TaskStatus.FAILED - task.error = str(e) - logger.error(f"Task failed: {e}") - - self._log_execution_event("task_failed", { - "task_id": task.id, - "error": str(e) - }) - - return {"error": str(e)} - - def _execute_task_impl(self, task: Task) -> Dict[str, Any]: - """ - Implement actual task execution. - This is a placeholder - real implementation would execute actual work. - """ - # Simulate execution time based on estimated cycles - cycle_time = 0.1 # seconds per cycle - execution_time = task.estimated_cycles * cycle_time - - if task.gpu_accelerated and self.gpu_available: - execution_time /= 4 # GPU acceleration - logger.info(f"Using GPU acceleration for task") - - time.sleep(min(execution_time, 2.0)) # Cap at 2 seconds for demo - - return { - "status": "completed", - "cycles_used": task.estimated_cycles, - "gpu_used": task.gpu_accelerated, - "result": f"Task '{task.description}' executed successfully" - } - - def execute_all_tasks(self) -> Dict[str, Any]: - """ - Execute all tasks in priority order. - - Returns: - Execution summary - """ - logger.info(f"Starting execution of {len(self.tasks)} tasks...") - - start_time = time.time() - completed = 0 - failed = 0 - - # Sort tasks by priority (highest first) - sorted_tasks = sorted(self.tasks, key=lambda t: t.priority, reverse=True) - - for task in sorted_tasks: - self.current_cycle += 1 - - result = self.execute_task(task) - - if task.status == TaskStatus.COMPLETED: - completed += 1 - elif task.status == TaskStatus.FAILED: - failed += 1 - - # Check for swarm re-priorization every 5 cycles - if self.current_cycle % 5 == 0: - logger.info(f"Cycle {self.current_cycle}/{self.total_cycles} - Checking for swarm re-prioritization...") - # In real implementation, would query swarm for updated priorities - - elapsed_time = time.time() - start_time - - summary = { - "total_tasks": len(self.tasks), - "completed": completed, - "failed": failed, - "total_cycles": self.current_cycle, - "elapsed_time_seconds": elapsed_time, - "gpu_available": self.gpu_available, - "gpu_type": self.gpu_type - } - - self._log_execution_event("execution_completed", summary) - - logger.info(f"Execution completed: {completed}/{len(self.tasks)} tasks successful in {elapsed_time:.2f}s") - - return summary - - def _log_execution_event(self, event_type: str, data: Dict[str, Any]): - """Log execution events for swarm monitoring.""" - event = { - "timestamp": time.time(), - "cycle": self.current_cycle, - "event_type": event_type, - "data": data - } - self.execution_log.append(event) - - def get_execution_status(self) -> Dict[str, Any]: - """Get current execution status.""" - return { - "current_cycle": self.current_cycle, - "total_cycles": self.total_cycles, - "tasks_pending": sum(1 for t in self.tasks if t.status == TaskStatus.PENDING), - "tasks_in_progress": sum(1 for t in self.tasks if t.status == TaskStatus.IN_PROGRESS), - "tasks_completed": sum(1 for t in self.tasks if t.status == TaskStatus.COMPLETED), - "tasks_failed": sum(1 for t in self.tasks if t.status == TaskStatus.FAILED), - "gpu_available": self.gpu_available, - "gpu_type": self.gpu_type - } - - def save_execution_log(self, filepath: str): - """Save execution log to file.""" - with open(filepath, 'w') as f: - json.dump(self.execution_log, f, indent=2) - logger.info(f"Execution log saved to {filepath}") - - -if __name__ == "__main__": - # Demo execution layer - from lean_unified_shim import LeanUnifiedShim - from enhanced_integrated_swarm import EnhancedIntegratedSwarm - - logger.info("Initializing Swarm Execution Layer") - - # Get swarm recommendations - swarm = EnhancedIntegratedSwarm() - prioritization = swarm.perform_self_prioritization() - - # Initialize execution layer - executor = SwarmExecutionLayer() - - # Receive and execute tasks - tasks = executor.receive_swarm_recommendations(prioritization) - - print("\n" + "=" * 60) - print("SWARM EXECUTION LAYER - TASK QUEUE") - print("=" * 60) - for task in tasks: - print(f" [{task.id}] {task.description}") - print(f" Priority: {task.priority:.2f}") - print(f" GPU Accelerated: {task.gpu_accelerated}") - print(f" Estimated Cycles: {task.estimated_cycles}") - print(f" Status: {task.status.value}") - - print("\n" + "=" * 60) - print("EXECUTING TASKS") - print("=" * 60) - - summary = executor.execute_all_tasks() - - print("\n" + "=" * 60) - print("EXECUTION SUMMARY") - print("=" * 60) - print(f" Total Tasks: {summary['total_tasks']}") - print(f" Completed: {summary['completed']}") - print(f" Failed: {summary['failed']}") - print(f" Total Cycles: {summary['total_cycles']}") - print(f" Elapsed Time: {summary['elapsed_time_seconds']:.2f}s") - print(f" GPU Available: {summary['gpu_available']}") - print(f" GPU Type: {summary['gpu_type']}") - - # Save execution log - executor.save_execution_log("execution_log.json") diff --git a/4-Infrastructure/infra/topological_engine_client.py b/4-Infrastructure/infra/topological_engine_client.py deleted file mode 100644 index 0dc6b3fb..00000000 --- a/4-Infrastructure/infra/topological_engine_client.py +++ /dev/null @@ -1,220 +0,0 @@ -#!/usr/bin/env python3 -""" -Topological Engine Client — Private connector to NoDupeLabs Obsidian + Neo4j graph. - -Provides: -- Obsidian vault read/write/search -- Neo4j Cypher execution, road upserts, and Obsidian link imports -- Health checks - -All endpoints require TOPOLOGICAL_ENGINE_TOKEN (Bearer auth). -""" - -import json -import os -from typing import Dict, List, Any, Optional -from urllib.parse import quote - -import requests - - -class TopologicalEngineClient: - """Client for the private topological engine (NoDupeLabs).""" - - def __init__( - self, - base_url: str = None, - token: str = None, - timeout: int = 30 - ): - self.base_url = (base_url or os.getenv( - "TOPOLOGICAL_ENGINE_URL", "http://localhost:3000" - )).rstrip("/") - self.token = token or os.getenv("TOPOLOGICAL_ENGINE_TOKEN", "") - self.timeout = timeout - self._session = requests.Session() - if self.token: - self._session.headers["Authorization"] = f"Bearer {self.token}" - self._session.headers["Content-Type"] = "application/json" - - # ───────────────────────────────────────────────────────────────────────── - # Health - # ───────────────────────────────────────────────────────────────────────── - - def health(self) -> Dict[str, Any]: - """Ping the topological engine health endpoint.""" - url = f"{self.base_url}/topology/health" - try: - resp = self._session.get(url, timeout=self.timeout) - resp.raise_for_status() - return resp.json() - except requests.RequestException as e: - return {"error": str(e), "healthy": False} - - # ───────────────────────────────────────────────────────────────────────── - # Obsidian - # ───────────────────────────────────────────────────────────────────────── - - def write_obsidian_note( - self, - title: str = None, - folder: str = "Research Stack", - body: str = "", - path: str = None, - metadata: Optional[Dict[str, Any]] = None - ) -> Dict[str, Any]: - """Write or overwrite an Obsidian note. - - Uses 'path' if provided (legacy compat), otherwise constructs from title+folder. - """ - url = f"{self.base_url}/topology/obsidian/write-note" - payload = { - "folder": folder, - "body": body, - } - if title: - payload["title"] = title - if path: - # Router uses title+folder; simulate by extracting title from path - payload["title"] = path.replace(".md", "").split("/")[-1] - if "/" in path: - payload["folder"] = "/".join(path.replace(".md", "").split("/")[:-1]) - if metadata: - payload.update({k: v for k, v in metadata.items() if k not in payload}) - try: - resp = self._session.post(url, json=payload, timeout=self.timeout) - resp.raise_for_status() - return resp.json() - except requests.RequestException as e: - return {"error": str(e)} - - def read_obsidian_note(self, path: str) -> Dict[str, Any]: - """Read an Obsidian note by vault-relative path.""" - encoded = quote(path, safe="") - url = f"{self.base_url}/topology/obsidian/read-note?path={encoded}" - try: - resp = self._session.get(url, timeout=self.timeout) - resp.raise_for_status() - return resp.json() - except requests.RequestException as e: - return {"error": str(e)} - - def search_obsidian(self, query: str, limit: int = 20) -> Dict[str, Any]: - """Full-text search across the Obsidian vault.""" - url = f"{self.base_url}/topology/obsidian/search" - payload = {"query": query, "limit": limit} - try: - resp = self._session.post(url, json=payload, timeout=self.timeout) - resp.raise_for_status() - return resp.json() - except requests.RequestException as e: - return {"error": str(e)} - - # ───────────────────────────────────────────────────────────────────────── - # Neo4j — simple pairwise road upsert (matches API exactly) - # ───────────────────────────────────────────────────────────────────────── - - def neo4j_upsert_road( - self, - source_id: str, - source_label: str, - target_id: str, - target_label: str, - relation: str, - authority_scope: str = "candidate", - outcome: str = "hold", - torsion: float = 0, - coherence: float = 1.0, - provenance_hash: str = None, - source_of_truth: str = None, - quarantine_status: str = "none" - ) -> Dict[str, Any]: - """Upsert a simple pairwise road (2 nodes + 1 relationship).""" - url = f"{self.base_url}/topology/neo4j/upsert-road" - payload = { - "sourceId": source_id, - "sourceLabel": source_label, - "targetId": target_id, - "targetLabel": target_label, - "relation": relation, - "authority_scope": authority_scope, - "outcome": outcome, - "torsion": torsion, - "coherence": coherence, - "provenance_hash": provenance_hash or "", - "source_of_truth": source_of_truth or "Neo4j candidate; Graph.lean audit required", - "quarantine_status": quarantine_status, - } - try: - resp = self._session.post(url, json=payload, timeout=self.timeout) - resp.raise_for_status() - return resp.json() - except requests.RequestException as e: - return {"error": str(e)} - - # ───────────────────────────────────────────────────────────────────────── - # Neo4j — arbitrary Cypher (for complex batch structures) - # ───────────────────────────────────────────────────────────────────────── - - def neo4j_cypher( - self, - query: str, - parameters: Optional[Dict[str, Any]] = None, - read_only: bool = False - ) -> Dict[str, Any]: - """Run an arbitrary Cypher query.""" - url = f"{self.base_url}/topology/neo4j/cypher" - payload = {"cypher": query, "params": parameters or {}, "readOnly": read_only} - try: - resp = self._session.post(url, json=payload, timeout=self.timeout) - resp.raise_for_status() - return resp.json() - except requests.RequestException as e: - return {"error": str(e)} - - def neo4j_import_obsidian_links( - self, - vault_path: Optional[str] = None, - limit: int = 1000 - ) -> Dict[str, Any]: - """Trigger import of Obsidian [[wiki-links]] into Neo4j.""" - url = f"{self.base_url}/topology/neo4j/import-obsidian-links" - payload = {"limit": limit} - if vault_path: - payload["vault_path"] = vault_path - try: - resp = self._session.post(url, json=payload, timeout=self.timeout) - resp.raise_for_status() - return resp.json() - except requests.RequestException as e: - return {"error": str(e)} - - # ───────────────────────────────────────────────────────────────────────── - # Batch helpers - # ───────────────────────────────────────────────────────────────────────── - - def batch_upsert_roads(self, roads: List[Dict[str, Any]]) -> List[Dict[str, Any]]: - """Upsert multiple simple roads, returning results per road.""" - results = [] - for road in roads: - results.append(self.neo4j_upsert_road(**road)) - return results - - def batch_cypher(self, statements: List[Dict[str, Any]]) -> List[Dict[str, Any]]: - """Run multiple Cypher statements.""" - results = [] - for stmt in statements: - results.append(self.neo4j_cypher( - query=stmt["query"], - parameters=stmt.get("parameters", {}), - read_only=stmt.get("read_only", False) - )) - return results - - def ensure_indexes(self) -> Dict[str, Any]: - """Idempotent index/constraints for Research Stack graph.""" - statements = [ - {"query": "CREATE INDEX rs_node_id IF NOT EXISTS FOR (n:Node) ON (n.id)", "read_only": False}, - {"query": "CREATE INDEX rs_obsidian_id IF NOT EXISTS FOR (n:ObsidianNote) ON (n.id)", "read_only": False}, - ] - return {"statements": statements, "results": self.batch_cypher(statements)} diff --git a/4-Infrastructure/infra/topological_storage_delta_gcl.py b/4-Infrastructure/infra/topological_storage_delta_gcl.py deleted file mode 100644 index 54fa18af..00000000 --- a/4-Infrastructure/infra/topological_storage_delta_gcl.py +++ /dev/null @@ -1,273 +0,0 @@ -#!/usr/bin/env python3 -""" -Topological Storage Delta GCL Integration - -Integrates Delta GCL compression with topological storage manifests. -Compresses metadata before storing to Google Drive topological storage. - -Features: -- Delta GCL compression for topological manifests -- Previous state tracking for delta encoding -- Integration with Rclone for storage operations -- Compression statistics tracking - -Per AGENTS.md: This is a shim layer - logic in Lean, only JSON marshaling here. -""" - -import sys -from pathlib import Path -# Add project root to Python path -project_root = Path(__file__).parent.parent -sys.path.insert(0, str(project_root)) - -import json -import hashlib -from typing import Dict, Any, List, Optional -from dataclasses import dataclass -from datetime import datetime - -from infra.delta_gcl_compression_service import DeltaGCLCompressionService, CompressionResult - - -@dataclass -class TopologicalManifest: - """Topological storage manifest.""" - manifest_id: str - version: int - timestamp: float - topology_type: str # "manifold", "connectome", "resource", etc. - data: Dict[str, Any] - compression_metadata: Optional[Dict[str, Any]] = None - - -@dataclass -class StoredManifest: - """Manifest stored in topological storage with compression.""" - manifest_id: str - version: int - compressed_data: str # Delta GCL compressed - original_size: int - compressed_size: int - stored_at: float - storage_path: str - - -class TopologicalStorageDeltaGCL: - """ - Delta GCL compression for topological storage manifests. - - Compresses manifests before storing to Google Drive topological storage. - Maintains previous state for delta encoding. - """ - - def __init__(self, storage_path: str = "gdrive:topological_storage/manifests"): - self.storage_path = storage_path - self.compression_service = DeltaGCLCompressionService() - self.previous_manifests: Dict[str, TopologicalManifest] = {} - self.stored_manifests: Dict[str, StoredManifest] = {} - - def compress_manifest(self, manifest: TopologicalManifest) -> CompressionResult: - """Compress manifest using Delta GCL.""" - # Convert to PTOS format for compression - ptos_manifest = { - "layer": manifest.topology_type, - "domain": "TOPOLOGICAL", - "tier": "STORAGE", - "condition": "ACTIVE", - "metadata": manifest.data - } - - # Check if we have previous version for delta encoding - previous = None - use_delta = False - if manifest.manifest_id in self.previous_manifests: - previous = self.previous_manifests[manifest.manifest_id] - use_delta = True - - # Compress - result = self.compression_service.compress_manifest( - ptos_manifest, - f"topo_{manifest.manifest_id}", - use_delta=use_delta - ) - - # Update previous manifest - self.previous_manifests[manifest.manifest_id] = manifest - - return result - - def store_manifest(self, manifest: TopologicalManifest, - storage_path: Optional[str] = None) -> StoredManifest: - """ - Store manifest to topological storage with Delta GCL compression. - - In production, this would call Rclone to upload to Google Drive. - For now, simulates storage. - """ - # Compress manifest - compression_result = self.compress_manifest(manifest) - - # Calculate sizes - original_size = len(json.dumps(manifest.data)) - compressed_size = len(compression_result.delta_gcl) - - # Determine storage path - path = storage_path or self.storage_path - full_path = f"{path}/{manifest.manifest_id}_v{manifest.version}.delta_gcl" - - # Create stored manifest record - stored = StoredManifest( - manifest_id=manifest.manifest_id, - version=manifest.version, - compressed_data=compression_result.delta_gcl, - original_size=original_size, - compressed_size=compressed_size, - stored_at=datetime.now().timestamp(), - storage_path=full_path - ) - - # Store in local cache (in production: upload via Rclone) - self.stored_manifests[manifest.manifest_id] = stored - - # Add compression metadata to manifest - manifest.compression_metadata = { - "original_size": original_size, - "compressed_size": compressed_size, - "reduction": original_size - compressed_size, - "reduction_percent": compression_result.stats["reduction_percent"], - "compression_method": "delta_gcl", - "use_delta": compression_result.stats["use_delta"] - } - - return stored - - def retrieve_manifest(self, manifest_id: str) -> Optional[TopologicalManifest]: - """ - Retrieve manifest from topological storage. - - In production, this would call Rclone to download from Google Drive. - For now, retrieves from local cache. - """ - if manifest_id not in self.stored_manifests: - return None - - stored = self.stored_manifests[manifest_id] - - # In production: decompress using Delta GCL - # For now: return placeholder indicating compression - return TopologicalManifest( - manifest_id=stored.manifest_id, - version=stored.version, - timestamp=stored.stored_at, - topology_type="retrieved", - data={"compressed": True, "path": stored.storage_path}, - compression_metadata={ - "original_size": stored.original_size, - "compressed_size": stored.compressed_size, - "reduction": stored.original_size - stored.compressed_size - } - ) - - def get_compression_stats(self) -> Dict[str, Any]: - """Get aggregate compression statistics.""" - if not self.stored_manifests: - return {"total_manifests": 0} - - total_original = sum(m.original_size for m in self.stored_manifests.values()) - total_compressed = sum(m.compressed_size for m in self.stored_manifests.values()) - total_reduction = total_original - total_compressed - avg_reduction = total_reduction / len(self.stored_manifests) if self.stored_manifests else 0 - - return { - "total_manifests": len(self.stored_manifests), - "total_original_size": total_original, - "total_compressed_size": total_compressed, - "total_reduction": total_reduction, - "average_reduction_percent": (avg_reduction / total_original * 100) if total_original > 0 else 0 - } - - def list_manifests(self) -> List[str]: - """List all stored manifest IDs.""" - return list(self.stored_manifests.keys()) - - -# Singleton instance -_storage_instance: Optional[TopologicalStorageDeltaGCL] = None - - -def get_topological_storage() -> TopologicalStorageDeltaGCL: - """Get singleton topological storage instance.""" - global _storage_instance - if _storage_instance is None: - _storage_instance = TopologicalStorageDeltaGCL() - return _storage_instance - - -def store_topological_manifest(manifest: TopologicalManifest) -> StoredManifest: - """Convenience function to store manifest.""" - storage = get_topological_storage() - return storage.store_manifest(manifest) - - -if __name__ == "__main__": - # Test topological storage with Delta GCL - print("=" * 70) - print("TOPOLOGICAL STORAGE DELTA GCL INTEGRATION") - print("=" * 70) - - storage = TopologicalStorageDeltaGCL() - - # Create test manifest - manifest1 = TopologicalManifest( - manifest_id="manifold_001", - version=1, - timestamp=datetime.now().timestamp(), - topology_type="manifold", - data={ - "vertices": 1000, - "edges": 2500, - "dimension": 14, - "embedding": "high_dim" - } - ) - - print("\n[1] Storing first manifest (full encoding)...") - stored1 = storage.store_manifest(manifest1) - print(f" Manifest ID: {stored1.manifest_id}") - print(f" Storage Path: {stored1.storage_path}") - print(f" Compression: {stored1.original_size} → {stored1.compressed_size} bytes") - print(f" Reduction: {stored1.original_size - stored1.compressed_size} bytes") - - # Create second version (delta encoding) - manifest2 = TopologicalManifest( - manifest_id="manifold_001", - version=2, - timestamp=datetime.now().timestamp(), - topology_type="manifold", - data={ - "vertices": 1005, # Small change - "edges": 2510, - "dimension": 14, - "embedding": "high_dim" - } - ) - - print("\n[2] Storing second manifest (delta encoding)...") - stored2 = storage.store_manifest(manifest2) - print(f" Manifest ID: {stored2.manifest_id}") - print(f" Version: {stored2.version}") - print(f" Compression: {stored2.original_size} → {stored2.compressed_size} bytes") - print(f" Reduction: {stored2.original_size - stored2.compressed_size} bytes") - - # Get aggregate stats - print("\n[3] Aggregate compression statistics...") - stats = storage.get_compression_stats() - print(f" Total Manifests: {stats['total_manifests']}") - print(f" Total Original: {stats['total_original_size']} bytes") - print(f" Total Compressed: {stats['total_compressed_size']} bytes") - print(f" Total Reduction: {stats['total_reduction']} bytes") - print(f" Avg Reduction: {stats['average_reduction_percent']:.2f}%") - - print("\n" + "=" * 70) - print("TOPOLOGICAL STORAGE DELTA GCL OPERATIONAL") - print("=" * 70) diff --git a/4-Infrastructure/infra/topology_controller.py b/4-Infrastructure/infra/topology_controller.py deleted file mode 100644 index 942e0fb0..00000000 --- a/4-Infrastructure/infra/topology_controller.py +++ /dev/null @@ -1,419 +0,0 @@ -#!/usr/bin/env python3 -""" -topology_controller.py — Geometric Topology Controller - -Python extraction of TopologyResilience.lean. - -Principles: -- The topology is a manifold, not a graph. -- Network segments are charts on the manifold. -- Traffic flows along geodesics (minimal curvature paths). -- Load is an energy density field. -- Services are placed where the Laplacian is minimized. -- Quorum is geometric coverage, not node counting. -- Resiliency = k-connected atlas + dynamic reconfiguration. - -Per AGENTS.md §6: Python may only serialize, spawn, wrap, display. -All cost/invariant/branching decisions are extracted from Lean. -""" - -import json -import sys -import time -import hashlib -import threading -from pathlib import Path -from dataclasses import dataclass, field, asdict -from typing import Dict, List, Optional, Tuple, Callable -from enum import Enum - - -# ═══════════════════════════════════════════════════════════════════════════ -# §0 Q16.16 Fixed-Point (Python extraction) -# ═══════════════════════════════════════════════════════════════════════════ - -class Q16_16: - """Fixed-point Q16.16 using Python int (backend: UInt32 semantics).""" - def __init__(self, val: int): - self.val = val & 0xFFFFFFFF - - @staticmethod - def zero() -> "Q16_16": - return Q16_16(0) - - @staticmethod - def one() -> "Q16_16": - return Q16_16(0x00010000) - - @staticmethod - def from_nat(n: int) -> "Q16_16": - return Q16_16((n * 65536) & 0xFFFFFFFF) - - @staticmethod - def from_float(f: float) -> "Q16_16": - return Q16_16(int(f * 65536) & 0xFFFFFFFF) - - def to_float(self) -> float: - v = self.val if self.val < 0x80000000 else self.val - 0x100000000 - return v / 65536.0 - - def __add__(self, other: "Q16_16") -> "Q16_16": - return Q16_16((self.val + other.val) & 0xFFFFFFFF) - - def __sub__(self, other: "Q16_16") -> "Q16_16": - return Q16_16((self.val - other.val) & 0xFFFFFFFF) - - def __mul__(self, other: "Q16_16") -> "Q16_16": - prod = (self.val * other.val) >> 16 - return Q16_16(prod & 0xFFFFFFFF) - - def __truediv__(self, other: "Q16_16") -> "Q16_16": - if other.val == 0: - return Q16_16(0xFFFFFFFF) - return Q16_16(((self.val << 16) // other.val) & 0xFFFFFFFF) - - def __lt__(self, other: "Q16_16") -> bool: - a = self.val if self.val < 0x80000000 else self.val - 0x100000000 - b = other.val if other.val < 0x80000000 else other.val - 0x100000000 - return a < b - - def __gt__(self, other: "Q16_16") -> bool: - return other < self - - def __le__(self, other: "Q16_16") -> bool: - return self == other or self < other - - def __ge__(self, other: "Q16_16") -> bool: - return other <= self - - def __eq__(self, other: object) -> bool: - if not isinstance(other, Q16_16): - return False - return self.val == other.val - - def to_dict(self) -> dict: - return {"val": self.val, "float": self.to_float()} - - @staticmethod - def from_dict(d: dict) -> "Q16_16": - return Q16_16(d["val"]) - - -# ═══════════════════════════════════════════════════════════════════════════ -# §1 Network Segment — Chart with capacity -# ═══════════════════════════════════════════════════════════════════════════ - -@dataclass -class NetworkSegment: - """A network segment is a chart on the manifold with capacity constraints.""" - point_id: str - dimension: int = 2 - center_coords: List[Q16_16] = field(default_factory=list) - capacity_qps: Q16_16 = field(default_factory=Q16_16.zero) - latency_ms: Q16_16 = field(default_factory=Q16_16.zero) - throughput_mbps: Q16_16 = field(default_factory=Q16_16.zero) - current_load: Q16_16 = field(default_factory=Q16_16.zero) - healthy: bool = True - - def utilization(self) -> Q16_16: - if self.capacity_qps == Q16_16.zero(): - return Q16_16.one() - return self.current_load / self.capacity_qps - - def is_overloaded(self) -> bool: - return self.utilization() > Q16_16(52428) # 0.8 in Q16.16 - - def traversal_cost(self) -> Q16_16: - return self.latency_ms + (self.current_load * Q16_16.from_nat(2)) - - def to_dict(self) -> dict: - return { - "point_id": self.point_id, - "dimension": self.dimension, - "center_coords": [c.to_dict() for c in self.center_coords], - "capacity_qps": self.capacity_qps.to_dict(), - "latency_ms": self.latency_ms.to_dict(), - "throughput_mbps": self.throughput_mbps.to_dict(), - "current_load": self.current_load.to_dict(), - "healthy": self.healthy, - } - - @staticmethod - def from_dict(d: dict) -> "NetworkSegment": - return NetworkSegment( - point_id=d["point_id"], - dimension=d["dimension"], - center_coords=[Q16_16.from_dict(c) for c in d.get("center_coords", [])], - capacity_qps=Q16_16.from_dict(d.get("capacity_qps", {"val": 0})), - latency_ms=Q16_16.from_dict(d.get("latency_ms", {"val": 0})), - throughput_mbps=Q16_16.from_dict(d.get("throughput_mbps", {"val": 0})), - current_load=Q16_16.from_dict(d.get("current_load", {"val": 0})), - healthy=d.get("healthy", True), - ) - - -# ═══════════════════════════════════════════════════════════════════════════ -# §2 Topology Controller — Manifold-aware orchestrator -# ═══════════════════════════════════════════════════════════════════════════ - -@dataclass -class PingProbe: - source_id: str - target_id: str - timestamp: float - ttl: int - accumulated_cost: Q16_16 - - def to_dict(self) -> dict: - return { - "source_id": self.source_id, - "target_id": self.target_id, - "timestamp": self.timestamp, - "ttl": self.ttl, - "accumulated_cost": self.accumulated_cost.to_dict(), - } - - -class TopologyController: - """ - Geometric topology controller. - - Manages the manifold as a collection of overlapping network segments. - All routing, load balancing, and service placement is manifold-native. - """ - - def __init__(self, controller_id: str): - self.controller_id = controller_id - self.segments: Dict[str, NetworkSegment] = {} - self._lock = threading.RLock() - self._running = False - - # ───────────────────────────────────────────────────────────────────── - # Segment Management - # ───────────────────────────────────────────────────────────────────── - - def register_segment(self, seg: NetworkSegment): - with self._lock: - self.segments[seg.point_id] = seg - self._log("segment_register", f"{seg.point_id} load={seg.current_load.to_float():.2f}") - - def unregister_segment(self, point_id: str): - with self._lock: - if point_id in self.segments: - del self.segments[point_id] - self._log("segment_unregister", point_id) - - def update_segment(self, point_id: str, load: Optional[Q16_16] = None, - healthy: Optional[bool] = None): - with self._lock: - if point_id not in self.segments: - return - seg = self.segments[point_id] - if load is not None: - seg.current_load = load - if healthy is not None: - seg.healthy = healthy - - def get_healthy_segments(self) -> List[NetworkSegment]: - with self._lock: - return [s for s in self.segments.values() if s.healthy] - - # ───────────────────────────────────────────────────────────────────── - # Load Balancing — Geodesic routing - # ───────────────────────────────────────────────────────────────────── - - def pick_best_segment(self, candidates: Optional[List[NetworkSegment]] = None) -> Optional[NetworkSegment]: - """Greedy geodesic: pick segment with minimal traversal cost.""" - segs = candidates or self.get_healthy_segments() - if not segs: - return None - return min(segs, key=lambda s: s.traversal_cost().val) - - def route_to(self, target_id: str) -> Optional[NetworkSegment]: - """Route to a specific segment. Returns None if unreachable.""" - with self._lock: - seg = self.segments.get(target_id) - if seg and seg.healthy: - return seg - return None - - def path_cost(self, point_ids: List[str]) -> Q16_16: - """Total traversal cost across a path of segments.""" - total = Q16_16.zero() - for pid in point_ids: - seg = self.segments.get(pid) - if seg: - total = total + seg.traversal_cost() - return total - - # ───────────────────────────────────────────────────────────────────── - # Bottleneck Detection - # ───────────────────────────────────────────────────────────────────── - - def detect_bottlenecks(self) -> List[NetworkSegment]: - """Segments with load above neighborhood mean.""" - healthy = self.get_healthy_segments() - if len(healthy) <= 1: - return [] - mean_load = sum(s.current_load.val for s in healthy) / len(healthy) - return [s for s in healthy if s.current_load.val > mean_load] - - def detect_overloaded(self) -> List[NetworkSegment]: - """Segments with utilization > 0.8.""" - return [s for s in self.get_healthy_segments() if s.is_overloaded()] - - # ───────────────────────────────────────────────────────────────────── - # Service Placement - # ───────────────────────────────────────────────────────────────────── - - def place_service(self, service_name: str, - candidates: Optional[List[NetworkSegment]] = None) -> Optional[NetworkSegment]: - """Place a service on the best available segment.""" - segs = candidates or self.get_healthy_segments() - if not segs: - return None - # Score = load + latency + health penalty - def score(seg: NetworkSegment) -> int: - s = seg.current_load.val + seg.latency_ms.val - if not seg.healthy: - s += 100 * 65536 - return s - return min(segs, key=score) - - # ───────────────────────────────────────────────────────────────────── - # Ping Protocol — Probe along geodesics - # ───────────────────────────────────────────────────────────────────── - - def process_ping(self, probe: PingProbe, seg: NetworkSegment) -> Tuple[Q16_16, PingProbe, bool]: - """Process a ping probe at a segment. Returns (load, updated_probe, should_forward).""" - new_cost = probe.accumulated_cost + seg.traversal_cost() - response = seg.current_load - should_forward = seg.healthy and probe.ttl > 0 - updated = PingProbe( - source_id=probe.source_id, - target_id=seg.point_id, - timestamp=time.time(), - ttl=probe.ttl - 1, - accumulated_cost=new_cost, - ) - return response, updated, should_forward - - def flood_ping(self, source_id: str) -> List[Tuple[str, Q16_16]]: - """Flood ping from source to all reachable segments.""" - source = self.segments.get(source_id) - if not source: - return [] - init_probe = PingProbe( - source_id=source_id, - target_id="", - timestamp=time.time(), - ttl=5, - accumulated_cost=Q16_16.zero(), - ) - results = [] - for seg in self.segments.values(): - if seg.healthy and seg.point_id != source_id: - _, _, reachable = self.process_ping(init_probe, seg) - if reachable: - results.append((seg.point_id, seg.traversal_cost())) - return results - - # ───────────────────────────────────────────────────────────────────── - # Quorum & Resilience - # ───────────────────────────────────────────────────────────────────── - - def has_quorum(self) -> bool: - """Geometric quorum: every segment overlaps with at least one other.""" - healthy = self.get_healthy_segments() - if len(healthy) < 2: - return False - for s1 in healthy: - has_overlap = any( - s1.point_id != s2.point_id and s1.dimension == s2.dimension - for s2 in healthy - ) - if not has_overlap: - return False - return True - - def k_resilient(self, k: int) -> bool: - """Every segment overlaps with at least k others.""" - healthy = self.get_healthy_segments() - for s1 in healthy: - overlap_count = sum( - 1 for s2 in healthy - if s1.point_id != s2.point_id and s1.dimension == s2.dimension - ) - if overlap_count < k: - return False - return True - - def reconfigure_quorum(self) -> bool: - """After failures, check if remaining segments still form quorum.""" - return self.has_quorum() - - # ───────────────────────────────────────────────────────────────────── - # Status & Logging - # ───────────────────────────────────────────────────────────────────── - - def get_status(self) -> dict: - healthy = self.get_healthy_segments() - overloaded = self.detect_overloaded() - bottlenecks = self.detect_bottlenecks() - return { - "controller_id": self.controller_id, - "total_segments": len(self.segments), - "healthy_segments": len(healthy), - "overloaded": [s.point_id for s in overloaded], - "bottlenecks": [s.point_id for s in bottlenecks], - "quorum": self.has_quorum(), - "k2_resilient": self.k_resilient(2), - } - - def _log(self, event: str, msg: str): - ts = time.strftime("%Y-%m-%dT%H:%M:%S") - print(f"[{ts}] [{self.controller_id}] [{event}] {msg}", flush=True) - - -# ═══════════════════════════════════════════════════════════════════════════ -# §3 CLI -# ═══════════════════════════════════════════════════════════════════════════ - -def main(): - if len(sys.argv) < 2: - print("Usage: topology_controller.py ") - sys.exit(1) - - with open(sys.argv[1]) as f: - cfg = json.load(f) - - ctrl = TopologyController(cfg.get("controller_id", "default")) - - # Register segments from config - for seg_raw in cfg.get("segments", []): - seg = NetworkSegment.from_dict(seg_raw) - ctrl.register_segment(seg) - - # Run ping flood and report - print("\n=== TOPOLOGY CONTROLLER STATUS ===") - print(json.dumps(ctrl.get_status(), indent=2)) - - print("\n=== PING FLOOD ===") - source = cfg.get("ping_source", list(ctrl.segments.keys())[0] if ctrl.segments else "") - if source: - results = ctrl.flood_ping(source) - for target, cost in results: - print(f" {source} -> {target}: cost={cost.to_float():.4f}") - - print("\n=== BOTTLENECKS ===") - for b in ctrl.detect_bottlenecks(): - print(f" {b.point_id}: load={b.current_load.to_float():.2f}") - - print("\n=== BEST PLACEMENT ===") - best = ctrl.pick_best_segment() - if best: - print(f" {best.point_id}: traversal_cost={best.traversal_cost().to_float():.4f}") - - -if __name__ == "__main__": - main() diff --git a/4-Infrastructure/infra/topology_node.py b/4-Infrastructure/infra/topology_node.py deleted file mode 100644 index 88480152..00000000 --- a/4-Infrastructure/infra/topology_node.py +++ /dev/null @@ -1,520 +0,0 @@ -#!/usr/bin/env python3 -""" -topology_node.py — Fit-to-Purpose Topology Node Runtime - -Python extraction of Semantics.TopologyNode Lean module. -Each node runs this runtime to participate in the Research Stack topology. - -Roles: - core — build, verify, synthesize (architect) - judge — arbitrate, attest, verify (BFT partner) - mirror — store, relay, backup (git mirror) - edge — filter, compress, route, attest (minimal resource) - foxTop — experimental, unindexed - -State machine: - BOOT → SELFTEST → ANNOUNCE → ACTIVE → (DEGRADED | FAILED) → BOOT - -Per AGENTS.md §6: Python may only serialize, spawn, wrap, display. -All cost/invariant/branching decisions are extracted from Lean. -""" - -import json -import sys -import time -import signal -import subprocess -import threading -import sqlite3 -import hashlib -from pathlib import Path -from dataclasses import dataclass, field, asdict -from typing import Dict, List, Optional, Any -from enum import Enum, auto - - -# ═══════════════════════════════════════════════════════════════════════════ -# §0 Q16.16 Fixed-Point (Python extraction) -# ═══════════════════════════════════════════════════════════════════════════ - -class Q16_16: - """Fixed-point Q16.16 using Python int (backend: UInt32 semantics).""" - def __init__(self, val: int): - self.val = val & 0xFFFFFFFF - - @staticmethod - def zero() -> "Q16_16": - return Q16_16(0) - - @staticmethod - def one() -> "Q16_16": - return Q16_16(0x00010000) - - @staticmethod - def from_nat(n: int) -> "Q16_16": - return Q16_16((n * 65536) & 0xFFFFFFFF) - - @staticmethod - def from_frac(num: int, denom: int) -> "Q16_16": - if denom == 0: - return Q16_16.zero() - return Q16_16((num * 65536 // denom) & 0xFFFFFFFF) - - def __add__(self, other: "Q16_16") -> "Q16_16": - return Q16_16((self.val + other.val) & 0xFFFFFFFF) - - def __sub__(self, other: "Q16_16") -> "Q16_16": - return Q16_16((self.val - other.val) & 0xFFFFFFFF) - - def __lt__(self, other: "Q16_16") -> bool: - # Signed comparison for Q16.16 - a = self.val if self.val < 0x80000000 else self.val - 0x100000000 - b = other.val if other.val < 0x80000000 else other.val - 0x100000000 - return a < b - - def __le__(self, other: "Q16_16") -> bool: - return self == other or self < other - - def __gt__(self, other: "Q16_16") -> bool: - return other < self - - def __ge__(self, other: "Q16_16") -> bool: - return other <= self - - def __eq__(self, other: object) -> bool: - if not isinstance(other, Q16_16): - return False - return self.val == other.val - - def to_float(self) -> float: - v = self.val if self.val < 0x80000000 else self.val - 0x100000000 - return v / 65536.0 - - def to_dict(self) -> dict: - return {"val": self.val, "float": self.to_float()} - - @staticmethod - def from_dict(d: dict) -> "Q16_16": - return Q16_16(d["val"]) - - -# ═══════════════════════════════════════════════════════════════════════════ -# §1 Enums -# ═══════════════════════════════════════════════════════════════════════════ - -class NodeRole(str, Enum): - CORE = "core" - JUDGE = "judge" - MIRROR = "mirror" - EDGE = "edge" - FOXTOP = "foxTop" - - -class NodeState(str, Enum): - BOOT = "boot" - SELFTEST = "selftest" - ANNOUNCE = "announce" - ACTIVE = "active" - DEGRADED = "degraded" - FAILED = "failed" - - -class NodeCapability(str, Enum): - LEAN_BUILD = "leanBuild" - FPGA_SYNTH = "fpgaSynth" - EQUATION_FOREST = "equationForest" - FULL_GIT = "fullGit" - BFT_ARBITRATE = "bftArbitrate" - MMR_VERIFY = "mmrVerify" - ATTESTATION = "attestation" - GIT_MIRROR = "gitMirror" - OBJECT_STORE = "objectStore" - RELAY = "relay" - BACKUP = "backup" - COMPRESS = "compress" - RGFLOW_FILTER = "rgflowFilter" - ROUTE = "route" - STORAGE = "storage" - COMPUTE = "compute" - EXPERIMENTAL = "experimental" - - -class BindClass(str, Enum): - INFORMATIONAL = "informational" - GEOMETRIC = "geometric" - THERMODYNAMIC = "thermodynamic" - PHYSICAL = "physical" - CONTROL = "control" - - -# ═══════════════════════════════════════════════════════════════════════════ -# §2 Capability & Role Defaults (extracted from Lean) -# ═══════════════════════════════════════════════════════════════════════════ - -CAPABILITY_COST: Dict[NodeCapability, Q16_16] = { - NodeCapability.LEAN_BUILD: Q16_16.from_nat(10), - NodeCapability.FPGA_SYNTH: Q16_16.from_nat(50), - NodeCapability.EQUATION_FOREST: Q16_16.from_nat(20), - NodeCapability.FULL_GIT: Q16_16.from_nat(5), - NodeCapability.BFT_ARBITRATE: Q16_16.from_nat(3), - NodeCapability.MMR_VERIFY: Q16_16.from_nat(2), - NodeCapability.ATTESTATION: Q16_16.from_nat(2), - NodeCapability.GIT_MIRROR: Q16_16.from_nat(2), - NodeCapability.OBJECT_STORE: Q16_16.from_nat(1), - NodeCapability.RELAY: Q16_16.from_nat(1), - NodeCapability.BACKUP: Q16_16.from_nat(1), - NodeCapability.COMPRESS: Q16_16.from_nat(4), - NodeCapability.RGFLOW_FILTER: Q16_16.from_nat(3), - NodeCapability.ROUTE: Q16_16.from_nat(1), - NodeCapability.STORAGE: Q16_16.from_nat(1), - NodeCapability.COMPUTE: Q16_16.from_nat(2), - NodeCapability.EXPERIMENTAL: Q16_16.from_nat(8), -} - -DEFAULT_CAPABILITIES: Dict[NodeRole, List[NodeCapability]] = { - NodeRole.CORE: [ - NodeCapability.LEAN_BUILD, NodeCapability.FPGA_SYNTH, - NodeCapability.EQUATION_FOREST, NodeCapability.FULL_GIT, - NodeCapability.STORAGE, NodeCapability.COMPUTE, - ], - NodeRole.JUDGE: [ - NodeCapability.BFT_ARBITRATE, NodeCapability.MMR_VERIFY, - NodeCapability.ATTESTATION, NodeCapability.COMPUTE, - ], - NodeRole.MIRROR: [ - NodeCapability.GIT_MIRROR, NodeCapability.OBJECT_STORE, - NodeCapability.RELAY, NodeCapability.BACKUP, NodeCapability.STORAGE, - ], - NodeRole.EDGE: [ - NodeCapability.COMPRESS, NodeCapability.RGFLOW_FILTER, - NodeCapability.ATTESTATION, NodeCapability.ROUTE, - NodeCapability.STORAGE, NodeCapability.COMPUTE, - ], - NodeRole.FOXTOP: [ - NodeCapability.EXPERIMENTAL, NodeCapability.COMPUTE, - NodeCapability.STORAGE, NodeCapability.ROUTE, - ], -} - -DEFAULT_BIND_CLASSES: Dict[NodeRole, List[BindClass]] = { - NodeRole.CORE: [BindClass.INFORMATIONAL, BindClass.GEOMETRIC, BindClass.THERMODYNAMIC, BindClass.PHYSICAL, BindClass.CONTROL], - NodeRole.JUDGE: [BindClass.INFORMATIONAL, BindClass.CONTROL], - NodeRole.MIRROR: [BindClass.INFORMATIONAL, BindClass.GEOMETRIC], - NodeRole.EDGE: [BindClass.PHYSICAL, BindClass.THERMODYNAMIC, BindClass.CONTROL], - NodeRole.FOXTOP: [BindClass.INFORMATIONAL, BindClass.PHYSICAL, BindClass.CONTROL], -} - -MAX_ENERGY: Dict[NodeRole, Q16_16] = { - NodeRole.CORE: Q16_16.from_nat(100), - NodeRole.JUDGE: Q16_16.from_nat(50), - NodeRole.MIRROR: Q16_16.from_nat(50), - NodeRole.EDGE: Q16_16.from_nat(25), - NodeRole.FOXTOP: Q16_16.from_nat(40), -} - -RECOVERY_RATE: Dict[NodeRole, Q16_16] = { - NodeRole.CORE: Q16_16.from_frac(1, 2), - NodeRole.JUDGE: Q16_16.from_frac(1, 4), - NodeRole.MIRROR: Q16_16.from_frac(1, 4), - NodeRole.EDGE: Q16_16.from_frac(1, 8), - NodeRole.FOXTOP: Q16_16.from_frac(1, 4), -} - -STATE_TRANSITION_COST: Dict[tuple, Q16_16] = { - (NodeState.BOOT, NodeState.SELFTEST): Q16_16.from_nat(1), - (NodeState.SELFTEST, NodeState.ANNOUNCE): Q16_16.from_nat(1), - (NodeState.ANNOUNCE, NodeState.ACTIVE): Q16_16.from_nat(2), - (NodeState.ACTIVE, NodeState.DEGRADED): Q16_16.from_nat(1), - (NodeState.DEGRADED, NodeState.ACTIVE): Q16_16.from_nat(3), - (NodeState.ACTIVE, NodeState.FAILED): Q16_16.zero(), - (NodeState.DEGRADED, NodeState.FAILED): Q16_16.zero(), - (NodeState.FAILED, NodeState.BOOT): Q16_16.from_nat(5), - (NodeState.SELFTEST, NodeState.FAILED): Q16_16.from_nat(1), - (NodeState.ANNOUNCE, NodeState.FAILED): Q16_16.from_nat(1), -} - -ALLOWED_TRANSITIONS: set = { - (NodeState.BOOT, NodeState.SELFTEST), - (NodeState.SELFTEST, NodeState.ANNOUNCE), - (NodeState.ANNOUNCE, NodeState.ACTIVE), - (NodeState.ACTIVE, NodeState.DEGRADED), - (NodeState.ACTIVE, NodeState.FAILED), - (NodeState.DEGRADED, NodeState.ACTIVE), - (NodeState.DEGRADED, NodeState.FAILED), - (NodeState.FAILED, NodeState.BOOT), - (NodeState.SELFTEST, NodeState.FAILED), - (NodeState.ANNOUNCE, NodeState.FAILED), -} - - -# ═══════════════════════════════════════════════════════════════════════════ -# §3 Node Config & State -# ═══════════════════════════════════════════════════════════════════════════ - -@dataclass -class NodeConfig: - node_id: str - role: NodeRole - memory_budget_mb: int - disk_budget_gb: int - jurisdiction: str - capabilities: List[NodeCapability] = field(default_factory=list) - bind_classes: List[BindClass] = field(default_factory=list) - services: Dict[str, List[str]] = field(default_factory=dict) - # services: {"warden": ["/usr/bin/python3", "warden.py", "--port", "8448"]} - - def __post_init__(self): - if not self.capabilities: - self.capabilities = DEFAULT_CAPABILITIES.get(self.role, []) - if not self.bind_classes: - self.bind_classes = DEFAULT_BIND_CLASSES.get(self.role, []) - - -@dataclass -class TopologyNodeState: - node_id: str - role: NodeRole - state: NodeState = NodeState.BOOT - capabilities: List[NodeCapability] = field(default_factory=list) - energy_budget: Q16_16 = field(default_factory=Q16_16.zero) - memory_budget_mb: int = 0 - disk_budget_gb: int = 0 - bind_classes: List[BindClass] = field(default_factory=list) - jurisdiction: str = "" - - def to_dict(self) -> dict: - return { - "node_id": self.node_id, - "role": self.role.value, - "state": self.state.value, - "capabilities": [c.value for c in self.capabilities], - "energy_budget": self.energy_budget.to_dict(), - "memory_budget_mb": self.memory_budget_mb, - "disk_budget_gb": self.disk_budget_gb, - "bind_classes": [b.value for b in self.bind_classes], - "jurisdiction": self.jurisdiction, - } - - @staticmethod - def from_config(cfg: NodeConfig) -> "TopologyNodeState": - return TopologyNodeState( - node_id=cfg.node_id, - role=cfg.role, - state=NodeState.BOOT, - capabilities=cfg.capabilities, - energy_budget=MAX_ENERGY.get(cfg.role, Q16_16.from_nat(25)), - memory_budget_mb=cfg.memory_budget_mb, - disk_budget_gb=cfg.disk_budget_gb, - bind_classes=cfg.bind_classes, - jurisdiction=cfg.jurisdiction, - ) - - -# ═══════════════════════════════════════════════════════════════════════════ -# §4 Topology Node Runtime -# ═══════════════════════════════════════════════════════════════════════════ - -class TopologyNodeRuntime: - """Fit-to-purpose topology node runtime.""" - - def __init__(self, config_path: str): - self.config_path = Path(config_path) - self.cfg = self._load_config() - self.state = TopologyNodeState.from_config(self.cfg) - self.db_path = Path.home() / f"var/db/topology-node-{self.cfg.node_id}.db" - self.db_path.parent.mkdir(parents=True, exist_ok=True) - self._init_db() - self._running = False - self._services: Dict[str, subprocess.Popen] = {} - self._threads: List[threading.Thread] = [] - - def _load_config(self) -> NodeConfig: - with open(self.config_path) as f: - raw = json.load(f) - return NodeConfig( - node_id=raw["node_id"], - role=NodeRole(raw["role"]), - memory_budget_mb=raw["memory_budget_mb"], - disk_budget_gb=raw["disk_budget_gb"], - jurisdiction=raw["jurisdiction"], - capabilities=[NodeCapability(c) for c in raw.get("capabilities", [])], - bind_classes=[BindClass(b) for b in raw.get("bind_classes", [])], - services=raw.get("services", {}), - ) - - def _init_db(self): - conn = sqlite3.connect(str(self.db_path)) - c = conn.cursor() - c.execute(""" - CREATE TABLE IF NOT EXISTS node_state_log ( - id INTEGER PRIMARY KEY AUTOINCREMENT, - timestamp REAL, - old_state TEXT, - new_state TEXT, - energy_budget_val INTEGER, - reason TEXT - ) - """) - c.execute(""" - CREATE TABLE IF NOT EXISTS bind_requests ( - id INTEGER PRIMARY KEY AUTOINCREMENT, - timestamp REAL, - bind_class TEXT, - cost_val INTEGER, - accepted INTEGER, - reason TEXT - ) - """) - conn.commit() - conn.close() - - # ───────────────────────────────────────────────────────────────────── - # State Machine - # ───────────────────────────────────────────────────────────────────── - - def transition(self, new_state: NodeState, reason: str = "") -> bool: - if (self.state.state, new_state) not in ALLOWED_TRANSITIONS: - self._log("transition_rejected", f"illegal: {self.state.state.value} -> {new_state.value}") - return False - cost = STATE_TRANSITION_COST.get((self.state.state, new_state), Q16_16.from_nat(1)) - if self.state.energy_budget < cost: - self._log("transition_rejected", f"insufficient_energy: need {cost.to_float()}, have {self.state.energy_budget.to_float()}") - return False - old = self.state.state - self.state.energy_budget = self.state.energy_budget - cost - self.state.state = new_state - self._log_state_change(old, new_state, reason) - return True - - def recover_energy(self): - rate = RECOVERY_RATE.get(self.state.role, Q16_16.from_frac(1, 8)) - max_cap = MAX_ENERGY.get(self.state.role, Q16_16.from_nat(25)) - new_energy = self.state.energy_budget + rate - self.state.energy_budget = new_energy if new_energy < max_cap else max_cap - - def can_accept_bind(self, bc: BindClass, cost: Q16_16) -> bool: - return ( - self.state.state == NodeState.ACTIVE - and bc in self.state.bind_classes - and self.state.energy_budget >= cost - ) - - def deduct_energy(self, cost: Q16_16) -> bool: - if self.state.energy_budget >= cost: - self.state.energy_budget = self.state.energy_budget - cost - return True - return False - - # ───────────────────────────────────────────────────────────────────── - # Service Management - # ───────────────────────────────────────────────────────────────────── - - def start_services(self): - """Start configured services as child processes.""" - for name, cmd in self.cfg.services.items(): - if name in self._services: - continue - self._log("service_start", f"spawning {name}: {' '.join(cmd)}") - proc = subprocess.Popen( - cmd, - stdout=subprocess.PIPE, - stderr=subprocess.PIPE, - ) - self._services[name] = proc - - def stop_services(self): - for name, proc in self._services.items(): - self._log("service_stop", f"terminating {name}") - proc.terminate() - try: - proc.wait(timeout=5) - except subprocess.TimeoutExpired: - proc.kill() - self._services.clear() - - def check_services(self) -> Dict[str, bool]: - health = {} - for name, proc in list(self._services.items()): - ret = proc.poll() - if ret is not None: - health[name] = False - self._log("service_died", f"{name} exited {ret}") - del self._services[name] - else: - health[name] = True - return health - - # ───────────────────────────────────────────────────────────────────── - # Lifecycle - # ───────────────────────────────────────────────────────────────────── - - def run(self): - self._running = True - signal.signal(signal.SIGTERM, self._on_signal) - signal.signal(signal.SIGINT, self._on_signal) - - # BOOT → SELFTEST → ANNOUNCE → ACTIVE - self.transition(NodeState.SELFTEST, "boot_complete") - self._run_selftest() - self.transition(NodeState.ANNOUNCE, "selftest_pass") - self._announce() - self.transition(NodeState.ACTIVE, "announce_complete") - self.start_services() - - # Main loop - while self._running: - self.recover_energy() - health = self.check_services() - if any(not v for v in health.values()): - if self.state.state == NodeState.ACTIVE: - self.transition(NodeState.DEGRADED, "service_failure") - else: - if self.state.state == NodeState.DEGRADED: - self.transition(NodeState.ACTIVE, "service_recovery") - time.sleep(5) - - self.stop_services() - self._log("shutdown", "runtime exiting") - - def _run_selftest(self): - """Verify capabilities.""" - self._log("selftest", f"capabilities: {[c.value for c in self.state.capabilities]}") - for cap in self.state.capabilities: - cost = CAPABILITY_COST.get(cap, Q16_16.from_nat(1)) - self._log("selftest", f"{cap.value} cost={cost.to_float()}") - time.sleep(0.5) - - def _announce(self): - """Broadcast presence to mesh.""" - self._log("announce", json.dumps(self.state.to_dict())) - - def _on_signal(self, signum, _frame): - self._log("signal", f"received {signum}") - self._running = False - - def _log(self, event: str, msg: str): - ts = time.strftime("%Y-%m-%dT%H:%M:%S") - print(f"[{ts}] [{self.cfg.node_id}] [{event}] {msg}", flush=True) - - def _log_state_change(self, old: NodeState, new: NodeState, reason: str): - self._log("state_change", f"{old.value} -> {new.value} ({reason})") - conn = sqlite3.connect(str(self.db_path)) - c = conn.cursor() - c.execute( - "INSERT INTO node_state_log (timestamp, old_state, new_state, energy_budget_val, reason) VALUES (?, ?, ?, ?, ?)", - (time.time(), old.value, new.value, self.state.energy_budget.val, reason) - ) - conn.commit() - conn.close() - - -def main(): - if len(sys.argv) < 2: - print("Usage: topology_node.py ") - sys.exit(1) - runtime = TopologyNodeRuntime(sys.argv[1]) - runtime.run() - - -if __name__ == "__main__": - main() diff --git a/4-Infrastructure/infra/web_interaction_surface.py b/4-Infrastructure/infra/web_interaction_surface.py deleted file mode 100644 index b60ce0f9..00000000 --- a/4-Infrastructure/infra/web_interaction_surface.py +++ /dev/null @@ -1,405 +0,0 @@ -#!/usr/bin/env python3 -""" -web_interaction_surface.py — Swarm Web Interaction Implementation - -Implements the 100% feasibility web interaction surface designed by the swarm. -Features: -- Browser automation via Playwright -- Swarm-coordinated distributed crawling -- GPU duty assignment integration -- Docker container sandboxing -- Redis-based coordination -- Session management with AES-256 encryption - -Per swarm design: SwarmWebSurface v2.0.0 -""" - -import asyncio -import hashlib -import json -import time -from dataclasses import dataclass, field -from typing import Dict, List, Optional, Any, Callable -from enum import Enum -from pathlib import Path - -# Try to import playwright, fallback to simulation -HAS_PLAYWRIGHT = False -HAS_REDIS = False - -class DutyType(Enum): - """GPU duty types for web interaction.""" - WEB_NAVIGATION = "WEB_NAVIGATION" - CONTENT_EXTRACTION = "CONTENT_EXTRACTION" - FORM_INTERACTION = "FORM_INTERACTION" - JAVASCRIPT_EXECUTION = "JAVASCRIPT_EXECUTION" - SCREENSHOT_CAPTURE = "SCREENSHOT_CAPTURE" - DISTRIBUTED_CRAWL = "DISTRIBUTED_CRAWL" - - -@dataclass -class BrowserSession: - """Browser session with security and state tracking.""" - session_id: str - url: str - cookies: Dict[str, str] = field(default_factory=dict) - local_storage: Dict[str, str] = field(default_factory=dict) - created_at: float = field(default_factory=time.time) - last_activity: float = field(default_factory=time.time) - - def touch(self): - """Update last activity timestamp.""" - self.last_activity = time.time() - - -@dataclass -class WebTask: - """Web interaction task.""" - task_id: str - duty_type: DutyType - url: str - priority: int - options: Dict[str, Any] = field(default_factory=dict) - status: str = "pending" - result: Optional[Dict] = None - error: Optional[str] = None - assigned_gpu: Optional[int] = None - - -class BrowserPool: - """ - Browser pool management with dynamic sizing. - - From swarm optimization: - - Dynamic pool sizing (min 2, max 20) - - LRU eviction for idle browsers - - Graceful shutdown with drain mode - - Health check endpoints - - Memory leak detection - """ - - def __init__(self, min_browsers: int = 2, max_browsers: int = 20): - self.min_browsers = min_browsers - self.max_browsers = max_browsers - self.active_browsers: Dict[str, Any] = {} - self.idle_browsers: List[str] = [] - self.health_status: Dict[str, bool] = {} - - def acquire_browser(self, session_id: str) -> Optional[str]: - """Acquire browser from pool.""" - # Check if we can create new browser - if len(self.active_browsers) < self.max_browsers: - browser_id = f"browser_{session_id}_{hash(session_id) % 10000}" - self.active_browsers[browser_id] = { - "session_id": session_id, - "created_at": time.time() - } - self.health_status[browser_id] = True - return browser_id - - # Reuse idle browser - if self.idle_browsers: - browser_id = self.idle_browsers.pop(0) - self.active_browsers[browser_id] = { - "session_id": session_id, - "reused_at": time.time() - } - return browser_id - - return None - - def release_browser(self, browser_id: str): - """Release browser back to pool.""" - if browser_id in self.active_browsers: - self.idle_browsers.append(browser_id) - del self.active_browsers[browser_id] - - def get_stats(self) -> Dict[str, int]: - """Get pool statistics.""" - return { - "active": len(self.active_browsers), - "idle": len(self.idle_browsers), - "healthy": sum(self.health_status.values()), - "capacity": self.max_browsers - } - - -class SessionManager: - """ - Session management with AES-256 encryption. - - From swarm optimization: - - AES-256-GCM cookie encryption - - SQLite + Redis session storage - - Cross-browser session sync - - Configurable TTL with auto-cleanup - """ - - def __init__(self, ttl_seconds: int = 3600): - self.sessions: Dict[str, BrowserSession] = {} - self.ttl_seconds = ttl_seconds - - def create_session(self, url: str) -> str: - """Create new browser session.""" - session_id = f"sess_{hash(url + str(time.time())) % 1000000000}" - session = BrowserSession(session_id=session_id, url=url) - self.sessions[session_id] = session - return session_id - - def get_session(self, session_id: str) -> Optional[BrowserSession]: - """Get session if not expired.""" - session = self.sessions.get(session_id) - if session: - if time.time() - session.last_activity > self.ttl_seconds: - del self.sessions[session_id] - return None - session.touch() - return session - - def cleanup_expired(self): - """Remove expired sessions.""" - now = time.time() - expired = [sid for sid, s in self.sessions.items() - if now - s.last_activity > self.ttl_seconds] - for sid in expired: - del self.sessions[sid] - return len(expired) - - def get_stats(self) -> Dict[str, int]: - """Get session statistics.""" - return { - "active_sessions": len(self.sessions), - "ttl_seconds": self.ttl_seconds - } - - -class WebInteractionSurface: - """ - Swarm Web Interaction Surface (100% Feasibility Implementation). - - Implements the optimized design from swarm_web_surface_design_optimized.json - with all 5 optimization categories: - - Security sandboxing (Docker containers) - - Swarm coordination (Redis-based) - - Session management (AES-256 encryption) - - Browser pool management (dynamic sizing) - """ - - def __init__(self): - self.browser_pool = BrowserPool(min_browsers=2, max_browsers=20) - self.session_manager = SessionManager(ttl_seconds=3600) - self.task_queue: List[WebTask] = [] - self.completed_tasks: List[WebTask] = [] - self.distributed_locks: Dict[str, bool] = {} - - def submit_task(self, duty_type: DutyType, url: str, - priority: int = 5, options: Dict = None) -> Dict[str, Any]: - """Submit web interaction task.""" - task_id = f"web_{hash(url + str(time.time())) % 1000000000}" - task = WebTask( - task_id=task_id, - duty_type=duty_type, - url=url, - priority=priority, - options=options or {} - ) - self.task_queue.append(task) - - # Sort by priority - self.task_queue.sort(key=lambda t: t.priority, reverse=True) - - return { - "task_id": task_id, - "duty_type": duty_type.value, - "url": url, - "priority": priority, - "status": "queued", - "position": self.task_queue.index(task) + 1 - } - - def execute_task(self, task_id: str) -> Dict[str, Any]: - """Execute web interaction task.""" - task = next((t for t in self.task_queue if t.task_id == task_id), None) - if not task: - return {"error": f"Task {task_id} not found"} - - task.status = "executing" - - # Acquire browser - session_id = self.session_manager.create_session(task.url) - browser_id = self.browser_pool.acquire_browser(session_id) - - if not browser_id: - task.status = "failed" - task.error = "No browsers available" - return {"error": "No browsers available"} - - # Simulate execution (Playwright would go here) - task.assigned_gpu = 0 # Would be assigned by GPU duty system - - # Generate simulated result - result = { - "task_id": task_id, - "url": task.url, - "duty_type": task.duty_type.value, - "executed_at": time.time(), - "result": self._simulate_execution(task), - "browser_id": browser_id, - "session_id": session_id - } - - task.result = result - task.status = "completed" - self.completed_tasks.append(task) - self.task_queue.remove(task) - - # Cleanup - self.browser_pool.release_browser(browser_id) - - return result - - def _simulate_execution(self, task: WebTask) -> Dict[str, Any]: - """Simulate web task execution.""" - simulations = { - DutyType.WEB_NAVIGATION: { - "page_loaded": True, - "title": "Simulated Page", - "status_code": 200 - }, - DutyType.CONTENT_EXTRACTION: { - "text_content": "Simulated content extraction...", - "links_found": 15, - "images_found": 8 - }, - DutyType.SCREENSHOT_CAPTURE: { - "screenshot_path": f"/tmp/screenshot_{task.task_id}.png", - "dimensions": [1920, 1080] - }, - DutyType.FORM_INTERACTION: { - "forms_found": 2, - "fields_filled": 5, - "submitted": True - }, - DutyType.JAVASCRIPT_EXECUTION: { - "script_executed": True, - "return_value": "Simulated JS result", - "console_logs": 3 - }, - DutyType.DISTRIBUTED_CRAWL: { - "pages_crawled": 25, - "links_discovered": 150, - "depth_reached": 3 - } - } - return simulations.get(task.duty_type, {"status": "unknown"}) - - def get_task_queue(self) -> List[Dict[str, Any]]: - """Get current task queue.""" - return [ - { - "task_id": t.task_id, - "duty_type": t.duty_type.value, - "url": t.url[:50] + "..." if len(t.url) > 50 else t.url, - "priority": t.priority, - "status": t.status - } - for t in self.task_queue - ] - - def get_system_health(self) -> Dict[str, Any]: - """Get comprehensive system health.""" - return { - "browser_pool": self.browser_pool.get_stats(), - "session_manager": self.session_manager.get_stats(), - "task_queue": { - "pending": len(self.task_queue), - "completed": len(self.completed_tasks) - }, - "feasibility": "100%", - "version": "2.0.0", - "optimizations_applied": [ - "Security sandboxing (Docker)", - "Swarm coordination (Redis)", - "Session management (AES-256)", - "Browser pool management (Dynamic sizing)" - ] - } - - -# ═══════════════════════════════════════════════════════════════════════════ -# Unified Interface -# ═══════════════════════════════════════════════════════════════════════════ - -class SwarmWebInterface: - """ - Unified swarm web interface combining all capabilities. - """ - - def __init__(self): - self.surface = WebInteractionSurface() - - def navigate(self, url: str, priority: int = 5) -> Dict[str, Any]: - """Navigate to URL.""" - return self.surface.submit_task(DutyType.WEB_NAVIGATION, url, priority) - - def extract_content(self, url: str, priority: int = 5) -> Dict[str, Any]: - """Extract content from URL.""" - return self.surface.submit_task(DutyType.CONTENT_EXTRACTION, url, priority) - - def screenshot(self, url: str, priority: int = 5) -> Dict[str, Any]: - """Capture screenshot of URL.""" - return self.surface.submit_task(DutyType.SCREENSHOT_CAPTURE, url, priority) - - def execute(self, task_id: str) -> Dict[str, Any]: - """Execute submitted task.""" - return self.surface.execute_task(task_id) - - def health(self) -> Dict[str, Any]: - """Get system health.""" - return self.surface.get_system_health() - - -# ═══════════════════════════════════════════════════════════════════════════ -# Example Usage -# ═══════════════════════════════════════════════════════════════════════════ - -if __name__ == "__main__": - print("=" * 70) - print("WEB INTERACTION SURFACE TEST (100% Feasibility)") - print("=" * 70) - - web = SwarmWebInterface() - - # Test 1: Navigation - print("\n[1] Navigation Task:") - nav = web.navigate("https://example.com", priority=8) - print(f" Task ID: {nav['task_id']}") - print(f" Position: {nav['position']}") - - # Test 2: Content extraction - print("\n[2] Content Extraction Task:") - extract = web.extract_content("https://example.com/docs", priority=5) - print(f" Task ID: {extract['task_id']}") - - # Test 3: Screenshot - print("\n[3] Screenshot Task:") - shot = web.screenshot("https://example.com/chart", priority=7) - print(f" Task ID: {shot['task_id']}") - - # Test 4: Execute navigation - print("\n[4] Execute Navigation:") - result = web.execute(nav['task_id']) - print(f" Status: {result.get('result', {}).get('page_loaded', False)}") - print(f" Code: {result.get('result', {}).get('status_code', 0)}") - - # Test 5: System health - print("\n[5] System Health:") - health = web.health() - print(f" Feasibility: {health['feasibility']}") - print(f" Browser Pool: {health['browser_pool']['active']}/{health['browser_pool']['capacity']} active") - print(f" Pending Tasks: {health['task_queue']['pending']}") - print(f" Optimizations Applied: {len(health['optimizations_applied'])}") - - print("\n" + "=" * 70) - print("WEB INTERACTION SURFACE TEST COMPLETE") - print("=" * 70)