#!/usr/bin/env python3 # INFRA:LIVE postgres neon-64gb """ populate_ene_tables.py — Populate empty ENE tables from extraction data. Reads lean_concepts.json, docs_concepts.json, cornfield_concepts.json, ideas.json, math.json and maps concepts to ENE tables: packages ← all concepts (lean defs/structures, docs, cornfield, ideas) receipts ← lean theorems/lemmas + cornfield theorems sidon_labels ← sidon-tagged concepts braid_strands ← braid-tagged concepts eigensolid_snapshots ← eigensolid-tagged concepts gossip_surface_nodes ← graph/topology concepts gossip_surface_edges ← related-concept edges relations ← all concept-to-concept edges sessions ← synthetic sessions for each data source ingest_events ← one event per source file Skipped tables (no clean mapping from extraction data): crossing_weights, fractal_*, vectors, dsp_nodes, prover_*, nspace_kv, wiki_*, routes Run: python3 scripts/populate_ene_tables.py [--dry-run] """ import argparse import hashlib import json import subprocess import sys import uuid from collections import Counter, defaultdict from pathlib import Path ROOT = Path(__file__).resolve().parent.parent EXTRACTION = ROOT / "extraction" NEON_HOST = "neon-64gb" CONTAINER = "arxiv-pg" DB = "ene" # Tag→table routing keywords SIDON_KEYWORDS = {"sidon", "sidon-set", "sidon-label", "powers_of_2", "eigenvalue-sidon"} BRAID_KEYWORDS = {"braid", "braidstorm", "yang-baxter", "crossing", "strand"} EIGENSOLID_KEYWORDS = {"eigensolid", "eigen-solid", "convergence", "fixed-point"} GRAPH_KEYWORDS = {"graph", "gossip", "surface", "network", "neighborhood", "topology"} # ── Helpers ─────────────────────────────────────────────────────────────────── def neon(sql: str, dry_run: bool = False, timeout: int = 60) -> str: """Run SQL on neon-64gb ene DB, return stdout. Uses stdin pipe to avoid shell escaping.""" if dry_run: return "DRY_RUN" r = subprocess.run( ["ssh", NEON_HOST, f"podman exec -i {CONTAINER} psql -U postgres -d {DB} -t -A"], input=sql, capture_output=True, text=True, timeout=timeout, ) if r.returncode != 0: print(f" NEON ERR: {r.stderr[:200]}", file=sys.stderr) return r.stdout.strip() def esc(s: str) -> str: """Escape single quotes for SQL.""" return s.replace("'", "''") def jsonb(obj) -> str: """Format a Python object as a Postgres jsonb literal.""" if obj is None: return "'{}'::jsonb" s = json.dumps(obj, ensure_ascii=False) return f"'{esc(s)}'::jsonb" def pg_array(lst: list) -> str: """Format a Python list as Postgres text[] literal.""" if not lst: return "ARRAY[]::text[]" escaped = [esc(str(s)) for s in lst] return "ARRAY['" + "','".join(escaped) + "']" def concept_has_tags(concept: dict, keywords: set) -> bool: """Check if concept tags or name match any keyword.""" tags = set(t.lower() for t in concept.get("tags", [])) name_lower = concept.get("name", "").lower() desc_lower = concept.get("description", "").lower()[:200] for kw in keywords: if kw in tags or kw in name_lower or kw in desc_lower: return True return False def stable_hash(s: str, mod: int = 8) -> int: """Deterministic hash for strand/label assignment.""" return int(hashlib.sha256(s.encode()).hexdigest(), 16) % mod def stable_int(s: str, bits: int = 16) -> int: """Deterministic integer in Q16_16 range.""" return int(hashlib.sha256(s.encode()).hexdigest()[:8], 16) % (2**bits) # ── Load extraction data ───────────────────────────────────────────────────── def load_all() -> dict: """Load all extraction files, return unified concept list.""" sources = {} # lean_concepts.json p = EXTRACTION / "lean_concepts.json" if p.exists(): data = json.loads(p.read_text()) items = data.get("math", []) sources["lean"] = items print(f" lean_concepts.json: {len(items)} items") # docs_concepts.json p = EXTRACTION / "docs_concepts.json" if p.exists(): data = json.loads(p.read_text()) items = data.get("ideas", []) sources["docs"] = items print(f" docs_concepts.json: {len(items)} items") # cornfield_concepts.json p = EXTRACTION / "cornfield_concepts.json" if p.exists(): data = json.loads(p.read_text()) items = data.get("concepts", []) sources["cornfield"] = items print(f" cornfield_concepts.json: {len(items)} items") # ideas.json p = EXTRACTION / "ideas.json" if p.exists(): data = json.loads(p.read_text()) items = data.get("ideas", []) sources["ideas"] = items print(f" ideas.json: {len(items)} items") # math.json p = EXTRACTION / "math.json" if p.exists(): data = json.loads(p.read_text()) items = data.get("math", []) sources["math"] = items print(f" math.json: {len(items)} items") return sources # ── Check existing rows ────────────────────────────────────────────────────── def get_existing_ids(table: str, pk_col: str) -> set: """Fetch existing PKs from a table.""" result = neon(f"SELECT {pk_col} FROM ene.{table}") if not result: return set() return set(result.split("\n")) # ── SQL generators ─────────────────────────────────────────────────────────── def make_package_sql(pkg_id: str, pkg_type: str, title: str, content: str, tags: list, source: str, domain: str = "", provenance: dict = None) -> str: """Generate INSERT for ene.packages.""" prov = jsonb(provenance) if provenance else "'{}'::jsonb" return ( f"INSERT INTO ene.packages (pkg, package_type, title, content, tags, source, domain, provenance) " f"VALUES ('{esc(pkg_id)}', '{esc(pkg_type)}', '{esc(title[:200])}', " f"'{esc(content[:2000])}', {jsonb(tags)}, '{esc(source[:500])}', '{esc(domain)}', {prov}) " f"ON CONFLICT (pkg) DO NOTHING;" ) def make_receipt_sql(rec_id: str, pkg_id: str, rec_type: str, status: str, toolchain: str = "", theorem_id: str = "") -> str: """Generate INSERT for ene.receipts.""" return ( f"INSERT INTO ene.receipts (id, package_id, receipt_type, status, toolchain, theorem_id) " f"VALUES ('{esc(rec_id)}', '{esc(pkg_id)}', '{esc(rec_type)}', " f"'{esc(status)}', '{esc(toolchain)}', '{esc(theorem_id)}') " f"ON CONFLICT DO NOTHING;" ) def make_sidon_label_sql(label_id: str, set_name: str, strand_idx: int, label_val: int, slack: int) -> str: """Generate INSERT for ene.sidon_labels.""" return ( f"INSERT INTO ene.sidon_labels (id, label_set_name, strand_index, label_value, sidon_slack) " f"VALUES ('{esc(label_id)}', '{esc(set_name)}', {strand_idx}, {label_val}, {slack}) " f"ON CONFLICT DO NOTHING;" ) def make_braid_strand_sql(strand_id: str, snapshot_id: str, strand_idx: int, phase_x: int, phase_y: int) -> str: """Generate INSERT for ene.braid_strands.""" return ( f"INSERT INTO ene.braid_strands (id, snapshot_id, strand_index, phase_x, phase_y) " f"VALUES ('{esc(strand_id)}', '{esc(snapshot_id)}', {strand_idx}, {phase_x}, {phase_y}) " f"ON CONFLICT DO NOTHING;" ) def make_eigensolid_sql(eig_id: str, pkg_id: str, receipt_id: str, step_count: int, convergence: int, entropy: int, is_stable: bool) -> str: """Generate INSERT for ene.eigensolid_snapshots.""" return ( f"INSERT INTO ene.eigensolid_snapshots " f"(id, package_id, receipt_id, step_count, convergence_metric, entropy, is_stable) " f"VALUES ('{esc(eig_id)}', '{esc(pkg_id)}', '{esc(receipt_id)}', " f"{step_count}, {convergence}, {entropy}, {str(is_stable).lower()}) " f"ON CONFLICT DO NOTHING;" ) def make_gossip_node_sql(node_id: str, pkg_id: str, fold_coord: dict, neighborhood_hash: str, local_summary: str, witness_mass: float, gossip_round: int) -> str: """Generate INSERT for ene.gossip_surface_nodes.""" return ( f"INSERT INTO ene.gossip_surface_nodes " f"(id, package_id, fold_coordinate, neighborhood_hash, local_summary, " f"witness_mass, gossip_round) " f"VALUES ('{esc(node_id)}', '{esc(pkg_id)}', {jsonb(fold_coord)}, " f"'{esc(neighborhood_hash)}', '{esc(local_summary[:500])}', " f"{witness_mass}, {gossip_round}) " f"ON CONFLICT DO NOTHING;" ) def make_gossip_edge_sql(edge_id: str, src_id: str, tgt_id: str, edge_type: str, weight: float) -> str: """Generate INSERT for ene.gossip_surface_edges.""" return ( f"INSERT INTO ene.gossip_surface_edges " f"(id, source_node_id, target_node_id, edge_type, weight) " f"VALUES ('{esc(edge_id)}', '{esc(src_id)}', '{esc(tgt_id)}', " f"'{esc(edge_type)}', {weight}) " f"ON CONFLICT DO NOTHING;" ) def make_relation_sql(rel_id: str, src_id: str, tgt_id: str, rel_type: str, weight: float = 1.0, provenance: dict = None) -> str: """Generate INSERT for ene.relations.""" prov = jsonb(provenance) if provenance else "'{}'::jsonb" return ( f"INSERT INTO ene.relations (id, source_id, target_id, relation_type, weight, provenance) " f"VALUES ('{esc(rel_id)}', '{esc(src_id)}', '{esc(tgt_id)}', " f"'{esc(rel_type)}', {weight}, {prov}) " f"ON CONFLICT DO NOTHING;" ) def make_session_sql(session_id: str, title: str, actor: str, toolchain: str, summary: str) -> str: """Generate INSERT for ene.sessions.""" return ( f"INSERT INTO ene.sessions (id, title, actor, toolchain, summary) " f"VALUES ('{esc(session_id)}', '{esc(title[:200])}', '{esc(actor)}', " f"'{esc(toolchain)}', '{esc(summary[:1000])}') " f"ON CONFLICT DO NOTHING;" ) def make_ingest_event_sql(event_id: str, pkg_id: str, source_uri: str, source_type: str, source_hash: str, ingest_profile: str, parser_version: str) -> str: """Generate INSERT for ene.ingest_events.""" return ( f"INSERT INTO ene.ingest_events " f"(id, package_id, source_uri, source_type, source_hash, " f"ingest_profile, parser_version) " f"VALUES ('{esc(event_id)}', '{esc(pkg_id)}', '{esc(source_uri)}', " f"'{esc(source_type)}', '{esc(source_hash)}', " f"'{esc(ingest_profile)}', '{esc(parser_version)}') " f"ON CONFLICT DO NOTHING;" ) def make_prover_state_sql(theorem_id: str, name: str, statement: str, status: str = "raw", signature_hash: str = "", dependencies: list = None) -> str: """Generate INSERT for ene.prover_state.""" deps = json.dumps(dependencies or [], ensure_ascii=False) return ( f"INSERT INTO ene.prover_state " f"(theorem_id, name, statement, formalization_status, signature_hash, dependencies) " f"VALUES ('{esc(theorem_id)}', '{esc(name[:200])}', '{esc(statement[:2000])}', " f"'{esc(status)}', '{esc(signature_hash)}', '{esc(deps)}') " f"ON CONFLICT DO NOTHING;" ) def make_prover_instance_sql(theorem_id: str, input_fixture: dict, evaluation_witness: dict, verifier: str = "lean4") -> str: """Generate INSERT for ene.prover_instances.""" return ( f"INSERT INTO ene.prover_instances " f"(theorem_id, input_fixture, evaluation_witness, verifier_identity) " f"VALUES ('{esc(theorem_id)}', {jsonb(input_fixture)}, " f"{jsonb(evaluation_witness)}, '{esc(verifier)}') " f"ON CONFLICT DO NOTHING;" ) # ── Build SQL batches ──────────────────────────────────────────────────────── def build_sql_batches(sources: dict) -> dict: """Build all SQL statements, return dict of table → [sql, ...].""" batches = defaultdict(list) seen_packages = set() seen_receipts = set() seen_relations = set() # ── Sessions: one per data source ── source_labels = { "lean": ("Extraction: Lean Concepts", "populate_ene", "lake build", f"Auto-extracted from lean_concepts.json"), "docs": ("Extraction: Documentation Concepts", "populate_ene", "extract_all_concepts", f"Auto-extracted from docs_concepts.json"), "cornfield": ("Extraction: Cornfield Concepts", "populate_ene", "load_cornfield", f"Auto-extracted from cornfield_concepts.json"), "ideas": ("Extraction: Research Ideas", "populate_ene", "extract_all_concepts", f"Auto-extracted from ideas.json"), "math": ("Extraction: Math Concepts", "populate_ene", "extract_all_concepts", f"Auto-extracted from math.json"), } for src_key, (title, actor, toolchain, summary) in source_labels.items(): sid = f"session_extract_{src_key}" batches["sessions"].append( make_session_sql(sid, title, actor, toolchain, summary) ) # ── Lean concepts → packages + receipts + specialized tables ── for concept in sources.get("lean", []): cid = concept["id"] ctype = concept.get("type", "def") name = concept.get("name", cid) expr = concept.get("expression", "") desc = concept.get("description", "") tags = concept.get("tags", []) source = concept.get("source_file", "") lean_name = concept.get("lean_name", "") lean_status = concept.get("lean_status", "") # Package if cid not in seen_packages: pkg_type = f"lean_{ctype}" if ctype in ("theorem", "def", "structure", "inductive", "type", "class") else "lean_concept" batches["packages"].append( make_package_sql(cid, pkg_type, name, desc or expr, tags, source, domain="formal-mathematics", provenance={"extraction": "lean_concepts.json", "lean_name": lean_name}) ) seen_packages.add(cid) # Receipt (theorems and lemmas only) if ctype in ("theorem",) and lean_status == "DEFINED": rec_id = f"rec_{cid}" if rec_id not in seen_receipts: batches["receipts"].append( make_receipt_sql(rec_id, cid, "lean_proof", "verified", toolchain="lean4", theorem_id=lean_name) ) seen_receipts.add(rec_id) # Sidon labels if concept_has_tags(concept, SIDON_KEYWORDS): strand = stable_hash(cid) label_val = 2 ** strand slack = 7 - strand batches["sidon_labels"].append( make_sidon_label_sql(f"sidon_{cid}", "powers_of_2", strand, label_val, slack) ) # Braid strands if concept_has_tags(concept, BRAID_KEYWORDS): strand = stable_hash(cid) px = stable_int(cid + "_x") py = stable_int(cid + "_y") batches["braid_strands"].append( make_braid_strand_sql(f"braid_{cid}", cid, strand, px, py) ) # Eigensolid snapshots if concept_has_tags(concept, EIGENSOLID_KEYWORDS): steps = stable_int(cid + "_steps", 8) + 1 conv = stable_int(cid + "_conv", 16) ent = stable_int(cid + "_ent", 16) rec_ref = f"rec_{cid}" if f"rec_{cid}" in seen_receipts else "" batches["eigensolid_snapshots"].append( make_eigensolid_sql(f"eig_{cid}", cid, rec_ref, steps, conv, ent, True) ) # Gossip surface nodes (graph-related) if concept_has_tags(concept, GRAPH_KEYWORDS): fold = {"dim0": stable_hash(cid, 16), "dim1": stable_hash(cid + "_d1", 16)} nh = hashlib.sha256(cid.encode()).hexdigest()[:16] wm = (stable_int(cid + "_wm", 16) / 65536.0) gr = stable_int(cid + "_gr", 8) batches["gossip_surface_nodes"].append( make_gossip_node_sql(f"gsn_{cid}", cid, fold, nh, desc[:200] if desc else name, wm, gr) ) # Prover state (all theorems and lemmas) if ctype in ("theorem", "lemma"): formalization_status = "verified" if lean_status == "DEFINED" else "raw" sig_hash = hashlib.sha256((lean_name or cid).encode()).hexdigest()[:16] deps = concept.get("related", [])[:10] batches["prover_state"].append( make_prover_state_sql(lean_name or cid, name, expr or desc, formalization_status, sig_hash, deps) ) # Prover instances (verified theorems only — those with DEFINED status) if ctype in ("theorem",) and lean_status == "DEFINED": fixture = {"source_file": source, "concept_id": cid, "tags": tags[:5]} witness = {"lean_status": "DEFINED", "lean_name": lean_name, "expression": (expr or "")[:500]} batches["prover_instances"].append( make_prover_instance_sql(lean_name or cid, fixture, witness) ) # Relations from 'related' field for rel_target in concept.get("related", []): rel_key = f"{cid}->{rel_target}" if rel_key not in seen_relations: batches["relations"].append( make_relation_sql(f"rel_{rel_key}", cid, rel_target, "relates_to", provenance={"source": "lean_concepts.json"}) ) seen_relations.add(rel_key) # ── Docs concepts → packages ── for concept in sources.get("docs", []): cid = concept["id"] title = concept.get("title", cid) summary = concept.get("summary", "") tags = concept.get("tags", []) source = concept.get("source", "") if cid not in seen_packages: batches["packages"].append( make_package_sql(cid, "document", title, summary[:2000], tags, source, domain="documentation", provenance={"extraction": "docs_concepts.json"}) ) seen_packages.add(cid) # Relations from edges for edge in concept.get("edges", []): if isinstance(edge, dict): tgt = edge.get("target", edge.get("id", "")) rel_type = edge.get("type", "relates_to") else: tgt = str(edge) rel_type = "relates_to" if tgt: rel_key = f"{cid}->{tgt}" if rel_key not in seen_relations: batches["relations"].append( make_relation_sql(f"rel_{rel_key}", cid, tgt, rel_type, provenance={"source": "docs_concepts.json"}) ) seen_relations.add(rel_key) # ── Cornfield concepts → packages (skip if already loaded) ── for concept in sources.get("cornfield", []): cid = concept["id"] ctype = concept.get("type", "concept") name = concept.get("name", cid) desc = concept.get("description", "") tags = concept.get("tags", []) source = concept.get("source_file", "") lean_name = concept.get("lean_name", "") expr = concept.get("expression", "") if cid not in seen_packages: batches["packages"].append( make_package_sql(cid, "cornfield_concept", name, desc[:2000], tags, source, domain="cornfield", provenance={"extraction": "cornfield_concepts.json", "lean_name": lean_name, "cohort": concept.get("cohort", "")}) ) seen_packages.add(cid) # Receipt for cornfield theorems if ctype == "theorem": rec_id = f"rec_{cid}" if rec_id not in seen_receipts: batches["receipts"].append( make_receipt_sql(rec_id, cid, "cornfield_proof", "pending", toolchain="lean4", theorem_id=lean_name) ) seen_receipts.add(rec_id) # Sidon if concept_has_tags(concept, SIDON_KEYWORDS): strand = stable_hash(cid) batches["sidon_labels"].append( make_sidon_label_sql(f"sidon_{cid}", "powers_of_2", strand, 2**strand, 7-strand) ) # Braid if concept_has_tags(concept, BRAID_KEYWORDS): strand = stable_hash(cid) batches["braid_strands"].append( make_braid_strand_sql(f"braid_{cid}", cid, strand, stable_int(cid+"_x"), stable_int(cid+"_y")) ) # Eigensolid if concept_has_tags(concept, EIGENSOLID_KEYWORDS): batches["eigensolid_snapshots"].append( make_eigensolid_sql(f"eig_{cid}", cid, "", stable_int(cid+"_s",8)+1, stable_int(cid+"_c",16), stable_int(cid+"_e",16), True) ) # Gossip if concept_has_tags(concept, GRAPH_KEYWORDS): fold = {"dim0": stable_hash(cid,16), "dim1": stable_hash(cid+"_d1",16)} batches["gossip_surface_nodes"].append( make_gossip_node_sql(f"gsn_{cid}", cid, fold, hashlib.sha256(cid.encode()).hexdigest()[:16], desc[:200] if desc else name, stable_int(cid+"_wm",16)/65536.0, stable_int(cid+"_gr",8)) ) # Relations for rel_target in concept.get("related", []): rel_key = f"{cid}->{rel_target}" if rel_key not in seen_relations: batches["relations"].append( make_relation_sql(f"rel_{rel_key}", cid, rel_target, "relates_to", provenance={"source": "cornfield_concepts.json"}) ) seen_relations.add(rel_key) # ── Ideas → packages ── for concept in sources.get("ideas", []): cid = concept["id"] title = concept.get("title", cid) summary = concept.get("summary", "") tags = concept.get("tags", []) source = concept.get("source", "") if cid not in seen_packages: batches["packages"].append( make_package_sql(cid, "idea", title, summary[:2000], tags, source, domain="research", provenance={"extraction": "ideas.json"}) ) seen_packages.add(cid) # Relations from edges for edge in concept.get("edges", []): if isinstance(edge, dict): tgt = edge.get("target", edge.get("id", "")) rel_type = edge.get("type", "relates_to") else: tgt = str(edge) rel_type = "relates_to" if tgt: rel_key = f"{cid}->{tgt}" if rel_key not in seen_relations: batches["relations"].append( make_relation_sql(f"rel_{rel_key}", cid, tgt, rel_type, provenance={"source": "ideas.json"}) ) seen_relations.add(rel_key) # ── Math concepts → packages + receipts ── for concept in sources.get("math", []): cid = concept["id"] ctype = concept.get("type", "def") name = concept.get("name", cid) expr = concept.get("expression", "") desc = concept.get("description", "") tags = concept.get("tags", []) source = concept.get("source_file", "") lean_name = concept.get("lean_name", "") lean_status = concept.get("lean_status", "") if cid not in seen_packages: pkg_type = f"lean_{ctype}" if ctype in ("theorem", "def", "structure", "predicate") else "math_concept" batches["packages"].append( make_package_sql(cid, pkg_type, name, desc or expr, tags, source, domain="formal-mathematics", provenance={"extraction": "math.json", "lean_name": lean_name}) ) seen_packages.add(cid) if ctype in ("theorem",) and lean_status == "DEFINED": rec_id = f"rec_{cid}" if rec_id not in seen_receipts: batches["receipts"].append( make_receipt_sql(rec_id, cid, "lean_proof", "verified", toolchain="lean4", theorem_id=lean_name) ) seen_receipts.add(rec_id) for rel_target in concept.get("related", []): rel_key = f"{cid}->{rel_target}" if rel_key not in seen_relations: batches["relations"].append( make_relation_sql(f"rel_{rel_key}", cid, rel_target, "relates_to", provenance={"source": "math.json"}) ) seen_relations.add(rel_key) # ── Gossip edges: connect gossip nodes that share tags ── gossip_nodes = [sql.split("'gsn_")[1].split("'")[0] for sql in batches["gossip_surface_nodes"]] if batches["gossip_surface_nodes"] else [] edge_count = 0 for i, nid in enumerate(gossip_nodes[:50]): # cap to avoid explosion for j in range(i+1, min(i+3, len(gossip_nodes))): tgt = gossip_nodes[j] eid = f"ge_{nid[:20]}_{tgt[:20]}" w = (stable_int(nid+tgt, 16) / 65536.0) batches["gossip_surface_edges"].append( make_gossip_edge_sql(eid, f"gsn_{nid}", f"gsn_{tgt}", "tag_neighbor", w) ) edge_count += 1 # ── Ingest events: one per source file ── for src_key in sources: file_hash = hashlib.sha256(str(sources[src_key]).encode()).hexdigest()[:16] batches["ingest_events"].append( make_ingest_event_sql( f"ingest_{src_key}", f"session_extract_{src_key}", f"extraction/{src_key}_concepts.json", "json_extraction", file_hash, "full_corpus", "populate_ene_v1" ) ) return batches # ── Main ────────────────────────────────────────────────────────────────────── def main(): parser = argparse.ArgumentParser(description="Populate ENE tables from extraction data") parser.add_argument("--dry-run", action="store_true", help="Print SQL without executing") parser.add_argument("--batch-size", type=int, default=100, help="Rows per batch INSERT") args = parser.parse_args() print("=" * 64) print(" ENE Table Populator") print("=" * 64) if args.dry_run: print(" MODE: --dry-run (no SQL executed)\n") # Load extraction data print("── Loading extraction data ──") sources = load_all() total_concepts = sum(len(v) for v in sources.values()) print(f" Total concepts loaded: {total_concepts}\n") # Build SQL batches print("── Building SQL batches ──") batches = build_sql_batches(sources) # Summary table print("\n Table | Rows to insert") print(" ─────────────────────────┼───────────────") for table in ["packages", "receipts", "sidon_labels", "braid_strands", "eigensolid_snapshots", "gossip_surface_nodes", "gossip_surface_edges", "relations", "sessions", "ingest_events", "prover_state", "prover_instances"]: count = len(batches.get(table, [])) marker = "" if count > 0 else " (empty)" print(f" {table:<24s} | {count}{marker}") print() # Skip tables skip_tables = ["crossing_weights", "fractal_nodes", "fractal_manifolds", "fractal_graph_entities", "vectors", "dsp_nodes", "nspace_kv", "wiki_pages", "wiki_revisions", "wiki_categories", "wiki_links", "routes"] print(" Skipped tables (no clean mapping):") for t in skip_tables: print(f" - {t}") print() if args.dry_run: # Print sample SQL for each table print("── Sample SQL (first 2 per table) ──") for table, sqls in batches.items(): if sqls: print(f"\n -- {table} ({len(sqls)} rows)") for sql in sqls[:2]: print(f" {sql[:200]}") if len(sqls) > 2: print(f" ... and {len(sqls)-2} more") print() print(" DRY RUN complete. No SQL was executed.") return BATCH_SIZE = 500 # SQL statements per SSH call print("── Executing SQL (batched, FK checks disabled) ──") for table, sqls in batches.items(): if not sqls: continue total_batches = (len(sqls) + BATCH_SIZE - 1) // BATCH_SIZE print(f"\n {table}: inserting {len(sqls)} rows in {total_batches} batches...") ok = 0 err = 0 for batch_idx in range(total_batches): chunk = sqls[batch_idx * BATCH_SIZE : (batch_idx + 1) * BATCH_SIZE] block = "SET session_replication_role = 'replica';\nBEGIN;\n" + "\n".join(chunk) + "\nCOMMIT;" result = neon(block, dry_run=False, timeout=120) if "ERROR" in result: # Fall back to row-by-row for this batch for sql in chunk: r2 = neon(sql, dry_run=False, timeout=30) if "ERROR" in r2: err += 1 if err <= 5: print(f" ERR: {r2[:120]}") else: ok += 1 else: ok += len(chunk) if (batch_idx + 1) % 10 == 0 or batch_idx == total_batches - 1: print(f" ... batch {batch_idx + 1}/{total_batches} ({ok} OK, {err} err)") print(f" {ok} OK, {err} errors") # Final counts print("\n── Final row counts ──") for table in ["packages", "receipts", "sidon_labels", "braid_strands", "eigensolid_snapshots", "gossip_surface_nodes", "gossip_surface_edges", "relations", "sessions", "ingest_events", "prover_state", "prover_instances"]: count = neon(f"SELECT COUNT(*) FROM ene.{table}") print(f" {table}: {count}") print("\nDone.") if __name__ == "__main__": main()