SilverSight/python/spectral_codebook_db.py
allaun e81caa4a6a feat(spectral): Neon ENE sync layer for the spectral codebook (dry-run by default)
Adds python/spectral_codebook_db.py: sync codebook rows into
ene.rrc_predictions on the neon-64gb Postgres (NEON_PG convention from
scripts/auto/auto_pipeline.py, default research_stack DB).

- Row shape (flat, SQL-typed, Spark-JDBC readable): equation_id,
  proxy_pred = cluster codeword C0..C8, exact_pred = shape from exact
  lambda under CURRENT ClassifyN.lean thresholds (1.5/4.0 Q16.16,
  integer semantics mirrored), matrix_hash =
  'charpoly=<c1..c8>;pos10=<base-10 positional hash>' (similarity +
  injective identity keys), confidence = 1.0 unique fingerprint / 1/k
  in k-way charpoly collision class; deterministic uuid5 ids so reruns
  upsert idempotently.
- SAFE BY DEFAULT: dry run prints summary + sample SQL and writes
  nothing; --apply required to insert (psycopg2, with --emit-sql
  data/spectral_codebook_sync.sql fallback when the driver is absent).
  --apply has NOT been run; live DB untouched. --verify-schema does a
  read-only column check; schema verified offline against
  scripts/auto/ene_schema.sql in tests (live check left to the user
  per the ask-before-DB-work rule).
- spectral_codebook.py gains --sync-db (always dry-run from that entry
  point). Dry-run counts: 250 rows; C0=35 C1=20 C2=13 C3=79 C4=29
  C5=15 C6=22 C7=19 C8=18; Logogram=69 Signal=131 CognitiveLoad=50;
  183 rows at confidence 1.0.
- docs: 'Neon data layer' section — ENE table map, stale
  ene.rrc_classifications finding (120 rows with artifact spectral
  radii 0.3-0.85 predating the exact-eigenvalue fix; recommend
  re-classification via this codebook), empty landing tables, Spark
  JDBC snippet, arxiv-pg (podman-exec only) citation layer note.
- tests: 8 new dry-run/no-network tests (23 total).

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-01 20:39:40 -05:00

278 lines
11 KiB
Python

#!/usr/bin/env python3
# /// script
# requires-python = ">=3.10"
# dependencies = []
# ///
"""
spectral_codebook_db.py — Sync the spectral codebook into the Neon ENE layer.
Writes one row per equation into `ene.rrc_predictions` on the neon-64gb
Postgres (schema in scripts/auto/ene_schema.sql):
id uuid — deterministic uuid5 of the equation_id (idempotent reruns)
equation_id text
proxy_pred text — gap-aware cluster codeword ("C0".."C8")
exact_pred text — exact-λ shape under the CURRENT ClassifyN.lean
thresholds (signalThreshold 1.5 / oberthHigh 4.0,
Q16.16 integer semantics mirrored exactly)
matrix_hash text — canonical fingerprint string:
"charpoly=<c1,..,c8>;pos10=<base-10 positional hash>"
charpoly = exact integer char-poly coefficients
(similarity key); pos10 = base-10 positional hash
(identity key — injective since entries ≤ 9,
unlike ClassifyN.hashMatrix's base 5)
confidence float — 1.0 for a unique charpoly fingerprint,
1/k inside a k-way charpoly collision class
SAFETY: the default is a DRY RUN — it prints the row summary and sample SQL
and writes NOTHING. `--apply` performs the insert via psycopg2; if psycopg2
is not importable, `--apply` (or `--emit-sql`) writes
data/spectral_codebook_sync.sql for `psql $NEON_PG/research_stack -f ...`.
The row shape is flat/SQL-typed so the Spark cluster on neon-64gb
(spark://100.92.88.64:7077, postgresql-42.7.5 JDBC) can read
ene.rrc_predictions directly — see docs/SPECTRAL_CODEBOOK_GENERATOR.md.
Usage:
python3 python/spectral_codebook_db.py # dry run
python3 python/spectral_codebook_db.py --verify-schema # + read-only DB check
python3 python/spectral_codebook_db.py --emit-sql # write .sql, no DB
python3 python/spectral_codebook_db.py --apply # actually insert
"""
from __future__ import annotations
import argparse
import os
import sys
import uuid
from pathlib import Path
from typing import Dict, List, Optional, Sequence
sys.path.insert(0, str(Path(__file__).resolve().parent))
import spectral_codebook as sc
NEON_PG_DEFAULT = "postgres://postgres:postgres@100.92.88.64:5432/research_stack"
DEFAULT_SQL_OUT = sc.REPO_ROOT / "data" / "spectral_codebook_sync.sql"
# Q16.16 thresholds from formal/SilverSight/PIST/ClassifyN.lean
Q16_SCALE = 65536
SIGNAL_THRESHOLD_RAW = 98304 # 1.5
OBERTH_HIGH_RAW = 262144 # 4.0
# uuid5 namespace for deterministic row ids (rerun-idempotent)
ROW_NS = uuid.uuid5(uuid.NAMESPACE_URL, "silversight/ene/rrc_predictions")
def neon_dsn() -> str:
"""NEON_PG env convention (scripts/auto/auto_pipeline.py); default db
research_stack is appended when the env var carries only host:port."""
dsn = os.environ.get("NEON_PG", NEON_PG_DEFAULT)
if dsn.rstrip("/").count("/") < 3: # postgres://user:pw@host:port (no db)
dsn = dsn.rstrip("/") + "/research_stack"
return dsn
def classify_exact_shape(lam: float) -> str:
"""Mirror ClassifyN.lean's spectralRadiusToColor → colorToShapeName on
the exact spectral radius, using the CURRENT Lean thresholds (1.5/4.0
Q16.16) with integer Q16.16 semantics."""
raw = int(lam * Q16_SCALE) # truncation, as in Q16.16 fixed point
if raw >= OBERTH_HIGH_RAW:
return "CognitiveLoadField" # red > 0
if raw >= SIGNAL_THRESHOLD_RAW:
green = (raw - SIGNAL_THRESHOLD_RAW) * Q16_SCALE // (OBERTH_HIGH_RAW - SIGNAL_THRESHOLD_RAW)
if green > 0:
return "SignalShapedRouteCompiler"
return "LogogramProjection" # blue-or-zero branch
def positional_hash(mat: Sequence[Sequence[int]], base: int = 10) -> int:
"""Positional hash in the loop order of ClassifyN.hashMatrix, but with a
base ≥ max_entry + 1 (corpus entries reach 9), which makes it injective
on entry patterns — ClassifyN's base 5 is not."""
acc, pw = 0, 1
for row in mat:
for val in row:
acc += val * pw
pw *= base
return acc
def matrix_hash(charpoly: Sequence[int], mat: Sequence[Sequence[int]]) -> str:
cp = ",".join(str(c) for c in charpoly)
return f"charpoly={cp};pos10={positional_hash(mat)}"
def build_rows(codebook: sc.Codebook) -> List[dict]:
"""Flat SQL-typed rows for ene.rrc_predictions, one per equation."""
# k-way charpoly collision classes → confidence 1/k
cp_size: Dict[tuple, int] = {}
for p in codebook.profiles.values():
key = tuple(p["charpoly"])
cp_size[key] = cp_size.get(key, 0) + 1
rows = []
for eid in sorted(codebook.profiles):
p = codebook.profiles[eid]
cw, _idx = codebook.codeword_of(eid)
k = cp_size[tuple(p["charpoly"])]
rows.append({
"id": str(uuid.uuid5(ROW_NS, eid)),
"equation_id": eid,
"proxy_pred": cw,
"exact_pred": classify_exact_shape(p["spectral_radius"]),
"matrix_hash": matrix_hash(p["charpoly"], codebook.matrices[eid]),
"confidence": round(1.0 / k, 6),
})
return rows
INSERT_SQL = (
"INSERT INTO ene.rrc_predictions "
"(id, equation_id, proxy_pred, exact_pred, matrix_hash, confidence) "
"VALUES (%(id)s, %(equation_id)s, %(proxy_pred)s, %(exact_pred)s, "
"%(matrix_hash)s, %(confidence)s) "
"ON CONFLICT (id) DO UPDATE SET "
"proxy_pred = EXCLUDED.proxy_pred, exact_pred = EXCLUDED.exact_pred, "
"matrix_hash = EXCLUDED.matrix_hash, confidence = EXCLUDED.confidence, "
"predicted_at = NOW()"
)
def _sql_literal(v) -> str:
if isinstance(v, str):
return "'" + v.replace("'", "''") + "'"
return str(v)
def render_sql(rows: List[dict]) -> str:
"""Standalone .sql rendering of the same upsert (for psql, no driver)."""
lines = [
"-- spectral_codebook → ene.rrc_predictions sync",
"-- generated by python/spectral_codebook_db.py (schema: scripts/auto/ene_schema.sql)",
"BEGIN;",
]
for r in rows:
vals = ", ".join(_sql_literal(r[c]) for c in
("id", "equation_id", "proxy_pred", "exact_pred",
"matrix_hash", "confidence"))
lines.append(
"INSERT INTO ene.rrc_predictions "
"(id, equation_id, proxy_pred, exact_pred, matrix_hash, confidence) "
f"VALUES ({vals}) "
"ON CONFLICT (id) DO UPDATE SET "
"proxy_pred = EXCLUDED.proxy_pred, exact_pred = EXCLUDED.exact_pred, "
"matrix_hash = EXCLUDED.matrix_hash, confidence = EXCLUDED.confidence, "
"predicted_at = NOW();"
)
lines.append("COMMIT;")
return "\n".join(lines) + "\n"
def summarize(rows: List[dict]) -> str:
by_cluster: Dict[str, int] = {}
by_shape: Dict[str, int] = {}
for r in rows:
by_cluster[r["proxy_pred"]] = by_cluster.get(r["proxy_pred"], 0) + 1
by_shape[r["exact_pred"]] = by_shape.get(r["exact_pred"], 0) + 1
unique = sum(1 for r in rows if r["confidence"] == 1.0)
out = [f"{len(rows)} rows → ene.rrc_predictions"]
out.append(" per cluster (proxy_pred): " + ", ".join(
f"{k}={by_cluster[k]}" for k in sorted(by_cluster, key=lambda c: int(c[1:]))))
out.append(" per shape (exact_pred): " + ", ".join(
f"{k}={v}" for k, v in sorted(by_shape.items())))
out.append(f" confidence: {unique} unique fingerprints at 1.0, "
f"{len(rows) - unique} in collision classes (<1.0)")
return "\n".join(out)
def verify_schema(dsn: str) -> List[str]:
"""READ-ONLY check that ene.rrc_predictions has the expected columns."""
import psycopg2 # only needed for this optional check
expected = {"id", "equation_id", "proxy_pred", "exact_pred",
"matrix_hash", "confidence", "predicted_at"}
with psycopg2.connect(dsn) as conn:
conn.set_session(readonly=True)
with conn.cursor() as cur:
cur.execute(
"SELECT column_name FROM information_schema.columns "
"WHERE table_schema = 'ene' AND table_name = 'rrc_predictions'"
)
cols = {r[0] for r in cur.fetchall()}
missing = sorted(expected - cols)
if missing:
raise RuntimeError(f"ene.rrc_predictions missing columns: {missing}")
return sorted(cols)
def apply_rows(dsn: str, rows: List[dict]) -> int:
import psycopg2
with psycopg2.connect(dsn) as conn:
with conn.cursor() as cur:
cur.executemany(INSERT_SQL, rows)
conn.commit()
return len(rows)
def sync_db(codebook: Optional[sc.Codebook] = None,
apply: bool = False,
emit_sql: Optional[Path] = None,
verify: bool = False,
dsn: Optional[str] = None) -> List[dict]:
"""Build rows and dry-run/emit/apply. Returns the rows.
Dry run (apply=False, emit_sql=None) touches neither network nor disk.
"""
cb = codebook or sc.build_codebook()
rows = build_rows(cb)
print(summarize(rows))
if verify:
cols = verify_schema(dsn or neon_dsn())
print(f" schema check (read-only): ene.rrc_predictions columns OK: {cols}")
if apply:
try:
import psycopg2 # noqa: F401
except ImportError:
emit_sql = emit_sql or DEFAULT_SQL_OUT
print(" psycopg2 not importable — falling back to SQL emission")
else:
n = apply_rows(dsn or neon_dsn(), rows)
print(f" APPLIED: upserted {n} rows into ene.rrc_predictions")
return rows
if emit_sql is not None:
emit_sql.parent.mkdir(parents=True, exist_ok=True)
emit_sql.write_text(render_sql(rows))
print(f" wrote {emit_sql} "
f"(psql $NEON_PG/research_stack -f {emit_sql.name})")
elif not apply:
print(" DRY RUN — nothing written. Sample statement:")
print(" " + render_sql(rows[:1]).splitlines()[3])
print(" Use --apply to insert, --emit-sql to write "
f"{DEFAULT_SQL_OUT.relative_to(sc.REPO_ROOT)}")
return rows
def main(argv: Optional[List[str]] = None) -> int:
ap = argparse.ArgumentParser(
description="Sync spectral codebook → ene.rrc_predictions (dry-run by default)")
ap.add_argument("--apply", action="store_true",
help="Actually upsert rows (default: dry run, writes nothing)")
ap.add_argument("--emit-sql", nargs="?", type=Path, const=DEFAULT_SQL_OUT,
default=None, metavar="PATH",
help=f"Write upsert SQL to PATH (default {DEFAULT_SQL_OUT}) "
"instead of connecting")
ap.add_argument("--verify-schema", action="store_true",
help="Read-only check of ene.rrc_predictions columns")
ap.add_argument("--dsn", default=None,
help="Postgres DSN (default: $NEON_PG or the neon-64gb URL)")
args = ap.parse_args(argv)
sync_db(apply=args.apply, emit_sql=args.emit_sql,
verify=args.verify_schema, dsn=args.dsn)
return 0
if __name__ == "__main__":
sys.exit(main())