From 0409a80929351e25e6e83310a5e565cbed784fb3 Mon Sep 17 00:00:00 2001 From: Allaun Silverfox <28494262+allaunthefox@users.noreply.github.com> Date: Tue, 26 May 2026 14:46:13 -0500 Subject: [PATCH] feat(pist): add RRC receipt-density backfill injector --- .../shim/pist_receipt_density_injector.py | 392 ++++++++++++++++++ 1 file changed, 392 insertions(+) create mode 100644 4-Infrastructure/shim/pist_receipt_density_injector.py diff --git a/4-Infrastructure/shim/pist_receipt_density_injector.py b/4-Infrastructure/shim/pist_receipt_density_injector.py new file mode 100644 index 00000000..7b2a8e52 --- /dev/null +++ b/4-Infrastructure/shim/pist_receipt_density_injector.py @@ -0,0 +1,392 @@ +#!/usr/bin/env python3 +"""PIST -> RRC receipt-density backfill injector. + +This script converts existing RRC equation projection rows plus optional PIST +classification output into receipt-density records. + +It is intentionally conservative: + +* It DOES NOT promote equations. +* It DOES NOT mutate RDS/ENE by default. +* It writes an audit JSON that can later be consumed by a database writer. +* It filters Markdown table header/separator rows that older validation scripts + accidentally treated as equations. + +Default inputs: + + 6-Documentation/docs/rrc_equation_classification.md + shared-data/rrc_pist_exact_validation.json + +Default output: + + shared-data/rrc_receipt_density_backfill.json + +The output is a receipt-density surface, not a truth claim. A populated density +axis means "this equation has enough structural/witness evidence for routing"; +it does not mean the equation is mathematically proved. +""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import math +import re +import sys +from collections import Counter +from dataclasses import asdict, dataclass +from pathlib import Path +from typing import Any, Iterable + +REPO_ROOT = Path(__file__).resolve().parents[2] +DEFAULT_RRC_FILE = REPO_ROOT / "6-Documentation/docs/rrc_equation_classification.md" +DEFAULT_PIST_REPORT = REPO_ROOT / "shared-data/rrc_pist_exact_validation.json" +DEFAULT_OUT = REPO_ROOT / "shared-data/rrc_receipt_density_backfill.json" + +TARGET_AXES = { + "projection_declared", + "negative_control_strength", + "witness_declared", + "scale_band_declared", + "shape_closure", +} + +STATUS_BASE = { + "BLOCKED": 0.0, + "HOLD": 0.12, + "CANDIDATE": 0.45, + "REVIEWED": 0.78, + "VERIFIED": 0.84, +} + +SHAPE_DOMAIN = { + "CognitiveLoadField": "analysis", + "SignalShapedRouteCompiler": "topology", + "ProjectableGeometryTopology": "geometry", + "CadForceProbeReceipt": "physics", + "LogogramProjection": "symbolic", + "HoldForUnlawfulOrUnderspecifiedShape": "unknown", +} + + +@dataclass(frozen=True) +class RRCEquationRow: + equation_id: str + rrc_shape: str + status: str + top_axes: list[str] + + +@dataclass(frozen=True) +class ReceiptDensityRecord: + receipt_version: str + equation_id: str + rrc_shape: str + domain: str + source_status: str + receipt_density: float + confidence: float + density_components: dict[str, float] + shape_prediction: dict[str, Any] + top_axes: list[str] + status: str + promotion: str + source: str + receipt_hash: str + warnings: list[str] + + +def stable_hash(payload: Any) -> str: + canonical = json.dumps(payload, sort_keys=True, separators=(",", ":")) + return hashlib.sha256(canonical.encode("utf-8")).hexdigest() + + +def clamp01(value: float) -> float: + if math.isnan(value) or math.isinf(value): + return 0.0 + return max(0.0, min(1.0, value)) + + +def parse_axis_list(raw: str) -> list[str]: + return [part.strip().strip("`") for part in raw.split(",") if part.strip()] + + +def is_table_noise(equation: str, shape: str, status: str) -> bool: + bad = {"", "---", "Equation", "RRC shape", "Status"} + if equation in bad or shape in bad or status in bad: + return True + return bool(re.fullmatch(r"-+", equation)) or bool(re.fullmatch(r"-+", shape)) + + +def parse_rrc_table(path: Path) -> list[RRCEquationRow]: + if not path.exists(): + raise FileNotFoundError(f"RRC classification file not found: {path}") + + text = path.read_text(encoding="utf-8") + if "## Sample Projections" not in text: + raise ValueError(f"No '## Sample Projections' section found in {path}") + + sample_section = text.split("## Sample Projections", 1)[1] + sample_section = sample_section.split("\n## ", 1)[0] + + rows: list[RRCEquationRow] = [] + for line in sample_section.splitlines(): + line = line.strip() + if not line.startswith("|"): + continue + parts = [p.strip().strip("`") for p in line.strip("|").split("|")] + if len(parts) < 4: + continue + equation, shape, status, axes = parts[:4] + if is_table_noise(equation, shape, status): + continue + rows.append( + RRCEquationRow( + equation_id=equation, + rrc_shape=shape, + status=status, + top_axes=parse_axis_list(axes), + ) + ) + return rows + + +def load_pist_predictions(path: Path) -> dict[str, dict[str, Any]]: + if not path.exists(): + return {} + + data = json.loads(path.read_text(encoding="utf-8")) + predictions = data.get("predictions", []) + out: dict[str, dict[str, Any]] = {} + for pred in predictions: + equation = str(pred.get("equation", "")) + ground_truth = str(pred.get("ground_truth", "")) + if is_table_noise(equation, ground_truth, pred.get("proxy_pred", "")): + continue + out[equation] = pred + return out + + +def spectral_quality(pred: dict[str, Any] | None) -> float: + if not pred: + return 0.0 + + rank = float(pred.get("rank_estimate") or 0.0) + gap = float(pred.get("spectral_gap") or 0.0) + crossing_density = float(pred.get("crossing_density") or 0.0) + entropy = float(pred.get("strand_entropy") or 0.0) + lap_zero = float(pred.get("laplacian_zero_count") or 0.0) + + rank_score = clamp01(rank / 8.0) + gap_score = clamp01(gap) + entropy_score = clamp01(entropy / 3.0) + density_score = clamp01(crossing_density / 0.5) + lap_score = 1.0 if lap_zero >= 1.0 else 0.45 + hash_score = 1.0 if pred.get("canonical_hash") and pred.get("matrix_hash") else 0.0 + + return round( + clamp01( + 0.24 * rank_score + + 0.18 * gap_score + + 0.18 * entropy_score + + 0.12 * density_score + + 0.12 * lap_score + + 0.16 * hash_score + ), + 6, + ) + + +def shape_agreement(row: RRCEquationRow, pred: dict[str, Any] | None) -> float: + if not pred: + return 0.0 + exact = pred.get("exact_pred") + proxy = pred.get("proxy_pred") + if exact == row.rrc_shape: + return 1.0 + if proxy == row.rrc_shape: + return 0.82 + if exact or proxy: + return 0.35 + return 0.0 + + +def axis_score(row: RRCEquationRow) -> float: + if not row.top_axes: + return 0.0 + hits = len(TARGET_AXES.intersection(row.top_axes)) + return round(clamp01(hits / 4.0), 6) + + +def status_score(row: RRCEquationRow) -> float: + return STATUS_BASE.get(row.status.upper(), 0.2) + + +def compute_density(row: RRCEquationRow, pred: dict[str, Any] | None) -> tuple[float, float, dict[str, float], list[str]]: + warnings: list[str] = [] + s_status = status_score(row) + s_axes = axis_score(row) + s_spectral = spectral_quality(pred) + s_shape = shape_agreement(row, pred) + + if pred is None: + warnings.append("missing_pist_prediction") + elif s_shape < 0.5: + warnings.append("pist_shape_disagreement") + + # Receipt density is deliberately not a proof score. It measures routing + # evidence density from declared axes + status + structural PIST signal. + density = clamp01( + 0.26 * s_status + + 0.24 * s_axes + + 0.26 * s_spectral + + 0.24 * s_shape + ) + + # Confidence is slightly more classifier-weighted than density. + confidence = clamp01( + 0.20 * s_status + + 0.20 * s_axes + + 0.28 * s_spectral + + 0.32 * s_shape + ) + + components = { + "status_score": round(s_status, 6), + "axis_score": round(s_axes, 6), + "spectral_quality": round(s_spectral, 6), + "shape_agreement": round(s_shape, 6), + } + return round(density, 6), round(confidence, 6), components, warnings + + +def build_record(row: RRCEquationRow, pred: dict[str, Any] | None) -> ReceiptDensityRecord: + density, confidence, components, warnings = compute_density(row, pred) + shape_prediction = { + "ground_truth_hint": row.rrc_shape, + "proxy_pred": pred.get("proxy_pred") if pred else None, + "exact_pred": pred.get("exact_pred") if pred else None, + "matrix_hash": pred.get("matrix_hash") if pred else None, + "canonical_hash": pred.get("canonical_hash") if pred else None, + "spectral_gap": pred.get("spectral_gap") if pred else None, + "rank_estimate": pred.get("rank_estimate") if pred else None, + "laplacian_zero_count": pred.get("laplacian_zero_count") if pred else None, + } + + unsigned_payload = { + "equation_id": row.equation_id, + "rrc_shape": row.rrc_shape, + "source_status": row.status, + "receipt_density": density, + "confidence": confidence, + "shape_prediction": shape_prediction, + "top_axes": row.top_axes, + "promotion": "not_promoted", + "source": "pist_receipt_density_injector_v1", + } + receipt_hash = stable_hash(unsigned_payload) + + return ReceiptDensityRecord( + receipt_version="pist-receipt-density-v1", + equation_id=row.equation_id, + rrc_shape=row.rrc_shape, + domain=SHAPE_DOMAIN.get(row.rrc_shape, "unknown"), + source_status=row.status, + receipt_density=density, + confidence=confidence, + density_components=components, + shape_prediction=shape_prediction, + top_axes=row.top_axes, + status="CANDIDATE" if density > 0.0 else "HOLD", + promotion="not_promoted", + source="pist_receipt_density_injector_v1", + receipt_hash=receipt_hash, + warnings=warnings, + ) + + +def summarize(records: list[ReceiptDensityRecord], total_rows: int, prediction_count: int) -> dict[str, Any]: + by_shape = Counter(r.rrc_shape for r in records) + by_status = Counter(r.status for r in records) + warning_counts: Counter[str] = Counter() + for r in records: + warning_counts.update(r.warnings) + + densities = [r.receipt_density for r in records] + confidences = [r.confidence for r in records] + populated = sum(1 for r in records if r.receipt_density > 0.0) + + return { + "receipt_version": "pist-receipt-density-v1", + "input_rows": total_rows, + "records": len(records), + "pist_predictions_loaded": prediction_count, + "receipt_density_populated": populated, + "receipt_density_missing": len(records) - populated, + "coverage": round(populated / len(records), 6) if records else 0.0, + "mean_receipt_density": round(sum(densities) / len(densities), 6) if densities else 0.0, + "min_receipt_density": round(min(densities), 6) if densities else 0.0, + "max_receipt_density": round(max(densities), 6) if densities else 0.0, + "mean_confidence": round(sum(confidences) / len(confidences), 6) if confidences else 0.0, + "by_shape": dict(sorted(by_shape.items())), + "by_status": dict(sorted(by_status.items())), + "warning_counts": dict(sorted(warning_counts.items())), + "promotion_policy": "no automatic promotion; density populates routing evidence only", + } + + +def emit_jsonl(records: Iterable[ReceiptDensityRecord], path: Path) -> None: + with path.open("w", encoding="utf-8") as f: + for record in records: + f.write(json.dumps(asdict(record), sort_keys=True) + "\n") + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description="Generate RRC receipt-density records from PIST outputs.") + parser.add_argument("--rrc-file", type=Path, default=DEFAULT_RRC_FILE) + parser.add_argument("--pist-report", type=Path, default=DEFAULT_PIST_REPORT) + parser.add_argument("--out", type=Path, default=DEFAULT_OUT) + parser.add_argument("--jsonl-out", type=Path, default=None, help="Optional JSONL output path for DB import.") + parser.add_argument("--fail-on-missing-pist", action="store_true", help="Exit nonzero if any row lacks a PIST prediction.") + args = parser.parse_args(argv) + + rows = parse_rrc_table(args.rrc_file) + predictions = load_pist_predictions(args.pist_report) + + records = [build_record(row, predictions.get(row.equation_id)) for row in rows] + summary = summarize(records, total_rows=len(rows), prediction_count=len(predictions)) + + payload = { + "summary": summary, + "inputs": { + "rrc_file": str(args.rrc_file), + "pist_report": str(args.pist_report), + }, + "claim_boundary": { + "receipt_density_means": "routing evidence is populated", + "receipt_density_does_not_mean": "mathematical proof or promotion", + "promotion_policy": "not_promoted for every generated record", + }, + "records": [asdict(r) for r in records], + } + + args.out.parent.mkdir(parents=True, exist_ok=True) + args.out.write_text(json.dumps(payload, indent=2, sort_keys=True) + "\n", encoding="utf-8") + + if args.jsonl_out is not None: + args.jsonl_out.parent.mkdir(parents=True, exist_ok=True) + emit_jsonl(records, args.jsonl_out) + + print(json.dumps(summary, indent=2, sort_keys=True)) + print(f"Wrote audit JSON: {args.out}", file=sys.stderr) + if args.jsonl_out is not None: + print(f"Wrote JSONL import file: {args.jsonl_out}", file=sys.stderr) + + if args.fail_on_missing_pist and summary["warning_counts"].get("missing_pist_prediction", 0) > 0: + return 2 + return 0 + + +if __name__ == "__main__": + raise SystemExit(main())