Research-Stack/4-Infrastructure/shim/rrc_dataset_kernel_build.py
allaun 77488ac0ae feat(lean): close gaussian_line_integral_unit_dir + consolidate infrastructure
Lean proof fixes:
- N3L_Energy.lean: fully close gaussian_line_integral_unit_dir (nlinarith+hab
  for unit-circle quadratic, sqrt_mul+neg_div for integral_gaussian_1d match,
  exp_sum_of_sq order fix, add_assoc for h_gauss_shift, sq_sqrt for field_simp,
  sq_abs for perpDistance hd)
- Add Adapters/AlphaProofNexus: 12 Erdos/graph adapter stubs (AlphaProof nexus)
- Add Adapters/ErgodicAdditive.lean, SidonMatroid.lean
- Add AntiDiophantine.lean, EffectiveBoundDQ.lean, PVGS_DQ_Bridge.lean
- Add FormalConjectures/Util/ProblemImports.lean
- Add RRC/EntropyCandidates/Candidates.lean
- Add OTOM external project (lakefile.toml, lake-manifest.json, lean-toolchain)

Infrastructure:
- Add 4-Infrastructure/shim/: 17 Python probes (RRC manifold, Sidon kernel,
  Wannier, arxiv harvest, math_symbols DB, coverage density, geometric entropy)
- Add 4-Infrastructure/NoDupeLabs/: Node server + package files
- Add 6-Documentation/docs/specs/DP_RRC_RECEIPT_ENCODING_SPEC.md
- Add fix_offloat.py

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-06-18 16:53:23 -05:00

258 lines
11 KiB
Python

#!/usr/bin/env python3
"""
rrc_dataset_kernel_build.py — Build kernels from math datasets for RRC pipeline.
Consumes:
- Big-Math-RL-Verified.parquet (251K rows, domain taxonomy + solve rates)
- AutoMathText_web.parquet (851K rows, web math corpus)
- TheoremQA.json (800 rows, theorem QA pairs)
Outputs:
- shared-data/data/domain_kernel_v1.json — domain taxonomy kernel
- shared-data/data/webmath_kernel_v1.json — web math pattern kernel
- shared-data/data/theorem_kernel_v1.json — theorem QA kernel
Usage:
python3 4-Infrastructure/shim/rrc_dataset_kernel_build.py
"""
from __future__ import annotations
import json
import re
import sys
from collections import Counter, defaultdict
from pathlib import Path
import numpy as np
import pandas as pd
ROOT = Path(__file__).resolve().parents[2]
DATA = ROOT / "shared-data" / "data" / "math-datasets"
OUT = ROOT / "shared-data" / "data"
STOPWORDS = {
"the", "and", "for", "where", "with", "this", "from", "that", "are",
"but", "not", "have", "has", "been", "was", "were", "will", "would",
"could", "should", "their", "them", "they", "its", "also", "can",
"may", "however", "thus", "proof", "theorem", "lemma", "corollary",
"proposition", "function", "functions", "using", "used", "use",
"given", "show", "shows", "paper", "result", "results", "method",
"methods", "well", "first", "new", "one", "two", "three",
"equation", "equations", "find", "value", "values", "let",
}
# ─────────────────────────────────────────────────────────────────────────────
# 1. Domain kernel (Big-Math-RL-Verified)
# ─────────────────────────────────────────────────────────────────────────────
def build_domain_kernel(df: pd.DataFrame) -> dict:
"""Build a domain taxonomy kernel from Big-Math-RL-Verified."""
# Extract domain paths → problem keywords
domain_problems: dict[str, list[str]] = defaultdict(list)
domain_stats: dict[str, dict] = defaultdict(lambda: {"count": 0, "avg_solve_rate": 0.0, "sources": set()})
for _, row in df.iterrows():
problem = str(row.get("problem", ""))
solve_rate = float(row.get("llama8b_solve_rate", 0))
source = str(row.get("source", ""))
domains_raw = row.get("domain", [])
if isinstance(domains_raw, np.ndarray):
for d in domains_raw:
d_str = str(d)
if d_str and d_str != "nan":
domain_problems[d_str].append(problem)
s = domain_stats[d_str]
s["count"] += 1
# Running average
n = s["count"]
s["avg_solve_rate"] = (s["avg_solve_rate"] * (n - 1) + solve_rate) / n
s["sources"].add(source)
# Build domain hierarchy and patterns
domains = []
for d_path in sorted(domain_problems.keys()):
parts = [p.strip() for p in d_path.split("->")]
stats = domain_stats[d_path]
# Extract keyword patterns from problem texts
problems = domain_problems[d_path]
all_text = " ".join(problems).lower()
tokens = re.findall(r"[a-z][a-z-]{2,}", all_text)
freq = Counter(t for t in tokens if t not in STOPWORDS)
top_kws = [kw for kw, _ in freq.most_common(10)]
domains.append({
"path": d_path,
"parts": parts,
"root": parts[0] if parts else "",
"leaf": parts[-1] if parts else "",
"count": stats["count"],
"avg_solve_rate": round(stats["avg_solve_rate"], 4),
"sources": list(stats["sources"]),
"keywords": top_kws,
})
return {
"schema": "domain_kernel_v1",
"source": "Big-Math-RL-Verified (251K rows)",
"domain_count": len(domains),
"root_categories": sorted(set(d["root"] for d in domains)),
"domains": sorted(domains, key=lambda x: -x["count"]),
}
# ─────────────────────────────────────────────────────────────────────────────
# 2. Web math kernel (AutoMathText)
# ─────────────────────────────────────────────────────────────────────────────
def build_webmath_kernel(df: pd.DataFrame, sample: int = 50000) -> dict:
"""Build web math pattern kernel from AutoMathText."""
# Sample to keep it fast
if len(df) > sample:
df = df.sample(sample, random_state=42)
# Extract equation patterns from web text
# Pattern types: inline math $...$, display math $$...$$, LaTeX equations
eq_patterns = re.compile(r"\$\$[^$]+\$\$|\$[^$]{4,200}\$|\\\\[[a-zA-Z]+|\\\\[[a-zA-Z]+")
math_patterns: dict[str, int] = Counter()
domain_urls: dict[str, list[str]] = defaultdict(list)
for _, row in df.iterrows():
text = str(row.get("text", ""))
url = str(row.get("url", ""))
meta = row.get("meta", {})
score = meta.get("openwebmath_score", 0) if isinstance(meta, dict) else 0
if score < 0.5:
continue
# Find LaTeX math patterns
found = eq_patterns.findall(text)
for m in found[:5]: # limit per doc
# Hash to pattern type
m_clean = re.sub(r"[0-9]+", "N", m)[:80]
math_patterns[m_clean] += 1
# Extract domain from URL
domain = url.split("/")[2] if "//" in url else "unknown"
domain_urls[domain].append(text[:200])
# Build the kernel
top_patterns = [{"pattern": p, "count": c} for p, c in math_patterns.most_common(50)]
return {
"schema": "webmath_kernel_v1",
"source": "AutoMathText_web (sampled 50K from 851K)",
"sampled_rows": sample,
"total_math_patterns": len(math_patterns),
"top_domains": sorted(
[{"domain": d, "count": len(u)} for d, u in domain_urls.items()],
key=lambda x: -x["count"],
)[:20],
"patterns": top_patterns,
}
# ─────────────────────────────────────────────────────────────────────────────
# 3. Theorem kernel (TheoremQA)
# ─────────────────────────────────────────────────────────────────────────────
def build_theorem_kernel(data: list) -> dict:
"""Build theorem QA kernel from TheoremQA."""
theorems = []
for item in data:
q = str(item.get("Question", ""))
a = str(item.get("Answer", ""))
at = str(item.get("Answer_type", ""))
# Extract keywords from the question
tokens = re.findall(r"[a-z][a-z-]{2,}", q.lower())
freq = Counter(t for t in tokens if t not in STOPWORDS)
kws = [kw for kw, _ in freq.most_common(8)]
# Detect the kind of math in the question
kind = detect_theorem_kind(q)
theorems.append({
"question": q[:200],
"answer": a[:100],
"answer_type": at,
"keywords": kws,
"kind": kind,
})
# Build kind-based index
by_kind: dict[str, list[str]] = defaultdict(list)
for t in theorems:
by_kind[t["kind"]].append(t["question"][:120])
return {
"schema": "theorem_kernel_v1",
"source": "TheoremQA (800 rows)",
"count": len(theorems),
"kinds": [{"kind": k, "count": len(v), "examples": v[:3]} for k, v in sorted(by_kind.items())],
"theorems": sorted(theorems, key=lambda x: -len(x["keywords"])),
}
def detect_theorem_kind(q: str) -> str:
ql = q.lower()
if any(kw in ql for kw in ["graph", "vertex", "edge", "tree", "chromatic", "matching"]):
return "graph_theory"
if any(kw in ql for kw in ["prime", "divisor", "gcd", "lcm", "modulo", "congruence"]):
return "number_theory"
if any(kw in ql for kw in ["matrix", "determinant", "eigenvalue", "vector space", "linear"]):
return "linear_algebra"
if any(kw in ql for kw in ["group", "ring", "field", "ideal", "module"]):
return "abstract_algebra"
if any(kw in ql for kw in ["integral", "derivative", "limit", "series", "converge"]):
return "calculus_analysis"
if any(kw in ql for kw in ["probability", "expectation", "variance", "random"]):
return "probability"
if any(kw in ql for kw in ["set", "subset", "union", "intersection", "cardinal"]):
return "set_theory"
if any(kw in ql for kw in ["combinatorics", "permutation", "combination", "binomial"]):
return "combinatorics"
if any(kw in ql for kw in ["geometry", "triangle", "circle", "angle", "polygon"]):
return "geometry"
return "other"
# ─────────────────────────────────────────────────────────────────────────────
# Main
# ─────────────────────────────────────────────────────────────────────────────
def main():
OUT.mkdir(parents=True, exist_ok=True)
print("=" * 60, file=sys.stderr)
print("RRC Dataset Kernel Build", file=sys.stderr)
print("=" * 60, file=sys.stderr)
# 1. Domain kernel
print("\n[1/3] Building domain kernel from Big-Math-RL-Verified...", file=sys.stderr)
df_math = pd.read_parquet(DATA / "Big-Math-RL-Verified.parquet")
domain_kernel = build_domain_kernel(df_math)
(OUT / "domain_kernel_v1.json").write_text(json.dumps(domain_kernel, indent=2))
print(f" {domain_kernel['domain_count']} domains indexed", file=sys.stderr)
# 2. Web math kernel
print("\n[2/3] Building web math kernel from AutoMathText...", file=sys.stderr)
df_web = pd.read_parquet(DATA / "AutoMathText_web.parquet")
web_kernel = build_webmath_kernel(df_web, sample=50000)
(OUT / "webmath_kernel_v1.json").write_text(json.dumps(web_kernel, indent=2))
print(f" {web_kernel['total_math_patterns']} math patterns found", file=sys.stderr)
# 3. Theorem kernel
print("\n[3/3] Building theorem kernel from TheoremQA...", file=sys.stderr)
theorem_data = json.loads((DATA / "TheoremQA.json").read_text())
theorem_kernel = build_theorem_kernel(theorem_data)
(OUT / "theorem_kernel_v1.json").write_text(json.dumps(theorem_kernel, indent=2))
for k in theorem_kernel["kinds"]:
print(f" {k['kind']:25s} {k['count']} theorems", file=sys.stderr)
print("\nDone. Kernels written to shared-data/data/", file=sys.stderr)
print(f" domain_kernel_v1.json — {domain_kernel['domain_count']} domains", file=sys.stderr)
print(f" webmath_kernel_v1.json — {web_kernel['total_math_patterns']} patterns", file=sys.stderr)
print(f" theorem_kernel_v1.json — {len(theorem_kernel['theorems'])} theorems", file=sys.stderr)
if __name__ == "__main__":
main()