Research-Stack/5-Applications/scripts/q16_lut_generator.py
Devin AI 0c9efac330 chore(consolidation): integrate E8Sidon stack (PRs #79 #80 #81 #89) into one PR
Squash the four overlapping feature branches into a single change set against
main, eliminating cross-PR merge conflicts and the duplicated CI-fix scripts.

What this brings in (merge order #79 -> #80 -> #81 -> #89):
- #79 refactor(infra): shared utilities (4-Infrastructure/lib/*: q16, hashing,
  jsonl, fraction_utils) + the scripts/math-first/* validators that the
  math-check CI requires.
- #80 feat(lean): Semantics.E8Sidon (1025 lines) -- Eisenstein coefficient
  identity E4^2 = E8 and the Sidon framework. E4_sq_eq_E8_coeff is fully proved
  (all Fourier-coefficient extraction machine-checked); the single residual gap
  is pinned to E4_sq_eq_E8_qExpansion (Mathlib lacks the valence formula /
  dim M8 = 1). 4 sorries + 1 axiom (e8_additive_completeness), all TODO(lean-port).
- #81 refactor(lean): Float-free FixedPoint core (integer-only sqrt/log2/expNeg).
  E8Sidon.lean kept at #80's final 1025-line version (the #81 intermediate
  438-line copy was overridden by merge order).
- #89 feat(lean): Semantics.RRC.PolyFactorIdentity -- short-sleeve polynomial
  detection at the zerocopy limb boundary; now imports Semantics.E8Sidon for
  sigma3/sigma7/convolutionLHS (single source of truth) instead of inlining them.

Conflict resolution:
- flake.nix -> canonical rs-surface removal (Garnix shutdown).
- scripts/math-first/* -> byte-identical across branches, clean.
- .cursorrules / AGENTS.md -> unified; baselines + sorry inventory refreshed.

Verification:
- lake build (default aggregator): 3573 jobs, 0 errors.
- lake build Semantics.RRC.PolyFactorIdentity (E8Sidon + FixedPoint + PolyFactor):
  3655 jobs, 0 errors. Witnesses verified (sigma7 4 = 16513, convolutionLHS 6 = 2350).
- Python tests: 68/68 pass.

Note: the "Workers Builds: researchstack" check is a preexisting external
Cloudflare build unrelated to this change (no branch touches 4-Infrastructure/cloudflare/).

Build: 3573 jobs (default), 3655 jobs (narrow), 0 errors
Co-Authored-By: Allaun Silverfox <bigdataiscoming+9i37y6j2@protonmail.com>
2026-06-16 02:01:31 +00:00

183 lines
5.7 KiB
Python

#!/usr/bin/env python3
"""
Q16_16 LUT Generator - Python fallback for LUT generation when WGSL runtime unavailable.
This script computes all combinations of Q16_16 operations and generates a lookup table.
For each pair (a, b) in Q16_16 space (65,536² = 4,294,967,296 combinations), it computes:
- add(a, b)
- sub(a, b)
- mul(a, b)
- div(a, b) (when b != 0)
- max(a, b)
- min(a, b)
- neg(a)
- abs(a)
The output is a packed LUT that can be used for fast hardware implementation.
"""
import numpy as np
import json
from pathlib import Path
def q16_add(a, b):
"""Q16_16 addition (wrapping)."""
return (a + b) & 0xFFFFFFFF
def q16_sub(a, b):
"""Q16_16 subtraction (wrapping)."""
return (a - b) & 0xFFFFFFFF
def q16_mul(a, b):
"""Q16_16 multiplication (high 32 bits of 64-bit product)."""
prod = (a * b) >> 16
return prod & 0xFFFFFFFF
def q16_div(a, b):
"""Q16_16 division (with division by zero handling)."""
if b == 0:
return 0xFFFFFFFF
numerator = (a << 16) & 0xFFFFFFFFFFFFFFFF
return (numerator // b) & 0xFFFFFFFF
def q16_max(a, b):
"""Q16_16 maximum (unsigned comparison)."""
return a if a > b else b
def q16_min(a, b):
"""Q16_16 minimum (unsigned comparison)."""
return a if a < b else b
def q16_neg(a):
"""Q16_16 negation (two's complement)."""
return (-a) & 0xFFFFFFFF
def q16_abs(a):
"""Q16_16 absolute value (sign bit check)."""
if a & 0x80000000:
return q16_neg(a)
return a
def generate_lut_sample():
"""Generate a sample LUT (not full 4.3B combinations due to memory constraints)."""
Q16_SPACE = 65536
SAMPLE_SIZE = 1000 # Sample size for demonstration
print(f"Generating sample Q16_16 LUT ({SAMPLE_SIZE} combinations)...")
print(f"Full LUT would require {Q16_SPACE * Q16_SPACE:,} combinations (4.3 billion)")
print(f"Full LUT size: {(Q16_SPACE * Q16_SPACE * 32):,} bytes ({(Q16_SPACE * Q16_SPACE * 32 / (1024**3)):.2f} GB)")
# Generate sample LUT
lut = []
for i in range(SAMPLE_SIZE):
a = (i * 137) % Q16_SPACE # Pseudo-random a
b = (i * 251) % Q16_SPACE # Pseudo-random b
entry = {
'a': a,
'b': b,
'add': q16_add(a, b),
'sub': q16_sub(a, b),
'mul': q16_mul(a, b),
'div': q16_div(a, b),
'max': q16_max(a, b),
'min': q16_min(a, b),
'neg': q16_neg(a),
'abs': q16_abs(a)
}
lut.append(entry)
# Save sample LUT
output_path = Path('/home/allaun/Documents/Research Stack/out/q16_lut_sample.json')
with open(output_path, 'w') as f:
json.dump({
'metadata': {
'q16_space': Q16_SPACE,
'total_combinations': Q16_SPACE * Q16_SPACE,
'sample_size': SAMPLE_SIZE,
'full_lut_size_bytes': Q16_SPACE * Q16_SPACE * 32,
'full_lut_size_gb': (Q16_SPACE * Q16_SPACE * 32) / (1024**3)
},
'lut': lut
}, f, indent=2)
print(f"Sample LUT saved to {output_path}")
print(f"Sample entries: {len(lut)}")
# Statistics
adds = [e['add'] for e in lut]
muls = [e['mul'] for e in lut]
divs = [e['div'] for e in lut]
print(f"\nStatistics:")
print(f" Add range: [{min(adds)}, {max(adds)}]")
print(f" Mul range: [{min(muls)}, {max(muls)}]")
print(f" Div by zero: {divs.count(0xFFFFFFFF)}")
return lut
def generate_full_lut_chunked():
"""
Generate full LUT in chunks to avoid memory issues.
This would generate the actual 4.3B combinations but is disabled by default.
"""
Q16_SPACE = 65536
CHUNK_SIZE = 10000 # Process in chunks
print(f"Generating full Q16_16 LUT in chunks of {CHUNK_SIZE}...")
print(f"Total combinations: {Q16_SPACE * Q16_SPACE:,}")
output_dir = Path('/home/allaun/Documents/Research Stack/out/q16_lut_chunks')
output_dir.mkdir(exist_ok=True)
chunk_count = 0
for chunk_start in range(0, Q16_SPACE * Q16_SPACE, CHUNK_SIZE):
chunk_end = min(chunk_start + CHUNK_SIZE, Q16_SPACE * Q16_SPACE)
lut_chunk = []
for idx in range(chunk_start, chunk_end):
a = idx // Q16_SPACE
b = idx % Q16_SPACE
entry = {
'a': a,
'b': b,
'add': q16_add(a, b),
'sub': q16_sub(a, b),
'mul': q16_mul(a, b),
'div': q16_div(a, b),
'max': q16_max(a, b),
'min': q16_min(a, b),
'neg': q16_neg(a),
'abs': q16_abs(a)
}
lut_chunk.append(entry)
# Save chunk
chunk_file = output_dir / f'chunk_{chunk_count:06d}.json'
with open(chunk_file, 'w') as f:
json.dump(lut_chunk, f)
chunk_count += 1
if chunk_count % 100 == 0:
print(f" Processed {chunk_end:,} / {Q16_SPACE * Q16_SPACE:,} combinations ({chunk_end / (Q16_SPACE * Q16_SPACE) * 100:.2f}%)")
print(f"Full LUT saved to {output_dir} in {chunk_count} chunks")
def main():
print("Q16_16 LUT Generator")
print("=" * 50)
# Generate sample LUT (for demonstration)
generate_lut_sample()
# Note: Full LUT generation is disabled by default due to memory constraints
# To generate the full 4.3B combination LUT, uncomment the following line:
# generate_full_lut_chunked()
print("\nNote: Full LUT generation (4.3B combinations) requires ~137 GB of storage.")
print("Use the WGSL shader (q16_lut_generator.wgsl) with GPU for full LUT generation.")
if __name__ == '__main__':
main()