Research-Stack/scripts/math-first/test_validate_deepseek_receipts.py
Devin AI 0639eae30a chore(consolidation): integrate E8Sidon stack (PRs #79 #80 #81 #89) into one PR
Squash the four overlapping feature branches into a single change set against
main, eliminating cross-PR merge conflicts and the duplicated CI-fix scripts.

What this brings in (merge order #79 -> #80 -> #81 -> #89):
- #79 refactor(infra): shared utilities (4-Infrastructure/lib/*: q16, hashing,
  jsonl, fraction_utils) + the scripts/math-first/* validators that the
  math-check CI requires.
- #80 feat(lean): Semantics.E8Sidon (1025 lines) -- Eisenstein coefficient
  identity E4^2 = E8 and the Sidon framework. E4_sq_eq_E8_coeff is fully proved
  (all Fourier-coefficient extraction machine-checked); the single residual gap
  is pinned to E4_sq_eq_E8_qExpansion (Mathlib lacks the valence formula /
  dim M8 = 1). 4 sorries + 1 axiom (e8_additive_completeness), all TODO(lean-port).
- #81 refactor(lean): Float-free FixedPoint core (integer-only sqrt/log2/expNeg).
  E8Sidon.lean kept at #80's final 1025-line version (the #81 intermediate
  438-line copy was overridden by merge order).
- #89 feat(lean): Semantics.RRC.PolyFactorIdentity -- short-sleeve polynomial
  detection at the zerocopy limb boundary; now imports Semantics.E8Sidon for
  sigma3/sigma7/convolutionLHS (single source of truth) instead of inlining them.

Conflict resolution:
- flake.nix -> canonical rs-surface removal (Garnix shutdown).
- scripts/math-first/* -> byte-identical across branches, clean.
- .cursorrules / AGENTS.md -> unified; baselines + sorry inventory refreshed.

Verification:
- lake build (default aggregator): 3573 jobs, 0 errors.
- lake build Semantics.RRC.PolyFactorIdentity (E8Sidon + FixedPoint + PolyFactor):
  3655 jobs, 0 errors. Witnesses verified (sigma7 4 = 16513, convolutionLHS 6 = 2350).
- Python tests: 68/68 pass.

Note: the "Workers Builds: researchstack" check is a preexisting external
Cloudflare build unrelated to this change (no branch touches 4-Infrastructure/cloudflare/).

Build: 3573 jobs (default), 3655 jobs (narrow), 0 errors
Co-Authored-By: Allaun Silverfox <bigdataiscoming+9i37y6j2@protonmail.com>
2026-06-16 02:01:31 +00:00

126 lines
4 KiB
Python
Executable file

#!/usr/bin/env python3
"""Self-tests for validate_deepseek_receipts.py.
Runs minimal smoke tests to verify the validator works against the live
schema and any existing receipts. Exits 0 on success, 1 on failure.
"""
from __future__ import annotations
import json
import sys
import tempfile
from pathlib import Path
# Ensure we can import the validator's logic
SCRIPT_DIR = Path(__file__).resolve().parent
REPO_ROOT = SCRIPT_DIR.parents[1]
SCHEMA_PATH = REPO_ROOT / "shared-data" / "schemas" / "deepseek-review-receipt.schema.json"
try:
from jsonschema import Draft202012Validator, ValidationError
except ImportError:
print("SKIP: jsonschema not installed")
sys.exit(0)
def test_schema_compiles() -> bool:
"""The schema itself must be valid JSON Schema."""
if not SCHEMA_PATH.exists():
print("SKIP: schema file not found")
return True
schema = json.loads(SCHEMA_PATH.read_text())
Draft202012Validator.check_schema(schema)
print("PASS: schema compiles")
return True
def test_valid_receipt_accepted() -> bool:
"""A minimal valid primary receipt must pass validation."""
if not SCHEMA_PATH.exists():
print("SKIP: schema file not found")
return True
schema = json.loads(SCHEMA_PATH.read_text())
validator = Draft202012Validator(schema)
valid_receipt = {
"schema": "ollama_deepseek_review_receipt_v1",
"created_at": "2026-01-01T00:00:00+00:00",
"model": "deepseek-v3.2",
"endpoint": "https://ollama.com/v1/chat/completions",
"prompt_sha256": "sha256:" + "a" * 64,
"answer_sha256": "sha256:" + "b" * 64,
"usage": {"prompt_tokens": 100, "completion_tokens": 200, "total_tokens": 300},
"context_files": ["some/file.lean"],
"answer_path": "shared-data/artifacts/deepseek_review/test.md",
}
errors = list(validator.iter_errors(valid_receipt))
if errors:
print("FAIL: valid receipt rejected:")
for err in errors:
print(f" {err.message}")
return False
print("PASS: valid receipt accepted")
return True
def test_invalid_receipt_rejected() -> bool:
"""A receipt missing required fields must be rejected."""
if not SCHEMA_PATH.exists():
print("SKIP: schema file not found")
return True
schema = json.loads(SCHEMA_PATH.read_text())
validator = Draft202012Validator(schema)
invalid_receipt = {"schema": "ollama_deepseek_review_receipt_v1"}
errors = list(validator.iter_errors(invalid_receipt))
if not errors:
print("FAIL: invalid receipt was accepted")
return False
print("PASS: invalid receipt rejected")
return True
def test_bad_sha256_rejected() -> bool:
"""A receipt with malformed SHA-256 must be rejected."""
if not SCHEMA_PATH.exists():
print("SKIP: schema file not found")
return True
schema = json.loads(SCHEMA_PATH.read_text())
validator = Draft202012Validator(schema)
receipt = {
"schema": "ollama_deepseek_review_receipt_v1",
"created_at": "2026-01-01T00:00:00+00:00",
"model": "deepseek-v3.2",
"endpoint": "https://ollama.com/v1/chat/completions",
"prompt_sha256": "not-a-hash",
"answer_sha256": "sha256:" + "b" * 64,
"usage": {"prompt_tokens": 100, "completion_tokens": 200, "total_tokens": 300},
"context_files": ["some/file.lean"],
"answer_path": "shared-data/artifacts/deepseek_review/test.md",
}
errors = list(validator.iter_errors(receipt))
if not errors:
print("FAIL: bad SHA-256 was accepted")
return False
print("PASS: bad SHA-256 rejected")
return True
def main() -> int:
tests = [
test_schema_compiles,
test_valid_receipt_accepted,
test_invalid_receipt_rejected,
test_bad_sha256_rejected,
]
passed = sum(1 for t in tests if t())
total = len(tests)
print(f"\n{passed}/{total} tests passed")
return 0 if passed == total else 1
if __name__ == "__main__":
raise SystemExit(main())