#!/usr/bin/env python3 """ model_theorem_matrix.py — Degenerate sector analysis + theorem attack matrix. Each model has strong and weak sectors. For each weak sector, there's a specific theorem attack that compensates. The panel selection accounts for these degeneracies by ensuring at least one model covers each sector. Usage: python3 model_theorem_matrix.py [--problem-type lean_proof] """ import json from dataclasses import dataclass from typing import Optional # ── Sector Definitions ─────────────────────────────────────────────────── SECTORS = [ "math_reasoning", # Pure mathematical reasoning "formal_proof", # Lean/Coq/Isabelle formal verification "code_generation", # Writing code "code_debugging", # Finding and fixing bugs "architecture", # System design, refactoring "tool_use", # Calling tools, APIs, structured output "reasoning_depth", # Deep multi-step reasoning "breadth", # Wide knowledge across domains "speed", # Fast iteration "cost_efficiency", # Quality per dollar ] # ── Model Capability Matrix ────────────────────────────────────────────── # Each model: {sector: score 0-1, ...} # Score > 0.7 = strong, 0.4-0.7 = mixed, < 0.4 = degenerate MODEL_CAPABILITIES = { "deepseek-v4-pro": { "math_reasoning": 0.95, "formal_proof": 0.70, "code_generation": 0.85, "code_debugging": 0.80, "architecture": 0.65, "tool_use": 0.70, "reasoning_depth": 0.90, "breadth": 0.60, "speed": 0.50, "cost_efficiency": 0.95, }, "deepseek-v4-flash": { "math_reasoning": 0.50, "formal_proof": 0.30, "code_generation": 0.60, "code_debugging": 0.55, "architecture": 0.40, "tool_use": 0.45, "reasoning_depth": 0.40, "breadth": 0.35, "speed": 0.95, "cost_efficiency": 0.98, }, "claude-opus": { "math_reasoning": 0.85, "formal_proof": 0.95, "code_generation": 0.80, "code_debugging": 0.85, "architecture": 0.90, "tool_use": 0.85, "reasoning_depth": 0.95, "breadth": 0.80, "speed": 0.30, "cost_efficiency": 0.20, }, "gpt-5.5": { "math_reasoning": 0.70, "formal_proof": 0.55, "code_generation": 0.80, "code_debugging": 0.75, "architecture": 0.85, "tool_use": 0.90, "reasoning_depth": 0.75, "breadth": 0.95, "speed": 0.60, "cost_efficiency": 0.35, }, "cohere-command": { "math_reasoning": 0.55, "formal_proof": 0.45, "code_generation": 0.70, "code_debugging": 0.65, "architecture": 0.60, "tool_use": 0.95, "reasoning_depth": 0.60, "breadth": 0.55, "speed": 0.65, "cost_efficiency": 0.60, }, "gemini-pro": { "math_reasoning": 0.65, "formal_proof": 0.50, "code_generation": 0.70, "code_debugging": 0.65, "architecture": 0.70, "tool_use": 0.75, "reasoning_depth": 0.65, "breadth": 0.75, "speed": 0.70, "cost_efficiency": 0.65, }, "glm-5.2": { "math_reasoning": 0.55, "formal_proof": 0.35, "code_generation": 0.60, "code_debugging": 0.50, "architecture": 0.50, "tool_use": 0.50, "reasoning_depth": 0.50, "breadth": 0.45, "speed": 0.55, "cost_efficiency": 0.65, }, "kimi-k2.6": { "math_reasoning": 0.60, "formal_proof": 0.40, "code_generation": 0.75, "code_debugging": 0.70, "architecture": 0.55, "tool_use": 0.60, "reasoning_depth": 0.55, "breadth": 0.50, "speed": 0.70, "cost_efficiency": 0.75, }, } # ── Theorem Attack Matrix ──────────────────────────────────────────────── # For each (model, sector) where the model is degenerate (< 0.4), # there's a specific theorem attack that compensates. THEOREM_ATTACKS = { ("deepseek-v4-flash", "formal_proof"): { "attack": "delegate_to_opus", "description": "Flash can't do formal proofs. Delegate to Claude Opus.", "compensation": "Use Flash for fast iteration, Opus for final proof.", }, ("deepseek-v4-flash", "reasoning_depth"): { "attack": "chain_of_thought", "description": "Flash lacks deep reasoning. Force chain-of-thought prompting.", "compensation": "Break problem into smaller steps, iterate.", }, ("deepseek-v4-flash", "breadth"): { "attack": "context_injection", "description": "Flash has narrow knowledge. Inject relevant context explicitly.", "compensation": "Provide AGENTS.md, file contents, build logs.", }, ("claude-opus", "speed"): { "attack": "parallel_delegation", "description": "Opus is slow. Delegate fast tasks to DeepSeek Flash.", "compensation": "Use Opus for synthesis only, Flash for iteration.", }, ("claude-opus", "cost_efficiency"): { "attack": "selective_invocation", "description": "Opus is expensive. Only invoke when cheaper models fail.", "compensation": "Escalation path: Flash → Pro → Opus.", }, ("gpt-5.5", "formal_proof"): { "attack": "lean_template", "description": "GPT struggles with Lean 4. Provide proof templates.", "compensation": "Give GPT the theorem statement + proof skeleton, let it fill in.", }, ("gpt-5.5", "math_reasoning"): { "attack": "step_by_step", "description": "GPT's math is weaker. Force step-by-step decomposition.", "compensation": "Break math problems into smaller steps.", }, ("gpt-5.5", "cost_efficiency"): { "attack": "selective_invocation", "description": "GPT is expensive. Only invoke for breadth-critical tasks.", "compensation": "Use DeepSeek for math/code, GPT for ops/breadth.", }, ("cohere-command", "math_reasoning"): { "attack": "structured_output", "description": "Cohere's math is weak. Use its structured output strength.", "compensation": "Cohere formats the output, DeepSeek does the math.", }, ("cohere-command", "formal_proof"): { "attack": "template_injection", "description": "Cohere can't do formal proofs. Inject proof templates.", "compensation": "Cohere structures the proof, Opus verifies.", }, ("cohere-command", "reasoning_depth"): { "attack": "decomposition", "description": "Cohere lacks deep reasoning. Decompose into smaller tasks.", "compensation": "Break problem into structured sub-tasks.", }, ("gemini-pro", "formal_proof"): { "attack": "lean_template", "description": "Gemini struggles with Lean. Provide templates.", "compensation": "Gemini suggests the approach, Opus writes the proof.", }, ("gemini-pro", "math_reasoning"): { "attack": "visual_reasoning", "description": "Gemini's math is weaker. Use its visual/multimodal strength.", "compensation": "Gemini draws diagrams, DeepSeek does the math.", }, ("glm-5.2", "formal_proof"): { "attack": "delegate", "description": "GLM can't do formal proofs. Delegate entirely.", "compensation": "GLM provides architectural diversity only.", }, ("glm-5.2", "math_reasoning"): { "attack": "delegate", "description": "GLM's math is weak. Delegate to DeepSeek.", "compensation": "GLM provides different perspective, DeepSeek does math.", }, ("glm-5.2", "breadth"): { "attack": "context_injection", "description": "GLM has narrow knowledge. Inject context.", "compensation": "Provide full codebase context.", }, ("kimi-k2.6", "formal_proof"): { "attack": "lean_template", "description": "Kimi can't do Lean proofs. Provide templates.", "compensation": "Kimi writes code, Opus writes proofs.", }, ("kimi-k2.6", "math_reasoning"): { "attack": "step_by_step", "description": "Kimi's math is weaker. Force step-by-step.", "compensation": "Break math into smaller steps.", }, ("kimi-k2.6", "architecture"): { "attack": "pattern_injection", "description": "Kimi lacks architectural vision. Inject patterns.", "compensation": "Provide architecture templates.", }, ("kimi-k2.6", "reasoning_depth"): { "attack": "chain_of_thought", "description": "Kimi lacks deep reasoning. Force chain-of-thought.", "compensation": "Break into smaller reasoning steps.", }, } # ── Analysis ───────────────────────────────────────────────────────────── def analyze_model(model_name: str, capabilities: dict) -> dict: """Analyze a model's strong and weak sectors.""" strong = {s: v for s, v in capabilities.items() if v >= 0.7} mixed = {s: v for s, v in capabilities.items() if 0.4 <= v < 0.7} degenerate = {s: v for s, v in capabilities.items() if v < 0.4} attacks = {} for sector in degenerate: key = (model_name, sector) if key in THEOREM_ATTACKS: attacks[sector] = THEOREM_ATTACKS[key] return { "model": model_name, "strong": strong, "mixed": mixed, "degenerate": degenerate, "attacks": attacks, "chi_strong": len(strong) / len(capabilities), "chi_degenerate": len(degenerate) / len(capabilities), } def analyze_panel(models: list[str]) -> dict: """Analyze a panel's coverage and degeneracies.""" coverage = {} for sector in SECTORS: best_score = 0 best_model = "" for model in models: score = MODEL_CAPABILITIES.get(model, {}).get(sector, 0) if score > best_score: best_score = score best_model = model coverage[sector] = { "best_model": best_model, "best_score": best_score, "covered": best_score >= 0.4, "strong": best_score >= 0.7, } uncovered = [s for s, c in coverage.items() if not c["covered"]] weak = [s for s, c in coverage.items() if not c["strong"]] return { "models": models, "coverage": coverage, "uncovered": uncovered, "weak": weak, "coverage_pct": round((len(SECTORS) - len(uncovered)) / len(SECTORS) * 100, 1), "strength_pct": round((len(SECTORS) - len(weak)) / len(SECTORS) * 100, 1), } def build_optimal_panel(problem_type: str, max_models: int = 8) -> dict: """Build an optimal panel that covers all priority sectors at strong level (≥ 0.7).""" priority = { "lean_proof": ["formal_proof", "math_reasoning", "reasoning_depth"], "code_bug": ["code_debugging", "code_generation", "speed"], "architecture": ["architecture", "reasoning_depth", "breadth"], "math_novel": ["math_reasoning", "reasoning_depth", "formal_proof"], "infra_debug": ["code_debugging", "breadth", "speed"], "hard_wall": SECTORS, } priority_sectors = priority.get(problem_type, SECTORS) # Greedy: keep adding models until all priority sectors are strong selected = [] strong = set() # sectors covered at ≥ 0.7 while len(selected) < max_models: # Check if all priority sectors are strong if all(s in strong for s in priority_sectors): break best_model = None best_score = 0 for model, caps in MODEL_CAPABILITIES.items(): if model in selected: continue # Score = number of uncovered priority sectors this model covers strongly new_strong = sum(1 for s in priority_sectors if s not in strong and caps.get(s, 0) >= 0.7) # Bonus for covering weak sectors weak = [s for s in priority_sectors if s not in strong] new_any = sum(1 for s in weak if caps.get(s, 0) >= 0.4) score = new_strong * 3 + new_any if score > best_score: best_score = score best_model = model if best_model is None or best_score == 0: break selected.append(best_model) for s in priority_sectors: if MODEL_CAPABILITIES[best_model].get(s, 0) >= 0.7: strong.add(s) return { "problem_type": problem_type, "selected": selected, "coverage": analyze_panel(selected), "attacks": {m: analyze_model(m, MODEL_CAPABILITIES[m])["attacks"] for m in selected}, } # ── CLI ────────────────────────────────────────────────────────────────── def main(): import argparse parser = argparse.ArgumentParser(description="Model theorem matrix") parser.add_argument("--problem-type", default="lean_proof") parser.add_argument("--max-models", type=int, default=8) parser.add_argument("--all-types", action="store_true") args = parser.parse_args() if args.all_types: for ptype in ["lean_proof", "code_bug", "architecture", "math_novel", "infra_debug", "hard_wall"]: result = build_optimal_panel(ptype, args.max_models) cov = result["coverage"] print(f"\n{'='*60}") print(f" {ptype}") print(f"{'='*60}") print(f" Models: {', '.join(result['selected'])}") print(f" Coverage: {cov['coverage_pct']}% | Strength: {cov['strength_pct']}%") if cov["uncovered"]: print(f" UNCOVERED: {', '.join(cov['uncovered'])}") if cov["weak"]: print(f" WEAK: {', '.join(cov['weak'])}") print(f" Theorem attacks:") for model, attacks in result["attacks"].items(): for sector, attack in attacks.items(): print(f" {model} × {sector}: {attack['attack']}") else: result = build_optimal_panel(args.problem_type, args.max_models) print(json.dumps(result, indent=2, default=str)) if __name__ == "__main__": main()