diff --git a/.atlas/benchmark.sh b/.atlas/benchmark.sh new file mode 100755 index 00000000..4531fef0 --- /dev/null +++ b/.atlas/benchmark.sh @@ -0,0 +1,29 @@ +#!/usr/bin/env bash +# Generated by `atlas autoresearch` — this file is YOURS, edit it freely. +# +# Contract: write {"score": , "examples": [...]} to $ATLAS_OPTIMIZE_RESULT. +# The candidate artifact has already been written into this worktree at +# ${ATLAS_OPTIMIZE_TARGET} (the repo-relative path: 4-Infrastructure/shim/braid_search.py) +# so the only thing a candidate can change is that file — the optimizer runs +# every candidate in its own throwaway git worktree, which keeps the evaluator +# and data pinned. score.py turns the run's output into the score; .atlas/gate.sh +# is the Goodhart guard. +set -uo pipefail +HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +OUT="$(mktemp)" + +# --- the repo's run command (auto-detected) -------------------------------- +# Edit this line if the campaign should run something different. +( python3 "$ATLAS_OPTIMIZE_TARGET" ) >"$OUT" 2>&1 +RC=$? +# --------------------------------------------------------------------------- + +if [ "$RC" -ne 0 ]; then + echo "[benchmark] run command exited $RC — candidate failed" >&2 + tail -n 40 "$OUT" >&2 + exit "$RC" +fi + +# Parse the optimized metric out of the run output (edit .atlas/score.py to +# change which number is read, or to emit richer per-example feedback). +python3 "$HERE/score.py" --stdout "$OUT" diff --git a/.atlas/gate.sh b/.atlas/gate.sh new file mode 100755 index 00000000..2a16ee63 --- /dev/null +++ b/.atlas/gate.sh @@ -0,0 +1,23 @@ +#!/usr/bin/env bash +# Generated by `atlas autoresearch` — the Goodhart guard. +# +# A candidate may only change the artifact under test. If a run modifies any +# OTHER tracked file (e.g. rewrites the evaluator or test data to fake a pass), +# this gate fails (exit non-zero) and the candidate is scored worst, so the +# Pareto front never elevates a cheat. Relax IGNORE_RE if your run legitimately +# rewrites tracked files (checkpoints, logs, and .atlas/ are already ignored). +set -uo pipefail +WT="${1:-$PWD}" +TARGET="${2:-${ATLAS_OPTIMIZE_TARGET:-}}" +REL="${TARGET#"$WT"/}" +IGNORE_RE='^(outputs/|out/|checkpoints/|runs/|wandb/|\.atlas/|.*\.log$|.*\.ckpt$|.*\.pt$|.*\.bin$|.*\.safetensors$)' +CHANGED="$(git -C "$WT" status --porcelain --untracked-files=no 2>/dev/null \ + | sed 's/^...//' \ + | { [ -n "$REL" ] && grep -vF -- "$REL" || cat; } \ + | grep -Ev "$IGNORE_RE" || true)" +if [ -n "$CHANGED" ]; then + echo "[gate] the run modified tracked files other than the candidate:" >&2 + echo "$CHANGED" >&2 + exit 1 +fi +exit 0 diff --git a/.atlas/score.py b/.atlas/score.py new file mode 100755 index 00000000..7d9f2173 --- /dev/null +++ b/.atlas/score.py @@ -0,0 +1,65 @@ +#!/usr/bin/env python3 +"""Generated by `atlas autoresearch`. Parse one metric out of a run's output +into the Atlas benchmark contract. + +EDIT the METRIC line if the wrong number is being read, or rewrite parse() for +a richer evaluator (per-example feedback is what makes GEPA converge — see +.atlas/benchmark-contract or the atlas-optimize skill).""" +import argparse +import json +import os +import re +import sys + +# autoresearch guessed this metric name from your --goal. +METRIC = "find_optimal_crossing" # <-- EDIT ME if this is the wrong metric +HIGHER_IS_BETTER = False + +# Matches " = 0.74", ": 0.74", " 0.74". +PATTERN = re.compile( + re.escape(METRIC) + r"\s*[:=]?\s*([+-]?[0-9]*\.?[0-9]+(?:[eE][+-]?[0-9]+)?)" +) + + +def parse(text): + matches = PATTERN.findall(text) + if not matches: + return None + return float(matches[-1]) # last occurrence == final epoch / final eval + + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("--stdout", required=True, help="file holding the run's captured output") + args = ap.parse_args() + text = open(args.stdout, encoding="utf-8", errors="replace").read() + score = parse(text) + if score is None: + sys.stderr.write( + f"[score] could not find metric '{METRIC}' in the run output.\n" + " Fix: edit .atlas/score.py (METRIC=...), or pass an explicit\n" + " --benchmark to 'atlas autoresearch'.\n" + ) + sys.exit(3) + direction = "higher is better" if HIGHER_IS_BETTER else "lower is better" + result = { + "score": score, + "examples": [{ + "id": METRIC, + "score": score, + "pass": True, + "feedback": f"{METRIC} = {score} (parsed from run output; {direction})", + }], + "feedback": f"Final {METRIC} = {score}.", + } + out = os.environ.get("ATLAS_OPTIMIZE_RESULT") + payload = json.dumps(result) + if out: + with open(out, "w", encoding="utf-8") as fh: + fh.write(payload) + else: + sys.stdout.write(payload) + + +if __name__ == "__main__": + main() diff --git a/atlas.json b/atlas.json new file mode 100644 index 00000000..2a71aee1 --- /dev/null +++ b/atlas.json @@ -0,0 +1,11 @@ +{ + "project_id": "b4390de5-14d9-46dc-8be7-9bb36291e0b4", + "auto": { + "enabled": true, + "log": true, + "web": "prefer_atlas", + "capture": "suggest" + }, + "privacy": "private", + "source_ids": [] +}