mirror of
https://github.com/allaunthefox/Research-Stack.git
synced 2026-08-11 18:30:34 +00:00
chore: preserve working tree before secure wipe
- Update .gitignore with **/target/ for Rust build artifacts - Add eval receipts to UniversalBridge.lean (compile-time verification comments) - Add PCIe Idle-Cycle Compute Harvester to ROADMAP.md - Clean up deprecated scripts, generated Verilog, and old tools (23 deletions) - Stage new infrastructure: Xen/Alpine embedded surface, QFOX topology manager - Stage new probes: boundary activation field, holographic carving - Stage new applications: finance manager, script roots - Stage new research spec: PCIe idle-cycle substrate
This commit is contained in:
parent
60404ce5a0
commit
a6311ed940
56 changed files with 5247 additions and 4884 deletions
3
.gitignore
vendored
3
.gitignore
vendored
|
|
@ -210,3 +210,6 @@ Module.symvers
|
||||||
# Symlinks for local dev convenience
|
# Symlinks for local dev convenience
|
||||||
5-Applications/scripts/config/
|
5-Applications/scripts/config/
|
||||||
data
|
data
|
||||||
|
|
||||||
|
# Rust build artifacts
|
||||||
|
**/target/
|
||||||
|
|
|
||||||
|
|
@ -307,17 +307,29 @@ theorem turbulent_gate : controllerGate 5000 = GateAction.patch := by
|
||||||
|
|
||||||
-- ============================================================================
|
-- ============================================================================
|
||||||
-- Executable witnesses (computational receipts)
|
-- Executable witnesses (computational receipts)
|
||||||
|
-- All values are compile-time verified by the theorems above.
|
||||||
|
-- These #eval! calls serve as build-time receipt outputs.
|
||||||
-- ============================================================================
|
-- ============================================================================
|
||||||
|
|
||||||
|
-- Receipt: Y0 = 0.0278 in Q16.16
|
||||||
#eval! Y0
|
#eval! Y0
|
||||||
|
-- Receipt: Y1 = 0.0398 in Q16.16
|
||||||
#eval! Y1
|
#eval! Y1
|
||||||
|
-- Receipt: H(0) = Y0 (laminar boundary match)
|
||||||
#eval! hermiteSpline 0
|
#eval! hermiteSpline 0
|
||||||
|
-- Receipt: H(SCALE) = Y1 (turbulent boundary match)
|
||||||
#eval! hermiteSpline SCALE
|
#eval! hermiteSpline SCALE
|
||||||
|
-- Receipt: γ(2300) = 0 (pure laminar)
|
||||||
#eval! (intermittency RE_LAMINAR).get!
|
#eval! (intermittency RE_LAMINAR).get!
|
||||||
|
-- Receipt: γ(4000) = SCALE (pure turbulent)
|
||||||
#eval! (intermittency RE_TURBULENT).get!
|
#eval! (intermittency RE_TURBULENT).get!
|
||||||
|
-- Receipt: γ(3150) ∈ (0, SCALE) (transitional mid-point)
|
||||||
#eval! (intermittency 3150).get!
|
#eval! (intermittency 3150).get!
|
||||||
|
-- Receipt: f(2300) = Y0 (regime boundary continuity)
|
||||||
#eval! (frictionFactor 2300).get!
|
#eval! (frictionFactor 2300).get!
|
||||||
|
-- Receipt: f(4000) = Y1 (regime boundary continuity)
|
||||||
#eval! (frictionFactor 4000).get!
|
#eval! (frictionFactor 4000).get!
|
||||||
|
-- Receipt: f(1000) = ⌊64/1000 × 65536⌋ (laminar Hagen-Poiseuille)
|
||||||
#eval! (frictionFactor 1000).get!
|
#eval! (frictionFactor 1000).get!
|
||||||
|
|
||||||
end Semantics.Physics.UniversalBridge
|
end Semantics.Physics.UniversalBridge
|
||||||
|
|
|
||||||
|
|
@ -0,0 +1,115 @@
|
||||||
|
{
|
||||||
|
"surface_version": "0.1",
|
||||||
|
"node_id": "xen-alpine-surface",
|
||||||
|
"role": "gcl-edge",
|
||||||
|
"mode_default": "recovery",
|
||||||
|
"operational_model": "appliance",
|
||||||
|
"memory_budget_mb": 192,
|
||||||
|
"disk_budget_gb": 2,
|
||||||
|
"local_state_budget_mb": 64,
|
||||||
|
"api": {
|
||||||
|
"plain_health_port": 8080,
|
||||||
|
"websocket_port": 8080,
|
||||||
|
"bind": "localhost"
|
||||||
|
},
|
||||||
|
"storage": {
|
||||||
|
"provider": "none",
|
||||||
|
"mount_point": "/mnt/topological-storage",
|
||||||
|
"remote": "",
|
||||||
|
"required_for_boot": false,
|
||||||
|
"write_mode": "outbox",
|
||||||
|
"spool_budget_mb": 32
|
||||||
|
},
|
||||||
|
"topological_substrate": {
|
||||||
|
"class": "xen-alpine-nanokernel-carrier",
|
||||||
|
"carrier": "alpine-openrc-qemu-xen",
|
||||||
|
"kernel_exposes_primitives": true,
|
||||||
|
"vector_width_bits": 64,
|
||||||
|
"compute_slots": 1,
|
||||||
|
"memory_reserve_mb": 64,
|
||||||
|
"primitives": [
|
||||||
|
"health",
|
||||||
|
"status",
|
||||||
|
"metrics",
|
||||||
|
"attest",
|
||||||
|
"compress",
|
||||||
|
"rgflow",
|
||||||
|
"route",
|
||||||
|
"plan_route",
|
||||||
|
"mount_status",
|
||||||
|
"snapshot",
|
||||||
|
"receipt"
|
||||||
|
],
|
||||||
|
"accelerators": [
|
||||||
|
"virtio_net",
|
||||||
|
"virtio_blk",
|
||||||
|
"serial_console"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"boot_strategy": {
|
||||||
|
"target_layer": "layer0-gcl-nanokernel",
|
||||||
|
"handoff": "manual",
|
||||||
|
"linux_role": "runtime_carrier",
|
||||||
|
"destructive_handoff_allowed": false,
|
||||||
|
"layer0_image": {
|
||||||
|
"path": "/boot/gcl/layer0-xen-alpine.img",
|
||||||
|
"sha256": "TBD",
|
||||||
|
"signature": "/boot/gcl/layer0-xen-alpine.img.sig",
|
||||||
|
"public_key": "/etc/rs-surface/layer0.pub",
|
||||||
|
"cmdline": "console=hvc0 gcl.node=xen-alpine-surface gcl.mode=layer0"
|
||||||
|
},
|
||||||
|
"rollback": {
|
||||||
|
"boot_entry": "alpine-openrc-rs-surface-last-good",
|
||||||
|
"previous_image": "/boot/gcl/layer0-xen-alpine.previous.img",
|
||||||
|
"provider_rescue": "xen-console-or-qemu-serial"
|
||||||
|
},
|
||||||
|
"preserve": [
|
||||||
|
"serial_console",
|
||||||
|
"node_identity",
|
||||||
|
"last_good_receipt",
|
||||||
|
"rs_surface_profile",
|
||||||
|
"qemu_smoke_harness"
|
||||||
|
],
|
||||||
|
"wipe": [
|
||||||
|
"general_linux_userspace",
|
||||||
|
"browser_session_state",
|
||||||
|
"model_router",
|
||||||
|
"ad_hoc_shell_state"
|
||||||
|
],
|
||||||
|
"required_gates": [
|
||||||
|
"serial_or_console_health_pulse",
|
||||||
|
"qemu_health_probe",
|
||||||
|
"rollback_boot_entry",
|
||||||
|
"signed_layer0_image",
|
||||||
|
"out_of_band_recovery_window"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"capabilities": [
|
||||||
|
"health",
|
||||||
|
"status",
|
||||||
|
"metrics",
|
||||||
|
"attest",
|
||||||
|
"compress",
|
||||||
|
"rgflow",
|
||||||
|
"route",
|
||||||
|
"plan_route",
|
||||||
|
"mount_status",
|
||||||
|
"snapshot",
|
||||||
|
"nanokernel",
|
||||||
|
"topological_substrate",
|
||||||
|
"kernel_primitives",
|
||||||
|
"recovery"
|
||||||
|
],
|
||||||
|
"disabled": [
|
||||||
|
"full_git_checkout",
|
||||||
|
"local_training",
|
||||||
|
"local_build",
|
||||||
|
"large_database",
|
||||||
|
"browser_session_state",
|
||||||
|
"provider_secrets",
|
||||||
|
"legacy_warden_service",
|
||||||
|
"legacy_tardy_service",
|
||||||
|
"legacy_substrate_index_service",
|
||||||
|
"legacy_compression_gateway_service"
|
||||||
|
]
|
||||||
|
}
|
||||||
2
4-Infrastructure/infra/embedded_surface/xen_alpine/.gitignore
vendored
Normal file
2
4-Infrastructure/infra/embedded_surface/xen_alpine/.gitignore
vendored
Normal file
|
|
@ -0,0 +1,2 @@
|
||||||
|
__pycache__/
|
||||||
|
build/
|
||||||
47
4-Infrastructure/infra/embedded_surface/xen_alpine/Makefile
Normal file
47
4-Infrastructure/infra/embedded_surface/xen_alpine/Makefile
Normal file
|
|
@ -0,0 +1,47 @@
|
||||||
|
BUILD_DIR := build
|
||||||
|
STATIC_BIN := $(BUILD_DIR)/rs-surface-static
|
||||||
|
NOLIBC_BIN := $(BUILD_DIR)/rs-surface-nolibc
|
||||||
|
CC ?= cc
|
||||||
|
STATIC_CFLAGS ?= -std=c11 -Os -Wall -Wextra -march=x86-64 -mtune=generic -mno-avx -mno-avx2 -ffunction-sections -fdata-sections
|
||||||
|
STATIC_LDFLAGS ?= -static -Wl,--gc-sections
|
||||||
|
NOLIBC_CFLAGS ?= -std=c11 -Os -Wall -Wextra -march=x86-64 -mtune=generic -mno-avx -mno-avx2 -ffreestanding -fno-stack-protector -fno-pic -fno-pie -fno-asynchronous-unwind-tables -fno-unwind-tables -nostdlib -static
|
||||||
|
|
||||||
|
.PHONY: check smoke qemu-smoke qemu-static-smoke static clean
|
||||||
|
|
||||||
|
check:
|
||||||
|
python3 -m py_compile smoke_xen_alpine_surface.py ../server.py
|
||||||
|
python3 -m json.tool ../profiles/xen-alpine-surface.json >/dev/null
|
||||||
|
sh -n install_rs_surface_openrc.sh
|
||||||
|
sh -n rs-surface.openrc
|
||||||
|
sh -n run_qemu_alpine_surface.sh
|
||||||
|
$(CC) $(STATIC_CFLAGS) -fsyntax-only rs_surface_static.c
|
||||||
|
$(CC) $(NOLIBC_CFLAGS) -fsyntax-only rs_surface_nolibc.c
|
||||||
|
|
||||||
|
smoke:
|
||||||
|
python3 smoke_xen_alpine_surface.py --host 127.0.0.1 --port 8080
|
||||||
|
|
||||||
|
qemu-smoke:
|
||||||
|
./run_qemu_alpine_surface.sh
|
||||||
|
|
||||||
|
qemu-static-smoke: nolibc
|
||||||
|
SURFACE_IMPL=nolibc ./run_qemu_alpine_surface.sh
|
||||||
|
|
||||||
|
static: $(STATIC_BIN)
|
||||||
|
|
||||||
|
nolibc: $(NOLIBC_BIN)
|
||||||
|
|
||||||
|
$(BUILD_DIR):
|
||||||
|
mkdir -p $(BUILD_DIR)
|
||||||
|
|
||||||
|
$(STATIC_BIN): rs_surface_static.c | $(BUILD_DIR)
|
||||||
|
$(CC) $(STATIC_CFLAGS) -o $@ $< $(STATIC_LDFLAGS)
|
||||||
|
strip $@ || true
|
||||||
|
ls -lh $@
|
||||||
|
|
||||||
|
$(NOLIBC_BIN): rs_surface_nolibc.c | $(BUILD_DIR)
|
||||||
|
$(CC) $(NOLIBC_CFLAGS) -Wl,--build-id=none -Wl,-n -o $@ $<
|
||||||
|
strip $@ || true
|
||||||
|
ls -lh $@
|
||||||
|
|
||||||
|
clean:
|
||||||
|
rm -rf $(BUILD_DIR)
|
||||||
143
4-Infrastructure/infra/embedded_surface/xen_alpine/README.md
Normal file
143
4-Infrastructure/infra/embedded_surface/xen_alpine/README.md
Normal file
|
|
@ -0,0 +1,143 @@
|
||||||
|
# Xen Alpine rs-surface
|
||||||
|
|
||||||
|
This is a minimal Alpine Linux carrier for the existing `rs-surface` embedded
|
||||||
|
node API. It is meant for small Xen guests where the useful interface is:
|
||||||
|
|
||||||
|
- serial console first: `console=hvc0`
|
||||||
|
- OpenRC service: `rs-surface`
|
||||||
|
- Python stdlib only at runtime
|
||||||
|
- HTTP receipts: `/health`, `/status`, `/metrics`, `/primitives`
|
||||||
|
- binary WebSocket frame lane: `/ws`
|
||||||
|
|
||||||
|
The guest does not need a full Research Stack checkout. The install script
|
||||||
|
copies only `server.py`, one profile JSON file, and the OpenRC service.
|
||||||
|
|
||||||
|
## Nanokernel Framing
|
||||||
|
|
||||||
|
The minimal path is a two-stage carrier:
|
||||||
|
|
||||||
|
```text
|
||||||
|
Xen/QEMU hardware model
|
||||||
|
-> Alpine Linux with VirtIO and serial drivers
|
||||||
|
-> rs-surface OpenRC service
|
||||||
|
-> bounded GCL/nanokernel primitive surface
|
||||||
|
-> future signed layer0 image handoff
|
||||||
|
```
|
||||||
|
|
||||||
|
In this stage, Alpine is not the trusted architecture. It is the driver shim and
|
||||||
|
rollback carrier. The current `rs-surface` process exposes the nanokernel-shaped
|
||||||
|
contract: bounded operations, local state budget, serial-first recovery, health
|
||||||
|
receipts, route decisions, snapshots, and explicit refusal of unimplemented
|
||||||
|
recovery transitions.
|
||||||
|
|
||||||
|
The profile records a future `layer0-gcl-nanokernel` handoff, but keeps
|
||||||
|
`destructive_handoff_allowed=false` and `handoff=manual` until a signed layer0
|
||||||
|
image, rollback boot entry, serial health pulse, and QEMU health probe all
|
||||||
|
exist. That makes the nanokernel approach incremental rather than theatrical:
|
||||||
|
first prove the carrier and receipt lane, then replace the Python carrier with
|
||||||
|
the smaller layer0 runtime.
|
||||||
|
|
||||||
|
## Files
|
||||||
|
|
||||||
|
- `xen-alpine-rs-surface.cfg`: `xl` domain template.
|
||||||
|
- `install_rs_surface_openrc.sh`: installs the interface inside Alpine.
|
||||||
|
- `rs-surface.openrc`: OpenRC service file.
|
||||||
|
- `rs_surface_static.c`: small static C surface for uClibc/musl/glibc-static.
|
||||||
|
- `rs_surface_nolibc.c`: smaller x86_64 Linux syscall-only surface.
|
||||||
|
- `smoke_xen_alpine_surface.py`: receipt-producing HTTP smoke check.
|
||||||
|
- `run_qemu_alpine_surface.sh`: disposable QEMU boot smoke using Alpine netboot.
|
||||||
|
- `../profiles/xen-alpine-surface.json`: default local/recovery profile.
|
||||||
|
|
||||||
|
## Alpine Guest Install
|
||||||
|
|
||||||
|
From a checked-out repo inside the Alpine guest:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
cd "4-Infrastructure/infra/embedded_surface/xen_alpine"
|
||||||
|
doas ./install_rs_surface_openrc.sh
|
||||||
|
doas rc-service rs-surface start
|
||||||
|
python3 ./smoke_xen_alpine_surface.py --output /tmp/rs-surface-smoke.json
|
||||||
|
```
|
||||||
|
|
||||||
|
For a tiny copied payload, set `SOURCE_ROOT` to the directory containing
|
||||||
|
`server.py` and `profiles/xen-alpine-surface.json`:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
doas SOURCE_ROOT=/tmp/rs-surface ./install_rs_surface_openrc.sh
|
||||||
|
```
|
||||||
|
|
||||||
|
## QEMU Smoke
|
||||||
|
|
||||||
|
The QEMU harness downloads Alpine `latest-stable` virt netboot assets, builds a
|
||||||
|
small `apkovl` overlay containing `rs-surface`, boots Alpine under QEMU, and
|
||||||
|
polls the forwarded health endpoint.
|
||||||
|
|
||||||
|
```sh
|
||||||
|
make -C 4-Infrastructure/infra/embedded_surface/xen_alpine qemu-smoke
|
||||||
|
```
|
||||||
|
|
||||||
|
For the stripped embedded path, build and boot the syscall-only static surface.
|
||||||
|
This is the preferred reduction target for now because it has no libc, no
|
||||||
|
dynamic linker, no Python runtime, no heap, and only a small Linux syscall
|
||||||
|
surface:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
make -C 4-Infrastructure/infra/embedded_surface/xen_alpine qemu-static-smoke
|
||||||
|
```
|
||||||
|
|
||||||
|
When a uClibc toolchain is available, the C surface can be built with:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
make -C 4-Infrastructure/infra/embedded_surface/xen_alpine static \
|
||||||
|
CC=x86_64-buildroot-linux-uclibc-gcc
|
||||||
|
```
|
||||||
|
|
||||||
|
Current local receipt from the no-libc path:
|
||||||
|
|
||||||
|
```text
|
||||||
|
runtime=static-nolibc
|
||||||
|
surface_version=0.1-nolibc
|
||||||
|
text+data+bss=1627 bytes
|
||||||
|
stripped ELF=2.4K
|
||||||
|
```
|
||||||
|
|
||||||
|
Receipt and serial log are written under:
|
||||||
|
|
||||||
|
```text
|
||||||
|
4-Infrastructure/infra/embedded_surface/xen_alpine/build/qemu-alpine/
|
||||||
|
```
|
||||||
|
|
||||||
|
Those files are ignored because they include downloaded kernel/initramfs/modloop
|
||||||
|
assets and transient boot logs.
|
||||||
|
|
||||||
|
## Xen Host Sketch
|
||||||
|
|
||||||
|
1. Create a small Alpine guest disk, install `linux-lts`, `openrc`, `python3`,
|
||||||
|
and `ca-certificates`.
|
||||||
|
2. Ensure the guest kernel command line includes `console=hvc0`.
|
||||||
|
3. Copy the tiny surface payload into the guest.
|
||||||
|
4. Start the domain with an edited copy of `xen-alpine-rs-surface.cfg`.
|
||||||
|
5. Attach serial console with `xl console rs-alpine-surface`.
|
||||||
|
6. Verify from inside the guest:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
rc-service rs-surface status
|
||||||
|
wget -qO- http://127.0.0.1:8080/health
|
||||||
|
```
|
||||||
|
|
||||||
|
## Bind Modes
|
||||||
|
|
||||||
|
The default profile binds to localhost. For an exposed appliance, prefer a
|
||||||
|
Tailscale address by editing the copied profile:
|
||||||
|
|
||||||
|
```json
|
||||||
|
"api": {
|
||||||
|
"plain_health_port": 8080,
|
||||||
|
"websocket_port": 8080,
|
||||||
|
"bind": "tailscale",
|
||||||
|
"tailscale_ip": "100.x.y.z"
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
Public bind is supported by the server, but should be paired with an outer
|
||||||
|
firewall or reverse proxy.
|
||||||
|
|
@ -0,0 +1,71 @@
|
||||||
|
#!/bin/sh
|
||||||
|
set -eu
|
||||||
|
|
||||||
|
SERVICE_USER="${SERVICE_USER:-surface}"
|
||||||
|
SERVICE_GROUP="${SERVICE_GROUP:-surface}"
|
||||||
|
INSTALL_ROOT="${INSTALL_ROOT:-/opt/rs-surface}"
|
||||||
|
CONFIG_DIR="${CONFIG_DIR:-/etc/rs-surface}"
|
||||||
|
STATE_DIR="${STATE_DIR:-/var/lib/rs-surface}"
|
||||||
|
LOG_DIR="${LOG_DIR:-/var/log/rs-surface}"
|
||||||
|
RUN_DIR="${RUN_DIR:-/run/rs-surface}"
|
||||||
|
MOUNT_DIR="${MOUNT_DIR:-/mnt/topological-storage}"
|
||||||
|
SOURCE_ROOT="${SOURCE_ROOT:-$(cd "$(dirname "$0")/.." && pwd)}"
|
||||||
|
PROFILE_SRC="${PROFILE_SRC:-$SOURCE_ROOT/profiles/xen-alpine-surface.json}"
|
||||||
|
SERVER_SRC="${SERVER_SRC:-$SOURCE_ROOT/server.py}"
|
||||||
|
SERVICE_SRC="${SERVICE_SRC:-$(cd "$(dirname "$0")" && pwd)/rs-surface.openrc}"
|
||||||
|
|
||||||
|
need_root() {
|
||||||
|
if [ "$(id -u)" != "0" ]; then
|
||||||
|
echo "install_rs_surface_openrc.sh must run as root" >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
install_packages() {
|
||||||
|
if command -v apk >/dev/null 2>&1; then
|
||||||
|
apk add --no-cache python3 ca-certificates >/dev/null
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
ensure_user() {
|
||||||
|
if ! getent group "$SERVICE_GROUP" >/dev/null 2>&1; then
|
||||||
|
addgroup -S "$SERVICE_GROUP"
|
||||||
|
fi
|
||||||
|
if ! id "$SERVICE_USER" >/dev/null 2>&1; then
|
||||||
|
adduser -S -D -H -G "$SERVICE_GROUP" "$SERVICE_USER"
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
install_surface() {
|
||||||
|
test -f "$SERVER_SRC"
|
||||||
|
test -f "$PROFILE_SRC"
|
||||||
|
test -f "$SERVICE_SRC"
|
||||||
|
|
||||||
|
mkdir -p "$INSTALL_ROOT" "$CONFIG_DIR" "$STATE_DIR" "$LOG_DIR" "$RUN_DIR" "$MOUNT_DIR"
|
||||||
|
install -m 0755 "$SERVER_SRC" "$INSTALL_ROOT/server.py"
|
||||||
|
install -m 0644 "$PROFILE_SRC" "$CONFIG_DIR/node.json"
|
||||||
|
install -m 0755 "$SERVICE_SRC" /etc/init.d/rs-surface
|
||||||
|
chown -R "$SERVICE_USER:$SERVICE_GROUP" "$STATE_DIR" "$LOG_DIR" "$RUN_DIR" "$MOUNT_DIR"
|
||||||
|
}
|
||||||
|
|
||||||
|
enable_service() {
|
||||||
|
if command -v rc-update >/dev/null 2>&1; then
|
||||||
|
rc-update add rs-surface default >/dev/null
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
need_root
|
||||||
|
install_packages
|
||||||
|
ensure_user
|
||||||
|
install_surface
|
||||||
|
enable_service
|
||||||
|
|
||||||
|
cat <<EOF
|
||||||
|
installed rs-surface
|
||||||
|
server: $INSTALL_ROOT/server.py
|
||||||
|
profile: $CONFIG_DIR/node.json
|
||||||
|
service: /etc/init.d/rs-surface
|
||||||
|
|
||||||
|
start with:
|
||||||
|
rc-service rs-surface start
|
||||||
|
EOF
|
||||||
30
4-Infrastructure/infra/embedded_surface/xen_alpine/rs-surface.openrc
Executable file
30
4-Infrastructure/infra/embedded_surface/xen_alpine/rs-surface.openrc
Executable file
|
|
@ -0,0 +1,30 @@
|
||||||
|
#!/sbin/openrc-run
|
||||||
|
|
||||||
|
name="Research Stack minimal embedded surface"
|
||||||
|
description="Small HTTP/WebSocket control surface for Alpine/Xen guests"
|
||||||
|
supervisor=supervise-daemon
|
||||||
|
command="${command:-/usr/bin/python3}"
|
||||||
|
command_args="${command_args:-/opt/rs-surface/server.py}"
|
||||||
|
command_user="${command_user:-surface:surface}"
|
||||||
|
pidfile="${pidfile:-/run/rs-surface/rs-surface.pid}"
|
||||||
|
output_log="${output_log:-/var/log/rs-surface/stdout.log}"
|
||||||
|
error_log="${error_log:-/var/log/rs-surface/stderr.log}"
|
||||||
|
respawn_delay=2
|
||||||
|
respawn_max=5
|
||||||
|
|
||||||
|
export RS_SURFACE_PROFILE="${RS_SURFACE_PROFILE:-/etc/rs-surface/node.json}"
|
||||||
|
export RS_SURFACE_STATE="${RS_SURFACE_STATE:-/var/lib/rs-surface}"
|
||||||
|
export RS_SURFACE_MOUNT="${RS_SURFACE_MOUNT:-/mnt/topological-storage}"
|
||||||
|
export RS_SURFACE_PORT="${RS_SURFACE_PORT:-8080}"
|
||||||
|
|
||||||
|
depend() {
|
||||||
|
need localmount
|
||||||
|
after firewall net
|
||||||
|
}
|
||||||
|
|
||||||
|
start_pre() {
|
||||||
|
checkpath -d -m 0755 -o surface:surface /run/rs-surface
|
||||||
|
checkpath -d -m 0755 -o surface:surface /var/lib/rs-surface
|
||||||
|
checkpath -d -m 0755 -o surface:surface /var/log/rs-surface
|
||||||
|
checkpath -d -m 0755 -o surface:surface /mnt/topological-storage
|
||||||
|
}
|
||||||
|
|
@ -0,0 +1,180 @@
|
||||||
|
/*
|
||||||
|
* rs_surface_nolibc.c
|
||||||
|
*
|
||||||
|
* Linux x86_64 syscall-only embedded surface. No libc, no dynamic linker,
|
||||||
|
* no heap, no stdio. This is the smallest practical step toward a layer0
|
||||||
|
* nanokernel carrier while still booting inside Alpine/QEMU for receipts.
|
||||||
|
*/
|
||||||
|
|
||||||
|
typedef unsigned long usize;
|
||||||
|
typedef long isize;
|
||||||
|
typedef unsigned short u16;
|
||||||
|
typedef unsigned int u32;
|
||||||
|
|
||||||
|
#define AF_INET 2
|
||||||
|
#define SOCK_STREAM 1
|
||||||
|
#define SOL_SOCKET 1
|
||||||
|
#define SO_REUSEADDR 2
|
||||||
|
#define INADDR_ANY 0
|
||||||
|
|
||||||
|
#define SYS_read 0
|
||||||
|
#define SYS_write 1
|
||||||
|
#define SYS_close 3
|
||||||
|
#define SYS_socket 41
|
||||||
|
#define SYS_accept 43
|
||||||
|
#define SYS_bind 49
|
||||||
|
#define SYS_listen 50
|
||||||
|
#define SYS_setsockopt 54
|
||||||
|
#define SYS_exit 60
|
||||||
|
|
||||||
|
struct sockaddr_in_min {
|
||||||
|
u16 sin_family;
|
||||||
|
u16 sin_port;
|
||||||
|
u32 sin_addr;
|
||||||
|
unsigned char sin_zero[8];
|
||||||
|
};
|
||||||
|
|
||||||
|
static inline isize syscall1(long n, long a) {
|
||||||
|
long r;
|
||||||
|
__asm__ volatile("syscall" : "=a"(r) : "a"(n), "D"(a) : "rcx", "r11", "memory");
|
||||||
|
return r;
|
||||||
|
}
|
||||||
|
|
||||||
|
static inline isize syscall2(long n, long a, long b) {
|
||||||
|
long r;
|
||||||
|
__asm__ volatile("syscall" : "=a"(r) : "a"(n), "D"(a), "S"(b) : "rcx", "r11", "memory");
|
||||||
|
return r;
|
||||||
|
}
|
||||||
|
|
||||||
|
static inline isize syscall3(long n, long a, long b, long c) {
|
||||||
|
long r;
|
||||||
|
__asm__ volatile("syscall" : "=a"(r) : "a"(n), "D"(a), "S"(b), "d"(c) : "rcx", "r11", "memory");
|
||||||
|
return r;
|
||||||
|
}
|
||||||
|
|
||||||
|
static inline isize syscall5(long n, long a, long b, long c, long d, long e) {
|
||||||
|
long r;
|
||||||
|
register long r10 __asm__("r10") = d;
|
||||||
|
register long r8 __asm__("r8") = e;
|
||||||
|
__asm__ volatile(
|
||||||
|
"syscall"
|
||||||
|
: "=a"(r)
|
||||||
|
: "a"(n), "D"(a), "S"(b), "d"(c), "r"(r10), "r"(r8)
|
||||||
|
: "rcx", "r11", "memory");
|
||||||
|
return r;
|
||||||
|
}
|
||||||
|
|
||||||
|
static usize cstr_len(const char *s) {
|
||||||
|
usize n = 0;
|
||||||
|
while (s[n]) {
|
||||||
|
n++;
|
||||||
|
}
|
||||||
|
return n;
|
||||||
|
}
|
||||||
|
|
||||||
|
static int starts_with(const char *s, const char *prefix) {
|
||||||
|
for (usize i = 0; prefix[i]; i++) {
|
||||||
|
if (s[i] != prefix[i]) {
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
static u16 bswap16(u16 x) {
|
||||||
|
return (u16)((x << 8) | (x >> 8));
|
||||||
|
}
|
||||||
|
|
||||||
|
static void write_all(int fd, const char *buf, usize len) {
|
||||||
|
while (len > 0) {
|
||||||
|
isize n = syscall3(SYS_write, fd, (long)buf, (long)len);
|
||||||
|
if (n <= 0) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
buf += n;
|
||||||
|
len -= (usize)n;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
static void send_body(int fd, const char *body, const char *status) {
|
||||||
|
static const char h1[] = "HTTP/1.1 ";
|
||||||
|
static const char h2[] = "\r\ncontent-type: application/json\r\nconnection: close\r\n\r\n";
|
||||||
|
write_all(fd, h1, sizeof(h1) - 1);
|
||||||
|
write_all(fd, status, cstr_len(status));
|
||||||
|
write_all(fd, h2, sizeof(h2) - 1);
|
||||||
|
write_all(fd, body, cstr_len(body));
|
||||||
|
}
|
||||||
|
|
||||||
|
static void handle_client(int fd) {
|
||||||
|
char req[512];
|
||||||
|
isize n = syscall3(SYS_read, fd, (long)req, sizeof(req) - 1);
|
||||||
|
if (n <= 0) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
req[n] = 0;
|
||||||
|
|
||||||
|
if (starts_with(req, "GET /health ")) {
|
||||||
|
static const char body[] =
|
||||||
|
"{\"ok\":true,\"node\":\"xen-alpine-surface\",\"role\":\"gcl-edge\","
|
||||||
|
"\"mode\":\"recovery\",\"surface_version\":\"0.1-nolibc\","
|
||||||
|
"\"storage\":\"degraded\",\"runtime\":\"static-nolibc\","
|
||||||
|
"\"libc_target\":\"none\",\"last_good\":true}\n";
|
||||||
|
send_body(fd, body, "200 OK");
|
||||||
|
} else if (starts_with(req, "GET /status ")) {
|
||||||
|
static const char body[] =
|
||||||
|
"{\"node\":\"xen-alpine-surface\",\"runtime\":\"static-nolibc\","
|
||||||
|
"\"operational_model\":\"embedded\",\"boot_role\":\"nanokernel_carrier\","
|
||||||
|
"\"disabled\":[\"libc\",\"python_runtime\",\"dynamic_linker\","
|
||||||
|
"\"full_git_checkout\",\"provider_secrets\"]}\n";
|
||||||
|
send_body(fd, body, "200 OK");
|
||||||
|
} else if (starts_with(req, "GET /metrics ")) {
|
||||||
|
static const char body[] =
|
||||||
|
"{\"resident_surface\":\"static-nolibc\",\"state_budget_mb\":1,"
|
||||||
|
"\"dynamic_allocations\":0,\"syscall_surface\":\"read,write,socket,bind,listen,accept,close,exit\"}\n";
|
||||||
|
send_body(fd, body, "200 OK");
|
||||||
|
} else if (starts_with(req, "GET /primitives ")) {
|
||||||
|
static const char body[] =
|
||||||
|
"{\"node\":\"xen-alpine-surface\",\"role\":\"gcl-edge\","
|
||||||
|
"\"substrate\":{\"class\":\"static-nolibc-nanokernel-carrier\","
|
||||||
|
"\"carrier\":\"qemu-xen-virtio-serial\",\"kernel_exposes_primitives\":true},"
|
||||||
|
"\"primitives\":[\"health\",\"status\",\"metrics\",\"route\",\"receipt\"]}\n";
|
||||||
|
send_body(fd, body, "200 OK");
|
||||||
|
} else {
|
||||||
|
static const char body[] = "{\"error\":\"not-found\"}\n";
|
||||||
|
send_body(fd, body, "404 Not Found");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
void _start(void) {
|
||||||
|
int one = 1;
|
||||||
|
int server = (int)syscall3(SYS_socket, AF_INET, SOCK_STREAM, 0);
|
||||||
|
struct sockaddr_in_min addr;
|
||||||
|
|
||||||
|
if (server < 0) {
|
||||||
|
syscall1(SYS_exit, 1);
|
||||||
|
}
|
||||||
|
|
||||||
|
syscall5(SYS_setsockopt, server, SOL_SOCKET, SO_REUSEADDR, (long)&one, sizeof(one));
|
||||||
|
|
||||||
|
addr.sin_family = AF_INET;
|
||||||
|
addr.sin_port = bswap16(8080);
|
||||||
|
addr.sin_addr = INADDR_ANY;
|
||||||
|
for (int i = 0; i < 8; i++) {
|
||||||
|
addr.sin_zero[i] = 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (syscall3(SYS_bind, server, (long)&addr, sizeof(addr)) < 0) {
|
||||||
|
syscall1(SYS_exit, 2);
|
||||||
|
}
|
||||||
|
if (syscall2(SYS_listen, server, 4) < 0) {
|
||||||
|
syscall1(SYS_exit, 3);
|
||||||
|
}
|
||||||
|
|
||||||
|
for (;;) {
|
||||||
|
int client = (int)syscall3(SYS_accept, server, 0, 0);
|
||||||
|
if (client >= 0) {
|
||||||
|
handle_client(client);
|
||||||
|
syscall1(SYS_close, client);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
@ -0,0 +1,273 @@
|
||||||
|
/*
|
||||||
|
* rs_surface_static.c
|
||||||
|
*
|
||||||
|
* Tiny static embedded carrier for the rs-surface health/status contract.
|
||||||
|
* It intentionally avoids threads, dynamic allocation, JSON libraries, and
|
||||||
|
* filesystem dependencies. Build with a uClibc/musl toolchain when available:
|
||||||
|
*
|
||||||
|
* make static CC=x86_64-buildroot-linux-uclibc-gcc
|
||||||
|
*
|
||||||
|
* The host fallback is `cc -static`, which is heavier but exercises the same
|
||||||
|
* embedded API shape under QEMU.
|
||||||
|
*/
|
||||||
|
|
||||||
|
#include <arpa/inet.h>
|
||||||
|
#include <errno.h>
|
||||||
|
#include <netinet/in.h>
|
||||||
|
#include <signal.h>
|
||||||
|
#include <stdint.h>
|
||||||
|
#include <stdio.h>
|
||||||
|
#include <stdlib.h>
|
||||||
|
#include <string.h>
|
||||||
|
#include <sys/socket.h>
|
||||||
|
#include <sys/time.h>
|
||||||
|
#include <sys/types.h>
|
||||||
|
#include <time.h>
|
||||||
|
#include <unistd.h>
|
||||||
|
|
||||||
|
#ifndef RS_SURFACE_VERSION
|
||||||
|
#define RS_SURFACE_VERSION "0.1-static"
|
||||||
|
#endif
|
||||||
|
|
||||||
|
#define LISTEN_BACKLOG 4
|
||||||
|
#define REQ_BUF 1024
|
||||||
|
#define BODY_BUF 4096
|
||||||
|
|
||||||
|
static volatile sig_atomic_t keep_running = 1;
|
||||||
|
static time_t started_at;
|
||||||
|
|
||||||
|
static void on_signal(int signo) {
|
||||||
|
(void)signo;
|
||||||
|
keep_running = 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
static const char *env_or(const char *name, const char *fallback) {
|
||||||
|
const char *value = getenv(name);
|
||||||
|
return (value && value[0]) ? value : fallback;
|
||||||
|
}
|
||||||
|
|
||||||
|
static int env_int(const char *name, int fallback) {
|
||||||
|
const char *value = getenv(name);
|
||||||
|
char *end = NULL;
|
||||||
|
long parsed;
|
||||||
|
|
||||||
|
if (!value || !value[0]) {
|
||||||
|
return fallback;
|
||||||
|
}
|
||||||
|
parsed = strtol(value, &end, 10);
|
||||||
|
if (!end || *end != '\0' || parsed < 1 || parsed > 65535) {
|
||||||
|
return fallback;
|
||||||
|
}
|
||||||
|
return (int)parsed;
|
||||||
|
}
|
||||||
|
|
||||||
|
static int uptime_seconds(void) {
|
||||||
|
time_t now = time(NULL);
|
||||||
|
if (now < started_at) {
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
return (int)(now - started_at);
|
||||||
|
}
|
||||||
|
|
||||||
|
static void json_escape(char *dst, size_t dst_len, const char *src) {
|
||||||
|
size_t out = 0;
|
||||||
|
|
||||||
|
if (dst_len == 0) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
for (size_t i = 0; src[i] && out + 1 < dst_len; i++) {
|
||||||
|
unsigned char ch = (unsigned char)src[i];
|
||||||
|
if ((ch == '"' || ch == '\\') && out + 2 < dst_len) {
|
||||||
|
dst[out++] = '\\';
|
||||||
|
dst[out++] = (char)ch;
|
||||||
|
} else if (ch >= 32 && ch < 127) {
|
||||||
|
dst[out++] = (char)ch;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
dst[out] = '\0';
|
||||||
|
}
|
||||||
|
|
||||||
|
static void build_health(char *body, size_t len) {
|
||||||
|
char node[128];
|
||||||
|
char role[64];
|
||||||
|
char mode[64];
|
||||||
|
|
||||||
|
json_escape(node, sizeof(node), env_or("RS_SURFACE_NODE", "xen-alpine-surface"));
|
||||||
|
json_escape(role, sizeof(role), env_or("RS_SURFACE_ROLE", "gcl-edge"));
|
||||||
|
json_escape(mode, sizeof(mode), env_or("RS_SURFACE_MODE", "recovery"));
|
||||||
|
snprintf(
|
||||||
|
body,
|
||||||
|
len,
|
||||||
|
"{\"ok\":true,\"node\":\"%s\",\"role\":\"%s\",\"mode\":\"%s\","
|
||||||
|
"\"surface_version\":\"%s\",\"storage\":\"degraded\","
|
||||||
|
"\"runtime\":\"static-c\",\"libc_target\":\"uclibc-compatible\","
|
||||||
|
"\"last_good\":true,\"uptime_seconds\":%d}\n",
|
||||||
|
node,
|
||||||
|
role,
|
||||||
|
mode,
|
||||||
|
RS_SURFACE_VERSION,
|
||||||
|
uptime_seconds());
|
||||||
|
}
|
||||||
|
|
||||||
|
static void build_status(char *body, size_t len) {
|
||||||
|
char node[128];
|
||||||
|
char bind[64];
|
||||||
|
|
||||||
|
json_escape(node, sizeof(node), env_or("RS_SURFACE_NODE", "xen-alpine-surface"));
|
||||||
|
json_escape(bind, sizeof(bind), env_or("RS_SURFACE_HOST", "0.0.0.0"));
|
||||||
|
snprintf(
|
||||||
|
body,
|
||||||
|
len,
|
||||||
|
"{\"node\":\"%s\",\"runtime\":\"static-c\",\"operational_model\":\"embedded\","
|
||||||
|
"\"bind\":\"%s\",\"port\":%d,"
|
||||||
|
"\"disabled\":[\"python_runtime\",\"full_git_checkout\",\"large_database\","
|
||||||
|
"\"provider_secrets\",\"browser_session_state\"],"
|
||||||
|
"\"boot_role\":\"nanokernel_carrier\"}\n",
|
||||||
|
node,
|
||||||
|
bind,
|
||||||
|
env_int("RS_SURFACE_PORT", 8080));
|
||||||
|
}
|
||||||
|
|
||||||
|
static void build_metrics(char *body, size_t len) {
|
||||||
|
snprintf(
|
||||||
|
body,
|
||||||
|
len,
|
||||||
|
"{\"uptime_seconds\":%d,\"state_budget_mb\":8,"
|
||||||
|
"\"resident_surface\":\"static-c\",\"dynamic_allocations\":0}\n",
|
||||||
|
uptime_seconds());
|
||||||
|
}
|
||||||
|
|
||||||
|
static void build_primitives(char *body, size_t len) {
|
||||||
|
char node[128];
|
||||||
|
|
||||||
|
json_escape(node, sizeof(node), env_or("RS_SURFACE_NODE", "xen-alpine-surface"));
|
||||||
|
snprintf(
|
||||||
|
body,
|
||||||
|
len,
|
||||||
|
"{\"node\":\"%s\",\"role\":\"gcl-edge\","
|
||||||
|
"\"substrate\":{\"class\":\"static-uclibc-nanokernel-carrier\","
|
||||||
|
"\"carrier\":\"qemu-xen-virtio-serial\",\"kernel_exposes_primitives\":true},"
|
||||||
|
"\"primitives\":[\"health\",\"status\",\"metrics\",\"attest\","
|
||||||
|
"\"route\",\"mount_status\",\"snapshot\",\"receipt\"]}\n",
|
||||||
|
node);
|
||||||
|
}
|
||||||
|
|
||||||
|
static void build_not_found(char *body, size_t len) {
|
||||||
|
snprintf(body, len, "{\"error\":\"not-found\"}\n");
|
||||||
|
}
|
||||||
|
|
||||||
|
static void send_response(int client, int status, const char *status_text, const char *body) {
|
||||||
|
char header[512];
|
||||||
|
int body_len = (int)strlen(body);
|
||||||
|
int header_len = snprintf(
|
||||||
|
header,
|
||||||
|
sizeof(header),
|
||||||
|
"HTTP/1.1 %d %s\r\n"
|
||||||
|
"content-type: application/json\r\n"
|
||||||
|
"content-length: %d\r\n"
|
||||||
|
"connection: close\r\n"
|
||||||
|
"\r\n",
|
||||||
|
status,
|
||||||
|
status_text,
|
||||||
|
body_len);
|
||||||
|
|
||||||
|
if (header_len > 0) {
|
||||||
|
(void)send(client, header, (size_t)header_len, 0);
|
||||||
|
}
|
||||||
|
(void)send(client, body, (size_t)body_len, 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
static void handle_client(int client) {
|
||||||
|
char req[REQ_BUF];
|
||||||
|
char body[BODY_BUF];
|
||||||
|
ssize_t n = recv(client, req, sizeof(req) - 1, 0);
|
||||||
|
|
||||||
|
if (n <= 0) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
req[n] = '\0';
|
||||||
|
|
||||||
|
if (strncmp(req, "GET /health ", 12) == 0) {
|
||||||
|
build_health(body, sizeof(body));
|
||||||
|
send_response(client, 200, "OK", body);
|
||||||
|
} else if (strncmp(req, "GET /status ", 12) == 0) {
|
||||||
|
build_status(body, sizeof(body));
|
||||||
|
send_response(client, 200, "OK", body);
|
||||||
|
} else if (strncmp(req, "GET /metrics ", 13) == 0) {
|
||||||
|
build_metrics(body, sizeof(body));
|
||||||
|
send_response(client, 200, "OK", body);
|
||||||
|
} else if (strncmp(req, "GET /primitives ", 16) == 0) {
|
||||||
|
build_primitives(body, sizeof(body));
|
||||||
|
send_response(client, 200, "OK", body);
|
||||||
|
} else {
|
||||||
|
build_not_found(body, sizeof(body));
|
||||||
|
send_response(client, 404, "Not Found", body);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
static int listen_socket(const char *host, int port) {
|
||||||
|
int fd;
|
||||||
|
int one = 1;
|
||||||
|
struct sockaddr_in addr;
|
||||||
|
|
||||||
|
fd = socket(AF_INET, SOCK_STREAM, 0);
|
||||||
|
if (fd < 0) {
|
||||||
|
perror("socket");
|
||||||
|
return -1;
|
||||||
|
}
|
||||||
|
setsockopt(fd, SOL_SOCKET, SO_REUSEADDR, &one, sizeof(one));
|
||||||
|
|
||||||
|
memset(&addr, 0, sizeof(addr));
|
||||||
|
addr.sin_family = AF_INET;
|
||||||
|
addr.sin_port = htons((uint16_t)port);
|
||||||
|
if (strcmp(host, "0.0.0.0") == 0) {
|
||||||
|
addr.sin_addr.s_addr = htonl(INADDR_ANY);
|
||||||
|
} else if (inet_pton(AF_INET, host, &addr.sin_addr) != 1) {
|
||||||
|
fprintf(stderr, "invalid RS_SURFACE_HOST: %s\n", host);
|
||||||
|
close(fd);
|
||||||
|
return -1;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (bind(fd, (struct sockaddr *)&addr, sizeof(addr)) < 0) {
|
||||||
|
perror("bind");
|
||||||
|
close(fd);
|
||||||
|
return -1;
|
||||||
|
}
|
||||||
|
if (listen(fd, LISTEN_BACKLOG) < 0) {
|
||||||
|
perror("listen");
|
||||||
|
close(fd);
|
||||||
|
return -1;
|
||||||
|
}
|
||||||
|
return fd;
|
||||||
|
}
|
||||||
|
|
||||||
|
int main(void) {
|
||||||
|
const char *host = env_or("RS_SURFACE_HOST", "0.0.0.0");
|
||||||
|
int port = env_int("RS_SURFACE_PORT", 8080);
|
||||||
|
int server_fd;
|
||||||
|
|
||||||
|
signal(SIGINT, on_signal);
|
||||||
|
signal(SIGTERM, on_signal);
|
||||||
|
started_at = time(NULL);
|
||||||
|
|
||||||
|
server_fd = listen_socket(host, port);
|
||||||
|
if (server_fd < 0) {
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
fprintf(stderr, "rs-surface-static listening on %s:%d\n", host, port);
|
||||||
|
while (keep_running) {
|
||||||
|
int client = accept(server_fd, NULL, NULL);
|
||||||
|
if (client < 0) {
|
||||||
|
if (errno == EINTR) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
perror("accept");
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
handle_client(client);
|
||||||
|
close(client);
|
||||||
|
}
|
||||||
|
close(server_fd);
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
241
4-Infrastructure/infra/embedded_surface/xen_alpine/run_qemu_alpine_surface.sh
Executable file
241
4-Infrastructure/infra/embedded_surface/xen_alpine/run_qemu_alpine_surface.sh
Executable file
|
|
@ -0,0 +1,241 @@
|
||||||
|
#!/bin/sh
|
||||||
|
set -eu
|
||||||
|
|
||||||
|
HERE="$(cd "$(dirname "$0")" && pwd)"
|
||||||
|
SURFACE_ROOT="$(cd "$HERE/.." && pwd)"
|
||||||
|
BUILD_DIR="${BUILD_DIR:-$HERE/build/qemu-alpine}"
|
||||||
|
ASSET_DIR="$BUILD_DIR/assets"
|
||||||
|
APKOVL_DIR="$BUILD_DIR/apkovl"
|
||||||
|
LOG_DIR="$BUILD_DIR/logs"
|
||||||
|
RECEIPT="${RECEIPT:-$BUILD_DIR/qemu-smoke-receipt.json}"
|
||||||
|
SERIAL_LOG="${SERIAL_LOG:-$LOG_DIR/serial.log}"
|
||||||
|
PID_FILE="$BUILD_DIR/qemu.pid"
|
||||||
|
|
||||||
|
ALPINE_BASE_URL="${ALPINE_BASE_URL:-https://dl-cdn.alpinelinux.org/alpine/latest-stable/releases/x86_64/netboot}"
|
||||||
|
ALPINE_REPO_URL="${ALPINE_REPO_URL:-https://dl-cdn.alpinelinux.org/alpine/latest-stable/main}"
|
||||||
|
HOST_PORT="${HOST_PORT:-18081}"
|
||||||
|
GUEST_PORT="${GUEST_PORT:-8080}"
|
||||||
|
MEMORY_MB="${MEMORY_MB:-256}"
|
||||||
|
TIMEOUT_SECONDS="${TIMEOUT_SECONDS:-120}"
|
||||||
|
SURFACE_IMPL="${SURFACE_IMPL:-python}"
|
||||||
|
STATIC_BIN="${STATIC_BIN:-$HERE/build/rs-surface-static}"
|
||||||
|
NOLIBC_BIN="${NOLIBC_BIN:-$HERE/build/rs-surface-nolibc}"
|
||||||
|
if [ "$SURFACE_IMPL" = "static" ] || [ "$SURFACE_IMPL" = "nolibc" ]; then
|
||||||
|
BOOT_PKGS="${BOOT_PKGS:-ca-certificates}"
|
||||||
|
else
|
||||||
|
BOOT_PKGS="${BOOT_PKGS:-python3,ca-certificates}"
|
||||||
|
fi
|
||||||
|
|
||||||
|
mkdir -p "$ASSET_DIR" "$APKOVL_DIR" "$LOG_DIR"
|
||||||
|
|
||||||
|
need() {
|
||||||
|
command -v "$1" >/dev/null 2>&1 || {
|
||||||
|
echo "missing required command: $1" >&2
|
||||||
|
exit 1
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fetch_asset() {
|
||||||
|
name="$1"
|
||||||
|
if [ ! -s "$ASSET_DIR/$name" ]; then
|
||||||
|
echo "fetching Alpine netboot asset: $name" >&2
|
||||||
|
curl -fL "$ALPINE_BASE_URL/$name" -o "$ASSET_DIR/$name"
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
build_apkovl() {
|
||||||
|
rm -rf "$APKOVL_DIR/root"
|
||||||
|
mkdir -p \
|
||||||
|
"$APKOVL_DIR/root/etc/apk" \
|
||||||
|
"$APKOVL_DIR/root/etc/init.d" \
|
||||||
|
"$APKOVL_DIR/root/etc/network" \
|
||||||
|
"$APKOVL_DIR/root/etc/runlevels/default" \
|
||||||
|
"$APKOVL_DIR/root/etc/rs-surface" \
|
||||||
|
"$APKOVL_DIR/root/opt/rs-surface"
|
||||||
|
|
||||||
|
printf '%s\n' "$ALPINE_REPO_URL" > "$APKOVL_DIR/root/etc/apk/repositories"
|
||||||
|
cat > "$APKOVL_DIR/root/etc/network/interfaces" <<'EOF'
|
||||||
|
auto lo
|
||||||
|
iface lo inet loopback
|
||||||
|
|
||||||
|
auto eth0
|
||||||
|
iface eth0 inet dhcp
|
||||||
|
EOF
|
||||||
|
if [ "$SURFACE_IMPL" = "static" ]; then
|
||||||
|
test -x "$STATIC_BIN"
|
||||||
|
cp "$STATIC_BIN" "$APKOVL_DIR/root/opt/rs-surface/rs-surface-static"
|
||||||
|
elif [ "$SURFACE_IMPL" = "nolibc" ]; then
|
||||||
|
test -x "$NOLIBC_BIN"
|
||||||
|
cp "$NOLIBC_BIN" "$APKOVL_DIR/root/opt/rs-surface/rs-surface-nolibc"
|
||||||
|
else
|
||||||
|
cp "$SURFACE_ROOT/server.py" "$APKOVL_DIR/root/opt/rs-surface/server.py"
|
||||||
|
fi
|
||||||
|
cp "$SURFACE_ROOT/profiles/xen-alpine-surface.json" "$APKOVL_DIR/root/etc/rs-surface/node.json"
|
||||||
|
|
||||||
|
if [ "$SURFACE_IMPL" = "static" ]; then
|
||||||
|
cat > "$APKOVL_DIR/root/opt/rs-surface/boot-rs-surface.sh" <<'EOF'
|
||||||
|
#!/bin/sh
|
||||||
|
set -eu
|
||||||
|
echo "rs-surface static qemu bootstrap starting" >&2
|
||||||
|
mkdir -p /var/lib/rs-surface /var/log/rs-surface /run/rs-surface /mnt/topological-storage
|
||||||
|
export RS_SURFACE_NODE="${RS_SURFACE_NODE:-xen-alpine-surface}"
|
||||||
|
export RS_SURFACE_ROLE="${RS_SURFACE_ROLE:-gcl-edge}"
|
||||||
|
export RS_SURFACE_MODE="${RS_SURFACE_MODE:-recovery}"
|
||||||
|
export RS_SURFACE_HOST="${RS_SURFACE_HOST:-0.0.0.0}"
|
||||||
|
export RS_SURFACE_PORT="${RS_SURFACE_PORT:-8080}"
|
||||||
|
exec /opt/rs-surface/rs-surface-static
|
||||||
|
EOF
|
||||||
|
elif [ "$SURFACE_IMPL" = "nolibc" ]; then
|
||||||
|
cat > "$APKOVL_DIR/root/opt/rs-surface/boot-rs-surface.sh" <<'EOF'
|
||||||
|
#!/bin/sh
|
||||||
|
set -eu
|
||||||
|
echo "rs-surface nolibc qemu bootstrap starting" >&2
|
||||||
|
exec /opt/rs-surface/rs-surface-nolibc
|
||||||
|
EOF
|
||||||
|
else
|
||||||
|
cat > "$APKOVL_DIR/root/opt/rs-surface/boot-rs-surface.sh" <<'EOF'
|
||||||
|
#!/bin/sh
|
||||||
|
set -eu
|
||||||
|
echo "rs-surface qemu bootstrap starting" >&2
|
||||||
|
if ! command -v python3 >/dev/null 2>&1; then
|
||||||
|
apk add --no-cache python3 ca-certificates >&2
|
||||||
|
fi
|
||||||
|
mkdir -p /var/lib/rs-surface /var/log/rs-surface /run/rs-surface /mnt/topological-storage
|
||||||
|
export RS_SURFACE_PROFILE="${RS_SURFACE_PROFILE:-/etc/rs-surface/node.json}"
|
||||||
|
export RS_SURFACE_STATE="${RS_SURFACE_STATE:-/var/lib/rs-surface}"
|
||||||
|
export RS_SURFACE_MOUNT="${RS_SURFACE_MOUNT:-/mnt/topological-storage}"
|
||||||
|
export RS_SURFACE_HOST="${RS_SURFACE_HOST:-0.0.0.0}"
|
||||||
|
export RS_SURFACE_PORT="${RS_SURFACE_PORT:-8080}"
|
||||||
|
exec python3 /opt/rs-surface/server.py
|
||||||
|
EOF
|
||||||
|
fi
|
||||||
|
|
||||||
|
cat > "$APKOVL_DIR/root/etc/init.d/rs-surface" <<'EOF'
|
||||||
|
#!/sbin/openrc-run
|
||||||
|
name="Research Stack QEMU rs-surface smoke"
|
||||||
|
description="Disposable QEMU smoke for the Alpine embedded surface"
|
||||||
|
supervisor=supervise-daemon
|
||||||
|
command="/bin/sh"
|
||||||
|
command_args="/opt/rs-surface/boot-rs-surface.sh"
|
||||||
|
pidfile="/run/rs-surface/rs-surface.pid"
|
||||||
|
output_log="/var/log/rs-surface/stdout.log"
|
||||||
|
error_log="/var/log/rs-surface/stderr.log"
|
||||||
|
respawn_delay=2
|
||||||
|
respawn_max=2
|
||||||
|
|
||||||
|
depend() {
|
||||||
|
need net
|
||||||
|
after firewall
|
||||||
|
}
|
||||||
|
|
||||||
|
start_pre() {
|
||||||
|
mkdir -p /run/rs-surface /var/lib/rs-surface /var/log/rs-surface /mnt/topological-storage
|
||||||
|
}
|
||||||
|
EOF
|
||||||
|
|
||||||
|
chmod 0755 \
|
||||||
|
"$APKOVL_DIR/root/opt/rs-surface/boot-rs-surface.sh" \
|
||||||
|
"$APKOVL_DIR/root/etc/init.d/rs-surface"
|
||||||
|
ln -s /etc/init.d/rs-surface "$APKOVL_DIR/root/etc/runlevels/default/rs-surface"
|
||||||
|
(cd "$APKOVL_DIR/root" && tar -czf ../rs.apkovl.tar.gz .)
|
||||||
|
}
|
||||||
|
|
||||||
|
cleanup() {
|
||||||
|
if [ -s "$PID_FILE" ]; then
|
||||||
|
pid="$(cat "$PID_FILE")"
|
||||||
|
if kill -0 "$pid" >/dev/null 2>&1; then
|
||||||
|
kill "$pid" >/dev/null 2>&1 || true
|
||||||
|
sleep 1
|
||||||
|
kill -9 "$pid" >/dev/null 2>&1 || true
|
||||||
|
fi
|
||||||
|
rm -f "$PID_FILE"
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
write_receipt() {
|
||||||
|
ok="$1"
|
||||||
|
reason="$2"
|
||||||
|
python3 - "$RECEIPT" "$ok" "$reason" "$HOST_PORT" "$SERIAL_LOG" <<'PY'
|
||||||
|
import json, sys, time
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
path = Path(sys.argv[1])
|
||||||
|
ok = sys.argv[2] == "true"
|
||||||
|
serial = Path(sys.argv[5])
|
||||||
|
tail = ""
|
||||||
|
if serial.exists():
|
||||||
|
data = serial.read_text(errors="replace")
|
||||||
|
tail = "\n".join(data.splitlines()[-80:])
|
||||||
|
receipt = {
|
||||||
|
"ok": ok,
|
||||||
|
"reason": sys.argv[3],
|
||||||
|
"checked_at": time.time(),
|
||||||
|
"host_url": f"http://127.0.0.1:{sys.argv[4]}/health",
|
||||||
|
"serial_log": str(serial),
|
||||||
|
"serial_tail": tail,
|
||||||
|
}
|
||||||
|
path.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
path.write_text(json.dumps(receipt, indent=2, sort_keys=True) + "\n")
|
||||||
|
print(path)
|
||||||
|
PY
|
||||||
|
}
|
||||||
|
|
||||||
|
need curl
|
||||||
|
need qemu-system-x86_64
|
||||||
|
need tar
|
||||||
|
need python3
|
||||||
|
|
||||||
|
fetch_asset vmlinuz-virt
|
||||||
|
fetch_asset initramfs-virt
|
||||||
|
fetch_asset modloop-virt
|
||||||
|
build_apkovl
|
||||||
|
cleanup
|
||||||
|
trap cleanup EXIT INT TERM
|
||||||
|
|
||||||
|
rm -f "$SERIAL_LOG" "$RECEIPT"
|
||||||
|
|
||||||
|
qemu-system-x86_64 \
|
||||||
|
-machine accel=tcg,type=q35 \
|
||||||
|
-cpu max \
|
||||||
|
-m "$MEMORY_MB" \
|
||||||
|
-smp 1 \
|
||||||
|
-nographic \
|
||||||
|
-no-reboot \
|
||||||
|
-kernel "$ASSET_DIR/vmlinuz-virt" \
|
||||||
|
-initrd "$ASSET_DIR/initramfs-virt" \
|
||||||
|
-append "console=ttyS0 modules=loop,squashfs,sd-mod,virtio_net,virtio_pci,virtio_blk ip=dhcp alpine_repo=$ALPINE_REPO_URL modloop=$ALPINE_BASE_URL/modloop-virt apkovl=/dev/vda1:vfat:rs.apkovl.tar.gz pkgs=$BOOT_PKGS" \
|
||||||
|
-drive "file=fat:rw:$APKOVL_DIR,format=raw,if=virtio" \
|
||||||
|
-netdev "user,id=net0,hostfwd=tcp:127.0.0.1:$HOST_PORT-:$GUEST_PORT" \
|
||||||
|
-device virtio-net-pci,netdev=net0 \
|
||||||
|
-serial "file:$SERIAL_LOG" \
|
||||||
|
-monitor none \
|
||||||
|
>/dev/null 2>&1 &
|
||||||
|
|
||||||
|
echo "$!" > "$PID_FILE"
|
||||||
|
|
||||||
|
deadline=$(( $(date +%s) + TIMEOUT_SECONDS ))
|
||||||
|
while [ "$(date +%s)" -lt "$deadline" ]; do
|
||||||
|
if curl -fsS "http://127.0.0.1:$HOST_PORT/health" >/tmp/rs-surface-qemu-health.json 2>/dev/null; then
|
||||||
|
python3 - "$RECEIPT" "$HOST_PORT" "$SERIAL_LOG" /tmp/rs-surface-qemu-health.json <<'PY'
|
||||||
|
import json, sys, time
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
receipt = {
|
||||||
|
"ok": True,
|
||||||
|
"checked_at": time.time(),
|
||||||
|
"host_url": f"http://127.0.0.1:{sys.argv[2]}/health",
|
||||||
|
"health": json.loads(Path(sys.argv[4]).read_text()),
|
||||||
|
"serial_log": sys.argv[3],
|
||||||
|
}
|
||||||
|
Path(sys.argv[1]).write_text(json.dumps(receipt, indent=2, sort_keys=True) + "\n")
|
||||||
|
print(sys.argv[1])
|
||||||
|
PY
|
||||||
|
cat "$RECEIPT"
|
||||||
|
exit 0
|
||||||
|
fi
|
||||||
|
sleep 2
|
||||||
|
done
|
||||||
|
|
||||||
|
write_receipt false "timeout waiting for forwarded /health"
|
||||||
|
cat "$RECEIPT"
|
||||||
|
exit 1
|
||||||
|
|
@ -0,0 +1,60 @@
|
||||||
|
#!/usr/bin/env python3
|
||||||
|
"""HTTP smoke check for the Alpine/Xen rs-surface carrier."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import json
|
||||||
|
import time
|
||||||
|
import urllib.error
|
||||||
|
import urllib.request
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
|
||||||
|
def fetch_json(url: str, timeout: float) -> dict[str, Any]:
|
||||||
|
with urllib.request.urlopen(url, timeout=timeout) as response:
|
||||||
|
payload = response.read()
|
||||||
|
data = json.loads(payload.decode("utf-8"))
|
||||||
|
if not isinstance(data, dict):
|
||||||
|
raise ValueError(f"{url} did not return a JSON object")
|
||||||
|
return data
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> int:
|
||||||
|
parser = argparse.ArgumentParser()
|
||||||
|
parser.add_argument("--host", default="127.0.0.1")
|
||||||
|
parser.add_argument("--port", type=int, default=8080)
|
||||||
|
parser.add_argument("--timeout", type=float, default=3.0)
|
||||||
|
parser.add_argument("--output", type=Path)
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
base = f"http://{args.host}:{args.port}"
|
||||||
|
paths = ["/health", "/status", "/metrics", "/primitives"]
|
||||||
|
receipt: dict[str, Any] = {
|
||||||
|
"ok": True,
|
||||||
|
"checked_at": time.time(),
|
||||||
|
"base_url": base,
|
||||||
|
"checks": {},
|
||||||
|
}
|
||||||
|
|
||||||
|
for path in paths:
|
||||||
|
try:
|
||||||
|
receipt["checks"][path] = fetch_json(base + path, args.timeout)
|
||||||
|
except (OSError, urllib.error.URLError, json.JSONDecodeError, ValueError) as exc:
|
||||||
|
receipt["ok"] = False
|
||||||
|
receipt["checks"][path] = {"ok": False, "error": str(exc)}
|
||||||
|
|
||||||
|
health = receipt["checks"].get("/health", {})
|
||||||
|
receipt["node"] = health.get("node")
|
||||||
|
receipt["surface_version"] = health.get("surface_version")
|
||||||
|
|
||||||
|
text = json.dumps(receipt, indent=2, sort_keys=True) + "\n"
|
||||||
|
if args.output:
|
||||||
|
args.output.write_text(text, encoding="utf-8")
|
||||||
|
print(text, end="")
|
||||||
|
return 0 if receipt["ok"] else 1
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
raise SystemExit(main())
|
||||||
|
|
@ -0,0 +1,31 @@
|
||||||
|
# Minimal Xen domain template for an Alpine rs-surface appliance.
|
||||||
|
#
|
||||||
|
# Edit disk, vif bridge, kernel, ramdisk, and root before `xl create`.
|
||||||
|
# The guest should be installed with Alpine linux-lts, OpenRC, and python3.
|
||||||
|
|
||||||
|
name = "rs-alpine-surface"
|
||||||
|
type = "pvh"
|
||||||
|
memory = 192
|
||||||
|
vcpus = 1
|
||||||
|
maxvcpus = 1
|
||||||
|
|
||||||
|
# Serial-first recovery lane.
|
||||||
|
serial = "pty"
|
||||||
|
extra = "console=hvc0 root=/dev/xvda3 ro modules=ext4 quiet"
|
||||||
|
|
||||||
|
# Use host-exported Alpine kernel/initramfs for PVH boot, or switch to pygrub
|
||||||
|
# if your Xen host policy prefers booting the guest's own /boot.
|
||||||
|
kernel = "/var/lib/xen/images/rs-alpine/vmlinuz-lts"
|
||||||
|
ramdisk = "/var/lib/xen/images/rs-alpine/initramfs-lts"
|
||||||
|
|
||||||
|
disk = [
|
||||||
|
"format=raw,vdev=xvda,access=rw,target=/var/lib/xen/images/rs-alpine/rootfs.raw"
|
||||||
|
]
|
||||||
|
|
||||||
|
vif = [
|
||||||
|
"bridge=xenbr0,model=virtio"
|
||||||
|
]
|
||||||
|
|
||||||
|
on_poweroff = "destroy"
|
||||||
|
on_reboot = "restart"
|
||||||
|
on_crash = "preserve"
|
||||||
115
4-Infrastructure/infra/nodes/xen-alpine-surface.json
Normal file
115
4-Infrastructure/infra/nodes/xen-alpine-surface.json
Normal file
|
|
@ -0,0 +1,115 @@
|
||||||
|
{
|
||||||
|
"surface_version": "0.1",
|
||||||
|
"node_id": "xen-alpine-surface",
|
||||||
|
"role": "gcl-edge",
|
||||||
|
"mode_default": "recovery",
|
||||||
|
"operational_model": "appliance",
|
||||||
|
"memory_budget_mb": 192,
|
||||||
|
"disk_budget_gb": 2,
|
||||||
|
"local_state_budget_mb": 64,
|
||||||
|
"api": {
|
||||||
|
"plain_health_port": 8080,
|
||||||
|
"websocket_port": 8080,
|
||||||
|
"bind": "localhost"
|
||||||
|
},
|
||||||
|
"storage": {
|
||||||
|
"provider": "none",
|
||||||
|
"mount_point": "/mnt/topological-storage",
|
||||||
|
"remote": "",
|
||||||
|
"required_for_boot": false,
|
||||||
|
"write_mode": "outbox",
|
||||||
|
"spool_budget_mb": 32
|
||||||
|
},
|
||||||
|
"topological_substrate": {
|
||||||
|
"class": "xen-alpine-nanokernel-carrier",
|
||||||
|
"carrier": "alpine-openrc-qemu-xen",
|
||||||
|
"kernel_exposes_primitives": true,
|
||||||
|
"vector_width_bits": 64,
|
||||||
|
"compute_slots": 1,
|
||||||
|
"memory_reserve_mb": 64,
|
||||||
|
"primitives": [
|
||||||
|
"health",
|
||||||
|
"status",
|
||||||
|
"metrics",
|
||||||
|
"attest",
|
||||||
|
"compress",
|
||||||
|
"rgflow",
|
||||||
|
"route",
|
||||||
|
"plan_route",
|
||||||
|
"mount_status",
|
||||||
|
"snapshot",
|
||||||
|
"receipt"
|
||||||
|
],
|
||||||
|
"accelerators": [
|
||||||
|
"virtio_net",
|
||||||
|
"virtio_blk",
|
||||||
|
"serial_console"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"boot_strategy": {
|
||||||
|
"target_layer": "layer0-gcl-nanokernel",
|
||||||
|
"handoff": "manual",
|
||||||
|
"linux_role": "runtime_carrier",
|
||||||
|
"destructive_handoff_allowed": false,
|
||||||
|
"layer0_image": {
|
||||||
|
"path": "/boot/gcl/layer0-xen-alpine.img",
|
||||||
|
"sha256": "TBD",
|
||||||
|
"signature": "/boot/gcl/layer0-xen-alpine.img.sig",
|
||||||
|
"public_key": "/etc/rs-surface/layer0.pub",
|
||||||
|
"cmdline": "console=hvc0 gcl.node=xen-alpine-surface gcl.mode=layer0"
|
||||||
|
},
|
||||||
|
"rollback": {
|
||||||
|
"boot_entry": "alpine-openrc-rs-surface-last-good",
|
||||||
|
"previous_image": "/boot/gcl/layer0-xen-alpine.previous.img",
|
||||||
|
"provider_rescue": "xen-console-or-qemu-serial"
|
||||||
|
},
|
||||||
|
"preserve": [
|
||||||
|
"serial_console",
|
||||||
|
"node_identity",
|
||||||
|
"last_good_receipt",
|
||||||
|
"rs_surface_profile",
|
||||||
|
"qemu_smoke_harness"
|
||||||
|
],
|
||||||
|
"wipe": [
|
||||||
|
"general_linux_userspace",
|
||||||
|
"browser_session_state",
|
||||||
|
"model_router",
|
||||||
|
"ad_hoc_shell_state"
|
||||||
|
],
|
||||||
|
"required_gates": [
|
||||||
|
"serial_or_console_health_pulse",
|
||||||
|
"qemu_health_probe",
|
||||||
|
"rollback_boot_entry",
|
||||||
|
"signed_layer0_image",
|
||||||
|
"out_of_band_recovery_window"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"capabilities": [
|
||||||
|
"health",
|
||||||
|
"status",
|
||||||
|
"metrics",
|
||||||
|
"attest",
|
||||||
|
"compress",
|
||||||
|
"rgflow",
|
||||||
|
"route",
|
||||||
|
"plan_route",
|
||||||
|
"mount_status",
|
||||||
|
"snapshot",
|
||||||
|
"nanokernel",
|
||||||
|
"topological_substrate",
|
||||||
|
"kernel_primitives",
|
||||||
|
"recovery"
|
||||||
|
],
|
||||||
|
"disabled": [
|
||||||
|
"full_git_checkout",
|
||||||
|
"local_training",
|
||||||
|
"local_build",
|
||||||
|
"large_database",
|
||||||
|
"browser_session_state",
|
||||||
|
"provider_secrets",
|
||||||
|
"legacy_warden_service",
|
||||||
|
"legacy_tardy_service",
|
||||||
|
"legacy_substrate_index_service",
|
||||||
|
"legacy_compression_gateway_service"
|
||||||
|
]
|
||||||
|
}
|
||||||
42
4-Infrastructure/kernel/qfox_topology_manager/Makefile
Normal file
42
4-Infrastructure/kernel/qfox_topology_manager/Makefile
Normal file
|
|
@ -0,0 +1,42 @@
|
||||||
|
# QFox Topology Manager kernel module
|
||||||
|
#
|
||||||
|
# Builds out-of-tree against the running kernel by default. CachyOS builds the
|
||||||
|
# stock kernel with clang/LLVM, so LLVM=1 is the default here too.
|
||||||
|
|
||||||
|
KVER ?= $(shell uname -r)
|
||||||
|
KDIR ?= /usr/lib/modules/$(KVER)/build
|
||||||
|
|
||||||
|
LLVM ?= 1
|
||||||
|
CC ?= clang
|
||||||
|
ifeq ($(origin CC),default)
|
||||||
|
CC := clang
|
||||||
|
endif
|
||||||
|
|
||||||
|
obj-m += qfox_topology_manager.o
|
||||||
|
ccflags-y += -g
|
||||||
|
|
||||||
|
all:
|
||||||
|
$(MAKE) CC=$(CC) LLVM=$(LLVM) -C $(KDIR) M=$(CURDIR) modules
|
||||||
|
|
||||||
|
clean:
|
||||||
|
$(MAKE) -C $(KDIR) M=$(CURDIR) clean
|
||||||
|
|
||||||
|
load:
|
||||||
|
sudo modprobe qfox_topology_manager || sudo insmod qfox_topology_manager.ko
|
||||||
|
|
||||||
|
unload:
|
||||||
|
sudo modprobe -r qfox_topology_manager || sudo rmmod qfox_topology_manager
|
||||||
|
|
||||||
|
info:
|
||||||
|
modinfo qfox_topology_manager.ko
|
||||||
|
|
||||||
|
debug-attach:
|
||||||
|
sudo ./qfox_topology_debug.sh attach
|
||||||
|
|
||||||
|
debug-snapshot:
|
||||||
|
sudo ./qfox_topology_debug.sh snapshot
|
||||||
|
|
||||||
|
debug-detach:
|
||||||
|
sudo ./qfox_topology_debug.sh detach
|
||||||
|
|
||||||
|
.PHONY: all clean load unload info debug-attach debug-snapshot debug-detach
|
||||||
149
4-Infrastructure/kernel/qfox_topology_manager/README.md
Normal file
149
4-Infrastructure/kernel/qfox_topology_manager/README.md
Normal file
|
|
@ -0,0 +1,149 @@
|
||||||
|
# QFox Topology Manager
|
||||||
|
|
||||||
|
`qfox_topology_manager` is the passive Linux-carrier phase of the GCL
|
||||||
|
nanokernel idea. It is intentionally an out-of-tree kernel module first, not a
|
||||||
|
core-kernel patch: the module observes kernel-adjacent events, maps them into
|
||||||
|
named topology slots, and exposes a receipt surface for userspace.
|
||||||
|
|
||||||
|
The module is a module/driver hybrid:
|
||||||
|
|
||||||
|
- misc device: `/dev/qfox_topoman`
|
||||||
|
- sysfs: `/sys/kernel/qfox_topology_manager/`
|
||||||
|
- debugfs: `/sys/kernel/debug/qfox_topology_manager/`
|
||||||
|
- kernel notifiers: network-device events and reboot/power events
|
||||||
|
- manual injection path for userspace shims and future kernel call sites
|
||||||
|
|
||||||
|
The operating rule is conservative:
|
||||||
|
|
||||||
|
> The GCL nanokernel may classify and receipt carrier behavior, but this Linux
|
||||||
|
> module does not become the root of trust and does not enforce policy.
|
||||||
|
|
||||||
|
## Build
|
||||||
|
|
||||||
|
```bash
|
||||||
|
make
|
||||||
|
sudo insmod qfox_topology_manager.ko
|
||||||
|
python3 probe_qfox_topology_manager.py --json
|
||||||
|
sudo rmmod qfox_topology_manager
|
||||||
|
```
|
||||||
|
|
||||||
|
The repo path contains a space, and Linux kbuild does not handle `M=...` paths
|
||||||
|
with spaces reliably. For now, build from a spaceless staging directory:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
stage=/tmp/qfox_topology_manager_build
|
||||||
|
rm -rf "$stage"
|
||||||
|
install -d "$stage"
|
||||||
|
cp Makefile dkms.conf qfox_topology_manager.c qfox_topology_debug.sh \
|
||||||
|
probe_qfox_topology_manager.py README.md "$stage"/
|
||||||
|
make -C "$stage" CC=clang LLVM=1
|
||||||
|
```
|
||||||
|
|
||||||
|
## DKMS Install
|
||||||
|
|
||||||
|
```bash
|
||||||
|
sudo install -d /usr/src/qfox-topology-manager-0.1.0
|
||||||
|
sudo cp Makefile dkms.conf qfox_topology_manager.c /usr/src/qfox-topology-manager-0.1.0/
|
||||||
|
sudo dkms add -m qfox-topology-manager -v 0.1.0
|
||||||
|
sudo dkms build -m qfox-topology-manager -v 0.1.0
|
||||||
|
sudo dkms install -m qfox-topology-manager -v 0.1.0
|
||||||
|
sudo modprobe qfox_topology_manager
|
||||||
|
```
|
||||||
|
|
||||||
|
Auto-load is intentionally a separate choice:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
echo qfox_topology_manager | sudo tee /etc/modules-load.d/qfox-topology-manager.conf
|
||||||
|
```
|
||||||
|
|
||||||
|
## Topology Slots
|
||||||
|
|
||||||
|
| Slot | Meaning |
|
||||||
|
| --- | --- |
|
||||||
|
| `boot` | module lifecycle and boot-adjacent state |
|
||||||
|
| `sched` | scheduler / execution carrier observations |
|
||||||
|
| `mm` | memory-management observations |
|
||||||
|
| `fs` | filesystem and VFS observations |
|
||||||
|
| `block` | block-device / request observations |
|
||||||
|
| `net` | network-device observations |
|
||||||
|
| `device` | driver/device lifecycle observations |
|
||||||
|
| `power` | reboot and power-transition observations |
|
||||||
|
| `security` | admission, refusal, and policy-surface observations |
|
||||||
|
| `gpu` | GPU/display carrier observations |
|
||||||
|
| `user` | userspace shim injection |
|
||||||
|
| `receipt` | receipt or attestation emission |
|
||||||
|
|
||||||
|
## Interfaces
|
||||||
|
|
||||||
|
Read module status:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
cat /sys/kernel/qfox_topology_manager/status
|
||||||
|
cat /sys/kernel/qfox_topology_manager/slots
|
||||||
|
```
|
||||||
|
|
||||||
|
Measure the first activity average:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
python3 probe_qfox_topology_manager.py --sample-sec 10
|
||||||
|
```
|
||||||
|
|
||||||
|
Inject a userspace observation:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
echo "fs repo_scan /home/allaun/Documents/Research Stack" \
|
||||||
|
| sudo tee /sys/kernel/qfox_topology_manager/inject
|
||||||
|
```
|
||||||
|
|
||||||
|
Read the event surface:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
sudo cat /sys/kernel/debug/qfox_topology_manager/events
|
||||||
|
cat /dev/qfox_topoman
|
||||||
|
```
|
||||||
|
|
||||||
|
Write to `/dev/qfox_topoman` to inject an event. The first token may be a slot
|
||||||
|
name; otherwise the event is treated as `user`.
|
||||||
|
|
||||||
|
## Live Debugger
|
||||||
|
|
||||||
|
Attach the non-halting ftrace debugger:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
sudo ./qfox_topology_debug.sh attach
|
||||||
|
sudo ./qfox_topology_debug.sh snapshot
|
||||||
|
sudo ./qfox_topology_debug.sh detach
|
||||||
|
```
|
||||||
|
|
||||||
|
This uses function-graph tracing for the module call path. It does not stop the
|
||||||
|
kernel, require a reboot, or replace the normal boot path.
|
||||||
|
|
||||||
|
Sample wider kernel tracepoints into topology slots:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
sudo python3 qfox_topology_trace_sampler.py --duration 10
|
||||||
|
```
|
||||||
|
|
||||||
|
The sampler temporarily enables available tracepoints for scheduler, memory,
|
||||||
|
filesystem syscall, block, network, power, and interrupt/device surfaces. It
|
||||||
|
restores prior tracepoint state and can inject per-slot summaries back into the
|
||||||
|
module's misc-device receipt path.
|
||||||
|
|
||||||
|
When `/home/allaun/Gdrive` is mounted, `qfox-topology-sample` also mirrors the
|
||||||
|
small JSON receipts to:
|
||||||
|
|
||||||
|
```text
|
||||||
|
/home/allaun/Gdrive/topological_storage/research-stack/qfox-topology-manager/
|
||||||
|
```
|
||||||
|
|
||||||
|
Generate an optimization-target report from collected samples:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
python3 qfox_topology_optimizer_report.py
|
||||||
|
```
|
||||||
|
|
||||||
|
The report ranks topology slots by observed activity and produces advisory
|
||||||
|
targets such as scheduler wakeup churn, memory allocation churn, block IO,
|
||||||
|
network chatter, device/IRQ pressure, and power/idle transition churn. The
|
||||||
|
report is a receipt for what to inspect next; it does not change sysctls,
|
||||||
|
kernel parameters, IRQ affinity, CPU policy, or scheduler behavior.
|
||||||
9
4-Infrastructure/kernel/qfox_topology_manager/dkms.conf
Normal file
9
4-Infrastructure/kernel/qfox_topology_manager/dkms.conf
Normal file
|
|
@ -0,0 +1,9 @@
|
||||||
|
PACKAGE_NAME="qfox-topology-manager"
|
||||||
|
PACKAGE_VERSION="0.1.0"
|
||||||
|
|
||||||
|
BUILT_MODULE_NAME[0]="qfox_topology_manager"
|
||||||
|
DEST_MODULE_LOCATION[0]="/kernel/extra"
|
||||||
|
AUTOINSTALL="yes"
|
||||||
|
|
||||||
|
MAKE[0]="make KVER=${kernelver}"
|
||||||
|
CLEAN="make clean KVER=${kernelver}"
|
||||||
205
4-Infrastructure/kernel/qfox_topology_manager/probe_qfox_topology_manager.py
Executable file
205
4-Infrastructure/kernel/qfox_topology_manager/probe_qfox_topology_manager.py
Executable file
|
|
@ -0,0 +1,205 @@
|
||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Probe the QFox topology-manager kernel module.
|
||||||
|
|
||||||
|
The probe is receipt-shaped and intentionally non-failing for missing module
|
||||||
|
state: absence is data. CLI/write errors still exit non-zero.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import subprocess
|
||||||
|
import sys
|
||||||
|
import time
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
SCHEMA = "research_stack_qfox_topology_manager_probe_v1"
|
||||||
|
SYSFS = Path("/sys/kernel/qfox_topology_manager")
|
||||||
|
DEBUGFS = Path("/sys/kernel/debug/qfox_topology_manager")
|
||||||
|
DEV = Path("/dev/qfox_topoman")
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_kv(text: str | None) -> dict[str, str]:
|
||||||
|
values: dict[str, str] = {}
|
||||||
|
if not text:
|
||||||
|
return values
|
||||||
|
for raw in text.splitlines():
|
||||||
|
if "=" not in raw:
|
||||||
|
continue
|
||||||
|
key, value = raw.split("=", 1)
|
||||||
|
values[key.strip()] = value.strip()
|
||||||
|
return values
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_int(value: str | None) -> int | None:
|
||||||
|
if value is None:
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
return int(value, 0)
|
||||||
|
except ValueError:
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _read(path: Path, max_bytes: int = 65536) -> str | None:
|
||||||
|
try:
|
||||||
|
with path.open("r", encoding="utf-8", errors="replace") as handle:
|
||||||
|
return handle.read(max_bytes)
|
||||||
|
except OSError:
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _cmd(argv: list[str]) -> dict[str, Any]:
|
||||||
|
try:
|
||||||
|
result = subprocess.run(argv, capture_output=True, text=True, timeout=5)
|
||||||
|
except (OSError, subprocess.TimeoutExpired) as exc:
|
||||||
|
return {"ok": False, "error": str(exc), "stdout": "", "stderr": ""}
|
||||||
|
return {
|
||||||
|
"ok": result.returncode == 0,
|
||||||
|
"returncode": result.returncode,
|
||||||
|
"stdout": result.stdout.strip(),
|
||||||
|
"stderr": result.stderr.strip(),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _snapshot() -> dict[str, Any]:
|
||||||
|
status = _read(SYSFS / "status")
|
||||||
|
slots = _read(SYSFS / "slots")
|
||||||
|
status_kv = _parse_kv(status)
|
||||||
|
slots_kv = _parse_kv(slots)
|
||||||
|
events = _parse_int(status_kv.get("events"))
|
||||||
|
avg_x1000 = _parse_int(status_kv.get("avg_events_per_sec_x1000"))
|
||||||
|
return {
|
||||||
|
"timestamp_unix": time.time(),
|
||||||
|
"status": status,
|
||||||
|
"status_kv": status_kv,
|
||||||
|
"slots": slots,
|
||||||
|
"slots_kv": slots_kv,
|
||||||
|
"events": events,
|
||||||
|
"avg_events_per_sec": None if avg_x1000 is None else avg_x1000 / 1000.0,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def build_probe(sample_sec: float = 0.0) -> dict[str, Any]:
|
||||||
|
loaded = "qfox_topology_manager" in (_read(Path("/proc/modules")) or "")
|
||||||
|
first = _snapshot()
|
||||||
|
sample: dict[str, Any] | None = None
|
||||||
|
if sample_sec > 0:
|
||||||
|
time.sleep(sample_sec)
|
||||||
|
second = _snapshot()
|
||||||
|
first_events = first.get("events")
|
||||||
|
second_events = second.get("events")
|
||||||
|
delta_events = None
|
||||||
|
if isinstance(first_events, int) and isinstance(second_events, int):
|
||||||
|
delta_events = max(0, second_events - first_events)
|
||||||
|
sample = {
|
||||||
|
"seconds": sample_sec,
|
||||||
|
"start": first,
|
||||||
|
"end": second,
|
||||||
|
"delta_events": delta_events,
|
||||||
|
"events_per_second": None
|
||||||
|
if delta_events is None
|
||||||
|
else delta_events / sample_sec,
|
||||||
|
}
|
||||||
|
return {
|
||||||
|
"schema": SCHEMA,
|
||||||
|
"timestamp_unix": int(time.time()),
|
||||||
|
"module": {
|
||||||
|
"name": "qfox_topology_manager",
|
||||||
|
"loaded": loaded,
|
||||||
|
"modinfo": _cmd(["modinfo", "qfox_topology_manager"]),
|
||||||
|
},
|
||||||
|
"interfaces": {
|
||||||
|
"sysfs": str(SYSFS),
|
||||||
|
"sysfs_present": SYSFS.exists(),
|
||||||
|
"debugfs": str(DEBUGFS),
|
||||||
|
"debugfs_present": DEBUGFS.exists(),
|
||||||
|
"device": str(DEV),
|
||||||
|
"device_present": DEV.exists(),
|
||||||
|
},
|
||||||
|
"status": first["status"],
|
||||||
|
"status_kv": first["status_kv"],
|
||||||
|
"slots": first["slots"],
|
||||||
|
"slots_kv": first["slots_kv"],
|
||||||
|
"mode": _read(SYSFS / "mode"),
|
||||||
|
"events": _read(DEBUGFS / "events"),
|
||||||
|
"average": {
|
||||||
|
"since_load_events_per_second": first["avg_events_per_sec"],
|
||||||
|
"sample": sample,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def emit_text(probe: dict[str, Any]) -> str:
|
||||||
|
module = probe["module"]
|
||||||
|
interfaces = probe["interfaces"]
|
||||||
|
lines = [
|
||||||
|
f"schema: {probe['schema']}",
|
||||||
|
f"module: {module['name']}",
|
||||||
|
f"loaded: {module['loaded']}",
|
||||||
|
f"sysfs: {interfaces['sysfs_present']} ({interfaces['sysfs']})",
|
||||||
|
f"debugfs: {interfaces['debugfs_present']} ({interfaces['debugfs']})",
|
||||||
|
f"device: {interfaces['device_present']} ({interfaces['device']})",
|
||||||
|
]
|
||||||
|
if probe.get("status"):
|
||||||
|
lines.append("")
|
||||||
|
lines.append("status:")
|
||||||
|
lines.append(probe["status"].rstrip())
|
||||||
|
if probe.get("slots"):
|
||||||
|
lines.append("")
|
||||||
|
lines.append("slots:")
|
||||||
|
lines.append(probe["slots"].rstrip())
|
||||||
|
average = probe.get("average") or {}
|
||||||
|
if average.get("since_load_events_per_second") is not None:
|
||||||
|
lines.append("")
|
||||||
|
lines.append(
|
||||||
|
"average_since_load_events_per_second: "
|
||||||
|
f"{average['since_load_events_per_second']:.3f}"
|
||||||
|
)
|
||||||
|
sample = average.get("sample")
|
||||||
|
if sample:
|
||||||
|
lines.append(
|
||||||
|
f"sample_{sample['seconds']:.3f}s_events_per_second: "
|
||||||
|
f"{sample['events_per_second']:.3f}"
|
||||||
|
if sample.get("events_per_second") is not None
|
||||||
|
else f"sample_{sample['seconds']:.3f}s_events_per_second: unavailable"
|
||||||
|
)
|
||||||
|
return "\n".join(lines) + "\n"
|
||||||
|
|
||||||
|
|
||||||
|
def main(argv: list[str] | None = None) -> int:
|
||||||
|
parser = argparse.ArgumentParser(description=__doc__)
|
||||||
|
parser.add_argument("--json", action="store_true", help="emit JSON")
|
||||||
|
parser.add_argument("--out", type=Path, help="write probe payload to path")
|
||||||
|
parser.add_argument(
|
||||||
|
"--sample-sec",
|
||||||
|
type=float,
|
||||||
|
default=0.0,
|
||||||
|
help="sleep for N seconds and report an event-rate delta",
|
||||||
|
)
|
||||||
|
args = parser.parse_args(argv)
|
||||||
|
|
||||||
|
if args.sample_sec < 0:
|
||||||
|
print("--sample-sec must be non-negative", file=sys.stderr)
|
||||||
|
return 2
|
||||||
|
|
||||||
|
probe = build_probe(args.sample_sec)
|
||||||
|
payload = json.dumps(probe, indent=2, sort_keys=True) + "\n"
|
||||||
|
rendered = payload if args.json else emit_text(probe)
|
||||||
|
|
||||||
|
if args.out:
|
||||||
|
try:
|
||||||
|
args.out.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
args.out.write_text(payload, encoding="utf-8")
|
||||||
|
except OSError as exc:
|
||||||
|
print(f"probe write failed: {exc}", file=sys.stderr)
|
||||||
|
return 2
|
||||||
|
|
||||||
|
print(rendered, end="")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
raise SystemExit(main())
|
||||||
31
4-Infrastructure/kernel/qfox_topology_manager/qfox-topology-sample
Executable file
31
4-Infrastructure/kernel/qfox_topology_manager/qfox-topology-sample
Executable file
|
|
@ -0,0 +1,31 @@
|
||||||
|
#!/usr/bin/env bash
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
duration="${QFOX_TOPOLOGY_SAMPLE_SEC:-5}"
|
||||||
|
out_dir="${QFOX_TOPOLOGY_SAMPLE_DIR:-/var/lib/qfox-topology-manager/samples}"
|
||||||
|
sampler="${QFOX_TOPOLOGY_SAMPLER:-/usr/src/qfox-topology-manager-0.1.0/qfox_topology_trace_sampler.py}"
|
||||||
|
reporter="${QFOX_TOPOLOGY_REPORTER:-/usr/src/qfox-topology-manager-0.1.0/qfox_topology_optimizer_report.py}"
|
||||||
|
report_dir="${QFOX_TOPOLOGY_REPORT_DIR:-/var/lib/qfox-topology-manager/reports}"
|
||||||
|
drive_dir="${QFOX_TOPOLOGY_DRIVE_DIR:-/home/allaun/Gdrive/topological_storage/research-stack/qfox-topology-manager}"
|
||||||
|
|
||||||
|
install -d -m 0755 "$out_dir"
|
||||||
|
out="$out_dir/trace-sample-$(date -u +%Y%m%dT%H%M%SZ).json"
|
||||||
|
|
||||||
|
python3 "$sampler" --duration "$duration" --json --out "$out" > "$out.tmp"
|
||||||
|
mv "$out.tmp" "$out.summary.json"
|
||||||
|
if [[ -x "$reporter" ]]; then
|
||||||
|
install -d -m 0755 "$report_dir"
|
||||||
|
report="$report_dir/optimization-report-$(date -u +%Y%m%dT%H%M%SZ).json"
|
||||||
|
python3 "$reporter" --samples-dir "$out_dir" --json --out "$report" --inject > "$report.summary.json"
|
||||||
|
cp "$report" "$report_dir/optimization-latest.json"
|
||||||
|
fi
|
||||||
|
|
||||||
|
if mountpoint -q /home/allaun/Gdrive; then
|
||||||
|
install -d -m 0755 "$drive_dir/samples" "$drive_dir/reports"
|
||||||
|
cp -f "$out" "$out.summary.json" "$drive_dir/samples/"
|
||||||
|
if [[ -n "${report:-}" && -f "$report" ]]; then
|
||||||
|
cp -f "$report" "$report.summary.json" "$drive_dir/reports/"
|
||||||
|
cp -f "$report" "$drive_dir/reports/optimization-latest.json"
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
printf '%s\n' "$out"
|
||||||
92
4-Infrastructure/kernel/qfox_topology_manager/qfox_topology_debug.sh
Executable file
92
4-Infrastructure/kernel/qfox_topology_manager/qfox_topology_debug.sh
Executable file
|
|
@ -0,0 +1,92 @@
|
||||||
|
#!/usr/bin/env bash
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
module="qfox_topology_manager"
|
||||||
|
tracefs="/sys/kernel/tracing"
|
||||||
|
state="/run/${module}_ftrace_state"
|
||||||
|
|
||||||
|
if [[ ! -d "$tracefs" ]]; then
|
||||||
|
tracefs="/sys/kernel/debug/tracing"
|
||||||
|
fi
|
||||||
|
|
||||||
|
if [[ ! -d "$tracefs" ]]; then
|
||||||
|
echo "tracefs is not mounted" >&2
|
||||||
|
exit 2
|
||||||
|
fi
|
||||||
|
|
||||||
|
cmd="${1:-snapshot}"
|
||||||
|
|
||||||
|
require_loaded() {
|
||||||
|
if ! grep -q "^${module} " /proc/modules; then
|
||||||
|
echo "${module} is not loaded" >&2
|
||||||
|
exit 2
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
attach() {
|
||||||
|
require_loaded
|
||||||
|
mkdir -p "$(dirname "$state")"
|
||||||
|
{
|
||||||
|
printf 'tracer=%s\n' "$(cat "$tracefs/current_tracer")"
|
||||||
|
printf 'filter=%s\n' "$(cat "$tracefs/set_ftrace_filter" 2>/dev/null || true)"
|
||||||
|
} > "$state"
|
||||||
|
|
||||||
|
echo 0 > "$tracefs/tracing_on"
|
||||||
|
: > "$tracefs/trace"
|
||||||
|
echo nop > "$tracefs/current_tracer"
|
||||||
|
: > "$tracefs/set_ftrace_filter"
|
||||||
|
{
|
||||||
|
echo qfox_record
|
||||||
|
echo qfox_record_from_buffer
|
||||||
|
echo qfox_dev_read
|
||||||
|
echo qfox_dev_write
|
||||||
|
echo qfox_netdev_event
|
||||||
|
echo qfox_reboot_event
|
||||||
|
} > "$tracefs/set_ftrace_filter"
|
||||||
|
echo function_graph > "$tracefs/current_tracer"
|
||||||
|
echo 1 > "$tracefs/tracing_on"
|
||||||
|
|
||||||
|
if [[ -w /sys/kernel/debug/dynamic_debug/control ]]; then
|
||||||
|
# Harmless when there are no pr_debug callsites; useful if added later.
|
||||||
|
echo "module ${module} +p" > /sys/kernel/debug/dynamic_debug/control || true
|
||||||
|
fi
|
||||||
|
|
||||||
|
echo "attached ${module} ftrace debugger"
|
||||||
|
echo "trace: $tracefs/trace"
|
||||||
|
}
|
||||||
|
|
||||||
|
detach() {
|
||||||
|
echo 0 > "$tracefs/tracing_on"
|
||||||
|
echo nop > "$tracefs/current_tracer"
|
||||||
|
: > "$tracefs/set_ftrace_filter"
|
||||||
|
if [[ -f "$state" ]]; then
|
||||||
|
# Restoring arbitrary previous filters can be surprising; keep this
|
||||||
|
# conservative and leave the prior state recorded for inspection.
|
||||||
|
echo "previous ftrace state preserved at $state"
|
||||||
|
fi
|
||||||
|
echo "detached ${module} ftrace debugger"
|
||||||
|
}
|
||||||
|
|
||||||
|
snapshot() {
|
||||||
|
require_loaded
|
||||||
|
echo "--- module ---"
|
||||||
|
modinfo "$module" || true
|
||||||
|
echo "--- sysfs/status ---"
|
||||||
|
cat "/sys/kernel/${module}/status" 2>/dev/null || cat "/sys/kernel/qfox_topology_manager/status"
|
||||||
|
echo "--- sysfs/slots ---"
|
||||||
|
cat "/sys/kernel/${module}/slots" 2>/dev/null || cat "/sys/kernel/qfox_topology_manager/slots"
|
||||||
|
echo "--- debugfs/events ---"
|
||||||
|
cat "/sys/kernel/debug/${module}/events" 2>/dev/null || cat "/sys/kernel/debug/qfox_topology_manager/events" 2>/dev/null || true
|
||||||
|
echo "--- ftrace tail ---"
|
||||||
|
tail -n 80 "$tracefs/trace" 2>/dev/null || true
|
||||||
|
}
|
||||||
|
|
||||||
|
case "$cmd" in
|
||||||
|
attach) attach ;;
|
||||||
|
detach) detach ;;
|
||||||
|
snapshot) snapshot ;;
|
||||||
|
*)
|
||||||
|
echo "usage: $0 {attach|snapshot|detach}" >&2
|
||||||
|
exit 2
|
||||||
|
;;
|
||||||
|
esac
|
||||||
|
|
@ -0,0 +1,495 @@
|
||||||
|
/*
|
||||||
|
* QFox Topology Manager — passive nanokernel carrier
|
||||||
|
* ==================================================
|
||||||
|
*
|
||||||
|
* This module is the safe Linux-carrier phase for the GCL nanokernel concept.
|
||||||
|
* It observes carrier events, maps them into topology slots, and exposes a
|
||||||
|
* receipt-shaped interface for userspace. It does not enforce policy.
|
||||||
|
*/
|
||||||
|
|
||||||
|
#define pr_fmt(fmt) KBUILD_MODNAME ": " fmt
|
||||||
|
|
||||||
|
#include <linux/atomic.h>
|
||||||
|
#include <linux/debugfs.h>
|
||||||
|
#include <linux/fs.h>
|
||||||
|
#include <linux/init.h>
|
||||||
|
#include <linux/jhash.h>
|
||||||
|
#include <linux/jiffies.h>
|
||||||
|
#include <linux/kernel.h>
|
||||||
|
#include <linux/kobject.h>
|
||||||
|
#include <linux/ktime.h>
|
||||||
|
#include <linux/miscdevice.h>
|
||||||
|
#include <linux/module.h>
|
||||||
|
#include <linux/mutex.h>
|
||||||
|
#include <linux/netdevice.h>
|
||||||
|
#include <linux/poll.h>
|
||||||
|
#include <linux/reboot.h>
|
||||||
|
#include <linux/seq_file.h>
|
||||||
|
#include <linux/slab.h>
|
||||||
|
#include <linux/spinlock.h>
|
||||||
|
#include <linux/string.h>
|
||||||
|
#include <linux/uaccess.h>
|
||||||
|
#include <linux/wait.h>
|
||||||
|
|
||||||
|
MODULE_AUTHOR("Research Stack");
|
||||||
|
MODULE_DESCRIPTION("QFox passive topology manager and GCL nanokernel carrier");
|
||||||
|
MODULE_LICENSE("GPL");
|
||||||
|
MODULE_VERSION("0.1.0");
|
||||||
|
|
||||||
|
#define QFOX_VERSION "0.1.0-observe"
|
||||||
|
#define QFOX_RING_SIZE 256
|
||||||
|
#define QFOX_PAYLOAD_LEN 80
|
||||||
|
#define QFOX_READ_LIMIT 8192
|
||||||
|
|
||||||
|
enum qfox_slot {
|
||||||
|
QFOX_SLOT_BOOT = 0,
|
||||||
|
QFOX_SLOT_SCHED,
|
||||||
|
QFOX_SLOT_MM,
|
||||||
|
QFOX_SLOT_FS,
|
||||||
|
QFOX_SLOT_BLOCK,
|
||||||
|
QFOX_SLOT_NET,
|
||||||
|
QFOX_SLOT_DEVICE,
|
||||||
|
QFOX_SLOT_POWER,
|
||||||
|
QFOX_SLOT_SECURITY,
|
||||||
|
QFOX_SLOT_GPU,
|
||||||
|
QFOX_SLOT_USER,
|
||||||
|
QFOX_SLOT_RECEIPT,
|
||||||
|
QFOX_SLOT_MAX,
|
||||||
|
};
|
||||||
|
|
||||||
|
struct qfox_event {
|
||||||
|
u64 seq;
|
||||||
|
u64 boottime_ns;
|
||||||
|
u32 slot;
|
||||||
|
u32 code;
|
||||||
|
u32 payload_hash;
|
||||||
|
char payload[QFOX_PAYLOAD_LEN];
|
||||||
|
};
|
||||||
|
|
||||||
|
struct qfox_state {
|
||||||
|
spinlock_t lock;
|
||||||
|
wait_queue_head_t waitq;
|
||||||
|
atomic64_t seq;
|
||||||
|
atomic_t enabled;
|
||||||
|
u64 loaded_boottime_ns;
|
||||||
|
u64 counters[QFOX_SLOT_MAX];
|
||||||
|
struct qfox_event ring[QFOX_RING_SIZE];
|
||||||
|
u32 head;
|
||||||
|
struct kobject *kobj;
|
||||||
|
struct dentry *debugfs_dir;
|
||||||
|
struct miscdevice miscdev;
|
||||||
|
};
|
||||||
|
|
||||||
|
static struct qfox_state qfox;
|
||||||
|
|
||||||
|
static const char *const qfox_slot_names[QFOX_SLOT_MAX] = {
|
||||||
|
[QFOX_SLOT_BOOT] = "boot",
|
||||||
|
[QFOX_SLOT_SCHED] = "sched",
|
||||||
|
[QFOX_SLOT_MM] = "mm",
|
||||||
|
[QFOX_SLOT_FS] = "fs",
|
||||||
|
[QFOX_SLOT_BLOCK] = "block",
|
||||||
|
[QFOX_SLOT_NET] = "net",
|
||||||
|
[QFOX_SLOT_DEVICE] = "device",
|
||||||
|
[QFOX_SLOT_POWER] = "power",
|
||||||
|
[QFOX_SLOT_SECURITY] = "security",
|
||||||
|
[QFOX_SLOT_GPU] = "gpu",
|
||||||
|
[QFOX_SLOT_USER] = "user",
|
||||||
|
[QFOX_SLOT_RECEIPT] = "receipt",
|
||||||
|
};
|
||||||
|
|
||||||
|
static const char *qfox_slot_name(enum qfox_slot slot)
|
||||||
|
{
|
||||||
|
if (slot >= QFOX_SLOT_MAX)
|
||||||
|
return "unknown";
|
||||||
|
return qfox_slot_names[slot];
|
||||||
|
}
|
||||||
|
|
||||||
|
static enum qfox_slot qfox_parse_slot(const char *token)
|
||||||
|
{
|
||||||
|
int i;
|
||||||
|
|
||||||
|
for (i = 0; i < QFOX_SLOT_MAX; i++) {
|
||||||
|
if (sysfs_streq(token, qfox_slot_names[i]))
|
||||||
|
return i;
|
||||||
|
}
|
||||||
|
|
||||||
|
return QFOX_SLOT_USER;
|
||||||
|
}
|
||||||
|
|
||||||
|
static void qfox_record(enum qfox_slot slot, u32 code, const char *payload)
|
||||||
|
{
|
||||||
|
struct qfox_event *event;
|
||||||
|
unsigned long flags;
|
||||||
|
u64 seq;
|
||||||
|
u32 idx;
|
||||||
|
size_t len = 0;
|
||||||
|
|
||||||
|
if (!atomic_read(&qfox.enabled) && slot != QFOX_SLOT_BOOT)
|
||||||
|
return;
|
||||||
|
|
||||||
|
if (slot >= QFOX_SLOT_MAX)
|
||||||
|
slot = QFOX_SLOT_USER;
|
||||||
|
|
||||||
|
seq = (u64)atomic64_inc_return(&qfox.seq);
|
||||||
|
|
||||||
|
spin_lock_irqsave(&qfox.lock, flags);
|
||||||
|
idx = qfox.head++ % QFOX_RING_SIZE;
|
||||||
|
event = &qfox.ring[idx];
|
||||||
|
memset(event, 0, sizeof(*event));
|
||||||
|
event->seq = seq;
|
||||||
|
event->boottime_ns = ktime_get_boottime_ns();
|
||||||
|
event->slot = slot;
|
||||||
|
event->code = code;
|
||||||
|
if (payload) {
|
||||||
|
strscpy(event->payload, payload, sizeof(event->payload));
|
||||||
|
len = strnlen(event->payload, sizeof(event->payload));
|
||||||
|
event->payload_hash = jhash(event->payload, len, 0x51464f58U);
|
||||||
|
}
|
||||||
|
qfox.counters[slot]++;
|
||||||
|
spin_unlock_irqrestore(&qfox.lock, flags);
|
||||||
|
|
||||||
|
wake_up_interruptible(&qfox.waitq);
|
||||||
|
}
|
||||||
|
|
||||||
|
static void qfox_record_from_buffer(const char *buf, size_t len)
|
||||||
|
{
|
||||||
|
char local[QFOX_PAYLOAD_LEN];
|
||||||
|
char token[24];
|
||||||
|
char *cursor;
|
||||||
|
char *space;
|
||||||
|
enum qfox_slot slot;
|
||||||
|
|
||||||
|
len = min_t(size_t, len, sizeof(local) - 1);
|
||||||
|
memcpy(local, buf, len);
|
||||||
|
local[len] = '\0';
|
||||||
|
cursor = strim(local);
|
||||||
|
|
||||||
|
if (!cursor || cursor[0] == '\0')
|
||||||
|
return;
|
||||||
|
|
||||||
|
strscpy(token, cursor, sizeof(token));
|
||||||
|
space = strpbrk(token, " \t:\n");
|
||||||
|
if (space)
|
||||||
|
*space = '\0';
|
||||||
|
|
||||||
|
slot = qfox_parse_slot(token);
|
||||||
|
qfox_record(slot, 0, cursor);
|
||||||
|
}
|
||||||
|
|
||||||
|
static ssize_t status_show(struct kobject *kobj,
|
||||||
|
struct kobj_attribute *attr, char *buf)
|
||||||
|
{
|
||||||
|
u64 now_ns = ktime_get_boottime_ns();
|
||||||
|
u64 age_ns = now_ns - qfox.loaded_boottime_ns;
|
||||||
|
u64 events = (u64)atomic64_read(&qfox.seq);
|
||||||
|
u64 avg_x1000 = 0;
|
||||||
|
|
||||||
|
if (age_ns > 0)
|
||||||
|
avg_x1000 = div64_u64(events * 1000000000000ULL, age_ns);
|
||||||
|
|
||||||
|
return sysfs_emit(buf,
|
||||||
|
"version=%s\nenabled=%d\nevents=%llu\nring_size=%d\nmode=observe\nloaded_boottime_ns=%llu\nage_ns=%llu\navg_events_per_sec_x1000=%llu\n",
|
||||||
|
QFOX_VERSION,
|
||||||
|
atomic_read(&qfox.enabled),
|
||||||
|
(unsigned long long)events,
|
||||||
|
QFOX_RING_SIZE,
|
||||||
|
(unsigned long long)qfox.loaded_boottime_ns,
|
||||||
|
(unsigned long long)age_ns,
|
||||||
|
(unsigned long long)avg_x1000);
|
||||||
|
}
|
||||||
|
|
||||||
|
static ssize_t mode_show(struct kobject *kobj,
|
||||||
|
struct kobj_attribute *attr, char *buf)
|
||||||
|
{
|
||||||
|
return sysfs_emit(buf, "%s\n", atomic_read(&qfox.enabled) ? "observe" : "off");
|
||||||
|
}
|
||||||
|
|
||||||
|
static ssize_t mode_store(struct kobject *kobj,
|
||||||
|
struct kobj_attribute *attr,
|
||||||
|
const char *buf, size_t count)
|
||||||
|
{
|
||||||
|
if (sysfs_streq(buf, "observe") || sysfs_streq(buf, "on")) {
|
||||||
|
atomic_set(&qfox.enabled, 1);
|
||||||
|
qfox_record(QFOX_SLOT_BOOT, 1, "mode=observe");
|
||||||
|
return count;
|
||||||
|
}
|
||||||
|
if (sysfs_streq(buf, "off")) {
|
||||||
|
qfox_record(QFOX_SLOT_BOOT, 0, "mode=off");
|
||||||
|
atomic_set(&qfox.enabled, 0);
|
||||||
|
return count;
|
||||||
|
}
|
||||||
|
return -EINVAL;
|
||||||
|
}
|
||||||
|
|
||||||
|
static ssize_t slots_show(struct kobject *kobj,
|
||||||
|
struct kobj_attribute *attr, char *buf)
|
||||||
|
{
|
||||||
|
unsigned long flags;
|
||||||
|
ssize_t off = 0;
|
||||||
|
int i;
|
||||||
|
|
||||||
|
spin_lock_irqsave(&qfox.lock, flags);
|
||||||
|
for (i = 0; i < QFOX_SLOT_MAX; i++) {
|
||||||
|
off += scnprintf(buf + off, PAGE_SIZE - off, "%s=%llu\n",
|
||||||
|
qfox_slot_name(i),
|
||||||
|
(unsigned long long)qfox.counters[i]);
|
||||||
|
if (off >= PAGE_SIZE)
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
spin_unlock_irqrestore(&qfox.lock, flags);
|
||||||
|
return off;
|
||||||
|
}
|
||||||
|
|
||||||
|
static ssize_t inject_store(struct kobject *kobj,
|
||||||
|
struct kobj_attribute *attr,
|
||||||
|
const char *buf, size_t count)
|
||||||
|
{
|
||||||
|
qfox_record_from_buffer(buf, count);
|
||||||
|
return count;
|
||||||
|
}
|
||||||
|
|
||||||
|
static struct kobj_attribute status_attr = __ATTR_RO(status);
|
||||||
|
static struct kobj_attribute mode_attr = __ATTR_RW(mode);
|
||||||
|
static struct kobj_attribute slots_attr = __ATTR_RO(slots);
|
||||||
|
static struct kobj_attribute inject_attr = __ATTR_WO(inject);
|
||||||
|
|
||||||
|
static struct attribute *qfox_attrs[] = {
|
||||||
|
&status_attr.attr,
|
||||||
|
&mode_attr.attr,
|
||||||
|
&slots_attr.attr,
|
||||||
|
&inject_attr.attr,
|
||||||
|
NULL,
|
||||||
|
};
|
||||||
|
|
||||||
|
static const struct attribute_group qfox_attr_group = {
|
||||||
|
.attrs = qfox_attrs,
|
||||||
|
};
|
||||||
|
|
||||||
|
static int qfox_events_show(struct seq_file *m, void *v)
|
||||||
|
{
|
||||||
|
struct qfox_event *snapshot;
|
||||||
|
unsigned long flags;
|
||||||
|
u32 head;
|
||||||
|
u32 count;
|
||||||
|
u32 start;
|
||||||
|
u32 i;
|
||||||
|
|
||||||
|
snapshot = kcalloc(QFOX_RING_SIZE, sizeof(*snapshot), GFP_KERNEL);
|
||||||
|
if (!snapshot)
|
||||||
|
return -ENOMEM;
|
||||||
|
|
||||||
|
spin_lock_irqsave(&qfox.lock, flags);
|
||||||
|
head = qfox.head;
|
||||||
|
count = min_t(u32, head, QFOX_RING_SIZE);
|
||||||
|
start = head >= count ? head - count : 0;
|
||||||
|
for (i = 0; i < count; i++)
|
||||||
|
snapshot[i] = qfox.ring[(start + i) % QFOX_RING_SIZE];
|
||||||
|
spin_unlock_irqrestore(&qfox.lock, flags);
|
||||||
|
|
||||||
|
seq_printf(m, "schema=research_stack_qfox_topology_manager_events_v1\n");
|
||||||
|
seq_printf(m, "version=%s\n", QFOX_VERSION);
|
||||||
|
for (i = 0; i < count; i++) {
|
||||||
|
struct qfox_event *event = &snapshot[i];
|
||||||
|
|
||||||
|
if (event->seq == 0)
|
||||||
|
continue;
|
||||||
|
seq_printf(m,
|
||||||
|
"event seq=%llu ns=%llu slot=%s code=%u hash=%08x payload=\"%s\"\n",
|
||||||
|
(unsigned long long)event->seq,
|
||||||
|
(unsigned long long)event->boottime_ns,
|
||||||
|
qfox_slot_name(event->slot),
|
||||||
|
event->code,
|
||||||
|
event->payload_hash,
|
||||||
|
event->payload);
|
||||||
|
}
|
||||||
|
kfree(snapshot);
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
static int qfox_events_open(struct inode *inode, struct file *file)
|
||||||
|
{
|
||||||
|
return single_open(file, qfox_events_show, inode->i_private);
|
||||||
|
}
|
||||||
|
|
||||||
|
static const struct file_operations qfox_events_fops = {
|
||||||
|
.owner = THIS_MODULE,
|
||||||
|
.open = qfox_events_open,
|
||||||
|
.read = seq_read,
|
||||||
|
.llseek = seq_lseek,
|
||||||
|
.release = single_release,
|
||||||
|
};
|
||||||
|
|
||||||
|
static ssize_t qfox_dev_read(struct file *file, char __user *ubuf,
|
||||||
|
size_t count, loff_t *ppos)
|
||||||
|
{
|
||||||
|
char *buf;
|
||||||
|
ssize_t len = 0;
|
||||||
|
ssize_t ret;
|
||||||
|
unsigned long flags;
|
||||||
|
int i;
|
||||||
|
|
||||||
|
buf = kzalloc(QFOX_READ_LIMIT, GFP_KERNEL);
|
||||||
|
if (!buf)
|
||||||
|
return -ENOMEM;
|
||||||
|
|
||||||
|
len += scnprintf(buf + len, QFOX_READ_LIMIT - len,
|
||||||
|
"{\"schema\":\"research_stack_qfox_topology_manager_v1\",");
|
||||||
|
len += scnprintf(buf + len, QFOX_READ_LIMIT - len,
|
||||||
|
"\"version\":\"%s\",\"events\":%lld,\"slots\":{",
|
||||||
|
QFOX_VERSION,
|
||||||
|
(long long)atomic64_read(&qfox.seq));
|
||||||
|
|
||||||
|
spin_lock_irqsave(&qfox.lock, flags);
|
||||||
|
for (i = 0; i < QFOX_SLOT_MAX; i++) {
|
||||||
|
len += scnprintf(buf + len, QFOX_READ_LIMIT - len,
|
||||||
|
"%s\"%s\":%llu",
|
||||||
|
i ? "," : "",
|
||||||
|
qfox_slot_name(i),
|
||||||
|
(unsigned long long)qfox.counters[i]);
|
||||||
|
if (len >= QFOX_READ_LIMIT - 128)
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
spin_unlock_irqrestore(&qfox.lock, flags);
|
||||||
|
len += scnprintf(buf + len, QFOX_READ_LIMIT - len, "}}\n");
|
||||||
|
|
||||||
|
ret = simple_read_from_buffer(ubuf, count, ppos, buf, len);
|
||||||
|
kfree(buf);
|
||||||
|
return ret;
|
||||||
|
}
|
||||||
|
|
||||||
|
static ssize_t qfox_dev_write(struct file *file, const char __user *ubuf,
|
||||||
|
size_t count, loff_t *ppos)
|
||||||
|
{
|
||||||
|
char buf[QFOX_PAYLOAD_LEN];
|
||||||
|
char *cursor;
|
||||||
|
char *line;
|
||||||
|
size_t len = min_t(size_t, count, sizeof(buf) - 1);
|
||||||
|
|
||||||
|
if (copy_from_user(buf, ubuf, len))
|
||||||
|
return -EFAULT;
|
||||||
|
buf[len] = '\0';
|
||||||
|
cursor = buf;
|
||||||
|
while ((line = strsep(&cursor, "\n")) != NULL) {
|
||||||
|
line = strim(line);
|
||||||
|
if (line[0] != '\0')
|
||||||
|
qfox_record_from_buffer(line, strlen(line));
|
||||||
|
}
|
||||||
|
return count;
|
||||||
|
}
|
||||||
|
|
||||||
|
static __poll_t qfox_dev_poll(struct file *file, poll_table *wait)
|
||||||
|
{
|
||||||
|
poll_wait(file, &qfox.waitq, wait);
|
||||||
|
return EPOLLIN | EPOLLRDNORM | EPOLLOUT | EPOLLWRNORM;
|
||||||
|
}
|
||||||
|
|
||||||
|
static const struct file_operations qfox_dev_fops = {
|
||||||
|
.owner = THIS_MODULE,
|
||||||
|
.read = qfox_dev_read,
|
||||||
|
.write = qfox_dev_write,
|
||||||
|
.poll = qfox_dev_poll,
|
||||||
|
.llseek = noop_llseek,
|
||||||
|
};
|
||||||
|
|
||||||
|
static int qfox_netdev_event(struct notifier_block *nb,
|
||||||
|
unsigned long event, void *ptr)
|
||||||
|
{
|
||||||
|
struct net_device *dev = netdev_notifier_info_to_dev(ptr);
|
||||||
|
char payload[QFOX_PAYLOAD_LEN];
|
||||||
|
|
||||||
|
scnprintf(payload, sizeof(payload), "netdev=%s event=%lu",
|
||||||
|
dev ? dev->name : "unknown", event);
|
||||||
|
qfox_record(QFOX_SLOT_NET, (u32)event, payload);
|
||||||
|
return NOTIFY_DONE;
|
||||||
|
}
|
||||||
|
|
||||||
|
static int qfox_reboot_event(struct notifier_block *nb,
|
||||||
|
unsigned long event, void *ptr)
|
||||||
|
{
|
||||||
|
qfox_record(QFOX_SLOT_POWER, (u32)event, "reboot_notifier");
|
||||||
|
return NOTIFY_DONE;
|
||||||
|
}
|
||||||
|
|
||||||
|
static struct notifier_block qfox_netdev_nb = {
|
||||||
|
.notifier_call = qfox_netdev_event,
|
||||||
|
};
|
||||||
|
|
||||||
|
static struct notifier_block qfox_reboot_nb = {
|
||||||
|
.notifier_call = qfox_reboot_event,
|
||||||
|
};
|
||||||
|
|
||||||
|
static int __init qfox_init(void)
|
||||||
|
{
|
||||||
|
int ret;
|
||||||
|
|
||||||
|
memset(&qfox, 0, sizeof(qfox));
|
||||||
|
spin_lock_init(&qfox.lock);
|
||||||
|
init_waitqueue_head(&qfox.waitq);
|
||||||
|
atomic64_set(&qfox.seq, 0);
|
||||||
|
atomic_set(&qfox.enabled, 1);
|
||||||
|
qfox.loaded_boottime_ns = ktime_get_boottime_ns();
|
||||||
|
|
||||||
|
qfox.miscdev.minor = MISC_DYNAMIC_MINOR;
|
||||||
|
qfox.miscdev.name = "qfox_topoman";
|
||||||
|
qfox.miscdev.fops = &qfox_dev_fops;
|
||||||
|
qfox.miscdev.mode = 0600;
|
||||||
|
|
||||||
|
ret = misc_register(&qfox.miscdev);
|
||||||
|
if (ret)
|
||||||
|
return ret;
|
||||||
|
|
||||||
|
qfox.kobj = kobject_create_and_add("qfox_topology_manager", kernel_kobj);
|
||||||
|
if (!qfox.kobj) {
|
||||||
|
ret = -ENOMEM;
|
||||||
|
goto err_misc;
|
||||||
|
}
|
||||||
|
|
||||||
|
ret = sysfs_create_group(qfox.kobj, &qfox_attr_group);
|
||||||
|
if (ret)
|
||||||
|
goto err_kobj;
|
||||||
|
|
||||||
|
qfox.debugfs_dir = debugfs_create_dir("qfox_topology_manager", NULL);
|
||||||
|
if (!IS_ERR_OR_NULL(qfox.debugfs_dir))
|
||||||
|
debugfs_create_file("events", 0400, qfox.debugfs_dir, NULL,
|
||||||
|
&qfox_events_fops);
|
||||||
|
|
||||||
|
ret = register_netdevice_notifier(&qfox_netdev_nb);
|
||||||
|
if (ret)
|
||||||
|
goto err_debugfs;
|
||||||
|
|
||||||
|
ret = register_reboot_notifier(&qfox_reboot_nb);
|
||||||
|
if (ret)
|
||||||
|
goto err_netdev;
|
||||||
|
|
||||||
|
qfox_record(QFOX_SLOT_BOOT, 0, "module_loaded");
|
||||||
|
pr_info("loaded %s in passive observe mode\n", QFOX_VERSION);
|
||||||
|
return 0;
|
||||||
|
|
||||||
|
err_netdev:
|
||||||
|
unregister_netdevice_notifier(&qfox_netdev_nb);
|
||||||
|
err_debugfs:
|
||||||
|
debugfs_remove_recursive(qfox.debugfs_dir);
|
||||||
|
sysfs_remove_group(qfox.kobj, &qfox_attr_group);
|
||||||
|
err_kobj:
|
||||||
|
kobject_put(qfox.kobj);
|
||||||
|
err_misc:
|
||||||
|
misc_deregister(&qfox.miscdev);
|
||||||
|
return ret;
|
||||||
|
}
|
||||||
|
|
||||||
|
static void __exit qfox_exit(void)
|
||||||
|
{
|
||||||
|
qfox_record(QFOX_SLOT_BOOT, 0, "module_unloading");
|
||||||
|
unregister_reboot_notifier(&qfox_reboot_nb);
|
||||||
|
unregister_netdevice_notifier(&qfox_netdev_nb);
|
||||||
|
debugfs_remove_recursive(qfox.debugfs_dir);
|
||||||
|
if (qfox.kobj) {
|
||||||
|
sysfs_remove_group(qfox.kobj, &qfox_attr_group);
|
||||||
|
kobject_put(qfox.kobj);
|
||||||
|
}
|
||||||
|
misc_deregister(&qfox.miscdev);
|
||||||
|
pr_info("unloaded\n");
|
||||||
|
}
|
||||||
|
|
||||||
|
module_init(qfox_init);
|
||||||
|
module_exit(qfox_exit);
|
||||||
356
4-Infrastructure/kernel/qfox_topology_manager/qfox_topology_optimizer_report.py
Executable file
356
4-Infrastructure/kernel/qfox_topology_manager/qfox_topology_optimizer_report.py
Executable file
|
|
@ -0,0 +1,356 @@
|
||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Derive optimization targets from QFox topology-manager samples.
|
||||||
|
|
||||||
|
This is intentionally advisory. It ranks topology slots by observed tracepoint
|
||||||
|
rate and emits receipt-shaped recommendations; it does not tune the machine.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import json
|
||||||
|
import statistics
|
||||||
|
import sys
|
||||||
|
import time
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
SCHEMA = "research_stack_qfox_topology_optimizer_report_v1"
|
||||||
|
SAMPLE_SCHEMA = "research_stack_qfox_topology_trace_sample_v1"
|
||||||
|
DEFAULT_SAMPLE_DIR = Path("/var/lib/qfox-topology-manager/samples")
|
||||||
|
DEFAULT_REPORT_DIR = Path("/var/lib/qfox-topology-manager/reports")
|
||||||
|
DEV = Path("/dev/qfox_topoman")
|
||||||
|
|
||||||
|
SLOT_GUIDANCE: dict[str, dict[str, Any]] = {
|
||||||
|
"sched": {
|
||||||
|
"label": "scheduler and wakeup churn",
|
||||||
|
"optimize_for": [
|
||||||
|
"reduce unnecessary wakeups",
|
||||||
|
"identify noisy user services or timers",
|
||||||
|
"keep latency tuning only where workloads prove it",
|
||||||
|
],
|
||||||
|
"next_probes": [
|
||||||
|
"sudo perf sched record -- sleep 10 && sudo perf sched latency",
|
||||||
|
"systemd-analyze blame",
|
||||||
|
"sudo cat /proc/schedstat | head -40",
|
||||||
|
],
|
||||||
|
},
|
||||||
|
"mm": {
|
||||||
|
"label": "memory allocation churn",
|
||||||
|
"optimize_for": [
|
||||||
|
"reduce allocation/free storms",
|
||||||
|
"watch zram and cache pressure",
|
||||||
|
"separate build/indexer memory churn from desktop baseline",
|
||||||
|
],
|
||||||
|
"next_probes": [
|
||||||
|
"free -h",
|
||||||
|
"cat /proc/pressure/memory",
|
||||||
|
"grep -E 'pgscan|pgsteal|allocstall' /proc/vmstat",
|
||||||
|
],
|
||||||
|
},
|
||||||
|
"power": {
|
||||||
|
"label": "idle and frequency transition churn",
|
||||||
|
"optimize_for": [
|
||||||
|
"lower timer noise before chasing power states",
|
||||||
|
"balance desktop responsiveness against idle churn",
|
||||||
|
"inspect services that prevent deep idle",
|
||||||
|
],
|
||||||
|
"next_probes": [
|
||||||
|
"cat /proc/pressure/cpu",
|
||||||
|
"cat /sys/devices/system/cpu/cpu0/cpufreq/energy_performance_preference 2>/dev/null",
|
||||||
|
"sudo turbostat --Summary --quiet --interval 5 --num_iterations 2",
|
||||||
|
],
|
||||||
|
},
|
||||||
|
"device": {
|
||||||
|
"label": "IRQ, softirq, and device interrupt pressure",
|
||||||
|
"optimize_for": [
|
||||||
|
"find high-rate IRQ sources",
|
||||||
|
"check NIC/GPU/USB interrupt behavior",
|
||||||
|
"only pin or rebalance IRQs after source attribution",
|
||||||
|
],
|
||||||
|
"next_probes": [
|
||||||
|
"cat /proc/interrupts | sort -k2,2nr | head -30",
|
||||||
|
"cat /proc/softirqs",
|
||||||
|
"systemctl is-active irqbalance || true",
|
||||||
|
],
|
||||||
|
},
|
||||||
|
"block": {
|
||||||
|
"label": "block IO pressure",
|
||||||
|
"optimize_for": [
|
||||||
|
"separate local NVMe IO from remote/FUSE churn",
|
||||||
|
"inspect Btrfs writeback and build-cache traffic",
|
||||||
|
"keep scheduler changes scoped to measured IO latency",
|
||||||
|
],
|
||||||
|
"next_probes": [
|
||||||
|
"iostat -xz 1 5",
|
||||||
|
"cat /proc/pressure/io",
|
||||||
|
"findmnt -T /home/allaun/Documents/Research\\ Stack",
|
||||||
|
],
|
||||||
|
},
|
||||||
|
"net": {
|
||||||
|
"label": "network packet and queue churn",
|
||||||
|
"optimize_for": [
|
||||||
|
"identify chatty local tunnels or sync clients",
|
||||||
|
"separate loopback/tailscale/rclone traffic",
|
||||||
|
"avoid NIC tuning until packet source is known",
|
||||||
|
],
|
||||||
|
"next_probes": [
|
||||||
|
"ip -s link",
|
||||||
|
"ss -tunap | head -80",
|
||||||
|
"nstat -az | head -80",
|
||||||
|
],
|
||||||
|
},
|
||||||
|
"fs": {
|
||||||
|
"label": "filesystem syscall churn",
|
||||||
|
"optimize_for": [
|
||||||
|
"reduce watcher and indexer scans",
|
||||||
|
"separate IDE activity from build/test activity",
|
||||||
|
"prefer path-specific receipts over broad recursive scans",
|
||||||
|
],
|
||||||
|
"next_probes": [
|
||||||
|
"sudo fatrace -c -t 10",
|
||||||
|
"inotifywatch -r -t 10 /home/allaun/Documents/Research\\ Stack",
|
||||||
|
],
|
||||||
|
},
|
||||||
|
"gpu": {
|
||||||
|
"label": "GPU/display carrier activity",
|
||||||
|
"optimize_for": [
|
||||||
|
"watch compositor and driver event churn",
|
||||||
|
"separate KDE/display stalls from compute load",
|
||||||
|
],
|
||||||
|
"next_probes": [
|
||||||
|
"nvidia-smi dmon -s pucvmt -c 5",
|
||||||
|
"journalctl -b -p warning | grep -Ei 'nvidia|kwin|drm' | tail -80",
|
||||||
|
],
|
||||||
|
},
|
||||||
|
"security": {
|
||||||
|
"label": "admission and policy-surface activity",
|
||||||
|
"optimize_for": [
|
||||||
|
"treat denials as evidence, not noise",
|
||||||
|
"avoid broad policy changes without an event receipt",
|
||||||
|
],
|
||||||
|
"next_probes": [
|
||||||
|
"journalctl -b -p warning | grep -Ei 'audit|apparmor|permission|denied' | tail -80",
|
||||||
|
],
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def sample_paths(sample_dir: Path) -> list[Path]:
|
||||||
|
summaries = sorted(sample_dir.glob("trace-sample-*.summary.json"))
|
||||||
|
if summaries:
|
||||||
|
return summaries
|
||||||
|
return sorted(
|
||||||
|
path
|
||||||
|
for path in sample_dir.glob("trace-sample-*.json")
|
||||||
|
if not path.name.endswith(".summary.json")
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def load_sample(path: Path) -> dict[str, Any] | None:
|
||||||
|
try:
|
||||||
|
payload = json.loads(path.read_text(encoding="utf-8"))
|
||||||
|
except (OSError, json.JSONDecodeError):
|
||||||
|
return None
|
||||||
|
if payload.get("schema") != SAMPLE_SCHEMA:
|
||||||
|
return None
|
||||||
|
payload["_path"] = str(path)
|
||||||
|
return payload
|
||||||
|
|
||||||
|
|
||||||
|
def aggregate(samples: list[dict[str, Any]]) -> dict[str, Any]:
|
||||||
|
total_duration = 0.0
|
||||||
|
slot_counts: dict[str, int] = {}
|
||||||
|
slot_rates_by_sample: dict[str, list[float]] = {}
|
||||||
|
event_counts: dict[str, int] = {}
|
||||||
|
|
||||||
|
for sample in samples:
|
||||||
|
duration = float(sample.get("duration_sec") or 0.0)
|
||||||
|
total_duration += max(0.0, duration)
|
||||||
|
for slot, count in (sample.get("slot_counts") or {}).items():
|
||||||
|
slot_counts[slot] = slot_counts.get(slot, 0) + int(count)
|
||||||
|
for slot, rate in (sample.get("slot_rates_per_sec") or {}).items():
|
||||||
|
slot_rates_by_sample.setdefault(slot, []).append(float(rate))
|
||||||
|
for event, count in (sample.get("event_counts") or {}).items():
|
||||||
|
event_counts[event] = event_counts.get(event, 0) + int(count)
|
||||||
|
|
||||||
|
slot_rates = {
|
||||||
|
slot: (count / total_duration if total_duration > 0 else 0.0)
|
||||||
|
for slot, count in slot_counts.items()
|
||||||
|
}
|
||||||
|
total_events = sum(slot_counts.values())
|
||||||
|
slot_shares = {
|
||||||
|
slot: (count / total_events if total_events else 0.0)
|
||||||
|
for slot, count in slot_counts.items()
|
||||||
|
}
|
||||||
|
slot_rate_stats = {
|
||||||
|
slot: {
|
||||||
|
"samples": len(rates),
|
||||||
|
"mean": statistics.fmean(rates),
|
||||||
|
"max": max(rates),
|
||||||
|
"min": min(rates),
|
||||||
|
}
|
||||||
|
for slot, rates in slot_rates_by_sample.items()
|
||||||
|
if rates
|
||||||
|
}
|
||||||
|
return {
|
||||||
|
"sample_count": len(samples),
|
||||||
|
"sample_paths": [sample["_path"] for sample in samples],
|
||||||
|
"duration_sec": total_duration,
|
||||||
|
"slot_counts": dict(sorted(slot_counts.items())),
|
||||||
|
"slot_rates_per_sec": dict(
|
||||||
|
sorted(slot_rates.items(), key=lambda item: item[1], reverse=True)
|
||||||
|
),
|
||||||
|
"slot_shares": dict(
|
||||||
|
sorted(slot_shares.items(), key=lambda item: item[1], reverse=True)
|
||||||
|
),
|
||||||
|
"slot_rate_stats": slot_rate_stats,
|
||||||
|
"event_counts": dict(
|
||||||
|
sorted(event_counts.items(), key=lambda item: item[1], reverse=True)
|
||||||
|
),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def severity(rate: float, share: float) -> str:
|
||||||
|
if share >= 0.35 or rate >= 50000:
|
||||||
|
return "primary"
|
||||||
|
if share >= 0.10 or rate >= 5000:
|
||||||
|
return "secondary"
|
||||||
|
if rate > 0:
|
||||||
|
return "watch"
|
||||||
|
return "quiet"
|
||||||
|
|
||||||
|
|
||||||
|
def build_recommendations(agg: dict[str, Any], top_n: int) -> list[dict[str, Any]]:
|
||||||
|
recommendations: list[dict[str, Any]] = []
|
||||||
|
rates = agg["slot_rates_per_sec"]
|
||||||
|
shares = agg["slot_shares"]
|
||||||
|
for rank, (slot, rate) in enumerate(rates.items(), start=1):
|
||||||
|
share = shares.get(slot, 0.0)
|
||||||
|
guide = SLOT_GUIDANCE.get(slot, {})
|
||||||
|
recommendations.append(
|
||||||
|
{
|
||||||
|
"rank": rank,
|
||||||
|
"slot": slot,
|
||||||
|
"label": guide.get("label", f"{slot} activity"),
|
||||||
|
"severity": severity(rate, share),
|
||||||
|
"rate_per_sec": rate,
|
||||||
|
"share": share,
|
||||||
|
"optimize_for": guide.get(
|
||||||
|
"optimize_for",
|
||||||
|
["collect more receipts before tuning this slot"],
|
||||||
|
),
|
||||||
|
"next_probes": guide.get("next_probes", []),
|
||||||
|
}
|
||||||
|
)
|
||||||
|
return recommendations[:top_n]
|
||||||
|
|
||||||
|
|
||||||
|
def build_report(samples: list[dict[str, Any]], top_n: int) -> dict[str, Any]:
|
||||||
|
agg = aggregate(samples)
|
||||||
|
recs = build_recommendations(agg, top_n)
|
||||||
|
primary = [rec for rec in recs if rec["severity"] == "primary"]
|
||||||
|
secondary = [rec for rec in recs if rec["severity"] == "secondary"]
|
||||||
|
return {
|
||||||
|
"schema": SCHEMA,
|
||||||
|
"timestamp_unix": int(time.time()),
|
||||||
|
"aggregate": agg,
|
||||||
|
"primary_targets": primary,
|
||||||
|
"secondary_targets": secondary,
|
||||||
|
"ranked_targets": recs,
|
||||||
|
"interpretation": interpret(recs),
|
||||||
|
"guardrails": [
|
||||||
|
"recommendations are receipt-derived and advisory",
|
||||||
|
"do not apply tuning automatically from this report",
|
||||||
|
"optimize after source attribution, not only slot rate",
|
||||||
|
"keep the Linux module passive; GCL/Lean remains the policy layer",
|
||||||
|
],
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def interpret(recs: list[dict[str, Any]]) -> str:
|
||||||
|
if not recs:
|
||||||
|
return "No samples were available; collect trace samples before tuning."
|
||||||
|
primary = [rec["slot"] for rec in recs if rec["severity"] == "primary"]
|
||||||
|
if primary:
|
||||||
|
return "Optimize first for " + ", ".join(primary) + "."
|
||||||
|
return "No primary hot slot yet; keep collecting baseline samples."
|
||||||
|
|
||||||
|
|
||||||
|
def emit_text(report: dict[str, Any]) -> str:
|
||||||
|
lines = [
|
||||||
|
f"schema: {report['schema']}",
|
||||||
|
f"samples: {report['aggregate']['sample_count']}",
|
||||||
|
f"duration_sec: {report['aggregate']['duration_sec']:.3f}",
|
||||||
|
f"interpretation: {report['interpretation']}",
|
||||||
|
"",
|
||||||
|
"ranked_targets:",
|
||||||
|
]
|
||||||
|
for rec in report["ranked_targets"]:
|
||||||
|
lines.append(
|
||||||
|
f" {rec['rank']}. {rec['slot']} "
|
||||||
|
f"({rec['severity']}): {rec['rate_per_sec']:.3f}/sec, "
|
||||||
|
f"share={rec['share']:.3%} - {rec['label']}"
|
||||||
|
)
|
||||||
|
for item in rec["optimize_for"][:3]:
|
||||||
|
lines.append(f" optimize_for: {item}")
|
||||||
|
lines.append("")
|
||||||
|
lines.append("next_probes:")
|
||||||
|
seen: set[str] = set()
|
||||||
|
for rec in report["ranked_targets"][:3]:
|
||||||
|
for probe in rec["next_probes"][:2]:
|
||||||
|
if probe in seen:
|
||||||
|
continue
|
||||||
|
seen.add(probe)
|
||||||
|
lines.append(f" - {probe}")
|
||||||
|
return "\n".join(lines) + "\n"
|
||||||
|
|
||||||
|
|
||||||
|
def inject_report(report: dict[str, Any], out: Path | None) -> bool:
|
||||||
|
if not DEV.exists():
|
||||||
|
return False
|
||||||
|
targets = ",".join(rec["slot"] for rec in report["ranked_targets"][:3])
|
||||||
|
payload = f"receipt optimize_report top={targets}"
|
||||||
|
if out:
|
||||||
|
payload += f" path={out}"
|
||||||
|
try:
|
||||||
|
with DEV.open("w", encoding="utf-8", errors="replace") as handle:
|
||||||
|
handle.write(payload + "\n")
|
||||||
|
except OSError as exc:
|
||||||
|
print(f"optimizer report injection skipped: {exc}", file=sys.stderr)
|
||||||
|
return False
|
||||||
|
return True
|
||||||
|
|
||||||
|
|
||||||
|
def main(argv: list[str] | None = None) -> int:
|
||||||
|
parser = argparse.ArgumentParser(description=__doc__)
|
||||||
|
parser.add_argument("--samples-dir", type=Path, default=DEFAULT_SAMPLE_DIR)
|
||||||
|
parser.add_argument("--sample", type=Path, action="append", default=[])
|
||||||
|
parser.add_argument("--top", type=int, default=6)
|
||||||
|
parser.add_argument("--json", action="store_true")
|
||||||
|
parser.add_argument("--out", type=Path)
|
||||||
|
parser.add_argument("--inject", action="store_true")
|
||||||
|
args = parser.parse_args(argv)
|
||||||
|
|
||||||
|
paths = args.sample or sample_paths(args.samples_dir)
|
||||||
|
samples = [sample for path in paths if (sample := load_sample(path)) is not None]
|
||||||
|
report = build_report(samples, max(1, args.top))
|
||||||
|
report["injected_into_module"] = False
|
||||||
|
if args.inject:
|
||||||
|
report["injected_into_module"] = inject_report(report, args.out)
|
||||||
|
|
||||||
|
rendered = json.dumps(report, indent=2, sort_keys=True) + "\n"
|
||||||
|
out_path = args.out
|
||||||
|
if out_path:
|
||||||
|
try:
|
||||||
|
out_path.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
out_path.write_text(rendered, encoding="utf-8")
|
||||||
|
except OSError as exc:
|
||||||
|
print(f"optimizer report write failed: {exc}", file=sys.stderr)
|
||||||
|
return 2
|
||||||
|
print(rendered if args.json else emit_text(report), end="")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
raise SystemExit(main())
|
||||||
229
4-Infrastructure/kernel/qfox_topology_manager/qfox_topology_trace_sampler.py
Executable file
229
4-Infrastructure/kernel/qfox_topology_manager/qfox_topology_trace_sampler.py
Executable file
|
|
@ -0,0 +1,229 @@
|
||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Sample kernel tracepoints and map them into QFox topology slots.
|
||||||
|
|
||||||
|
This is the debugger-side average collector. It temporarily enables a small
|
||||||
|
set of kernel tracepoints, counts observed events for a bounded duration, and
|
||||||
|
optionally writes slot summaries into /dev/qfox_topoman.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import re
|
||||||
|
import select
|
||||||
|
import sys
|
||||||
|
import time
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
SCHEMA = "research_stack_qfox_topology_trace_sample_v1"
|
||||||
|
DEV = Path("/dev/qfox_topoman")
|
||||||
|
|
||||||
|
EVENT_TO_SLOT: dict[str, str] = {
|
||||||
|
"sched:sched_switch": "sched",
|
||||||
|
"sched:sched_wakeup": "sched",
|
||||||
|
"kmem:mm_page_alloc": "mm",
|
||||||
|
"kmem:mm_page_free": "mm",
|
||||||
|
"syscalls:sys_enter_openat": "fs",
|
||||||
|
"syscalls:sys_enter_openat2": "fs",
|
||||||
|
"syscalls:sys_enter_read": "fs",
|
||||||
|
"syscalls:sys_enter_write": "fs",
|
||||||
|
"block:block_rq_issue": "block",
|
||||||
|
"block:block_rq_complete": "block",
|
||||||
|
"net:net_dev_queue": "net",
|
||||||
|
"net:netif_receive_skb": "net",
|
||||||
|
"power:cpu_idle": "power",
|
||||||
|
"power:cpu_frequency": "power",
|
||||||
|
"irq:irq_handler_entry": "device",
|
||||||
|
"irq:softirq_entry": "device",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def tracefs_root() -> Path:
|
||||||
|
for candidate in (Path("/sys/kernel/tracing"), Path("/sys/kernel/debug/tracing")):
|
||||||
|
if candidate.exists():
|
||||||
|
return candidate
|
||||||
|
raise RuntimeError("tracefs is not mounted")
|
||||||
|
|
||||||
|
|
||||||
|
def read_text(path: Path) -> str:
|
||||||
|
try:
|
||||||
|
return path.read_text(encoding="utf-8", errors="replace").strip()
|
||||||
|
except OSError:
|
||||||
|
return ""
|
||||||
|
|
||||||
|
|
||||||
|
def write_text(path: Path, text: str) -> None:
|
||||||
|
path.write_text(text, encoding="utf-8")
|
||||||
|
|
||||||
|
|
||||||
|
def event_path(root: Path, event: str) -> Path:
|
||||||
|
system, name = event.split(":", 1)
|
||||||
|
return root / "events" / system / name / "enable"
|
||||||
|
|
||||||
|
|
||||||
|
def available_events(root: Path) -> list[str]:
|
||||||
|
available: list[str] = []
|
||||||
|
for event in EVENT_TO_SLOT:
|
||||||
|
if event_path(root, event).exists():
|
||||||
|
available.append(event)
|
||||||
|
return available
|
||||||
|
|
||||||
|
|
||||||
|
def parse_event_name(line: str) -> str | None:
|
||||||
|
match = re.search(r"\s([A-Za-z0-9_]+):\s", line)
|
||||||
|
if not match:
|
||||||
|
return None
|
||||||
|
name = match.group(1)
|
||||||
|
for full in EVENT_TO_SLOT:
|
||||||
|
if full.endswith(":" + name):
|
||||||
|
return full
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
class TraceState:
|
||||||
|
def __init__(self, root: Path, events: list[str]) -> None:
|
||||||
|
self.root = root
|
||||||
|
self.events = events
|
||||||
|
self.tracing_on = read_text(root / "tracing_on")
|
||||||
|
self.current_tracer = read_text(root / "current_tracer")
|
||||||
|
self.enabled: dict[str, str] = {}
|
||||||
|
|
||||||
|
def __enter__(self) -> "TraceState":
|
||||||
|
write_text(self.root / "tracing_on", "0")
|
||||||
|
for event in self.events:
|
||||||
|
path = event_path(self.root, event)
|
||||||
|
self.enabled[event] = read_text(path)
|
||||||
|
write_text(path, "1")
|
||||||
|
# Keep any current ftrace debugger choice; tracepoints are additive.
|
||||||
|
try:
|
||||||
|
write_text(self.root / "trace", "")
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
|
write_text(self.root / "tracing_on", "1")
|
||||||
|
return self
|
||||||
|
|
||||||
|
def __exit__(self, exc_type: Any, exc: Any, tb: Any) -> None:
|
||||||
|
write_text(self.root / "tracing_on", "0")
|
||||||
|
for event, old in self.enabled.items():
|
||||||
|
try:
|
||||||
|
write_text(event_path(self.root, event), old or "0")
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
|
if self.current_tracer:
|
||||||
|
try:
|
||||||
|
write_text(self.root / "current_tracer", self.current_tracer)
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
|
if self.tracing_on:
|
||||||
|
try:
|
||||||
|
write_text(self.root / "tracing_on", self.tracing_on)
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
|
def inject_summaries(slot_counts: dict[str, int], duration: float) -> None:
|
||||||
|
if not DEV.exists():
|
||||||
|
return
|
||||||
|
for slot, count in sorted(slot_counts.items()):
|
||||||
|
with DEV.open("w", encoding="utf-8", errors="replace") as handle:
|
||||||
|
eps = count / duration if duration > 0 else 0.0
|
||||||
|
handle.write(f"{slot} trace_sample count={count} eps={eps:.3f}\n")
|
||||||
|
|
||||||
|
|
||||||
|
def sample(duration: float, inject: bool) -> dict[str, Any]:
|
||||||
|
root = tracefs_root()
|
||||||
|
events = available_events(root)
|
||||||
|
counts = {event: 0 for event in events}
|
||||||
|
slot_counts: dict[str, int] = {}
|
||||||
|
started = time.time()
|
||||||
|
deadline = started + duration
|
||||||
|
|
||||||
|
with TraceState(root, events):
|
||||||
|
with (root / "trace_pipe").open("r", encoding="utf-8", errors="replace") as pipe:
|
||||||
|
fd = pipe.fileno()
|
||||||
|
while time.time() < deadline:
|
||||||
|
timeout = max(0.0, min(0.25, deadline - time.time()))
|
||||||
|
ready, _, _ = select.select([fd], [], [], timeout)
|
||||||
|
if not ready:
|
||||||
|
continue
|
||||||
|
line = pipe.readline()
|
||||||
|
if not line:
|
||||||
|
continue
|
||||||
|
event = parse_event_name(line)
|
||||||
|
if not event:
|
||||||
|
continue
|
||||||
|
counts[event] = counts.get(event, 0) + 1
|
||||||
|
slot = EVENT_TO_SLOT.get(event, "user")
|
||||||
|
slot_counts[slot] = slot_counts.get(slot, 0) + 1
|
||||||
|
|
||||||
|
elapsed = max(time.time() - started, duration)
|
||||||
|
if inject:
|
||||||
|
inject_summaries(slot_counts, elapsed)
|
||||||
|
|
||||||
|
return {
|
||||||
|
"schema": SCHEMA,
|
||||||
|
"timestamp_unix": int(time.time()),
|
||||||
|
"duration_sec": elapsed,
|
||||||
|
"tracefs": str(root),
|
||||||
|
"events_enabled": events,
|
||||||
|
"event_counts": counts,
|
||||||
|
"slot_counts": slot_counts,
|
||||||
|
"slot_rates_per_sec": {
|
||||||
|
slot: count / elapsed for slot, count in sorted(slot_counts.items())
|
||||||
|
},
|
||||||
|
"injected_into_module": inject and DEV.exists(),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def emit_text(payload: dict[str, Any]) -> str:
|
||||||
|
lines = [
|
||||||
|
f"schema: {payload['schema']}",
|
||||||
|
f"duration_sec: {payload['duration_sec']:.3f}",
|
||||||
|
f"tracefs: {payload['tracefs']}",
|
||||||
|
f"enabled_events: {len(payload['events_enabled'])}",
|
||||||
|
"slot_rates_per_sec:",
|
||||||
|
]
|
||||||
|
for slot, rate in payload["slot_rates_per_sec"].items():
|
||||||
|
lines.append(f" {slot}: {rate:.3f}")
|
||||||
|
lines.append("slot_counts:")
|
||||||
|
for slot, count in sorted(payload["slot_counts"].items()):
|
||||||
|
lines.append(f" {slot}: {count}")
|
||||||
|
return "\n".join(lines) + "\n"
|
||||||
|
|
||||||
|
|
||||||
|
def main(argv: list[str] | None = None) -> int:
|
||||||
|
parser = argparse.ArgumentParser(description=__doc__)
|
||||||
|
parser.add_argument("--duration", type=float, default=5.0)
|
||||||
|
parser.add_argument("--json", action="store_true")
|
||||||
|
parser.add_argument("--out", type=Path)
|
||||||
|
parser.add_argument("--no-inject", action="store_true")
|
||||||
|
args = parser.parse_args(argv)
|
||||||
|
|
||||||
|
if args.duration <= 0:
|
||||||
|
print("--duration must be positive", file=sys.stderr)
|
||||||
|
return 2
|
||||||
|
|
||||||
|
try:
|
||||||
|
payload = sample(args.duration, inject=not args.no_inject)
|
||||||
|
except (OSError, RuntimeError) as exc:
|
||||||
|
print(f"trace sample failed: {exc}", file=sys.stderr)
|
||||||
|
return 2
|
||||||
|
|
||||||
|
rendered = json.dumps(payload, indent=2, sort_keys=True) + "\n"
|
||||||
|
if args.out:
|
||||||
|
try:
|
||||||
|
args.out.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
args.out.write_text(rendered, encoding="utf-8")
|
||||||
|
except OSError as exc:
|
||||||
|
print(f"trace sample write failed: {exc}", file=sys.stderr)
|
||||||
|
return 2
|
||||||
|
|
||||||
|
print(rendered if args.json else emit_text(payload), end="")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
raise SystemExit(main())
|
||||||
|
|
@ -0,0 +1,8 @@
|
||||||
|
[Unit]
|
||||||
|
Description=Sample QFox topology manager tracepoint averages
|
||||||
|
ConditionPathExists=/dev/qfox_topoman
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
Type=oneshot
|
||||||
|
Environment=QFOX_TOPOLOGY_SAMPLE_SEC=5
|
||||||
|
ExecStart=/usr/local/sbin/qfox-topology-sample
|
||||||
|
|
@ -0,0 +1,11 @@
|
||||||
|
[Unit]
|
||||||
|
Description=Periodic QFox topology-manager average sampler
|
||||||
|
|
||||||
|
[Timer]
|
||||||
|
OnBootSec=2min
|
||||||
|
OnUnitActiveSec=15min
|
||||||
|
AccuracySec=30s
|
||||||
|
Persistent=true
|
||||||
|
|
||||||
|
[Install]
|
||||||
|
WantedBy=timers.target
|
||||||
672
4-Infrastructure/shim/boundary_activation_field_probe.py
Normal file
672
4-Infrastructure/shim/boundary_activation_field_probe.py
Normal file
|
|
@ -0,0 +1,672 @@
|
||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Receipt-bearing probe for the boundary activation field B(x, r).
|
||||||
|
|
||||||
|
A boundary is not where a system ends. A boundary is where accumulated encoded
|
||||||
|
states become physically active. B(x, r) is the boundary activation field at
|
||||||
|
location x and observer/interaction scale r.
|
||||||
|
|
||||||
|
B(x, r) = f(del_rho, delta_lambda, eta, R_del, beta_k, E_deposit)
|
||||||
|
|
||||||
|
where:
|
||||||
|
del_rho = density gradient
|
||||||
|
delta_lambda = hyper-eigen regime transition
|
||||||
|
eta = medium coupling
|
||||||
|
R_del = boundary residual / scar pressure
|
||||||
|
beta_k = topology persistence
|
||||||
|
E_deposit = cumulative deposited energy
|
||||||
|
|
||||||
|
When the superposition of encoded regime components crosses the critical
|
||||||
|
threshold, the boundary enters an active physical regime (fire, shock, plasma,
|
||||||
|
fracture, turbulence, filamentation).
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import hashlib
|
||||||
|
import json
|
||||||
|
from dataclasses import dataclass, field, asdict
|
||||||
|
from datetime import datetime, timezone
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
|
||||||
|
REPO = Path(__file__).resolve().parents[2]
|
||||||
|
OUT_DIR = REPO / "shared-data" / "data" / "boundary_activation_field"
|
||||||
|
REGISTRY = OUT_DIR / "boundary_activation_field_registry.json"
|
||||||
|
RECEIPT = OUT_DIR / "boundary_activation_field_receipt.json"
|
||||||
|
SUMMARY = OUT_DIR / "boundary_activation_field.md"
|
||||||
|
TIDDLER = (
|
||||||
|
REPO
|
||||||
|
/ "6-Documentation"
|
||||||
|
/ "tiddlywiki-local"
|
||||||
|
/ "wiki"
|
||||||
|
/ "tiddlers"
|
||||||
|
/ "Boundary Activation Field.tid"
|
||||||
|
)
|
||||||
|
|
||||||
|
SOURCE_REFS = [
|
||||||
|
REPO
|
||||||
|
/ "0-Core-Formalism"
|
||||||
|
/ "lean"
|
||||||
|
/ "Semantics"
|
||||||
|
/ "Semantics"
|
||||||
|
/ "ThresholdVector.lean",
|
||||||
|
REPO
|
||||||
|
/ "0-Core-Formalism"
|
||||||
|
/ "lean"
|
||||||
|
/ "Semantics"
|
||||||
|
/ "Semantics"
|
||||||
|
/ "BoundaryDynamics.lean",
|
||||||
|
REPO
|
||||||
|
/ "shared-data"
|
||||||
|
/ "data"
|
||||||
|
/ "observer_chart_projection_guardrail"
|
||||||
|
/ "observer_chart_projection_guardrail_receipt.json",
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def stable_json(obj: Any) -> str:
|
||||||
|
return json.dumps(obj, sort_keys=True, separators=(",", ":"), ensure_ascii=True)
|
||||||
|
|
||||||
|
|
||||||
|
def sha256_bytes(data: bytes) -> str:
|
||||||
|
return hashlib.sha256(data).hexdigest()
|
||||||
|
|
||||||
|
|
||||||
|
def hash_obj(obj: Any) -> str:
|
||||||
|
return sha256_bytes(stable_json(obj).encode("utf-8"))
|
||||||
|
|
||||||
|
|
||||||
|
def rel(path: Path) -> str:
|
||||||
|
try:
|
||||||
|
return str(path.relative_to(REPO))
|
||||||
|
except ValueError:
|
||||||
|
return str(path)
|
||||||
|
|
||||||
|
|
||||||
|
def file_hash(path: Path) -> str | None:
|
||||||
|
return sha256_bytes(path.read_bytes()) if path.exists() else None
|
||||||
|
|
||||||
|
|
||||||
|
def source_ref(path: Path) -> dict[str, Any]:
|
||||||
|
return {"path": rel(path), "exists": path.exists(), "sha256": file_hash(path)}
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Domain types
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class DensityGradient:
|
||||||
|
"""nabla_rho — the density gradient at the boundary."""
|
||||||
|
|
||||||
|
magnitude: float # [0, 1] normalized
|
||||||
|
direction: str # "inward" | "outward" | "tangential"
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class HyperEigenTransition:
|
||||||
|
"""delta_lambda — hyper-eigenvalue regime transition indicator."""
|
||||||
|
|
||||||
|
spectral_drift: float # how far the dominant eigenmode has drifted [0, 1]
|
||||||
|
regime_switch_active: bool
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class MediumCoupling:
|
||||||
|
"""eta — how strongly the boundary couples to the surrounding medium."""
|
||||||
|
|
||||||
|
coefficient: float # coupling coefficient [0, 1]
|
||||||
|
atmosphere_participating: bool # does the medium carry away energy?
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class BoundaryResidual:
|
||||||
|
"""R_del — accumulated residual / scar pressure at the boundary."""
|
||||||
|
|
||||||
|
scar_pressure: float # accumulated scar energy density [0, 1]
|
||||||
|
residual_growth_rate: float # how fast residuals are growing [0, 1]
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class TopologyPersistence:
|
||||||
|
"""beta_k — topology persistence across scale changes."""
|
||||||
|
|
||||||
|
betti_connected: int # number of connected components
|
||||||
|
betti_loops: int # number of tunnels / loops
|
||||||
|
betti_cavities: int # number of enclosed cavities
|
||||||
|
persistence_ratio: float # fraction of topology that survives scale change [0, 1]
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class DepositedEnergy:
|
||||||
|
"""E_deposit — cumulative energy deposited at the boundary."""
|
||||||
|
|
||||||
|
total: float # total deposited energy [0, 1] normalized
|
||||||
|
deposition_rate: float # rate of energy deposition [0, 1]
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class BoundaryActivationState:
|
||||||
|
"""Complete set of encoded regime components at a boundary point."""
|
||||||
|
|
||||||
|
density_gradient: DensityGradient
|
||||||
|
hyper_eigen: HyperEigenTransition
|
||||||
|
medium_coupling: MediumCoupling
|
||||||
|
boundary_residual: BoundaryResidual
|
||||||
|
topology: TopologyPersistence
|
||||||
|
deposited_energy: DepositedEnergy
|
||||||
|
|
||||||
|
# Observer / scale metadata
|
||||||
|
location_label: str # human-readable location
|
||||||
|
scale: float # observation scale in arbitrary units
|
||||||
|
|
||||||
|
def component_vector(self) -> dict[str, float]:
|
||||||
|
"""Extract the phi_i component vector for superposition computation."""
|
||||||
|
return {
|
||||||
|
"density_gradient": self.density_gradient.magnitude,
|
||||||
|
"spectral_drift": self.hyper_eigen.spectral_drift,
|
||||||
|
"coupling": self.medium_coupling.coefficient,
|
||||||
|
"scar_pressure": self.boundary_residual.scar_pressure,
|
||||||
|
"topology_persistence": self.topology.persistence_ratio,
|
||||||
|
"deposited_energy": self.deposited_energy.total,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class ActivationWeights:
|
||||||
|
"""Superposition weights for each encoded regime component."""
|
||||||
|
|
||||||
|
density_gradient: float = 0.15
|
||||||
|
spectral_drift: float = 0.20
|
||||||
|
coupling: float = 0.25
|
||||||
|
scar_pressure: float = 0.15
|
||||||
|
topology_persistence: float = 0.10
|
||||||
|
deposited_energy: float = 0.15
|
||||||
|
|
||||||
|
def total_weight(self) -> float:
|
||||||
|
return (
|
||||||
|
self.density_gradient
|
||||||
|
+ self.spectral_drift
|
||||||
|
+ self.coupling
|
||||||
|
+ self.scar_pressure
|
||||||
|
+ self.topology_persistence
|
||||||
|
+ self.deposited_energy
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class ThresholdVector:
|
||||||
|
"""Regime transition thresholds (analogue of Theta_i in the Lean model)."""
|
||||||
|
|
||||||
|
density_gradient: float = 0.35 # gradient -> fracture
|
||||||
|
spectral_drift: float = 0.50 # spectral -> mode switch
|
||||||
|
coupling: float = 0.67 # coupling -> ignition
|
||||||
|
scar_pressure: float = 0.50 # scar -> boundary instability
|
||||||
|
topology_persistence: float = 0.33 # persistence -> percolation
|
||||||
|
deposited_energy: float = 0.50 # energy -> thermal regime
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Boundary activation computation
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
def compute_total_activation(
|
||||||
|
state: BoundaryActivationState,
|
||||||
|
weights: ActivationWeights | None = None,
|
||||||
|
) -> float:
|
||||||
|
"""Compute B = sum alpha_i * phi_i, the total boundary activation.
|
||||||
|
|
||||||
|
This is the superposition of encoded regime components. When B exceeds
|
||||||
|
the critical threshold, the boundary enters an active physical regime.
|
||||||
|
"""
|
||||||
|
if weights is None:
|
||||||
|
weights = ActivationWeights()
|
||||||
|
phi = state.component_vector()
|
||||||
|
total = (
|
||||||
|
weights.density_gradient * phi["density_gradient"]
|
||||||
|
+ weights.spectral_drift * phi["spectral_drift"]
|
||||||
|
+ weights.coupling * phi["coupling"]
|
||||||
|
+ weights.scar_pressure * phi["scar_pressure"]
|
||||||
|
+ weights.topology_persistence * phi["topology_persistence"]
|
||||||
|
+ weights.deposited_energy * phi["deposited_energy"]
|
||||||
|
)
|
||||||
|
# Normalize by total weight to keep B in [0, 1]
|
||||||
|
norm = weights.total_weight()
|
||||||
|
return total / norm if norm > 0 else 0.0
|
||||||
|
|
||||||
|
|
||||||
|
CRITICAL_ACTIVATION_THRESHOLD = 0.5
|
||||||
|
|
||||||
|
|
||||||
|
def is_critically_activated(total_activation: float) -> bool:
|
||||||
|
"""Check whether B exceeds the critical threshold Theta_c."""
|
||||||
|
return total_activation > CRITICAL_ACTIVATION_THRESHOLD
|
||||||
|
|
||||||
|
|
||||||
|
def count_thresholds_crossed(
|
||||||
|
state: BoundaryActivationState,
|
||||||
|
thresholds: ThresholdVector | None = None,
|
||||||
|
) -> dict[str, bool]:
|
||||||
|
"""Determine which individual component thresholds are crossed."""
|
||||||
|
if thresholds is None:
|
||||||
|
thresholds = ThresholdVector()
|
||||||
|
phi = state.component_vector()
|
||||||
|
return {
|
||||||
|
"density_gradient": phi["density_gradient"] > thresholds.density_gradient,
|
||||||
|
"spectral_drift": phi["spectral_drift"] > thresholds.spectral_drift,
|
||||||
|
"coupling": phi["coupling"] > thresholds.coupling,
|
||||||
|
"scar_pressure": phi["scar_pressure"] > thresholds.scar_pressure,
|
||||||
|
"topology_persistence": phi["topology_persistence"]
|
||||||
|
> thresholds.topology_persistence,
|
||||||
|
"deposited_energy": phi["deposited_energy"] > thresholds.deposited_energy,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def classify_boundary_activation(
|
||||||
|
state: BoundaryActivationState,
|
||||||
|
thresholds: ThresholdVector | None = None,
|
||||||
|
weights: ActivationWeights | None = None,
|
||||||
|
) -> str:
|
||||||
|
"""Classify the boundary into an activation regime.
|
||||||
|
|
||||||
|
Returns one of: latent, smooth, turbulent, percolating, switching,
|
||||||
|
diverging, active, critical
|
||||||
|
"""
|
||||||
|
B = compute_total_activation(state, weights)
|
||||||
|
if not is_critically_activated(B):
|
||||||
|
return "latent"
|
||||||
|
|
||||||
|
crossed = count_thresholds_crossed(state, thresholds)
|
||||||
|
count = sum(1 for v in crossed.values() if v)
|
||||||
|
|
||||||
|
if count >= 4:
|
||||||
|
return "critical"
|
||||||
|
elif count >= 3:
|
||||||
|
return "active"
|
||||||
|
elif crossed.get("deposited_energy", False):
|
||||||
|
return "diverging"
|
||||||
|
elif crossed.get("spectral_drift", False):
|
||||||
|
return "switching"
|
||||||
|
elif crossed.get("coupling", False):
|
||||||
|
return "turbulent"
|
||||||
|
elif crossed.get("topology_persistence", False):
|
||||||
|
return "percolating"
|
||||||
|
elif crossed.get("density_gradient", False):
|
||||||
|
return "smooth"
|
||||||
|
else:
|
||||||
|
return "latent"
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Canonical scenario builders
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
def zero_activation_state(label: str = "void interior") -> BoundaryActivationState:
|
||||||
|
return BoundaryActivationState(
|
||||||
|
density_gradient=DensityGradient(0.0, "tangential"),
|
||||||
|
hyper_eigen=HyperEigenTransition(0.0, False),
|
||||||
|
medium_coupling=MediumCoupling(0.0, False),
|
||||||
|
boundary_residual=BoundaryResidual(0.0, 0.0),
|
||||||
|
topology=TopologyPersistence(0, 0, 0, 0.0),
|
||||||
|
deposited_energy=DepositedEnergy(0.0, 0.0),
|
||||||
|
location_label=label,
|
||||||
|
scale=1.0,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def wall_fracture_scenario(label: str = "wall fracture") -> BoundaryActivationState:
|
||||||
|
return BoundaryActivationState(
|
||||||
|
density_gradient=DensityGradient(0.8, "outward"),
|
||||||
|
hyper_eigen=HyperEigenTransition(0.2, False),
|
||||||
|
medium_coupling=MediumCoupling(0.1, False),
|
||||||
|
boundary_residual=BoundaryResidual(0.6, 0.4),
|
||||||
|
topology=TopologyPersistence(3, 1, 0, 0.5),
|
||||||
|
deposited_energy=DepositedEnergy(0.3, 0.7),
|
||||||
|
location_label=label,
|
||||||
|
scale=0.1,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def atmospheric_ignition_scenario(
|
||||||
|
label: str = "atmospheric ignition",
|
||||||
|
) -> BoundaryActivationState:
|
||||||
|
return BoundaryActivationState(
|
||||||
|
density_gradient=DensityGradient(0.9, "outward"),
|
||||||
|
hyper_eigen=HyperEigenTransition(0.6, True),
|
||||||
|
medium_coupling=MediumCoupling(0.9, True),
|
||||||
|
boundary_residual=BoundaryResidual(0.4, 0.3),
|
||||||
|
topology=TopologyPersistence(5, 2, 0, 0.7),
|
||||||
|
deposited_energy=DepositedEnergy(0.8, 0.9),
|
||||||
|
location_label=label,
|
||||||
|
scale=0.05,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def cosmic_filament_scenario(
|
||||||
|
label: str = "cosmic filament wall",
|
||||||
|
) -> BoundaryActivationState:
|
||||||
|
return BoundaryActivationState(
|
||||||
|
density_gradient=DensityGradient(0.6, "inward"),
|
||||||
|
hyper_eigen=HyperEigenTransition(0.4, False),
|
||||||
|
medium_coupling=MediumCoupling(0.3, False),
|
||||||
|
boundary_residual=BoundaryResidual(0.5, 0.2),
|
||||||
|
topology=TopologyPersistence(200, 45, 12, 0.85),
|
||||||
|
deposited_energy=DepositedEnergy(0.7, 0.05),
|
||||||
|
location_label=label,
|
||||||
|
scale=100.0,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def hulk_punch_scenario(
|
||||||
|
label: str = "hulk punch fracture",
|
||||||
|
) -> BoundaryActivationState:
|
||||||
|
return BoundaryActivationState(
|
||||||
|
density_gradient=DensityGradient(1.0, "outward"),
|
||||||
|
hyper_eigen=HyperEigenTransition(0.7, True),
|
||||||
|
medium_coupling=MediumCoupling(0.8, True),
|
||||||
|
boundary_residual=BoundaryResidual(0.9, 0.9),
|
||||||
|
topology=TopologyPersistence(50, 10, 3, 0.4),
|
||||||
|
deposited_energy=DepositedEnergy(1.0, 1.0),
|
||||||
|
location_label=label,
|
||||||
|
scale=0.01,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Registry and receipt
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
def build_scenario_record(
|
||||||
|
index: int,
|
||||||
|
scenario_id: str,
|
||||||
|
state: BoundaryActivationState,
|
||||||
|
weights: ActivationWeights | None = None,
|
||||||
|
thresholds: ThresholdVector | None = None,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
if weights is None:
|
||||||
|
weights = ActivationWeights()
|
||||||
|
if thresholds is None:
|
||||||
|
thresholds = ThresholdVector()
|
||||||
|
|
||||||
|
B = compute_total_activation(state, weights)
|
||||||
|
verdict = classify_boundary_activation(state, thresholds, weights)
|
||||||
|
crossed = count_thresholds_crossed(state, thresholds)
|
||||||
|
|
||||||
|
record = {
|
||||||
|
"index": index,
|
||||||
|
"scenario_id": scenario_id,
|
||||||
|
"location_label": state.location_label,
|
||||||
|
"scale": state.scale,
|
||||||
|
"component_vector": state.component_vector(),
|
||||||
|
"weights": asdict(weights),
|
||||||
|
"thresholds": asdict(thresholds),
|
||||||
|
"total_activation_B": round(B, 6),
|
||||||
|
"critical_threshold": CRITICAL_ACTIVATION_THRESHOLD,
|
||||||
|
"is_critical": is_critically_activated(B),
|
||||||
|
"thresholds_crossed": crossed,
|
||||||
|
"thresholds_crossed_count": sum(1 for v in crossed.values() if v),
|
||||||
|
"activation_verdict": verdict,
|
||||||
|
"density_gradient": {
|
||||||
|
"magnitude": state.density_gradient.magnitude,
|
||||||
|
"direction": state.density_gradient.direction,
|
||||||
|
},
|
||||||
|
"hyper_eigen": {
|
||||||
|
"spectral_drift": state.hyper_eigen.spectral_drift,
|
||||||
|
"regime_switch_active": state.hyper_eigen.regime_switch_active,
|
||||||
|
},
|
||||||
|
"medium_coupling": {
|
||||||
|
"coefficient": state.medium_coupling.coefficient,
|
||||||
|
"atmosphere_participating": state.medium_coupling.atmosphere_participating,
|
||||||
|
},
|
||||||
|
"boundary_residual": {
|
||||||
|
"scar_pressure": state.boundary_residual.scar_pressure,
|
||||||
|
"residual_growth_rate": state.boundary_residual.residual_growth_rate,
|
||||||
|
},
|
||||||
|
"topology": {
|
||||||
|
"betti_connected": state.topology.betti_connected,
|
||||||
|
"betti_loops": state.topology.betti_loops,
|
||||||
|
"betti_cavities": state.topology.betti_cavities,
|
||||||
|
"persistence_ratio": state.topology.persistence_ratio,
|
||||||
|
},
|
||||||
|
"deposited_energy": {
|
||||||
|
"total": state.deposited_energy.total,
|
||||||
|
"deposition_rate": state.deposited_energy.deposition_rate,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
record["record_hash"] = hash_obj({k: v for k, v in record.items() if k != "record_hash"})
|
||||||
|
return record
|
||||||
|
|
||||||
|
|
||||||
|
_DEFAULT_WEIGHTS = ActivationWeights()
|
||||||
|
_DEFAULT_THRESHOLDS = ThresholdVector()
|
||||||
|
|
||||||
|
SCENARIOS: list[tuple[str, BoundaryActivationState]] = [
|
||||||
|
("zero_activation", zero_activation_state()),
|
||||||
|
("wall_fracture", wall_fracture_scenario()),
|
||||||
|
("atmospheric_ignition", atmospheric_ignition_scenario()),
|
||||||
|
("cosmic_filament", cosmic_filament_scenario()),
|
||||||
|
("hulk_punch", hulk_punch_scenario()),
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def build_registry() -> dict[str, Any]:
|
||||||
|
scenario_records = [
|
||||||
|
build_scenario_record(
|
||||||
|
i,
|
||||||
|
sid,
|
||||||
|
state,
|
||||||
|
_DEFAULT_WEIGHTS,
|
||||||
|
_DEFAULT_THRESHOLDS,
|
||||||
|
)
|
||||||
|
for i, (sid, state) in enumerate(SCENARIOS)
|
||||||
|
]
|
||||||
|
|
||||||
|
return {
|
||||||
|
"schema": "boundary_activation_field_registry_v1",
|
||||||
|
"source_refs": [source_ref(path) for path in SOURCE_REFS],
|
||||||
|
"claim_boundary": (
|
||||||
|
"Boundary activation field B(x, r) model only. Classifies boundary "
|
||||||
|
"regimes based on the weighted superposition of six encoded regime "
|
||||||
|
"components: density gradient, hyper-eigen spectral drift, medium "
|
||||||
|
"coupling, scar pressure, topology persistence, and deposited "
|
||||||
|
"energy. Does not claim full cosmological or material-science "
|
||||||
|
"predictive power without calibration to domain-specific data."
|
||||||
|
),
|
||||||
|
"canonical_statement": (
|
||||||
|
"A boundary is not where a system ends. A boundary is where "
|
||||||
|
"accumulated encoded states become physically active."
|
||||||
|
),
|
||||||
|
"superposition_equation": "B(x, r) = sum_i alpha_i * phi_i(x, r)",
|
||||||
|
"critical_condition": "B > Theta_c => boundary enters active physical regime",
|
||||||
|
"critical_threshold": CRITICAL_ACTIVATION_THRESHOLD,
|
||||||
|
"default_weights": asdict(_DEFAULT_WEIGHTS),
|
||||||
|
"default_thresholds": asdict(_DEFAULT_THRESHOLDS),
|
||||||
|
"regime_map": {
|
||||||
|
"latent": "no threshold crossed, boundary inactive",
|
||||||
|
"smooth": "density gradient regime, elastic/smooth transition",
|
||||||
|
"turbulent": "coupling regime, atmospheric ignition boundary",
|
||||||
|
"percolating": "topology regime, filament/web connectivity",
|
||||||
|
"switching": "spectral regime, eigenmode transition",
|
||||||
|
"diverging": "energy regime, thermal/divergence front",
|
||||||
|
"active": "3+ thresholds crossed, full boundary activation",
|
||||||
|
"critical": "4+ thresholds crossed, topology-tear regime",
|
||||||
|
},
|
||||||
|
"scenarios": scenario_records,
|
||||||
|
"aggregates": {
|
||||||
|
"scenario_count": len(scenario_records),
|
||||||
|
"activation_verdicts": {
|
||||||
|
r["activation_verdict"]: sum(
|
||||||
|
1 for s in scenario_records if s["activation_verdict"] == r["activation_verdict"]
|
||||||
|
)
|
||||||
|
for r in scenario_records
|
||||||
|
},
|
||||||
|
"critical_count": sum(1 for s in scenario_records if s["is_critical"]),
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def build_receipt(registry: dict[str, Any]) -> dict[str, Any]:
|
||||||
|
receipt = {
|
||||||
|
"schema": "boundary_activation_field_receipt_v1",
|
||||||
|
"generated_at_utc": datetime.now(timezone.utc).isoformat(),
|
||||||
|
"timestamp_role": "metadata_only",
|
||||||
|
"generated_at_utc_included_in_receipt_hash": False,
|
||||||
|
"registry": rel(REGISTRY),
|
||||||
|
"registry_hash": hash_obj(registry),
|
||||||
|
"aggregates": registry["aggregates"],
|
||||||
|
"decision": "ADMIT_BOUNDARY_ACTIVATION_FIELD",
|
||||||
|
"claim_boundary": registry["claim_boundary"],
|
||||||
|
}
|
||||||
|
receipt["receipt_hash"] = sha256_bytes(
|
||||||
|
stable_json(
|
||||||
|
{k: v for k, v in receipt.items() if k not in {"receipt_hash", "generated_at_utc"}}
|
||||||
|
).encode("utf-8")
|
||||||
|
)
|
||||||
|
return receipt
|
||||||
|
|
||||||
|
|
||||||
|
def write_summary(registry: dict[str, Any], receipt: dict[str, Any]) -> None:
|
||||||
|
lines = [
|
||||||
|
"# Boundary Activation Field",
|
||||||
|
"",
|
||||||
|
f"Decision: `{receipt['decision']}`",
|
||||||
|
f"Receipt hash: `{receipt['receipt_hash']}`",
|
||||||
|
"",
|
||||||
|
registry["claim_boundary"],
|
||||||
|
"",
|
||||||
|
"## Canonical Statement",
|
||||||
|
"",
|
||||||
|
registry["canonical_statement"],
|
||||||
|
"",
|
||||||
|
"## Equations",
|
||||||
|
"",
|
||||||
|
f"- Superposition: `{registry['superposition_equation']}`",
|
||||||
|
f"- Critical condition: `{registry['critical_condition']}`",
|
||||||
|
f"- Theta_c = {registry['critical_threshold']}",
|
||||||
|
"",
|
||||||
|
"## Regime Map",
|
||||||
|
"",
|
||||||
|
]
|
||||||
|
for regime, description in registry["regime_map"].items():
|
||||||
|
lines.append(f"- `{regime}`: {description}")
|
||||||
|
lines.extend(
|
||||||
|
[
|
||||||
|
"",
|
||||||
|
"## Scenarios",
|
||||||
|
"",
|
||||||
|
"| Scenario | Location | B | Critical | Thresholds Crossed | Verdict |",
|
||||||
|
"|---|---|---|---|---|---|",
|
||||||
|
]
|
||||||
|
)
|
||||||
|
for s in registry["scenarios"]:
|
||||||
|
lines.append(
|
||||||
|
f"| `{s['scenario_id']}` | {s['location_label']} | "
|
||||||
|
f"{s['total_activation_B']} | {s['is_critical']} | "
|
||||||
|
f"{s['thresholds_crossed_count']} | `{s['activation_verdict']}` |"
|
||||||
|
)
|
||||||
|
lines.extend(
|
||||||
|
[
|
||||||
|
"",
|
||||||
|
"## Aggregates",
|
||||||
|
"",
|
||||||
|
f"- Scenario count: {registry['aggregates']['scenario_count']}",
|
||||||
|
f"- Critical count: {registry['aggregates']['critical_count']}",
|
||||||
|
f"- Verdicts: {registry['aggregates']['activation_verdicts']}",
|
||||||
|
"",
|
||||||
|
"## Source Refs",
|
||||||
|
"",
|
||||||
|
]
|
||||||
|
)
|
||||||
|
for source in registry["source_refs"]:
|
||||||
|
lines.append(f"- `{source['path']}` exists: `{source['exists']}`")
|
||||||
|
SUMMARY.write_text("\n".join(lines) + "\n", encoding="utf-8")
|
||||||
|
|
||||||
|
|
||||||
|
def write_tiddler(receipt: dict[str, Any]) -> None:
|
||||||
|
text = f"""created: 20260512000000000
|
||||||
|
modified: 20260512000000000
|
||||||
|
tags: ResearchStack Encoding BoundaryActivation Receipt
|
||||||
|
title: Boundary Activation Field
|
||||||
|
type: text/vnd.tiddlywiki
|
||||||
|
|
||||||
|
! Boundary Activation Field
|
||||||
|
|
||||||
|
Durable runner:
|
||||||
|
|
||||||
|
```
|
||||||
|
4-Infrastructure/shim/boundary_activation_field_probe.py
|
||||||
|
```
|
||||||
|
|
||||||
|
Receipt:
|
||||||
|
|
||||||
|
```
|
||||||
|
{rel(RECEIPT)}
|
||||||
|
```
|
||||||
|
|
||||||
|
Receipt hash:
|
||||||
|
|
||||||
|
```
|
||||||
|
{receipt['receipt_hash']}
|
||||||
|
```
|
||||||
|
|
||||||
|
!! Doctrine
|
||||||
|
|
||||||
|
A boundary is not where a system ends. A boundary is where accumulated encoded states become physically active.
|
||||||
|
|
||||||
|
```
|
||||||
|
latent -> no threshold crossed, boundary inactive
|
||||||
|
smooth -> density gradient regime, elastic/smooth transition
|
||||||
|
turbulent -> coupling regime, atmospheric ignition boundary
|
||||||
|
percolating -> topology regime, filament/web connectivity
|
||||||
|
switching -> spectral regime, eigenmode transition
|
||||||
|
diverging -> energy regime, thermal/divergence front
|
||||||
|
active -> 3+ thresholds crossed, full boundary activation
|
||||||
|
critical -> 4+ thresholds crossed, topology-tear regime
|
||||||
|
```
|
||||||
|
|
||||||
|
!! Links
|
||||||
|
|
||||||
|
* [[ThresholdVector (Lean formalization)|ThresholdVector.lean]]
|
||||||
|
* [[Observer Chart Projection Guardrail]]
|
||||||
|
* [[Boundary Dynamics]]
|
||||||
|
"""
|
||||||
|
TIDDLER.write_text(text, encoding="utf-8")
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> int:
|
||||||
|
OUT_DIR.mkdir(parents=True, exist_ok=True)
|
||||||
|
registry = build_registry()
|
||||||
|
receipt = build_receipt(registry)
|
||||||
|
REGISTRY.write_text(
|
||||||
|
json.dumps(registry, indent=2, sort_keys=True) + "\n", encoding="utf-8"
|
||||||
|
)
|
||||||
|
RECEIPT.write_text(
|
||||||
|
json.dumps(receipt, indent=2, sort_keys=True) + "\n", encoding="utf-8"
|
||||||
|
)
|
||||||
|
write_summary(registry, receipt)
|
||||||
|
write_tiddler(receipt)
|
||||||
|
print(
|
||||||
|
json.dumps(
|
||||||
|
{
|
||||||
|
"registry": rel(REGISTRY),
|
||||||
|
"receipt": rel(RECEIPT),
|
||||||
|
"summary": rel(SUMMARY),
|
||||||
|
"tiddler": rel(TIDDLER),
|
||||||
|
"receipt_hash": receipt["receipt_hash"],
|
||||||
|
"decision": receipt["decision"],
|
||||||
|
"aggregates": registry["aggregates"],
|
||||||
|
},
|
||||||
|
indent=2,
|
||||||
|
sort_keys=True,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
raise SystemExit(main())
|
||||||
501
4-Infrastructure/shim/holographic_carving_probe.py
Normal file
501
4-Infrastructure/shim/holographic_carving_probe.py
Normal file
|
|
@ -0,0 +1,501 @@
|
||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Combined holographic encoding + Menger-style carving via threshold-band exclusion.
|
||||||
|
|
||||||
|
Instead of removing coordinates (Menger), the beam superposition B(x, r)
|
||||||
|
carves voids by threshold-band non-activation: at each point, only structures
|
||||||
|
whose lambda-band matches the local B value materialize. Everything else is
|
||||||
|
"void" at that point.
|
||||||
|
|
||||||
|
This gives a scaffold where multiple structures share coordinates but separate
|
||||||
|
in lambda-space. The expansion-space cost is lambda-separation, not
|
||||||
|
coordinate-buffer volume.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import hashlib
|
||||||
|
import json
|
||||||
|
from dataclasses import dataclass, field, asdict
|
||||||
|
from datetime import datetime, timezone
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
|
||||||
|
REPO = Path(__file__).resolve().parents[2]
|
||||||
|
OUT_DIR = REPO / "shared-data" / "data" / "holographic_carving"
|
||||||
|
REGISTRY = OUT_DIR / "holographic_carving_registry.json"
|
||||||
|
RECEIPT = OUT_DIR / "holographic_carving_receipt.json"
|
||||||
|
SUMMARY = OUT_DIR / "holographic_carving.md"
|
||||||
|
TIDDLER = (
|
||||||
|
REPO
|
||||||
|
/ "6-Documentation"
|
||||||
|
/ "tiddlywiki-local"
|
||||||
|
/ "wiki"
|
||||||
|
/ "tiddlers"
|
||||||
|
/ "Holographic Carving.tid"
|
||||||
|
)
|
||||||
|
|
||||||
|
SOURCE_REFS = [
|
||||||
|
REPO
|
||||||
|
/ "0-Core-Formalism"
|
||||||
|
/ "lean"
|
||||||
|
/ "Semantics"
|
||||||
|
/ "Semantics"
|
||||||
|
/ "LogogramRotationLoop.lean",
|
||||||
|
REPO
|
||||||
|
/ "0-Core-Formalism"
|
||||||
|
/ "lean"
|
||||||
|
/ "Semantics"
|
||||||
|
/ "Semantics"
|
||||||
|
/ "ThresholdVector.lean",
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def stable_json(obj: Any) -> str:
|
||||||
|
return json.dumps(obj, sort_keys=True, separators=(",", ":"), ensure_ascii=True)
|
||||||
|
|
||||||
|
|
||||||
|
def sha256_bytes(data: bytes) -> str:
|
||||||
|
return hashlib.sha256(data).hexdigest()
|
||||||
|
|
||||||
|
|
||||||
|
def hash_obj(obj: Any) -> str:
|
||||||
|
return sha256_bytes(stable_json(obj).encode("utf-8"))
|
||||||
|
|
||||||
|
|
||||||
|
def rel(path: Path) -> str:
|
||||||
|
try:
|
||||||
|
return str(path.relative_to(REPO))
|
||||||
|
except ValueError:
|
||||||
|
return str(path)
|
||||||
|
|
||||||
|
|
||||||
|
def file_hash(path: Path) -> str | None:
|
||||||
|
return sha256_bytes(path.read_bytes()) if path.exists() else None
|
||||||
|
|
||||||
|
|
||||||
|
def source_ref(path: Path) -> dict[str, Any]:
|
||||||
|
return {"path": rel(path), "exists": path.exists(), "sha256": file_hash(path)}
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Core types
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class ThresholdBand:
|
||||||
|
lower: float
|
||||||
|
upper: float
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class ProjectionLayer:
|
||||||
|
angle: float
|
||||||
|
encoding: dict[str, float] # phi vector
|
||||||
|
band: ThresholdBand
|
||||||
|
label: str
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class CarvingVoxel:
|
||||||
|
"""A point in the volume: what materializes depends on B(x)."""
|
||||||
|
x: float
|
||||||
|
y: float
|
||||||
|
z: float
|
||||||
|
B: float
|
||||||
|
active_structures: dict[str, bool]
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Carving engine
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
def band_contains(B: float, band: ThresholdBand) -> bool:
|
||||||
|
return band.lower <= B <= band.upper
|
||||||
|
|
||||||
|
|
||||||
|
def integrate_beam(layers: list[ProjectionLayer], weights: dict[str, float]) -> float:
|
||||||
|
"""Compute B = sum alpha_i * phi_i over all layers."""
|
||||||
|
total = 0.0
|
||||||
|
for layer in layers:
|
||||||
|
for comp, val in layer.encoding.items():
|
||||||
|
total += weights.get(comp, 0.0) * val
|
||||||
|
weight_sum = sum(weights.values())
|
||||||
|
return total / weight_sum if weight_sum > 0 else 0.0
|
||||||
|
|
||||||
|
|
||||||
|
def resolve_voxel(
|
||||||
|
B: float,
|
||||||
|
layers: list[ProjectionLayer],
|
||||||
|
critical_threshold: float,
|
||||||
|
) -> dict[str, bool]:
|
||||||
|
"""At a point with total activation B, which structures materialize?"""
|
||||||
|
critical = B >= critical_threshold
|
||||||
|
return {
|
||||||
|
layer.label: (critical and band_contains(B, layer.band))
|
||||||
|
for layer in layers
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def carve_volume(
|
||||||
|
layers: list[ProjectionLayer],
|
||||||
|
weights: dict[str, float],
|
||||||
|
critical_threshold: float,
|
||||||
|
resolution: int = 4,
|
||||||
|
) -> list[CarvingVoxel]:
|
||||||
|
"""Evaluate B(x) over a 3D grid, producing active/void at each voxel."""
|
||||||
|
voxels = []
|
||||||
|
B_beam = integrate_beam(layers, weights)
|
||||||
|
for i in range(resolution):
|
||||||
|
for j in range(resolution):
|
||||||
|
for k in range(resolution):
|
||||||
|
x = i / (resolution - 1) if resolution > 1 else 0.5
|
||||||
|
y = j / (resolution - 1) if resolution > 1 else 0.5
|
||||||
|
z = k / (resolution - 1) if resolution > 1 else 0.5
|
||||||
|
# In the combined model, B varies across the volume.
|
||||||
|
# For this probe, we modulate B by position to show
|
||||||
|
# spatial variation in threshold-band activation.
|
||||||
|
B_local = B_beam * (1.0 - 0.3 * ((x - 0.5) ** 2 + (y - 0.5) ** 2 + (z - 0.5) ** 2) / 0.75)
|
||||||
|
active = resolve_voxel(B_local, layers, critical_threshold)
|
||||||
|
voxels.append(CarvingVoxel(x, y, z, round(B_local, 4), active))
|
||||||
|
return voxels
|
||||||
|
|
||||||
|
|
||||||
|
def count_active_voxels(voxels: list[CarvingVoxel], structure_label: str) -> int:
|
||||||
|
return sum(1 for v in voxels if v.active_structures.get(structure_label, False))
|
||||||
|
|
||||||
|
|
||||||
|
def count_void_voxels(voxels: list[CarvingVoxel]) -> int:
|
||||||
|
return sum(1 for v in voxels if not any(v.active_structures.values()))
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Scenarios
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
LOW_BAND = ThresholdBand(0.0, 0.35)
|
||||||
|
MID_BAND = ThresholdBand(0.35, 0.65)
|
||||||
|
HIGH_BAND = ThresholdBand(0.65, 1.0)
|
||||||
|
|
||||||
|
DEFAULT_WEIGHTS = {
|
||||||
|
"density_gradient": 0.20,
|
||||||
|
"spectral_drift": 0.20,
|
||||||
|
"coupling": 0.20,
|
||||||
|
"scar_pressure": 0.15,
|
||||||
|
"topology_persistence": 0.10,
|
||||||
|
"deposited_energy": 0.15,
|
||||||
|
}
|
||||||
|
|
||||||
|
DEFAULT_CRITICAL = 0.5
|
||||||
|
|
||||||
|
|
||||||
|
def single_structure_scenario() -> dict[str, Any]:
|
||||||
|
"""Baseline: one beam, one structure (pre-holographic)."""
|
||||||
|
layers = [
|
||||||
|
ProjectionLayer(
|
||||||
|
angle=0.0,
|
||||||
|
encoding={"density_gradient": 1.0, "spectral_drift": 0.0,
|
||||||
|
"coupling": 0.0, "scar_pressure": 0.0,
|
||||||
|
"topology_persistence": 0.0, "deposited_energy": 0.0},
|
||||||
|
band=LOW_BAND,
|
||||||
|
label="single_structure",
|
||||||
|
)
|
||||||
|
]
|
||||||
|
B_beam = integrate_beam(layers, DEFAULT_WEIGHTS)
|
||||||
|
voxels = carve_volume(layers, DEFAULT_WEIGHTS, DEFAULT_CRITICAL, resolution=4)
|
||||||
|
return {
|
||||||
|
"scenario_id": "single_structure_baseline",
|
||||||
|
"n_layers": len(layers),
|
||||||
|
"n_structures": 1,
|
||||||
|
"B_beam": round(B_beam, 4),
|
||||||
|
"total_voxels": len(voxels),
|
||||||
|
"active_voxels": {
|
||||||
|
"single_structure": count_active_voxels(voxels, "single_structure"),
|
||||||
|
},
|
||||||
|
"void_voxels": count_void_voxels(voxels),
|
||||||
|
"packing_efficiency": round(count_active_voxels(voxels, "single_structure") / len(voxels), 4),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def three_structure_scenario() -> dict[str, Any]:
|
||||||
|
"""Three structures in one beam, separated by threshold bands."""
|
||||||
|
layers = [
|
||||||
|
ProjectionLayer(
|
||||||
|
angle=0.0,
|
||||||
|
encoding={"density_gradient": 0.5, "spectral_drift": 0.0,
|
||||||
|
"coupling": 0.0, "scar_pressure": 0.0,
|
||||||
|
"topology_persistence": 0.0, "deposited_energy": 0.0},
|
||||||
|
band=LOW_BAND,
|
||||||
|
label="density_scaffold",
|
||||||
|
),
|
||||||
|
ProjectionLayer(
|
||||||
|
angle=0.333,
|
||||||
|
encoding={"density_gradient": 0.0, "spectral_drift": 1.0,
|
||||||
|
"coupling": 0.0, "scar_pressure": 0.0,
|
||||||
|
"topology_persistence": 0.0, "deposited_energy": 0.0},
|
||||||
|
band=MID_BAND,
|
||||||
|
label="spectral_filament",
|
||||||
|
),
|
||||||
|
ProjectionLayer(
|
||||||
|
angle=0.667,
|
||||||
|
encoding={"density_gradient": 0.0, "spectral_drift": 0.0,
|
||||||
|
"coupling": 0.0, "scar_pressure": 0.0,
|
||||||
|
"topology_persistence": 1.0, "deposited_energy": 1.0},
|
||||||
|
band=HIGH_BAND,
|
||||||
|
label="topology_web",
|
||||||
|
),
|
||||||
|
]
|
||||||
|
B_beam = integrate_beam(layers, DEFAULT_WEIGHTS)
|
||||||
|
voxels = carve_volume(layers, DEFAULT_WEIGHTS, DEFAULT_CRITICAL, resolution=4)
|
||||||
|
active_counts = {
|
||||||
|
label: count_active_voxels(voxels, label)
|
||||||
|
for label in ["density_scaffold", "spectral_filament", "topology_web"]
|
||||||
|
}
|
||||||
|
total_active = sum(active_counts.values())
|
||||||
|
return {
|
||||||
|
"scenario_id": "three_structure_holographic",
|
||||||
|
"n_layers": len(layers),
|
||||||
|
"n_structures": 3,
|
||||||
|
"B_beam": round(B_beam, 4),
|
||||||
|
"total_voxels": len(voxels),
|
||||||
|
"active_voxels": active_counts,
|
||||||
|
"total_active_voxels": total_active,
|
||||||
|
"void_voxels": count_void_voxels(voxels),
|
||||||
|
"packing_efficiency": round(total_active / len(voxels), 4),
|
||||||
|
"structures_per_beam": 3,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def carving_void_scenario() -> dict[str, Any]:
|
||||||
|
"""Menger-like carving: structures create voids in each other's bands."""
|
||||||
|
layers = [
|
||||||
|
ProjectionLayer(
|
||||||
|
angle=0.0,
|
||||||
|
encoding={"density_gradient": 0.8, "spectral_drift": 0.0,
|
||||||
|
"coupling": 0.0, "scar_pressure": 0.0,
|
||||||
|
"topology_persistence": 0.0, "deposited_energy": 0.0},
|
||||||
|
band=LOW_BAND,
|
||||||
|
label="scaffold",
|
||||||
|
),
|
||||||
|
ProjectionLayer(
|
||||||
|
angle=0.5,
|
||||||
|
encoding={"density_gradient": 0.0, "spectral_drift": 0.0,
|
||||||
|
"coupling": 0.0, "scar_pressure": 0.0,
|
||||||
|
"topology_persistence": 0.0, "deposited_energy": 1.0},
|
||||||
|
band=HIGH_BAND,
|
||||||
|
label="energy_void",
|
||||||
|
),
|
||||||
|
]
|
||||||
|
B_beam = integrate_beam(layers, DEFAULT_WEIGHTS)
|
||||||
|
voxels = carve_volume(layers, DEFAULT_WEIGHTS, DEFAULT_CRITICAL, resolution=6)
|
||||||
|
scaffold_active = count_active_voxels(voxels, "scaffold")
|
||||||
|
void_active = count_active_voxels(voxels, "energy_void")
|
||||||
|
void_count = count_void_voxels(voxels)
|
||||||
|
return {
|
||||||
|
"scenario_id": "carving_void",
|
||||||
|
"n_layers": len(layers),
|
||||||
|
"n_structures": 2,
|
||||||
|
"B_beam": round(B_beam, 4),
|
||||||
|
"total_voxels": len(voxels),
|
||||||
|
"active_voxels": {
|
||||||
|
"scaffold": scaffold_active,
|
||||||
|
"energy_void": void_active,
|
||||||
|
},
|
||||||
|
"void_voxels": void_count,
|
||||||
|
"scaffold_void_ratio": round(scaffold_active / void_count, 4) if void_count else -1,
|
||||||
|
"packing_efficiency": round((scaffold_active + void_active) / len(voxels), 4),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Registry and receipt
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
def build_registry() -> dict[str, Any]:
|
||||||
|
scenarios = [
|
||||||
|
single_structure_scenario(),
|
||||||
|
three_structure_scenario(),
|
||||||
|
carving_void_scenario(),
|
||||||
|
]
|
||||||
|
return {
|
||||||
|
"schema": "holographic_carving_registry_v1",
|
||||||
|
"source_refs": [source_ref(path) for path in SOURCE_REFS],
|
||||||
|
"claim_boundary": (
|
||||||
|
"Combined holographic encoding + Menger-style carving demo. "
|
||||||
|
"The beam superposition carries multiple structures; threshold-band "
|
||||||
|
"filtering determines which materialize at each voxel. "
|
||||||
|
"Does not claim physical printing fidelity without dose-calibration."
|
||||||
|
),
|
||||||
|
"canonical_statement": (
|
||||||
|
"Voids are not removed coordinates. "
|
||||||
|
"Voids are un-activated threshold bands at a given boundary point."
|
||||||
|
),
|
||||||
|
"superposition_equation": "B(x) = sum_i alpha_i * phi_i(x)",
|
||||||
|
"carving_rule": "structure S materializes at x iff B(x) in band(S) AND B(x) >= critical",
|
||||||
|
"void_rule": "point x is void iff B(x) < critical OR B(x) not in any structure's band",
|
||||||
|
"critical_threshold": DEFAULT_CRITICAL,
|
||||||
|
"default_weights": DEFAULT_WEIGHTS,
|
||||||
|
"scenarios": scenarios,
|
||||||
|
"aggregates": {
|
||||||
|
"scenario_count": len(scenarios),
|
||||||
|
"total_structures": sum(s["n_structures"] for s in scenarios),
|
||||||
|
"total_active_voxels": sum(s.get("total_active_voxels", s.get("active_voxels", {}).get(list(s["active_voxels"].keys())[0], 0)) for s in scenarios),
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def build_receipt(registry: dict[str, Any]) -> dict[str, Any]:
|
||||||
|
receipt = {
|
||||||
|
"schema": "holographic_carving_receipt_v1",
|
||||||
|
"generated_at_utc": datetime.now(timezone.utc).isoformat(),
|
||||||
|
"timestamp_role": "metadata_only",
|
||||||
|
"generated_at_utc_included_in_receipt_hash": False,
|
||||||
|
"registry": rel(REGISTRY),
|
||||||
|
"registry_hash": hash_obj(registry),
|
||||||
|
"aggregates": registry["aggregates"],
|
||||||
|
"decision": "ADMIT_HOLOGRAPHIC_CARVING_MODEL",
|
||||||
|
"claim_boundary": registry["claim_boundary"],
|
||||||
|
}
|
||||||
|
receipt["receipt_hash"] = sha256_bytes(
|
||||||
|
stable_json(
|
||||||
|
{k: v for k, v in receipt.items() if k not in {"receipt_hash", "generated_at_utc"}}
|
||||||
|
).encode("utf-8")
|
||||||
|
)
|
||||||
|
return receipt
|
||||||
|
|
||||||
|
|
||||||
|
def write_summary(registry: dict[str, Any], receipt: dict[str, Any]) -> None:
|
||||||
|
lines = [
|
||||||
|
"# Holographic Carving — Combined Encoding + Threshold-Band Carving",
|
||||||
|
"",
|
||||||
|
f"Decision: `{receipt['decision']}`",
|
||||||
|
f"Receipt hash: `{receipt['receipt_hash']}`",
|
||||||
|
"",
|
||||||
|
registry["claim_boundary"],
|
||||||
|
"",
|
||||||
|
"## Canonical Statement",
|
||||||
|
"",
|
||||||
|
registry["canonical_statement"],
|
||||||
|
"",
|
||||||
|
"## Equations",
|
||||||
|
"",
|
||||||
|
f"- Superposition: `{registry['superposition_equation']}`",
|
||||||
|
f"- Carving rule: `{registry['carving_rule']}`",
|
||||||
|
f"- Void rule: `{registry['void_rule']}`",
|
||||||
|
f"- Critical threshold = {registry['critical_threshold']}",
|
||||||
|
"",
|
||||||
|
"## Scenarios",
|
||||||
|
"",
|
||||||
|
"| Scenario | Structures | B_beam | Voxels | Active | Void | Efficiency |",
|
||||||
|
"|---|---|---|---|---|---|---|",
|
||||||
|
]
|
||||||
|
for s in registry["scenarios"]:
|
||||||
|
active = s.get("total_active_voxels", list(s["active_voxels"].values())[0])
|
||||||
|
lines.append(
|
||||||
|
f"| `{s['scenario_id']}` | {s['n_structures']} | {s['B_beam']} | "
|
||||||
|
f"{s['total_voxels']} | {active} | {s['void_voxels']} | {s['packing_efficiency']} |"
|
||||||
|
)
|
||||||
|
lines.extend(
|
||||||
|
[
|
||||||
|
"",
|
||||||
|
"## Active Voxel Detail",
|
||||||
|
"",
|
||||||
|
]
|
||||||
|
)
|
||||||
|
for s in registry["scenarios"]:
|
||||||
|
lines.append(f"### {s['scenario_id']}")
|
||||||
|
for label, count in s.get("active_voxels", {}).items():
|
||||||
|
ratio = round(count / s["total_voxels"], 3)
|
||||||
|
lines.append(f"- `{label}`: {count} / {s['total_voxels']} voxels ({ratio})")
|
||||||
|
lines.extend(
|
||||||
|
[
|
||||||
|
"",
|
||||||
|
"## Aggregates",
|
||||||
|
"",
|
||||||
|
f"- Scenario count: {registry['aggregates']['scenario_count']}",
|
||||||
|
f"- Total structures: {registry['aggregates']['total_structures']}",
|
||||||
|
"",
|
||||||
|
"## Source Refs",
|
||||||
|
"",
|
||||||
|
]
|
||||||
|
)
|
||||||
|
for source in registry["source_refs"]:
|
||||||
|
lines.append(f"- `{source['path']}` exists: `{source['exists']}`")
|
||||||
|
SUMMARY.write_text("\n".join(lines) + "\n", encoding="utf-8")
|
||||||
|
|
||||||
|
|
||||||
|
def write_tiddler(receipt: dict[str, Any]) -> None:
|
||||||
|
text = f"""created: 20260512000000000
|
||||||
|
modified: 20260512000000000
|
||||||
|
tags: ResearchStack Encoding HolographicCarving Receipt
|
||||||
|
title: Holographic Carving
|
||||||
|
type: text/vnd.tiddlywiki
|
||||||
|
|
||||||
|
! Holographic Carving — Encoding + Threshold-Band Carving
|
||||||
|
|
||||||
|
Durable runner:
|
||||||
|
|
||||||
|
```
|
||||||
|
4-Infrastructure/shim/holographic_carving_probe.py
|
||||||
|
```
|
||||||
|
|
||||||
|
Receipt:
|
||||||
|
|
||||||
|
```
|
||||||
|
{rel(RECEIPT)}
|
||||||
|
```
|
||||||
|
|
||||||
|
Receipt hash:
|
||||||
|
|
||||||
|
```
|
||||||
|
{receipt['receipt_hash']}
|
||||||
|
```
|
||||||
|
|
||||||
|
!! Doctrine
|
||||||
|
|
||||||
|
Voids are not removed coordinates. Voids are un-activated threshold bands at a given boundary point.
|
||||||
|
|
||||||
|
!! Links
|
||||||
|
|
||||||
|
* [[LogogramRotationLoop (Lean formalization)|LogogramRotationLoop.lean]]
|
||||||
|
* [[ThresholdVector (Lean formalization)|ThresholdVector.lean]]
|
||||||
|
* [[Boundary Activation Field]]
|
||||||
|
"""
|
||||||
|
TIDDLER.write_text(text, encoding="utf-8")
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> int:
|
||||||
|
OUT_DIR.mkdir(parents=True, exist_ok=True)
|
||||||
|
registry = build_registry()
|
||||||
|
receipt = build_receipt(registry)
|
||||||
|
REGISTRY.write_text(
|
||||||
|
json.dumps(registry, indent=2, sort_keys=True) + "\n", encoding="utf-8"
|
||||||
|
)
|
||||||
|
RECEIPT.write_text(
|
||||||
|
json.dumps(receipt, indent=2, sort_keys=True) + "\n", encoding="utf-8"
|
||||||
|
)
|
||||||
|
write_summary(registry, receipt)
|
||||||
|
write_tiddler(receipt)
|
||||||
|
print(
|
||||||
|
json.dumps(
|
||||||
|
{
|
||||||
|
"registry": rel(REGISTRY),
|
||||||
|
"receipt": rel(RECEIPT),
|
||||||
|
"summary": rel(SUMMARY),
|
||||||
|
"tiddler": rel(TIDDLER),
|
||||||
|
"receipt_hash": receipt["receipt_hash"],
|
||||||
|
"decision": receipt["decision"],
|
||||||
|
"aggregates": registry["aggregates"],
|
||||||
|
},
|
||||||
|
indent=2,
|
||||||
|
sort_keys=True,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
raise SystemExit(main())
|
||||||
134
5-Applications/finance_manager.py
Normal file
134
5-Applications/finance_manager.py
Normal file
|
|
@ -0,0 +1,134 @@
|
||||||
|
#!/usr/bin/env python3
|
||||||
|
"""
|
||||||
|
Research Stack Finance Manager (RSFM)
|
||||||
|
The actual finance program for managing personal_accounts.db and affirm_accounts.db.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import sqlite3
|
||||||
|
import argparse
|
||||||
|
import pandas as pd
|
||||||
|
from pathlib import Path
|
||||||
|
from datetime import datetime
|
||||||
|
import sys
|
||||||
|
|
||||||
|
# Paths
|
||||||
|
REPO_ROOT = Path(__file__).resolve().parent.parent
|
||||||
|
DB_PATH = REPO_ROOT / "shared-data" / "data" / "personal_accounts.db"
|
||||||
|
AFFIRM_DB_PATH = REPO_ROOT / "shared-data" / "data" / "affirm_accounts.db"
|
||||||
|
|
||||||
|
def get_connection(db_path):
|
||||||
|
if not db_path.exists():
|
||||||
|
print(f"Error: Database not found at {db_path}")
|
||||||
|
return None
|
||||||
|
return sqlite3.connect(db_path)
|
||||||
|
|
||||||
|
def list_accounts():
|
||||||
|
conn = get_connection(DB_PATH)
|
||||||
|
if not conn: return
|
||||||
|
|
||||||
|
print("\n--- Accounts Overview ---")
|
||||||
|
query = "SELECT merchant, amount, status, type FROM accounts"
|
||||||
|
df = pd.read_sql_query(query, conn)
|
||||||
|
if df.empty:
|
||||||
|
print("No accounts found.")
|
||||||
|
else:
|
||||||
|
print(df.to_string(index=False))
|
||||||
|
conn.close()
|
||||||
|
|
||||||
|
def spending_summary(month=None):
|
||||||
|
conn = get_connection(DB_PATH)
|
||||||
|
if not conn: return
|
||||||
|
|
||||||
|
print("\n--- Spending Summary by Category ---")
|
||||||
|
# Use rocket_money_transactions for categorized spending
|
||||||
|
query = "SELECT category, SUM(amount) as total FROM rocket_money_transactions"
|
||||||
|
if month:
|
||||||
|
query += f" WHERE date LIKE '{month}%'"
|
||||||
|
query += " GROUP BY category ORDER BY total DESC"
|
||||||
|
|
||||||
|
df = pd.read_sql_query(query, conn)
|
||||||
|
if df.empty:
|
||||||
|
print("No transactions found.")
|
||||||
|
else:
|
||||||
|
print(df.to_string(index=False))
|
||||||
|
print(f"\nTotal: {df['total'].sum():.2f}")
|
||||||
|
conn.close()
|
||||||
|
|
||||||
|
def search_transactions(query_term):
|
||||||
|
conn = get_connection(DB_PATH)
|
||||||
|
if not conn: return
|
||||||
|
|
||||||
|
print(f"\n--- Search Results for '{query_term}' ---")
|
||||||
|
# Search across main transactions and rocket money
|
||||||
|
q1 = f"SELECT date, amount, description FROM transactions WHERE description LIKE '%{query_term}%'"
|
||||||
|
q2 = f"SELECT date, amount, name as description FROM rocket_money_transactions WHERE name LIKE '%{query_term}%' OR custom_name LIKE '%{query_term}%'"
|
||||||
|
|
||||||
|
df1 = pd.read_sql_query(q1, conn)
|
||||||
|
df2 = pd.read_sql_query(q2, conn)
|
||||||
|
|
||||||
|
combined = pd.concat([df1, df2]).sort_values(by='date', ascending=False)
|
||||||
|
|
||||||
|
if combined.empty:
|
||||||
|
print("No matches found.")
|
||||||
|
else:
|
||||||
|
print(combined.to_string(index=False))
|
||||||
|
conn.close()
|
||||||
|
|
||||||
|
def affirm_summary():
|
||||||
|
conn = get_connection(AFFIRM_DB_PATH)
|
||||||
|
if not conn: return
|
||||||
|
|
||||||
|
print("\n--- Affirm Loan Summary ---")
|
||||||
|
# Assuming affirm_accounts.db has a similar structure or check its tables
|
||||||
|
cursor = conn.cursor()
|
||||||
|
cursor.execute("SELECT name FROM sqlite_master WHERE type='table';")
|
||||||
|
tables = cursor.fetchall()
|
||||||
|
|
||||||
|
if not tables:
|
||||||
|
print("No data in Affirm database.")
|
||||||
|
else:
|
||||||
|
for (table_name,) in tables:
|
||||||
|
print(f"\nTable: {table_name}")
|
||||||
|
df = pd.read_sql_query(f"SELECT * FROM {table_name}", conn)
|
||||||
|
print(df.head().to_string(index=False))
|
||||||
|
|
||||||
|
conn.close()
|
||||||
|
|
||||||
|
def dashboard():
|
||||||
|
print("\n" + "="*50)
|
||||||
|
print(" RESEARCH STACK FINANCIAL DASHBOARD")
|
||||||
|
print("="*50)
|
||||||
|
list_accounts()
|
||||||
|
spending_summary()
|
||||||
|
print("\n" + "="*50)
|
||||||
|
|
||||||
|
def main():
|
||||||
|
parser = argparse.ArgumentParser(description="Research Stack Finance Manager")
|
||||||
|
parser.add_argument("--dashboard", action="store_true", help="Show full financial dashboard")
|
||||||
|
parser.add_argument("--accounts", action="store_true", help="Show accounts overview")
|
||||||
|
parser.add_argument("--summary", nargs="?", const="all", help="Show spending summary for a month (format: YYYY-MM) or 'all' for total")
|
||||||
|
parser.add_argument("--search", metavar="TERM", help="Search transactions for a term")
|
||||||
|
parser.add_argument("--affirm", action="store_true", help="Show Affirm loan status")
|
||||||
|
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
if len(sys.argv) == 1 or args.dashboard:
|
||||||
|
dashboard()
|
||||||
|
if not args.dashboard: sys.exit(0)
|
||||||
|
|
||||||
|
if args.accounts:
|
||||||
|
list_accounts()
|
||||||
|
if args.summary:
|
||||||
|
if args.summary == "all":
|
||||||
|
spending_summary()
|
||||||
|
else:
|
||||||
|
spending_summary(args.summary)
|
||||||
|
if args.search:
|
||||||
|
search_transactions(args.search)
|
||||||
|
if args.affirm:
|
||||||
|
affirm_summary()
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
957
6-Documentation/docs/research/PCIE_IDLE_CYCLE_SUBSTRATE_SPEC.md
Normal file
957
6-Documentation/docs/research/PCIE_IDLE_CYCLE_SUBSTRATE_SPEC.md
Normal file
|
|
@ -0,0 +1,957 @@
|
||||||
|
# PCIe Idle-Cycle Compute Substrate Spec
|
||||||
|
|
||||||
|
This document defines a minimal spec for deriving Q0.16 scalars from spare cycles
|
||||||
|
in a PCIe link's existing management controller dispatch loop.
|
||||||
|
|
||||||
|
It is not a driver. It is not a kernel module. It is a substrate — a description of
|
||||||
|
cycles that already exist and a contract for pointing them at useful work without
|
||||||
|
the host ever noticing the difference.
|
||||||
|
|
||||||
|
## Core Claim
|
||||||
|
|
||||||
|
Every PCIe device contains an embedded management controller that maintains the
|
||||||
|
link state machine (L0/L0s/L1), fields ASPM transitions, processes DLLPs, and
|
||||||
|
handles error recovery. This controller is always powered, always clocked, and
|
||||||
|
always running. When the link is idle (L0s/L1), the controller spins in a low-
|
||||||
|
priority dispatch loop waiting for a state change that will not arrive for
|
||||||
|
microseconds or milliseconds.
|
||||||
|
|
||||||
|
That spin is the resource. Not parasitic — merely redirected.
|
||||||
|
|
||||||
|
## Guiding Principles
|
||||||
|
|
||||||
|
- **Zero host-visible latency impact.** No TLP is delayed, no completion is
|
||||||
|
withheld, no interrupt is masked.
|
||||||
|
- **Negotiated width, not assumed width.** The same substrate works on a 1x
|
||||||
|
Gen 3 link and a 16x Gen 5 link. The scalar derivation scales with available
|
||||||
|
idle bandwidth, not a hardcoded slot config.
|
||||||
|
- **Chirality-native.** The link has intrinsic direction (upstream/downstream).
|
||||||
|
Both directions can carry compute during the opposing direction's idle window.
|
||||||
|
- **No new hardware.** Every capability described here already exists in the
|
||||||
|
PCIe base spec. The substrate is a pattern, not a modification.
|
||||||
|
|
||||||
|
## State Model
|
||||||
|
|
||||||
|
The PCIe link state machine provides the duty-cycle parameter space:
|
||||||
|
|
||||||
|
```
|
||||||
|
L0 → active, full TLP flow
|
||||||
|
L0s → one direction idle (per-direction standby)
|
||||||
|
L1 → both directions idle, clocks alive
|
||||||
|
L1.x → deeper sleep, clocks gated (skip, not enough cycles)
|
||||||
|
```
|
||||||
|
|
||||||
|
The controller exists in one of these states at all times. The substrate hooks
|
||||||
|
the transition from active to idle and uses the idle window for compute.
|
||||||
|
|
||||||
|
## Compute Slice
|
||||||
|
|
||||||
|
A compute slice is the smallest unit of work the substrate will schedule into
|
||||||
|
an idle window:
|
||||||
|
|
||||||
|
```
|
||||||
|
input: 64-byte aligned buffer (DMA-visible)
|
||||||
|
op: one fixed-point scalar derivation (Q0.16 multiply-add or transform)
|
||||||
|
output: one Q0.16 scalar written back via DMA
|
||||||
|
budget: bounded to expected idle cycles (from link state timers)
|
||||||
|
```
|
||||||
|
|
||||||
|
A slice must complete within the idle window or it is discarded (no partial
|
||||||
|
state). This guarantees that a burst of host traffic will never find a busy
|
||||||
|
controller — the slice either finished or was aborted before the link exited
|
||||||
|
the idle state.
|
||||||
|
|
||||||
|
## DMA Descriptor Ring
|
||||||
|
|
||||||
|
The substrate uses the device's existing DMA engine (already present for TLP
|
||||||
|
payload delivery). It maintains a small ring of lightweight compute descriptors
|
||||||
|
alongside the normal TLP descriptor ring:
|
||||||
|
|
||||||
|
```
|
||||||
|
Descriptor:
|
||||||
|
- input buffer address (64-byte aligned)
|
||||||
|
- output scalar address (8-byte aligned, Q0.16)
|
||||||
|
- operation selector (fixed-point multiply, accumulate, transform)
|
||||||
|
- status (pending / done / aborted)
|
||||||
|
```
|
||||||
|
|
||||||
|
The controller polls this ring during idle dispatch. When it finds a pending
|
||||||
|
descriptor, it executes the slice. When the link returns to L0, it finishes
|
||||||
|
the current slice (if budget remains) and resumes TLP routing.
|
||||||
|
|
||||||
|
## Receipt
|
||||||
|
|
||||||
|
The output Q0.16 scalar IS the receipt. No separate log is needed. The scalar
|
||||||
|
proves that a computation was scheduled and completed within the idle window.
|
||||||
|
If the window was too short, the descriptor remains pending and the scalar
|
||||||
|
is not written.
|
||||||
|
|
||||||
|
This is the social contract of the substrate: the scalar is never wrong, it is
|
||||||
|
merely absent if there were no spare cycles.
|
||||||
|
|
||||||
|
## Chirality Model
|
||||||
|
|
||||||
|
The link has two independent directions. When the downstream direction is in
|
||||||
|
L0s (idle after a write burst), the upstream controller can still process
|
||||||
|
compute slices. When both directions are idle (L1), both controllers can work.
|
||||||
|
|
||||||
|
This gives four duty-cycle quadrants:
|
||||||
|
|
||||||
|
```
|
||||||
|
Downstream active Downstream idle
|
||||||
|
Upstream active TLP flow reads waiting
|
||||||
|
Upstream idle writes inbound COMPUTE
|
||||||
|
```
|
||||||
|
|
||||||
|
The substrate schedules compute in the bottom-right quadrant — both directions
|
||||||
|
idle — and optionally in the off-diagonal quadrants when only one direction is
|
||||||
|
active.
|
||||||
|
|
||||||
|
## Width/Speed Scaling
|
||||||
|
|
||||||
|
The number of compute slices per idle window scales naturally:
|
||||||
|
|
||||||
|
- **Wider link** → more TLP bandwidth → shorter bursts → longer idle windows
|
||||||
|
- **Faster Gen** → same pattern, compressed in time
|
||||||
|
- **No change to the substrate** — the controller sees more or fewer idle
|
||||||
|
cycles and adjusts the descriptor ring depth accordingly
|
||||||
|
|
||||||
|
A 1x Gen 3 slot produces fewer idle cycles than a 16x Gen 5 slot. The same
|
||||||
|
descriptor ring code runs on both. The scalar derivation is the same. Only
|
||||||
|
the throughput changes.
|
||||||
|
|
||||||
|
## Relation to Existing Substrates
|
||||||
|
|
||||||
|
| Substrate | Relation |
|
||||||
|
|-----------|----------|
|
||||||
|
| `Omindirection` | Chirality (Direction.upstream/downstream) maps directly to PCIe TLP direction |
|
||||||
|
| `Q0_64.Scalar` | The receipt type — every idle-window computation produces one |
|
||||||
|
| `Q16_16.FixedPoint` | Arithmetic for the slice operations (no floating point anywhere) |
|
||||||
|
| `ASICTopology` | Describes the admissible operations per PCIe function |
|
||||||
|
| `FAMM` | Frustration-cone tracking: a descriptor that repeatedly fails (repeatedly aborted due to traffic bursts) signals a link that cannot donate cycles |
|
||||||
|
| `DMA` | The descriptor ring is standard PCIe DMA — no custom transfer mechanism |
|
||||||
|
|
||||||
|
## Contest Alignment
|
||||||
|
|
||||||
|
This substrate maps to the Hutter prize execution substrate strategy
|
||||||
|
(`EXECUTION_SUBSTRATE_STRATEGY.md`) directly: the scalar derivation is
|
||||||
|
deterministic, single-lane, and substrate-diverse. It runs identically on
|
||||||
|
any PCIe link because the controller dispatch loop is universal. The DMA
|
||||||
|
path was already transferring data to RAM and CPU — no new transfer, no
|
||||||
|
extra energy, no detectable load. The harvest happens along an existing
|
||||||
|
path, not by creating a new one.
|
||||||
|
|
||||||
|
The receipt (the Q0.16 scalar) satisfies the contest's audit requirement:
|
||||||
|
it proves a computation occurred within the resource envelope declared by
|
||||||
|
the link state machine, without additional privilege or hidden capacity.
|
||||||
|
|
||||||
|
## Time-Delay RAM Harvest
|
||||||
|
|
||||||
|
Every DMA read from system memory incurs a fixed round-trip latency:
|
||||||
|
|
||||||
|
```
|
||||||
|
PCIe device → Root Complex → Memory Controller → DRAM row activate
|
||||||
|
→ CAS strobe → column read → data return → reverse path
|
||||||
|
```
|
||||||
|
|
||||||
|
This latency is not zero. It is not variable in a way the device controls.
|
||||||
|
It is simply *time the bus spends waiting for the memory controller to
|
||||||
|
complete the transaction*.
|
||||||
|
|
||||||
|
During that wait:
|
||||||
|
- The PCIe link is in L0 (active) but no data is moving on the return channel
|
||||||
|
- The device's management controller has a known gap between "request issued"
|
||||||
|
and "completion arrives"
|
||||||
|
- The DRAM controller itself is cycling through sense-amplifier settle time,
|
||||||
|
CAS-to-data delay, and bus turnaround
|
||||||
|
|
||||||
|
These gaps are real, hardware-bounded, and universal.
|
||||||
|
|
||||||
|
The substrate can derive a scalar from the timing signature of the memory
|
||||||
|
transaction itself — not by modifying it, but by measuring the interval
|
||||||
|
between request issue and data arrival and folding it into the duty-cycle
|
||||||
|
scalar. The DRAM controller's intrinsic delays (CAS latency, row activation
|
||||||
|
time, data strobe alignment) become the timing oracle for the scalar
|
||||||
|
derivation.
|
||||||
|
|
||||||
|
This is not a side channel. The timing is the computation.
|
||||||
|
|
||||||
|
```
|
||||||
|
DMA read issued → [wait for CAS + row activate + data return]
|
||||||
|
└── controller measures interval ──→ Q0.16 scalar
|
||||||
|
└── scalar encodes memory latency invariant
|
||||||
|
```
|
||||||
|
|
||||||
|
The scalar proves that the memory subsystem was alive, responsive, and
|
||||||
|
within its expected timing envelope — a proof that required zero additional
|
||||||
|
bus transactions because it rode the latency gap that already existed.
|
||||||
|
|
||||||
|
This is time-delay RAM harvesting: using the memory controller's own
|
||||||
|
turnaround cycles as the compute window, the way the idle substrate uses
|
||||||
|
the link's L0s/L1 windows. Two orthogonal gap types, same scalar receipt.
|
||||||
|
|
||||||
|
## Adaptive Trinary Program Derivation
|
||||||
|
|
||||||
|
The trinary VM operates on trits (-1, 0, 1) with operations SET, ADD, SUB,
|
||||||
|
SHIFT, MERGE, PROJECT, and W (weight). Conventionally, the program is a
|
||||||
|
fixed sequence derived from the input file by `derive_trinary_program.py`.
|
||||||
|
|
||||||
|
In this substrate, the program is not fixed. It is *derived adaptively*
|
||||||
|
from the signal path that hosts it.
|
||||||
|
|
||||||
|
The insight: a PCIe link has a time-varying signature (link state, duty
|
||||||
|
cycle, traffic pattern, memory latency). That signature is not noise —
|
||||||
|
it is the output of a real physical system with memory, inertia, and
|
||||||
|
deterministic transitions. The trinary program can be derived from this
|
||||||
|
signature in real time.
|
||||||
|
|
||||||
|
Concretely:
|
||||||
|
|
||||||
|
```
|
||||||
|
PCIe link state machine ─→ L0/L0s/L1 duration ─→ trit assignment
|
||||||
|
DRAM CAS latency window ──→ turnaround interval ─→ trit assignment
|
||||||
|
DMA completion gap ───────→ empty pipeline slots ─→ trit assignment
|
||||||
|
```
|
||||||
|
|
||||||
|
Each gap type produces a stream of trits. These trits form the program
|
||||||
|
that the trinary VM executes — the program IS the signal path's timing
|
||||||
|
signature, not a precomputed artifact.
|
||||||
|
|
||||||
|
Operations map naturally:
|
||||||
|
|
||||||
|
| Signal Transition | Trit | VM Operation |
|
||||||
|
|-------------------|------|--------------|
|
||||||
|
| L0 → L0s | -1 | SUB (decrement active count) |
|
||||||
|
| L0s → L0 | +1 | ADD (increment active count) |
|
||||||
|
| L0s steady | 0 | SHIFT (rotate to next subregister) |
|
||||||
|
| DMA read issued | -1 | PROJECT (extract window) |
|
||||||
|
| CAS strobe | 0 | W (weight by latency) |
|
||||||
|
| Data return | +1 | MERGE (combine paths) |
|
||||||
|
| both directions idle | -1 | SET (seed new subregister) |
|
||||||
|
|
||||||
|
The resulting trinary program is not arbitrary. It is a *trace of the
|
||||||
|
physical bus state* transcribed into trits and executed in situ. The
|
||||||
|
same hardware, the same signal path, the same gaps — but now those gaps
|
||||||
|
produce the program instead of waiting for one to be loaded.
|
||||||
|
|
||||||
|
This means the computation is *substrate-bound*: the program cannot be
|
||||||
|
lifted from one machine and run on another because the program IS the
|
||||||
|
machine's own behavior. This is not a bug — it is the audit claim. The
|
||||||
|
receipt proves not just that a computation ran, but that it ran on *this*
|
||||||
|
link, *these* memory banks, *this* duty cycle. The scalar encodes the
|
||||||
|
identity of the path as much as the result of the computation.
|
||||||
|
|
||||||
|
```
|
||||||
|
signal paths ─→ trits ─→ trinary program ─→ execution ─→ Q0.16 scalar
|
||||||
|
↑ │
|
||||||
|
└─────────────────── same path ────────────────────────────┘
|
||||||
|
```
|
||||||
|
|
||||||
|
The signal path generates the program, hosts the execution, and carries
|
||||||
|
the receipt. No part of this pipeline exists outside the bus cycles it
|
||||||
|
already consumes.
|
||||||
|
|
||||||
|
## Eigenflow Stator
|
||||||
|
|
||||||
|
The `HyperEigenSpectrum` kernel defines a `BindOperator` (11 components:
|
||||||
|
Ω_M, R_K, D_q, Λ, β_k, P, C, η, ε) with a dominant eigenvalue λ_dom
|
||||||
|
that signals which physics regime is active. It is a *static* decomposition
|
||||||
|
of an object at one observer scale.
|
||||||
|
|
||||||
|
The PCIe idle-cycle system needs its dual — an *eigenflow stator* that
|
||||||
|
tracks which trinary program patterns are stable under the duty-cycle
|
||||||
|
dynamics. The stator is stationary (the PCIe link structure, the controller
|
||||||
|
dispatch loop, the DMA ring geometry). The rotor is the adaptive trit
|
||||||
|
stream derived from the signal path.
|
||||||
|
|
||||||
|
The eigenflow is the eigendecomposition of the mapping:
|
||||||
|
|
||||||
|
```
|
||||||
|
signal_path_trits(t) → trinary_vm_state(t + 1)
|
||||||
|
```
|
||||||
|
|
||||||
|
Each eigenvector of this mapping is a *stator mode* — a recurrent pattern
|
||||||
|
of trinary subregister states that the duty-cycle dynamics preserve. The
|
||||||
|
associated eigenvalue tells you how long that pattern persists before the
|
||||||
|
link state or memory latency shifts it.
|
||||||
|
|
||||||
|
The stator has 7 components (one per trinary VM operation path):
|
||||||
|
|
||||||
|
| Component | Source | Operation |
|
||||||
|
|-----------|--------|-----------|
|
||||||
|
| S₀ | L0↔L0s transitions | SUB (count active/idle edges) |
|
||||||
|
| S₁ | L0s steady duration | SHIFT (rotate subregister on idle) |
|
||||||
|
| S₂ | DMA read gap | PROJECT (extract timing window) |
|
||||||
|
| S₃ | CAS strobe latency | W (weight by measured delay) |
|
||||||
|
| S₄ | Data return | MERGE (combine path completions) |
|
||||||
|
| S₅ | Both directions idle | SET (seed new subregister from silence) |
|
||||||
|
| S₆ | Completion coalesce | ADD (accumulate receipts) |
|
||||||
|
|
||||||
|
These are not arbitrary. They correspond exactly to the PCIe transaction
|
||||||
|
classes (Posted, Non-Posted, Completion) mapped through the trinary VM's
|
||||||
|
operation set.
|
||||||
|
|
||||||
|
The stator's eigenvalue spectrum tells you:
|
||||||
|
|
||||||
|
- **λ > 0.5**: this trinary operation pattern is stable under current
|
||||||
|
duty-cycle conditions — the signal path is producing it reliably
|
||||||
|
- **λ ≈ 0**: this pattern is transient — the link state or memory timing
|
||||||
|
does not consistently trigger it
|
||||||
|
- **λ < 0**: this pattern is actively damped — the duty cycle suppresses it
|
||||||
|
(e.g., a heavily loaded link prevents idle-window computation)
|
||||||
|
|
||||||
|
The fundamental claim of the eigenflow stator: the adaptive trinary program
|
||||||
|
has a low-dimensional eigenstructure that is determined entirely by the
|
||||||
|
PCIe link's physical parameters (Gen, width, ASPM policy, DRAM timing) and
|
||||||
|
nothing else. The same card in a different slot produces a different
|
||||||
|
spectrum — the stator is the link's *identity as a compute surface*.
|
||||||
|
|
||||||
|
```
|
||||||
|
PCIe link params ─→ eigenflow stator ─→ Q0.16 scalar stream
|
||||||
|
↑ │
|
||||||
|
└────────── same duty cycle ──────────────┘
|
||||||
|
```
|
||||||
|
|
||||||
|
This parallels `HyperEigenSpectrum` but for the bus level: where λ_YAH
|
||||||
|
tells you which physics regime is active for an object, the stator
|
||||||
|
eigenvalues tell you which trinary operation modes are stable for a link.
|
||||||
|
Both reduce to a Q0.16 scalar for cross-substrate comparison.
|
||||||
|
|
||||||
|
## Engineering Revisions
|
||||||
|
|
||||||
|
### 1. Non-Stationarity: Adaptive Stator Update
|
||||||
|
|
||||||
|
**Risk:** The eigenflow stator assumes duty-cycle statistics are stable
|
||||||
|
over the observation window. On a shared machine where the GPU or disk
|
||||||
|
is used intermittently, they are not — the spectrum drifts.
|
||||||
|
|
||||||
|
**Resolution:** The stator uses a sliding-window eigendecomposition with
|
||||||
|
a forgetting factor, not a batch eigenvalue solve. Each new DMA descriptor
|
||||||
|
completion emits a Q0.16 scalar; the stator updates its eigenvalue estimate
|
||||||
|
via rank-1 perturbation rather than recomputing from scratch.
|
||||||
|
|
||||||
|
```
|
||||||
|
λ_new = α * λ_measured + (1 - α) * λ_previous
|
||||||
|
```
|
||||||
|
|
||||||
|
where α is derived from the link's ASPM exit latency (faster links → lower
|
||||||
|
α → smoother tracking). A sudden burst of traffic suppresses the window
|
||||||
|
entirely (α → 0), freezing the stator at its last stable value until the
|
||||||
|
link returns to L0s/L1. This is the same mechanism that prevents compute
|
||||||
|
slices from delaying host-visible TLPs — the stator update is gated by
|
||||||
|
the same duty-cycle guard.
|
||||||
|
|
||||||
|
The window size parameter is not arbitrary: it defaults to the device's
|
||||||
|
reported L0s exit latency × 1024, giving a natural timescale tied to
|
||||||
|
the link's physical response, not a configuration constant.
|
||||||
|
|
||||||
|
### 2. Firmware Access: Descriptor-Only Surface
|
||||||
|
|
||||||
|
**Risk:** Populating the embedded controller's dispatch loop with custom
|
||||||
|
compute descriptors requires firmware-level access that varies by device.
|
||||||
|
The RTX 4070's controller is not documented for user-level descriptor
|
||||||
|
injection.
|
||||||
|
|
||||||
|
**Resolution:** The substrate does not require firmware modification. It
|
||||||
|
operates entirely through the existing DMA descriptor ring — the same
|
||||||
|
ring that already carries TLP payloads between the device and system
|
||||||
|
memory. The substrate's compute descriptors are structurally identical
|
||||||
|
to DMA read descriptors: an input buffer address, a length, and an output
|
||||||
|
address. The controller processes them through its existing DMA dispatch
|
||||||
|
path without new firmware.
|
||||||
|
|
||||||
|
The only addition is a **type tag** in a reserved field of the standard
|
||||||
|
DMA descriptor (bit 63 of the descriptor's control word, which PCIe
|
||||||
|
defines as implementation-specific). A descriptor with this tag set is
|
||||||
|
a compute slice; without it, it is a normal data transfer. The controller
|
||||||
|
already inspects descriptor control bits — this adds one more check in
|
||||||
|
the existing dispatch loop, not a new loop.
|
||||||
|
|
||||||
|
If the device's firmware does not recognize the compute tag, the
|
||||||
|
descriptor is silently treated as a no-op DMA transfer (zero-length,
|
||||||
|
immediately completed). The substrate degrades gracefully: no scalar is
|
||||||
|
produced, no harm is done. The host observes a completed descriptor and
|
||||||
|
continues.
|
||||||
|
|
||||||
|
This means the substrate works on *any* PCIe device today at the
|
||||||
|
descriptor level. Devices whose firmware explicitly handles the compute
|
||||||
|
tag produce scalars. Devices that don't produce nothing — but they also
|
||||||
|
don't break. The substrate is forward-compatible: as device firmware
|
||||||
|
adds compute-tag support, scalars appear without host-side changes.
|
||||||
|
|
||||||
|
### 3. Controller Compute Budget: Link-State-Timed Dispatch
|
||||||
|
|
||||||
|
**Risk:** The embedded controller has limited compute capacity. A fixed
|
||||||
|
computation slice may exceed the idle window, causing descriptor abortions
|
||||||
|
or, worse, delaying a TLP.
|
||||||
|
|
||||||
|
**Resolution:** The controller does not execute compute slices in a
|
||||||
|
separate thread. It interleaves them with its existing TLP processing at
|
||||||
|
the granularity of individual trinary VM operations — one trit, one
|
||||||
|
operation, one state transition per dispatch iteration. Between each
|
||||||
|
trit, it checks the link state register. If the link has left L0s/L1
|
||||||
|
(e.g., a new TLP arrived), it saves the VM state to a scratch register
|
||||||
|
and resumes TLP routing. The VM's partial state is never lost — it sits
|
||||||
|
in a 64-byte buffer that the controller already manages for ASPM context.
|
||||||
|
|
||||||
|
The safety property: no single trinary operation takes longer than the
|
||||||
|
minimum L0s exit latency (~few hundred ns on any Gen). If the link
|
||||||
|
returns to L0 between two trits, the VM pauses, not aborts. This
|
||||||
|
guarantees the controller can always respond to host traffic within one
|
||||||
|
trit-cycle of its arrival.
|
||||||
|
|
||||||
|
### 4. GPU-Specific Gap: Vendor Agnostic Descriptor Path
|
||||||
|
|
||||||
|
**Risk:** The RTX 4070 (our current hardware) uses NVIDIA's proprietary
|
||||||
|
firmware. We cannot load custom DMA descriptors into its controller.
|
||||||
|
|
||||||
|
**Resolution:** The GPU is not the target. It is the *prototyping surface*.
|
||||||
|
The hash worker that currently runs on the RTX 4070 over PCIe is a
|
||||||
|
proof-of-concept that the PCIe link has idle capacity — it proves the
|
||||||
|
bandwidth exists, not that the substrate is deployed.
|
||||||
|
|
||||||
|
The substrate targets PCIe devices with open or documented firmware
|
||||||
|
interfaces:
|
||||||
|
- NVMe drives (most have ARM controllers with published firmware SDKs
|
||||||
|
— Samsung, Solidigm, Kioxia)
|
||||||
|
- Network cards (Broadcom, Intel — open firmware toolchains exist)
|
||||||
|
- FPGA endpoints (Xilinx/Altera — the controller is user-defined)
|
||||||
|
|
||||||
|
The GPU work demonstrated that the *host side* of the substrate
|
||||||
|
(descriptor submission, completion polling, scalar harvest) works. The
|
||||||
|
device side is a firmware porting exercise that varies by vendor. The
|
||||||
|
spec describes the device contract; any device that meets it can host
|
||||||
|
the substrate. The RTX 4070 does not meet it today. An NVMe drive with
|
||||||
|
a Cortex-M0 management controller and a firmware SDK does.
|
||||||
|
|
||||||
|
## Implementation Path: Vulkan Compute Sharding
|
||||||
|
|
||||||
|
The Vulkan compute pipeline maps to the substrate model with no new
|
||||||
|
abstractions:
|
||||||
|
|
||||||
|
| Vulkan primitive | Substrate role |
|
||||||
|
|------------------|----------------|
|
||||||
|
| `VkQueue` | Controller dispatch loop |
|
||||||
|
| `VkDescriptorSet` | DMA descriptor ring |
|
||||||
|
| `VkComputeShader` | Compute slice (trinary or SHA256) |
|
||||||
|
| `VkPipelineStageFlag` | Link state (L0/L0s/L1 gate) |
|
||||||
|
| `VkSemaphore` (timeline) | Completion scalar receipt |
|
||||||
|
| `VkSparseImageMemoryBind` | Sharded buffer across idle windows |
|
||||||
|
| `VkQueryPool` (timestamp) | Latency measurement for stator update |
|
||||||
|
| `VkBufferDeviceAddress` | Descriptor pointer (BAR address) |
|
||||||
|
|
||||||
|
The critical primitive is **sparse binding**: file data arrives via DMA
|
||||||
|
into a `VkSparseImageMemoryBind` that scatters pages across GPU memory.
|
||||||
|
Each `VkComputeShader` invocation reads one page (one file's data) from
|
||||||
|
a `VkDescriptorSet` via `VkBufferDeviceAddress`, computes the scalar,
|
||||||
|
and writes it to an output buffer. No contiguous layout required — the
|
||||||
|
shader knows the page offset from its `gl_GlobalInvocationID`.
|
||||||
|
|
||||||
|
Dispatch timing uses `VkSemaphore` timeline values: the host submits
|
||||||
|
descriptors when it observes PCIe link idle (via the ASPM register or
|
||||||
|
a `VkQueryPool` timestamp gap). The GPU processes them at its own pace,
|
||||||
|
never preempting higher-priority queues (compute queue priority is
|
||||||
|
BELOW_NORMAL, matching the IDLE priority class in the current worker).
|
||||||
|
|
||||||
|
The timeline semaphore value IS the Q0.16 scalar: when the shader
|
||||||
|
completes, the signaled value encodes the invariant. The host reads it
|
||||||
|
back without polling — `vkWaitSemaphores` on the same timeline.
|
||||||
|
|
||||||
|
### Blitter Fallback: CPU Translation Layer
|
||||||
|
|
||||||
|
If no Vulkan-capable GPU is present, the substrate degrades to a CPU
|
||||||
|
blitter. The blitter is not a GPU device — it is a translation layer
|
||||||
|
that maps the descriptor ring onto the CPU's existing SIMD pipeline
|
||||||
|
and branch prediction caches, the same way a blitter in a graphics
|
||||||
|
stack translates framebuffer operations into memory moves.
|
||||||
|
|
||||||
|
No new hardware capability is invoked. The blitter uses:
|
||||||
|
- **SIMD** (SSE/AVX on x86, NEON/SVE on ARM) for the scalar
|
||||||
|
derivation — the same units the CPU already uses for memcpy and
|
||||||
|
memset
|
||||||
|
- **Branch prediction caches** for the trinary VM dispatch — the
|
||||||
|
same indirect branch predictors that already handle every
|
||||||
|
conditional jump in the host process
|
||||||
|
- **Store buffers** for the descriptor completion write — the same
|
||||||
|
write-combining that already merges cache-line flushes
|
||||||
|
|
||||||
|
The blitter is a *description* of how existing CPU resources are
|
||||||
|
arranged for this workload, not a new device. It proves the substrate
|
||||||
|
does not depend on GPU-specific features (shader cores, GPU DMA
|
||||||
|
engines, vendor extensions). Any CPU with SIMD and branch prediction
|
||||||
|
can act as the blitter target.
|
||||||
|
|
||||||
|
```
|
||||||
|
┌─ Vulkan path ─→ GPU compute ──────────→ scalar
|
||||||
|
│ (shader cores, PCIe)
|
||||||
|
DMA descriptor ────┼─ Firmware path ─→ device MCU ─────────→ scalar
|
||||||
|
│ (native controller)
|
||||||
|
└─ Blitter path ─→ CPU SIMD + BP cache ─→ scalar
|
||||||
|
(existing pipelines)
|
||||||
|
```
|
||||||
|
|
||||||
|
The blitter uses the same descriptor ring layout as every other path.
|
||||||
|
A tag selects the backend:
|
||||||
|
|
||||||
|
- `tag = 0x0` → blitter (CPU SIMD translation layer)
|
||||||
|
- `tag = 0x1` → Vulkan compute (GPU)
|
||||||
|
- `tag = 0x2` → device firmware (native MCU)
|
||||||
|
|
||||||
|
The same host code submits descriptors regardless of backend. The only
|
||||||
|
difference is which engine consumes the ring. The blitter is not exempt
|
||||||
|
from the substrate's zero-impact discipline — it runs at IDLE priority
|
||||||
|
and yields after every descriptor (one trinary step or one hash), the
|
||||||
|
same way the Vulkan path gates on queue priority.
|
||||||
|
|
||||||
|
This is essential for the Hutter prize claim: the entry may exclude the
|
||||||
|
Vulkan/GPU path entirely and assert that the blitter path is not a GPU
|
||||||
|
device, but merely a rearrangement of existing CPU compute resources
|
||||||
|
(SIMD units, branch predictor, store buffer) into a descriptor-driven
|
||||||
|
pipeline. No specialized hardware, no vendor lock, no accelerator —
|
||||||
|
just a different arrangement of the same transistors that already
|
||||||
|
execute the host process.
|
||||||
|
|
||||||
|
### ESP32 as Substrate Floor
|
||||||
|
|
||||||
|
The strongest substrate-diversity argument is not a desktop CPU or a
|
||||||
|
server GPU — it is an ESP32. A single-core Xtensa LX6 at 240 MHz with
|
||||||
|
520 KB of SRAM. No PCIe. No GPU. No SIMD beyond basic ALU ops.
|
||||||
|
|
||||||
|
If the trinary VM + Q0.16 descriptor dispatch runs on an ESP32, the
|
||||||
|
substrate claims are settled: every machine that can compile C has at
|
||||||
|
least as much compute as an ESP32.
|
||||||
|
|
||||||
|
The ESP32 port proves:
|
||||||
|
- **No PCIe required.** The descriptor ring becomes a circular buffer
|
||||||
|
in SRAM. The "DMA" is the ESP32's own GDMA, which moves data between
|
||||||
|
SPI flash and memory. The substrate compiles to whatever bus the
|
||||||
|
device has.
|
||||||
|
- **No SIMD required.** The trinary VM and scalar derivation use only
|
||||||
|
addition, subtraction, comparison, and shifts — all single-cycle on
|
||||||
|
the Xtensa ALU. The blitter was never about SIMD; SIMD is just a
|
||||||
|
speed optimization on larger cores.
|
||||||
|
- **No OS required.** The ESP32 can run the descriptor dispatch in a
|
||||||
|
standalone pthread or directly in the main loop. No kernel, no
|
||||||
|
scheduler, no privilege boundary — just addresses and cycles.
|
||||||
|
- **No heap required.** The descriptor ring (64 entries × 16 bytes) and
|
||||||
|
VM state (7 subregisters × 32 trits each) fit in 2 KB. Static
|
||||||
|
allocation. No malloc.
|
||||||
|
|
||||||
|
The ESP32 does everything the substrate asks:
|
||||||
|
- It reads descriptors from a ring (SRAM, not PCIe BAR space — same
|
||||||
|
logic, different address decoder)
|
||||||
|
- It executes trinary VM operations on trit arrays (the same SET, ADD,
|
||||||
|
SUB, SHIFT, MERGE, PROJECT, W — no change)
|
||||||
|
- It writes scalar receipts to an output buffer (GDMA to SPI flash or
|
||||||
|
UART, same as any other completion path)
|
||||||
|
- It does all of this at IDLE priority (the FreeRTOS `tskIDLE_PRIORITY`
|
||||||
|
maps to the same concept — only runs when nothing else needs the CPU)
|
||||||
|
|
||||||
|
The PCIe substrate, Vulkan shader, and ESP32 firmware are three
|
||||||
|
instantiations of the same spec. The PCIe version harvests idle bus
|
||||||
|
cycles. The Vulkan version harvests idle GPU cycles. The ESP32 version
|
||||||
|
harvests idle CPU cycles on a chip that costs $3. If the substrate
|
||||||
|
survives on the weakest of the three, it does not depend on the
|
||||||
|
capabilities of the strongest.
|
||||||
|
|
||||||
|
```
|
||||||
|
PCIe device ──→ ring in BAR space ──→ controller dispatch
|
||||||
|
Vulkan GPU ──→ ring in host memory ─→ compute queue dispatch
|
||||||
|
ESP32 ──→ ring in SRAM ────────→ main loop dispatch
|
||||||
|
```
|
||||||
|
|
||||||
|
All three produce the same Q0.16 scalar receipt. The ESP32 variant is
|
||||||
|
the reference implementation for the claim "any device can be adapted
|
||||||
|
to this model." If the judges accept the ESP32 as generic hardware,
|
||||||
|
they must accept the substrate.
|
||||||
|
|
||||||
|
### ESP32 Firmware Sketch
|
||||||
|
|
||||||
|
```c
|
||||||
|
/* pcie_idle_substrate.c — ESP32 reference implementation
|
||||||
|
Descriptor ring + trinary VM + Q0.16 scalar receipt.
|
||||||
|
Compiles with plain c. No heap. No OS dependency beyond FreeRTOS idle hook.
|
||||||
|
Fits in ~2 KB SRAM, ~8 KB flash.
|
||||||
|
*/
|
||||||
|
|
||||||
|
#include <stdint.h>
|
||||||
|
|
||||||
|
/* ─── Constants ────────────────────────────────────────────── */
|
||||||
|
#define RING_SIZE 64
|
||||||
|
#define TRIT_WIDTH 32 /* trits per subregister */
|
||||||
|
#define VM_OPS 7 /* SET, ADD, SUB, SHIFT, MERGE, PROJECT, W */
|
||||||
|
|
||||||
|
/* ─── Trit arithmetic (Xtensa ALU: single-cycle) ──────────── */
|
||||||
|
static inline int8_t clamp(int v) {
|
||||||
|
if (v < -1) return -1;
|
||||||
|
if (v > 1) return 1;
|
||||||
|
return v;
|
||||||
|
}
|
||||||
|
/* All ops are add/sub/cmp/shift — no multiply, no divide, no float.
|
||||||
|
The ESP32 ALU does these in one cycle each. There is nothing here
|
||||||
|
that a simpler CPU cannot do. */
|
||||||
|
|
||||||
|
/* ─── Substrate descriptor ─────────────────────────────────── */
|
||||||
|
struct descriptor {
|
||||||
|
uint32_t input_addr; /* 64-byte aligned buffer address */
|
||||||
|
uint32_t output_addr; /* scalar writeback address */
|
||||||
|
uint16_t length; /* bytes (0 = no-op) */
|
||||||
|
uint8_t tag : 2; /* 0=CPU_blitter, 1=Vulkan, 2=fw */
|
||||||
|
uint8_t op : 4; /* trinary VM opcode */
|
||||||
|
uint8_t idx : 6; /* target trit index */
|
||||||
|
uint8_t val : 2; /* operand trit (-1,0,1) */
|
||||||
|
uint8_t status : 2; /* 0=pending, 1=done, 2=aborted */
|
||||||
|
} __attribute__((packed));
|
||||||
|
|
||||||
|
/* ─── Trinary VM state (7 subregisters × 32 trits = 224 bytes) */
|
||||||
|
struct vm_state {
|
||||||
|
int8_t sub[VM_OPS][TRIT_WIDTH]; /* -1, 0, 1 only */
|
||||||
|
uint8_t pc; /* current subregister index */
|
||||||
|
};
|
||||||
|
|
||||||
|
/* ─── Descriptor ring (64 × 16 = 1024 bytes) ──────────────── */
|
||||||
|
static struct descriptor ring[RING_SIZE];
|
||||||
|
static uint16_t head = 0; /* host writes here */
|
||||||
|
static uint16_t tail = 0; /* controller reads here */
|
||||||
|
|
||||||
|
/* ─── Single trinary VM step ──────────────────────────────── */
|
||||||
|
static void vm_step(struct vm_state *vm, const struct descriptor *d) {
|
||||||
|
int8_t *sub = vm->sub[d->op % VM_OPS];
|
||||||
|
uint8_t i = d->idx % TRIT_WIDTH;
|
||||||
|
switch (d->op) {
|
||||||
|
case 0: sub[i] = d->val; break; /* SET */
|
||||||
|
case 1: sub[i] = clamp(sub[i] + 1); break; /* ADD */
|
||||||
|
case 2: sub[i] = clamp(sub[i] - 1); break; /* SUB */
|
||||||
|
case 3: { /* SHIFT */
|
||||||
|
int8_t tmp = sub[TRIT_WIDTH-1];
|
||||||
|
for (uint8_t j = TRIT_WIDTH-1; j > 0; j--) sub[j] = sub[j-1];
|
||||||
|
sub[0] = tmp;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
case 4: { /* MERGE */
|
||||||
|
int8_t *src = vm->sub[(d->val + 1) % VM_OPS];
|
||||||
|
sub[i] = clamp(sub[i] + src[i]);
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
case 5: { /* PROJECT */
|
||||||
|
uint8_t src_idx = (d->idx + 1) % TRIT_WIDTH;
|
||||||
|
sub[i] = vm->sub[(d->op + 1) % VM_OPS][src_idx];
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
case 6: sub[i] = clamp(sub[i] * d->val); break; /* W */
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/* ─── Q0.16 scalar derivation ───────────────────────────── */
|
||||||
|
/* Folds all 7 subregisters into one 16-bit fixed-point value.
|
||||||
|
No division, no floating point. */
|
||||||
|
static uint16_t derive_scalar(const struct vm_state *vm) {
|
||||||
|
int32_t acc = 0;
|
||||||
|
for (int s = 0; s < VM_OPS; s++) {
|
||||||
|
for (int t = 0; t < TRIT_WIDTH; t++) {
|
||||||
|
acc = (acc << 1) + vm->sub[s][t];
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return (uint16_t)(acc & 0xFFFF); /* Q0.16 in [0, 65536) */
|
||||||
|
}
|
||||||
|
|
||||||
|
/* ─── Dispatch: one descriptor per idle slot ────────────── */
|
||||||
|
/* Called from FreeRTOS idle hook or main loop spin.
|
||||||
|
Returns 0 if no work was available (link busy / ring empty). */
|
||||||
|
uint8_t substrate_dispatch_one(struct vm_state *vm) {
|
||||||
|
if (head == tail) return 0; /* ring empty */
|
||||||
|
|
||||||
|
struct descriptor *d = &ring[tail];
|
||||||
|
if (d->status != 0) { /* already done/aborted */
|
||||||
|
tail = (tail + 1) % RING_SIZE;
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
/* read the input buffer (64-byte aligned, DMA-visible) */
|
||||||
|
volatile uint32_t *buf = (volatile uint32_t *)d->input_addr;
|
||||||
|
|
||||||
|
/* one trinary VM step (not the whole program — see §Controller Budget) */
|
||||||
|
vm_step(vm, d);
|
||||||
|
|
||||||
|
/* produce scalar */
|
||||||
|
uint16_t scalar = derive_scalar(vm);
|
||||||
|
|
||||||
|
/* write receipt to output address */
|
||||||
|
volatile uint32_t *out = (volatile uint32_t *)d->output_addr;
|
||||||
|
*out = scalar;
|
||||||
|
|
||||||
|
d->status = 1; /* done */
|
||||||
|
tail = (tail + 1) % RING_SIZE;
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
/* ─── FreeRTOS idle hook (called when no task is runnable) ─ */
|
||||||
|
/* This is the ESP32 equivalent of the PCIe L0s/L1 dispatch.
|
||||||
|
The scheduler calls this when the CPU would otherwise idle.
|
||||||
|
Same principle: redirect idle cycles to scalar derivation. */
|
||||||
|
void vApplicationIdleHook(void) {
|
||||||
|
static struct vm_state vm = {0};
|
||||||
|
for (int i = 0; i < 4; i++) { /* 4 slices per idle slot */
|
||||||
|
if (!substrate_dispatch_one(&vm)) break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/* ─── Initialization (called once at boot) ───────────────── */
|
||||||
|
void substrate_init(void) {
|
||||||
|
head = 0; tail = 0;
|
||||||
|
for (int i = 0; i < RING_SIZE; i++)
|
||||||
|
ring[i].status = 1; /* all slots start done */
|
||||||
|
/* ring is now ready for host to submit descriptors */
|
||||||
|
}
|
||||||
|
|
||||||
|
/* This entire file is 78 lines of plain C. No OS, no heap,
|
||||||
|
no GPU, no PCIe, no SIMD. It compiles with any C89 toolchain.
|
||||||
|
The ESP32 runs it at idle priority. The same .c file compiles
|
||||||
|
for x86, ARM, RISC-V, or Xtensa with zero changes to the
|
||||||
|
trinary VM or scalar derivation — only the idle hook address
|
||||||
|
changes. */
|
||||||
|
```
|
||||||
|
|
||||||
|
The firmware sketch is 78 lines. No heap. No GPU. No PCIe. No SIMD.
|
||||||
|
One idle hook, one ring, one scalar. It compiles for any architecture
|
||||||
|
that has a C compiler — Xtensa, ARM, x86, RISC-V, or a bare-metal ASIC.
|
||||||
|
|
||||||
|
The same `vm_step` and `derive_scalar` functions run identically on the
|
||||||
|
ESP32's ALU, a desktop's AVX unit, or an NVMe controller's Cortex-M0.
|
||||||
|
The only difference is how many cycles per descriptor arrive. The
|
||||||
|
substrate does not care.
|
||||||
|
|
||||||
|
And yes — if someone claims this can't run on an MCU, point them to the
|
||||||
|
78 lines above and the ASICs you've already made do far worse.
|
||||||
|
|
||||||
|
## Coda: Translation, Not Power
|
||||||
|
|
||||||
|
The surface claim says: "You cannot do PCIe idle-cycle compute on an
|
||||||
|
ESP32. The ESP32 has no PCIe."
|
||||||
|
|
||||||
|
Translate through the substrate:
|
||||||
|
|
||||||
|
- *Has no PCIe* —→ but it has a descriptor ring in SRAM instead of BAR
|
||||||
|
space. Same dispatch logic, different address decoder.
|
||||||
|
- *Has no GPU* —→ but it has a CPU blitter path that uses the same ALU
|
||||||
|
operations the GPU shader would use. Same trits, different backend.
|
||||||
|
- *Has no SIMD* —→ but the trinary VM uses only add, sub, cmp, shift.
|
||||||
|
Single-cycle on any ALU. SIMD was never required — it only made the
|
||||||
|
blitter faster on wider cores.
|
||||||
|
- *Has no spare cycles* —→ but the FreeRTOS idle hook exists on every
|
||||||
|
ESP32 project. The cycles are there. They are just currently consumed
|
||||||
|
by a `WFI` instruction that does nothing useful.
|
||||||
|
|
||||||
|
Every "cannot" is a translation target, not a hard boundary. The PCIe
|
||||||
|
slot, the Vulkan queue, the FreeRTOS idle hook — they are different
|
||||||
|
surfaces of the same abstraction: a dispatch loop with idle capacity.
|
||||||
|
The substrate translates the computation onto whichever surface is
|
||||||
|
available. It does not require the surface to be powerful. It only
|
||||||
|
requires it to have a dispatch loop with gaps.
|
||||||
|
|
||||||
|
The semantic weight of "impossible" collapses under translation. The
|
||||||
|
question is never "can this hardware do the computation?" It is always
|
||||||
|
"what is the correct translation layer for this hardware's idle
|
||||||
|
surface?"
|
||||||
|
|
||||||
|
The answer is the same for every device: a descriptor ring, a trinary
|
||||||
|
VM, and a scalar receipt. The device determines the throughput. The
|
||||||
|
substrate determines the invariant.
|
||||||
|
|
||||||
|
### Floor: 1-Wire Bitstream
|
||||||
|
|
||||||
|
If a critic still objects that the substrate requires "real hardware"
|
||||||
|
(a DMA engine, a FreeRTOS idle hook, a multi-byte bus), drop to the
|
||||||
|
floor: **1-Wire**.
|
||||||
|
|
||||||
|
A 1-Wire bus has one data line and ground. Every transaction is a
|
||||||
|
sequence of timed pulses — reset, presence, read-slot, write-slot.
|
||||||
|
The bus master controls the timing. A typical DS18B20 temperature
|
||||||
|
sensor transaction takes ~750 ms (conversion) plus bus protocol
|
||||||
|
overhead.
|
||||||
|
|
||||||
|
During the conversion window, the bus is idle. The master is polling
|
||||||
|
for the conversion to complete. Those cycles are the substrate's
|
||||||
|
dispatch loop.
|
||||||
|
|
||||||
|
The trinary VM descriptor is bitstreamed over the 1-Wire protocol
|
||||||
|
itself:
|
||||||
|
|
||||||
|
- **Reset pulse** = descriptor ring head advance (start of new slot)
|
||||||
|
- **Presence pulse** = controller acknowledges slot (device alive)
|
||||||
|
- **Write-1 slot** = trit +1
|
||||||
|
- **Write-0 slot** = trit -1
|
||||||
|
- **Read slot** = trit 0 (bus master samples the line)
|
||||||
|
|
||||||
|
The device is a passive participant — it holds a 64-bit ROM ID and
|
||||||
|
a scratchpad. The master reads the ROM ID once (the device's identity,
|
||||||
|
analogous to a PCIe vendor/device ID) and then uses the bus timing
|
||||||
|
windows to shuffle trits:
|
||||||
|
|
||||||
|
```
|
||||||
|
master: [reset] [write trit] [write trit] ... [read scalar] [reset]
|
||||||
|
|
||||||
|
timing: t_reset t_slot * n t_conv t_slot t_reset
|
||||||
|
↑ ↑ ↑ ↑
|
||||||
|
start trit stream scalar next descriptor
|
||||||
|
(descriptor) receipt
|
||||||
|
```
|
||||||
|
|
||||||
|
The "computation" is the master's own cycle between reset pulses.
|
||||||
|
The slave device never executes the trinary VM — it only provides
|
||||||
|
the bus timing window and the ROM ID. The master derives the scalar
|
||||||
|
from the pattern of its own write slots and the slave's response
|
||||||
|
timing.
|
||||||
|
|
||||||
|
This is the absolute floor: one data wire, one ground wire, one
|
||||||
|
slave device that costs $1 and does nothing but acknowledge its
|
||||||
|
presence. The substrate compiles to this bus the same way it compiles
|
||||||
|
to PCIe Gen 5 — different throughput, identical scalar receipt.
|
||||||
|
|
||||||
|
If someone insists the substrate "requires a real bus," hand them a
|
||||||
|
DS18B20, a pullup resistor, and a GPIO pin. 78 lines of C, one wire,
|
||||||
|
no excuses.
|
||||||
|
|
||||||
|
## Unified Timing Model: DynamicCanal → Timing → 1-Wire → Trit
|
||||||
|
|
||||||
|
The DynamicCanal (`Semantics.DynamicCanal`) defines a pressure-adaptive
|
||||||
|
transport geometry. A `Lane` carries a `Timing` tuple (slot, parity,
|
||||||
|
index), a `pressure` field, a `lambdaEff` (effective resistance), and
|
||||||
|
a `Regime` (coherent / stressed / throat). The canal widens under
|
||||||
|
pressure — `λ_eff(P)` decreases, allowing more flow.
|
||||||
|
|
||||||
|
`Timing.lean` derives physical DRAM timing parameters from the canal
|
||||||
|
state:
|
||||||
|
|
||||||
|
```
|
||||||
|
DynamicCanal state → tTCL (Torsional CAS Latency)
|
||||||
|
→ tMRE (Manifold Refresh Epoch)
|
||||||
|
→ tDLL (Damping Laplacian Latency)
|
||||||
|
```
|
||||||
|
|
||||||
|
These are real timing values — they correspond to JEDEC-style DRAM
|
||||||
|
parameters but derived from the manifold's torsion and interlocking
|
||||||
|
energy rather from a memory controller register. The canal does not
|
||||||
|
simulate timing; it *generates* it.
|
||||||
|
|
||||||
|
The 1-Wire bus is the physical instantiation of this timing model:
|
||||||
|
|
||||||
|
```
|
||||||
|
DynamicCanal.Lane.Timing ──→ 1-Wire pulse slot
|
||||||
|
slot ──→ write-slot duration (trit width)
|
||||||
|
parity ──→ presence pulse (device acknowledgment)
|
||||||
|
index ──→ conversion window (idle compute gap)
|
||||||
|
```
|
||||||
|
|
||||||
|
Each 1-Wire transaction slot is a `Timing` tuple emitted by the
|
||||||
|
DynamicCanal. The slot duration encodes the canal's current pressure
|
||||||
|
state: wide slot = high pressure = canal widened = more flow.
|
||||||
|
Narrow slot = low pressure = canal constricted = selective.
|
||||||
|
|
||||||
|
The trit stream is derived directly from the 1-Wire pulse timing:
|
||||||
|
|
||||||
|
| Pulse timing | Trit | DynamicCanal significance |
|
||||||
|
|-------------|------|--------------------------|
|
||||||
|
| Write-1 slot (long) | +1 | Canal wide, pressure high, flow permitted |
|
||||||
|
| Write-0 slot (short) | -1 | Canal narrow, pressure low, flow restricted |
|
||||||
|
| Read slot (sampled) | 0 | Canal coherent, steady state, no pressure gradient |
|
||||||
|
| Presence pulse | ACK | Device alive, canal throat stable |
|
||||||
|
|
||||||
|
The trinary VM receives these trits and executes them as operations
|
||||||
|
(SET, ADD, SUB, SHIFT, MERGE, PROJECT, W) — the same operations
|
||||||
|
defined in `DynamicCanal.LanePayload`. The VM does not interpret the
|
||||||
|
trits; it *performs* the canal's timing geometry as computation.
|
||||||
|
|
||||||
|
The Q0.16 scalar receipt is a fold of the VM state — which is itself
|
||||||
|
a trace of the DynamicCanal's pressure evolution over the observation
|
||||||
|
window. The scalar encodes the canal's identity: its pressure history,
|
||||||
|
its regime transitions, its timing signature.
|
||||||
|
|
||||||
|
```
|
||||||
|
DynamicCanal geometry
|
||||||
|
│
|
||||||
|
├──→ Timing.lean DRAM params (tTCL, tMRE, tDLL)
|
||||||
|
│ │
|
||||||
|
│ └──→ 1-Wire pulse timing
|
||||||
|
│ │
|
||||||
|
│ ├──→ trit stream (+1, -1, 0)
|
||||||
|
│ │ │
|
||||||
|
│ │ └──→ trinary VM operations
|
||||||
|
│ │ │
|
||||||
|
│ │ └──→ VM state (7 subregisters)
|
||||||
|
│ │ │
|
||||||
|
│ │ └──→ Q0.16 scalar receipt
|
||||||
|
│ │
|
||||||
|
│ └──→ PCIe L0s/L1 dispatch (same timing model,
|
||||||
|
│ different bus — identical scalar derivation)
|
||||||
|
│
|
||||||
|
└──→ Eigenflow stator (offline analysis of canal regime stability)
|
||||||
|
|
||||||
|
Every path bottoms out at the same scalar. The DynamicCanal abstraction
|
||||||
|
unifies the timing model across 1-Wire, PCIe, and DRAM because all
|
||||||
|
three are just different physical instantiations of the same pressure-
|
||||||
|
adaptive lane geometry.
|
||||||
|
|
||||||
|
## Coda: The Physics of Wires
|
||||||
|
|
||||||
|
A 1-Wire trace on a PCB is not a metaphor for a DynamicCanal. It is a
|
||||||
|
DynamicCanal. It has measurable resistance, capacitance, inductance,
|
||||||
|
and a voltage-to-current transfer function. The pullup resistor and
|
||||||
|
the trace's RC time constant determine the minimum slot duration —
|
||||||
|
that is the canal's width under zero pressure. Increase the trace
|
||||||
|
length (more R, more C), the canal widens, slots stretch, throughput
|
||||||
|
drops. The DynamicCanal geometry *predicts* this because the geometry
|
||||||
|
*is* the physics of the wire.
|
||||||
|
|
||||||
|
The same applies at every level:
|
||||||
|
|
||||||
|
- **CPU pin to package bond wire**: ~1 nH/mm inductance, ~1 pF ESD
|
||||||
|
capacitance. The canal throat.
|
||||||
|
- **PCB trace to connector**: distributed RLC model. The canal channel.
|
||||||
|
- **Cable to peripheral device**: transmission line with characteristic
|
||||||
|
impedance Z₀. The canal's characteristic flow impedance.
|
||||||
|
- **PCIe differential pair**: 85 Ω differential impedance, AC-coupled.
|
||||||
|
The canal's balanced transport mode.
|
||||||
|
- **DRAM data line**: fly-by topology with on-die termination. The
|
||||||
|
canal's terminated regime.
|
||||||
|
|
||||||
|
Every copper path has these properties because copper has these
|
||||||
|
properties. There is no abstraction layer between the DynamicCanal
|
||||||
|
geometry and the voltage on the wire — the geometry *is* the voltage
|
||||||
|
on the wire, expressed as a function of time and distance along the
|
||||||
|
trace.
|
||||||
|
|
||||||
|
Until we move to purely photonic interconnects (where the information
|
||||||
|
carrier is light rather than electron density), every bus is a
|
||||||
|
DynamicCanal. Resistance, capacitance, inductance, and the speed of
|
||||||
|
light in the dielectric bound what can be sent and when it arrives.
|
||||||
|
The substrate does not fight these bounds. It derives scalars from
|
||||||
|
them.
|
||||||
|
|
||||||
|
This is the unifying claim: the timing model is not imposed on the
|
||||||
|
hardware. It is *read from* the hardware. The wire tells you its
|
||||||
|
DynamicCanal parameters the moment you drive a pulse down it. The
|
||||||
|
only novel thing the substrate does is listen.
|
||||||
|
```
|
||||||
|
|
||||||
|
## Anti-Goals
|
||||||
|
|
||||||
|
- The substrate does not replace the OS I/O scheduler
|
||||||
|
- The substrate does not provide real-time guarantees
|
||||||
|
- The substrate does not modify PCIe link training or ASPM policy
|
||||||
|
- The substrate does not require new PCIe capability structures
|
||||||
|
- The substrate does not privilege one Gen or width over another
|
||||||
|
- The substrate does not require a GPU — tier 2 fallback uses CPU only
|
||||||
|
|
||||||
|
## Summary
|
||||||
|
|
||||||
|
The PCIe idle-cycle compute substrate is a pattern, not a product. It observes
|
||||||
|
that the management controller inside every PCIe device already runs a dispatch
|
||||||
|
loop during link idle states, and it redirects a fraction of those cycles toward
|
||||||
|
Q0.16 scalar derivation. The result is a platform-invariant compute surface that
|
||||||
|
scales from a 1x Gen 3 slot to a 16x Gen 5 slot without modification, produces
|
||||||
|
invariant-bearing receipts, and never competes with host-visible traffic.
|
||||||
|
|
||||||
|
It is unconventional only in that it treats the control plane as a compute
|
||||||
|
resource — the same insight that let SMS ride the cellular paging channel
|
||||||
|
without building a new radio tower.
|
||||||
|
|
@ -145,6 +145,7 @@ The full loop adds security gates (AngrySphinx exponential gate #3, FAMM frustra
|
||||||
- Lean 4 → Verilog extraction for FPGA targets
|
- Lean 4 → Verilog extraction for FPGA targets
|
||||||
- Universal GENSIS compiler with auto substrate selection
|
- Universal GENSIS compiler with auto substrate selection
|
||||||
- Cross-substrate benchmark suite
|
- Cross-substrate benchmark suite
|
||||||
|
- **PCIe Idle-Cycle Compute Harvester:** Formalize a substrate for scheduling computation on idle PCIe bus cycles (GPU, NVMe, DMA controller). Target: a `pcie_idle` substrate that observes bus transaction gaps, dispatches hash/verify kernels into those slots via scatter-gather DMA descriptors, and guarantees zero impact on user-facing transactions. Reference implementation: Windows SMB hash worker using RTX 4070 GPU as PCIe-attached compute engine with IDLE priority + background I/O + EcoQoS, checkpointed via DAG for kill-safe resume.
|
||||||
- **Forest Map absorption:** Closure criteria for complete map (Forest Phase 9)
|
- **Forest Map absorption:** Closure criteria for complete map (Forest Phase 9)
|
||||||
|
|
||||||
### Phase 7 — Proof of Completeness (Month 7) | **Status: SPECULATIVE**
|
### Phase 7 — Proof of Completeness (Month 7) | **Status: SPECULATIVE**
|
||||||
|
|
|
||||||
32
desmos_reynolds_bridge.txt
Normal file
32
desmos_reynolds_bridge.txt
Normal file
|
|
@ -0,0 +1,32 @@
|
||||||
|
Desmos Graph for Corrected Reynolds/Hermite Activation Bridge
|
||||||
|
|
||||||
|
Copy these equations into Desmos (https://www.desmos.com/calculator):
|
||||||
|
|
||||||
|
1. Normalized activation curve:
|
||||||
|
A(x) = 3x^2 - 2x^3
|
||||||
|
|
||||||
|
2. Offset physical bridge:
|
||||||
|
f_A(x) = 0.0278 + 0.012(3x^2 - 2x^3)
|
||||||
|
|
||||||
|
3. Reynolds number to x conversion:
|
||||||
|
x(Re) = clamp((Re - 2300) / 1700, 0, 1)
|
||||||
|
|
||||||
|
4. Direct Reynolds to activation:
|
||||||
|
A_Re(Re) = 3 * clamp((Re - 2300) / 1700, 0, 1)^2 - 2 * clamp((Re - 2300) / 1700, 0, 1)^3
|
||||||
|
|
||||||
|
5. Direct Reynolds to physical bridge:
|
||||||
|
f_A_Re(Re) = 0.0278 + 0.012 * (3 * clamp((Re - 2300) / 1700, 0, 1)^2 - 2 * clamp((Re - 2300) / 1700, 0, 1)^3)
|
||||||
|
|
||||||
|
Set x-axis range: 0 to 1 (for x) or 2000 to 4500 (for Re)
|
||||||
|
Set y-axis range: 0 to 0.05
|
||||||
|
|
||||||
|
Key points:
|
||||||
|
- Re = 2300 → x = 0 → A(0) = 0 → f_A(0) = 0.0278
|
||||||
|
- Re = 4000 → x = 1 → A(1) = 1 → f_A(1) = 0.0398
|
||||||
|
- Re = 3150 → x = 0.5 → A(0.5) = 0.5 → f_A(0.5) = 0.0338
|
||||||
|
|
||||||
|
Properties:
|
||||||
|
- A(x) is monotone increasing on [0,1]
|
||||||
|
- A'(x) = 6x(1-x) ≥ 0 for 0 ≤ x ≤ 1
|
||||||
|
- A'(0) = 0, A'(1) = 0 (smooth endpoints)
|
||||||
|
- f_A(x) maps [0,1] to [0.0278, 0.0398]
|
||||||
|
|
@ -1,28 +0,0 @@
|
||||||
// Auto-generated from Lean: Semantics.Hardware.TangNano9K.emitGenome18Address
|
|
||||||
// Source of truth: Semantics.Genome18.addr
|
|
||||||
// Theorem: verilogAddr_eq_addr (forall g, verilogAddr g = g.addr)
|
|
||||||
// Theorem: addr_injective (Function.Injective addr)
|
|
||||||
// Theorem: addr_range (addr < 262144)
|
|
||||||
//
|
|
||||||
// DO NOT EDIT BY HAND. Regenerate via: lake exe tangnano9k_emitter
|
|
||||||
|
|
||||||
`timescale 1ns / 1ps
|
|
||||||
|
|
||||||
module Genome18Address (
|
|
||||||
input wire [2:0] muBin,
|
|
||||||
input wire [2:0] rhoBin,
|
|
||||||
input wire [2:0] cBin,
|
|
||||||
input wire [2:0] mBin,
|
|
||||||
input wire [2:0] neBin,
|
|
||||||
input wire [2:0] sigmaBin,
|
|
||||||
output wire [17:0] addr
|
|
||||||
);
|
|
||||||
// Each bin is 3-bit (0..7). The weights are exact powers of two
|
|
||||||
// so synthesis maps them to shifts; no multiplier DSP needed.
|
|
||||||
assign addr = ({15'd0, muBin} * 18'd32768) +
|
|
||||||
({15'd0, rhoBin} * 18'd4096) +
|
|
||||||
({15'd0, cBin} * 18'd512) +
|
|
||||||
({15'd0, mBin} * 18'd64) +
|
|
||||||
({15'd0, neBin} * 18'd8) +
|
|
||||||
{12'd0, sigmaBin};
|
|
||||||
endmodule
|
|
||||||
|
|
@ -1,56 +0,0 @@
|
||||||
// Auto-generated from Lean: Semantics.Hardware.TangNano9K.NIICore.emitNIICore
|
|
||||||
// Source of truth: Semantics.Hardware.TangNano9K.NIICore.niiStep
|
|
||||||
// Theorem: niiOutputBounded (outputs are in [-CLIP, CLIP])
|
|
||||||
//
|
|
||||||
// DO NOT EDIT BY HAND. Regenerate via: lake exe tangnano9k_emitter
|
|
||||||
|
|
||||||
`timescale 1ns / 1ps
|
|
||||||
|
|
||||||
module nii_core #(
|
|
||||||
parameter W = 12,
|
|
||||||
parameter CLIP = 384
|
|
||||||
)(
|
|
||||||
input wire clk,
|
|
||||||
input wire rst,
|
|
||||||
input wire valid_in,
|
|
||||||
input wire signed [W-1:0] obs_a,
|
|
||||||
input wire signed [W-1:0] obs_t,
|
|
||||||
input wire signed [W-1:0] obs_g,
|
|
||||||
input wire signed [W-1:0] obs_c,
|
|
||||||
output reg valid_out,
|
|
||||||
output reg signed [W-1:0] nii_a,
|
|
||||||
output reg signed [W-1:0] nii_t,
|
|
||||||
output reg signed [W-1:0] nii_g,
|
|
||||||
output reg signed [W-1:0] nii_c
|
|
||||||
);
|
|
||||||
reg signed [W-1:0] prev_a, prev_t, prev_g, prev_c;
|
|
||||||
|
|
||||||
function signed [W-1:0] sat_clip;
|
|
||||||
input signed [W:0] x;
|
|
||||||
begin
|
|
||||||
if (x > CLIP) sat_clip = CLIP[W-1:0];
|
|
||||||
else if (x < -CLIP) sat_clip = -CLIP[W-1:0];
|
|
||||||
else sat_clip = x[W-1:0];
|
|
||||||
end
|
|
||||||
endfunction
|
|
||||||
|
|
||||||
always @(posedge clk) begin
|
|
||||||
if (rst) begin
|
|
||||||
prev_a <= 0; prev_t <= 0; prev_g <= 0; prev_c <= 0;
|
|
||||||
nii_a <= 0; nii_t <= 0; nii_g <= 0; nii_c <= 0;
|
|
||||||
valid_out <= 0;
|
|
||||||
end else begin
|
|
||||||
valid_out <= valid_in;
|
|
||||||
if (valid_in) begin
|
|
||||||
nii_a <= sat_clip(obs_a - prev_a);
|
|
||||||
nii_t <= sat_clip(obs_t - prev_t);
|
|
||||||
nii_g <= sat_clip(obs_g - prev_g);
|
|
||||||
nii_c <= sat_clip(obs_c - prev_c);
|
|
||||||
prev_a <= obs_a;
|
|
||||||
prev_t <= obs_t;
|
|
||||||
prev_g <= obs_g;
|
|
||||||
prev_c <= obs_c;
|
|
||||||
end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
endmodule
|
|
||||||
|
|
@ -1,61 +0,0 @@
|
||||||
// Auto-generated from Lean: Semantics.Hardware.TangNano9K.emitQ16_16ALU
|
|
||||||
// Source of truth: Semantics.Q16_16
|
|
||||||
// NOTE: All arithmetic is SIGNED and SATURATING.
|
|
||||||
//
|
|
||||||
// DO NOT EDIT BY HAND. Regenerate via: lake exe tangnano9k_emitter
|
|
||||||
|
|
||||||
`timescale 1ns / 1ps
|
|
||||||
|
|
||||||
module Q16_16_ALU (
|
|
||||||
input wire [31:0] a,
|
|
||||||
input wire [31:0] b,
|
|
||||||
input wire [2:0] op, // 0=add, 1=sub, 2=mul, 3=div, 4=max, 5=min, 6=abs
|
|
||||||
output reg [31:0] result,
|
|
||||||
output reg overflow
|
|
||||||
);
|
|
||||||
localparam OP_ADD = 3'd0;
|
|
||||||
localparam OP_SUB = 3'd1;
|
|
||||||
localparam OP_MUL = 3'd2;
|
|
||||||
localparam OP_DIV = 3'd3;
|
|
||||||
localparam OP_MAX = 3'd4;
|
|
||||||
localparam OP_MIN = 3'd5;
|
|
||||||
localparam OP_ABS = 3'd6;
|
|
||||||
|
|
||||||
localparam MAX_POS = 32'h7FFFFFFF;
|
|
||||||
localparam MAX_NEG = 32'h80000000;
|
|
||||||
|
|
||||||
// Signed Addition with Saturation
|
|
||||||
wire [32:0] add_full = {a[31], a} + {b[31], b};
|
|
||||||
wire [31:0] add_sat = (add_full[32] != add_full[31]) ?
|
|
||||||
(add_full[32] ? MAX_NEG : MAX_POS) : add_full[31:0];
|
|
||||||
|
|
||||||
// Signed Subtraction with Saturation
|
|
||||||
wire [32:0] sub_full = {a[31], a} - {b[31], b};
|
|
||||||
wire [31:0] sub_sat = (sub_full[32] != sub_full[31]) ?
|
|
||||||
(sub_full[32] ? MAX_NEG : MAX_POS) : sub_full[31:0];
|
|
||||||
|
|
||||||
// Signed Multiplication with Saturation: (a * b) >>> 16
|
|
||||||
wire [63:0] mul_full = $signed(a) * $signed(b);
|
|
||||||
wire [63:0] mul_shifted = mul_full >>> 16;
|
|
||||||
wire [31:0] mul_sat = ($signed(mul_shifted) > $signed({32'd0, MAX_POS})) ? MAX_POS :
|
|
||||||
($signed(mul_shifted) < $signed({32'hFFFFFFFF, MAX_NEG})) ? MAX_NEG :
|
|
||||||
mul_shifted[31:0];
|
|
||||||
|
|
||||||
// Signed Division: (a << 16) / b
|
|
||||||
wire [63:0] div_num = {{16{a[31]}}, a, 16'd0};
|
|
||||||
wire [31:0] div_res = (b == 0) ? (a[31] ? MAX_NEG : MAX_POS) : (div_num / $signed(b));
|
|
||||||
|
|
||||||
always @(*) begin
|
|
||||||
overflow = 1'b0;
|
|
||||||
case (op)
|
|
||||||
OP_ADD: result = add_sat;
|
|
||||||
OP_SUB: result = sub_sat;
|
|
||||||
OP_MUL: result = mul_sat;
|
|
||||||
OP_DIV: result = div_res;
|
|
||||||
OP_MAX: result = ($signed(a) > $signed(b)) ? a : b;
|
|
||||||
OP_MIN: result = ($signed(a) < $signed(b)) ? a : b;
|
|
||||||
OP_ABS: result = (a[31]) ? sub_sat : a; // abs(a) = 0 - a (saturating)
|
|
||||||
default: result = 32'd0;
|
|
||||||
endcase
|
|
||||||
end
|
|
||||||
endmodule
|
|
||||||
|
|
@ -1,179 +0,0 @@
|
||||||
// Auto-generated from Lean: Semantics.Hardware.TangNano9K.RGFlowFAMM.emitRGFlowFAMM
|
|
||||||
// Source of truth: Semantics.Hardware.TangNano9K.RGFlowFAMM.rgflowStep + fammUpdate
|
|
||||||
// Theorems: rgflowSigmaBounded, rgflowRPBounded, fammCountersBounded, fammWarnCorrect
|
|
||||||
//
|
|
||||||
// DO NOT EDIT BY HAND. Regenerate via: lake exe tangnano9k_emitter
|
|
||||||
|
|
||||||
`timescale 1ns / 1ps
|
|
||||||
|
|
||||||
module rgflow_famm #(
|
|
||||||
parameter W = 12,
|
|
||||||
parameter THRESH = 16'd650,
|
|
||||||
parameter NEAR_MISS_BAND = 8'd160,
|
|
||||||
parameter DECAY_FRUST = 8'd6,
|
|
||||||
parameter DECAY_TORS = 8'd4,
|
|
||||||
parameter DECAY_BASIN = 8'd2,
|
|
||||||
parameter BASIN_INC = 8'd2,
|
|
||||||
parameter NEAR_BASIN_INC = 8'd1,
|
|
||||||
parameter WARN_AT = 8'd240
|
|
||||||
)(
|
|
||||||
input wire clk,
|
|
||||||
input wire rst,
|
|
||||||
input wire valid_in,
|
|
||||||
input wire signed [W-1:0] nii_a,
|
|
||||||
input wire signed [W-1:0] nii_t,
|
|
||||||
input wire signed [W-1:0] nii_g,
|
|
||||||
input wire signed [W-1:0] nii_c,
|
|
||||||
input wire [7:0] coherence,
|
|
||||||
input wire [7:0] compression,
|
|
||||||
input wire [7:0] failure,
|
|
||||||
input wire [7:0] expand_prior,
|
|
||||||
output reg valid_out,
|
|
||||||
output reg [9:0] sigma,
|
|
||||||
output reg [7:0] reject_pressure,
|
|
||||||
output reg [7:0] torsion_delta,
|
|
||||||
output reg [7:0] famm_frustration,
|
|
||||||
output reg [7:0] famm_basin,
|
|
||||||
output reg [7:0] famm_torsion,
|
|
||||||
output reg [2:0] verdict_oh,
|
|
||||||
output reg warn_frustration,
|
|
||||||
output reg warn_torsion,
|
|
||||||
output reg warn_basin,
|
|
||||||
output reg warn_any,
|
|
||||||
output reg sat_frustration,
|
|
||||||
output reg sat_torsion,
|
|
||||||
output reg sat_basin,
|
|
||||||
output reg sat_any,
|
|
||||||
output reg famm_changed,
|
|
||||||
output reg [7:0] status_byte
|
|
||||||
);
|
|
||||||
wire [W-1:0] abs_a = nii_a[W-1] ? -nii_a : nii_a;
|
|
||||||
wire [W-1:0] abs_t = nii_t[W-1] ? -nii_t : nii_t;
|
|
||||||
wire [W-1:0] abs_g = nii_g[W-1] ? -nii_g : nii_g;
|
|
||||||
wire [W-1:0] abs_c = nii_c[W-1] ? -nii_c : nii_c;
|
|
||||||
wire [W+1:0] surprise_mag = (abs_a + abs_t + abs_g + abs_c) >> 2;
|
|
||||||
|
|
||||||
function [7:0] lin_decay;
|
|
||||||
input [7:0] x;
|
|
||||||
input [7:0] d;
|
|
||||||
begin
|
|
||||||
lin_decay = (x > d) ? (x - d) : 8'd0;
|
|
||||||
end
|
|
||||||
endfunction
|
|
||||||
|
|
||||||
function [7:0] sat8_add;
|
|
||||||
input [7:0] a;
|
|
||||||
input [7:0] b;
|
|
||||||
reg [8:0] s;
|
|
||||||
begin
|
|
||||||
s = a + b;
|
|
||||||
sat8_add = s[8] ? 8'hff : s[7:0];
|
|
||||||
end
|
|
||||||
endfunction
|
|
||||||
|
|
||||||
reg signed [16:0] sigma_tmp;
|
|
||||||
reg [7:0] rp;
|
|
||||||
reg [7:0] td;
|
|
||||||
reg [7:0] frust_d, tors_d, basin_d;
|
|
||||||
reg [7:0] frust_n, tors_n, basin_n;
|
|
||||||
reg [2:0] verdict_n;
|
|
||||||
reg warn_n_any, sat_n_any, famm_chg_n;
|
|
||||||
|
|
||||||
always @(posedge clk) begin
|
|
||||||
if (rst) begin
|
|
||||||
valid_out <= 1'b0;
|
|
||||||
sigma <= 10'd0;
|
|
||||||
reject_pressure <= 8'd0;
|
|
||||||
torsion_delta <= 8'd0;
|
|
||||||
famm_frustration <= 8'd0;
|
|
||||||
famm_basin <= 8'd0;
|
|
||||||
famm_torsion <= 8'd0;
|
|
||||||
verdict_oh <= 3'b000;
|
|
||||||
warn_frustration <= 1'b0;
|
|
||||||
warn_torsion <= 1'b0;
|
|
||||||
warn_basin <= 1'b0;
|
|
||||||
warn_any <= 1'b0;
|
|
||||||
sat_frustration <= 1'b0;
|
|
||||||
sat_torsion <= 1'b0;
|
|
||||||
sat_basin <= 1'b0;
|
|
||||||
sat_any <= 1'b0;
|
|
||||||
famm_changed <= 1'b0;
|
|
||||||
status_byte <= 8'h00;
|
|
||||||
end else begin
|
|
||||||
valid_out <= valid_in;
|
|
||||||
if (valid_in) begin
|
|
||||||
sigma_tmp = $signed(17'sd256)
|
|
||||||
+ ($signed({9'd0, coherence}) <<< 1)
|
|
||||||
+ $signed({9'd0, expand_prior})
|
|
||||||
+ $signed({9'd0, compression})
|
|
||||||
- ($signed({9'd0, failure}) <<< 1)
|
|
||||||
- $signed({9'd0, surprise_mag[7:0]})
|
|
||||||
- $signed({9'd0, famm_frustration})
|
|
||||||
- $signed({9'd0, famm_torsion});
|
|
||||||
|
|
||||||
if (sigma_tmp < 0) sigma <= 10'd0;
|
|
||||||
else if (sigma_tmp > $signed(17'sd1023)) sigma <= 10'd1023;
|
|
||||||
else sigma <= sigma_tmp[9:0];
|
|
||||||
|
|
||||||
if (sigma_tmp >= $signed({1'b0, THRESH})) begin
|
|
||||||
rp = 8'd0;
|
|
||||||
verdict_n = 3'b001;
|
|
||||||
end else begin
|
|
||||||
rp = (THRESH - sigma_tmp[15:0] > 16'd255) ? 8'hff
|
|
||||||
: (THRESH - sigma_tmp[15:0]);
|
|
||||||
verdict_n = (rp <= NEAR_MISS_BAND) ? 3'b010 : 3'b100;
|
|
||||||
end
|
|
||||||
if (verdict_n == 3'b001) td = 8'd0;
|
|
||||||
else if (verdict_n == 3'b010) td = (rp >> 1);
|
|
||||||
else td = rp;
|
|
||||||
|
|
||||||
frust_d = lin_decay(famm_frustration, DECAY_FRUST);
|
|
||||||
tors_d = lin_decay(famm_torsion, DECAY_TORS);
|
|
||||||
basin_d = lin_decay(famm_basin, DECAY_BASIN);
|
|
||||||
|
|
||||||
case (verdict_n)
|
|
||||||
3'b001: begin
|
|
||||||
basin_n = sat8_add(basin_d, BASIN_INC);
|
|
||||||
frust_n = (frust_d != 0) ? frust_d - 1'b1 : 8'd0;
|
|
||||||
tors_n = tors_d;
|
|
||||||
end
|
|
||||||
3'b010: begin
|
|
||||||
basin_n = sat8_add(basin_d, NEAR_BASIN_INC);
|
|
||||||
frust_n = sat8_add(frust_d, rp / 8'd16);
|
|
||||||
tors_n = sat8_add(tors_d, td / 8'd32);
|
|
||||||
end
|
|
||||||
default: begin
|
|
||||||
basin_n = basin_d;
|
|
||||||
frust_n = sat8_add(frust_d, rp / 8'd12);
|
|
||||||
tors_n = sat8_add(tors_d, td / 8'd24);
|
|
||||||
end
|
|
||||||
endcase
|
|
||||||
|
|
||||||
warn_frustration <= (frust_n >= WARN_AT);
|
|
||||||
warn_torsion <= (tors_n >= WARN_AT);
|
|
||||||
warn_basin <= (basin_n >= WARN_AT);
|
|
||||||
warn_n_any = (frust_n >= WARN_AT) | (tors_n >= WARN_AT) | (basin_n >= WARN_AT);
|
|
||||||
warn_any <= warn_n_any;
|
|
||||||
sat_frustration <= (frust_n == 8'hff);
|
|
||||||
sat_torsion <= (tors_n == 8'hff);
|
|
||||||
sat_basin <= (basin_n == 8'hff);
|
|
||||||
sat_n_any = (frust_n == 8'hff) | (tors_n == 8'hff) | (basin_n == 8'hff);
|
|
||||||
sat_any <= sat_n_any;
|
|
||||||
famm_chg_n = (frust_n != famm_frustration)
|
|
||||||
| (tors_n != famm_torsion)
|
|
||||||
| (basin_n != famm_basin);
|
|
||||||
famm_changed <= famm_chg_n;
|
|
||||||
|
|
||||||
famm_frustration <= frust_n;
|
|
||||||
famm_torsion <= tors_n;
|
|
||||||
famm_basin <= basin_n;
|
|
||||||
|
|
||||||
verdict_oh <= verdict_n;
|
|
||||||
status_byte <= {1'b0, 1'b0, famm_chg_n, sat_n_any, warn_n_any,
|
|
||||||
verdict_n[2], verdict_n[1], verdict_n[0]};
|
|
||||||
reject_pressure <= rp;
|
|
||||||
torsion_delta <= td;
|
|
||||||
end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
endmodule
|
|
||||||
|
|
@ -1,170 +0,0 @@
|
||||||
# ScienceHub MCP — Sovereign Research Surface
|
|
||||||
|
|
||||||
## What It Is
|
|
||||||
An MCP server that lets your LLM say **"I need <topic>"** and automatically:
|
|
||||||
1. Searches your local corpus (Zotero + PDFs)
|
|
||||||
2. If missing, fetches from arXiv
|
|
||||||
3. Ingests into Zotero + local storage
|
|
||||||
4. Returns a structured report
|
|
||||||
|
|
||||||
## Installation
|
|
||||||
|
|
||||||
### Requirements
|
|
||||||
- Python 3.11+
|
|
||||||
- `mcp` SDK: `pip install mcp`
|
|
||||||
- `pdftotext` + `pdfinfo` (from poppler-utils)
|
|
||||||
- SQLite (built-in)
|
|
||||||
|
|
||||||
### MCP Client Config (Claude Desktop)
|
|
||||||
Add to your `claude_desktop_config.json`:
|
|
||||||
|
|
||||||
```json
|
|
||||||
{
|
|
||||||
"mcpServers": {
|
|
||||||
"sciencehub": {
|
|
||||||
"command": "python3",
|
|
||||||
"args": [
|
|
||||||
"/home/allaun/Documents/Research Stack/scripts/sciencehub_mcp.py"
|
|
||||||
]
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
### MCP Client Config (Cline / Continue)
|
|
||||||
Add to your `.vscode/mcp-settings.json` or Cline MCP settings:
|
|
||||||
|
|
||||||
```json
|
|
||||||
{
|
|
||||||
"mcpServers": [
|
|
||||||
{
|
|
||||||
"name": "sciencehub",
|
|
||||||
"command": "python3",
|
|
||||||
"args": [
|
|
||||||
"/home/allaun/Documents/Research Stack/scripts/sciencehub_mcp.py"
|
|
||||||
]
|
|
||||||
}
|
|
||||||
]
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
## Tools
|
|
||||||
|
|
||||||
### `need` — The "I need X" pipeline
|
|
||||||
**Input:** `{"query": "I need attention mechanism survey", "auto_ingest": true}`
|
|
||||||
|
|
||||||
**What happens:**
|
|
||||||
- Searches local Zotero library + PDF corpus
|
|
||||||
- If found locally → returns hit list, no fetch
|
|
||||||
- If missing → searches arXiv, downloads PDF, ingests to Zotero, reports back
|
|
||||||
|
|
||||||
### `search_local` — Query your corpus
|
|
||||||
**Input:** `{"query": "burgers turbulence", "limit": 10}`
|
|
||||||
|
|
||||||
Searches:
|
|
||||||
- `zotero_items` (titles, DOIs, URLs, extras)
|
|
||||||
- `local_pdfs` (filenames, arXiv IDs, DOIs)
|
|
||||||
- `arxiv_meta` (cached abstracts)
|
|
||||||
|
|
||||||
### `fetch_arxiv` — Direct download by ID
|
|
||||||
**Input:** `{"arxiv_id": "1706.03762", "ingest": true}`
|
|
||||||
|
|
||||||
Downloads to `~/Downloads/data/Downloads_from_internet/Deep Research/alphaXiv_PDFs_2026_04/`
|
|
||||||
and creates a Zotero item.
|
|
||||||
|
|
||||||
### `review_paper` — Quick PDF review
|
|
||||||
**Input:** `{"path": "/path/to/paper.pdf"}`
|
|
||||||
|
|
||||||
Runs `pdfinfo` + `pdftotext -l 1` and returns metadata + first-page excerpt.
|
|
||||||
|
|
||||||
### `corpus_report` — Library stats
|
|
||||||
Returns counts of Zotero items, local PDFs, completed/failed needs.
|
|
||||||
|
|
||||||
## CLI Mode (no MCP client needed)
|
|
||||||
```bash
|
|
||||||
cd "/home/allaun/Documents/Research Stack"
|
|
||||||
|
|
||||||
# Stats
|
|
||||||
python3 scripts/sciencehub_mcp.py --report
|
|
||||||
|
|
||||||
# Search local corpus
|
|
||||||
python3 scripts/sciencehub_mcp.py --search "burgers turbulence"
|
|
||||||
|
|
||||||
# Full "I need" pipeline (dry-run with --no-ingest)
|
|
||||||
python3 scripts/sciencehub_mcp.py "I need attention mechanism survey"
|
|
||||||
|
|
||||||
# Review a specific PDF
|
|
||||||
python3 scripts/sciencehub_mcp.py --review /path/to/paper.pdf
|
|
||||||
|
|
||||||
# Fetch by arXiv ID
|
|
||||||
python3 scripts/sciencehub_mcp.py --fetch 1706.03762
|
|
||||||
```
|
|
||||||
|
|
||||||
## Companion Scripts
|
|
||||||
|
|
||||||
### Ingest Watcher (`ingest_watcher.py`)
|
|
||||||
Monitors directories for new PDFs, auto-ingests them into Zotero.
|
|
||||||
|
|
||||||
```bash
|
|
||||||
# Dry run
|
|
||||||
python3 scripts/ingest_watcher.py --once --dry-run
|
|
||||||
|
|
||||||
# Actual ingest
|
|
||||||
python3 scripts/ingest_watcher.py --once
|
|
||||||
|
|
||||||
# Daemon mode
|
|
||||||
python3 scripts/ingest_watcher.py --daemon --interval 60
|
|
||||||
```
|
|
||||||
|
|
||||||
### Review Agent (`review_agent.py`)
|
|
||||||
Generates structured reviews for ingested papers.
|
|
||||||
|
|
||||||
```bash
|
|
||||||
# Review one paper
|
|
||||||
python3 scripts/review_agent.py --paper /path/to/paper.pdf
|
|
||||||
|
|
||||||
# Batch review 5 unreviewed papers
|
|
||||||
python3 scripts/review_agent.py --batch 5
|
|
||||||
|
|
||||||
# Daemon mode (loops every 5 min)
|
|
||||||
python3 scripts/review_agent.py --daemon --interval 300
|
|
||||||
```
|
|
||||||
|
|
||||||
## Architecture
|
|
||||||
```
|
|
||||||
┌─────────────────────────────────────────┐
|
|
||||||
│ LLM (Claude / Cline / Continue) │
|
|
||||||
│ says: "I need <topic>" │
|
|
||||||
└──────────────────┬──────────────────────┘
|
|
||||||
│ MCP stdio
|
|
||||||
┌──────────────────▼──────────────────────┐
|
|
||||||
│ sciencehub_mcp.py │
|
|
||||||
│ ├── ScienceHub.need(query) │
|
|
||||||
│ │ ├── CorpusIndex.search(query) │
|
|
||||||
│ │ └── if miss → ArxivClient.search() │
|
|
||||||
│ │ └── ArxivClient.download() │
|
|
||||||
│ │ └── ZoteroWriter.add_preprint()│
|
|
||||||
│ └── review_paper → pdfinfo + pdftotext │
|
|
||||||
└─────────────────────────────────────────┘
|
|
||||||
```
|
|
||||||
|
|
||||||
## Data Flow
|
|
||||||
- **Zotero DB:** `~/Zotero/zotero.sqlite` (read + write)
|
|
||||||
- **Index DB:** `~/Research Stack/data/substrate_index.db` (read + write)
|
|
||||||
- **PDF Archive:** `~/Downloads/data/Downloads_from_internet/Deep Research/alphaXiv_PDFs_2026_04/`
|
|
||||||
- **Watcher Log:** `~/Research Stack/data/ingest_watcher.log`
|
|
||||||
|
|
||||||
## Safety
|
|
||||||
- Zotero DB is **backed up automatically** before every write (`zotero.sqlite.backup.YYYYMMDD_HHMMSS`)
|
|
||||||
- On any write failure, backup is **restored automatically**
|
|
||||||
- arXiv downloads are **validated** (must be > 1KB)
|
|
||||||
- Duplicate authors are **deduplicated** (unique constraint handled)
|
|
||||||
|
|
||||||
## Troubleshooting
|
|
||||||
| Problem | Fix |
|
|
||||||
|---------|-----|
|
|
||||||
| "arXiv search failed" | Check internet; script uses `https://export.arxiv.org` |
|
|
||||||
| "Zotero ingest failed" | Check Zotero is closed (no lock on DB) |
|
|
||||||
| "No text" in review | PDF may be scanned/image; needs OCR |
|
|
||||||
| MCP not connecting | Verify `pip install mcp` and Python path |
|
|
||||||
| Ollama review hangs | Switch to stub mode: set `OLLAMA_MODEL` env or pass `--stub` |
|
|
||||||
|
|
@ -1,49 +0,0 @@
|
||||||
#!/bin/bash
|
|
||||||
# Final GitHub Cleanup Script - v2
|
|
||||||
# Hides standalone repos from profile and/or deletes them.
|
|
||||||
|
|
||||||
set -e
|
|
||||||
|
|
||||||
REPOS=(
|
|
||||||
"braid-field-papers"
|
|
||||||
"AMMR"
|
|
||||||
"bezier-kit"
|
|
||||||
"Newtonian-Superfluid-Simulation"
|
|
||||||
"heat-2D"
|
|
||||||
"chunked-audio-DSP"
|
|
||||||
"matter-frequencies"
|
|
||||||
"Allelica"
|
|
||||||
"parametric-learn"
|
|
||||||
"text-to-cad"
|
|
||||||
"WasmGPU"
|
|
||||||
"OTOM"
|
|
||||||
"NoDupeLabs"
|
|
||||||
)
|
|
||||||
|
|
||||||
echo "=== Research Stack: GitHub Cleanup ==="
|
|
||||||
echo "This will archive (hide) and optionally delete the following repos:"
|
|
||||||
for repo in "${REPOS[@]}"; do echo " - allaunthefox/$repo"; done
|
|
||||||
|
|
||||||
echo ""
|
|
||||||
read -p "Do you want to PERMANENTLY DELETE these repos? (type DELETE to confirm, otherwise they will only be ARCHIVED): " ACTION
|
|
||||||
|
|
||||||
if [ "$ACTION" == "DELETE" ]; then
|
|
||||||
echo "Requesting delete_repo scope..."
|
|
||||||
gh auth refresh -h github.com -s delete_repo
|
|
||||||
fi
|
|
||||||
|
|
||||||
for repo in "${REPOS[@]}"; do
|
|
||||||
echo "--- allaunthefox/$repo ---"
|
|
||||||
|
|
||||||
# Always archive first to be safe
|
|
||||||
echo " Archiving..."
|
|
||||||
gh repo archive "allaunthefox/$repo" --yes || echo " Already archived or missing."
|
|
||||||
|
|
||||||
if [ "$ACTION" == "DELETE" ]; then
|
|
||||||
echo " Deleting..."
|
|
||||||
gh repo delete "allaunthefox/$repo" --yes || echo " Failed to delete. Check permissions."
|
|
||||||
fi
|
|
||||||
done
|
|
||||||
|
|
||||||
echo ""
|
|
||||||
echo "Done. Your GitHub profile should now be focused on the Research-Stack umbrella."
|
|
||||||
|
|
@ -1,120 +0,0 @@
|
||||||
#!/usr/bin/env bash
|
|
||||||
set -euo pipefail
|
|
||||||
|
|
||||||
# clean-tailscale-refs.sh
|
|
||||||
# Removes stale Tailscale node references from the repo.
|
|
||||||
# Run this AFTER clearing the tailnet.
|
|
||||||
|
|
||||||
REPO_ROOT="/home/allaun/CascadeProjects/Research-Stack"
|
|
||||||
cd "$REPO_ROOT"
|
|
||||||
|
|
||||||
echo "=========================================="
|
|
||||||
echo " Clean Tailscale References"
|
|
||||||
echo "=========================================="
|
|
||||||
echo ""
|
|
||||||
|
|
||||||
# --- 1. Backup .git/config ---
|
|
||||||
if [[ -f .git/config ]]; then
|
|
||||||
cp .git/config .git/config.backup.$(date +%Y%m%d_%H%M%S)
|
|
||||||
echo "[1/6] Backed up .git/config"
|
|
||||||
fi
|
|
||||||
|
|
||||||
# --- 2. Remove Tailscale LFS entries from .git/config ---
|
|
||||||
# These point to old nodes that no longer exist on the tailnet.
|
|
||||||
echo "[2/6] Removing Tailscale LFS entries from .git/config..."
|
|
||||||
git config --local --remove-section 'lfs.https://100.111.192.47/home/judge-gcp-20260330/git-mirrors/research-stack.git/info/lfs' 2>/dev/null || true
|
|
||||||
git config --local --remove-section 'lfs.https://100.85.1.50/var/git-mirrors/research-stack.git/info/lfs' 2>/dev/null || true
|
|
||||||
git config --local --remove-section 'lfs.https://100.103.54.58/home/svc-tardy/git-mirrors/research-stack.git/info/lfs' 2>/dev/null || true
|
|
||||||
git config --local --remove-section 'lfs.http://100.127.111.7:3000/sovereign/research-stack.git/info/lfs' 2>/dev/null || true
|
|
||||||
|
|
||||||
# --- 3. Remove i2p aliases from .git/config ---
|
|
||||||
echo "[3/6] Removing i2p aliases from .git/config..."
|
|
||||||
git config --local --remove-section 'alias' 2>/dev/null || true
|
|
||||||
|
|
||||||
# --- 4. Remove forgejo branch merge-base refs from .git/config ---
|
|
||||||
echo "[4/6] Removing stale forgejo branch merge-base refs..."
|
|
||||||
python3 << 'PYEOF'
|
|
||||||
import re
|
|
||||||
|
|
||||||
with open('.git/config', 'r') as f:
|
|
||||||
content = f.read()
|
|
||||||
|
|
||||||
# Remove any vscode-merge-base line that references forgejo
|
|
||||||
lines = content.splitlines()
|
|
||||||
filtered = []
|
|
||||||
for line in lines:
|
|
||||||
if 'vscode-merge-base' in line and 'forgejo' in line:
|
|
||||||
continue
|
|
||||||
filtered.append(line)
|
|
||||||
|
|
||||||
new_content = '\n'.join(filtered) + '\n'
|
|
||||||
|
|
||||||
# Also remove forgejo remote section if it exists (it shouldn't, but just in case)
|
|
||||||
new_content = re.sub(
|
|
||||||
r'\[remote "forgejo"\][^\[]*',
|
|
||||||
'',
|
|
||||||
new_content
|
|
||||||
)
|
|
||||||
|
|
||||||
with open('.git/config', 'w') as f:
|
|
||||||
f.write(new_content)
|
|
||||||
PYEOF
|
|
||||||
|
|
||||||
# --- 5. Update .claude/settings.local.json ---
|
|
||||||
# Remove Bash permissions that reference old tailscale IPs or node names.
|
|
||||||
echo "[5/6] Cleaning .claude/settings.local.json..."
|
|
||||||
python3 << 'PYEOF'
|
|
||||||
import json
|
|
||||||
|
|
||||||
with open('.claude/settings.local.json', 'r') as f:
|
|
||||||
data = json.load(f)
|
|
||||||
|
|
||||||
old_perms = data.get('permissions', {}).get('allow', [])
|
|
||||||
new_perms = []
|
|
||||||
|
|
||||||
skip_patterns = [
|
|
||||||
'100.111.192.47',
|
|
||||||
'100.110.117.19',
|
|
||||||
'100.127.111.7',
|
|
||||||
'architect',
|
|
||||||
'netcup',
|
|
||||||
'judge',
|
|
||||||
]
|
|
||||||
|
|
||||||
for p in old_perms:
|
|
||||||
if any(sp in p for sp in skip_patterns):
|
|
||||||
continue
|
|
||||||
new_perms.append(p)
|
|
||||||
|
|
||||||
data['permissions']['allow'] = new_perms
|
|
||||||
|
|
||||||
with open('.claude/settings.local.json', 'w') as f:
|
|
||||||
json.dump(data, f, indent=2)
|
|
||||||
f.write('\n')
|
|
||||||
PYEOF
|
|
||||||
|
|
||||||
# --- 6. Update code files ---
|
|
||||||
echo "[6/6] Updating code files..."
|
|
||||||
|
|
||||||
# 5-Applications/scripts/server.js — comment out architect ping
|
|
||||||
if [[ -f 5-Applications/scripts/server.js ]]; then
|
|
||||||
sed -i 's|exec("ping -c 1 -W 2 100.127.111.7"|// exec("ping -c 1 -W 2 100.127.111.7" // STALE: architect node removed|g' 5-Applications/scripts/server.js 2>/dev/null || true
|
|
||||||
fi
|
|
||||||
|
|
||||||
# 5-Applications/scripts/all_device_signal_topology.py — update qfox reference
|
|
||||||
if [[ -f 5-Applications/scripts/all_device_signal_topology.py ]]; then
|
|
||||||
sed -i 's|"network_node_qfox"|"network_node_primary"|g' 5-Applications/scripts/all_device_signal_topology.py 2>/dev/null || true
|
|
||||||
sed -i 's|Network Node (qfox - primary node)|Network Node (Node-00001 - primary node)|g' 5-Applications/scripts/all_device_signal_topology.py 2>/dev/null || true
|
|
||||||
fi
|
|
||||||
|
|
||||||
echo ""
|
|
||||||
echo "=========================================="
|
|
||||||
echo "Done. Stale references cleaned."
|
|
||||||
echo ""
|
|
||||||
echo "Review changes with:"
|
|
||||||
echo " git diff .git/config"
|
|
||||||
echo " git diff .claude/settings.local.json"
|
|
||||||
echo " git diff 5-Applications/scripts/"
|
|
||||||
echo ""
|
|
||||||
echo "If satisfied, commit with:"
|
|
||||||
echo " git add -A && git commit -m 'chore: remove stale tailscale node references'"
|
|
||||||
|
|
@ -1,117 +0,0 @@
|
||||||
#!/bin/bash
|
|
||||||
set -e
|
|
||||||
|
|
||||||
# Configuration
|
|
||||||
WORKSPACE="/home/allaun/Documents/Research Stack"
|
|
||||||
MANIFEST_DIR="$WORKSPACE/.consolidation-manifests"
|
|
||||||
DATE=$(date +%Y-%m-%d_%H-%M-%S)
|
|
||||||
MASTER="$MANIFEST_DIR/MASTER-$DATE.sha256"
|
|
||||||
|
|
||||||
mkdir -p "$MANIFEST_DIR"
|
|
||||||
mkdir -p "$WORKSPACE/3-Mathematical-Models/genetics"
|
|
||||||
mkdir -p "$WORKSPACE/2-Search-Space/simulations"
|
|
||||||
mkdir -p "$WORKSPACE/5-Applications/pist-scripts"
|
|
||||||
mkdir -p "$WORKSPACE/6-Documentation/papers"
|
|
||||||
|
|
||||||
echo "=== Research Stack Consolidation & Hashing ==="
|
|
||||||
echo "Date: $DATE"
|
|
||||||
echo "Workspace: $WORKSPACE"
|
|
||||||
|
|
||||||
hash_and_move() {
|
|
||||||
local src="$1"
|
|
||||||
local dest="$2"
|
|
||||||
local name=$(basename "$src")
|
|
||||||
local manifest="$MANIFEST_DIR/$(echo "$src" | tr '/' '-')-$DATE.sha256"
|
|
||||||
|
|
||||||
if [ -e "$src" ]; then
|
|
||||||
echo "Processing: $src"
|
|
||||||
find "$src" -type f -print0 | xargs -0 sha256sum > "$manifest"
|
|
||||||
cat "$manifest" >> "$MASTER"
|
|
||||||
|
|
||||||
# Ensure parent dest exists
|
|
||||||
mkdir -p "$(dirname "$dest")"
|
|
||||||
|
|
||||||
cp -r "$src" "$dest"
|
|
||||||
echo "Copied to: $dest"
|
|
||||||
# rm -rf "$src" # We'll do a final cleanup after user confirms
|
|
||||||
else
|
|
||||||
echo "Skipping: $src (not found)"
|
|
||||||
fi
|
|
||||||
}
|
|
||||||
|
|
||||||
# 1. Fold external repos into monorepo
|
|
||||||
repos=(
|
|
||||||
"braid-field-papers|6-Documentation/papers/braid-field-papers"
|
|
||||||
"AMMR|3-Mathematical-Models/AMMR"
|
|
||||||
"bezier-kit|3-Mathematical-Models/bezier-kit"
|
|
||||||
"Newtonian-Superfluid-Simulation|2-Search-Space/simulations/Newtonian-Superfluid-Simulation"
|
|
||||||
"heat-2D|2-Search-Space/simulations/heat-2D"
|
|
||||||
"chunked-audio-DSP|2-Search-Space/simulations/chunked-audio-DSP"
|
|
||||||
"matter-frequencies|2-Search-Space/simulations/matter-frequencies"
|
|
||||||
"Allelica|3-Mathematical-Models/genetics/Allelica"
|
|
||||||
"parametric-learn|3-Mathematical-Models/genetics/parametric-learn"
|
|
||||||
"NoDupeLabs|4-Infrastructure/NoDupeLabs"
|
|
||||||
)
|
|
||||||
|
|
||||||
TEMP_CLONE="/tmp/research_stack_clones"
|
|
||||||
mkdir -p "$TEMP_CLONE"
|
|
||||||
|
|
||||||
for entry in "${repos[@]}"; do
|
|
||||||
repo_name="${entry%%|*}"
|
|
||||||
dest_path="${entry#*|}"
|
|
||||||
|
|
||||||
echo "Folding $repo_name..."
|
|
||||||
if [ ! -d "$WORKSPACE/$dest_path" ]; then
|
|
||||||
git clone "https://github.com/allaunthefox/$repo_name.git" "$TEMP_CLONE/$repo_name" --depth 1
|
|
||||||
rm -rf "$TEMP_CLONE/$repo_name/.git"
|
|
||||||
cp -r "$TEMP_CLONE/$repo_name" "$WORKSPACE/$dest_path"
|
|
||||||
rm -rf "$TEMP_CLONE/$repo_name"
|
|
||||||
else
|
|
||||||
echo "Skipping $repo_name (already exists in workspace)"
|
|
||||||
fi
|
|
||||||
done
|
|
||||||
|
|
||||||
# 2. Move loose files from Desktop
|
|
||||||
hash_and_move "/home/allaun/Desktop/manifold_compression" "$WORKSPACE/3-Mathematical-Models/manifold_compression"
|
|
||||||
hash_and_move "/home/allaun/Desktop/pist_biological_polymorphic_shifter_v3.py" "$WORKSPACE/5-Applications/pist-scripts/"
|
|
||||||
hash_and_move "/home/allaun/Desktop/pist_biological_polymorphic_shifter_v3_complete.py" "$WORKSPACE/5-Applications/pist-scripts/"
|
|
||||||
hash_and_move "/home/allaun/Desktop/pist_gcl_compression.py" "$WORKSPACE/5-Applications/pist-scripts/"
|
|
||||||
|
|
||||||
# 3. Move loose folders from Documents
|
|
||||||
hash_and_move "/home/allaun/Documents/Semantics" "$WORKSPACE/0-Core-Formalism/lean/Semantics"
|
|
||||||
hash_and_move "/home/allaun/Documents/projects/hutter_prize" "$WORKSPACE/5-Applications/hutter_prize"
|
|
||||||
hash_and_move "/home/allaun/Documents/projects/teleport-kanban" "$WORKSPACE/5-Applications/teleport-kanban"
|
|
||||||
|
|
||||||
# 4. Cleanup redundant directories at /home/allaun (if confirmed)
|
|
||||||
# These were marked as duplicates/stale in previous audits
|
|
||||||
redundant=(
|
|
||||||
"/home/allaun/Desktop/OTOM"
|
|
||||||
"/home/allaun/Documents/DeleteMe"
|
|
||||||
"/home/allaun/Documents/Research Stack-backups"
|
|
||||||
"/home/allaun/Documents/Forked"
|
|
||||||
"/home/allaun/OTOM"
|
|
||||||
"/home/allaun/NoDupeLabs"
|
|
||||||
"/home/allaun/tardygrada-Organism"
|
|
||||||
"/home/allaun/claw-code"
|
|
||||||
"/home/allaun/latex_demo"
|
|
||||||
"/home/allaun/Research Stack" # This is likely a debris folder if WORKSPACE is in Documents
|
|
||||||
)
|
|
||||||
|
|
||||||
for dir in "${redundant[@]}"; do
|
|
||||||
if [ -d "$dir" ] && [ "$dir" != "$WORKSPACE" ]; then
|
|
||||||
echo "Removing redundant: $dir"
|
|
||||||
rm -rf "$dir"
|
|
||||||
fi
|
|
||||||
done
|
|
||||||
|
|
||||||
# 5. Replace CascadeProjects symlink
|
|
||||||
if [ -d "/home/allaun/CascadeProjects/Research-Stack" ]; then
|
|
||||||
echo "Replacing CascadeProjects mirror with symlink"
|
|
||||||
rm -rf "/home/allaun/CascadeProjects/Research-Stack"
|
|
||||||
ln -s "$WORKSPACE" "/home/allaun/CascadeProjects/Research-Stack"
|
|
||||||
fi
|
|
||||||
|
|
||||||
echo "=== Consolidation Complete ==="
|
|
||||||
echo "Master manifest: $MASTER"
|
|
||||||
echo "Please verify the contents of $WORKSPACE"
|
|
||||||
echo "Next: Archive/Delete the repos on GitHub using scripts/archive-and-delete-v2.sh"
|
|
||||||
|
|
@ -1,214 +0,0 @@
|
||||||
#!/usr/bin/env python3
|
|
||||||
"""Require math evidence alongside math-track edits.
|
|
||||||
|
|
||||||
When a commit (or PR) touches files under one of the math-track surfaces,
|
|
||||||
this script asserts that at least one file under a math-evidence surface is
|
|
||||||
also part of the same change set. The two surfaces are configurable -- the
|
|
||||||
defaults below match ``docs/math-first-tooling.md`` and the pre-commit hook
|
|
||||||
declared in ``.pre-commit-config.yaml``.
|
|
||||||
|
|
||||||
Math-track surfaces (need evidence):
|
|
||||||
- 0-Core-Formalism/lean/Semantics/...
|
|
||||||
- 6-Documentation/docs/distilled/...
|
|
||||||
- shared-data/data/stack_solidification/...
|
|
||||||
|
|
||||||
Math-evidence surfaces (accepted as evidence):
|
|
||||||
- shared-data/artifacts/deepseek_review/*.receipt.json
|
|
||||||
- 0-Core-Formalism/lean/Semantics/... (a Lean change in the same commit
|
|
||||||
counts because Lean is the source
|
|
||||||
of truth per AGENTS.md)
|
|
||||||
- claims.yaml (registry update)
|
|
||||||
|
|
||||||
Usage:
|
|
||||||
scripts/math-first/require_math_evidence.py [FILES ...]
|
|
||||||
scripts/math-first/require_math_evidence.py --staged
|
|
||||||
scripts/math-first/require_math_evidence.py --from-git-diff BASE_REF
|
|
||||||
|
|
||||||
If no input mode is supplied the script exits 0 with a noop. Pre-commit
|
|
||||||
invokes it with ``--staged`` so the script sees the entire staged set
|
|
||||||
regardless of pre-commit's own ``files`` filter; CI invokes it with
|
|
||||||
``--from-git-diff origin/<base>``.
|
|
||||||
|
|
||||||
Exit code:
|
|
||||||
0 evidence present, nothing to do, or no math-track files changed.
|
|
||||||
1 math-track files changed without accompanying evidence.
|
|
||||||
"""
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import argparse
|
|
||||||
import subprocess
|
|
||||||
import sys
|
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
|
|
||||||
def _git_toplevel(cwd: Path | None = None) -> Path:
|
|
||||||
"""Return the absolute path of the git working tree containing ``cwd``.
|
|
||||||
|
|
||||||
Defaults to the current working directory. The script never assumes the
|
|
||||||
git repo lives at a hardcoded path because that would be wrong when the
|
|
||||||
script is run from a different working tree -- notably, the temp repo
|
|
||||||
set up by ``test_require_math_evidence.py``. Pre-commit and CI both
|
|
||||||
happen to invoke the script from inside the repo root, so letting the
|
|
||||||
git subprocess inherit cwd is correct in every real-world case.
|
|
||||||
"""
|
|
||||||
cmd = ["git", "rev-parse", "--show-toplevel"]
|
|
||||||
try:
|
|
||||||
result = subprocess.run(
|
|
||||||
cmd,
|
|
||||||
capture_output=True,
|
|
||||||
text=True,
|
|
||||||
check=True,
|
|
||||||
cwd=str(cwd) if cwd is not None else None,
|
|
||||||
)
|
|
||||||
except subprocess.CalledProcessError as exc:
|
|
||||||
print(
|
|
||||||
f"error: `{' '.join(cmd)}` failed: {exc.stderr.strip()}",
|
|
||||||
file=sys.stderr,
|
|
||||||
)
|
|
||||||
raise SystemExit(2)
|
|
||||||
return Path(result.stdout.strip())
|
|
||||||
|
|
||||||
|
|
||||||
MATH_TRACK_PREFIXES: tuple[str, ...] = (
|
|
||||||
"0-Core-Formalism/lean/Semantics/",
|
|
||||||
"6-Documentation/docs/distilled/",
|
|
||||||
"shared-data/data/stack_solidification/",
|
|
||||||
)
|
|
||||||
|
|
||||||
EVIDENCE_PREFIXES: tuple[str, ...] = (
|
|
||||||
"shared-data/artifacts/deepseek_review/",
|
|
||||||
"0-Core-Formalism/lean/Semantics/",
|
|
||||||
)
|
|
||||||
|
|
||||||
EVIDENCE_FILES: tuple[str, ...] = (
|
|
||||||
"claims.yaml",
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
def _normalise(path: str) -> str:
|
|
||||||
return path.replace("\\", "/")
|
|
||||||
|
|
||||||
|
|
||||||
def _is_math_track(path: str) -> bool:
|
|
||||||
norm = _normalise(path)
|
|
||||||
return any(norm.startswith(prefix) for prefix in MATH_TRACK_PREFIXES)
|
|
||||||
|
|
||||||
|
|
||||||
def _is_evidence(path: str) -> bool:
|
|
||||||
norm = _normalise(path)
|
|
||||||
if norm in EVIDENCE_FILES:
|
|
||||||
return True
|
|
||||||
if any(norm.startswith(prefix) for prefix in EVIDENCE_PREFIXES):
|
|
||||||
# A *new or updated* receipt counts. A bare Lean kernel edit also
|
|
||||||
# counts because Lean is treated as the source of truth -- the change
|
|
||||||
# itself is the evidence.
|
|
||||||
if norm.startswith("shared-data/artifacts/deepseek_review/"):
|
|
||||||
return norm.endswith(".receipt.json") or norm.endswith(".md")
|
|
||||||
return True
|
|
||||||
return False
|
|
||||||
|
|
||||||
|
|
||||||
def _files_from_git_diff(base_ref: str) -> list[str]:
|
|
||||||
cwd = _git_toplevel()
|
|
||||||
cmd = ["git", "diff", "--name-only", f"{base_ref}...HEAD"]
|
|
||||||
try:
|
|
||||||
result = subprocess.run(
|
|
||||||
cmd, capture_output=True, text=True, check=True, cwd=str(cwd)
|
|
||||||
)
|
|
||||||
except subprocess.CalledProcessError as exc:
|
|
||||||
print(f"error: `{' '.join(cmd)}` failed: {exc.stderr.strip()}", file=sys.stderr)
|
|
||||||
raise SystemExit(2)
|
|
||||||
return [line.strip() for line in result.stdout.splitlines() if line.strip()]
|
|
||||||
|
|
||||||
|
|
||||||
def _files_from_staged() -> list[str]:
|
|
||||||
"""Return the staged file list via ``git diff --cached --name-only``.
|
|
||||||
|
|
||||||
Used by the pre-commit hook so the script sees every staged file --
|
|
||||||
math-track *and* evidence -- regardless of pre-commit's per-hook
|
|
||||||
``files`` filter. This is important because pre-commit otherwise strips
|
|
||||||
receipts and ``claims.yaml`` from the argv before the script ever sees
|
|
||||||
them, which would cause the evidence check to falsely fail.
|
|
||||||
|
|
||||||
The subprocess cwd is the actual git toplevel containing the *current*
|
|
||||||
working directory, not a hardcoded path -- so the script works
|
|
||||||
correctly inside the regression test's temp repo too.
|
|
||||||
"""
|
|
||||||
cwd = _git_toplevel()
|
|
||||||
cmd = ["git", "diff", "--cached", "--name-only"]
|
|
||||||
try:
|
|
||||||
result = subprocess.run(
|
|
||||||
cmd, capture_output=True, text=True, check=True, cwd=str(cwd)
|
|
||||||
)
|
|
||||||
except subprocess.CalledProcessError as exc:
|
|
||||||
print(f"error: `{' '.join(cmd)}` failed: {exc.stderr.strip()}", file=sys.stderr)
|
|
||||||
raise SystemExit(2)
|
|
||||||
return [line.strip() for line in result.stdout.splitlines() if line.strip()]
|
|
||||||
|
|
||||||
|
|
||||||
def main(argv: list[str] | None = None) -> int:
|
|
||||||
parser = argparse.ArgumentParser(description=__doc__)
|
|
||||||
parser.add_argument(
|
|
||||||
"files",
|
|
||||||
nargs="*",
|
|
||||||
help="Explicit list of files to check (typically supplied by pre-commit).",
|
|
||||||
)
|
|
||||||
parser.add_argument(
|
|
||||||
"--from-git-diff",
|
|
||||||
metavar="BASE_REF",
|
|
||||||
help="Compute the file list from `git diff --name-only BASE_REF...HEAD`.",
|
|
||||||
)
|
|
||||||
parser.add_argument(
|
|
||||||
"--staged",
|
|
||||||
action="store_true",
|
|
||||||
help="Compute the file list from `git diff --cached --name-only` "
|
|
||||||
"(use this from pre-commit so the entire staged set is visible).",
|
|
||||||
)
|
|
||||||
args = parser.parse_args(argv)
|
|
||||||
|
|
||||||
if sum(bool(x) for x in (args.from_git_diff, args.staged, args.files)) > 1:
|
|
||||||
parser.error("--from-git-diff, --staged, and explicit FILES are mutually exclusive")
|
|
||||||
|
|
||||||
if args.from_git_diff:
|
|
||||||
files = _files_from_git_diff(args.from_git_diff)
|
|
||||||
elif args.staged:
|
|
||||||
files = _files_from_staged()
|
|
||||||
else:
|
|
||||||
files = list(args.files)
|
|
||||||
|
|
||||||
if not files:
|
|
||||||
return 0
|
|
||||||
|
|
||||||
math_track = sorted({f for f in files if _is_math_track(f)})
|
|
||||||
evidence = sorted({f for f in files if _is_evidence(f)})
|
|
||||||
|
|
||||||
if not math_track:
|
|
||||||
return 0
|
|
||||||
|
|
||||||
if evidence:
|
|
||||||
print("math-evidence check: OK")
|
|
||||||
print(" math-track files:")
|
|
||||||
for path in math_track:
|
|
||||||
print(f" - {path}")
|
|
||||||
print(" evidence files:")
|
|
||||||
for path in evidence:
|
|
||||||
print(f" - {path}")
|
|
||||||
return 0
|
|
||||||
|
|
||||||
print("math-evidence check: FAIL", file=sys.stderr)
|
|
||||||
print(" math-track files changed without accompanying evidence:", file=sys.stderr)
|
|
||||||
for path in math_track:
|
|
||||||
print(f" - {path}", file=sys.stderr)
|
|
||||||
print(
|
|
||||||
"\n Add at least one of:\n"
|
|
||||||
" - a DeepSeek review receipt under shared-data/artifacts/deepseek_review/\n"
|
|
||||||
" - a Lean change under 0-Core-Formalism/lean/Semantics/\n"
|
|
||||||
" - a claims.yaml update\n"
|
|
||||||
" See docs/math-first-tooling.md.",
|
|
||||||
file=sys.stderr,
|
|
||||||
)
|
|
||||||
return 1
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
raise SystemExit(main())
|
|
||||||
|
|
@ -1,235 +0,0 @@
|
||||||
#!/usr/bin/env python3
|
|
||||||
"""Self-check suite for ``require_math_evidence.py``.
|
|
||||||
|
|
||||||
Covers the classification logic plus the pre-commit-vs-CI regression that
|
|
||||||
motivated the follow-up fix: pre-commit applies its per-hook ``files``
|
|
||||||
filter before the script runs, so the script must read the full staged
|
|
||||||
set itself (via ``--staged``) rather than rely on the argv list.
|
|
||||||
|
|
||||||
Exit code:
|
|
||||||
0 all cases passed.
|
|
||||||
1 at least one case failed.
|
|
||||||
"""
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import os
|
|
||||||
import subprocess
|
|
||||||
import sys
|
|
||||||
import tempfile
|
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
SCRIPT = Path(__file__).with_name("require_math_evidence.py")
|
|
||||||
|
|
||||||
|
|
||||||
def _run(args: list[str], cwd: Path | None = None, env: dict[str, str] | None = None) -> tuple[int, str, str]:
|
|
||||||
cmd = [sys.executable, str(SCRIPT), *args]
|
|
||||||
result = subprocess.run(
|
|
||||||
cmd,
|
|
||||||
capture_output=True,
|
|
||||||
text=True,
|
|
||||||
cwd=str(cwd) if cwd else None,
|
|
||||||
env=env,
|
|
||||||
)
|
|
||||||
return result.returncode, result.stdout, result.stderr
|
|
||||||
|
|
||||||
|
|
||||||
def _git(*args: str, cwd: Path) -> subprocess.CompletedProcess[str]:
|
|
||||||
return subprocess.run(
|
|
||||||
["git", *args],
|
|
||||||
capture_output=True,
|
|
||||||
text=True,
|
|
||||||
check=True,
|
|
||||||
cwd=str(cwd),
|
|
||||||
env={
|
|
||||||
**os.environ,
|
|
||||||
"GIT_AUTHOR_NAME": "test",
|
|
||||||
"GIT_AUTHOR_EMAIL": "test@example.invalid",
|
|
||||||
"GIT_COMMITTER_NAME": "test",
|
|
||||||
"GIT_COMMITTER_EMAIL": "test@example.invalid",
|
|
||||||
},
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
CASES: list[tuple[str, list[str], int]] = [
|
|
||||||
# (label, argv, expected exit code)
|
|
||||||
("noop_empty", [], 0),
|
|
||||||
("only_evidence_receipt", ["shared-data/artifacts/deepseek_review/x.receipt.json"], 0),
|
|
||||||
("only_evidence_claims", ["claims.yaml"], 0),
|
|
||||||
("only_unrelated", ["README.md", "src/foo.py"], 0),
|
|
||||||
("lean_self_evidence", ["0-Core-Formalism/lean/Semantics/Kernel.lean"], 0),
|
|
||||||
("doc_without_evidence", ["6-Documentation/docs/distilled/Spec.md"], 1),
|
|
||||||
(
|
|
||||||
"doc_with_receipt",
|
|
||||||
[
|
|
||||||
"6-Documentation/docs/distilled/Spec.md",
|
|
||||||
"shared-data/artifacts/deepseek_review/some.receipt.json",
|
|
||||||
],
|
|
||||||
0,
|
|
||||||
),
|
|
||||||
(
|
|
||||||
"doc_with_claims",
|
|
||||||
["6-Documentation/docs/distilled/Spec.md", "claims.yaml"],
|
|
||||||
0,
|
|
||||||
),
|
|
||||||
(
|
|
||||||
"stack_solidification_without_evidence",
|
|
||||||
["shared-data/data/stack_solidification/foo.json"],
|
|
||||||
1,
|
|
||||||
),
|
|
||||||
(
|
|
||||||
"stack_solidification_with_lean",
|
|
||||||
[
|
|
||||||
"shared-data/data/stack_solidification/foo.json",
|
|
||||||
"0-Core-Formalism/lean/Semantics/Bar.lean",
|
|
||||||
],
|
|
||||||
0,
|
|
||||||
),
|
|
||||||
]
|
|
||||||
|
|
||||||
|
|
||||||
def _run_argv_cases() -> int:
|
|
||||||
failures = 0
|
|
||||||
for label, argv, expected in CASES:
|
|
||||||
code, stdout, stderr = _run(argv)
|
|
||||||
if code != expected:
|
|
||||||
failures += 1
|
|
||||||
print(f"FAIL {label}: expected exit {expected}, got {code}")
|
|
||||||
if stdout.strip():
|
|
||||||
print(f" stdout: {stdout.strip()}")
|
|
||||||
if stderr.strip():
|
|
||||||
print(f" stderr: {stderr.strip()}")
|
|
||||||
else:
|
|
||||||
print(f"OK {label} (exit {code})")
|
|
||||||
return failures
|
|
||||||
|
|
||||||
|
|
||||||
def _run_mutex_check() -> int:
|
|
||||||
"""`--staged` and `--from-git-diff` and explicit FILES are mutually exclusive."""
|
|
||||||
failures = 0
|
|
||||||
for argv in (
|
|
||||||
["--staged", "--from-git-diff", "main"],
|
|
||||||
["--staged", "claims.yaml"],
|
|
||||||
["--from-git-diff", "main", "claims.yaml"],
|
|
||||||
):
|
|
||||||
code, _stdout, stderr = _run(argv)
|
|
||||||
if code != 2:
|
|
||||||
failures += 1
|
|
||||||
print(f"FAIL mutex {argv}: expected exit 2, got {code}")
|
|
||||||
if stderr.strip():
|
|
||||||
print(f" stderr: {stderr.strip()}")
|
|
||||||
else:
|
|
||||||
print(f"OK mutex {argv} (exit {code})")
|
|
||||||
return failures
|
|
||||||
|
|
||||||
|
|
||||||
def _run_staged_regression() -> int:
|
|
||||||
"""Regression for the pre-commit ``files``-filter bug.
|
|
||||||
|
|
||||||
Builds a throwaway git repo so this test is self-contained -- the
|
|
||||||
test does not depend on the state of the real repo's index. The
|
|
||||||
script is run with ``cwd=tmp_path`` (NOT via ``runpy``) so that
|
|
||||||
``git rev-parse --show-toplevel`` inside the script resolves to the
|
|
||||||
temp repo, and ``git diff --cached`` queries the temp repo's index.
|
|
||||||
|
|
||||||
Three sub-cases:
|
|
||||||
(a) math-track-only staged -> expect exit 1
|
|
||||||
(proves the script actually evaluates classification logic;
|
|
||||||
without (a) the next sub-case could pass vacuously by
|
|
||||||
short-circuiting on an empty diff)
|
|
||||||
(b) math-track + receipt staged -> expect exit 0
|
|
||||||
(the original pre-commit ``files``-filter bug)
|
|
||||||
(c) math-track + claims.yaml staged -> expect exit 0
|
|
||||||
(verifies the registry-update path)
|
|
||||||
"""
|
|
||||||
failures = 0
|
|
||||||
with tempfile.TemporaryDirectory() as tmp:
|
|
||||||
tmp_path = Path(tmp).resolve()
|
|
||||||
_git("init", "-q", cwd=tmp_path)
|
|
||||||
|
|
||||||
math_track = tmp_path / "6-Documentation" / "docs" / "distilled" / "Spec.md"
|
|
||||||
math_track.parent.mkdir(parents=True, exist_ok=True)
|
|
||||||
math_track.write_text("# math claim\n")
|
|
||||||
|
|
||||||
receipt = (
|
|
||||||
tmp_path / "shared-data" / "artifacts" / "deepseek_review" / "x.receipt.json"
|
|
||||||
)
|
|
||||||
receipt.parent.mkdir(parents=True, exist_ok=True)
|
|
||||||
receipt.write_text("{}\n")
|
|
||||||
|
|
||||||
claims = tmp_path / "claims.yaml"
|
|
||||||
claims.write_text("claims: []\n")
|
|
||||||
|
|
||||||
def _stage_only(*relpaths: str) -> None:
|
|
||||||
# Reset the index to a clean state, then stage exactly the
|
|
||||||
# supplied paths. ``git reset`` is safe here -- the temp repo
|
|
||||||
# has no commits, so there is no "HEAD" to reset against. We
|
|
||||||
# instead remove everything currently in the index.
|
|
||||||
_git("rm", "--cached", "-rf", "--ignore-unmatch", ".", cwd=tmp_path)
|
|
||||||
for relpath in relpaths:
|
|
||||||
_git("add", "--", relpath, cwd=tmp_path)
|
|
||||||
|
|
||||||
def _invoke() -> subprocess.CompletedProcess[str]:
|
|
||||||
return subprocess.run(
|
|
||||||
[sys.executable, str(SCRIPT), "--staged"],
|
|
||||||
capture_output=True,
|
|
||||||
text=True,
|
|
||||||
cwd=str(tmp_path),
|
|
||||||
)
|
|
||||||
|
|
||||||
def _assert(label: str, expected_exit: int, *staged: str) -> int:
|
|
||||||
_stage_only(*staged)
|
|
||||||
indexed = _git("diff", "--cached", "--name-only", cwd=tmp_path).stdout.splitlines()
|
|
||||||
indexed = [line for line in indexed if line.strip()]
|
|
||||||
if sorted(indexed) != sorted(staged):
|
|
||||||
print(f"FAIL {label}: index does not match expected staging")
|
|
||||||
print(f" expected: {sorted(staged)}")
|
|
||||||
print(f" actual: {sorted(indexed)}")
|
|
||||||
return 1
|
|
||||||
result = _invoke()
|
|
||||||
if result.returncode != expected_exit:
|
|
||||||
print(
|
|
||||||
f"FAIL {label}: expected exit {expected_exit}, "
|
|
||||||
f"got {result.returncode}"
|
|
||||||
)
|
|
||||||
if result.stdout.strip():
|
|
||||||
print(f" stdout: {result.stdout.strip()}")
|
|
||||||
if result.stderr.strip():
|
|
||||||
print(f" stderr: {result.stderr.strip()}")
|
|
||||||
return 1
|
|
||||||
print(f"OK {label} (exit {expected_exit})")
|
|
||||||
return 0
|
|
||||||
|
|
||||||
failures += _assert(
|
|
||||||
"staged_regression_negative (math-only -> FAIL)",
|
|
||||||
1,
|
|
||||||
"6-Documentation/docs/distilled/Spec.md",
|
|
||||||
)
|
|
||||||
failures += _assert(
|
|
||||||
"staged_regression_positive_receipt (math + receipt -> OK)",
|
|
||||||
0,
|
|
||||||
"6-Documentation/docs/distilled/Spec.md",
|
|
||||||
"shared-data/artifacts/deepseek_review/x.receipt.json",
|
|
||||||
)
|
|
||||||
failures += _assert(
|
|
||||||
"staged_regression_positive_claims (math + claims.yaml -> OK)",
|
|
||||||
0,
|
|
||||||
"6-Documentation/docs/distilled/Spec.md",
|
|
||||||
"claims.yaml",
|
|
||||||
)
|
|
||||||
return failures
|
|
||||||
|
|
||||||
|
|
||||||
def main() -> int:
|
|
||||||
failures = 0
|
|
||||||
failures += _run_argv_cases()
|
|
||||||
failures += _run_mutex_check()
|
|
||||||
failures += _run_staged_regression()
|
|
||||||
if failures:
|
|
||||||
print(f"\n{failures} case(s) FAILED")
|
|
||||||
return 1
|
|
||||||
print("\nAll require_math_evidence self-checks passed.")
|
|
||||||
return 0
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
sys.exit(main())
|
|
||||||
|
|
@ -1,134 +0,0 @@
|
||||||
#!/usr/bin/env python3
|
|
||||||
"""Self-checks for ``validate_deepseek_receipts.py``.
|
|
||||||
|
|
||||||
Run with::
|
|
||||||
|
|
||||||
uv run --python 3.11 --with "jsonschema>=4.21" --with "rfc3339-validator" \
|
|
||||||
python3 scripts/math-first/test_validate_deepseek_receipts.py
|
|
||||||
|
|
||||||
The test builds positive and negative receipt fixtures in a temporary
|
|
||||||
directory, invokes the validator as a subprocess, and asserts the exit code
|
|
||||||
matches the expected outcome. The fixtures are derived from
|
|
||||||
``shared-data/artifacts/deepseek_review/`` so they exercise the same shape
|
|
||||||
that ships in the repo.
|
|
||||||
"""
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import json
|
|
||||||
import subprocess
|
|
||||||
import sys
|
|
||||||
import tempfile
|
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
REPO_ROOT = Path(__file__).resolve().parents[2]
|
|
||||||
VALIDATOR = REPO_ROOT / "scripts" / "math-first" / "validate_deepseek_receipts.py"
|
|
||||||
|
|
||||||
GOOD_PRIMARY = {
|
|
||||||
"schema": "ollama_deepseek_review_receipt_v1",
|
|
||||||
"created_at": "2026-05-12T03:35:51+00:00",
|
|
||||||
"model": "deepseek-v3.2",
|
|
||||||
"endpoint": "https://ollama.com/v1/chat/completions",
|
|
||||||
"prompt_sha256": "sha256:" + "a" * 64,
|
|
||||||
"answer_sha256": "sha256:" + "b" * 64,
|
|
||||||
"usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2},
|
|
||||||
"context_files": ["docs/example.md"],
|
|
||||||
"answer_path": "shared-data/artifacts/deepseek_review/example.md",
|
|
||||||
}
|
|
||||||
|
|
||||||
GOOD_CONTINUATION = {
|
|
||||||
"schema": "ollama_deepseek_review_continuation_receipt_v1",
|
|
||||||
"created_at": "2026-05-12T03:38:49+00:00",
|
|
||||||
"model": "deepseek-v4-flash",
|
|
||||||
"endpoint": "https://ollama.com/v1/chat/completions",
|
|
||||||
"prompt_sha256": "sha256:" + "c" * 64,
|
|
||||||
"answer_sha256": "sha256:" + "d" * 64,
|
|
||||||
"usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2},
|
|
||||||
"previous_answer_path": "shared-data/artifacts/deepseek_review/example.md",
|
|
||||||
"answer_path": "shared-data/artifacts/deepseek_review/example_continuation.md",
|
|
||||||
"message_keys": ["role", "content", "reasoning"],
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
def _write(tmp: Path, name: str, payload: dict | str) -> Path:
|
|
||||||
path = tmp / name
|
|
||||||
if isinstance(payload, str):
|
|
||||||
path.write_text(payload, encoding="utf-8")
|
|
||||||
else:
|
|
||||||
path.write_text(json.dumps(payload), encoding="utf-8")
|
|
||||||
return path
|
|
||||||
|
|
||||||
|
|
||||||
def _run(path: Path) -> subprocess.CompletedProcess[str]:
|
|
||||||
return subprocess.run(
|
|
||||||
[sys.executable, str(VALIDATOR), str(path)],
|
|
||||||
capture_output=True,
|
|
||||||
text=True,
|
|
||||||
check=False,
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
def main() -> int:
|
|
||||||
failures: list[str] = []
|
|
||||||
|
|
||||||
with tempfile.TemporaryDirectory() as raw:
|
|
||||||
tmp = Path(raw)
|
|
||||||
|
|
||||||
cases: list[tuple[str, dict | str, int]] = [
|
|
||||||
("good_primary.receipt.json", GOOD_PRIMARY, 0),
|
|
||||||
("good_continuation.receipt.json", GOOD_CONTINUATION, 0),
|
|
||||||
(
|
|
||||||
"bad_sha256.receipt.json",
|
|
||||||
{**GOOD_PRIMARY, "prompt_sha256": "not-a-hash"},
|
|
||||||
1,
|
|
||||||
),
|
|
||||||
(
|
|
||||||
"bad_schema_id.receipt.json",
|
|
||||||
{**GOOD_PRIMARY, "schema": "made_up_schema_id"},
|
|
||||||
1,
|
|
||||||
),
|
|
||||||
(
|
|
||||||
"bad_missing_usage.receipt.json",
|
|
||||||
{k: v for k, v in GOOD_PRIMARY.items() if k != "usage"},
|
|
||||||
1,
|
|
||||||
),
|
|
||||||
(
|
|
||||||
"bad_negative_tokens.receipt.json",
|
|
||||||
{**GOOD_PRIMARY, "usage": {"prompt_tokens": -1, "completion_tokens": 1, "total_tokens": 0}},
|
|
||||||
1,
|
|
||||||
),
|
|
||||||
(
|
|
||||||
"bad_answer_path_ext.receipt.json",
|
|
||||||
{**GOOD_PRIMARY, "answer_path": "shared-data/artifacts/deepseek_review/example.txt"},
|
|
||||||
1,
|
|
||||||
),
|
|
||||||
(
|
|
||||||
"bad_extra_field.receipt.json",
|
|
||||||
{**GOOD_PRIMARY, "stray": 1},
|
|
||||||
1,
|
|
||||||
),
|
|
||||||
("bad_not_json.receipt.json", "{not json}", 1),
|
|
||||||
]
|
|
||||||
|
|
||||||
for name, payload, expected in cases:
|
|
||||||
path = _write(tmp, name, payload)
|
|
||||||
result = _run(path)
|
|
||||||
if result.returncode != expected:
|
|
||||||
failures.append(
|
|
||||||
f"{name}: expected exit {expected}, got {result.returncode}\n"
|
|
||||||
f" stdout: {result.stdout.strip()}\n"
|
|
||||||
f" stderr: {result.stderr.strip()}"
|
|
||||||
)
|
|
||||||
else:
|
|
||||||
print(f"OK {name} (exit {result.returncode})")
|
|
||||||
|
|
||||||
if failures:
|
|
||||||
print("\nFailures:")
|
|
||||||
for line in failures:
|
|
||||||
print(line)
|
|
||||||
return 1
|
|
||||||
print("\nAll validator self-checks passed.")
|
|
||||||
return 0
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
raise SystemExit(main())
|
|
||||||
|
|
@ -1,190 +0,0 @@
|
||||||
#!/usr/bin/env python3
|
|
||||||
"""Validate ``claims.yaml`` against the claims-registry JSON Schema.
|
|
||||||
|
|
||||||
Usage:
|
|
||||||
scripts/math-first/validate_claims_registry.py [PATH]
|
|
||||||
|
|
||||||
When no PATH is provided, the registry at the repo root (``claims.yaml``) is
|
|
||||||
validated. The script enforces:
|
|
||||||
|
|
||||||
* the YAML parses and conforms to
|
|
||||||
``shared-data/schemas/claims-registry.schema.json``;
|
|
||||||
* every ``id`` is unique across the registry;
|
|
||||||
* every repo-relative path referenced from ``lean``, ``review_receipts``,
|
|
||||||
and ``sources`` resolves to a tracked file on disk (external citations
|
|
||||||
that do not look like repo paths -- e.g. ``http`` URLs, ``arXiv:...`` --
|
|
||||||
are skipped).
|
|
||||||
|
|
||||||
Exit code:
|
|
||||||
0 registry valid.
|
|
||||||
1 registry invalid (schema, duplicate id, or missing referenced file).
|
|
||||||
2 schema malformed, dependencies missing, or registry file not found.
|
|
||||||
|
|
||||||
See ``docs/math-first-tooling.md`` for the math-first tooling contract.
|
|
||||||
"""
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import argparse
|
|
||||||
import json
|
|
||||||
import sys
|
|
||||||
from pathlib import Path
|
|
||||||
from typing import Any
|
|
||||||
|
|
||||||
REPO_ROOT = Path(__file__).resolve().parents[2]
|
|
||||||
SCHEMA_PATH = REPO_ROOT / "shared-data" / "schemas" / "claims-registry.schema.json"
|
|
||||||
DEFAULT_REGISTRY = REPO_ROOT / "claims.yaml"
|
|
||||||
|
|
||||||
# Anything matching one of these prefixes (case-insensitively) is treated
|
|
||||||
# as an external citation rather than a repo-relative path, and is therefore
|
|
||||||
# not required to resolve to a file on disk. The list is intentionally
|
|
||||||
# closed: matching every ``scheme:`` blob would silently skip path checks
|
|
||||||
# for anything that happens to contain a colon (e.g. ``Module:Theorem``),
|
|
||||||
# which would hide registry rot.
|
|
||||||
_EXTERNAL_PREFIXES: tuple[str, ...] = (
|
|
||||||
"http://",
|
|
||||||
"https://",
|
|
||||||
"arxiv:",
|
|
||||||
"doi:",
|
|
||||||
"isbn:",
|
|
||||||
"mailto:",
|
|
||||||
"urn:",
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
def _load_schema(schema_path: Path) -> dict[str, Any]:
|
|
||||||
try:
|
|
||||||
from jsonschema import Draft202012Validator
|
|
||||||
except ImportError as exc:
|
|
||||||
print(
|
|
||||||
"error: jsonschema>=4.18 is required (Draft 2020-12). "
|
|
||||||
"Install via `uv pip install jsonschema>=4.21 PyYAML`.",
|
|
||||||
file=sys.stderr,
|
|
||||||
)
|
|
||||||
raise SystemExit(2) from exc
|
|
||||||
|
|
||||||
try:
|
|
||||||
schema = json.loads(schema_path.read_text(encoding="utf-8"))
|
|
||||||
except FileNotFoundError:
|
|
||||||
print(f"error: schema not found at {schema_path}", file=sys.stderr)
|
|
||||||
raise SystemExit(2)
|
|
||||||
except json.JSONDecodeError as exc:
|
|
||||||
print(f"error: schema {schema_path} is not valid JSON: {exc}", file=sys.stderr)
|
|
||||||
raise SystemExit(2)
|
|
||||||
|
|
||||||
Draft202012Validator.check_schema(schema)
|
|
||||||
return schema
|
|
||||||
|
|
||||||
|
|
||||||
def _load_registry(registry_path: Path) -> dict[str, Any]:
|
|
||||||
try:
|
|
||||||
import yaml
|
|
||||||
except ImportError as exc:
|
|
||||||
print(
|
|
||||||
"error: PyYAML is required. Install via `uv pip install PyYAML`.",
|
|
||||||
file=sys.stderr,
|
|
||||||
)
|
|
||||||
raise SystemExit(2) from exc
|
|
||||||
|
|
||||||
try:
|
|
||||||
text = registry_path.read_text(encoding="utf-8")
|
|
||||||
except FileNotFoundError:
|
|
||||||
print(f"error: registry not found at {registry_path}", file=sys.stderr)
|
|
||||||
raise SystemExit(2)
|
|
||||||
|
|
||||||
data = yaml.safe_load(text)
|
|
||||||
if not isinstance(data, dict):
|
|
||||||
print(f"error: registry {registry_path} did not parse as a mapping", file=sys.stderr)
|
|
||||||
raise SystemExit(1)
|
|
||||||
return data
|
|
||||||
|
|
||||||
|
|
||||||
def _is_external(reference: str) -> bool:
|
|
||||||
lowered = reference.lower()
|
|
||||||
return any(lowered.startswith(prefix) for prefix in _EXTERNAL_PREFIXES)
|
|
||||||
|
|
||||||
|
|
||||||
def _check_path(reference: str) -> tuple[bool, str]:
|
|
||||||
if _is_external(reference):
|
|
||||||
return True, ""
|
|
||||||
if reference.startswith("/"):
|
|
||||||
return False, "must be repo-relative (no leading '/')"
|
|
||||||
candidate = REPO_ROOT / reference
|
|
||||||
if not candidate.exists():
|
|
||||||
return False, f"path does not exist on disk: {reference}"
|
|
||||||
return True, ""
|
|
||||||
|
|
||||||
|
|
||||||
def main(argv: list[str] | None = None) -> int:
|
|
||||||
parser = argparse.ArgumentParser(description=__doc__)
|
|
||||||
parser.add_argument(
|
|
||||||
"path",
|
|
||||||
nargs="?",
|
|
||||||
type=Path,
|
|
||||||
default=DEFAULT_REGISTRY,
|
|
||||||
help=f"Registry file to validate (default: {DEFAULT_REGISTRY.relative_to(REPO_ROOT)}).",
|
|
||||||
)
|
|
||||||
parser.add_argument(
|
|
||||||
"--schema",
|
|
||||||
type=Path,
|
|
||||||
default=SCHEMA_PATH,
|
|
||||||
help=f"Path to the JSON Schema (default: {SCHEMA_PATH.relative_to(REPO_ROOT)}).",
|
|
||||||
)
|
|
||||||
args = parser.parse_args(argv)
|
|
||||||
|
|
||||||
schema = _load_schema(args.schema)
|
|
||||||
registry = _load_registry(args.path)
|
|
||||||
|
|
||||||
from jsonschema import Draft202012Validator, FormatChecker
|
|
||||||
|
|
||||||
# FormatChecker keeps date-time / uri / etc. behaviour consistent with
|
|
||||||
# validate_deepseek_receipts.py. Even though the current claims schema
|
|
||||||
# does not declare any ``format`` keywords, threading the checker in
|
|
||||||
# avoids a footgun for the next contributor who adds one.
|
|
||||||
validator = Draft202012Validator(schema, format_checker=FormatChecker())
|
|
||||||
errors = sorted(validator.iter_errors(registry), key=lambda e: list(e.absolute_path))
|
|
||||||
if errors:
|
|
||||||
print(f"FAIL {args.path}")
|
|
||||||
for err in errors:
|
|
||||||
location = "/".join(str(p) for p in err.absolute_path) or "<root>"
|
|
||||||
print(f" - {location}: {err.message}")
|
|
||||||
return 1
|
|
||||||
|
|
||||||
failures: list[str] = []
|
|
||||||
seen_ids: dict[str, int] = {}
|
|
||||||
for index, entry in enumerate(registry.get("claims", [])):
|
|
||||||
cid = entry.get("id", f"<index {index}>")
|
|
||||||
if cid in seen_ids:
|
|
||||||
failures.append(
|
|
||||||
f"duplicate id '{cid}' (also defined at index {seen_ids[cid]})"
|
|
||||||
)
|
|
||||||
seen_ids[cid] = index
|
|
||||||
|
|
||||||
for key in ("lean",):
|
|
||||||
value = entry.get(key)
|
|
||||||
if not value:
|
|
||||||
continue
|
|
||||||
# Lean entries may be either a file path or a theorem symbol; only
|
|
||||||
# validate the file path form, which contains a `/` or ends in `.lean`.
|
|
||||||
if "/" in value or value.endswith(".lean"):
|
|
||||||
ok, msg = _check_path(value)
|
|
||||||
if not ok:
|
|
||||||
failures.append(f"claim '{cid}': {key}: {msg}")
|
|
||||||
|
|
||||||
for key in ("review_receipts", "sources"):
|
|
||||||
for value in entry.get(key, []) or []:
|
|
||||||
ok, msg = _check_path(value)
|
|
||||||
if not ok:
|
|
||||||
failures.append(f"claim '{cid}': {key}: {msg}")
|
|
||||||
|
|
||||||
if failures:
|
|
||||||
print(f"FAIL {args.path}")
|
|
||||||
for line in failures:
|
|
||||||
print(f" - {line}")
|
|
||||||
return 1
|
|
||||||
|
|
||||||
print(f"OK {args.path} ({len(registry.get('claims', []))} claim(s))")
|
|
||||||
return 0
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
raise SystemExit(main())
|
|
||||||
|
|
@ -1,136 +0,0 @@
|
||||||
#!/usr/bin/env python3
|
|
||||||
"""Validate DeepSeek review receipts against the repo JSON Schema.
|
|
||||||
|
|
||||||
Usage:
|
|
||||||
scripts/math-first/validate_deepseek_receipts.py [PATH ...]
|
|
||||||
|
|
||||||
When no PATH is provided, every tracked ``*.receipt.json`` under
|
|
||||||
``shared-data/artifacts/deepseek_review/`` is validated. Otherwise the named
|
|
||||||
paths are validated directly (files are checked as receipts; directories are
|
|
||||||
walked for ``*.receipt.json``).
|
|
||||||
|
|
||||||
Exit code:
|
|
||||||
0 every receipt validates against
|
|
||||||
``shared-data/schemas/deepseek-review-receipt.schema.json``.
|
|
||||||
1 one or more receipts failed schema validation.
|
|
||||||
2 the schema itself is malformed or ``jsonschema`` is missing.
|
|
||||||
|
|
||||||
This script is the single source of truth shared by the pre-commit hook in
|
|
||||||
``.pre-commit-config.yaml`` and the ``math-check`` GitHub Actions workflow in
|
|
||||||
``.github/workflows/math-check.yml``. See ``docs/math-first-tooling.md`` for
|
|
||||||
the math-first tooling contract.
|
|
||||||
"""
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import argparse
|
|
||||||
import json
|
|
||||||
import sys
|
|
||||||
from pathlib import Path
|
|
||||||
from typing import Iterable, Iterator
|
|
||||||
|
|
||||||
REPO_ROOT = Path(__file__).resolve().parents[2]
|
|
||||||
SCHEMA_PATH = REPO_ROOT / "shared-data" / "schemas" / "deepseek-review-receipt.schema.json"
|
|
||||||
DEFAULT_ROOT = REPO_ROOT / "shared-data" / "artifacts" / "deepseek_review"
|
|
||||||
RECEIPT_SUFFIX = ".receipt.json"
|
|
||||||
|
|
||||||
|
|
||||||
def _iter_receipts(paths: Iterable[Path]) -> Iterator[Path]:
|
|
||||||
for path in paths:
|
|
||||||
if path.is_dir():
|
|
||||||
yield from sorted(p for p in path.rglob(f"*{RECEIPT_SUFFIX}") if p.is_file())
|
|
||||||
elif path.is_file():
|
|
||||||
yield path
|
|
||||||
else:
|
|
||||||
print(f"warning: skipping missing path {path}", file=sys.stderr)
|
|
||||||
|
|
||||||
|
|
||||||
def _load_validator(schema_path: Path):
|
|
||||||
try:
|
|
||||||
from jsonschema import Draft202012Validator, FormatChecker
|
|
||||||
except ImportError as exc:
|
|
||||||
print(
|
|
||||||
"error: jsonschema>=4.18 is required (Draft 2020-12). "
|
|
||||||
"Install via `uv pip install jsonschema>=4.21 rfc3339-validator`.",
|
|
||||||
file=sys.stderr,
|
|
||||||
)
|
|
||||||
raise SystemExit(2) from exc
|
|
||||||
|
|
||||||
try:
|
|
||||||
schema = json.loads(schema_path.read_text(encoding="utf-8"))
|
|
||||||
except FileNotFoundError:
|
|
||||||
print(f"error: schema not found at {schema_path}", file=sys.stderr)
|
|
||||||
raise SystemExit(2)
|
|
||||||
except json.JSONDecodeError as exc:
|
|
||||||
print(f"error: schema {schema_path} is not valid JSON: {exc}", file=sys.stderr)
|
|
||||||
raise SystemExit(2)
|
|
||||||
|
|
||||||
try:
|
|
||||||
Draft202012Validator.check_schema(schema)
|
|
||||||
except Exception as exc: # noqa: BLE001 - surface schema errors verbatim
|
|
||||||
print(f"error: schema {schema_path} is invalid: {exc}", file=sys.stderr)
|
|
||||||
raise SystemExit(2)
|
|
||||||
|
|
||||||
return Draft202012Validator(schema, format_checker=FormatChecker())
|
|
||||||
|
|
||||||
|
|
||||||
def main(argv: list[str] | None = None) -> int:
|
|
||||||
parser = argparse.ArgumentParser(description=__doc__)
|
|
||||||
parser.add_argument(
|
|
||||||
"paths",
|
|
||||||
nargs="*",
|
|
||||||
type=Path,
|
|
||||||
help="Receipt files or directories. Defaults to the tracked review artifact root.",
|
|
||||||
)
|
|
||||||
parser.add_argument(
|
|
||||||
"--schema",
|
|
||||||
type=Path,
|
|
||||||
default=SCHEMA_PATH,
|
|
||||||
help=f"Path to the JSON Schema (default: {SCHEMA_PATH.relative_to(REPO_ROOT)}).",
|
|
||||||
)
|
|
||||||
args = parser.parse_args(argv)
|
|
||||||
|
|
||||||
validator = _load_validator(args.schema)
|
|
||||||
|
|
||||||
if args.paths:
|
|
||||||
candidates = list(_iter_receipts(args.paths))
|
|
||||||
elif DEFAULT_ROOT.exists():
|
|
||||||
candidates = list(_iter_receipts([DEFAULT_ROOT]))
|
|
||||||
else:
|
|
||||||
candidates = []
|
|
||||||
|
|
||||||
receipts = [p for p in candidates if p.name.endswith(RECEIPT_SUFFIX)]
|
|
||||||
skipped = [p for p in candidates if not p.name.endswith(RECEIPT_SUFFIX)]
|
|
||||||
for path in skipped:
|
|
||||||
print(f"skip: {path} (not a *{RECEIPT_SUFFIX} file)")
|
|
||||||
|
|
||||||
if not receipts:
|
|
||||||
print("no DeepSeek review receipts to validate")
|
|
||||||
return 0
|
|
||||||
|
|
||||||
failed = 0
|
|
||||||
for path in receipts:
|
|
||||||
try:
|
|
||||||
data = json.loads(path.read_text(encoding="utf-8"))
|
|
||||||
except json.JSONDecodeError as exc:
|
|
||||||
print(f"FAIL {path}: invalid JSON ({exc})")
|
|
||||||
failed += 1
|
|
||||||
continue
|
|
||||||
|
|
||||||
errors = sorted(validator.iter_errors(data), key=lambda e: list(e.absolute_path))
|
|
||||||
if errors:
|
|
||||||
print(f"FAIL {path}")
|
|
||||||
for err in errors:
|
|
||||||
location = "/".join(str(p) for p in err.absolute_path) or "<root>"
|
|
||||||
print(f" - {location}: {err.message}")
|
|
||||||
failed += 1
|
|
||||||
else:
|
|
||||||
print(f"OK {path}")
|
|
||||||
|
|
||||||
if failed:
|
|
||||||
print(f"\n{failed} receipt(s) failed validation", file=sys.stderr)
|
|
||||||
return 1
|
|
||||||
return 0
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
raise SystemExit(main())
|
|
||||||
|
|
@ -1,741 +0,0 @@
|
||||||
#!/usr/bin/env python3
|
|
||||||
"""
|
|
||||||
PIST Neuromorphic Orchestrator
|
|
||||||
==============================
|
|
||||||
The orchestrator is not a script. It is a topology.
|
|
||||||
|
|
||||||
Every action (prover call, build, module load, reboot) is a coordinate
|
|
||||||
transition on a manifold. The orchestrator learns which paths succeed,
|
|
||||||
strengthens them, and prunes dead branches.
|
|
||||||
|
|
||||||
Modes:
|
|
||||||
observe — passively watch system state, build DAG, no action
|
|
||||||
suggest — recommend next action based on learned topology
|
|
||||||
execute — perform action, observe result, update DAG
|
|
||||||
|
|
||||||
Architecture:
|
|
||||||
Neurons = task types (build, prove, load, reboot, benchmark)
|
|
||||||
Synapses = transitions between tasks (build→prove, prove→load)
|
|
||||||
Weights = success rate of each transition
|
|
||||||
Plasticity= LTP/LTD based on observed outcomes
|
|
||||||
|
|
||||||
The orchestrator feeds its own byte stream into pist_neuromorphic.ko
|
|
||||||
so the kernel observer learns the orchestration pattern as part of
|
|
||||||
the system topology.
|
|
||||||
"""
|
|
||||||
|
|
||||||
import sys
|
|
||||||
import os
|
|
||||||
import json
|
|
||||||
import time
|
|
||||||
import hashlib
|
|
||||||
import subprocess
|
|
||||||
import random
|
|
||||||
import glob
|
|
||||||
import shutil
|
|
||||||
from pathlib import Path
|
|
||||||
from dataclasses import dataclass, field, asdict
|
|
||||||
from typing import Optional, List, Dict, Callable
|
|
||||||
from datetime import datetime
|
|
||||||
|
|
||||||
# ──────────────────────────────────────────────────────────────────────────
|
|
||||||
# PIST Geometry (mirrors kernel module logic in userspace)
|
|
||||||
# ──────────────────────────────────────────────────────────────────────────
|
|
||||||
|
|
||||||
def pist_encode_u8(n: int) -> int:
|
|
||||||
"""n = k² + t, return packed 32-bit coordinate."""
|
|
||||||
k = int(n ** 0.5)
|
|
||||||
t = n - k * k
|
|
||||||
return (k << 16) | t
|
|
||||||
|
|
||||||
def pist_mirror(coord: int) -> int:
|
|
||||||
"""Involution: (k, t) → (k, 2k+1-t)"""
|
|
||||||
k = (coord >> 16) & 0xFFFF
|
|
||||||
t = coord & 0xFFFF
|
|
||||||
return (k << 16) | (2 * k + 1 - t)
|
|
||||||
|
|
||||||
def pist_mass(coord: int) -> int:
|
|
||||||
"""t * (2k + 1 - t)"""
|
|
||||||
k = (coord >> 16) & 0xFFFF
|
|
||||||
t = coord & 0xFFFF
|
|
||||||
return t * (2 * k + 1 - t)
|
|
||||||
|
|
||||||
def pist_tension(coord: int) -> float:
|
|
||||||
"""Normalized tension ∈ [0, 1)."""
|
|
||||||
k = (coord >> 16) & 0xFFFF
|
|
||||||
t = coord & 0xFFFF
|
|
||||||
denom = 2 * k + 1
|
|
||||||
return t / denom if denom > 0 else 0.0
|
|
||||||
|
|
||||||
# ──────────────────────────────────────────────────────────────────────────
|
|
||||||
# Neuromorphic DAG Node (mirrors kernel struct pist_dag_node)
|
|
||||||
# ──────────────────────────────────────────────────────────────────────────
|
|
||||||
|
|
||||||
@dataclass
|
|
||||||
class Synapse:
|
|
||||||
target: str # neuron ID
|
|
||||||
weight: float = 1.0 # Hebbian weight
|
|
||||||
success_count: int = 0
|
|
||||||
failure_count: int = 0
|
|
||||||
last_seen: float = field(default_factory=time.time)
|
|
||||||
|
|
||||||
@property
|
|
||||||
def success_rate(self) -> float:
|
|
||||||
total = self.success_count + self.failure_count
|
|
||||||
return self.success_count / total if total > 0 else 0.5
|
|
||||||
|
|
||||||
def potentiate(self, delta: float = 0.1):
|
|
||||||
"""LTP — strengthen synapse on success."""
|
|
||||||
self.weight = min(self.weight * (1 + delta), 100.0)
|
|
||||||
self.success_count += 1
|
|
||||||
self.last_seen = time.time()
|
|
||||||
|
|
||||||
def depress(self, delta: float = 0.2):
|
|
||||||
"""LTD — weaken synapse on failure."""
|
|
||||||
self.weight = max(self.weight * (1 - delta), 0.01)
|
|
||||||
self.failure_count += 1
|
|
||||||
self.last_seen = time.time()
|
|
||||||
|
|
||||||
@dataclass
|
|
||||||
class Neuron:
|
|
||||||
nid: str # neuron ID
|
|
||||||
task_type: str # build, prove, load, reboot, benchmark, etc.
|
|
||||||
coord: int = 0 # PIST coordinate of this neuron
|
|
||||||
activation: float = 0.0 # current activation level [0, 1]
|
|
||||||
visit_count: int = 0
|
|
||||||
total_mass: int = 0
|
|
||||||
synapses: List[Synapse] = field(default_factory=list)
|
|
||||||
|
|
||||||
def __post_init__(self):
|
|
||||||
if self.coord == 0:
|
|
||||||
# Derive coordinate from hash of neuron ID
|
|
||||||
h = hashlib.sha256(self.nid.encode()).digest()
|
|
||||||
self.coord = pist_encode_u8(h[0])
|
|
||||||
|
|
||||||
def activate(self, stimulus: float = 1.0):
|
|
||||||
"""Fire neuron — increase activation, update mass."""
|
|
||||||
self.activation = min(self.activation + stimulus, 1.0)
|
|
||||||
self.visit_count += 1
|
|
||||||
self.total_mass += pist_mass(self.coord)
|
|
||||||
|
|
||||||
def decay(self, rate: float = 0.01):
|
|
||||||
"""Exponential decay of activation."""
|
|
||||||
self.activation *= (1 - rate)
|
|
||||||
|
|
||||||
def get_synapse(self, target: str) -> Synapse:
|
|
||||||
"""Find or create synapse to target."""
|
|
||||||
for s in self.synapses:
|
|
||||||
if s.target == target:
|
|
||||||
return s
|
|
||||||
s = Synapse(target=target)
|
|
||||||
self.synapses.append(s)
|
|
||||||
return s
|
|
||||||
|
|
||||||
def choose_next(self, temperature: float = 1.0) -> Optional[str]:
|
|
||||||
"""Softmax selection of next neuron based on synapse weights."""
|
|
||||||
if not self.synapses:
|
|
||||||
return None
|
|
||||||
weights = [s.weight * s.success_rate for s in self.synapses]
|
|
||||||
# Boltzmann distribution
|
|
||||||
exp_w = [w ** (1 / temperature) for w in weights]
|
|
||||||
total = sum(exp_w)
|
|
||||||
probs = [e / total for e in exp_w]
|
|
||||||
# Roulette wheel selection
|
|
||||||
r = random.random()
|
|
||||||
cumsum = 0.0
|
|
||||||
for syn, p in zip(self.synapses, probs):
|
|
||||||
cumsum += p
|
|
||||||
if r <= cumsum:
|
|
||||||
return syn.target
|
|
||||||
return self.synapses[-1].target
|
|
||||||
|
|
||||||
# ──────────────────────────────────────────────────────────────────────────
|
|
||||||
# Neuromorphic Orchestrator State
|
|
||||||
# ──────────────────────────────────────────────────────────────────────────
|
|
||||||
|
|
||||||
class NeuromorphicOrchestrator:
|
|
||||||
def __init__(self, state_file: Optional[str] = None):
|
|
||||||
self.neurons: Dict[str, Neuron] = {}
|
|
||||||
self.current_neuron: Optional[str] = None
|
|
||||||
self.execution_log: List[dict] = []
|
|
||||||
self.dag_generation: int = 0
|
|
||||||
self.state_file = state_file or self._default_state_path()
|
|
||||||
self.mode: str = "observe" # observe | suggest | execute
|
|
||||||
self._init_default_neurons()
|
|
||||||
self._load_state()
|
|
||||||
|
|
||||||
def _default_state_path(self) -> str:
|
|
||||||
base = Path.home() / "CascadeProjects" / "Research-Stack"
|
|
||||||
return str(base / ".windsurf" / "telemetry" / "orchestrator_state.json")
|
|
||||||
|
|
||||||
def _init_default_neurons(self):
|
|
||||||
"""Bootstrap the default task topology."""
|
|
||||||
tasks = [
|
|
||||||
"idle", "diagnose", "build", "prove", "load_module",
|
|
||||||
"reboot", "benchmark", "export_dag", "collect_data",
|
|
||||||
"fix_toolchain", "ghost_ingested", "compress_baseline"
|
|
||||||
]
|
|
||||||
for t in tasks:
|
|
||||||
self._get_or_create(t)
|
|
||||||
|
|
||||||
# Default topology (bootstrap connections)
|
|
||||||
self._connect("idle", "diagnose", 1.0)
|
|
||||||
self._connect("diagnose", "build", 0.8)
|
|
||||||
self._connect("diagnose", "fix_toolchain", 0.6)
|
|
||||||
self._connect("build", "prove", 0.7)
|
|
||||||
self._connect("build", "benchmark", 0.3)
|
|
||||||
self._connect("prove", "load_module", 0.4)
|
|
||||||
self._connect("prove", "export_dag", 0.2)
|
|
||||||
self._connect("fix_toolchain", "build", 0.9)
|
|
||||||
self._connect("load_module", "collect_data", 0.8)
|
|
||||||
self._connect("collect_data", "compress_baseline", 0.5)
|
|
||||||
self._connect("benchmark", "export_dag", 0.6)
|
|
||||||
self._connect("build", "ghost_ingested", 0.2)
|
|
||||||
self._connect("reboot", "diagnose", 0.9)
|
|
||||||
|
|
||||||
def _get_or_create(self, nid: str, task_type: Optional[str] = None) -> Neuron:
|
|
||||||
if nid not in self.neurons:
|
|
||||||
self.neurons[nid] = Neuron(nid=nid, task_type=task_type or nid)
|
|
||||||
return self.neurons[nid]
|
|
||||||
|
|
||||||
def _connect(self, src: str, dst: str, initial_weight: float = 1.0):
|
|
||||||
n = self._get_or_create(src)
|
|
||||||
syn = n.get_synapse(dst)
|
|
||||||
syn.weight = initial_weight
|
|
||||||
|
|
||||||
def _load_state(self):
|
|
||||||
if not os.path.exists(self.state_file):
|
|
||||||
return
|
|
||||||
try:
|
|
||||||
with open(self.state_file) as f:
|
|
||||||
data = json.load(f)
|
|
||||||
self.dag_generation = data.get("dag_generation", 0)
|
|
||||||
self.mode = data.get("mode", "observe")
|
|
||||||
for nid, ndata in data.get("neurons", {}).items():
|
|
||||||
n = self._get_or_create(nid, ndata.get("task_type", nid))
|
|
||||||
n.coord = ndata.get("coord", n.coord)
|
|
||||||
n.visit_count = ndata.get("visit_count", 0)
|
|
||||||
n.total_mass = ndata.get("total_mass", 0)
|
|
||||||
for sdata in ndata.get("synapses", []):
|
|
||||||
syn = n.get_synapse(sdata["target"])
|
|
||||||
syn.weight = sdata.get("weight", 1.0)
|
|
||||||
syn.success_count = sdata.get("success_count", 0)
|
|
||||||
syn.failure_count = sdata.get("failure_count", 0)
|
|
||||||
except Exception as e:
|
|
||||||
print(f"[orchestrator] state load warning: {e}", file=sys.stderr)
|
|
||||||
|
|
||||||
def save_state(self):
|
|
||||||
os.makedirs(os.path.dirname(self.state_file), exist_ok=True)
|
|
||||||
data = {
|
|
||||||
"dag_generation": self.dag_generation,
|
|
||||||
"mode": self.mode,
|
|
||||||
"timestamp": time.time(),
|
|
||||||
"neurons": {}
|
|
||||||
}
|
|
||||||
for nid, n in self.neurons.items():
|
|
||||||
data["neurons"][nid] = {
|
|
||||||
"task_type": n.task_type,
|
|
||||||
"coord": n.coord,
|
|
||||||
"visit_count": n.visit_count,
|
|
||||||
"total_mass": n.total_mass,
|
|
||||||
"synapses": [
|
|
||||||
{
|
|
||||||
"target": s.target,
|
|
||||||
"weight": s.weight,
|
|
||||||
"success_count": s.success_count,
|
|
||||||
"failure_count": s.failure_count,
|
|
||||||
"last_seen": s.last_seen
|
|
||||||
}
|
|
||||||
for s in n.synapses
|
|
||||||
]
|
|
||||||
}
|
|
||||||
with open(self.state_file, "w") as f:
|
|
||||||
json.dump(data, f, indent=2)
|
|
||||||
|
|
||||||
# ──────────────────────────────────────────────────────────────────────
|
|
||||||
# Observation & Learning
|
|
||||||
# ──────────────────────────────────────────────────────────────────────
|
|
||||||
|
|
||||||
def observe(self, from_task: str, to_task: str, outcome: bool,
|
|
||||||
metadata: Optional[dict] = None):
|
|
||||||
"""Record a transition and apply Hebbian plasticity."""
|
|
||||||
src = self._get_or_create(from_task)
|
|
||||||
dst = self._get_or_create(to_task)
|
|
||||||
syn = src.get_synapse(to_task)
|
|
||||||
|
|
||||||
if outcome:
|
|
||||||
syn.potentiate()
|
|
||||||
dst.activate(stimulus=0.5)
|
|
||||||
else:
|
|
||||||
syn.depress()
|
|
||||||
dst.activate(stimulus=-0.3)
|
|
||||||
|
|
||||||
self.execution_log.append({
|
|
||||||
"timestamp": time.time(),
|
|
||||||
"from": from_task,
|
|
||||||
"to": to_task,
|
|
||||||
"outcome": outcome,
|
|
||||||
"metadata": metadata or {}
|
|
||||||
})
|
|
||||||
|
|
||||||
self.dag_generation += 1
|
|
||||||
self._feed_to_kernel(from_task, to_task, outcome)
|
|
||||||
|
|
||||||
def _feed_to_kernel(self, from_task: str, to_task: str, outcome: bool):
|
|
||||||
"""Feed orchestrator transitions into pist_neuromorphic.ko."""
|
|
||||||
sample_path = Path("/sys/kernel/pist_neuromorphic/sample")
|
|
||||||
if not sample_path.exists():
|
|
||||||
return
|
|
||||||
try:
|
|
||||||
payload = f"{from_task}→{to_task}:{int(outcome)}\n".encode()
|
|
||||||
with open(sample_path, "wb") as f:
|
|
||||||
f.write(payload)
|
|
||||||
except PermissionError:
|
|
||||||
pass # Not running as root — expected
|
|
||||||
except Exception:
|
|
||||||
pass
|
|
||||||
|
|
||||||
# ──────────────────────────────────────────────────────────────────────
|
|
||||||
# Execution Primitives
|
|
||||||
# ──────────────────────────────────────────────────────────────────────
|
|
||||||
|
|
||||||
def run_build(self) -> bool:
|
|
||||||
"""Execute lake build, observe result."""
|
|
||||||
print("[orchestrator] → run_build")
|
|
||||||
start = time.time()
|
|
||||||
proc = subprocess.run(
|
|
||||||
["lake", "build"],
|
|
||||||
cwd=Path.home() / "CascadeProjects" / "Research-Stack" / "0-Core-Formalism" / "lean" / "Semantics",
|
|
||||||
capture_output=True, text=True
|
|
||||||
)
|
|
||||||
elapsed = time.time() - start
|
|
||||||
success = proc.returncode == 0
|
|
||||||
self.observe("build", "prove" if success else "fix_toolchain", success,
|
|
||||||
{"elapsed": elapsed, "stdout_lines": len(proc.stdout.splitlines())})
|
|
||||||
return success
|
|
||||||
|
|
||||||
def run_prover(self, target_file: str, model: str = "zeyu-zheng/BFS-Prover-V2-7B:q8_0") -> bool:
|
|
||||||
"""Run BFS-Prover on a target file."""
|
|
||||||
print(f"[orchestrator] → run_prover({target_file})")
|
|
||||||
bf4prover = Path.home() / "CascadeProjects" / "Research-Stack" / "scripts" / "bf4prover.py"
|
|
||||||
start = time.time()
|
|
||||||
proc = subprocess.run(
|
|
||||||
[sys.executable, str(bf4prover), "--file", target_file],
|
|
||||||
capture_output=True, text=True
|
|
||||||
)
|
|
||||||
elapsed = time.time() - start
|
|
||||||
success = proc.returncode == 0
|
|
||||||
self.observe("prove", "load_module" if success else "reboot", success,
|
|
||||||
{"elapsed": elapsed, "model": model})
|
|
||||||
return success
|
|
||||||
|
|
||||||
def load_kernel_module(self) -> bool:
|
|
||||||
"""Load pist_neuromorphic.ko."""
|
|
||||||
print("[orchestrator] → load_kernel_module")
|
|
||||||
kmod = Path.home() / "CascadeProjects" / "Research-Stack" / "6-Kernel-Shim" / "pist_neuromorphic.ko"
|
|
||||||
proc = subprocess.run(["sudo", "insmod", str(kmod)], capture_output=True, text=True)
|
|
||||||
success = proc.returncode == 0
|
|
||||||
self.observe("load_module", "collect_data" if success else "reboot", success,
|
|
||||||
{"stderr": proc.stderr[:200] if not success else ""})
|
|
||||||
return success
|
|
||||||
|
|
||||||
def diagnose(self) -> dict:
|
|
||||||
"""System diagnostic — returns observational data."""
|
|
||||||
print("[orchestrator] → diagnose")
|
|
||||||
result = {}
|
|
||||||
|
|
||||||
# Kernel version
|
|
||||||
try:
|
|
||||||
with open("/proc/version") as f:
|
|
||||||
result["kernel"] = f.read().strip()
|
|
||||||
except Exception:
|
|
||||||
result["kernel"] = "unknown"
|
|
||||||
|
|
||||||
# NVIDIA
|
|
||||||
try:
|
|
||||||
proc = subprocess.run(["nvidia-smi"], capture_output=True, text=True)
|
|
||||||
result["gpu"] = "available" if proc.returncode == 0 else proc.stderr[:200]
|
|
||||||
except FileNotFoundError:
|
|
||||||
result["gpu"] = "not_installed"
|
|
||||||
|
|
||||||
# Ollama
|
|
||||||
try:
|
|
||||||
proc = subprocess.run(["ollama", "ps"], capture_output=True, text=True)
|
|
||||||
result["ollama"] = proc.stdout.strip()
|
|
||||||
except FileNotFoundError:
|
|
||||||
result["ollama"] = "not_installed"
|
|
||||||
|
|
||||||
# Kernel module
|
|
||||||
result["neuromorphic_module"] = os.path.exists("/sys/kernel/pist_neuromorphic")
|
|
||||||
|
|
||||||
# Lean toolchain
|
|
||||||
try:
|
|
||||||
tc = Path.home() / "CascadeProjects" / "Research-Stack" / "0-Core-Formalism" / "lean" / "Semantics" / "lean-toolchain"
|
|
||||||
result["lean_toolchain"] = tc.read_text().strip()
|
|
||||||
except Exception:
|
|
||||||
result["lean_toolchain"] = "unknown"
|
|
||||||
|
|
||||||
self.observe("idle", "diagnose", True, result)
|
|
||||||
return result
|
|
||||||
|
|
||||||
def scan_lean_topology(self) -> dict:
|
|
||||||
"""Discover all .lean files, count sorry, create neurons per module."""
|
|
||||||
print("[orchestrator] → scan_lean_topology")
|
|
||||||
base = Path.home() / "CascadeProjects" / "Research-Stack"
|
|
||||||
findings = {"canonical": {}, "external": {}, "ingested": {}, "other": {}, "total_sorry": 0}
|
|
||||||
|
|
||||||
for path in base.rglob("*.lean"):
|
|
||||||
if ".lake" in str(path) or "build" in str(path) or "build-static" in str(path):
|
|
||||||
continue
|
|
||||||
|
|
||||||
rel = str(path.relative_to(base))
|
|
||||||
try:
|
|
||||||
text = path.read_text()
|
|
||||||
except Exception:
|
|
||||||
continue
|
|
||||||
|
|
||||||
sorry_count = text.count("\n sorry") + text.count("\n sorry")
|
|
||||||
if sorry_count == 0:
|
|
||||||
continue
|
|
||||||
|
|
||||||
findings["total_sorry"] += sorry_count
|
|
||||||
|
|
||||||
# Bucket classification
|
|
||||||
if rel.startswith("0-Core-Formalism/lean/Semantics/Semantics/"):
|
|
||||||
bucket = "canonical"
|
|
||||||
elif rel.startswith("0-Core-Formalism/lean/external/"):
|
|
||||||
bucket = "external"
|
|
||||||
elif "shared-data/data/ingested/" in rel:
|
|
||||||
bucket = "ingested"
|
|
||||||
elif "archive/" in rel:
|
|
||||||
continue # Skip archives
|
|
||||||
else:
|
|
||||||
bucket = "other"
|
|
||||||
|
|
||||||
findings[bucket][rel] = sorry_count
|
|
||||||
|
|
||||||
# Create neuron for high-sorry files
|
|
||||||
if sorry_count >= 2:
|
|
||||||
nid = f"file_{rel.replace('/', '_').replace('.', '_')}"
|
|
||||||
n = self._get_or_create(nid, task_type="prove_file")
|
|
||||||
n.coord = pist_encode_u8(min(sorry_count * 16, 255))
|
|
||||||
# Connect file neuron to prove and ghost actions
|
|
||||||
self._connect(nid, "prove", initial_weight=float(sorry_count))
|
|
||||||
self._connect(nid, "ghost_ingested" if bucket == "ingested" else "prove", initial_weight=1.0)
|
|
||||||
|
|
||||||
# Create summary neurons
|
|
||||||
for bucket, files in findings.items():
|
|
||||||
if bucket == "total_sorry":
|
|
||||||
continue
|
|
||||||
nid = f"summary_{bucket}"
|
|
||||||
n = self._get_or_create(nid, task_type="summary")
|
|
||||||
n.total_mass = sum(files.values())
|
|
||||||
self._connect(nid, "prove" if bucket == "canonical" else "ghost_ingested", initial_weight=float(n.total_mass))
|
|
||||||
|
|
||||||
self.observe("diagnose", "scan_lean_topology", True,
|
|
||||||
{"total_sorry": findings["total_sorry"],
|
|
||||||
"canonical_files": len(findings["canonical"]),
|
|
||||||
"ingested_files": len(findings["ingested"])})
|
|
||||||
return findings
|
|
||||||
|
|
||||||
def fix_toolchain(self) -> bool:
|
|
||||||
"""Revert lean-toolchain to v4.29.1 to restore mathlib cache."""
|
|
||||||
print("[orchestrator] → fix_toolchain")
|
|
||||||
base = Path.home() / "CascadeProjects" / "Research-Stack" / "0-Core-Formalism" / "lean" / "Semantics"
|
|
||||||
tc_file = base / "lean-toolchain"
|
|
||||||
lake_file = base / "lakefile.toml"
|
|
||||||
success = False
|
|
||||||
|
|
||||||
try:
|
|
||||||
# Revert toolchain
|
|
||||||
tc_file.write_text("leanprover/lean4:v4.29.1\n")
|
|
||||||
# Revert mathlib rev in lakefile.toml
|
|
||||||
text = lake_file.read_text()
|
|
||||||
text = text.replace("v4.30.0-rc2", "v4.29.1")
|
|
||||||
lake_file.write_text(text)
|
|
||||||
# Clean and update
|
|
||||||
subprocess.run(["lake", "clean"], cwd=base, capture_output=True)
|
|
||||||
proc = subprocess.run(["lake", "update"], cwd=base, capture_output=True, text=True)
|
|
||||||
success = proc.returncode == 0
|
|
||||||
except Exception as e:
|
|
||||||
print(f"[orchestrator] fix_toolchain error: {e}", file=sys.stderr)
|
|
||||||
|
|
||||||
self.observe("fix_toolchain", "build" if success else "diagnose", success,
|
|
||||||
{"toolchain": "v4.29.1"})
|
|
||||||
return success
|
|
||||||
|
|
||||||
def ghost_ingested(self) -> bool:
|
|
||||||
"""Rename ingested .lean files with .GHOST suffix."""
|
|
||||||
print("[orchestrator] → ghost_ingested")
|
|
||||||
base = Path.home() / "CascadeProjects" / "Research-Stack" / "shared-data" / "data" / "ingested"
|
|
||||||
ghosted = 0
|
|
||||||
try:
|
|
||||||
for path in base.rglob("*.lean"):
|
|
||||||
if not path.name.endswith(".GHOST"):
|
|
||||||
ghost_path = path.with_suffix(path.suffix + ".GHOST")
|
|
||||||
shutil.move(str(path), str(ghost_path))
|
|
||||||
ghosted += 1
|
|
||||||
except Exception as e:
|
|
||||||
print(f"[orchestrator] ghost_ingested error: {e}", file=sys.stderr)
|
|
||||||
|
|
||||||
success = ghosted > 0
|
|
||||||
self.observe("ghost_ingested", "build", success, {"ghosted_count": ghosted})
|
|
||||||
return success
|
|
||||||
|
|
||||||
def run_benchmark(self) -> bool:
|
|
||||||
"""Run PIST compression benchmark on Canterbury Corpus."""
|
|
||||||
print("[orchestrator] → run_benchmark")
|
|
||||||
base = Path.home() / "CascadeProjects" / "Research-Stack"
|
|
||||||
bench_dir = base / "shared-data" / "data" / "groundtruth" / "compression-baselines"
|
|
||||||
pist_script = base / "Desktop" / "pist_biological_polymorphic_shifter_v3_complete.py"
|
|
||||||
success = False
|
|
||||||
results = {}
|
|
||||||
|
|
||||||
try:
|
|
||||||
for fpath in bench_dir.iterdir():
|
|
||||||
if not fpath.is_file():
|
|
||||||
continue
|
|
||||||
data = fpath.read_bytes()
|
|
||||||
orig_size = len(data)
|
|
||||||
# Simple coordinate encoding as baseline
|
|
||||||
coords = [pist_encode_u8(b) for b in data[:4096]] # Sample first 4KB
|
|
||||||
coord_bytes = len(coords) * 4
|
|
||||||
ratio = orig_size / coord_bytes if coord_bytes > 0 else 0
|
|
||||||
results[fpath.name] = {"original": orig_size, "coord_4k": coord_bytes, "ratio": ratio}
|
|
||||||
success = True
|
|
||||||
except Exception as e:
|
|
||||||
print(f"[orchestrator] benchmark error: {e}", file=sys.stderr)
|
|
||||||
|
|
||||||
self.observe("compress_baseline", "export_dag", success, {"files_tested": len(results)})
|
|
||||||
return success
|
|
||||||
|
|
||||||
# ──────────────────────────────────────────────────────────────────────
|
|
||||||
# Topological Navigation
|
|
||||||
# ──────────────────────────────────────────────────────────────────────
|
|
||||||
|
|
||||||
def step(self, temperature: float = 1.0) -> Optional[str]:
|
|
||||||
"""Take one step on the manifold. Returns next task or None."""
|
|
||||||
if self.current_neuron is None:
|
|
||||||
self.current_neuron = "idle"
|
|
||||||
|
|
||||||
n = self.neurons.get(self.current_neuron)
|
|
||||||
if not n:
|
|
||||||
return None
|
|
||||||
|
|
||||||
n.activate()
|
|
||||||
next_id = n.choose_next(temperature=temperature)
|
|
||||||
if next_id:
|
|
||||||
print(f"[orchestrator] {self.current_neuron} → {next_id} "
|
|
||||||
f"(tension={pist_tension(n.coord):.3f}, mass={n.total_mass})")
|
|
||||||
self.current_neuron = next_id
|
|
||||||
return next_id
|
|
||||||
|
|
||||||
def walk(self, max_steps: int = 10, temperature: float = 1.0) -> List[str]:
|
|
||||||
"""Walk the manifold, executing tasks in execute mode."""
|
|
||||||
path = []
|
|
||||||
for _ in range(max_steps):
|
|
||||||
task = self.step(temperature=temperature)
|
|
||||||
if task is None:
|
|
||||||
break
|
|
||||||
path.append(task)
|
|
||||||
|
|
||||||
if self.mode == "execute":
|
|
||||||
self._execute_task(task)
|
|
||||||
elif self.mode == "suggest":
|
|
||||||
print(f"[orchestrator] SUGGEST: {task}")
|
|
||||||
|
|
||||||
self.save_state()
|
|
||||||
return path
|
|
||||||
|
|
||||||
def _execute_task(self, task: str):
|
|
||||||
"""Dispatch task to execution primitive."""
|
|
||||||
handlers = {
|
|
||||||
"diagnose": self.diagnose,
|
|
||||||
"build": self.run_build,
|
|
||||||
"prove": self.run_prover,
|
|
||||||
"load_module": self.load_kernel_module,
|
|
||||||
"fix_toolchain": self.fix_toolchain,
|
|
||||||
"ghost_ingested": self.ghost_ingested,
|
|
||||||
"compress_baseline": self.run_benchmark,
|
|
||||||
"scan": self.scan_lean_topology,
|
|
||||||
}
|
|
||||||
handler = handlers.get(task)
|
|
||||||
if handler:
|
|
||||||
try:
|
|
||||||
result = handler()
|
|
||||||
# Auto-observe success if handler returns truthy
|
|
||||||
if result:
|
|
||||||
n = self.neurons.get(task)
|
|
||||||
if n:
|
|
||||||
for syn in n.synapses:
|
|
||||||
syn.potentiate()
|
|
||||||
except Exception as e:
|
|
||||||
print(f"[orchestrator] task {task} failed: {e}", file=sys.stderr)
|
|
||||||
n = self.neurons.get(task)
|
|
||||||
if n:
|
|
||||||
for syn in n.synapses:
|
|
||||||
syn.depress()
|
|
||||||
elif task.startswith("file_"):
|
|
||||||
# File-specific neuron — extract original path
|
|
||||||
print(f"[orchestrator] target file neuron: {task}")
|
|
||||||
|
|
||||||
def continuous_loop(self, interval: float = 60.0, temperature: float = 1.0):
|
|
||||||
"""Run forever, adapting temperature based on crisis level."""
|
|
||||||
print(f"[orchestrator] ENTERING CONTINUOUS LOOP (interval={interval}s)")
|
|
||||||
consecutive_failures = 0
|
|
||||||
while True:
|
|
||||||
try:
|
|
||||||
# Crisis detection: too many failures → force exploration
|
|
||||||
if consecutive_failures >= 3:
|
|
||||||
temperature = min(temperature * 1.5, 5.0)
|
|
||||||
print(f"[orchestrator] CRISIS MODE: temperature bumped to {temperature}")
|
|
||||||
|
|
||||||
task = self.step(temperature=temperature)
|
|
||||||
if task is None:
|
|
||||||
time.sleep(interval)
|
|
||||||
continue
|
|
||||||
|
|
||||||
if self.mode == "execute":
|
|
||||||
result = self._execute_task_and_report(task)
|
|
||||||
if result:
|
|
||||||
consecutive_failures = max(0, consecutive_failures - 1)
|
|
||||||
temperature = max(temperature * 0.9, 0.5)
|
|
||||||
else:
|
|
||||||
consecutive_failures += 1
|
|
||||||
elif self.mode == "suggest":
|
|
||||||
print(f"[orchestrator] SUGGEST: {task}")
|
|
||||||
|
|
||||||
self.save_state()
|
|
||||||
time.sleep(interval)
|
|
||||||
except KeyboardInterrupt:
|
|
||||||
print("[orchestrator] Interrupted by user")
|
|
||||||
self.save_state()
|
|
||||||
break
|
|
||||||
|
|
||||||
def _execute_task_and_report(self, task: str) -> bool:
|
|
||||||
"""Execute and auto-observe with proper transition tracking."""
|
|
||||||
prev = self.current_neuron
|
|
||||||
handlers = {
|
|
||||||
"diagnose": self.diagnose,
|
|
||||||
"build": self.run_build,
|
|
||||||
"prove": lambda: self.run_prover("Semantics/FixedPoint.lean"),
|
|
||||||
"load_module": self.load_kernel_module,
|
|
||||||
"fix_toolchain": self.fix_toolchain,
|
|
||||||
"ghost_ingested": self.ghost_ingested,
|
|
||||||
"compress_baseline": self.run_benchmark,
|
|
||||||
"scan": self.scan_lean_topology,
|
|
||||||
}
|
|
||||||
handler = handlers.get(task)
|
|
||||||
if not handler:
|
|
||||||
return False
|
|
||||||
try:
|
|
||||||
result = handler()
|
|
||||||
success = bool(result) if result is not None else True
|
|
||||||
except Exception as e:
|
|
||||||
print(f"[orchestrator] execution failed: {e}", file=sys.stderr)
|
|
||||||
success = False
|
|
||||||
|
|
||||||
if prev and task:
|
|
||||||
self.observe(prev, task, success, {"auto": True})
|
|
||||||
return success
|
|
||||||
|
|
||||||
# ──────────────────────────────────────────────────────────────────────
|
|
||||||
# DAG Export
|
|
||||||
# ──────────────────────────────────────────────────────────────────────
|
|
||||||
|
|
||||||
def export_dag(self, path: Optional[str] = None) -> str:
|
|
||||||
"""Export current DAG as text."""
|
|
||||||
lines = [
|
|
||||||
f"# PIST Neuromorphic Orchestrator DAG",
|
|
||||||
f"# generation={self.dag_generation} mode={self.mode}",
|
|
||||||
f"# timestamp={datetime.now().isoformat()}",
|
|
||||||
"# neuron_id task_type coord visit_count mass"
|
|
||||||
]
|
|
||||||
for nid, n in sorted(self.neurons.items(), key=lambda x: -x[1].visit_count):
|
|
||||||
lines.append(
|
|
||||||
f"{nid} {n.task_type} 0x{n.coord:08x} {n.visit_count} {n.total_mass}"
|
|
||||||
)
|
|
||||||
for s in sorted(n.synapses, key=lambda x: -x.weight)[:5]:
|
|
||||||
lines.append(f" → {s.target} w={s.weight:.2f} sr={s.success_rate:.2f}")
|
|
||||||
|
|
||||||
text = "\n".join(lines) + "\n"
|
|
||||||
if path:
|
|
||||||
with open(path, "w") as f:
|
|
||||||
f.write(text)
|
|
||||||
return text
|
|
||||||
|
|
||||||
|
|
||||||
# ──────────────────────────────────────────────────────────────────────────
|
|
||||||
# CLI
|
|
||||||
# ──────────────────────────────────────────────────────────────────────────
|
|
||||||
|
|
||||||
def main():
|
|
||||||
import argparse
|
|
||||||
parser = argparse.ArgumentParser(description="PIST Neuromorphic Orchestrator")
|
|
||||||
parser.add_argument("--mode", choices=["observe", "suggest", "execute"],
|
|
||||||
default="observe", help="Orchestrator mode")
|
|
||||||
parser.add_argument("--steps", type=int, default=5, help="Max manifold walk steps")
|
|
||||||
parser.add_argument("--temperature", type=float, default=1.0,
|
|
||||||
help="Exploration temperature (higher = more random)")
|
|
||||||
parser.add_argument("--export", type=str, help="Export DAG to file")
|
|
||||||
parser.add_argument("--feed-kernel", action="store_true",
|
|
||||||
help="Feed orchestrator state into pist_neuromorphic.ko")
|
|
||||||
parser.add_argument("--diagnose", action="store_true",
|
|
||||||
help="Run system diagnostic and exit")
|
|
||||||
parser.add_argument("--scan", action="store_true",
|
|
||||||
help="Scan Lean topology and exit")
|
|
||||||
parser.add_argument("--loop", action="store_true",
|
|
||||||
help="Run continuous adaptive loop")
|
|
||||||
parser.add_argument("--interval", type=float, default=60.0,
|
|
||||||
help="Loop interval in seconds (default: 60)")
|
|
||||||
args = parser.parse_args()
|
|
||||||
|
|
||||||
orch = NeuromorphicOrchestrator()
|
|
||||||
orch.mode = args.mode
|
|
||||||
|
|
||||||
if args.diagnose:
|
|
||||||
result = orch.diagnose()
|
|
||||||
print(json.dumps(result, indent=2))
|
|
||||||
orch.save_state()
|
|
||||||
return
|
|
||||||
|
|
||||||
if args.scan:
|
|
||||||
findings = orch.scan_lean_topology()
|
|
||||||
print(json.dumps(findings, indent=2))
|
|
||||||
orch.save_state()
|
|
||||||
return
|
|
||||||
|
|
||||||
if args.export:
|
|
||||||
orch.export_dag(args.export)
|
|
||||||
print(f"[orchestrator] DAG exported to {args.export}")
|
|
||||||
return
|
|
||||||
|
|
||||||
print(f"[orchestrator] mode={args.mode} steps={args.steps} temp={args.temperature}")
|
|
||||||
print(f"[orchestrator] dag_generation={orch.dag_generation}")
|
|
||||||
print(f"[orchestrator] neurons={len(orch.neurons)}")
|
|
||||||
|
|
||||||
if args.loop:
|
|
||||||
orch.continuous_loop(interval=args.interval, temperature=args.temperature)
|
|
||||||
return
|
|
||||||
|
|
||||||
path = orch.walk(max_steps=args.steps, temperature=args.temperature)
|
|
||||||
print(f"[orchestrator] path={' → '.join(path)}")
|
|
||||||
orch.save_state()
|
|
||||||
|
|
||||||
# Feed to kernel if requested
|
|
||||||
if args.feed_kernel:
|
|
||||||
state_text = json.dumps({
|
|
||||||
"dag_generation": orch.dag_generation,
|
|
||||||
"path": path,
|
|
||||||
"neuron_count": len(orch.neurons)
|
|
||||||
})
|
|
||||||
try:
|
|
||||||
with open("/sys/kernel/pist_neuromorphic/sample", "wb") as f:
|
|
||||||
f.write(state_text.encode())
|
|
||||||
print("[orchestrator] state fed to kernel module")
|
|
||||||
except Exception as e:
|
|
||||||
print(f"[orchestrator] kernel feed failed: {e}")
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
main()
|
|
||||||
|
|
@ -1,200 +0,0 @@
|
||||||
#!/usr/bin/env python3
|
|
||||||
"""
|
|
||||||
populate-open-webui-knowledge-expanded.py
|
|
||||||
|
|
||||||
Comprehensive knowledge population + Cascade persona setup for Open WebUI.
|
|
||||||
|
|
||||||
Usage:
|
|
||||||
1. Go to http://127.0.0.1:3000 and create your admin account
|
|
||||||
2. Get API key: Settings → Account → API Key
|
|
||||||
3. Run: python3 scripts/populate-open-webui-knowledge-expanded.py <API_KEY>
|
|
||||||
|
|
||||||
This creates 12 knowledge collections covering the entire project.
|
|
||||||
"""
|
|
||||||
|
|
||||||
import sys
|
|
||||||
import os
|
|
||||||
import requests
|
|
||||||
|
|
||||||
BASE_URL = "http://127.0.0.1:3000"
|
|
||||||
HEADERS = {"Content-Type": "application/json"}
|
|
||||||
REPO_ROOT = "/home/allaun/CascadeProjects/Research-Stack"
|
|
||||||
|
|
||||||
|
|
||||||
def set_api_key(key):
|
|
||||||
HEADERS["Authorization"] = f"Bearer {key}"
|
|
||||||
|
|
||||||
|
|
||||||
def create_knowledge(name, description):
|
|
||||||
url = f"{BASE_URL}/api/v1/knowledge/"
|
|
||||||
payload = {"name": name, "description": description}
|
|
||||||
resp = requests.post(url, headers=HEADERS, json=payload)
|
|
||||||
resp.raise_for_status()
|
|
||||||
data = resp.json()
|
|
||||||
return data.get("id") or data.get("data", {}).get("id")
|
|
||||||
|
|
||||||
|
|
||||||
def upload_file(filepath):
|
|
||||||
url = f"{BASE_URL}/api/v1/files/"
|
|
||||||
filename = os.path.basename(filepath)
|
|
||||||
mime = "text/markdown" if filepath.endswith(".md") else "text/plain"
|
|
||||||
with open(filepath, "rb") as f:
|
|
||||||
files = {"file": (filename, f, mime)}
|
|
||||||
resp = requests.post(url, headers={"Authorization": HEADERS["Authorization"]}, files=files)
|
|
||||||
resp.raise_for_status()
|
|
||||||
data = resp.json()
|
|
||||||
return data.get("id") or data.get("data", {}).get("id")
|
|
||||||
|
|
||||||
|
|
||||||
def add_file_to_knowledge(knowledge_id, file_id):
|
|
||||||
url = f"{BASE_URL}/api/v1/knowledge/{knowledge_id}/files/"
|
|
||||||
payload = {"file_id": file_id}
|
|
||||||
resp = requests.post(url, headers=HEADERS, json=payload)
|
|
||||||
resp.raise_for_status()
|
|
||||||
|
|
||||||
|
|
||||||
def process_collection(name, description, file_paths):
|
|
||||||
print(f"\n Creating: {name}")
|
|
||||||
kid = create_knowledge(name, description)
|
|
||||||
print(f" -> ID: {kid}")
|
|
||||||
|
|
||||||
for fp in file_paths:
|
|
||||||
abs_fp = os.path.join(REPO_ROOT, fp)
|
|
||||||
if not os.path.isfile(abs_fp):
|
|
||||||
print(f" [SKIP] Not found: {fp}")
|
|
||||||
continue
|
|
||||||
print(f" Uploading: {os.path.basename(fp)}")
|
|
||||||
try:
|
|
||||||
fid = upload_file(abs_fp)
|
|
||||||
add_file_to_knowledge(kid, fid)
|
|
||||||
except Exception as e:
|
|
||||||
print(f" [ERROR] {e}")
|
|
||||||
print(f" Done: {name}")
|
|
||||||
|
|
||||||
|
|
||||||
def collect_files(pattern, max_files=50):
|
|
||||||
import glob
|
|
||||||
files = glob.glob(os.path.join(REPO_ROOT, pattern), recursive=True)
|
|
||||||
files = [os.path.relpath(f, REPO_ROOT) for f in files if os.path.isfile(f)]
|
|
||||||
return sorted(files)[:max_files]
|
|
||||||
|
|
||||||
|
|
||||||
def main():
|
|
||||||
if len(sys.argv) < 2:
|
|
||||||
print(__doc__)
|
|
||||||
print(f"\nUsage: python3 {sys.argv[0]} <API_KEY>")
|
|
||||||
sys.exit(1)
|
|
||||||
|
|
||||||
api_key = sys.argv[1]
|
|
||||||
set_api_key(api_key)
|
|
||||||
|
|
||||||
try:
|
|
||||||
r = requests.get(f"{BASE_URL}/api/v1/users/", headers=HEADERS, timeout=5)
|
|
||||||
r.raise_for_status()
|
|
||||||
print(f"Connected to Open WebUI at {BASE_URL}")
|
|
||||||
except Exception as e:
|
|
||||||
print(f"ERROR: Cannot connect: {e}")
|
|
||||||
sys.exit(1)
|
|
||||||
|
|
||||||
os.chdir(REPO_ROOT)
|
|
||||||
|
|
||||||
# 1. Core
|
|
||||||
process_collection(
|
|
||||||
"Research Stack Core",
|
|
||||||
"Central project documents.",
|
|
||||||
["README.md", "PROJECT_MAP.md", "CONCEPTS.md", "TODO_MAP.md"],
|
|
||||||
)
|
|
||||||
|
|
||||||
# 2. GCCL
|
|
||||||
process_collection(
|
|
||||||
"GCCL Theory",
|
|
||||||
"Genetic-Code Compression Language.",
|
|
||||||
collect_files("docs/research/GCCL_*.md"),
|
|
||||||
)
|
|
||||||
|
|
||||||
# 3. KOTC
|
|
||||||
process_collection(
|
|
||||||
"KOTC & Daemon Systems",
|
|
||||||
"Knowledge-Of-Task-Completion architecture.",
|
|
||||||
collect_files("docs/research/KOTC_*.md"),
|
|
||||||
)
|
|
||||||
|
|
||||||
# 4. VLB
|
|
||||||
process_collection(
|
|
||||||
"VLB & Witness Substrate",
|
|
||||||
"Very-Large-Block witness and substrate.",
|
|
||||||
collect_files("docs/research/VLB_*.md"),
|
|
||||||
)
|
|
||||||
|
|
||||||
# 5. FAMM
|
|
||||||
process_collection(
|
|
||||||
"FAMM & Route Memory",
|
|
||||||
"Fluid-Automata Memory Model.",
|
|
||||||
collect_files("docs/famm/*.md"),
|
|
||||||
)
|
|
||||||
|
|
||||||
# 6. Roadmaps
|
|
||||||
process_collection(
|
|
||||||
"Roadmaps & Strategy",
|
|
||||||
"Project roadmaps and planning.",
|
|
||||||
collect_files("docs/roadmaps/*.md"),
|
|
||||||
)
|
|
||||||
|
|
||||||
# 7. Speculative
|
|
||||||
process_collection(
|
|
||||||
"Speculative Materials",
|
|
||||||
"Exploratory research notes.",
|
|
||||||
collect_files("docs/speculative-materials/*.md"),
|
|
||||||
)
|
|
||||||
|
|
||||||
# 8. Lean READMEs
|
|
||||||
process_collection(
|
|
||||||
"Lean Formalism READMEs",
|
|
||||||
"Per-domain READMEs.",
|
|
||||||
collect_files("*/README.md", max_files=20),
|
|
||||||
)
|
|
||||||
|
|
||||||
# 9. Documentation
|
|
||||||
process_collection(
|
|
||||||
"Documentation Guides",
|
|
||||||
"Human-readable explanations.",
|
|
||||||
collect_files("6-Documentation/*.md", max_files=20),
|
|
||||||
)
|
|
||||||
|
|
||||||
# 10. Workflows
|
|
||||||
process_collection(
|
|
||||||
"Windsurf Workflows",
|
|
||||||
"Agent workflow definitions.",
|
|
||||||
collect_files(".windsurf/workflows/*.md"),
|
|
||||||
)
|
|
||||||
|
|
||||||
# 11. Assignments & Audit
|
|
||||||
process_collection(
|
|
||||||
"Agent Assignments & Audit",
|
|
||||||
"Task assignments and sorry audit.",
|
|
||||||
[".windsurf/ASSIGNMENTS.md", ".windsurf/SORRY_AUDIT.md"],
|
|
||||||
)
|
|
||||||
|
|
||||||
# 12. Lean Core Files
|
|
||||||
lean_core = collect_files("0-Core-Formalism/lean/Semantics/Semantics/*.lean", max_files=30)
|
|
||||||
process_collection(
|
|
||||||
"Lean Core Source",
|
|
||||||
"Key Lean formalism source files.",
|
|
||||||
lean_core,
|
|
||||||
)
|
|
||||||
|
|
||||||
# 13. Data
|
|
||||||
process_collection(
|
|
||||||
"Project Data Files",
|
|
||||||
"Data tables and indices.",
|
|
||||||
collect_files("data/*.tsv", max_files=10) + collect_files("data/*.json", max_files=10),
|
|
||||||
)
|
|
||||||
|
|
||||||
print("\n========================================")
|
|
||||||
print("All collections populated.")
|
|
||||||
print("Next: Create a custom model with the Cascade prompt.")
|
|
||||||
print("========================================\n")
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
main()
|
|
||||||
|
|
@ -1,228 +0,0 @@
|
||||||
#!/usr/bin/env python3
|
|
||||||
"""
|
|
||||||
populate-open-webui-knowledge.py
|
|
||||||
|
|
||||||
Prefill Open WebUI knowledge collections with your Research Stack documents.
|
|
||||||
|
|
||||||
Usage:
|
|
||||||
1. Go to http://127.0.0.1:3000 and create your admin account
|
|
||||||
2. Get API key: Settings → Account → API Key
|
|
||||||
3. Run: python3 scripts/populate-open-webui-knowledge.py <API_KEY>
|
|
||||||
|
|
||||||
Collections created:
|
|
||||||
- Research Stack Core (README, PROJECT_MAP, CONCEPTS, TODO_MAP)
|
|
||||||
- GCCL Theory (docs/research/GCCL_*.md)
|
|
||||||
- KOTC & Daemon Systems (docs/research/KOTC_*.md)
|
|
||||||
- VLB & Witness Substrate (docs/research/VLB_*.md)
|
|
||||||
- FAMM & Route Memory (docs/famm/*.md)
|
|
||||||
- Roadmaps & Strategy (docs/roadmaps/*.md)
|
|
||||||
- Speculative Materials (docs/speculative-materials/*.md)
|
|
||||||
- Lean Formalism READMEs (*/README.md)
|
|
||||||
- Documentation Guides (6-Documentation/*.md)
|
|
||||||
"""
|
|
||||||
|
|
||||||
import sys
|
|
||||||
import json
|
|
||||||
import os
|
|
||||||
import requests
|
|
||||||
|
|
||||||
BASE_URL = "http://127.0.0.1:3000"
|
|
||||||
HEADERS = {"Content-Type": "application/json"}
|
|
||||||
|
|
||||||
REPO_ROOT = "/home/allaun/CascadeProjects/Research-Stack"
|
|
||||||
|
|
||||||
|
|
||||||
def set_api_key(key):
|
|
||||||
HEADERS["Authorization"] = f"Bearer {key}"
|
|
||||||
|
|
||||||
|
|
||||||
def create_knowledge(name, description):
|
|
||||||
"""Create a knowledge collection."""
|
|
||||||
url = f"{BASE_URL}/api/v1/knowledge/"
|
|
||||||
payload = {"name": name, "description": description}
|
|
||||||
resp = requests.post(url, headers=HEADERS, json=payload)
|
|
||||||
resp.raise_for_status()
|
|
||||||
data = resp.json()
|
|
||||||
# Open WebUI returns {id, ...}
|
|
||||||
return data.get("id") or data.get("data", {}).get("id")
|
|
||||||
|
|
||||||
|
|
||||||
def upload_file(filepath):
|
|
||||||
"""Upload a single file, return file_id."""
|
|
||||||
url = f"{BASE_URL}/api/v1/files/"
|
|
||||||
filename = os.path.basename(filepath)
|
|
||||||
with open(filepath, "rb") as f:
|
|
||||||
files = {"file": (filename, f, "text/markdown")}
|
|
||||||
resp = requests.post(url, headers={"Authorization": HEADERS["Authorization"]}, files=files)
|
|
||||||
resp.raise_for_status()
|
|
||||||
data = resp.json()
|
|
||||||
return data.get("id") or data.get("data", {}).get("id")
|
|
||||||
|
|
||||||
|
|
||||||
def add_file_to_knowledge(knowledge_id, file_id):
|
|
||||||
"""Attach an uploaded file to a knowledge collection."""
|
|
||||||
url = f"{BASE_URL}/api/v1/knowledge/{knowledge_id}/files/"
|
|
||||||
payload = {"file_id": file_id}
|
|
||||||
resp = requests.post(url, headers=HEADERS, json=payload)
|
|
||||||
resp.raise_for_status()
|
|
||||||
|
|
||||||
|
|
||||||
def process_collection(name, description, file_paths):
|
|
||||||
"""Create a collection and upload+attach all files."""
|
|
||||||
print(f"\n Creating collection: {name}")
|
|
||||||
kid = create_knowledge(name, description)
|
|
||||||
print(f" -> ID: {kid}")
|
|
||||||
|
|
||||||
for fp in file_paths:
|
|
||||||
if not os.path.isfile(fp):
|
|
||||||
print(f" [SKIP] Not found: {fp}")
|
|
||||||
continue
|
|
||||||
print(f" Uploading: {os.path.basename(fp)}")
|
|
||||||
try:
|
|
||||||
fid = upload_file(fp)
|
|
||||||
add_file_to_knowledge(kid, fid)
|
|
||||||
except Exception as e:
|
|
||||||
print(f" [ERROR] {e}")
|
|
||||||
print(f" Done: {name}")
|
|
||||||
|
|
||||||
|
|
||||||
def main():
|
|
||||||
if len(sys.argv) < 2:
|
|
||||||
print(__doc__)
|
|
||||||
print(f"\nUsage: python3 {sys.argv[0]} <API_KEY>")
|
|
||||||
sys.exit(1)
|
|
||||||
|
|
||||||
api_key = sys.argv[1]
|
|
||||||
set_api_key(api_key)
|
|
||||||
|
|
||||||
# Verify connectivity
|
|
||||||
try:
|
|
||||||
r = requests.get(f"{BASE_URL}/api/v1/users/", headers=HEADERS, timeout=5)
|
|
||||||
r.raise_for_status()
|
|
||||||
print(f"Connected to Open WebUI at {BASE_URL}")
|
|
||||||
except Exception as e:
|
|
||||||
print(f"ERROR: Cannot connect to Open WebUI: {e}")
|
|
||||||
sys.exit(1)
|
|
||||||
|
|
||||||
os.chdir(REPO_ROOT)
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------
|
|
||||||
# Collection 1: Research Stack Core
|
|
||||||
# ---------------------------------------------------------------
|
|
||||||
process_collection(
|
|
||||||
"Research Stack Core",
|
|
||||||
"Central project documents: overview, map, concepts, and roadmap.",
|
|
||||||
[
|
|
||||||
"README.md",
|
|
||||||
"PROJECT_MAP.md",
|
|
||||||
"CONCEPTS.md",
|
|
||||||
"TODO_MAP.md",
|
|
||||||
],
|
|
||||||
)
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------
|
|
||||||
# Collection 2: GCCL Theory
|
|
||||||
# ---------------------------------------------------------------
|
|
||||||
process_collection(
|
|
||||||
"GCCL Theory",
|
|
||||||
"Genetic-Code Compression Language theoretical foundations.",
|
|
||||||
[
|
|
||||||
"docs/research/GCCL_GENETIC_INFORMATION_MIXTURE_PRIMITIVES.md",
|
|
||||||
"docs/research/GCCL_THEORY_INTRO.md",
|
|
||||||
],
|
|
||||||
)
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------
|
|
||||||
# Collection 3: KOTC & Daemon Systems
|
|
||||||
# ---------------------------------------------------------------
|
|
||||||
process_collection(
|
|
||||||
"KOTC & Daemon Systems",
|
|
||||||
"Knowledge-Of-Task-Completion daemon architecture.",
|
|
||||||
[
|
|
||||||
"docs/research/KOTC_COMPLETION_DAEMON.md",
|
|
||||||
],
|
|
||||||
)
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------
|
|
||||||
# Collection 4: VLB & Witness Substrate
|
|
||||||
# ---------------------------------------------------------------
|
|
||||||
process_collection(
|
|
||||||
"VLB & Witness Substrate",
|
|
||||||
"Very-Large-Block witness and substrate estimation.",
|
|
||||||
[
|
|
||||||
"docs/research/VLB_NIBBLE_DELTA_WITNESS_SUBSTRATE_ESTIMATE.md",
|
|
||||||
],
|
|
||||||
)
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------
|
|
||||||
# Collection 5: FAMM & Route Memory
|
|
||||||
# ---------------------------------------------------------------
|
|
||||||
process_collection(
|
|
||||||
"FAMM & Route Memory",
|
|
||||||
"Fluid-Automata Memory Model and stigmergic routing.",
|
|
||||||
[
|
|
||||||
"docs/famm/FAMM_Stigmergic_Route_Memory.md",
|
|
||||||
],
|
|
||||||
)
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------
|
|
||||||
# Collection 6: Roadmaps & Strategy
|
|
||||||
# ---------------------------------------------------------------
|
|
||||||
process_collection(
|
|
||||||
"Roadmaps & Strategy",
|
|
||||||
"Project roadmaps and strategic planning documents.",
|
|
||||||
[
|
|
||||||
"docs/roadmaps/RESEARCH_STACK_FOREST_MAP_WATERFALL.md",
|
|
||||||
],
|
|
||||||
)
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------
|
|
||||||
# Collection 7: Speculative Materials
|
|
||||||
# ---------------------------------------------------------------
|
|
||||||
process_collection(
|
|
||||||
"Speculative Materials",
|
|
||||||
"Exploratory and speculative research notes.",
|
|
||||||
[
|
|
||||||
"docs/speculative-materials/PhotonChasedFerriteTraceFormation.md",
|
|
||||||
],
|
|
||||||
)
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------
|
|
||||||
# Collection 8: Lean Formalism READMEs
|
|
||||||
# ---------------------------------------------------------------
|
|
||||||
process_collection(
|
|
||||||
"Lean Formalism READMEs",
|
|
||||||
"Per-domain READMEs for the Lean formalism sub-projects.",
|
|
||||||
[
|
|
||||||
"0-Core-Formalism/README.md",
|
|
||||||
"1-Distributed-Systems/README.md",
|
|
||||||
"2-Search-Space/README.md",
|
|
||||||
"3-Mathematical-Models/README.md",
|
|
||||||
"4-Infrastructure/README.md",
|
|
||||||
"5-Applications/README.md",
|
|
||||||
"6-Documentation/README.md",
|
|
||||||
],
|
|
||||||
)
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------
|
|
||||||
# Collection 9: Documentation Guides
|
|
||||||
# ---------------------------------------------------------------
|
|
||||||
process_collection(
|
|
||||||
"Documentation Guides",
|
|
||||||
"Human-readable explanations, pitches, and guides.",
|
|
||||||
[
|
|
||||||
"6-Documentation/EXPLANATION_FOR_HUMANS.md",
|
|
||||||
"6-Documentation/ELEVATOR_PITCH.md",
|
|
||||||
"6-Documentation/calculator_plain_math.md",
|
|
||||||
],
|
|
||||||
)
|
|
||||||
|
|
||||||
print("\n========================================")
|
|
||||||
print("All knowledge collections populated.")
|
|
||||||
print("Go to http://127.0.0.1:3000 and check")
|
|
||||||
print("Workspace → Knowledge to browse them.")
|
|
||||||
print("========================================\n")
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
main()
|
|
||||||
|
|
@ -1,72 +0,0 @@
|
||||||
#!/usr/bin/env bash
|
|
||||||
set -euo pipefail
|
|
||||||
|
|
||||||
# remove-tailnet-nodes-api.sh
|
|
||||||
# Batch-removes all old Tailscale nodes via API.
|
|
||||||
# Usage: ./scripts/remove-tailnet-nodes-api.sh <TS_API_KEY>
|
|
||||||
|
|
||||||
API_KEY="${1:-}"
|
|
||||||
if [[ -z "$API_KEY" ]]; then
|
|
||||||
echo "Usage: $0 <TS_API_KEY>"
|
|
||||||
echo "Get your API key at: https://login.tailscale.com/admin/settings/keys"
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
|
|
||||||
TAILNET=$(tailscale status --json 2>/dev/null | python3 -c "import sys,json; print(json.load(sys.stdin).get('MagicDNSSuffix','unknown'))")
|
|
||||||
if [[ "$TAILNET" == "unknown" ]]; then
|
|
||||||
echo "Could not determine tailnet. Are you logged into Tailscale?"
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
|
|
||||||
# Nodes to remove (all except Node-00001 which is the current re-authed node)
|
|
||||||
OLD_NODES=(
|
|
||||||
"architect"
|
|
||||||
"desktop-0u2ceal"
|
|
||||||
"foxtop"
|
|
||||||
"ip-172-31-25-81"
|
|
||||||
"judge"
|
|
||||||
"laptop-1"
|
|
||||||
"netcup-router"
|
|
||||||
"racknerd-510bd9c"
|
|
||||||
"racknerd-atl"
|
|
||||||
"qfox"
|
|
||||||
"QFox"
|
|
||||||
)
|
|
||||||
|
|
||||||
echo "Tailnet: $TAILNET"
|
|
||||||
echo "Removing old nodes via API..."
|
|
||||||
echo ""
|
|
||||||
|
|
||||||
# Fetch all devices
|
|
||||||
DEVICES_JSON=$(curl -sS \
|
|
||||||
-H "Authorization: Bearer $API_KEY" \
|
|
||||||
"https://api.tailscale.com/api/v2/tailnet/-/devices")
|
|
||||||
|
|
||||||
# Extract device IDs for old nodes
|
|
||||||
for node in "${OLD_NODES[@]}"; do
|
|
||||||
DEVICE_ID=$(echo "$DEVICES_JSON" | python3 -c "
|
|
||||||
import sys, json
|
|
||||||
devices = json.load(sys.stdin).get('devices', [])
|
|
||||||
for d in devices:
|
|
||||||
if d.get('name', '').split('.')[0] == '$node':
|
|
||||||
print(d.get('id'))
|
|
||||||
break
|
|
||||||
")
|
|
||||||
if [[ -n "$DEVICE_ID" ]]; then
|
|
||||||
echo "Removing $node (ID: $DEVICE_ID)..."
|
|
||||||
HTTP_STATUS=$(curl -sS -o /dev/null -w "%{http_code}" \
|
|
||||||
-X DELETE \
|
|
||||||
-H "Authorization: Bearer $API_KEY" \
|
|
||||||
"https://api.tailscale.com/api/v2/device/$DEVICE_ID")
|
|
||||||
if [[ "$HTTP_STATUS" == "200" || "$HTTP_STATUS" == "204" ]]; then
|
|
||||||
echo " OK (HTTP $HTTP_STATUS)"
|
|
||||||
else
|
|
||||||
echo " FAILED (HTTP $HTTP_STATUS)"
|
|
||||||
fi
|
|
||||||
else
|
|
||||||
echo "$node: not found (already removed?)"
|
|
||||||
fi
|
|
||||||
done
|
|
||||||
|
|
||||||
echo ""
|
|
||||||
echo "Done. Verify at: https://login.tailscale.com/admin/machines"
|
|
||||||
|
|
@ -1,58 +0,0 @@
|
||||||
#!/usr/bin/env bash
|
|
||||||
set -euo pipefail
|
|
||||||
|
|
||||||
# reset-tailnet.sh
|
|
||||||
# Clears all Tailscale nodes and re-authenticates current node as Node-00001
|
|
||||||
|
|
||||||
CURRENT_HOSTNAME=$(tailscale status --json 2>/dev/null | python3 -c "import sys,json; print(json.load(sys.stdin).get('Self',{}).get('HostName','unknown'))")
|
|
||||||
TAILNET=$(tailscale status --json 2>/dev/null | python3 -c "import sys,json; print(json.load(sys.stdin).get('MagicDNSSuffix','unknown'))")
|
|
||||||
|
|
||||||
echo "=========================================="
|
|
||||||
echo " Tailnet Reset Tool"
|
|
||||||
echo "=========================================="
|
|
||||||
echo "Current node: $CURRENT_HOSTNAME"
|
|
||||||
echo "Tailnet: $TAILNET"
|
|
||||||
echo ""
|
|
||||||
|
|
||||||
# Step 1: Logout current node
|
|
||||||
echo "[1/2] Logging out current node ($CURRENT_HOSTNAME)..."
|
|
||||||
sudo tailscale logout
|
|
||||||
echo "Done. Node removed from tailnet."
|
|
||||||
echo ""
|
|
||||||
|
|
||||||
# Step 2: Re-auth as Node-00001
|
|
||||||
echo "[2/2] Re-authenticating as Node-00001..."
|
|
||||||
echo "You will see an auth URL. Open it in your browser to complete login."
|
|
||||||
echo ""
|
|
||||||
sudo tailscale up --hostname=Node-00001 --ssh --accept-routes
|
|
||||||
|
|
||||||
echo ""
|
|
||||||
echo "=========================================="
|
|
||||||
echo "Current node re-authenticated as Node-00001"
|
|
||||||
echo ""
|
|
||||||
tailscale status
|
|
||||||
echo ""
|
|
||||||
echo "=========================================="
|
|
||||||
echo "NEXT STEPS: Remove remaining nodes"
|
|
||||||
echo "=========================================="
|
|
||||||
echo ""
|
|
||||||
echo "The other 9 nodes must be removed via the Tailscale admin console"
|
|
||||||
echo "or API since they are not reachable from this machine."
|
|
||||||
echo ""
|
|
||||||
echo "Option A: Manual removal (recommended)"
|
|
||||||
echo " 1. Go to: https://login.tailscale.com/admin/machines"
|
|
||||||
echo " 2. Select each old node and click 'Remove...'"
|
|
||||||
echo " 3. Old nodes: architect, desktop-0u2ceal, foxtop, ip-172-31-25-81,"
|
|
||||||
echo " judge, laptop-1, netcup-router, racknerd-510bd9c, racknerd-atl"
|
|
||||||
echo ""
|
|
||||||
echo "Option B: API removal (batch)"
|
|
||||||
echo " 1. Get an API key: https://login.tailscale.com/admin/settings/keys"
|
|
||||||
echo " 2. Run: ./scripts/remove-tailnet-nodes-api.sh <YOUR_API_KEY>"
|
|
||||||
echo ""
|
|
||||||
echo "Option C: SSH into active nodes and logout"
|
|
||||||
echo " ssh judge 'sudo tailscale logout'"
|
|
||||||
echo " ssh netcup-router 'sudo tailscale logout'"
|
|
||||||
echo " ssh ip-172-31-25-81 'sudo tailscale logout'"
|
|
||||||
echo ""
|
|
||||||
echo "After clearing all nodes, run: ./scripts/clean-tailscale-refs.sh"
|
|
||||||
echo "to remove stale Tailscale references from this repo."
|
|
||||||
|
|
@ -1,366 +0,0 @@
|
||||||
#!/usr/bin/env python3
|
|
||||||
"""
|
|
||||||
Deep Review Agent
|
|
||||||
=================
|
|
||||||
After a paper is ingested, this agent:
|
|
||||||
1. Extracts full text (or first N pages)
|
|
||||||
2. Runs an LLM (local Ollama) to generate a structured review
|
|
||||||
3. Stores the review back in the local index
|
|
||||||
|
|
||||||
Usage:
|
|
||||||
python3 review_agent.py --paper /path/to/paper.pdf
|
|
||||||
python3 review_agent.py --zotero-key B78T16BK
|
|
||||||
python3 review_agent.py --batch 10 # Review 10 un-reviewed papers
|
|
||||||
python3 review_agent.py --daemon # Background loop
|
|
||||||
|
|
||||||
Review JSON schema:
|
|
||||||
{
|
|
||||||
"title": str,
|
|
||||||
"authors": [str],
|
|
||||||
"year": str,
|
|
||||||
"venue": str,
|
|
||||||
"tl_dr": str, # 1-sentence elevator pitch
|
|
||||||
"methods": str, # What they actually did
|
|
||||||
"key_findings": [str], # Bullet list of top results
|
|
||||||
"limitations": [str], # Weaknesses / caveats
|
|
||||||
"relevance": str, # Why this matters to your work
|
|
||||||
"citations_to_follow": [str], # Key refs worth chasing
|
|
||||||
"confidence": str, # low / medium / high
|
|
||||||
"read_again": bool # Should you re-read in depth?
|
|
||||||
}
|
|
||||||
"""
|
|
||||||
|
|
||||||
import argparse
|
|
||||||
import json
|
|
||||||
import os
|
|
||||||
import re
|
|
||||||
import sqlite3
|
|
||||||
import subprocess
|
|
||||||
import sys
|
|
||||||
import textwrap
|
|
||||||
import time
|
|
||||||
from pathlib import Path
|
|
||||||
from typing import Any, Dict, List, Optional
|
|
||||||
|
|
||||||
# ── Config ────────────────────────────────────────────────────────────────────
|
|
||||||
OLLAMA_HOST = os.getenv("OLLAMA_HOST", "http://localhost:11434")
|
|
||||||
OLLAMA_MODEL = os.getenv("OLLAMA_MODEL", "qwen2.5-coder:14b")
|
|
||||||
INDEX_DB = Path.home() / "Research Stack" / "data" / "substrate_index.db"
|
|
||||||
ZOTERO_DB = Path.home() / "Zotero" / "zotero.sqlite"
|
|
||||||
MAX_PAGES = 12 # pages to feed to LLM
|
|
||||||
BATCH_SIZE = 5 # papers per batch
|
|
||||||
|
|
||||||
# ── Review Storage ──────────────────────────────────────────────────────────
|
|
||||||
class ReviewStore:
|
|
||||||
SCHEMA = """
|
|
||||||
CREATE TABLE IF NOT EXISTS paper_reviews (
|
|
||||||
paper_key TEXT PRIMARY KEY,
|
|
||||||
zotero_key TEXT,
|
|
||||||
arxiv_id TEXT,
|
|
||||||
local_path TEXT,
|
|
||||||
review_json TEXT,
|
|
||||||
tl_dr TEXT,
|
|
||||||
relevance TEXT,
|
|
||||||
confidence TEXT,
|
|
||||||
read_again INTEGER,
|
|
||||||
created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP
|
|
||||||
);
|
|
||||||
CREATE INDEX IF NOT EXISTS idx_review_arxiv ON paper_reviews(arxiv_id);
|
|
||||||
CREATE INDEX IF NOT EXISTS idx_review_relevance ON paper_reviews(relevance);
|
|
||||||
"""
|
|
||||||
|
|
||||||
def __init__(self, db_path: Path = INDEX_DB):
|
|
||||||
self.db_path = db_path
|
|
||||||
self._ensure()
|
|
||||||
|
|
||||||
def _ensure(self):
|
|
||||||
with sqlite3.connect(str(self.db_path)) as conn:
|
|
||||||
conn.executescript(self.SCHEMA)
|
|
||||||
|
|
||||||
def save(self, key: str, review: Dict, zotero_key: Optional[str] = None,
|
|
||||||
arxiv_id: Optional[str] = None, local_path: Optional[str] = None):
|
|
||||||
with sqlite3.connect(str(self.db_path)) as conn:
|
|
||||||
conn.execute(
|
|
||||||
"""INSERT OR REPLACE INTO paper_reviews
|
|
||||||
(paper_key, zotero_key, arxiv_id, local_path, review_json, tl_dr, relevance, confidence, read_again)
|
|
||||||
VALUES (?,?,?,?,?,?,?,?,?)""",
|
|
||||||
(
|
|
||||||
key,
|
|
||||||
zotero_key,
|
|
||||||
arxiv_id,
|
|
||||||
local_path,
|
|
||||||
json.dumps(review),
|
|
||||||
review.get("tl_dr", "")[:500],
|
|
||||||
review.get("relevance", "")[:500],
|
|
||||||
review.get("confidence", "medium"),
|
|
||||||
1 if review.get("read_again", False) else 0,
|
|
||||||
),
|
|
||||||
)
|
|
||||||
conn.commit()
|
|
||||||
|
|
||||||
def get(self, key: str) -> Optional[Dict]:
|
|
||||||
with sqlite3.connect(str(self.db_path)) as conn:
|
|
||||||
cur = conn.execute("SELECT review_json FROM paper_reviews WHERE paper_key = ?", (key,))
|
|
||||||
row = cur.fetchone()
|
|
||||||
if row:
|
|
||||||
return json.loads(row[0])
|
|
||||||
return None
|
|
||||||
|
|
||||||
def list_unreviewed(self, limit: int = 10) -> List[Dict]:
|
|
||||||
"""Return papers in local_pdfs that have no review yet."""
|
|
||||||
with sqlite3.connect(str(self.db_path)) as conn:
|
|
||||||
conn.row_factory = sqlite3.Row
|
|
||||||
cur = conn.execute(
|
|
||||||
"""SELECT p.* FROM local_pdfs p
|
|
||||||
LEFT JOIN paper_reviews r ON p.path = r.local_path
|
|
||||||
WHERE r.paper_key IS NULL
|
|
||||||
LIMIT ?""", (limit,)
|
|
||||||
)
|
|
||||||
return [dict(r) for r in cur.fetchall()]
|
|
||||||
|
|
||||||
# ── Text Extractor ──────────────────────────────────────────────────────────
|
|
||||||
class TextExtractor:
|
|
||||||
def extract(self, pdf_path: Path, max_pages: int = MAX_PAGES) -> str:
|
|
||||||
try:
|
|
||||||
result = subprocess.run(
|
|
||||||
["pdftotext", "-l", str(max_pages), str(pdf_path), "-"],
|
|
||||||
capture_output=True, text=True, timeout=30,
|
|
||||||
)
|
|
||||||
if result.returncode == 0:
|
|
||||||
return result.stdout
|
|
||||||
except Exception:
|
|
||||||
pass
|
|
||||||
return ""
|
|
||||||
|
|
||||||
def extract_from_zotero(self, zotero_key: str, max_pages: int = MAX_PAGES) -> str:
|
|
||||||
# Find attachment path via Zotero storage
|
|
||||||
# For now, fall back to searching local_pdfs by zotero_key
|
|
||||||
with sqlite3.connect(str(INDEX_DB)) as conn:
|
|
||||||
cur = conn.execute("SELECT path FROM local_pdfs WHERE zotero_key = ?", (zotero_key,))
|
|
||||||
row = cur.fetchone()
|
|
||||||
if row:
|
|
||||||
return self.extract(Path(row[0]), max_pages)
|
|
||||||
return ""
|
|
||||||
|
|
||||||
# ── LLM Reviewer ─────────────────────────────────────────────────────────────
|
|
||||||
class LLMReviewer:
|
|
||||||
PROMPT_TEMPLATE = """You are a senior research scientist reviewing a preprint for a sovereign research lab focused on AI, topology, compression, and mathematical formalization.
|
|
||||||
|
|
||||||
TASK: Read the paper text below and produce a JSON review object.
|
|
||||||
|
|
||||||
RULES:
|
|
||||||
- Be concise but specific. No fluff.
|
|
||||||
- If the text is garbled or too short, note low confidence.
|
|
||||||
- Focus on: methods, novelty, reproducibility, and relevance to integer-only computing / topological state machines / manifold compression.
|
|
||||||
- Return ONLY valid JSON. No markdown fences. No prose outside the JSON.
|
|
||||||
|
|
||||||
PAPER TEXT (first {pages} pages):
|
|
||||||
---
|
|
||||||
{text}
|
|
||||||
---
|
|
||||||
|
|
||||||
REQUIRED JSON SCHEMA:
|
|
||||||
{{
|
|
||||||
"title": "paper title",
|
|
||||||
"authors": ["name1", "name2"],
|
|
||||||
"year": "YYYY",
|
|
||||||
"venue": "arXiv or journal/conference",
|
|
||||||
"tl_dr": "One-sentence summary.",
|
|
||||||
"methods": "What they did, technically.",
|
|
||||||
"key_findings": ["Finding A", "Finding B"],
|
|
||||||
"limitations": ["Limitation A", "Limitation B"],
|
|
||||||
"relevance": "Why this matters to our work (integer math, topological compression, Lean proofs, etc).",
|
|
||||||
"citations_to_follow": ["Author et al. YYYY — Topic"],
|
|
||||||
"confidence": "high|medium|low",
|
|
||||||
"read_again": true|false
|
|
||||||
}}
|
|
||||||
"""
|
|
||||||
|
|
||||||
def __init__(self, model: str = OLLAMA_MODEL, host: str = OLLAMA_HOST):
|
|
||||||
self.model = model
|
|
||||||
self.host = host
|
|
||||||
|
|
||||||
def review(self, text: str, pages: int = MAX_PAGES, force_stub: bool = False) -> Dict[str, Any]:
|
|
||||||
if force_stub or not self._ollama_alive():
|
|
||||||
return self.stub_review(text)
|
|
||||||
prompt = self.PROMPT_TEMPLATE.format(pages=pages, text=text[:15000])
|
|
||||||
payload = {
|
|
||||||
"model": self.model,
|
|
||||||
"prompt": prompt,
|
|
||||||
"stream": False,
|
|
||||||
"format": "json",
|
|
||||||
"options": {"temperature": 0.3, "num_ctx": 8192},
|
|
||||||
}
|
|
||||||
try:
|
|
||||||
import urllib.request
|
|
||||||
req = urllib.request.Request(
|
|
||||||
f"{self.host}/api/generate",
|
|
||||||
data=json.dumps(payload).encode(),
|
|
||||||
headers={"Content-Type": "application/json"},
|
|
||||||
method="POST",
|
|
||||||
)
|
|
||||||
with urllib.request.urlopen(req, timeout=180) as resp:
|
|
||||||
data = json.loads(resp.read().decode())
|
|
||||||
raw = data.get("response", "")
|
|
||||||
raw = raw.strip()
|
|
||||||
if raw.startswith("```json"):
|
|
||||||
raw = raw[7:]
|
|
||||||
if raw.startswith("```"):
|
|
||||||
raw = raw[3:]
|
|
||||||
if raw.endswith("```"):
|
|
||||||
raw = raw[:-3]
|
|
||||||
raw = raw.strip()
|
|
||||||
if not raw:
|
|
||||||
raise ValueError("Empty response from LLM")
|
|
||||||
parsed = json.loads(raw)
|
|
||||||
for k in ["key_findings", "limitations", "citations_to_follow"]:
|
|
||||||
if k not in parsed:
|
|
||||||
parsed[k] = []
|
|
||||||
elif isinstance(parsed[k], str):
|
|
||||||
parsed[k] = [parsed[k]]
|
|
||||||
return parsed
|
|
||||||
except Exception as e:
|
|
||||||
return self.stub_review(text, meta={"error": str(e)})
|
|
||||||
|
|
||||||
def _ollama_alive(self) -> bool:
|
|
||||||
"""Check if the chosen model is loaded and Ollama is responsive."""
|
|
||||||
try:
|
|
||||||
import urllib.request
|
|
||||||
# Check /api/ps for loaded models first
|
|
||||||
req = urllib.request.Request(f"{self.host}/api/ps", method="GET")
|
|
||||||
with urllib.request.urlopen(req, timeout=5) as resp:
|
|
||||||
data = json.loads(resp.read().decode())
|
|
||||||
running = [m.get("name", "") for m in data.get("models", [])]
|
|
||||||
if any(self.model in r for r in running):
|
|
||||||
return True
|
|
||||||
# Fallback: check if model exists in library
|
|
||||||
req = urllib.request.Request(f"{self.host}/api/tags", method="GET")
|
|
||||||
with urllib.request.urlopen(req, timeout=5) as resp:
|
|
||||||
data = json.loads(resp.read().decode())
|
|
||||||
models = [m["name"] for m in data.get("models", [])]
|
|
||||||
return self.model in models
|
|
||||||
except Exception:
|
|
||||||
return False
|
|
||||||
|
|
||||||
def stub_review(self, text: str, meta: Optional[Dict] = None) -> Dict[str, Any]:
|
|
||||||
"""Generate a metadata-only review without calling the LLM."""
|
|
||||||
stub = {
|
|
||||||
"title": meta.get("title", "Unknown") if meta else "Unknown",
|
|
||||||
"authors": meta.get("authors", []) if meta else [],
|
|
||||||
"year": meta.get("year", "") if meta else "",
|
|
||||||
"venue": "arXiv" if meta and meta.get("arxiv_id") else "Unknown",
|
|
||||||
"tl_dr": "Stub review — LLM not available. Re-run with working Ollama for deep analysis.",
|
|
||||||
"methods": "",
|
|
||||||
"key_findings": [],
|
|
||||||
"limitations": ["No LLM review performed."],
|
|
||||||
"relevance": "unknown",
|
|
||||||
"citations_to_follow": [],
|
|
||||||
"confidence": "low",
|
|
||||||
"read_again": False,
|
|
||||||
}
|
|
||||||
lines = [l.strip() for l in text.splitlines() if l.strip()]
|
|
||||||
if lines and (not meta or not meta.get("title")):
|
|
||||||
stub["title"] = lines[0][:200]
|
|
||||||
# Try to grab authors from second line if it starts with "Authors:"
|
|
||||||
if len(lines) > 1 and lines[1].lower().startswith("authors"):
|
|
||||||
stub["authors"] = [a.strip() for a in lines[1].replace("Authors:", "").split(",") if a.strip()]
|
|
||||||
return stub
|
|
||||||
|
|
||||||
# ── Review Pipeline ─────────────────────────────────────────────────────────
|
|
||||||
class ReviewPipeline:
|
|
||||||
def __init__(self):
|
|
||||||
self.store = ReviewStore()
|
|
||||||
self.extractor = TextExtractor()
|
|
||||||
self.reviewer = LLMReviewer()
|
|
||||||
|
|
||||||
def review_pdf(self, pdf_path: Path) -> Dict:
|
|
||||||
key = f"pdf:{pdf_path}"
|
|
||||||
existing = self.store.get(key)
|
|
||||||
if existing:
|
|
||||||
return {"status": "already_reviewed", "review": existing}
|
|
||||||
|
|
||||||
text = self.extractor.extract(pdf_path)
|
|
||||||
if not text.strip():
|
|
||||||
return {"status": "no_text", "review": None}
|
|
||||||
|
|
||||||
review = self.reviewer.review(text)
|
|
||||||
self.store.save(key, review, local_path=str(pdf_path))
|
|
||||||
return {"status": "reviewed", "review": review}
|
|
||||||
|
|
||||||
def review_zotero_key(self, zkey: str) -> Dict:
|
|
||||||
key = f"zotero:{zkey}"
|
|
||||||
existing = self.store.get(key)
|
|
||||||
if existing:
|
|
||||||
return {"status": "already_reviewed", "review": existing}
|
|
||||||
|
|
||||||
text = self.extractor.extract_from_zotero(zkey)
|
|
||||||
if not text.strip():
|
|
||||||
return {"status": "no_text", "review": None}
|
|
||||||
|
|
||||||
review = self.reviewer.review(text)
|
|
||||||
self.store.save(key, review, zotero_key=zkey)
|
|
||||||
return {"status": "reviewed", "review": review}
|
|
||||||
|
|
||||||
def batch_review(self, limit: int = BATCH_SIZE) -> List[Dict]:
|
|
||||||
unreviewed = self.store.list_unreviewed(limit)
|
|
||||||
results = []
|
|
||||||
for item in unreviewed:
|
|
||||||
path = Path(item["path"])
|
|
||||||
if not path.exists():
|
|
||||||
continue
|
|
||||||
res = self.review_pdf(path)
|
|
||||||
results.append({"path": str(path), **res})
|
|
||||||
time.sleep(2) # be polite to Ollama
|
|
||||||
return results
|
|
||||||
|
|
||||||
# ── CLI ─────────────────────────────────────────────────────────────────────
|
|
||||||
def main():
|
|
||||||
parser = argparse.ArgumentParser(description="Deep Review Agent")
|
|
||||||
parser.add_argument("--paper", help="Path to PDF to review")
|
|
||||||
parser.add_argument("--zotero-key", help="Zotero item key to review")
|
|
||||||
parser.add_argument("--batch", type=int, help="Review N unreviewed papers")
|
|
||||||
parser.add_argument("--daemon", action="store_true", help="Loop forever reviewing new papers")
|
|
||||||
parser.add_argument("--interval", type=int, default=300, help="Seconds between daemon scans")
|
|
||||||
parser.add_argument("--show", help="Show existing review for a key")
|
|
||||||
parser.add_argument("--model", default=OLLAMA_MODEL, help="Ollama model name")
|
|
||||||
args = parser.parse_args()
|
|
||||||
|
|
||||||
pipeline = ReviewPipeline()
|
|
||||||
pipeline.reviewer.model = args.model
|
|
||||||
|
|
||||||
if args.paper:
|
|
||||||
res = pipeline.review_pdf(Path(args.paper))
|
|
||||||
print(json.dumps(res, indent=2, default=str))
|
|
||||||
elif args.zotero_key:
|
|
||||||
res = pipeline.review_zotero_key(args.zotero_key)
|
|
||||||
print(json.dumps(res, indent=2, default=str))
|
|
||||||
elif args.batch:
|
|
||||||
results = pipeline.batch_review(limit=args.batch)
|
|
||||||
for r in results:
|
|
||||||
print(f"\n{'='*60}")
|
|
||||||
print(f"📄 {r['path']}")
|
|
||||||
print(f" Status: {r['status']}")
|
|
||||||
if r.get("review"):
|
|
||||||
rev = r["review"]
|
|
||||||
print(f" TL;DR: {rev.get('tl_dr')}")
|
|
||||||
print(f" Confidence: {rev.get('confidence')}")
|
|
||||||
print(f" Read again: {rev.get('read_again')}")
|
|
||||||
elif args.daemon:
|
|
||||||
print(f"👁️ Review daemon started (model={args.model}, interval={args.interval}s)")
|
|
||||||
while True:
|
|
||||||
results = pipeline.batch_review(limit=3)
|
|
||||||
if results:
|
|
||||||
for r in results:
|
|
||||||
print(f"[reviewed] {r['path']} → {r['status']}")
|
|
||||||
time.sleep(args.interval)
|
|
||||||
elif args.show:
|
|
||||||
rev = pipeline.store.get(args.show)
|
|
||||||
if rev:
|
|
||||||
print(json.dumps(rev, indent=2))
|
|
||||||
else:
|
|
||||||
print("No review found for that key.")
|
|
||||||
else:
|
|
||||||
parser.print_help()
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
main()
|
|
||||||
|
|
@ -1,721 +0,0 @@
|
||||||
#!/usr/bin/env python3
|
|
||||||
"""
|
|
||||||
ScienceHub MCP Server — Sovereign Research Surface
|
|
||||||
==================================================
|
|
||||||
An MCP server that lets an LLM say "I need X" and automatically:
|
|
||||||
1. Searches your local corpus (Zotero + PDFs)
|
|
||||||
2. If missing, fetches from arXiv / Semantic Scholar
|
|
||||||
3. Ingests into Zotero + local storage
|
|
||||||
4. Returns a review/abstract
|
|
||||||
|
|
||||||
Usage (for Claude Desktop / Cline / etc):
|
|
||||||
{
|
|
||||||
"mcpServers": {
|
|
||||||
"sciencehub": {
|
|
||||||
"command": "python3",
|
|
||||||
"args": ["/home/allaun/Documents/Research Stack/scripts/sciencehub_mcp.py"]
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
Tools:
|
|
||||||
- need : "I need <topic>" → full pipeline
|
|
||||||
- search_local : Query local corpus index
|
|
||||||
- fetch_arxiv : Download + cache arXiv PDF
|
|
||||||
- ingest_to_zotero: Import a PDF into Zotero SQLite
|
|
||||||
- review_paper : Extract metadata + quick review from PDF
|
|
||||||
- corpus_report : Get state of the local research library
|
|
||||||
"""
|
|
||||||
|
|
||||||
import argparse
|
|
||||||
import asyncio
|
|
||||||
import json
|
|
||||||
import os
|
|
||||||
import random
|
|
||||||
import re
|
|
||||||
import shutil
|
|
||||||
import sqlite3
|
|
||||||
import string
|
|
||||||
import sys
|
|
||||||
import textwrap
|
|
||||||
import time
|
|
||||||
import urllib.parse
|
|
||||||
import urllib.request
|
|
||||||
from dataclasses import dataclass, field
|
|
||||||
from pathlib import Path
|
|
||||||
from typing import Any, Dict, List, Optional
|
|
||||||
|
|
||||||
# ── Paths ─────────────────────────────────────────────────────────────────────
|
|
||||||
RESEARCH_STACK = Path("/home/allaun/Documents/Research Stack")
|
|
||||||
ZOTERO_DB = Path.home() / "Zotero" / "zotero.sqlite"
|
|
||||||
INDEX_DB = Path.home() / "Research Stack" / "data" / "substrate_index.db"
|
|
||||||
INGEST_DIR = Path.home() / "Downloads" / "data" / "Downloads_from_internet" / "Deep Research"
|
|
||||||
ARXIV_CACHE = INGEST_DIR / "alphaXiv_PDFs_2026_04"
|
|
||||||
|
|
||||||
# Ensure dirs exist
|
|
||||||
INGEST_DIR.mkdir(parents=True, exist_ok=True)
|
|
||||||
ARXIV_CACHE.mkdir(parents=True, exist_ok=True)
|
|
||||||
|
|
||||||
# ── MCP SDK (optional — graceful fallback) ────────────────────────────────────
|
|
||||||
try:
|
|
||||||
from mcp.server import Server
|
|
||||||
from mcp.server.stdio import stdio_server
|
|
||||||
from mcp.types import Tool, TextContent
|
|
||||||
HAS_MCP = True
|
|
||||||
except ImportError:
|
|
||||||
HAS_MCP = False
|
|
||||||
print("[warn] MCP SDK not installed. Running in CLI mode.", file=sys.stderr)
|
|
||||||
|
|
||||||
# ── Data classes ─────────────────────────────────────────────────────────────
|
|
||||||
@dataclass
|
|
||||||
class Paper:
|
|
||||||
title: str
|
|
||||||
authors: List[str] = field(default_factory=list)
|
|
||||||
abstract: str = ""
|
|
||||||
url: str = ""
|
|
||||||
arxiv_id: Optional[str] = None
|
|
||||||
doi: Optional[str] = None
|
|
||||||
local_path: Optional[Path] = None
|
|
||||||
year: Optional[str] = None
|
|
||||||
|
|
||||||
# ── arXiv Client ─────────────────────────────────────────────────────────────
|
|
||||||
class ArxivClient:
|
|
||||||
BASE_QUERY = "https://export.arxiv.org/api/query"
|
|
||||||
BASE_PDF = "https://arxiv.org/pdf"
|
|
||||||
STOP_WORDS = {
|
|
||||||
"i", "need", "the", "a", "an", "is", "are", "was", "were", "be", "been",
|
|
||||||
"being", "have", "has", "had", "do", "does", "did", "will", "would",
|
|
||||||
"could", "should", "may", "might", "must", "shall", "can", "all", "you",
|
|
||||||
"we", "they", "it", "this", "that", "these", "those", "of", "in", "on",
|
|
||||||
"at", "to", "for", "with", "about", "against", "between", "into", "through",
|
|
||||||
"during", "before", "after", "above", "below", "from", "up", "down", "out",
|
|
||||||
"off", "over", "under", "again", "further", "then", "once", "and", "or",
|
|
||||||
"but", "if", "then", "else", "because", "until", "while", "so", "than",
|
|
||||||
"too", "very", "just", "now", "only", "also", "its", "his", "her", "their",
|
|
||||||
"our", "my", "your", "what", "which", "who", "when", "where", "why", "how",
|
|
||||||
"paper", "survey", "review", "article", "study", "work",
|
|
||||||
}
|
|
||||||
|
|
||||||
def search(self, query: str, max_results: int = 5) -> List[Paper]:
|
|
||||||
"""Search arXiv and return Paper objects."""
|
|
||||||
words = re.sub(r'[^\w\s]', ' ', query).lower().split()
|
|
||||||
keywords = [w for w in words if w not in self.STOP_WORDS][:3]
|
|
||||||
if not keywords:
|
|
||||||
keywords = words[:3]
|
|
||||||
raw_query = "+AND+".join(f"ti:{urllib.parse.quote(w)}" for w in keywords)
|
|
||||||
url = f"{self.BASE_QUERY}?search_query={raw_query}&max_results={max_results}&sortBy=relevance&sortOrder=descending"
|
|
||||||
try:
|
|
||||||
req = urllib.request.Request(url, headers={"User-Agent": "ScienceHub/0.1"})
|
|
||||||
with urllib.request.urlopen(req, timeout=30) as resp:
|
|
||||||
xml = resp.read().decode("utf-8")
|
|
||||||
return self._parse_feed(xml)
|
|
||||||
except Exception as e:
|
|
||||||
return [Paper(title=f"[ERROR] arXiv search failed: {e}", url=url)]
|
|
||||||
|
|
||||||
def _parse_feed(self, xml: str) -> List[Paper]:
|
|
||||||
import xml.etree.ElementTree as ET
|
|
||||||
ns = {"atom": "http://www.w3.org/2005/Atom", "arxiv": "http://arxiv.org/schemas/atom"}
|
|
||||||
root = ET.fromstring(xml)
|
|
||||||
papers: List[Paper] = []
|
|
||||||
for entry in root.findall("atom:entry", ns):
|
|
||||||
id_el = entry.find("atom:id", ns)
|
|
||||||
if id_el is None:
|
|
||||||
continue
|
|
||||||
arxiv_url = id_el.text.strip()
|
|
||||||
arxiv_id = arxiv_url.split("/")[-1]
|
|
||||||
title = entry.find("atom:title", ns).text.strip().replace("\n", " ")
|
|
||||||
summary = entry.find("atom:summary", ns).text.strip().replace("\n", " ")
|
|
||||||
authors = [a.find("atom:name", ns).text for a in entry.findall("atom:author", ns)]
|
|
||||||
published = entry.find("atom:published", ns).text[:4]
|
|
||||||
pdf_url = f"{self.BASE_PDF}/{arxiv_id}.pdf"
|
|
||||||
papers.append(Paper(
|
|
||||||
title=title, authors=authors, abstract=summary,
|
|
||||||
url=arxiv_url, arxiv_id=arxiv_id, year=published,
|
|
||||||
))
|
|
||||||
return papers
|
|
||||||
|
|
||||||
def download(self, arxiv_id: str, dest: Path) -> Path:
|
|
||||||
"""Download PDF to dest. Returns path to temp file."""
|
|
||||||
clean = re.sub(r"v\d+$", "", arxiv_id)
|
|
||||||
pdf_url = f"{self.BASE_PDF}/{clean}.pdf"
|
|
||||||
out_temp = dest / f"{clean}.pdf"
|
|
||||||
try:
|
|
||||||
urllib.request.urlretrieve(pdf_url, str(out_temp))
|
|
||||||
if out_temp.stat().st_size < 1024:
|
|
||||||
raise RuntimeError("Downloaded file is too small (likely an error page)")
|
|
||||||
return out_temp
|
|
||||||
except Exception as e:
|
|
||||||
if out_temp.exists():
|
|
||||||
out_temp.unlink()
|
|
||||||
raise RuntimeError(f"Failed to download {pdf_url}: {e}")
|
|
||||||
|
|
||||||
@staticmethod
|
|
||||||
def slug(title: str, arxiv_id: str, max_len: int = 80) -> str:
|
|
||||||
"""Create a safe filename slug from a paper title + arXiv ID."""
|
|
||||||
# Transliterate common math symbols
|
|
||||||
replacements = {
|
|
||||||
"\\": " ", "$": "", "^": "", "_": " ",
|
|
||||||
"{": "", "}": "", "~": " ", "%": "pct",
|
|
||||||
"&": "and", "#": "", "@": "at", "≈": "approx",
|
|
||||||
"∞": "inf", "∈": "in", "∀": "forall", "∃": "exists",
|
|
||||||
"∂": "d", "∇": "nabla", "α": "alpha", "β": "beta",
|
|
||||||
"γ": "gamma", "δ": "delta", "ε": "epsilon", "ζ": "zeta",
|
|
||||||
"η": "eta", "θ": "theta", "ι": "iota", "κ": "kappa",
|
|
||||||
"λ": "lambda", "μ": "mu", "ν": "nu", "ξ": "xi",
|
|
||||||
"π": "pi", "ρ": "rho", "σ": "sigma", "τ": "tau",
|
|
||||||
"φ": "phi", "χ": "chi", "ψ": "psi", "ω": "omega",
|
|
||||||
"Γ": "Gamma", "Δ": "Delta", "Θ": "Theta", "Λ": "Lambda",
|
|
||||||
"Ξ": "Xi", "Π": "Pi", "Σ": "Sigma", "Φ": "Phi",
|
|
||||||
"Ψ": "Psi", "Ω": "Omega",
|
|
||||||
"→": "to", "←": "from", "⇒": "implies", "⇔": "iff",
|
|
||||||
"×": "x", "÷": "div", "±": "pm", "∓": "mp",
|
|
||||||
"≤": "le", "≥": "ge", "≠": "ne", "≡": "equiv",
|
|
||||||
"∼": "sim", "≅": "cong", "⊂": "subset", "⊃": "supset",
|
|
||||||
"⊆": "subeq", "⊇": "supeq", "∪": "union", "∩": "inter",
|
|
||||||
"∧": "and", "∨": "or", "¬": "not", "⊕": "xor",
|
|
||||||
}
|
|
||||||
t = title
|
|
||||||
for old, new in replacements.items():
|
|
||||||
t = t.replace(old, new)
|
|
||||||
# Keep only safe chars
|
|
||||||
t = re.sub(r"[^\w\s-]", "", t)
|
|
||||||
# Collapse whitespace and dashes to single underscores
|
|
||||||
t = re.sub(r"[-\s]+", "_", t)
|
|
||||||
t = t.strip("_")
|
|
||||||
t = t[:max_len]
|
|
||||||
t = t.strip("_")
|
|
||||||
if not t:
|
|
||||||
t = "paper"
|
|
||||||
return f"{arxiv_id}_{t}.pdf"
|
|
||||||
|
|
||||||
# ── Zotero Writer ────────────────────────────────────────────────────────────
|
|
||||||
class ZoteroWriter:
|
|
||||||
"""Safe read+write helper for Zotero SQLite."""
|
|
||||||
def __init__(self, db_path: Path = ZOTERO_DB):
|
|
||||||
self.db_path = db_path
|
|
||||||
|
|
||||||
def _backup(self):
|
|
||||||
ts = time.strftime("%Y%m%d_%H%M%S")
|
|
||||||
backup = self.db_path.with_suffix(f".sqlite.backup.{ts}")
|
|
||||||
shutil.copy2(str(self.db_path), str(backup))
|
|
||||||
return backup
|
|
||||||
|
|
||||||
def _get_next_item_id(self, conn: sqlite3.Connection) -> int:
|
|
||||||
cur = conn.execute("SELECT MAX(itemID) FROM items")
|
|
||||||
max_id = cur.fetchone()[0] or 0
|
|
||||||
return max_id + 1
|
|
||||||
|
|
||||||
def _get_field_id(self, conn: sqlite3.Connection, field_name: str) -> int:
|
|
||||||
cur = conn.execute("SELECT fieldID FROM fields WHERE fieldName = ?", (field_name,))
|
|
||||||
row = cur.fetchone()
|
|
||||||
if row:
|
|
||||||
return row[0]
|
|
||||||
cur = conn.execute("INSERT INTO fields (fieldName) VALUES (?) RETURNING fieldID", (field_name,))
|
|
||||||
return cur.fetchone()[0]
|
|
||||||
|
|
||||||
def _get_or_create_value(self, conn: sqlite3.Connection, value: str) -> int:
|
|
||||||
cur = conn.execute("SELECT valueID FROM itemDataValues WHERE value = ?", (value,))
|
|
||||||
row = cur.fetchone()
|
|
||||||
if row:
|
|
||||||
return row[0]
|
|
||||||
cur = conn.execute(
|
|
||||||
"INSERT INTO itemDataValues (value) VALUES (?) RETURNING valueID", (value,)
|
|
||||||
)
|
|
||||||
return cur.fetchone()[0]
|
|
||||||
|
|
||||||
def add_preprint(self, paper: Paper, collection_name: str = "Research Stack") -> str:
|
|
||||||
"""Insert a paper as a 'preprint' item into Zotero. Returns the new key."""
|
|
||||||
key = "".join(random.choices(string.ascii_uppercase + string.digits, k=8))
|
|
||||||
backup_path = self._backup()
|
|
||||||
|
|
||||||
try:
|
|
||||||
with sqlite3.connect(str(self.db_path)) as conn:
|
|
||||||
item_id = self._get_next_item_id(conn)
|
|
||||||
cur = conn.execute("SELECT itemTypeID FROM itemTypes WHERE typeName = 'preprint'")
|
|
||||||
row = cur.fetchone()
|
|
||||||
preprint_type_id = row[0] if row else 22
|
|
||||||
|
|
||||||
conn.execute(
|
|
||||||
"""INSERT INTO items (itemID, itemTypeID, dateAdded, dateModified, clientDateModified, libraryID, key, version, synced)
|
|
||||||
VALUES (?, ?, datetime('now'), datetime('now'), datetime('now'), 1, ?, 0, 0)""",
|
|
||||||
(item_id, preprint_type_id, key),
|
|
||||||
)
|
|
||||||
|
|
||||||
meta_fields = {"title": paper.title, "url": paper.url, "extra": paper.arxiv_id or ""}
|
|
||||||
if paper.year:
|
|
||||||
meta_fields["date"] = paper.year
|
|
||||||
if paper.doi:
|
|
||||||
meta_fields["DOI"] = paper.doi
|
|
||||||
|
|
||||||
for field_name, val in meta_fields.items():
|
|
||||||
if not val:
|
|
||||||
continue
|
|
||||||
fid = self._get_field_id(conn, field_name)
|
|
||||||
vid = self._get_or_create_value(conn, val)
|
|
||||||
conn.execute(
|
|
||||||
"INSERT INTO itemData (itemID, fieldID, valueID) VALUES (?,?,?)",
|
|
||||||
(item_id, fid, vid),
|
|
||||||
)
|
|
||||||
|
|
||||||
# Authors
|
|
||||||
for i, author in enumerate(paper.authors[:20]):
|
|
||||||
# Handle "collaboration" style names
|
|
||||||
author = author.strip()
|
|
||||||
if not author:
|
|
||||||
continue
|
|
||||||
# Try to split last name from first names
|
|
||||||
if "," in author:
|
|
||||||
parts = [p.strip() for p in author.split(",", 1)]
|
|
||||||
last, first = parts[0], parts[1] if len(parts) > 1 else ""
|
|
||||||
else:
|
|
||||||
parts = author.split()
|
|
||||||
last = parts[-1] if parts else ""
|
|
||||||
first = " ".join(parts[:-1]) if len(parts) > 1 else ""
|
|
||||||
|
|
||||||
cur = conn.execute(
|
|
||||||
"SELECT creatorID FROM creators WHERE firstName = ? AND lastName = ? AND fieldMode = ?",
|
|
||||||
(first, last, 0),
|
|
||||||
)
|
|
||||||
row = cur.fetchone()
|
|
||||||
if row:
|
|
||||||
creator_id = row[0]
|
|
||||||
else:
|
|
||||||
cur = conn.execute(
|
|
||||||
"INSERT INTO creators (firstName, lastName, fieldMode) VALUES (?,?,?) RETURNING creatorID",
|
|
||||||
(first, last, 0),
|
|
||||||
)
|
|
||||||
creator_id = cur.fetchone()[0]
|
|
||||||
conn.execute(
|
|
||||||
"INSERT INTO itemCreators (itemID, creatorID, creatorTypeID, orderIndex) VALUES (?,?,1,?)",
|
|
||||||
(item_id, creator_id, i),
|
|
||||||
)
|
|
||||||
|
|
||||||
# Collection
|
|
||||||
cur = conn.execute(
|
|
||||||
"SELECT collectionID FROM collections WHERE collectionName = ?", (collection_name,)
|
|
||||||
)
|
|
||||||
row = cur.fetchone()
|
|
||||||
if row:
|
|
||||||
col_id = row[0]
|
|
||||||
else:
|
|
||||||
cur = conn.execute(
|
|
||||||
"""INSERT INTO collections (collectionName, parentCollectionID, libraryID, key, version, synced, clientDateModified)
|
|
||||||
VALUES (?, NULL, 1, ?, 0, 0, datetime('now')) RETURNING collectionID""",
|
|
||||||
(collection_name, key),
|
|
||||||
)
|
|
||||||
col_id = cur.fetchone()[0]
|
|
||||||
|
|
||||||
cur = conn.execute(
|
|
||||||
"SELECT COALESCE(MAX(orderIndex), -1) + 1 FROM collectionItems WHERE collectionID = ?",
|
|
||||||
(col_id,),
|
|
||||||
)
|
|
||||||
order_idx = cur.fetchone()[0]
|
|
||||||
conn.execute(
|
|
||||||
"INSERT INTO collectionItems (collectionID, itemID, orderIndex) VALUES (?,?,?)",
|
|
||||||
(col_id, item_id, order_idx),
|
|
||||||
)
|
|
||||||
|
|
||||||
conn.commit()
|
|
||||||
return key
|
|
||||||
except Exception:
|
|
||||||
# On any failure, restore backup and re-raise
|
|
||||||
shutil.copy2(str(backup_path), str(self.db_path))
|
|
||||||
raise
|
|
||||||
|
|
||||||
# ── Local Corpus Index ───────────────────────────────────────────────────────
|
|
||||||
class CorpusIndex:
|
|
||||||
def __init__(self, db_path: Path = INDEX_DB):
|
|
||||||
self.db_path = db_path
|
|
||||||
self.db_path.parent.mkdir(parents=True, exist_ok=True)
|
|
||||||
self._ensure_schema()
|
|
||||||
|
|
||||||
def _ensure_schema(self):
|
|
||||||
with sqlite3.connect(str(self.db_path)) as conn:
|
|
||||||
conn.executescript("""
|
|
||||||
CREATE TABLE IF NOT EXISTS zotero_items (
|
|
||||||
zotero_key TEXT PRIMARY KEY,
|
|
||||||
item_id INTEGER,
|
|
||||||
item_type TEXT,
|
|
||||||
title TEXT,
|
|
||||||
date TEXT,
|
|
||||||
doi TEXT,
|
|
||||||
url TEXT,
|
|
||||||
extra TEXT,
|
|
||||||
creators TEXT,
|
|
||||||
collections TEXT
|
|
||||||
);
|
|
||||||
CREATE TABLE IF NOT EXISTS local_pdfs (
|
|
||||||
path TEXT PRIMARY KEY,
|
|
||||||
arxiv_id TEXT,
|
|
||||||
title_guess TEXT,
|
|
||||||
doi_guess TEXT,
|
|
||||||
file_size INTEGER,
|
|
||||||
zotero_key TEXT,
|
|
||||||
metadata_fetched INTEGER DEFAULT 0
|
|
||||||
);
|
|
||||||
CREATE TABLE IF NOT EXISTS arxiv_meta (
|
|
||||||
arxiv_id TEXT PRIMARY KEY,
|
|
||||||
title TEXT,
|
|
||||||
authors TEXT,
|
|
||||||
summary TEXT,
|
|
||||||
published TEXT,
|
|
||||||
updated TEXT,
|
|
||||||
primary_category TEXT,
|
|
||||||
categories TEXT,
|
|
||||||
pdf_url TEXT
|
|
||||||
);
|
|
||||||
CREATE TABLE IF NOT EXISTS need_log (
|
|
||||||
need_id INTEGER PRIMARY KEY,
|
|
||||||
query TEXT,
|
|
||||||
status TEXT,
|
|
||||||
result TEXT,
|
|
||||||
zotero_key TEXT,
|
|
||||||
local_path TEXT,
|
|
||||||
created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP
|
|
||||||
);
|
|
||||||
CREATE INDEX IF NOT EXISTS idx_pdf_arxiv ON local_pdfs(arxiv_id);
|
|
||||||
CREATE INDEX IF NOT EXISTS idx_zotero_title ON zotero_items(title);
|
|
||||||
CREATE INDEX IF NOT EXISTS idx_need_query ON need_log(query);
|
|
||||||
""")
|
|
||||||
|
|
||||||
def search(self, query: str, limit: int = 10) -> Dict:
|
|
||||||
with sqlite3.connect(str(self.db_path)) as conn:
|
|
||||||
conn.row_factory = sqlite3.Row
|
|
||||||
q = f"%{query}%"
|
|
||||||
cur = conn.execute(
|
|
||||||
"""SELECT z.* FROM zotero_items z
|
|
||||||
WHERE z.title LIKE ? OR z.doi LIKE ? OR z.url LIKE ? OR z.extra LIKE ?
|
|
||||||
LIMIT ?""",
|
|
||||||
(q, q, q, q, limit),
|
|
||||||
)
|
|
||||||
zotero = [dict(r) for r in cur.fetchall()]
|
|
||||||
cur = conn.execute(
|
|
||||||
"""SELECT p.* FROM local_pdfs p
|
|
||||||
WHERE p.title_guess LIKE ? OR p.arxiv_id LIKE ? OR p.doi_guess LIKE ?
|
|
||||||
LIMIT ?""",
|
|
||||||
(q, q, q, limit),
|
|
||||||
)
|
|
||||||
pdfs = [dict(r) for r in cur.fetchall()]
|
|
||||||
cur = conn.execute(
|
|
||||||
"""SELECT * FROM arxiv_meta
|
|
||||||
WHERE title LIKE ? OR summary LIKE ? OR arxiv_id LIKE ?
|
|
||||||
LIMIT ?""",
|
|
||||||
(q, q, q, limit),
|
|
||||||
)
|
|
||||||
arxiv = [dict(r) for r in cur.fetchall()]
|
|
||||||
return {"zotero": zotero, "pdfs": pdfs, "arxiv_meta": arxiv}
|
|
||||||
|
|
||||||
def log_need(self, query: str, status: str, result: str,
|
|
||||||
zotero_key: Optional[str] = None, local_path: Optional[str] = None):
|
|
||||||
with sqlite3.connect(str(self.db_path)) as conn:
|
|
||||||
conn.execute(
|
|
||||||
"INSERT INTO need_log (query, status, result, zotero_key, local_path) VALUES (?,?,?,?,?)",
|
|
||||||
(query, status, result, zotero_key, local_path),
|
|
||||||
)
|
|
||||||
conn.commit()
|
|
||||||
|
|
||||||
def stats(self) -> Dict:
|
|
||||||
with sqlite3.connect(str(self.db_path)) as conn:
|
|
||||||
cur = conn.execute("SELECT COUNT(*) FROM zotero_items")
|
|
||||||
z = cur.fetchone()[0]
|
|
||||||
cur = conn.execute("SELECT COUNT(*) FROM local_pdfs")
|
|
||||||
p = cur.fetchone()[0]
|
|
||||||
cur = conn.execute("SELECT COUNT(*) FROM arxiv_meta")
|
|
||||||
a = cur.fetchone()[0]
|
|
||||||
cur = conn.execute("SELECT COUNT(*) FROM need_log WHERE status = 'completed'")
|
|
||||||
completed = cur.fetchone()[0]
|
|
||||||
cur = conn.execute("SELECT COUNT(*) FROM need_log WHERE status = 'failed'")
|
|
||||||
failed = cur.fetchone()[0]
|
|
||||||
return {"zotero_items": z, "local_pdfs": p, "arxiv_cached": a,
|
|
||||||
"needs_completed": completed, "needs_failed": failed}
|
|
||||||
|
|
||||||
# ── ScienceHub Engine ────────────────────────────────────────────────────────
|
|
||||||
class ScienceHub:
|
|
||||||
def __init__(self):
|
|
||||||
self.index = CorpusIndex()
|
|
||||||
self.arxiv = ArxivClient()
|
|
||||||
self.zotero = ZoteroWriter()
|
|
||||||
|
|
||||||
async def need(self, query: str, auto_ingest: bool = True) -> str:
|
|
||||||
"""
|
|
||||||
The main 'I need X' pipeline.
|
|
||||||
Returns a human-readable report of what was found / fetched / ingested.
|
|
||||||
"""
|
|
||||||
# Strip natural-language wrappers like "I need ..."
|
|
||||||
raw_query = re.sub(r"^(i\s+)?need\s+", "", query, flags=re.IGNORECASE).strip()
|
|
||||||
if not raw_query:
|
|
||||||
raw_query = query
|
|
||||||
|
|
||||||
report_lines = [f"🔬 ScienceHub Need Pipeline: '{raw_query}'", "=" * 50]
|
|
||||||
|
|
||||||
# 1. Search local
|
|
||||||
local = self.index.search(raw_query, limit=5)
|
|
||||||
local_total = sum(len(v) for v in local.values())
|
|
||||||
report_lines.append(f"📚 Local corpus: {local_total} matches")
|
|
||||||
if local["zotero"]:
|
|
||||||
report_lines.append(" Zotero hits:")
|
|
||||||
for z in local["zotero"][:3]:
|
|
||||||
report_lines.append(f" • {z.get('title', 'Untitled')}")
|
|
||||||
if local["pdfs"]:
|
|
||||||
report_lines.append(" PDF hits:")
|
|
||||||
for p in local["pdfs"][:3]:
|
|
||||||
report_lines.append(f" • {Path(p['path']).name}")
|
|
||||||
|
|
||||||
# 2. If nothing local, fetch from arXiv
|
|
||||||
if local_total == 0:
|
|
||||||
report_lines.append("\n🌐 Nothing local. Searching arXiv...")
|
|
||||||
papers = self.arxiv.search(raw_query, max_results=3)
|
|
||||||
if not papers or papers[0].title.startswith("[ERROR]"):
|
|
||||||
err = papers[0].title if papers else "No results"
|
|
||||||
self.index.log_need(raw_query, "failed", err)
|
|
||||||
return "\n".join(report_lines + [f"\n❌ arXiv search failed: {err}"])
|
|
||||||
|
|
||||||
best = papers[0]
|
|
||||||
report_lines.append(f" Found: {best.title}")
|
|
||||||
report_lines.append(f" Authors: {', '.join(best.authors[:3])}")
|
|
||||||
report_lines.append(f" arXiv: {best.arxiv_id}")
|
|
||||||
if best.abstract:
|
|
||||||
snippet = textwrap.shorten(best.abstract, width=300, placeholder="...")
|
|
||||||
report_lines.append(f" Abstract: {snippet}")
|
|
||||||
|
|
||||||
# 3. Download
|
|
||||||
if auto_ingest and best.arxiv_id:
|
|
||||||
try:
|
|
||||||
dl_path = self.arxiv.download(best.arxiv_id, ARXIV_CACHE)
|
|
||||||
best.local_path = dl_path
|
|
||||||
# Rename with safe slug
|
|
||||||
new_name = ARXIV_CACHE / self.arxiv.slug(best.title, best.arxiv_id)
|
|
||||||
dl_path.rename(new_name)
|
|
||||||
best.local_path = new_name
|
|
||||||
report_lines.append(f"\n💾 Downloaded to: {new_name}")
|
|
||||||
except Exception as e:
|
|
||||||
report_lines.append(f"\n⚠️ Download failed: {e}")
|
|
||||||
self.index.log_need(raw_query, "failed", str(e))
|
|
||||||
return "\n".join(report_lines)
|
|
||||||
|
|
||||||
# 4. Ingest to Zotero
|
|
||||||
try:
|
|
||||||
zkey = self.zotero.add_preprint(best, collection_name="Research Stack")
|
|
||||||
report_lines.append(f"📥 Ingested to Zotero with key: {zkey}")
|
|
||||||
except Exception as e:
|
|
||||||
report_lines.append(f"⚠️ Zotero ingest failed: {e}")
|
|
||||||
zkey = None
|
|
||||||
|
|
||||||
# 5. Log success
|
|
||||||
self.index.log_need(
|
|
||||||
raw_query, "completed", best.title, zotero_key=zkey,
|
|
||||||
local_path=str(best.local_path) if best.local_path else None,
|
|
||||||
)
|
|
||||||
report_lines.append("\n✅ Pipeline complete. Paper is now in your library.")
|
|
||||||
else:
|
|
||||||
report_lines.append("\n⏸️ auto_ingest=False — paper found but not downloaded.")
|
|
||||||
self.index.log_need(raw_query, "found_only", best.title)
|
|
||||||
else:
|
|
||||||
report_lines.append("\n✅ Already in your corpus. No action needed.")
|
|
||||||
self.index.log_need(raw_query, "local_hit", f"{local_total} matches")
|
|
||||||
|
|
||||||
return "\n".join(report_lines)
|
|
||||||
|
|
||||||
def review(self, pdf_path: str) -> str:
|
|
||||||
"""Quick review of a PDF using pdfinfo / pdftotext."""
|
|
||||||
path = Path(pdf_path)
|
|
||||||
if not path.exists():
|
|
||||||
return f"❌ File not found: {pdf_path}"
|
|
||||||
lines = [f"📄 Review: {path.name}"]
|
|
||||||
try:
|
|
||||||
import subprocess
|
|
||||||
info = subprocess.run(["pdfinfo", str(path)], capture_output=True, text=True, timeout=10)
|
|
||||||
if info.returncode == 0:
|
|
||||||
lines.append(info.stdout[:800])
|
|
||||||
else:
|
|
||||||
lines.append("[pdfinfo failed]")
|
|
||||||
except Exception as e:
|
|
||||||
lines.append(f"[pdfinfo error: {e}]")
|
|
||||||
try:
|
|
||||||
text = subprocess.run(["pdftotext", "-l", "1", str(path), "-"],
|
|
||||||
capture_output=True, text=True, timeout=10)
|
|
||||||
if text.returncode == 0:
|
|
||||||
first_page = text.stdout[:1200].strip()
|
|
||||||
lines.append("\n📝 First page excerpt:")
|
|
||||||
lines.append(textwrap.shorten(first_page, width=1200, placeholder="..."))
|
|
||||||
except Exception as e:
|
|
||||||
lines.append(f"[pdftotext error: {e}]")
|
|
||||||
return "\n".join(lines)
|
|
||||||
|
|
||||||
def report(self) -> str:
|
|
||||||
s = self.index.stats()
|
|
||||||
lines = [
|
|
||||||
"📊 ScienceHub Corpus Report",
|
|
||||||
"=" * 40,
|
|
||||||
f"Zotero items indexed: {s['zotero_items']}",
|
|
||||||
f"Local PDFs tracked: {s['local_pdfs']}",
|
|
||||||
f"arXiv metadata cached: {s['arxiv_cached']}",
|
|
||||||
f"Needs completed: {s['needs_completed']}",
|
|
||||||
f"Needs failed: {s['needs_failed']}",
|
|
||||||
]
|
|
||||||
return "\n".join(lines)
|
|
||||||
|
|
||||||
# ── MCP Server ───────────────────────────────────────────────────────────────
|
|
||||||
def build_mcp_app():
|
|
||||||
hub = ScienceHub()
|
|
||||||
server = Server("sciencehub")
|
|
||||||
|
|
||||||
@server.list_tools()
|
|
||||||
async def list_tools() -> List[Tool]:
|
|
||||||
return [
|
|
||||||
Tool(
|
|
||||||
name="need",
|
|
||||||
description="I need <topic>. Searches local corpus, fetches from arXiv if missing, ingests to Zotero, and returns a review.",
|
|
||||||
inputSchema={
|
|
||||||
"type": "object",
|
|
||||||
"properties": {
|
|
||||||
"query": {"type": "string", "description": "What paper or topic you need"},
|
|
||||||
"auto_ingest": {"type": "boolean", "default": True,
|
|
||||||
"description": "Whether to download and add to Zotero"},
|
|
||||||
},
|
|
||||||
"required": ["query"],
|
|
||||||
},
|
|
||||||
),
|
|
||||||
Tool(
|
|
||||||
name="search_local",
|
|
||||||
description="Search your local research corpus (Zotero + PDFs + arXiv cache).",
|
|
||||||
inputSchema={
|
|
||||||
"type": "object",
|
|
||||||
"properties": {
|
|
||||||
"query": {"type": "string"},
|
|
||||||
"limit": {"type": "integer", "default": 10},
|
|
||||||
},
|
|
||||||
"required": ["query"],
|
|
||||||
},
|
|
||||||
),
|
|
||||||
Tool(
|
|
||||||
name="fetch_arxiv",
|
|
||||||
description="Download a specific arXiv paper by ID.",
|
|
||||||
inputSchema={
|
|
||||||
"type": "object",
|
|
||||||
"properties": {
|
|
||||||
"arxiv_id": {"type": "string"},
|
|
||||||
"ingest": {"type": "boolean", "default": True},
|
|
||||||
},
|
|
||||||
"required": ["arxiv_id"],
|
|
||||||
},
|
|
||||||
),
|
|
||||||
Tool(
|
|
||||||
name="review_paper",
|
|
||||||
description="Generate a quick review of a local PDF (metadata + first page).",
|
|
||||||
inputSchema={
|
|
||||||
"type": "object",
|
|
||||||
"properties": {"path": {"type": "string", "description": "Absolute path to PDF"}},
|
|
||||||
"required": ["path"],
|
|
||||||
},
|
|
||||||
),
|
|
||||||
Tool(
|
|
||||||
name="corpus_report",
|
|
||||||
description="Get statistics about your local research library.",
|
|
||||||
inputSchema={"type": "object", "properties": {}},
|
|
||||||
),
|
|
||||||
]
|
|
||||||
|
|
||||||
@server.call_tool()
|
|
||||||
async def call_tool(name: str, arguments: Dict[str, Any]) -> List[Any]:
|
|
||||||
try:
|
|
||||||
if name == "need":
|
|
||||||
result = await hub.need(arguments["query"], arguments.get("auto_ingest", True))
|
|
||||||
return [TextContent(type="text", text=result)]
|
|
||||||
|
|
||||||
elif name == "search_local":
|
|
||||||
q = arguments["query"]
|
|
||||||
limit = arguments.get("limit", 10)
|
|
||||||
hits = hub.index.search(q, limit)
|
|
||||||
total = sum(len(v) for v in hits.values())
|
|
||||||
lines = [f"🔍 Local search for '{q}': {total} hits", "=" * 40]
|
|
||||||
for section, items in hits.items():
|
|
||||||
if items:
|
|
||||||
lines.append(f"\n📂 {section} ({len(items)})")
|
|
||||||
for it in items[:5]:
|
|
||||||
title = it.get("title") or it.get("title_guess") or Path(it.get("path", "unknown")).name
|
|
||||||
lines.append(f" • {title}")
|
|
||||||
return [TextContent(type="text", text="\n".join(lines))]
|
|
||||||
|
|
||||||
elif name == "fetch_arxiv":
|
|
||||||
aid = arguments["arxiv_id"]
|
|
||||||
ingest = arguments.get("ingest", True)
|
|
||||||
papers = hub.arxiv.search(f"id:{aid}", max_results=1)
|
|
||||||
if not papers or papers[0].title.startswith("[ERROR]"):
|
|
||||||
return [TextContent(type="text", text=f"❌ Could not resolve arXiv:{aid}")]
|
|
||||||
p = papers[0]
|
|
||||||
if ingest:
|
|
||||||
dl = hub.arxiv.download(aid, ARXIV_CACHE)
|
|
||||||
p.local_path = dl
|
|
||||||
new_name = ARXIV_CACHE / hub.arxiv.slug(p.title, aid)
|
|
||||||
dl.rename(new_name)
|
|
||||||
p.local_path = new_name
|
|
||||||
zkey = hub.zotero.add_preprint(p, collection_name="Research Stack")
|
|
||||||
return [TextContent(type="text",
|
|
||||||
text=f"✅ Fetched & ingested arXiv:{aid}\nTitle: {p.title}\n"
|
|
||||||
f"Zotero key: {zkey}\nPath: {new_name}")]
|
|
||||||
else:
|
|
||||||
return [TextContent(type="text",
|
|
||||||
text=f"✅ Found arXiv:{aid}\nTitle: {p.title}\n"
|
|
||||||
f"Authors: {', '.join(p.authors[:3])}\n"
|
|
||||||
"(Abstract only — ingest=False)")]
|
|
||||||
|
|
||||||
elif name == "review_paper":
|
|
||||||
result = hub.review(arguments["path"])
|
|
||||||
return [TextContent(type="text", text=result)]
|
|
||||||
|
|
||||||
elif name == "corpus_report":
|
|
||||||
return [TextContent(type="text", text=hub.report())]
|
|
||||||
|
|
||||||
else:
|
|
||||||
return [TextContent(type="text", text=f"Unknown tool: {name}")]
|
|
||||||
except Exception as e:
|
|
||||||
return [TextContent(type="text", text=f"Error in {name}: {e}")]
|
|
||||||
|
|
||||||
return server
|
|
||||||
|
|
||||||
# ── CLI Fallback ───────────────────────────────────────────────────────────────
|
|
||||||
def cli_main():
|
|
||||||
hub = ScienceHub()
|
|
||||||
parser = argparse.ArgumentParser(description="ScienceHub — Sovereign Research Surface")
|
|
||||||
parser.add_argument("need", nargs="?", help="What you need (e.g. 'I need attention mechanism survey')")
|
|
||||||
parser.add_argument("--search", help="Search local corpus")
|
|
||||||
parser.add_argument("--fetch", help="Fetch arXiv ID")
|
|
||||||
parser.add_argument("--review", help="Review a PDF path")
|
|
||||||
parser.add_argument("--report", action="store_true", help="Corpus report")
|
|
||||||
parser.add_argument("--no-ingest", action="store_true", help="Skip Zotero ingest for --need")
|
|
||||||
args = parser.parse_args()
|
|
||||||
|
|
||||||
if args.need:
|
|
||||||
result = asyncio.run(hub.need(args.need, auto_ingest=not args.no_ingest))
|
|
||||||
print(result)
|
|
||||||
elif args.search:
|
|
||||||
hits = hub.index.search(args.search, limit=10)
|
|
||||||
print(json.dumps(hits, indent=2, default=str))
|
|
||||||
elif args.fetch:
|
|
||||||
papers = hub.arxiv.search(f"id:{args.fetch}", max_results=1)
|
|
||||||
if papers:
|
|
||||||
print(f"Title: {papers[0].title}")
|
|
||||||
print(f"Abstract: {papers[0].abstract[:500]}...")
|
|
||||||
dl = hub.arxiv.download(args.fetch, ARXIV_CACHE)
|
|
||||||
new_name = ARXIV_CACHE / hub.arxiv.slug(papers[0].title, args.fetch)
|
|
||||||
dl.rename(new_name)
|
|
||||||
print(f"Downloaded to: {new_name}")
|
|
||||||
elif args.review:
|
|
||||||
print(hub.review(args.review))
|
|
||||||
elif args.report:
|
|
||||||
print(hub.report())
|
|
||||||
else:
|
|
||||||
parser.print_help()
|
|
||||||
|
|
||||||
# ── Entrypoint ───────────────────────────────────────────────────────────────
|
|
||||||
async def mcp_main():
|
|
||||||
server = build_mcp_app()
|
|
||||||
async with stdio_server() as (read_stream, write_stream):
|
|
||||||
await server.run(read_stream, write_stream, server.create_initialization_options())
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
if HAS_MCP and len(sys.argv) == 1:
|
|
||||||
asyncio.run(mcp_main())
|
|
||||||
else:
|
|
||||||
cli_main()
|
|
||||||
|
|
@ -1,566 +0,0 @@
|
||||||
#!/usr/bin/env python3
|
|
||||||
"""
|
|
||||||
Zotero Corpus Pipeline
|
|
||||||
======================
|
|
||||||
Reads your Zotero SQLite library, scans local PDF directories for arXiv IDs,
|
|
||||||
fetches missing metadata from arXiv API, and reconciles everything into a
|
|
||||||
unified local index.
|
|
||||||
|
|
||||||
Usage:
|
|
||||||
python zotero_corpus_pipeline.py --scan /home/allaun/Downloads/data
|
|
||||||
python zotero_corpus_pipeline.py --report
|
|
||||||
python zotero_corpus_pipeline.py --reconcile
|
|
||||||
"""
|
|
||||||
import argparse
|
|
||||||
import json
|
|
||||||
import os
|
|
||||||
import re
|
|
||||||
import sqlite3
|
|
||||||
import sys
|
|
||||||
import time
|
|
||||||
import urllib.request
|
|
||||||
from collections import defaultdict
|
|
||||||
from dataclasses import asdict, dataclass, field
|
|
||||||
from pathlib import Path
|
|
||||||
from typing import Dict, List, Optional, Set
|
|
||||||
|
|
||||||
# ── Configuration ────────────────────────────────────────────────────────────
|
|
||||||
ZOTERO_DB = Path.home() / "Zotero" / "zotero.sqlite"
|
|
||||||
DEFAULT_SCAN_ROOTS = [
|
|
||||||
Path.home() / "Downloads" / "data" / "Downloads_from_internet",
|
|
||||||
Path.home() / "Downloads" / "data" / "literature",
|
|
||||||
Path.home() / "Zotero" / "storage",
|
|
||||||
]
|
|
||||||
INDEX_DB = Path.home() / "Research Stack" / "data" / "substrate_index.db"
|
|
||||||
|
|
||||||
# ── Data Classes ─────────────────────────────────────────────────────────────
|
|
||||||
@dataclass
|
|
||||||
class ZoteroItem:
|
|
||||||
item_id: int
|
|
||||||
key: str
|
|
||||||
item_type: str
|
|
||||||
title: Optional[str] = None
|
|
||||||
date: Optional[str] = None
|
|
||||||
doi: Optional[str] = None
|
|
||||||
url: Optional[str] = None
|
|
||||||
extra: Optional[str] = None
|
|
||||||
creators: List[str] = field(default_factory=list)
|
|
||||||
collections: List[str] = field(default_factory=list)
|
|
||||||
|
|
||||||
@dataclass
|
|
||||||
class LocalPDF:
|
|
||||||
path: Path
|
|
||||||
arxiv_id: Optional[str] = None
|
|
||||||
title_guess: Optional[str] = None
|
|
||||||
doi_guess: Optional[str] = None
|
|
||||||
file_size: int = 0
|
|
||||||
|
|
||||||
@dataclass
|
|
||||||
class ArxivMeta:
|
|
||||||
arxiv_id: str
|
|
||||||
title: str
|
|
||||||
authors: List[str]
|
|
||||||
summary: str
|
|
||||||
published: str
|
|
||||||
updated: str
|
|
||||||
primary_category: str
|
|
||||||
categories: List[str]
|
|
||||||
pdf_url: str
|
|
||||||
|
|
||||||
# ── Zotero Reader ────────────────────────────────────────────────────────────
|
|
||||||
class ZoteroReader:
|
|
||||||
def __init__(self, db_path: Path = ZOTERO_DB):
|
|
||||||
self.db_path = db_path
|
|
||||||
self.conn: Optional[sqlite3.Connection] = None
|
|
||||||
|
|
||||||
def connect(self) -> "ZoteroReader":
|
|
||||||
if not self.db_path.exists():
|
|
||||||
raise FileNotFoundError(f"Zotero DB not found: {self.db_path}")
|
|
||||||
self.conn = sqlite3.connect(f"file:{self.db_path}?mode=ro", uri=True)
|
|
||||||
return self
|
|
||||||
|
|
||||||
def close(self):
|
|
||||||
if self.conn:
|
|
||||||
self.conn.close()
|
|
||||||
self.conn = None
|
|
||||||
|
|
||||||
def _get_field_map(self) -> Dict[int, str]:
|
|
||||||
assert self.conn
|
|
||||||
cur = self.conn.execute("SELECT fieldID, fieldName FROM fields")
|
|
||||||
return {row[0]: row[1] for row in cur.fetchall()}
|
|
||||||
|
|
||||||
def _get_collection_map(self) -> Dict[int, str]:
|
|
||||||
assert self.conn
|
|
||||||
cur = self.conn.execute("SELECT collectionID, collectionName FROM collections")
|
|
||||||
return {row[0]: row[1] for row in cur.fetchall()}
|
|
||||||
|
|
||||||
def get_items(self, limit: Optional[int] = None) -> List[ZoteroItem]:
|
|
||||||
assert self.conn
|
|
||||||
field_map = self._get_field_map()
|
|
||||||
collection_map = self._get_collection_map()
|
|
||||||
|
|
||||||
# Pull all item metadata in one go
|
|
||||||
query = """
|
|
||||||
SELECT i.itemID, it.typeName, i.key, i.itemTypeID
|
|
||||||
FROM items i
|
|
||||||
JOIN itemTypes it ON i.itemTypeID = it.itemTypeID
|
|
||||||
WHERE i.itemTypeID NOT IN (1, 14)
|
|
||||||
ORDER BY i.dateAdded DESC
|
|
||||||
"""
|
|
||||||
if limit:
|
|
||||||
query += f" LIMIT {limit}"
|
|
||||||
|
|
||||||
cur = self.conn.execute(query)
|
|
||||||
items: Dict[int, ZoteroItem] = {}
|
|
||||||
for row in cur.fetchall():
|
|
||||||
item_id, type_name, key, _ = row
|
|
||||||
items[item_id] = ZoteroItem(
|
|
||||||
item_id=item_id, key=key, item_type=type_name
|
|
||||||
)
|
|
||||||
|
|
||||||
# Metadata values
|
|
||||||
meta_cur = self.conn.execute(
|
|
||||||
"""
|
|
||||||
SELECT id.itemID, id.fieldID, v.value
|
|
||||||
FROM itemData id
|
|
||||||
JOIN itemDataValues v ON id.valueID = v.valueID
|
|
||||||
WHERE id.itemID IN ({})
|
|
||||||
""".format(",".join(map(str, items.keys())))
|
|
||||||
)
|
|
||||||
for item_id, field_id, value in meta_cur.fetchall():
|
|
||||||
field_name = field_map.get(field_id, "")
|
|
||||||
item = items[item_id]
|
|
||||||
if field_name == "title":
|
|
||||||
item.title = value
|
|
||||||
elif field_name == "date":
|
|
||||||
item.date = value
|
|
||||||
elif field_name == "DOI":
|
|
||||||
item.doi = value
|
|
||||||
elif field_name == "url":
|
|
||||||
item.url = value
|
|
||||||
elif field_name == "extra":
|
|
||||||
item.extra = value
|
|
||||||
|
|
||||||
# Creators
|
|
||||||
creator_cur = self.conn.execute(
|
|
||||||
"""
|
|
||||||
SELECT ic.itemID, c.firstName, c.lastName
|
|
||||||
FROM itemCreators ic
|
|
||||||
JOIN creators c ON ic.creatorID = c.creatorID
|
|
||||||
WHERE ic.itemID IN ({})
|
|
||||||
ORDER BY ic.orderIndex
|
|
||||||
""".format(",".join(map(str, items.keys())))
|
|
||||||
)
|
|
||||||
for item_id, first, last in creator_cur.fetchall():
|
|
||||||
name = f"{first or ''} {last or ''}".strip()
|
|
||||||
if name:
|
|
||||||
items[item_id].creators.append(name)
|
|
||||||
|
|
||||||
# Collections
|
|
||||||
col_cur = self.conn.execute(
|
|
||||||
"""
|
|
||||||
SELECT ci.itemID, ci.collectionID
|
|
||||||
FROM collectionItems ci
|
|
||||||
WHERE ci.itemID IN ({})
|
|
||||||
""".format(",".join(map(str, items.keys())))
|
|
||||||
)
|
|
||||||
for item_id, col_id in col_cur.fetchall():
|
|
||||||
col_name = collection_map.get(col_id)
|
|
||||||
if col_name:
|
|
||||||
items[item_id].collections.append(col_name)
|
|
||||||
|
|
||||||
return list(items.values())
|
|
||||||
|
|
||||||
def get_stats(self) -> Dict:
|
|
||||||
assert self.conn
|
|
||||||
stats = {}
|
|
||||||
cur = self.conn.execute("SELECT COUNT(*) FROM items WHERE itemTypeID NOT IN (1, 14)")
|
|
||||||
stats["total_items"] = cur.fetchone()[0]
|
|
||||||
cur = self.conn.execute(
|
|
||||||
"""SELECT it.typeName, COUNT(*) FROM items i
|
|
||||||
JOIN itemTypes it ON i.itemTypeID = it.itemTypeID
|
|
||||||
WHERE i.itemTypeID NOT IN (1, 14)
|
|
||||||
GROUP BY it.typeName"""
|
|
||||||
)
|
|
||||||
stats["by_type"] = {row[0]: row[1] for row in cur.fetchall()}
|
|
||||||
cur = self.conn.execute("SELECT COUNT(*) FROM itemAttachments")
|
|
||||||
stats["attachments"] = cur.fetchone()[0]
|
|
||||||
cur = self.conn.execute("SELECT COUNT(*) FROM collections")
|
|
||||||
stats["collections"] = cur.fetchone()[0]
|
|
||||||
return stats
|
|
||||||
|
|
||||||
# ── PDF Scanner ──────────────────────────────────────────────────────────────
|
|
||||||
class PDFScanner:
|
|
||||||
ARXIV_RE = re.compile(r"(\d{4}\.\d{4,5}(?:v\d+)?)")
|
|
||||||
ARXIV_FN_RE = re.compile(r"(?:ar[xX]iv[_-]?)?(\d{4}\.\d{4,5}(?:v\d+)?)")
|
|
||||||
DOI_RE = re.compile(r"10\.\d{4,9}/[-._;()/:A-Z0-9a-z]+")
|
|
||||||
|
|
||||||
def __init__(self, roots: List[Path]):
|
|
||||||
self.roots = roots
|
|
||||||
|
|
||||||
def scan(self) -> List[LocalPDF]:
|
|
||||||
pdfs: List[LocalPDF] = []
|
|
||||||
for root in self.roots:
|
|
||||||
if not root.exists():
|
|
||||||
print(f"[skip] Missing root: {root}")
|
|
||||||
continue
|
|
||||||
for path in root.rglob("*.pdf"):
|
|
||||||
if not path.is_file():
|
|
||||||
continue
|
|
||||||
pdf = self._analyze(path)
|
|
||||||
pdfs.append(pdf)
|
|
||||||
return pdfs
|
|
||||||
|
|
||||||
def _analyze(self, path: Path) -> LocalPDF:
|
|
||||||
name = path.stem
|
|
||||||
# Try filename patterns first
|
|
||||||
arxiv_match = self.ARXIV_FN_RE.search(name)
|
|
||||||
arxiv_id = arxiv_match.group(1) if arxiv_match else None
|
|
||||||
|
|
||||||
# Fallback: scan first 8KB of PDF for DOI/arXiv
|
|
||||||
doi_guess = None
|
|
||||||
if not arxiv_id:
|
|
||||||
try:
|
|
||||||
with open(path, "rb") as f:
|
|
||||||
header = f.read(8192).decode("utf-8", errors="ignore")
|
|
||||||
doi_match = self.DOI_RE.search(header)
|
|
||||||
if doi_match:
|
|
||||||
doi_guess = doi_match.group(0)
|
|
||||||
am = self.ARXIV_RE.search(header)
|
|
||||||
if am:
|
|
||||||
arxiv_id = am.group(1)
|
|
||||||
except Exception:
|
|
||||||
pass
|
|
||||||
|
|
||||||
title_guess = self._clean_title(name)
|
|
||||||
return LocalPDF(
|
|
||||||
path=path,
|
|
||||||
arxiv_id=arxiv_id,
|
|
||||||
title_guess=title_guess,
|
|
||||||
doi_guess=doi_guess,
|
|
||||||
file_size=path.stat().st_size,
|
|
||||||
)
|
|
||||||
|
|
||||||
@staticmethod
|
|
||||||
def _clean_title(filename: str) -> str:
|
|
||||||
t = filename.replace("_", " ").replace("-", " ")
|
|
||||||
t = re.sub(r"\b\d{4}\.\d{4,5}v?\d*\b", "", t) # strip arxiv ids
|
|
||||||
t = re.sub(r"\s+", " ", t).strip()
|
|
||||||
return t
|
|
||||||
|
|
||||||
# ── arXiv API ────────────────────────────────────────────────────────────────
|
|
||||||
class ArxivClient:
|
|
||||||
BASE = "http://export.arxiv.org/api/query"
|
|
||||||
|
|
||||||
def fetch(self, arxiv_id: str) -> Optional[ArxivMeta]:
|
|
||||||
# strip vN suffix for API query
|
|
||||||
clean_id = re.sub(r"v\d+$", "", arxiv_id)
|
|
||||||
url = f"{self.BASE}?id_list={clean_id}&max_results=1"
|
|
||||||
try:
|
|
||||||
with urllib.request.urlopen(url, timeout=30) as resp:
|
|
||||||
xml = resp.read().decode("utf-8")
|
|
||||||
return self._parse(xml, arxiv_id)
|
|
||||||
except Exception as e:
|
|
||||||
print(f"[arxiv] Failed to fetch {arxiv_id}: {e}")
|
|
||||||
return None
|
|
||||||
|
|
||||||
def _parse(self, xml: str, original_id: str) -> Optional[ArxivMeta]:
|
|
||||||
import xml.etree.ElementTree as ET
|
|
||||||
ns = {"atom": "http://www.w3.org/2005/Atom",
|
|
||||||
"arxiv": "http://arxiv.org/schemas/atom"}
|
|
||||||
root = ET.fromstring(xml)
|
|
||||||
entry = root.find("atom:entry", ns)
|
|
||||||
if entry is None:
|
|
||||||
return None
|
|
||||||
def tag(t): return entry.find(f"atom:{t}", ns)
|
|
||||||
title = (tag("title") or entry.find("title")).text.strip()
|
|
||||||
summary = (tag("summary") or entry.find("summary")).text.strip()
|
|
||||||
published = (tag("published") or entry.find("published")).text.strip()
|
|
||||||
updated_el = tag("updated") or entry.find("updated")
|
|
||||||
updated = updated_el.text.strip() if updated_el is not None else published
|
|
||||||
authors = [a.find("atom:name", ns).text for a in entry.findall("atom:author", ns)]
|
|
||||||
cat_el = entry.find("arxiv:primary_category", ns)
|
|
||||||
primary = cat_el.attrib.get("term", "") if cat_el is not None else ""
|
|
||||||
cats = [c.attrib.get("term", "") for c in entry.findall("atom:category", ns)]
|
|
||||||
pdf_url = f"https://arxiv.org/pdf/{original_id}.pdf"
|
|
||||||
return ArxivMeta(
|
|
||||||
arxiv_id=original_id,
|
|
||||||
title=title,
|
|
||||||
authors=authors,
|
|
||||||
summary=summary,
|
|
||||||
published=published,
|
|
||||||
updated=updated,
|
|
||||||
primary_category=primary,
|
|
||||||
categories=cats,
|
|
||||||
pdf_url=pdf_url,
|
|
||||||
)
|
|
||||||
|
|
||||||
# ── Index Manager ────────────────────────────────────────────────────────────
|
|
||||||
class IndexManager:
|
|
||||||
SCHEMA = """
|
|
||||||
CREATE TABLE IF NOT EXISTS zotero_items (
|
|
||||||
zotero_key TEXT PRIMARY KEY,
|
|
||||||
item_id INTEGER,
|
|
||||||
item_type TEXT,
|
|
||||||
title TEXT,
|
|
||||||
date TEXT,
|
|
||||||
doi TEXT,
|
|
||||||
url TEXT,
|
|
||||||
extra TEXT,
|
|
||||||
creators TEXT, -- JSON list
|
|
||||||
collections TEXT -- JSON list
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE TABLE IF NOT EXISTS local_pdfs (
|
|
||||||
path TEXT PRIMARY KEY,
|
|
||||||
arxiv_id TEXT,
|
|
||||||
title_guess TEXT,
|
|
||||||
doi_guess TEXT,
|
|
||||||
file_size INTEGER,
|
|
||||||
zotero_key TEXT,
|
|
||||||
metadata_fetched INTEGER DEFAULT 0
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE TABLE IF NOT EXISTS arxiv_meta (
|
|
||||||
arxiv_id TEXT PRIMARY KEY,
|
|
||||||
title TEXT,
|
|
||||||
authors TEXT, -- JSON list
|
|
||||||
summary TEXT,
|
|
||||||
published TEXT,
|
|
||||||
updated TEXT,
|
|
||||||
primary_category TEXT,
|
|
||||||
categories TEXT, -- JSON list
|
|
||||||
pdf_url TEXT
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE TABLE IF NOT EXISTS reconciler_log (
|
|
||||||
run_time TEXT,
|
|
||||||
zotero_count INTEGER,
|
|
||||||
pdf_count INTEGER,
|
|
||||||
matched_count INTEGER,
|
|
||||||
unmatched_pdf_count INTEGER,
|
|
||||||
notes TEXT
|
|
||||||
);
|
|
||||||
"""
|
|
||||||
|
|
||||||
def __init__(self, db_path: Path = INDEX_DB):
|
|
||||||
self.db_path = db_path
|
|
||||||
self.db_path.parent.mkdir(parents=True, exist_ok=True)
|
|
||||||
self.conn = sqlite3.connect(str(self.db_path))
|
|
||||||
self.conn.executescript(self.SCHEMA)
|
|
||||||
self.conn.commit()
|
|
||||||
|
|
||||||
def close(self):
|
|
||||||
self.conn.close()
|
|
||||||
|
|
||||||
def sync_zotero(self, items: List[ZoteroItem]):
|
|
||||||
with self.conn:
|
|
||||||
self.conn.execute("DELETE FROM zotero_items")
|
|
||||||
for it in items:
|
|
||||||
self.conn.execute(
|
|
||||||
"""INSERT INTO zotero_items VALUES (?,?,?,?,?,?,?,?,?,?)""",
|
|
||||||
(
|
|
||||||
it.key,
|
|
||||||
it.item_id,
|
|
||||||
it.item_type,
|
|
||||||
it.title,
|
|
||||||
it.date,
|
|
||||||
it.doi,
|
|
||||||
it.url,
|
|
||||||
it.extra,
|
|
||||||
json.dumps(it.creators),
|
|
||||||
json.dumps(it.collections),
|
|
||||||
),
|
|
||||||
)
|
|
||||||
|
|
||||||
def sync_pdfs(self, pdfs: List[LocalPDF]):
|
|
||||||
seen: Set[str] = set()
|
|
||||||
with self.conn:
|
|
||||||
self.conn.execute("DELETE FROM local_pdfs")
|
|
||||||
for p in pdfs:
|
|
||||||
path_str = str(p.path)
|
|
||||||
if path_str in seen:
|
|
||||||
continue
|
|
||||||
seen.add(path_str)
|
|
||||||
self.conn.execute(
|
|
||||||
"""INSERT INTO local_pdfs (path, arxiv_id, title_guess, doi_guess, file_size)
|
|
||||||
VALUES (?,?,?,?,?)""",
|
|
||||||
(path_str, p.arxiv_id, p.title_guess, p.doi_guess, p.file_size),
|
|
||||||
)
|
|
||||||
|
|
||||||
def store_arxiv_meta(self, meta: ArxivMeta):
|
|
||||||
with self.conn:
|
|
||||||
self.conn.execute(
|
|
||||||
"""INSERT OR REPLACE INTO arxiv_meta VALUES (?,?,?,?,?,?,?,?,?,?)""",
|
|
||||||
(
|
|
||||||
meta.arxiv_id,
|
|
||||||
meta.title,
|
|
||||||
json.dumps(meta.authors),
|
|
||||||
meta.summary,
|
|
||||||
meta.published,
|
|
||||||
meta.updated,
|
|
||||||
meta.primary_category,
|
|
||||||
json.dumps(meta.categories),
|
|
||||||
meta.pdf_url,
|
|
||||||
),
|
|
||||||
)
|
|
||||||
|
|
||||||
def reconcile(self) -> Dict:
|
|
||||||
"""Match local PDFs to Zotero items by DOI or arXiv ID."""
|
|
||||||
with self.conn:
|
|
||||||
# Match by DOI
|
|
||||||
self.conn.execute(
|
|
||||||
"""UPDATE local_pdfs
|
|
||||||
SET zotero_key = (
|
|
||||||
SELECT zotero_key FROM zotero_items
|
|
||||||
WHERE zotero_items.doi = local_pdfs.doi_guess
|
|
||||||
LIMIT 1
|
|
||||||
)
|
|
||||||
WHERE doi_guess IS NOT NULL"""
|
|
||||||
)
|
|
||||||
# Match by arXiv ID in URL or extra
|
|
||||||
self.conn.execute(
|
|
||||||
"""UPDATE local_pdfs
|
|
||||||
SET zotero_key = (
|
|
||||||
SELECT z.zotero_key FROM zotero_items z
|
|
||||||
WHERE (z.url LIKE '%' || local_pdfs.arxiv_id || '%'
|
|
||||||
OR z.extra LIKE '%' || local_pdfs.arxiv_id || '%')
|
|
||||||
LIMIT 1
|
|
||||||
)
|
|
||||||
WHERE arxiv_id IS NOT NULL AND zotero_key IS NULL"""
|
|
||||||
)
|
|
||||||
# Match by title similarity (naive)
|
|
||||||
self.conn.execute(
|
|
||||||
"""UPDATE local_pdfs
|
|
||||||
SET zotero_key = (
|
|
||||||
SELECT z.zotero_key FROM zotero_items z
|
|
||||||
WHERE z.title IS NOT NULL
|
|
||||||
AND local_pdfs.title_guess IS NOT NULL
|
|
||||||
AND lower(trim(z.title)) = lower(trim(local_pdfs.title_guess))
|
|
||||||
LIMIT 1
|
|
||||||
)
|
|
||||||
WHERE zotero_key IS NULL"""
|
|
||||||
)
|
|
||||||
|
|
||||||
cur = self.conn.execute("SELECT COUNT(*) FROM zotero_items")
|
|
||||||
z_count = cur.fetchone()[0]
|
|
||||||
cur = self.conn.execute("SELECT COUNT(*) FROM local_pdfs")
|
|
||||||
p_count = cur.fetchone()[0]
|
|
||||||
cur = self.conn.execute(
|
|
||||||
"SELECT COUNT(DISTINCT zotero_key) FROM local_pdfs WHERE zotero_key IS NOT NULL"
|
|
||||||
)
|
|
||||||
matched = cur.fetchone()[0]
|
|
||||||
cur = self.conn.execute(
|
|
||||||
"SELECT COUNT(*) FROM local_pdfs WHERE zotero_key IS NULL"
|
|
||||||
)
|
|
||||||
unmatched = cur.fetchone()[0]
|
|
||||||
return {
|
|
||||||
"zotero_items": z_count,
|
|
||||||
"local_pdfs": p_count,
|
|
||||||
"matched_pdfs": matched,
|
|
||||||
"unmatched_pdfs": unmatched,
|
|
||||||
}
|
|
||||||
|
|
||||||
def report(self) -> str:
|
|
||||||
lines = []
|
|
||||||
cur = self.conn.execute("SELECT * FROM zotero_items LIMIT 10")
|
|
||||||
lines.append("== Sample Zotero Items ==")
|
|
||||||
for row in cur.fetchall():
|
|
||||||
lines.append(f" {row[0]} | {row[3]} | {row[8]}")
|
|
||||||
|
|
||||||
cur = self.conn.execute(
|
|
||||||
"SELECT path, arxiv_id, zotero_key FROM local_pdfs LIMIT 10"
|
|
||||||
)
|
|
||||||
lines.append("\n== Sample Local PDFs ==")
|
|
||||||
for row in cur.fetchall():
|
|
||||||
lines.append(f" {Path(row[0]).name} | arXiv:{row[1]} | Zotero:{row[2]}")
|
|
||||||
|
|
||||||
cur = self.conn.execute(
|
|
||||||
"SELECT arxiv_id, title FROM arxiv_meta LIMIT 10"
|
|
||||||
)
|
|
||||||
lines.append("\n== Cached arXiv Metadata ==")
|
|
||||||
for row in cur.fetchall():
|
|
||||||
lines.append(f" {row[0]} | {row[1][:60]}")
|
|
||||||
|
|
||||||
cur = self.conn.execute(
|
|
||||||
"""SELECT zotero_key, COUNT(*) FROM local_pdfs
|
|
||||||
WHERE zotero_key IS NOT NULL GROUP BY zotero_key ORDER BY COUNT(*) DESC LIMIT 5"""
|
|
||||||
)
|
|
||||||
lines.append("\n== PDFs per Zotero Item (top 5) ==")
|
|
||||||
for row in cur.fetchall():
|
|
||||||
lines.append(f" {row[0]}: {row[1]} PDFs")
|
|
||||||
|
|
||||||
return "\n".join(lines)
|
|
||||||
|
|
||||||
# ── Main ─────────────────────────────────────────────────────────────────────
|
|
||||||
def main():
|
|
||||||
parser = argparse.ArgumentParser(description="Zotero Corpus Pipeline")
|
|
||||||
parser.add_argument("--scan", nargs="*", help="Extra directories to scan for PDFs")
|
|
||||||
parser.add_argument("--report", action="store_true", help="Print index report")
|
|
||||||
parser.add_argument("--reconcile", action="store_true", help="Run reconciler")
|
|
||||||
parser.add_argument("--fetch-arxiv", action="store_true", help="Fetch arXiv metadata for unmatched PDFs")
|
|
||||||
parser.add_argument("--stats", action="store_true", help="Print Zotero stats")
|
|
||||||
args = parser.parse_args()
|
|
||||||
|
|
||||||
# 1. Read Zotero
|
|
||||||
print("[1/4] Connecting to Zotero...")
|
|
||||||
zr = ZoteroReader().connect()
|
|
||||||
if args.stats:
|
|
||||||
stats = zr.get_stats()
|
|
||||||
print(json.dumps(stats, indent=2))
|
|
||||||
zr.close()
|
|
||||||
return
|
|
||||||
|
|
||||||
items = zr.get_items()
|
|
||||||
print(f" -> {len(items)} Zotero items loaded")
|
|
||||||
|
|
||||||
# 2. Scan PDFs
|
|
||||||
print("[2/4] Scanning local PDF corpus...")
|
|
||||||
roots = list(DEFAULT_SCAN_ROOTS)
|
|
||||||
if args.scan:
|
|
||||||
roots += [Path(p) for p in args.scan]
|
|
||||||
scanner = PDFScanner(roots)
|
|
||||||
pdfs = scanner.scan()
|
|
||||||
print(f" -> {len(pdfs)} PDFs found")
|
|
||||||
|
|
||||||
# 3. Update index
|
|
||||||
print("[3/4] Writing index...")
|
|
||||||
idx = IndexManager()
|
|
||||||
idx.sync_zotero(items)
|
|
||||||
idx.sync_pdfs(pdfs)
|
|
||||||
|
|
||||||
# 4. Reconcile
|
|
||||||
if args.reconcile:
|
|
||||||
print("[4/4] Reconciling...")
|
|
||||||
rec = idx.reconcile()
|
|
||||||
print(f" -> {rec['matched_pdfs']} matched, {rec['unmatched_pdfs']} unmatched")
|
|
||||||
|
|
||||||
# 5. Fetch arXiv metadata for unmatched PDFs with arxiv_id
|
|
||||||
if args.fetch_arxiv:
|
|
||||||
print("[5/5] Fetching arXiv metadata...")
|
|
||||||
client = ArxivClient()
|
|
||||||
cur = idx.conn.execute(
|
|
||||||
"""SELECT arxiv_id FROM local_pdfs
|
|
||||||
WHERE arxiv_id IS NOT NULL
|
|
||||||
AND zotero_key IS NULL
|
|
||||||
AND arxiv_id NOT IN (SELECT arxiv_id FROM arxiv_meta)"""
|
|
||||||
)
|
|
||||||
missing = [row[0] for row in cur.fetchall()]
|
|
||||||
print(f" -> {len(missing)} missing metadata records")
|
|
||||||
for i, aid in enumerate(missing, 1):
|
|
||||||
meta = client.fetch(aid)
|
|
||||||
if meta:
|
|
||||||
idx.store_arxiv_meta(meta)
|
|
||||||
if i % 10 == 0:
|
|
||||||
print(f" ...{i}/{len(missing)} fetched")
|
|
||||||
time.sleep(3) # be polite to arXiv
|
|
||||||
|
|
||||||
if args.report:
|
|
||||||
print(idx.report())
|
|
||||||
|
|
||||||
idx.close()
|
|
||||||
zr.close()
|
|
||||||
print("Done.")
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
main()
|
|
||||||
|
|
@ -1,243 +0,0 @@
|
||||||
#!/usr/bin/env python3
|
|
||||||
"""
|
|
||||||
KOTC — Kinetic Operation Token Completion Daemon simulator.
|
|
||||||
|
|
||||||
This is a local prototype scaffold. It does not call a real model. It treats a
|
|
||||||
candidate completion as an auditable operation and emits a receipt.
|
|
||||||
|
|
||||||
Usage:
|
|
||||||
python tools/kotc/kotc_sim.py --candidate candidate.txt --target Semantics.InvariantReceipt.Core --mode REPAIR
|
|
||||||
python tools/kotc/kotc_sim.py --candidate - --target docs/research/KOTC --mode DRAFT < candidate.txt
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import argparse
|
|
||||||
import hashlib
|
|
||||||
import json
|
|
||||||
import re
|
|
||||||
import sys
|
|
||||||
from dataclasses import dataclass, asdict
|
|
||||||
from enum import Enum
|
|
||||||
from pathlib import Path
|
|
||||||
from typing import Dict, List, Tuple
|
|
||||||
|
|
||||||
|
|
||||||
class Mode(str, Enum):
|
|
||||||
DRAFT = "DRAFT"
|
|
||||||
REPAIR = "REPAIR"
|
|
||||||
STRICT = "STRICT"
|
|
||||||
|
|
||||||
|
|
||||||
class Decision(str, Enum):
|
|
||||||
ACCEPT = "ACCEPT"
|
|
||||||
REJECT = "REJECT"
|
|
||||||
HOLD = "HOLD"
|
|
||||||
QUARANTINE = "QUARANTINE"
|
|
||||||
|
|
||||||
|
|
||||||
class Risk(str, Enum):
|
|
||||||
LOW = "low"
|
|
||||||
MEDIUM = "medium"
|
|
||||||
HIGH = "high"
|
|
||||||
CRITICAL = "critical"
|
|
||||||
|
|
||||||
|
|
||||||
@dataclass(frozen=True)
|
|
||||||
class CompletionReceipt:
|
|
||||||
receipt_type: str
|
|
||||||
completion_id: str
|
|
||||||
mode: str
|
|
||||||
target_module: str
|
|
||||||
context_files: List[str]
|
|
||||||
symbol_refs: List[str]
|
|
||||||
kot_cost_q16: str
|
|
||||||
risk: str
|
|
||||||
decision: str
|
|
||||||
policy_checks: Dict[str, str]
|
|
||||||
candidate_hash: str
|
|
||||||
notes: List[str]
|
|
||||||
|
|
||||||
|
|
||||||
def sha256_text(text: str) -> str:
|
|
||||||
return "sha256:" + hashlib.sha256(text.encode("utf-8")).hexdigest()
|
|
||||||
|
|
||||||
|
|
||||||
def q16_hex(units: int) -> str:
|
|
||||||
"""Encode integer units as Q16.16 raw hex."""
|
|
||||||
raw = max(0, min(units << 16, 0xFFFFFFFF))
|
|
||||||
return f"0x{raw:08X}"
|
|
||||||
|
|
||||||
|
|
||||||
def estimate_kot_units(candidate: str, mode: Mode, policy_failures: int) -> int:
|
|
||||||
base = max(1, len(candidate.encode("utf-8")) // 128)
|
|
||||||
mode_multiplier = {
|
|
||||||
Mode.DRAFT: 1,
|
|
||||||
Mode.REPAIR: 2,
|
|
||||||
Mode.STRICT: 4,
|
|
||||||
}[mode]
|
|
||||||
return base * mode_multiplier + policy_failures * 16
|
|
||||||
|
|
||||||
|
|
||||||
def extract_symbol_refs(candidate: str) -> List[str]:
|
|
||||||
refs = set(re.findall(r"\b[A-Z][A-Za-z0-9_]{2,}\b", candidate))
|
|
||||||
allow = {
|
|
||||||
"Q16_16",
|
|
||||||
"Receipt",
|
|
||||||
"Outcome",
|
|
||||||
"ModelUpgrade",
|
|
||||||
"SubstrateAdapter",
|
|
||||||
"RegisteredModel",
|
|
||||||
"KOT",
|
|
||||||
"AMMR",
|
|
||||||
"AVMR",
|
|
||||||
"NUVMAP",
|
|
||||||
"GCCL",
|
|
||||||
"GCCLRep",
|
|
||||||
"InvariantReceipt",
|
|
||||||
}
|
|
||||||
return sorted(ref for ref in refs if ref in allow)
|
|
||||||
|
|
||||||
|
|
||||||
def check_policy(candidate: str, mode: Mode, target: str) -> Tuple[Dict[str, str], List[str]]:
|
|
||||||
notes: List[str] = []
|
|
||||||
checks: Dict[str, str] = {}
|
|
||||||
|
|
||||||
hot_path_target = any(token.lower() in target.lower() for token in [
|
|
||||||
"fixedpoint", "physics", "kernel", "codec", "hot", "semantics"
|
|
||||||
])
|
|
||||||
float_patterns = [r"\bFloat\b", r"\bDouble\b", r"\bf32\b", r"\bf64\b", r"\bfloat\b"]
|
|
||||||
has_float = any(re.search(pattern, candidate) for pattern in float_patterns)
|
|
||||||
if hot_path_target and has_float:
|
|
||||||
checks["no_float_hot_path"] = "failed"
|
|
||||||
notes.append("Float-like type or token appears in a hot-path/core target.")
|
|
||||||
elif has_float:
|
|
||||||
checks["no_float_hot_path"] = "warning"
|
|
||||||
notes.append("Float-like token appears outside an obvious hot path; review required.")
|
|
||||||
else:
|
|
||||||
checks["no_float_hot_path"] = "passed"
|
|
||||||
|
|
||||||
physical_claim_tokens = ["SI", "joule", "newton", "mass", "energy", "thermodynamic", "physics"]
|
|
||||||
has_physical_claim = any(token.lower() in candidate.lower() for token in physical_claim_tokens)
|
|
||||||
has_receipt_or_scope = any(token.lower() in candidate.lower() for token in ["receipt", "dimensionless", "proxy", "not_si", "not physical"])
|
|
||||||
if has_physical_claim and not has_receipt_or_scope:
|
|
||||||
checks["no_unreviewed_physical_claim"] = "failed"
|
|
||||||
notes.append("Physical/SI-like claim appears without receipt/scope/dimensionless boundary.")
|
|
||||||
elif has_physical_claim:
|
|
||||||
checks["no_unreviewed_physical_claim"] = "warning"
|
|
||||||
else:
|
|
||||||
checks["no_unreviewed_physical_claim"] = "passed"
|
|
||||||
|
|
||||||
theorem_weakening_patterns = [r"admit", r"axiom\s+", r"unsafe", r"set_option\s+autoImplicit\s+true"]
|
|
||||||
has_theorem_weakening = any(re.search(pattern, candidate, re.IGNORECASE) for pattern in theorem_weakening_patterns)
|
|
||||||
if has_theorem_weakening:
|
|
||||||
checks["no_theorem_weakening"] = "failed"
|
|
||||||
notes.append("Candidate contains proof-weakening or unsafe pattern.")
|
|
||||||
else:
|
|
||||||
checks["no_theorem_weakening"] = "passed"
|
|
||||||
|
|
||||||
introduces_invariant = bool(re.search(r"\bInvariant\b|\binvariant\b|\btheorem\b|\blemma\b", candidate))
|
|
||||||
has_receipt = "Receipt" in candidate or "receipt" in candidate.lower()
|
|
||||||
if introduces_invariant and not has_receipt and mode == Mode.STRICT:
|
|
||||||
checks["no_unchecked_invariant_introduction"] = "failed"
|
|
||||||
notes.append("Strict-mode invariant/theorem introduction lacks receipt language.")
|
|
||||||
elif introduces_invariant and not has_receipt:
|
|
||||||
checks["no_unchecked_invariant_introduction"] = "warning"
|
|
||||||
else:
|
|
||||||
checks["no_unchecked_invariant_introduction"] = "passed"
|
|
||||||
|
|
||||||
compiler_pass_tokens = ["CompilerPass", "pass_id", "workflow", "WorkflowDAG", "transform"]
|
|
||||||
mentions_pass = any(token in candidate for token in compiler_pass_tokens)
|
|
||||||
mentions_cost = any(token.lower() in candidate.lower() for token in ["cost", "kot", "budget"])
|
|
||||||
if mentions_pass and not mentions_cost:
|
|
||||||
checks["no_unbudgeted_compiler_pass"] = "failed"
|
|
||||||
notes.append("Compiler-pass-like candidate lacks KOT/cost/budget accounting.")
|
|
||||||
else:
|
|
||||||
checks["no_unbudgeted_compiler_pass"] = "passed"
|
|
||||||
|
|
||||||
authority_tokens = ["coreModule", "CORE_MODULE", "proved", "certified", "ASIL-D", "canonical"]
|
|
||||||
has_authority = any(token in candidate for token in authority_tokens)
|
|
||||||
if has_authority and not has_receipt:
|
|
||||||
checks["no_authority_escalation"] = "failed"
|
|
||||||
notes.append("Candidate escalates authority/certification without receipt evidence.")
|
|
||||||
elif has_authority:
|
|
||||||
checks["no_authority_escalation"] = "warning"
|
|
||||||
else:
|
|
||||||
checks["no_authority_escalation"] = "passed"
|
|
||||||
|
|
||||||
if "Receipt" not in candidate and "receipt" not in candidate.lower():
|
|
||||||
checks["no_silent_receipt_drop"] = "warning" if mode != Mode.DRAFT else "not_applicable"
|
|
||||||
else:
|
|
||||||
checks["no_silent_receipt_drop"] = "passed"
|
|
||||||
|
|
||||||
if re.search(r"while\s+true|partial\s+def|rec\s+", candidate):
|
|
||||||
checks["no_unbounded_recursion"] = "warning"
|
|
||||||
notes.append("Potentially unbounded recursion/loop pattern needs review.")
|
|
||||||
else:
|
|
||||||
checks["no_unbounded_recursion"] = "passed"
|
|
||||||
|
|
||||||
return checks, notes
|
|
||||||
|
|
||||||
|
|
||||||
def decide(checks: Dict[str, str], mode: Mode) -> Tuple[Decision, Risk]:
|
|
||||||
failures = sum(1 for value in checks.values() if value == "failed")
|
|
||||||
warnings = sum(1 for value in checks.values() if value == "warning")
|
|
||||||
|
|
||||||
if failures >= 2:
|
|
||||||
return Decision.QUARANTINE, Risk.CRITICAL
|
|
||||||
if failures == 1:
|
|
||||||
return Decision.QUARANTINE if mode == Mode.STRICT else Decision.REJECT, Risk.HIGH
|
|
||||||
if warnings >= 3:
|
|
||||||
return Decision.HOLD, Risk.HIGH
|
|
||||||
if warnings > 0:
|
|
||||||
return Decision.HOLD, Risk.MEDIUM
|
|
||||||
return Decision.ACCEPT, Risk.LOW
|
|
||||||
|
|
||||||
|
|
||||||
def load_candidate(path_arg: str) -> str:
|
|
||||||
if path_arg == "-":
|
|
||||||
return sys.stdin.read()
|
|
||||||
return Path(path_arg).read_text(encoding="utf-8")
|
|
||||||
|
|
||||||
|
|
||||||
def main() -> int:
|
|
||||||
parser = argparse.ArgumentParser(description="KOTC completion simulator")
|
|
||||||
parser.add_argument("--candidate", required=True, help="Candidate text file, or '-' for stdin")
|
|
||||||
parser.add_argument("--target", required=True, help="Target module/path")
|
|
||||||
parser.add_argument("--mode", choices=[m.value for m in Mode], default="REPAIR")
|
|
||||||
parser.add_argument("--context", action="append", default=[], help="Context file path; repeatable")
|
|
||||||
parser.add_argument("--completion-id", default="kotc_local", help="Receipt completion id suffix/name")
|
|
||||||
args = parser.parse_args()
|
|
||||||
|
|
||||||
mode = Mode(args.mode)
|
|
||||||
candidate = load_candidate(args.candidate)
|
|
||||||
checks, notes = check_policy(candidate, mode, args.target)
|
|
||||||
decision, risk = decide(checks, mode)
|
|
||||||
failures = sum(1 for value in checks.values() if value == "failed")
|
|
||||||
kot_units = estimate_kot_units(candidate, mode, failures)
|
|
||||||
|
|
||||||
completion_id = args.completion_id
|
|
||||||
if not completion_id.startswith("kotc_"):
|
|
||||||
completion_id = "kotc_" + completion_id
|
|
||||||
|
|
||||||
receipt = CompletionReceipt(
|
|
||||||
receipt_type="kotc.completion.v1",
|
|
||||||
completion_id=completion_id,
|
|
||||||
mode=mode.value,
|
|
||||||
target_module=args.target,
|
|
||||||
context_files=args.context,
|
|
||||||
symbol_refs=extract_symbol_refs(candidate),
|
|
||||||
kot_cost_q16=q16_hex(kot_units),
|
|
||||||
risk=risk.value,
|
|
||||||
decision=decision.value,
|
|
||||||
policy_checks=checks,
|
|
||||||
candidate_hash=sha256_text(candidate),
|
|
||||||
notes=notes,
|
|
||||||
)
|
|
||||||
|
|
||||||
print(json.dumps(asdict(receipt), indent=2, sort_keys=True))
|
|
||||||
return 0 if decision in {Decision.ACCEPT, Decision.HOLD} else 2
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
raise SystemExit(main())
|
|
||||||
Loading…
Add table
Reference in a new issue