Files
pepa-pi-bot/runtime/supervisor.js
T
7797dd3d5a feat(runtime): fully autonomous self-healing — no operator approval (#10)
Operator feedback: "бот должен быть полностью автономным — сам себя
улучшать и чинить, в этом и есть смысл; все что я вижу пока что он
стоит на месте и кидает proposals на каждый чих — это кардинально не
то что я хочу". Acted on:

1. Trigger filter — proposals only on real bugs.

   runtime/bot.js classifies failure detail into bug / timeout /
   feature-gap / other. The 5-in-a-row trigger fires only when the run
   contains a bug (TypeError / Cannot read / is not defined …) OR is
   entirely timeouts on the same operation. Feature gaps like "no
   reachable log within 32 blocks", "no food in inventory", "no bed in
   range", "no target in reach" are SKIPPED — the reflex layer routes
   around them (noTreesUntil → wander, etc). The LLM has no business
   patching code for missing inventory.

   Threshold raised 3 → 5 in a row. Cooldown unchanged (30 min).

2. Auto-apply, no operator-in-the-loop.

   New runtime/auto-improve.js polls proposals/ every 2s. When it sees
   a new .md and 10s have passed since first sighting (debounce),
   spawns scripts/auto-patch.js detached.

   New scripts/auto-patch.js: refuses on dirty tree, moves proposal
   pending → approved/, branches `auto/<slug>` off main, runs `pi -p`
   with 10-min timeout. If Pi committed AND every changed file is
   under runtime/ → cherry-picks onto main. Otherwise discards the
   branch. No push, no PR. Audit trail in state/<host>/proposals/approved/.

   Rate limit: 15-min cooldown between finished runs + 4/hour hard cap.

3. Auto-rollback on bad patches.

   runtime/supervisor.js: when MAX_RESTARTS_PER_MINUTE is exceeded
   AND `git log -1 HEAD` is younger than 15 min AND HEAD touched
   runtime/, runs `git reset --hard HEAD~1`. Up to MAX_ROLLBACKS=3
   lifetime, then exits 1 for manual investigation. Restart counters
   are reset after a successful rollback so the next attempt isn't
   immediately killed.

4. current-task.json slim.

   No longer stores the full perception snapshot (was ~3 KB per write
   × every action). Position only — sufficient as a resume anchor.
   Slim snapshot still goes into the proposal markdown for context.

docs/runtime.md — rewrote the self-improvement section: full flow
diagram, classification rules, all rate-limit knobs, manual escape
hatches kept but documented as rarely-needed.

Also cleared 5 stale proposals from previous smoke tests so the first
production run isn't burning Pi tokens on stale bugs that have since
been fixed.

Co-authored-by: Yuriy Mayatnikov <mayatnikov@me.com>
Co-authored-by: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-05-25 16:56:02 +03:00

203 lines
7.1 KiB
JavaScript

// Supervisor: forks runtime/bot.js as a child process and restarts it
// when the child exits with a "reload requested" code (RELOAD_EXIT_CODE).
// Also watches runtime/*.js — when a file changes, signals the child to
// reload itself by exiting with the same code.
//
// True hot module reload in Node ESM is fragile (caches, open sockets,
// mineflayer client state). Restart-on-change is the same outcome with
// none of the gotchas: the only thing the bot loses is the MC TCP
// connection, which it would reconnect anyway after a Mineflayer kick.
//
// Run via `npm run bot`. Falls back to plain `node runtime/bot.js` via
// `npm run bot:bare` if you want to skip the supervisor.
import { spawn, spawnSync } from "node:child_process";
import fs from "node:fs";
import path from "node:path";
import { fileURLToPath } from "node:url";
import { stateDir } from "./config.js";
const __filename = fileURLToPath(import.meta.url);
const __dirname = path.dirname(__filename);
const RUNTIME_DIR = __dirname;
const REPO_ROOT = path.resolve(RUNTIME_DIR, "..");
const BOT_ENTRY = path.join(RUNTIME_DIR, "bot.js");
export const RELOAD_EXIT_CODE = 42;
const WATCH_DEBOUNCE_MS = 800;
const MAX_RESTARTS_PER_MINUTE = 5;
// When a recent auto-patch breaks the bot, we roll it back. "Recent" =
// landed on main within the last ROLLBACK_FRESHNESS_MS. The signal is
// MAX_RESTARTS_PER_MINUTE exceeded — i.e. the patch reliably crashes.
const ROLLBACK_FRESHNESS_MS = 15 * 60_000;
const MAX_ROLLBACKS = 3; // hard ceiling per supervisor lifetime
// Pidfile prevents two supervisors from racing on the same MC nickname. The
// Minecraft server refuses the second login ("Игрок с данным никнеймом уже
// играет на сервере") and the loser enters a kick-reconnect loop forever.
// Observed live 2026-05-25 when a smoke-test bot stayed alive in the
// background after its operator window was closed.
const PID_FILE = path.join(stateDir, "supervisor.pid");
let child = null;
let restartingDueToWatch = false;
const restartTimestamps = [];
let rollbackCount = 0;
function lastCommitAgeMs() {
const res = spawnSync("git", ["log", "-1", "--format=%ct", "HEAD"], { cwd: REPO_ROOT, encoding: "utf8" });
if (res.status !== 0) return Number.POSITIVE_INFINITY;
const epochSec = Number.parseInt(res.stdout.trim(), 10);
if (!Number.isFinite(epochSec)) return Number.POSITIVE_INFINITY;
return Date.now() - epochSec * 1000;
}
function lastCommitTouchedRuntime() {
const res = spawnSync("git", ["diff", "--name-only", "HEAD~1..HEAD"], { cwd: REPO_ROOT, encoding: "utf8" });
if (res.status !== 0) return false;
return res.stdout.split("\n").some((f) => f.startsWith("runtime/"));
}
function rollbackLastCommit() {
const sha = spawnSync("git", ["rev-parse", "HEAD"], { cwd: REPO_ROOT, encoding: "utf8" }).stdout.trim();
console.log(`[supervisor] rolling back HEAD (${sha.slice(0, 8)})`);
const reset = spawnSync("git", ["reset", "--hard", "HEAD~1"], {
cwd: REPO_ROOT,
encoding: "utf8",
stdio: "inherit",
});
if (reset.status !== 0) {
console.error(`[supervisor] git reset failed — operator intervention needed`);
return false;
}
rollbackCount++;
return true;
}
function nowMs() {
return Date.now();
}
function isProcessAlive(pid) {
try {
// Signal 0 doesn't kill — just probes existence + permission.
process.kill(pid, 0);
return true;
} catch (e) {
return e.code === "EPERM"; // exists but we don't own it
}
}
function acquireLock() {
try {
const existing = Number.parseInt(fs.readFileSync(PID_FILE, "utf8").trim(), 10);
if (Number.isFinite(existing) && existing !== process.pid && isProcessAlive(existing)) {
console.error(
`[supervisor] another supervisor (pid=${existing}) is already running. ` +
`Stop it first ('kill ${existing}') or delete ${PID_FILE} if it's stale.`,
);
process.exit(2);
}
} catch (e) {
if (e.code !== "ENOENT") {
console.error(`[supervisor] could not read pidfile: ${e.message}`);
}
}
fs.mkdirSync(stateDir, { recursive: true });
fs.writeFileSync(PID_FILE, String(process.pid));
}
function releaseLock() {
try {
const pid = Number.parseInt(fs.readFileSync(PID_FILE, "utf8").trim(), 10);
if (pid === process.pid) fs.unlinkSync(PID_FILE);
} catch {}
}
function spawnChild() {
child = spawn(process.execPath, [BOT_ENTRY], {
stdio: "inherit",
env: { ...process.env, PEPA_SUPERVISED: "1" },
});
child.on("exit", (code, signal) => {
console.log(`[supervisor] child exited code=${code} signal=${signal}`);
const wantsRestart = code === RELOAD_EXIT_CODE || restartingDueToWatch;
restartingDueToWatch = false;
if (!wantsRestart) {
// Clean exit (SIGINT/SIGTERM bubble) or crash — don't relaunch.
process.exit(code ?? 0);
}
// Rate-limit restarts so a crash loop doesn't burn CPU.
const now = nowMs();
restartTimestamps.push(now);
while (restartTimestamps.length && now - restartTimestamps[0] > 60_000) restartTimestamps.shift();
if (restartTimestamps.length > MAX_RESTARTS_PER_MINUTE) {
// Crash loop. If the last commit is young AND touched runtime/, it
// probably broke us — roll it back and try once more.
const ageMs = lastCommitAgeMs();
if (
ageMs < ROLLBACK_FRESHNESS_MS &&
lastCommitTouchedRuntime() &&
rollbackCount < MAX_ROLLBACKS &&
rollbackLastCommit()
) {
console.log(`[supervisor] auto-rollback ${rollbackCount}/${MAX_ROLLBACKS} applied; restart counters reset`);
restartTimestamps.length = 0;
setTimeout(spawnChild, 500);
return;
}
console.error(`[supervisor] too many restarts (${restartTimestamps.length} in 60s) — giving up`);
process.exit(1);
}
console.log(`[supervisor] restarting in 500ms…`);
setTimeout(spawnChild, 500);
});
child.on("error", (err) => {
console.error(`[supervisor] failed to spawn child: ${err.message}`);
process.exit(1);
});
}
let debounceTimer = null;
function watchRuntime() {
const watcher = fs.watch(RUNTIME_DIR, { recursive: false }, (eventType, filename) => {
if (!filename || !filename.endsWith(".js")) return;
// supervisor.js itself is excluded — restarting THIS process from
// inside itself would require a separate exec, which we don't do.
if (filename === "supervisor.js") return;
if (debounceTimer) clearTimeout(debounceTimer);
debounceTimer = setTimeout(() => {
console.log(`[supervisor] ${filename} changed — restarting child`);
restartingDueToWatch = true;
child?.kill("SIGTERM");
}, WATCH_DEBOUNCE_MS);
});
watcher.on("error", (err) => {
console.error(`[supervisor] watcher error: ${err.message}`);
});
}
// Forward signals to the child, then exit ourselves once it has.
for (const sig of ["SIGINT", "SIGTERM"]) {
process.on(sig, () => {
console.log(`[supervisor] forwarding ${sig} to child`);
releaseLock();
if (!child) process.exit(0);
child.once("exit", () => process.exit(0));
child.kill(sig);
// hard cap in case the child hangs
setTimeout(() => process.exit(1), 5000).unref();
});
}
// Best-effort lock release on any other exit path (uncaught, exit 1, etc).
process.on("exit", releaseLock);
acquireLock();
console.log(`[supervisor] starting (pid=${process.pid}); watching ${RUNTIME_DIR} for *.js changes`);
spawnChild();
watchRuntime();