v0.2.0-rc.2: P0 hardening — Pi headless, test state isolation, advice fixes (#21)
P0 (correctness):
1. PEPA_HEADLESS=1 guard in extensions/mineflayer-bridge.ts. When `pi -p`
spawns a subprocess (banter, coach, planner, reflect, auto-patch), the
bridge no longer attempts a second MC connect — the hybrid runtime
already owns the nickname. runtime/pi-bridge.js sets the env var on
every spawn. Root cause of the "two pepa_bot's racing for the slot"
bug seen in reply-pi stderr.
2. Test state isolation in runtime/config.js. When running under the node
test runner (detected via execArgv/argv) — or when PEPA_STATE_DIR is
set — stateDir redirects to /tmp/pepa-test-state-<pid>/. log.js,
scenario-memory, world-journal, and knowledge.db all follow.
`npm test` no longer pollutes live scenarios.jsonl, world-journal.jsonl,
or daily log files. Verified empirically: post-fix run added 0 test
rows to the live scenarios file. Cleaned ~550 historical test rows
from live state in the same change.
3. defendReflex outcome reporting (runtime/reflex.js). Previously a
creeper-rule override marked the lesson succeeded=false BEFORE the
flee skill returned. Now dispatchDefendFlee accepts {lessonId} and
the onComplete fires reportAdviceOutcome with the actual flee result.
4. Mode-name → skill-id translation in runtime/coach/advice.js. Pi-coach
occasionally returns prefer_skill values that are mode names
("night_shelter", "self_preservation", "hunger"). normalisePreferSkill
maps these to SAFE_OVERRIDES entries before dispatch. Also handles
"tunnel-out", "survive_flee", "survive flee" shapes.
New behavior:
5. Self-reflection loop (runtime/coach/reflect.js). Every 30 min, the
bot asks Pi: "Are you making progress, or stuck in a loop? What
should you do differently?" Pi answers with a verdict
(progress/loop/recovering/idle/emergency), summary, next-action, and
0-N new lessons. The reflection is written to
state/<host>/reflections/<ts>.md and lessons land in the DB with
source="pi-reflect". Rate-limited to 2 calls/hour. Wired through
bot.js with the existing askPi + lastSnapshot accessor.
Tests: 246/246 green (+9 new: 6 advice mode-name + 4 reflect).
Co-authored-by: Yuriy Mayatnikov <mayatnikov@me.com>
Co-authored-by: Claude Opus 4.7 <noreply@anthropic.com>
This commit was merged in pull request #21.
This commit is contained in:
+38
-4
@@ -26,6 +26,36 @@ const SAFE_OVERRIDES = new Set([
|
||||
"village.build-shelter",
|
||||
]);
|
||||
|
||||
// Pi-coach occasionally suggests prefer_skill values that are mode names
|
||||
// (from runtime/modes.js) rather than registered skill ids. We translate
|
||||
// them to the closest equivalent skill before the SAFE_OVERRIDES check.
|
||||
// Unknown values are returned as-is and will fall through to 'avoid'.
|
||||
const MODE_TO_SKILL = Object.freeze({
|
||||
self_preservation: "survive.flee",
|
||||
night_shelter: "survive.sleep",
|
||||
hunger: "survive.eat",
|
||||
shelter: "village.build-shelter",
|
||||
flee: "survive.flee",
|
||||
sleep: "survive.sleep",
|
||||
eat: "survive.eat",
|
||||
tunnel_out: "recovery.tunnel-out",
|
||||
"tunnel-out": "recovery.tunnel-out",
|
||||
explore: "explore.far",
|
||||
wander: "explore.far",
|
||||
});
|
||||
|
||||
function normalisePreferSkill(raw) {
|
||||
if (!raw || typeof raw !== "string") return raw;
|
||||
if (SAFE_OVERRIDES.has(raw)) return raw;
|
||||
const lower = raw.toLowerCase().trim();
|
||||
if (MODE_TO_SKILL[lower]) return MODE_TO_SKILL[lower];
|
||||
// Pi sometimes writes "survive_flee" or "survive flee"; normalise.
|
||||
const dot = lower.replace(/[_\s]+/g, ".");
|
||||
if (SAFE_OVERRIDES.has(dot)) return dot;
|
||||
if (MODE_TO_SKILL[dot]) return MODE_TO_SKILL[dot];
|
||||
return raw;
|
||||
}
|
||||
|
||||
/**
|
||||
* consult({ plannedSkillId, snapshot })
|
||||
* → { action: 'override'|'avoid'|'proceed', overrideSkillId?, lessonId?, lesson? }
|
||||
@@ -48,11 +78,15 @@ export function consult({ plannedSkillId, snapshot } = {}) {
|
||||
|
||||
// avoid_skill matches?
|
||||
if (advice.avoid && advice.avoid === plannedSkillId) {
|
||||
if (advice.prefer && SAFE_OVERRIDES.has(advice.prefer)) {
|
||||
info("coach", `advice: override ${plannedSkillId} → ${advice.prefer} (lesson #${advice.lessonId})`);
|
||||
const normalisedPrefer = normalisePreferSkill(advice.prefer);
|
||||
if (normalisedPrefer && SAFE_OVERRIDES.has(normalisedPrefer)) {
|
||||
if (normalisedPrefer !== advice.prefer) {
|
||||
info("coach", `advice: normalised prefer "${advice.prefer}" → "${normalisedPrefer}"`);
|
||||
}
|
||||
info("coach", `advice: override ${plannedSkillId} → ${normalisedPrefer} (lesson #${advice.lessonId})`);
|
||||
return {
|
||||
action: "override",
|
||||
overrideSkillId: advice.prefer,
|
||||
overrideSkillId: normalisedPrefer,
|
||||
lessonId: advice.lessonId,
|
||||
lesson: advice.lesson,
|
||||
};
|
||||
@@ -76,4 +110,4 @@ export function reportOutcome({ lessonId, succeeded }) {
|
||||
const PROCEED = Object.freeze({ action: "proceed", lessonId: null, lesson: null });
|
||||
|
||||
// Test exports
|
||||
export const __testing = { SAFE_OVERRIDES };
|
||||
export const __testing = { SAFE_OVERRIDES, MODE_TO_SKILL, normalisePreferSkill };
|
||||
|
||||
Reference in New Issue
Block a user