The wedge wasn't only in the manifesto layer; several mechanical bugs kept the bot in a dead random-walk: - storyline / manifesto / curriculum: "local food" now means an edible passive mob within <=32 blocks. A distant chicken or a cod no longer fools the bot into dispatching acquire-food (which then fails on no_path). Long-range food goes through scout-food instead. - scout-food: partial approach to a target now counts as progress (approached_target, e.g. moved:14); a blocked heading is NOT counted as movement; added blind/tunnel fallback so it doesn't die when the pathfinder can't route cleanly. - acquire-food: on no_path it now also tries a blind/tunnel approach to the animal; no_drop routes back into food scouting instead of giving up. - explore.far / relocate / flee: fewer false "done" results (micro-steps no longer counted as success), more genuine escapes from stuck. - scripts/show-story.js: live IPC now actually renders the current storyline step. Verification: scripts/lint-patch.js clean; npm test 404/404 green; bot relaunched in tmux `pepa`. Live logs show real progress — bot switched to survive.scout-food, approached the chicken (approached_target moved:14), then reached survive.acquire-food: hunting chicken. Food isn't fully closed yet but the remaining issue is concrete pickup/drop, not dead random-walk. Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
130 lines
4.9 KiB
JavaScript
130 lines
4.9 KiB
JavaScript
// coach/advice.js — turn knowledge.lessons into actionable dispatch overrides.
|
|
//
|
|
// The reflex chain calls consult() right before it would dispatch a
|
|
// planned skill. If a high-confidence lesson in the knowledge DB says
|
|
// "avoid that skill in this situation", we either swap in the lesson's
|
|
// preferred alternative or back off (which the curriculum reflex
|
|
// translates into wander / cooldown).
|
|
//
|
|
// This is the closing of the learning loop: post-mortem → lesson →
|
|
// recall → behavioural change. Without this, the DB is just a log.
|
|
|
|
import { isAvailable as knowledgeAvailable, topAdvice, markApplied } from "../knowledge/index.js";
|
|
import { isRegistered } from "../skill-registry.js";
|
|
import { info, warn } from "../log.js";
|
|
|
|
// Skills we will not blindly swap into — they require their own
|
|
// preconditions (e.g. survive.flee needs a known threat direction).
|
|
// The dispatcher will still run runSkill on them, which performs the
|
|
// real precondition check.
|
|
const SAFE_OVERRIDES = new Set([
|
|
"survive.flee",
|
|
"survive.sleep",
|
|
"survive.eat",
|
|
"survive.acquire-food",
|
|
"survive.scout-food",
|
|
"survive.pillar-up",
|
|
"recovery.tunnel-out",
|
|
"explore.far",
|
|
"explore.wander",
|
|
"village.relocate",
|
|
"village.build-shelter",
|
|
"village.choose-base",
|
|
]);
|
|
|
|
// Pi-coach occasionally suggests prefer_skill values that are mode names
|
|
// (from runtime/modes.js) rather than registered skill ids. We translate
|
|
// them to the closest equivalent skill before the SAFE_OVERRIDES check.
|
|
// Unknown values are returned as-is and will fall through to 'avoid'.
|
|
const MODE_TO_SKILL = Object.freeze({
|
|
self_preservation: "survive.flee",
|
|
night_shelter: "survive.sleep",
|
|
hunger: "survive.eat",
|
|
shelter: "village.build-shelter",
|
|
flee: "survive.flee",
|
|
sleep: "survive.sleep",
|
|
eat: "survive.eat",
|
|
tunnel_out: "recovery.tunnel-out",
|
|
"tunnel-out": "recovery.tunnel-out",
|
|
explore: "explore.far",
|
|
wander: "explore.far",
|
|
scout_food: "survive.scout-food",
|
|
"scout-food": "survive.scout-food",
|
|
relocate: "village.relocate",
|
|
});
|
|
|
|
function normalisePreferSkill(raw) {
|
|
if (!raw || typeof raw !== "string") return null;
|
|
if (SAFE_OVERRIDES.has(raw) && isRegistered(raw)) return raw;
|
|
const lower = raw.toLowerCase().trim();
|
|
if (MODE_TO_SKILL[lower]) return MODE_TO_SKILL[lower];
|
|
// Pi sometimes writes "survive_flee" or "survive flee"; normalise.
|
|
const dot = lower.replace(/[_\s]+/g, ".");
|
|
if (SAFE_OVERRIDES.has(dot) && isRegistered(dot)) return dot;
|
|
if (MODE_TO_SKILL[dot]) return MODE_TO_SKILL[dot];
|
|
// Anything else (Pi hallucinated names like "relocate.surface",
|
|
// "choose.safe.surface", "survive.shelter", "gather.visible_log") —
|
|
// hard reject. We'd rather fall through to 'avoid' / 'proceed' than
|
|
// dispatch a nonexistent skill.
|
|
return null;
|
|
}
|
|
|
|
/**
|
|
* consult({ plannedSkillId, snapshot })
|
|
* → { action: 'override'|'avoid'|'proceed', overrideSkillId?, lessonId?, lesson? }
|
|
*
|
|
* 'override' — dispatch overrideSkillId instead of plannedSkillId
|
|
* 'avoid' — don't dispatch plannedSkillId; caller falls back to wander/idle
|
|
* 'proceed' — no high-confidence lesson applies; dispatch as planned
|
|
*/
|
|
export function consult({ plannedSkillId, snapshot } = {}) {
|
|
if (!knowledgeAvailable()) return PROCEED;
|
|
if (!plannedSkillId) return PROCEED;
|
|
const hostile = snapshot?.closestHostile?.name ?? snapshot?.threats?.[0]?.name ?? null;
|
|
const situation = snapshot?.situationHash ?? null;
|
|
const advice = topAdvice({
|
|
skill: plannedSkillId,
|
|
hostile,
|
|
situation,
|
|
});
|
|
if (!advice.lessonId) return PROCEED;
|
|
|
|
// avoid_skill matches?
|
|
if (advice.avoid && advice.avoid === plannedSkillId) {
|
|
const normalisedPrefer = normalisePreferSkill(advice.prefer);
|
|
if (normalisedPrefer && SAFE_OVERRIDES.has(normalisedPrefer) && isRegistered(normalisedPrefer)) {
|
|
if (normalisedPrefer !== advice.prefer) {
|
|
info("coach", `advice: normalised prefer "${advice.prefer}" → "${normalisedPrefer}"`);
|
|
}
|
|
info("coach", `advice: override ${plannedSkillId} → ${normalisedPrefer} (lesson #${advice.lessonId})`);
|
|
return {
|
|
action: "override",
|
|
overrideSkillId: normalisedPrefer,
|
|
lessonId: advice.lessonId,
|
|
lesson: advice.lesson,
|
|
};
|
|
}
|
|
if (advice.prefer && !normalisedPrefer) {
|
|
warn("coach", `advice: rejected hallucinated prefer_skill "${advice.prefer}" (lesson #${advice.lessonId})`);
|
|
}
|
|
info("coach", `advice: avoid ${plannedSkillId} (lesson #${advice.lessonId})`);
|
|
return { action: "avoid", lessonId: advice.lessonId, lesson: advice.lesson };
|
|
}
|
|
return PROCEED;
|
|
}
|
|
|
|
/**
|
|
* After the dispatcher runs the (possibly overridden) skill, call this
|
|
* with the lesson id and whether the outcome was good. Increments the
|
|
* lesson's applied/succeeded counters and nudges its confidence.
|
|
*/
|
|
export function reportOutcome({ lessonId, succeeded }) {
|
|
if (!lessonId) return;
|
|
markApplied(lessonId, { succeeded: !!succeeded });
|
|
}
|
|
|
|
const PROCEED = Object.freeze({ action: "proceed", lessonId: null, lesson: null });
|
|
|
|
// Test exports
|
|
export const __testing = { SAFE_OVERRIDES, MODE_TO_SKILL, normalisePreferSkill };
|