Files
pepa-pi-bot/runtime/coach/advice.js
T
mayatnikovandClaude Opus 4.7 fcfa2277ba v0.3.0-rc.1: live skill registry + fast advisor scaffold
Roots out the v0.2.x failure mode: Pi-extracted lessons routinely named
hallucinated skill ids (relocate.surface, choose.safe.surface,
survive.shelter, gather.visible_log, …). All 47 Pi-lessons in the live DB
had applied_count=0 because normalisePreferSkill couldn't find them.

Fix:
1. runtime/skill-registry.js — single source of truth derived from
   skills/index.js. Exports listSkillIds, isRegistered, and a
   prompt-ready block (skillRegistryPrompt) grouped by namespace.
2. Pi prompts (coach/postmortem, coach/reflect) embed the live registry
   with a "USE ONLY THESE, never invent" instruction. Lessons are
   filtered at write-time too — anything not in the registry and not a
   known mode name gets dropped.
3. coach/advice.js — normalisePreferSkill now returns null for unknown
   ids, hardening consult() against any hallucinations that slip
   through. Warn-logged for visibility.

Also lays the LLM substrate for the rest of v0.3.0:

- runtime/llm/provider.js — OpenAI-compatible chat client. Configured
  via PEPA_FAST_LLM_{BASE_URL,API_KEY,MODEL,TIMEOUT_MS}. Safe no-op
  unless API_KEY is set. Supports JSON-mode.
- runtime/coach/fast-advisor.js — tactical advisor tier (scaffold).
  Exposes advise() that asks the fast LLM what to do RIGHT NOW when
  the reflex is wedged/stuck. Rejects hallucinated skill ids using the
  registry. Rate-limited 6/h, 30s cooldown. Not auto-triggered yet —
  wired into reflex in rc.3 (awareness layer).

Tests: 279 green (+24 vs rc.3): 5 registry, 9 provider, 10 advisor.

See dev/v0.3.0/PLAN.md for the full iteration design (manifesto needs
ladder, event-driven awareness, skill pre-emption) and STATUS.md for
shipped/pending tracking.

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
2026-05-27 17:39:43 +03:00

124 lines
4.8 KiB
JavaScript

// coach/advice.js — turn knowledge.lessons into actionable dispatch overrides.
//
// The reflex chain calls consult() right before it would dispatch a
// planned skill. If a high-confidence lesson in the knowledge DB says
// "avoid that skill in this situation", we either swap in the lesson's
// preferred alternative or back off (which the curriculum reflex
// translates into wander / cooldown).
//
// This is the closing of the learning loop: post-mortem → lesson →
// recall → behavioural change. Without this, the DB is just a log.
import { isAvailable as knowledgeAvailable, topAdvice, markApplied } from "../knowledge/index.js";
import { isRegistered } from "../skill-registry.js";
import { info, warn } from "../log.js";
// Skills we will not blindly swap into — they require their own
// preconditions (e.g. survive.flee needs a known threat direction).
// The dispatcher will still run runSkill on them, which performs the
// real precondition check.
const SAFE_OVERRIDES = new Set([
"survive.flee",
"survive.sleep",
"survive.eat",
"survive.pillar-up",
"recovery.tunnel-out",
"explore.far",
"explore.wander",
"village.build-shelter",
"village.choose-base",
]);
// Pi-coach occasionally suggests prefer_skill values that are mode names
// (from runtime/modes.js) rather than registered skill ids. We translate
// them to the closest equivalent skill before the SAFE_OVERRIDES check.
// Unknown values are returned as-is and will fall through to 'avoid'.
const MODE_TO_SKILL = Object.freeze({
self_preservation: "survive.flee",
night_shelter: "survive.sleep",
hunger: "survive.eat",
shelter: "village.build-shelter",
flee: "survive.flee",
sleep: "survive.sleep",
eat: "survive.eat",
tunnel_out: "recovery.tunnel-out",
"tunnel-out": "recovery.tunnel-out",
explore: "explore.far",
wander: "explore.far",
});
function normalisePreferSkill(raw) {
if (!raw || typeof raw !== "string") return null;
if (SAFE_OVERRIDES.has(raw) && isRegistered(raw)) return raw;
const lower = raw.toLowerCase().trim();
if (MODE_TO_SKILL[lower]) return MODE_TO_SKILL[lower];
// Pi sometimes writes "survive_flee" or "survive flee"; normalise.
const dot = lower.replace(/[_\s]+/g, ".");
if (SAFE_OVERRIDES.has(dot) && isRegistered(dot)) return dot;
if (MODE_TO_SKILL[dot]) return MODE_TO_SKILL[dot];
// Anything else (Pi hallucinated names like "relocate.surface",
// "choose.safe.surface", "survive.shelter", "gather.visible_log") —
// hard reject. We'd rather fall through to 'avoid' / 'proceed' than
// dispatch a nonexistent skill.
return null;
}
/**
* consult({ plannedSkillId, snapshot })
* → { action: 'override'|'avoid'|'proceed', overrideSkillId?, lessonId?, lesson? }
*
* 'override' — dispatch overrideSkillId instead of plannedSkillId
* 'avoid' — don't dispatch plannedSkillId; caller falls back to wander/idle
* 'proceed' — no high-confidence lesson applies; dispatch as planned
*/
export function consult({ plannedSkillId, snapshot } = {}) {
if (!knowledgeAvailable()) return PROCEED;
if (!plannedSkillId) return PROCEED;
const hostile = snapshot?.closestHostile?.name ?? snapshot?.threats?.[0]?.name ?? null;
const situation = snapshot?.situationHash ?? null;
const advice = topAdvice({
skill: plannedSkillId,
hostile,
situation,
});
if (!advice.lessonId) return PROCEED;
// avoid_skill matches?
if (advice.avoid && advice.avoid === plannedSkillId) {
const normalisedPrefer = normalisePreferSkill(advice.prefer);
if (normalisedPrefer && SAFE_OVERRIDES.has(normalisedPrefer) && isRegistered(normalisedPrefer)) {
if (normalisedPrefer !== advice.prefer) {
info("coach", `advice: normalised prefer "${advice.prefer}" → "${normalisedPrefer}"`);
}
info("coach", `advice: override ${plannedSkillId}${normalisedPrefer} (lesson #${advice.lessonId})`);
return {
action: "override",
overrideSkillId: normalisedPrefer,
lessonId: advice.lessonId,
lesson: advice.lesson,
};
}
if (advice.prefer && !normalisedPrefer) {
warn("coach", `advice: rejected hallucinated prefer_skill "${advice.prefer}" (lesson #${advice.lessonId})`);
}
info("coach", `advice: avoid ${plannedSkillId} (lesson #${advice.lessonId})`);
return { action: "avoid", lessonId: advice.lessonId, lesson: advice.lesson };
}
return PROCEED;
}
/**
* After the dispatcher runs the (possibly overridden) skill, call this
* with the lesson id and whether the outcome was good. Increments the
* lesson's applied/succeeded counters and nudges its confidence.
*/
export function reportOutcome({ lessonId, succeeded }) {
if (!lessonId) return;
markApplied(lessonId, { succeeded: !!succeeded });
}
const PROCEED = Object.freeze({ action: "proceed", lessonId: null, lesson: null });
// Test exports
export const __testing = { SAFE_OVERRIDES, MODE_TO_SKILL, normalisePreferSkill };