feat(runtime/coach,reflex): retrieval-augmented dispatch via learned lessons
This closes the learning loop. Lessons in knowledge.db now actually
influence behaviour:
- runtime/coach/advice.js: consult({plannedSkillId, snapshot}) reads
knowledge.topAdvice() and returns 'override' / 'avoid' / 'proceed'.
When a lesson says "avoid <skill>" with prefer="survive.flee" (etc.),
the dispatcher swaps in the alternative.
- runtime/reflex.js:
* curriculumReflex now consults advice before dispatch; on 'avoid'
backs off the planned skill + sets wander hint; on 'override'
dispatches the lesson's preferred alternative.
* defendReflex (dist≤4 melee branch) consults advice too — so a
creeper at 4m honours the starter rule "attack creeper → flee".
Failure outcomes feed back via markApplied so confidence stays
grounded.
SAFE_OVERRIDES whitelist contains only known runSkill targets
(survive.flee, survive.sleep, survive.eat, recovery.tunnel-out,
explore.far/wander, village.build-shelter); unknown prefers fall back
to plain 'avoid'.
7 advice tests; total suite 237 green.
Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
+1
-1
@@ -16,7 +16,7 @@
|
||||
"tui": "tsx tui/tui.tsx",
|
||||
"propose:apply": "node scripts/propose-apply.js",
|
||||
"stop": "bash scripts/stop.sh",
|
||||
"test": "node --test runtime/skills/contract.test.js runtime/skills/groups.test.js runtime/skills/compat.test.js runtime/skills/recovery-tunnel-out.test.js runtime/curriculum.test.js runtime/social/social.test.js runtime/social/conversation.test.js runtime/social/chat-history.test.js runtime/social/reply-pi.test.js runtime/stuck-incident.test.js runtime/compat.test.js runtime/reflex.test.js runtime/base-site.test.js runtime/locations.test.js runtime/watch-filter.test.js runtime/world-journal.test.js runtime/scenario-memory.test.js runtime/critic.test.js runtime/skill-library.test.js runtime/modes.test.js runtime/pathfinder-watchdog.test.js runtime/knowledge/knowledge.test.js runtime/coach/postmortem.test.js runtime/persona/chatter.test.js scripts/edit-scope.test.js scripts/lint-patch.test.js"
|
||||
"test": "node --test runtime/skills/contract.test.js runtime/skills/groups.test.js runtime/skills/compat.test.js runtime/skills/recovery-tunnel-out.test.js runtime/curriculum.test.js runtime/social/social.test.js runtime/social/conversation.test.js runtime/social/chat-history.test.js runtime/social/reply-pi.test.js runtime/stuck-incident.test.js runtime/compat.test.js runtime/reflex.test.js runtime/base-site.test.js runtime/locations.test.js runtime/watch-filter.test.js runtime/world-journal.test.js runtime/scenario-memory.test.js runtime/critic.test.js runtime/skill-library.test.js runtime/modes.test.js runtime/pathfinder-watchdog.test.js runtime/knowledge/knowledge.test.js runtime/coach/postmortem.test.js runtime/coach/advice.test.js runtime/persona/chatter.test.js scripts/edit-scope.test.js scripts/lint-patch.test.js"
|
||||
},
|
||||
"dependencies": {
|
||||
"better-sqlite3": "^11.10.0",
|
||||
|
||||
@@ -0,0 +1,79 @@
|
||||
// coach/advice.js — turn knowledge.lessons into actionable dispatch overrides.
|
||||
//
|
||||
// The reflex chain calls consult() right before it would dispatch a
|
||||
// planned skill. If a high-confidence lesson in the knowledge DB says
|
||||
// "avoid that skill in this situation", we either swap in the lesson's
|
||||
// preferred alternative or back off (which the curriculum reflex
|
||||
// translates into wander / cooldown).
|
||||
//
|
||||
// This is the closing of the learning loop: post-mortem → lesson →
|
||||
// recall → behavioural change. Without this, the DB is just a log.
|
||||
|
||||
import { isAvailable as knowledgeAvailable, topAdvice, markApplied } from "../knowledge/index.js";
|
||||
import { info } from "../log.js";
|
||||
|
||||
// Skills we will not blindly swap into — they require their own
|
||||
// preconditions (e.g. survive.flee needs a known threat direction).
|
||||
// The dispatcher will still run runSkill on them, which performs the
|
||||
// real precondition check.
|
||||
const SAFE_OVERRIDES = new Set([
|
||||
"survive.flee",
|
||||
"survive.sleep",
|
||||
"survive.eat",
|
||||
"recovery.tunnel-out",
|
||||
"explore.far",
|
||||
"explore.wander",
|
||||
"village.build-shelter",
|
||||
]);
|
||||
|
||||
/**
|
||||
* consult({ plannedSkillId, snapshot })
|
||||
* → { action: 'override'|'avoid'|'proceed', overrideSkillId?, lessonId?, lesson? }
|
||||
*
|
||||
* 'override' — dispatch overrideSkillId instead of plannedSkillId
|
||||
* 'avoid' — don't dispatch plannedSkillId; caller falls back to wander/idle
|
||||
* 'proceed' — no high-confidence lesson applies; dispatch as planned
|
||||
*/
|
||||
export function consult({ plannedSkillId, snapshot } = {}) {
|
||||
if (!knowledgeAvailable()) return PROCEED;
|
||||
if (!plannedSkillId) return PROCEED;
|
||||
const hostile = snapshot?.closestHostile?.name ?? snapshot?.threats?.[0]?.name ?? null;
|
||||
const situation = snapshot?.situationHash ?? null;
|
||||
const advice = topAdvice({
|
||||
skill: plannedSkillId,
|
||||
hostile,
|
||||
situation,
|
||||
});
|
||||
if (!advice.lessonId) return PROCEED;
|
||||
|
||||
// avoid_skill matches?
|
||||
if (advice.avoid && advice.avoid === plannedSkillId) {
|
||||
if (advice.prefer && SAFE_OVERRIDES.has(advice.prefer)) {
|
||||
info("coach", `advice: override ${plannedSkillId} → ${advice.prefer} (lesson #${advice.lessonId})`);
|
||||
return {
|
||||
action: "override",
|
||||
overrideSkillId: advice.prefer,
|
||||
lessonId: advice.lessonId,
|
||||
lesson: advice.lesson,
|
||||
};
|
||||
}
|
||||
info("coach", `advice: avoid ${plannedSkillId} (lesson #${advice.lessonId})`);
|
||||
return { action: "avoid", lessonId: advice.lessonId, lesson: advice.lesson };
|
||||
}
|
||||
return PROCEED;
|
||||
}
|
||||
|
||||
/**
|
||||
* After the dispatcher runs the (possibly overridden) skill, call this
|
||||
* with the lesson id and whether the outcome was good. Increments the
|
||||
* lesson's applied/succeeded counters and nudges its confidence.
|
||||
*/
|
||||
export function reportOutcome({ lessonId, succeeded }) {
|
||||
if (!lessonId) return;
|
||||
markApplied(lessonId, { succeeded: !!succeeded });
|
||||
}
|
||||
|
||||
const PROCEED = Object.freeze({ action: "proceed", lessonId: null, lesson: null });
|
||||
|
||||
// Test exports
|
||||
export const __testing = { SAFE_OVERRIDES };
|
||||
@@ -0,0 +1,97 @@
|
||||
import { test } from "node:test";
|
||||
import assert from "node:assert/strict";
|
||||
import { mkdtempSync, rmSync } from "node:fs";
|
||||
import { tmpdir } from "node:os";
|
||||
import { join } from "node:path";
|
||||
|
||||
import { initKnowledge, record } from "../knowledge/index.js";
|
||||
import { __resetForTests, isAvailable, closeStore } from "../knowledge/store.js";
|
||||
import { consult, reportOutcome, __testing } from "./advice.js";
|
||||
|
||||
const { SAFE_OVERRIDES } = __testing;
|
||||
|
||||
async function bootstrap() {
|
||||
__resetForTests();
|
||||
const tmp = mkdtempSync(join(tmpdir(), "pepa-advice-test-"));
|
||||
await initKnowledge({ stateDir: tmp });
|
||||
return tmp;
|
||||
}
|
||||
|
||||
function cleanup(tmp) {
|
||||
closeStore();
|
||||
try { rmSync(tmp, { recursive: true, force: true }); } catch {}
|
||||
}
|
||||
|
||||
test("consult: returns proceed when knowledge disabled", () => {
|
||||
__resetForTests();
|
||||
const res = consult({ plannedSkillId: "gather.logs", snapshot: {} });
|
||||
assert.equal(res.action, "proceed");
|
||||
});
|
||||
|
||||
test("consult: returns proceed when no relevant lesson", async () => {
|
||||
const tmp = await bootstrap();
|
||||
if (!isAvailable()) { cleanup(tmp); return; }
|
||||
const res = consult({ plannedSkillId: "gather.unknown-skill", snapshot: {} });
|
||||
assert.equal(res.action, "proceed");
|
||||
cleanup(tmp);
|
||||
});
|
||||
|
||||
test("consult: starter creeper rule routes attack → survive.flee", async () => {
|
||||
const tmp = await bootstrap();
|
||||
if (!isAvailable()) { cleanup(tmp); return; }
|
||||
const res = consult({
|
||||
plannedSkillId: "attack creeper",
|
||||
snapshot: { closestHostile: { name: "creeper", distance: 4 } },
|
||||
});
|
||||
assert.equal(res.action, "override");
|
||||
assert.equal(res.overrideSkillId, "survive.flee");
|
||||
assert.ok(res.lessonId);
|
||||
assert.ok(res.lesson);
|
||||
cleanup(tmp);
|
||||
});
|
||||
|
||||
test("consult: avoid lesson without prefer → 'avoid' action", async () => {
|
||||
const tmp = await bootstrap();
|
||||
if (!isAvailable()) { cleanup(tmp); return; }
|
||||
record({
|
||||
text: "Don't gather.stone — confirmed flaky.",
|
||||
category: "pathing",
|
||||
triggerSkill: "gather.stone",
|
||||
avoidSkill: "gather.stone",
|
||||
preferSkill: null,
|
||||
confidence: 0.9,
|
||||
source: "test",
|
||||
});
|
||||
const res = consult({ plannedSkillId: "gather.stone", snapshot: {} });
|
||||
assert.equal(res.action, "avoid");
|
||||
assert.ok(res.lessonId);
|
||||
cleanup(tmp);
|
||||
});
|
||||
|
||||
test("consult: prefer outside SAFE_OVERRIDES set → falls to avoid", async () => {
|
||||
const tmp = await bootstrap();
|
||||
if (!isAvailable()) { cleanup(tmp); return; }
|
||||
record({
|
||||
text: "test fallback",
|
||||
category: "combat",
|
||||
triggerSkill: "gather.logs",
|
||||
avoidSkill: "gather.logs",
|
||||
preferSkill: "non.standard.skill",
|
||||
confidence: 0.9,
|
||||
source: "test",
|
||||
});
|
||||
const res = consult({ plannedSkillId: "gather.logs", snapshot: {} });
|
||||
assert.equal(res.action, "avoid", "unsafe prefer falls back to avoid, not override");
|
||||
cleanup(tmp);
|
||||
});
|
||||
|
||||
test("reportOutcome: no-op without lessonId", () => {
|
||||
reportOutcome({ lessonId: null });
|
||||
assert.ok(true);
|
||||
});
|
||||
|
||||
test("SAFE_OVERRIDES: only contains known reflex skills", () => {
|
||||
for (const id of SAFE_OVERRIDES) {
|
||||
assert.ok(typeof id === "string" && id.includes("."), `${id} looks like a real skill id`);
|
||||
}
|
||||
});
|
||||
+29
-3
@@ -25,6 +25,7 @@ import {
|
||||
wander,
|
||||
} from "./actions.js";
|
||||
import { runSkill, getSkill } from "./skills/index.js";
|
||||
import { consult as consultAdvice, reportOutcome as reportAdviceOutcome } from "./coach/advice.js";
|
||||
import { situationHash } from "./scenario-memory.js";
|
||||
import { tickModes } from "./modes.js";
|
||||
|
||||
@@ -200,6 +201,15 @@ function defendReflex(ctx) {
|
||||
ctx.defendAttackStuck = null;
|
||||
return dispatchDefendFlee(ctx, hostile, dist, { ignoreCooldown: true });
|
||||
}
|
||||
// v0.2.0 — consult learned lessons. If knowledge says "do not
|
||||
// attack <hostile> in this state" (e.g. creeper rule, or no-weapon
|
||||
// rule learned from post-mortems), flee instead. This is the
|
||||
// closing of the learning loop for emergency combat.
|
||||
const advice = consultAdvice({ plannedSkillId: `attack ${hostile.name}`, snapshot: s });
|
||||
if (advice.action === "avoid" || advice.action === "override") {
|
||||
if (advice.lessonId) reportAdviceOutcome({ lessonId: advice.lessonId, succeeded: false });
|
||||
return dispatchDefendFlee(ctx, hostile, dist, { ignoreCooldown: true });
|
||||
}
|
||||
ctx.dispatch(
|
||||
() => attackNearestUntilClear(ctx.bot, hostile.name, {
|
||||
maxSwings: ctx.defendAttackMaxSwings,
|
||||
@@ -441,10 +451,26 @@ function curriculumReflex(ctx) {
|
||||
}
|
||||
}
|
||||
|
||||
// v0.2.0 — consult learned lessons. If a high-confidence lesson says
|
||||
// "avoid <skillId> in this situation", swap to its preferred
|
||||
// alternative (or back off entirely if no safe alternative is named).
|
||||
const advice = consultAdvice({ plannedSkillId: skillId, snapshot: ctx.snapshot });
|
||||
let dispatchSkillId = skillId;
|
||||
if (advice.action === "avoid") {
|
||||
ctx.skillBackoff = ctx.skillBackoff ?? {};
|
||||
ctx.skillBackoff[skillId] = Date.now() + SKILL_BACKOFF_MS;
|
||||
ctx.skillBackoff["__wander_hint__"] = Date.now() + SKILL_BACKOFF_MS;
|
||||
return { action: "noop", kind: "curriculum-advice-avoid", label: skillId, lessonId: advice.lessonId };
|
||||
}
|
||||
if (advice.action === "override" && advice.overrideSkillId) {
|
||||
dispatchSkillId = advice.overrideSkillId;
|
||||
}
|
||||
|
||||
ctx.lastCurriculumAt = Date.now();
|
||||
ctx.dispatch(() => runSkill(skillId, ctx), skillId, {
|
||||
ctx.dispatch(() => runSkill(dispatchSkillId, ctx), dispatchSkillId, {
|
||||
onComplete: (res) => {
|
||||
ctx.skillBackoff = ctx.skillBackoff ?? {};
|
||||
if (advice.lessonId) reportAdviceOutcome({ lessonId: advice.lessonId, succeeded: !!res?.ok });
|
||||
if (res?.recovery?.hint === "wander") {
|
||||
// Same fix the old autonomous reflex applied for "no reachable
|
||||
// log" — switch to exploration for a minute.
|
||||
@@ -456,7 +482,7 @@ function curriculumReflex(ctx) {
|
||||
// retried on the very next tick. Hold for SKILL_BACKOFF_MS.
|
||||
const cooldownCodes = new Set(["missing_tool", "missing_material", "no_target", "no_food_source", "unsupported_version", "no_chest", "no_space", "nothing_to_deposit"]);
|
||||
if (cooldownCodes.has(res?.code)) {
|
||||
ctx.skillBackoff[skillId] = Date.now() + SKILL_BACKOFF_MS;
|
||||
ctx.skillBackoff[dispatchSkillId] = Date.now() + SKILL_BACKOFF_MS;
|
||||
}
|
||||
} else {
|
||||
// Success clears the wander hint immediately.
|
||||
@@ -465,7 +491,7 @@ function curriculumReflex(ctx) {
|
||||
}
|
||||
},
|
||||
});
|
||||
return { action: "dispatched", kind: "curriculum-skill", label: skillId };
|
||||
return { action: "dispatched", kind: "curriculum-skill", label: dispatchSkillId };
|
||||
}
|
||||
|
||||
// ---- idle ------------------------------------------------------------------
|
||||
|
||||
Reference in New Issue
Block a user