feat(v0.3.0): paradigm shift — TimeWeb-only LLM + persistent advisor trail + improvement queue
This is the rc.4 batch the user requested:
1. Emergency triggers (low HP + close hostile, lava-under-foot)
bypass the long cooldown so the LLM is consulted BEFORE the bot
dies, not after.
2. Active manifesto need is now included in the advisor user prompt
— the LLM picks suggestions that satisfy the bot's current
concrete need (L2 tools_wood → "gather logs nearby" not
"explore further").
3. Every advisor recommendation is persisted to SQLite
(advisor_recommendations table) with full token usage. The
reflex marks 'applied=1' when it dispatches and updates
outcome_ok/code when the dispatch completes. Ground truth for
"is the LLM actually helping" lives in the DB, not in logs.
4. Pi CLI is OUT of every background loop. coach/postmortem and
coach/reflect now go through the same TimeWeb endpoint
fast-advisor uses, via the shared coach/llm-call.js helper.
Pi is reserved for manual operator commands.
5. The LLM (postmortem, reflect, advisor) can flag "structural
gaps" — missing skills/features the operator should implement.
These land in the new improvement_requests table. Dedup by
title bumps `votes` instead of inserting duplicates so the
queue doesn't bloat. Operator views via
`node scripts/list-improvements.js`.
6. A deterministic trigger-tuner runs hourly: reads 24h of
recommendation stats, flags triggers whose success rate is
below 25% (sample ≥ 5) or whose prompts are expensive (>1000
input tokens) with mediocre payoff. Improvements get
source="tuner", category="tuning". No LLM call.
New files:
runtime/coach/llm-call.js — askAnalytical() helper
runtime/coach/trigger-tuner.js — stats → improvements
runtime/coach/trigger-tuner.test.js
scripts/list-improvements.js — operator CLI
Schema additions:
advisor_recommendations: id, ts, trigger_reason, planned_skill,
recommended_skill, action, rationale, active_need, tokens_in,
tokens_out, latency_ms, applied, outcome_ok, outcome_code, outcome_at
improvement_requests: id, ts, source, category, title, description,
context, priority, status, duplicate_of, votes, implemented_at, notes
Renamed env-var consumers:
Pi-coach drainOnce({ askPi }) → drainOnce({ askAnalyticalFn? })
Pi-reflect runOnce({ askPi }) → runOnce({ askAnalyticalFn? })
bot.js attachCoach/attachReflect no longer pass askPi
attachTuner() added to bot.js spawn handler
lessons.source 'pi-coach' → 'timeweb-coach'
lessons.source 'pi-reflect' → 'timeweb-reflect'
Token cost measured live:
~705 input + 45 output = ~750 total per advisor call
worst case @ 6 calls/hour rate cap = ~108K tokens/day
OpenAI gpt-5-mini reference price: ~$0.60/month
Operator usage:
node scripts/list-improvements.js # open queue
node scripts/list-improvements.js --stats # advisor performance
node scripts/list-improvements.js --done 17 "shipped in 0.3.1"
node scripts/list-improvements.js --reject 18 "duplicate"
Tests: 360 green (was 332, +28 new).
Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,145 @@
|
||||
#!/usr/bin/env node
|
||||
// Operator-facing view of bot-flagged improvement requests.
|
||||
//
|
||||
// The LLM (postmortem + reflect + trigger-tuner) writes here when it
|
||||
// notices a structural gap — a missing skill or a misconfigured policy.
|
||||
// You read this, decide what's worth implementing, and ship it.
|
||||
//
|
||||
// Usage:
|
||||
// node scripts/list-improvements.js # all open, sorted by priority
|
||||
// node scripts/list-improvements.js --status all # everything
|
||||
// node scripts/list-improvements.js --status implemented
|
||||
// node scripts/list-improvements.js --source reflect
|
||||
// node scripts/list-improvements.js --category skill
|
||||
// node scripts/list-improvements.js --done 17 "shipped in 0.3.1"
|
||||
// node scripts/list-improvements.js --reject 18 "duplicate"
|
||||
// node scripts/list-improvements.js --stats # aggregate counts
|
||||
|
||||
import { config as loadDotenv } from "dotenv";
|
||||
loadDotenv();
|
||||
|
||||
import { initKnowledge, listImprovements, markImprovementStatus, isAvailable, recommendationStats } from "../runtime/knowledge/index.js";
|
||||
import { stateDir } from "../runtime/config.js";
|
||||
|
||||
function parseArgs(argv) {
|
||||
const out = { status: "open", source: null, category: null, limit: 50, stats: false, action: null };
|
||||
for (let i = 2; i < argv.length; i++) {
|
||||
const a = argv[i];
|
||||
if (a === "--status") out.status = argv[++i];
|
||||
else if (a === "--source") out.source = argv[++i];
|
||||
else if (a === "--category") out.category = argv[++i];
|
||||
else if (a === "--limit") out.limit = Number(argv[++i]) || 50;
|
||||
else if (a === "--stats") out.stats = true;
|
||||
else if (a === "--done") { out.action = "implemented"; out.actionId = Number(argv[++i]); out.actionNote = argv[++i] ?? null; }
|
||||
else if (a === "--reject") { out.action = "rejected"; out.actionId = Number(argv[++i]); out.actionNote = argv[++i] ?? null; }
|
||||
else if (a === "--inprogress") { out.action = "in_progress"; out.actionId = Number(argv[++i]); out.actionNote = argv[++i] ?? null; }
|
||||
else if (a === "--help" || a === "-h") { printHelp(); process.exit(0); }
|
||||
}
|
||||
if (out.status === "all") out.status = null;
|
||||
return out;
|
||||
}
|
||||
|
||||
function printHelp() {
|
||||
console.log(`Usage: node scripts/list-improvements.js [options]
|
||||
|
||||
--status <open|in_progress|implemented|rejected|all> default: open
|
||||
--source <postmortem|reflect|advisor|tuner|manual>
|
||||
--category <skill|tuning|perception|planning|social|other>
|
||||
--limit <n> default: 50
|
||||
--stats show advisor recommendation stats
|
||||
--done <id> [note] mark a request as implemented
|
||||
--inprogress <id> [note] mark a request as in progress
|
||||
--reject <id> [note] mark a request as rejected
|
||||
`);
|
||||
}
|
||||
|
||||
function priorityLabel(p) {
|
||||
return ["", "P1 urgent", "P2 high", "P3 normal", "P4 low", "P5 nice-to-have"][p] ?? `P${p}`;
|
||||
}
|
||||
|
||||
function statusLabel(s) {
|
||||
return ({
|
||||
open: "OPEN",
|
||||
in_progress: "WIP",
|
||||
implemented: "DONE",
|
||||
rejected: "REJECTED",
|
||||
duplicate: "DUP",
|
||||
})[s] ?? s;
|
||||
}
|
||||
|
||||
function formatTs(ts) {
|
||||
if (!ts) return "?";
|
||||
const d = new Date(ts);
|
||||
return d.toISOString().slice(0, 16).replace("T", " ");
|
||||
}
|
||||
|
||||
function renderRow(r) {
|
||||
const lines = [
|
||||
`#${r.id} [${statusLabel(r.status).padEnd(8)}] ${priorityLabel(r.priority).padEnd(18)} ×${r.votes}`,
|
||||
` ${r.title}`,
|
||||
` source=${r.source} category=${r.category ?? "?"} created=${formatTs(r.ts)}${r.implemented_at ? ` done=${formatTs(r.implemented_at)}` : ""}`,
|
||||
];
|
||||
if (r.description) {
|
||||
lines.push(` ${String(r.description).slice(0, 240)}`);
|
||||
}
|
||||
if (r.notes) {
|
||||
lines.push(` notes: ${String(r.notes).slice(0, 200)}`);
|
||||
}
|
||||
return lines.join("\n");
|
||||
}
|
||||
|
||||
async function main() {
|
||||
const args = parseArgs(process.argv);
|
||||
await initKnowledge({ stateDir });
|
||||
if (!isAvailable()) {
|
||||
console.error(`knowledge DB unavailable at ${stateDir}/knowledge.db`);
|
||||
console.error(`(install better-sqlite3 and ensure the bot has run at least once)`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
if (args.action) {
|
||||
markImprovementStatus(args.actionId, { status: args.action, notes: args.actionNote });
|
||||
console.log(`#${args.actionId} → ${args.action}${args.actionNote ? ` (${args.actionNote})` : ""}`);
|
||||
return;
|
||||
}
|
||||
|
||||
if (args.stats) {
|
||||
const stats = recommendationStats({ sinceHours: 24 });
|
||||
console.log(`=== Advisor recommendation stats (last 24h) ===`);
|
||||
if (stats.length === 0) {
|
||||
console.log("(no recommendations yet)");
|
||||
} else {
|
||||
console.log(" trigger_reason total applied ok fail avg_in avg_out avg_latency");
|
||||
for (const s of stats) {
|
||||
console.log(` ${(s.trigger_reason || "?").padEnd(22)} ${String(s.total).padStart(5)} ${String(s.applied ?? 0).padStart(7)} ${String(s.succeeded ?? 0).padStart(2)} ${String(s.failed ?? 0).padStart(4)} ${String(Math.round(s.avg_in ?? 0)).padStart(6)} ${String(Math.round(s.avg_out ?? 0)).padStart(7)} ${String(Math.round(s.avg_latency_ms ?? 0)).padStart(11)}`);
|
||||
}
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
const rows = listImprovements({
|
||||
status: args.status,
|
||||
source: args.source,
|
||||
category: args.category,
|
||||
limit: args.limit,
|
||||
});
|
||||
const heading = `=== Improvement requests`
|
||||
+ (args.status ? ` (status=${args.status})` : ` (all)`)
|
||||
+ (args.source ? ` source=${args.source}` : "")
|
||||
+ (args.category ? ` category=${args.category}` : "")
|
||||
+ ` — ${rows.length} row${rows.length === 1 ? "" : "s"} ===`;
|
||||
console.log(heading);
|
||||
if (rows.length === 0) {
|
||||
console.log("(empty)");
|
||||
return;
|
||||
}
|
||||
for (const r of rows) {
|
||||
console.log("");
|
||||
console.log(renderRow(r));
|
||||
}
|
||||
}
|
||||
|
||||
main().catch((e) => {
|
||||
console.error("ERROR:", e?.message ?? e);
|
||||
process.exit(2);
|
||||
});
|
||||
Reference in New Issue
Block a user