mirror of
https://github.com/rohitg00/agentmemory.git
synced 2026-09-14 20:16:33 +08:00
7fb72f4010
* feat(eval): pluggable benchmark harness with in-house coding-agent corpus Adds eval/ tree (outside files field so npm tarball stays thin) with Adapter interface, three reference adapters (grep / vector / agentmemory-hybrid), two benchmarks (LongMemEval _s public, coding-agent-life-v1 in-house 15 sessions), scoring (P@K, R@K, hit, top-gold-rank), NDJSON output, sandbox script. coding-agent-life-v1 published scorecard at docs/benchmarks/2026-05-20-coding-agent-life-v1.md: agentmemory-hybrid R@5=0.967 P@5=0.578 (100% hit) vs grep R@5=0.967 P@5=0.267. 2.2x better precision on identical input, sandbox-reproducible. Adapter contract: init(sessions, config) -> State; query(q, state, k) -> RankedDoc[] npm scripts: npm run eval:coding-life (no download, no API key for grep) npm run eval:longmemeval (needs OPENAI key + 278MB download) eval/scripts/sandbox.sh boots clean agentmemory + iii-engine on ports 3411/3412 with isolated data dir; tears down on exit. README headline updated. 1072/1072 tests pass + 5 new eval tests. * fix(eval): address review findings on benchmark harness - agentmemory adapter: prefer row.sessionId before observationToSession lookup - vector adapter: validate embedBatch response (length, indexes, non-empty rows) - coding-life: positive-int guard on --k; wrap query loop in try/finally so teardown runs - longmemeval: positive-int guards on --k/--limit/--stratify; per-question try/finally - load: throw on haystack_session_ids vs haystack_sessions length mismatch - score: P@K denominator is k (requested cutoff) not topK.length - sandbox.sh: guard rm -rf with non-empty + /tmp/ prefix check - README: drop unsafe rm "$(which iii)"; instruct ~/.local/bin + PATH instead; add language tag to repo-layout fenced block - sessions.json: fix "two-phase" -> "three-phase" wording mismatch
55 lines
1.8 KiB
TypeScript
55 lines
1.8 KiB
TypeScript
import { readFileSync } from "node:fs";
|
|
import type { Question, Session } from "./types.js";
|
|
|
|
interface LongMemEvalRaw {
|
|
question_id: string;
|
|
question_type: string;
|
|
question: string;
|
|
answer?: string;
|
|
answer_session_ids: string[];
|
|
haystack_session_ids: string[];
|
|
haystack_sessions: Array<Array<{ role: string; content: string }>>;
|
|
}
|
|
|
|
function flattenSession(turns: Array<{ role: string; content: string }>): string {
|
|
return turns.map((t) => `[${t.role}] ${t.content}`).join("\n\n");
|
|
}
|
|
|
|
export function loadLongMemEval(path: string, limit?: number): Question[] {
|
|
const raw = JSON.parse(readFileSync(path, "utf8")) as LongMemEvalRaw[];
|
|
const slice = typeof limit === "number" ? raw.slice(0, limit) : raw;
|
|
const questions: Question[] = [];
|
|
for (const r of slice) {
|
|
if (r.haystack_session_ids.length !== r.haystack_sessions.length) {
|
|
throw new Error(
|
|
`LongMemEval row ${r.question_id}: haystack_session_ids (${r.haystack_session_ids.length}) and haystack_sessions (${r.haystack_sessions.length}) length mismatch`,
|
|
);
|
|
}
|
|
const haystack: Session[] = r.haystack_session_ids.map((id, i) => ({
|
|
id,
|
|
content: flattenSession(r.haystack_sessions[i]),
|
|
}));
|
|
questions.push({
|
|
id: r.question_id,
|
|
type: r.question_type,
|
|
question: r.question,
|
|
answer: r.answer,
|
|
goldSessionIds: r.answer_session_ids,
|
|
haystack,
|
|
});
|
|
}
|
|
return questions;
|
|
}
|
|
|
|
export function stratifySample(questions: Question[], perType: number): Question[] {
|
|
const buckets: Record<string, Question[]> = {};
|
|
for (const q of questions) {
|
|
(buckets[q.type] ??= []).push(q);
|
|
}
|
|
const out: Question[] = [];
|
|
for (const type of Object.keys(buckets).sort()) {
|
|
out.push(...buckets[type].slice(0, perType));
|
|
}
|
|
return out;
|
|
}
|