LibreChat/scripts/activity-labels/report.js
Danny Avila a07c0e4ae8
🧪 chore: Add the Activity-Label Prose Eval Harness (#14527)
Grades fast-model activity-label headers against a fixed corpus so
instruction changes are measured rather than eyeballed on one
conversation. This existed untracked while the continuity work was
developed; committing it because it is the only reproducible record of
WHY `ACTIVITY_INSTRUCTION` is ordered and capped the way it is.

- captured.json: 9 real production payloads pulled verbatim from Langfuse
  with the headers that shipped. Irreplaceable — traces age out.
- corpus.js: 17 cases / 28 steps. The captured run replays as one
  sequence, plus synthetic cases for the modes it never exercised
  (all-failed, partial, parallel batches, rapid near-duplicates, entry
  overflow, truncated output, error-shaped success). Multi-step cases
  chain each generated label into the next step's context, which is what
  makes cross-batch redundancy measurable at all.
- prompt.js: faithful port of the SDK's buildActivityLabelPrompt so
  synthetic cases render the bytes production sends, plus a
  previousLabelCap knob for continuity-window experiments.
- variants.js: single-factor instruction variants. The baseline is read
  from the BUILT package (workspace resolution, then dist, then
  LABEL_EVAL_DIST) so a variant can never be graded against a stale copy
  of the shipped instruction.
- checks.js: length/punctuation/markdown/tool-echo/count-echo, plus
  overlap split into `restate` (adds nothing over an earlier header) vs
  `template` (same frame, new payload — often fine).
- run.js / rescore.js: live runner on the production wire shape
  (max_tokens 256) and an offline re-grader, so metric fixes never
  require re-spending on the API.

Results are gitignored — regenerable, and 292K of the 364K. A full sweep
is ~$0.03 per variant and ~45s.

Findings are recorded in the README, two of them counter-intuitive:
enumerating acceptable opening verbs ANCHORED the model rather than
diversifying it (Confirmed 18→23, opener diversity halved), and diverse
examples alone changed nothing. Sentence order is load-bearing, so a
tidying reshuffle of ACTIVITY_INSTRUCTION regresses real output.
2026-07-30 09:22:34 -04:00

148 lines
4.8 KiB
JavaScript
Raw Permalink Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

/** Aggregation + markdown rendering, shared by the live runner and the
* offline rescorer so metric fixes never require re-spending on the API. */
const FLAG_TYPES = [
'len',
'punct',
'quote',
'md',
'opener',
'tool-echo',
'count-echo',
'restate',
'template',
];
const PRICES = { 'claude-haiku-4-5': { input: 1, output: 5 } };
function flagType(flag) {
return flag.split(':')[0];
}
function aggregate(records, model) {
const byVariant = new Map();
for (const record of records) {
if (!byVariant.has(record.variant)) {
byVariant.set(record.variant, {
steps: 0,
errors: 0,
flagCounts: {},
firstWords: {},
totalWords: 0,
latencies: [],
inputTokens: 0,
outputTokens: 0,
});
}
const agg = byVariant.get(record.variant);
if (record.error) {
agg.errors += 1;
continue;
}
agg.steps += 1;
agg.totalWords += record.wordCount;
agg.latencies.push(record.latencyMs);
agg.inputTokens += record.inputTokens;
agg.outputTokens += record.outputTokens;
agg.firstWords[record.firstWord] = (agg.firstWords[record.firstWord] ?? 0) + 1;
for (const flag of record.flags) {
const type = flagType(flag);
agg.flagCounts[type] = (agg.flagCounts[type] ?? 0) + 1;
}
}
const price = PRICES[model];
return [...byVariant.entries()].map(([name, agg]) => {
const sortedFirst = Object.entries(agg.firstWords).sort((a, b) => b[1] - a[1]);
const topOpener = sortedFirst[0] ?? ['—', 0];
return {
variant: name,
steps: agg.steps,
errors: agg.errors,
flagCounts: agg.flagCounts,
distinctOpeners: sortedFirst.length,
topOpener: `${topOpener[0]} ×${topOpener[1]}`,
avgWords: agg.steps > 0 ? (agg.totalWords / agg.steps).toFixed(1) : '—',
meanLatencyMs: agg.latencies.length
? Math.round(agg.latencies.reduce((a, b) => a + b, 0) / agg.latencies.length)
: 0,
inputTokens: agg.inputTokens,
outputTokens: agg.outputTokens,
costUsd: price
? ((agg.inputTokens * price.input + agg.outputTokens * price.output) / 1e6).toFixed(4)
: 'n/a',
};
});
}
function markdownReport({ records, aggregates, runCases, variantNames, model, samples }) {
const lines = [];
lines.push(`# Activity-label eval — ${new Date().toISOString()}`);
lines.push('');
lines.push(`model: \`${model}\` · samples: ${samples} · cases: ${runCases.length}`);
lines.push('');
lines.push('## Aggregate');
lines.push('');
lines.push(
`| variant | steps | ${FLAG_TYPES.join(' | ')} | distinct openers | top opener | avg words | mean ms | cost |`,
);
lines.push(`|---|---:|${FLAG_TYPES.map(() => '---:').join('|')}|---:|---|---:|---:|---:|`);
for (const agg of aggregates) {
lines.push(
`| ${agg.variant} | ${agg.steps}${agg.errors ? ` (+${agg.errors} err)` : ''} | ` +
FLAG_TYPES.map((type) => agg.flagCounts[type] ?? 0).join(' | ') +
` | ${agg.distinctOpeners} | ${agg.topOpener} | ${agg.avgWords} | ${agg.meanLatencyMs} | $${agg.costUsd} |`,
);
}
lines.push('');
lines.push('## Per-case');
for (const testCase of runCases) {
lines.push('');
lines.push(`### ${testCase.id}`);
lines.push('');
lines.push(`*${testCase.notes}*`);
lines.push('');
const sampleList = [...new Set(records.map((r) => r.sample))].sort();
const header = ['step'];
if (samples > 1) {
header.push('s');
}
if (testCase.steps.some((step) => step.productionLabel)) {
header.push('production');
}
header.push(...variantNames);
lines.push(`| ${header.join(' | ')} |`);
lines.push(`|${header.map(() => '---').join('|')}|`);
for (const step of testCase.steps) {
const stepId = step.id ?? testCase.id;
for (const sample of sampleList) {
const row = [stepId];
if (samples > 1) {
row.push(String(sample));
}
if (header.includes('production')) {
row.push(step.productionLabel ?? '');
}
for (const variantName of variantNames) {
const record = records.find(
(r) =>
r.variant === variantName &&
r.sample === sample &&
r.caseId === testCase.id &&
r.stepId === stepId,
);
if (!record) {
row.push('');
} else if (record.error) {
row.push(`${record.error}`);
} else {
const flagNote = record.flags.length > 0 ? `${record.flags.join(' ⚠')}` : '';
row.push(`${record.label}${flagNote}`);
}
}
lines.push(`| ${row.map((cell) => cell.replace(/\|/g, '\\|')).join(' | ')} |`);
}
}
}
return lines.join('\n') + '\n';
}
module.exports = { aggregate, markdownReport, FLAG_TYPES };