Commit a promptfoo harness (config + arm prompts + LOC metric + vendored caveman SKILL) so anyone can re-run the comparison: no-skill vs caveman vs ponytail, across Haiku / Sonnet / Opus, 10 runs per cell, median reported. Replace the old unreproducible 6-task chart with assets/benchmark-3model.svg from this run, and reframe the README to the reproducible numbers: ponytail writes 80-94% less code, costs 47-77% less, and runs 3-6x faster than a no-skill agent on every model. benchmarks/README.md carries the median tables and the reproduce command. Drops nothing that is not measured. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
13 lines
643 B
JavaScript
13 lines
643 B
JavaScript
// Deterministic code-size metric: non-blank, non-comment lines inside fenced code blocks.
|
|
// Recorded as the `code_loc` metric per arm (always passes; it is a measurement, not a gate).
|
|
module.exports = (output) => {
|
|
const text = String(output || '');
|
|
const blocks = [...text.matchAll(/```[a-zA-Z0-9_+-]*\n([\s\S]*?)```/g)].map((m) => m[1]);
|
|
const code = blocks.join('\n');
|
|
const loc = code
|
|
.split('\n')
|
|
.map((l) => l.trim())
|
|
.filter((l) => l && !l.startsWith('//') && !l.startsWith('#') && l !== '*/' && !l.startsWith('/*') && !l.startsWith('*')).length;
|
|
return { pass: true, score: loc, reason: loc + ' code LOC' };
|
|
};
|