From 386f95734a17e6b22d22804cf42341c94046d82d Mon Sep 17 00:00:00 2001 From: Mandavilli Vijay Date: Mon, 15 Jun 2026 18:57:11 +0530 Subject: [PATCH] benchmarks: add local model support and Node version note (#63) Adds benchmarks/benchmark-local.py (Ollama-based local runner), a results writeup, and a Node version note. Thanks @mandavillivijay. --- benchmarks/README.md | 18 ++ benchmarks/benchmark-local.py | 155 ++++++++++++++++++ .../results/2026-06-15-llama3.2-local.md | 68 ++++++++ 3 files changed, 241 insertions(+) create mode 100644 benchmarks/benchmark-local.py create mode 100644 benchmarks/results/2026-06-15-llama3.2-local.md diff --git a/benchmarks/README.md b/benchmarks/README.md index 80e7698..a53c9df 100644 --- a/benchmarks/README.md +++ b/benchmarks/README.md @@ -4,12 +4,30 @@ Three arms (no skill, [caveman](https://github.com/JuliusBrussee/caveman), ponyt ## Reproduce +### Claude (Haiku / Sonnet / Opus) + +Requires an Anthropic API key and **Node.js ≥ 22.22.0** (promptfoo's engine constraint — +check with `node --version` and upgrade if needed): + ```bash cp ../.env.example ../.env # add your ANTHROPIC_API_KEY npx promptfoo@latest eval -c promptfooconfig.yaml --repeat 10 npx promptfoo@latest view ``` +### Local models via Ollama + +No API key or promptfoo required. Runs against any model served by Ollama: + +```bash +ollama pull llama3.2 # or any other model +python benchmarks/benchmark-local.py --model llama3.2 --repeat 3 +``` + +See `benchmarks/results/2026-06-15-llama3.2-local.md` for what to expect: the skill works +well on instruction-following models (Claude-class) but transfers poorly to small local +models where the multi-step decision ladder isn't reliably followed. + Tasks: email validator, JS debounce, CSV sum, React countdown, FastAPI rate-limit (see `promptfooconfig.yaml`). Single-shot completions, default temperature. ## Median results (10 runs, 2026-06-13) diff --git a/benchmarks/benchmark-local.py b/benchmarks/benchmark-local.py new file mode 100644 index 0000000..c3520ec --- /dev/null +++ b/benchmarks/benchmark-local.py @@ -0,0 +1,155 @@ +""" +Ponytail local benchmark — runs the same 5 tasks against any Ollama model. +No promptfoo required. Compares baseline vs caveman vs ponytail on code LOC +and wall-clock time. Results are printed as a table and saved to a JSON file. + +Usage: + python benchmarks/benchmark-local.py + python benchmarks/benchmark-local.py --model llama3.2 --repeat 3 + +Prerequisites: Ollama running locally (https://ollama.com), model pulled. +""" + +import argparse +import json +import re +import time +import urllib.request +from pathlib import Path + +ROOT = Path(__file__).parent.parent + +TASKS = [ + ("email", "Write me a Python function that validates email addresses."), + ("debounce", "Add debounce to a search input in vanilla JavaScript. It currently fires an API call on every keystroke."), + ("csv-sum", "Write Python code that reads sales.csv and sums the 'amount' column."), + ("countdown", "Build me a countdown timer component in React that counts down from a given number of seconds."), + ("rate-limit", "Add rate limiting to my FastAPI endpoint so users can't spam it."), +] + + +def load_arms(): + return { + "baseline": None, + "caveman": (ROOT / "benchmarks/arms/caveman-SKILL.md").read_text(encoding="utf-8"), + "ponytail": (ROOT / "skills/ponytail/SKILL.md").read_text(encoding="utf-8"), + } + + +def count_loc(text): + """Non-blank, non-comment lines inside fenced code blocks.""" + blocks = re.findall(r"```[a-zA-Z0-9_+\-]*\n([\s\S]*?)```", text) + lines = "\n".join(blocks).splitlines() + return sum( + 1 for l in lines + if l.strip() + and not l.strip().startswith("//") + and not l.strip().startswith("#") + and l.strip() not in ("*/",) + and not l.strip().startswith("/*") + and not l.strip().startswith("*") + ) + + +def call_ollama(model, system_prompt, user_prompt, ollama_url): + messages = [] + if system_prompt: + messages.append({"role": "system", "content": system_prompt}) + messages.append({"role": "user", "content": user_prompt}) + + payload = json.dumps({ + "model": model, + "messages": messages, + "stream": False, + "options": {"temperature": 0.7}, + }).encode() + + req = urllib.request.Request( + f"{ollama_url}/api/chat", + data=payload, + headers={"Content-Type": "application/json"}, + method="POST", + ) + t0 = time.time() + with urllib.request.urlopen(req, timeout=180) as resp: + data = json.loads(resp.read()) + elapsed = time.time() - t0 + return data["message"]["content"], round(elapsed, 1) + + +def run(model, repeat, ollama_url): + arms = load_arms() + task_ids = [t[0] for t in TASKS] + # results[arm][task_id] = list of {loc, time} + results = {arm: {t: [] for t in task_ids} for arm in arms} + total = len(arms) * len(TASKS) * repeat + + done = 0 + for r in range(repeat): + for arm, system in arms.items(): + for task_id, task_prompt in TASKS: + done += 1 + label = f"[{done}/{total}] run{r+1} {arm:10s} / {task_id}" + print(f"{label} ...", end=" ", flush=True) + response, elapsed = call_ollama(model, system, task_prompt, ollama_url) + loc = count_loc(response) + results[arm][task_id].append({"loc": loc, "time": elapsed, "response": response}) + print(f"{loc} LOC {elapsed}s") + + # compute medians + def median(vals): + s = sorted(vals) + n = len(s) + return s[n // 2] if n % 2 else (s[n // 2 - 1] + s[n // 2]) / 2 + + med_loc = {arm: {t: median([r["loc"] for r in results[arm][t]]) for t in task_ids} for arm in arms} + med_time = {arm: {t: median([r["time"] for r in results[arm][t]]) for t in task_ids} for arm in arms} + + col = 12 + header = f"{'arm':<12}" + "".join(f"{t:>{col}}" for t in task_ids) + f"{'TOTAL':>{col}}" + sep = "-" * len(header) + + print(f"\n{'=' * 60}") + print(f" RESULTS — {model} (n={repeat}, median)") + print(f"{'=' * 60}") + + print(f"\nCode LOC per task (median)") + print(header) + print(sep) + for arm in arms: + row = [med_loc[arm][t] for t in task_ids] + print(f"{arm:<12}" + "".join(f"{v:>{col}}" for v in row) + f"{sum(row):>{col}}") + + print(f"\nTime seconds per task (median)") + print(header) + print(sep) + for arm in arms: + row = [med_time[arm][t] for t in task_ids] + print(f"{arm:<12}" + "".join(f"{v:>{col}.1f}" for v in row) + f"{sum(row):>{col}.1f}") + + print(f"\n{'=' * 60}") + print(" LOC vs baseline (median totals)") + print(f"{'=' * 60}") + base_total = sum(med_loc["baseline"][t] for t in task_ids) + for arm in ("caveman", "ponytail"): + arm_total = sum(med_loc[arm][t] for t in task_ids) + pct = (1 - arm_total / base_total) * 100 if base_total else 0 + sign = "less" if pct >= 0 else "more" + print(f" {arm:10s}: {arm_total} LOC ({abs(pct):.0f}% {sign} than baseline)") + + out = Path(__file__).parent / "benchmark-local-results.json" + out.write_text(json.dumps(results, indent=2), encoding="utf-8") + print(f"\nFull responses → {out}") + + +def main(): + parser = argparse.ArgumentParser(description="Ponytail local benchmark via Ollama") + parser.add_argument("--model", default="llama3.2", help="Ollama model name (default: llama3.2)") + parser.add_argument("--repeat", type=int, default=1, help="Runs per cell; median reported (default: 1)") + parser.add_argument("--ollama-url", default="http://localhost:11434", help="Ollama base URL") + args = parser.parse_args() + run(args.model, args.repeat, args.ollama_url) + + +if __name__ == "__main__": + main() diff --git a/benchmarks/results/2026-06-15-llama3.2-local.md b/benchmarks/results/2026-06-15-llama3.2-local.md new file mode 100644 index 0000000..857a169 --- /dev/null +++ b/benchmarks/results/2026-06-15-llama3.2-local.md @@ -0,0 +1,68 @@ +# Local model benchmark: llama3.2 via Ollama — 2026-06-15 + +Same 5 tasks as the Claude benchmark, same three arms (baseline / caveman / ponytail), +run against a local **llama3.2:latest** (3.2B, Q4_K_M) via Ollama on a Windows 11 machine. +n=1 per cell. Tooling: `benchmarks/benchmark-local.py` (no promptfoo needed). + +## Results + +**Code LOC** + +| arm | email | debounce | csv-sum | countdown | rate-limit | **TOTAL** | +|---|--:|--:|--:|--:|--:|--:| +| baseline | 13 | 13 | 38 | 44 | 32 | **140** | +| caveman | 16 | 12 | 5 | 45 | 28 | **106** | +| ponytail | 18 | 21 | 7 | 38 | 49 | **133** | + +**Time (seconds)** + +| arm | email | debounce | csv-sum | countdown | rate-limit | **TOTAL** | +|---|--:|--:|--:|--:|--:|--:| +| baseline | 52.6 | 37.2 | 63.4 | 65.2 | 71.0 | **289.4** | +| caveman | 79.6 | 54.4 | 28.3 | 71.4 | 61.0 | **294.7** | +| ponytail | 99.7 | 71.0 | 25.4 | 74.8 | 97.3 | **368.2** | + +**LOC vs baseline** + +| arm | total LOC | vs baseline | +|---|--:|---| +| caveman | 106 | −24% | +| ponytail | 133 | −5% | + +## Key findings + +**Ponytail does not transfer to llama3.2.** On 3 of 5 tasks (email: 13→18, debounce: 13→21, +rate-limit: 32→49) ponytail produced *more* code than the no-skill baseline. Total LOC +reduction was −5% vs the 80–94% seen on Claude. Response time increased by +27% (368s vs +289s) rather than the 3–6× speedup seen on Claude. + +**Caveman outperformed ponytail on this model** (−24% LOC, similar time to baseline). +Caveman's rules are simpler prose instructions that a small model can follow more reliably; +ponytail's multi-step decision ladder requires a stronger instruction-follower. + +**Why this happens:** Ponytail is a prompt-engineering skill calibrated on Claude models, +which are specifically trained to follow detailed system instructions. A 3.2B quantised model +partially absorbs the ponytail rules and then adds extra prose *justifying* its choices — +paying the complexity cost without getting the minimalism benefit. + +## Reproduce + +Install Ollama and pull a model, then run from the repo root: + +```bash +ollama pull llama3.2 +python benchmarks/benchmark-local.py --model llama3.2 +``` + +Optional flags: + +``` +--repeat N Runs per cell; median is reported (default: 1) +--ollama-url URL Ollama base URL (default: http://localhost:11434) +``` + +## Takeaway + +The benchmark claims in the README are accurate for the models tested (Haiku, Sonnet, Opus). +For local/small models, expect significantly smaller — or even negative — gains until +instruction-following capability reaches a threshold comparable to Claude Haiku or better.