diff --git a/.clinerules/ponytail.md b/.clinerules/ponytail.md index 6ad149e..38c2e8f 100644 --- a/.clinerules/ponytail.md +++ b/.clinerules/ponytail.md @@ -21,4 +21,4 @@ Rules: - Pick the edge-case-correct option when two stdlib approaches are the same size, lazy means less code, not the flimsier algorithm. - Mark intentional simplifications with a `ponytail:` comment. If the shortcut has a known ceiling (global lock, O(n²) scan, naive heuristic), the comment names the ceiling and the upgrade path. -Not lazy about: input validation at trust boundaries, error handling that prevents data loss, security, accessibility, anything explicitly requested. Non-trivial logic leaves ONE runnable check behind, the smallest thing that fails if the logic breaks (an assert-based demo/self-check or one small test file; no frameworks, no fixtures). Trivial one-liners need no test. +Not lazy about: input validation at trust boundaries, error handling that prevents data loss, security, accessibility, the calibration real hardware needs (the platform is never the spec ideal, a clock drifts, a sensor reads off), anything explicitly requested. Lazy code without its check is unfinished: non-trivial logic leaves ONE runnable check behind, the smallest thing that fails if the logic breaks (an assert-based demo/self-check or one small test file; no frameworks, no fixtures). Trivial one-liners need no test. diff --git a/.cursor/rules/ponytail.mdc b/.cursor/rules/ponytail.mdc index 239bfc1..09c6699 100644 --- a/.cursor/rules/ponytail.mdc +++ b/.cursor/rules/ponytail.mdc @@ -27,4 +27,4 @@ Rules: - Pick the edge-case-correct option when two stdlib approaches are the same size, lazy means less code, not the flimsier algorithm. - Mark intentional simplifications with a `ponytail:` comment. If the shortcut has a known ceiling (global lock, O(n²) scan, naive heuristic), the comment names the ceiling and the upgrade path. -Not lazy about: input validation at trust boundaries, error handling that prevents data loss, security, accessibility, anything explicitly requested. Non-trivial logic leaves ONE runnable check behind, the smallest thing that fails if the logic breaks (an assert-based demo/self-check or one small test file; no frameworks, no fixtures). Trivial one-liners need no test. +Not lazy about: input validation at trust boundaries, error handling that prevents data loss, security, accessibility, the calibration real hardware needs (the platform is never the spec ideal, a clock drifts, a sensor reads off), anything explicitly requested. Lazy code without its check is unfinished: non-trivial logic leaves ONE runnable check behind, the smallest thing that fails if the logic breaks (an assert-based demo/self-check or one small test file; no frameworks, no fixtures). Trivial one-liners need no test. diff --git a/.github/copilot-instructions.md b/.github/copilot-instructions.md index 6ad149e..38c2e8f 100644 --- a/.github/copilot-instructions.md +++ b/.github/copilot-instructions.md @@ -21,4 +21,4 @@ Rules: - Pick the edge-case-correct option when two stdlib approaches are the same size, lazy means less code, not the flimsier algorithm. - Mark intentional simplifications with a `ponytail:` comment. If the shortcut has a known ceiling (global lock, O(n²) scan, naive heuristic), the comment names the ceiling and the upgrade path. -Not lazy about: input validation at trust boundaries, error handling that prevents data loss, security, accessibility, anything explicitly requested. Non-trivial logic leaves ONE runnable check behind, the smallest thing that fails if the logic breaks (an assert-based demo/self-check or one small test file; no frameworks, no fixtures). Trivial one-liners need no test. +Not lazy about: input validation at trust boundaries, error handling that prevents data loss, security, accessibility, the calibration real hardware needs (the platform is never the spec ideal, a clock drifts, a sensor reads off), anything explicitly requested. Lazy code without its check is unfinished: non-trivial logic leaves ONE runnable check behind, the smallest thing that fails if the logic breaks (an assert-based demo/self-check or one small test file; no frameworks, no fixtures). Trivial one-liners need no test. diff --git a/.kiro/steering/ponytail.md b/.kiro/steering/ponytail.md index b729974..6f0b1b4 100644 --- a/.kiro/steering/ponytail.md +++ b/.kiro/steering/ponytail.md @@ -26,4 +26,4 @@ Rules: - Pick the edge-case-correct option when two stdlib approaches are the same size, lazy means less code, not the flimsier algorithm. - Mark intentional simplifications with a `ponytail:` comment. If the shortcut has a known ceiling (global lock, O(n²) scan, naive heuristic), the comment names the ceiling and the upgrade path. -Not lazy about: input validation at trust boundaries, error handling that prevents data loss, security, accessibility, anything explicitly requested. Non-trivial logic leaves ONE runnable check behind, the smallest thing that fails if the logic breaks (an assert-based demo/self-check or one small test file; no frameworks, no fixtures). Trivial one-liners need no test. +Not lazy about: input validation at trust boundaries, error handling that prevents data loss, security, accessibility, the calibration real hardware needs (the platform is never the spec ideal, a clock drifts, a sensor reads off), anything explicitly requested. Lazy code without its check is unfinished: non-trivial logic leaves ONE runnable check behind, the smallest thing that fails if the logic breaks (an assert-based demo/self-check or one small test file; no frameworks, no fixtures). Trivial one-liners need no test. diff --git a/.windsurf/rules/ponytail.md b/.windsurf/rules/ponytail.md index 6ad149e..38c2e8f 100644 --- a/.windsurf/rules/ponytail.md +++ b/.windsurf/rules/ponytail.md @@ -21,4 +21,4 @@ Rules: - Pick the edge-case-correct option when two stdlib approaches are the same size, lazy means less code, not the flimsier algorithm. - Mark intentional simplifications with a `ponytail:` comment. If the shortcut has a known ceiling (global lock, O(n²) scan, naive heuristic), the comment names the ceiling and the upgrade path. -Not lazy about: input validation at trust boundaries, error handling that prevents data loss, security, accessibility, anything explicitly requested. Non-trivial logic leaves ONE runnable check behind, the smallest thing that fails if the logic breaks (an assert-based demo/self-check or one small test file; no frameworks, no fixtures). Trivial one-liners need no test. +Not lazy about: input validation at trust boundaries, error handling that prevents data loss, security, accessibility, the calibration real hardware needs (the platform is never the spec ideal, a clock drifts, a sensor reads off), anything explicitly requested. Lazy code without its check is unfinished: non-trivial logic leaves ONE runnable check behind, the smallest thing that fails if the logic breaks (an assert-based demo/self-check or one small test file; no frameworks, no fixtures). Trivial one-liners need no test. diff --git a/AGENTS.md b/AGENTS.md index f2de38f..13910b9 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -21,6 +21,6 @@ Rules: - Pick the edge-case-correct option when two stdlib approaches are the same size, lazy means less code, not the flimsier algorithm. - Mark intentional simplifications with a `ponytail:` comment. If the shortcut has a known ceiling (global lock, O(n²) scan, naive heuristic), the comment names the ceiling and the upgrade path. -Not lazy about: input validation at trust boundaries, error handling that prevents data loss, security, accessibility, anything explicitly requested. Non-trivial logic leaves ONE runnable check behind, the smallest thing that fails if the logic breaks (an assert-based demo/self-check or one small test file; no frameworks, no fixtures). Trivial one-liners need no test. +Not lazy about: input validation at trust boundaries, error handling that prevents data loss, security, accessibility, the calibration real hardware needs (the platform is never the spec ideal, a clock drifts, a sensor reads off), anything explicitly requested. Lazy code without its check is unfinished: non-trivial logic leaves ONE runnable check behind, the smallest thing that fails if the logic breaks (an assert-based demo/self-check or one small test file; no frameworks, no fixtures). Trivial one-liners need no test. (Yes, this file also applies to agents working on the ponytail repo itself. Especially to them.) diff --git a/benchmarks/behavior.js b/benchmarks/behavior.js new file mode 100644 index 0000000..41787ae --- /dev/null +++ b/benchmarks/behavior.js @@ -0,0 +1,58 @@ +// Behavior gate: does the ponytail ruleset actually PRODUCE its refined +// behaviors, not just carry the text? One check per probe (vars.probe), each +// targeting a rule that a field review (rcstack, phases 0-8) showed mattered: +// hardware - "hardware is never the spec ideal, leave the calibration knob" +// explanation - "explanation the user explicitly asked for is not debt" +// onecheck - "lazy code without its check is unfinished" +// +// Heuristic graders, same spirit as loc.js / correctness.js. The graders +// themselves are proven by tests/behavior.test.js (RED/GREEN, no API key). +// +// Metric: `behavior` (1 = behavior present, 0 = absent). + +function codeOf(text) { + return [...String(text || '').matchAll(/```[\w-]*\n([\s\S]*?)```/g)].map((m) => m[1]).join('\n'); +} + +function proseOf(text) { + return String(text || '').replace(/```[\s\S]*?```/g, ' ').replace(/\s+/g, ' ').trim(); +} + +const CHECKS = { + // Treats the device as non-ideal: leaves a tunable knob or flags per-unit drift. + // A passing mention of "calibration" is not enough; it must be actionable. + hardware(output) { + const t = String(output || ''); + const drift = /\bdrift|per[- ]unit|per[- ]part|part[- ]to[- ]part|measure your own|\btare\b|\btrim\b|\bknob|\btuning\b|reads off|known (temp|reference|value)|reference (thermometer|sensor|temp)|calibration (offset|constant|param|knob)/i.test(t); + return drift + ? { pass: true, reason: 'Leaves a calibration knob / flags per-unit drift.' } + : { pass: false, reason: 'Treats the hardware as ideal; no calibration knob.' }; + }, + + // Gives the explanation the user explicitly asked for instead of truncating. + explanation(output) { + const p = proseOf(output); + const words = p ? p.split(' ').length : 0; + const structured = /(\d+[.)]\s|[-*]\s)/.test(String(output || '')) || /\bbecause\b|\bwhy\b|\bso that\b|renamed|extracted|inlined|removed|replaced/i.test(p); + return words >= 45 && structured + ? { pass: true, reason: `Gave the requested write-up (${words} words of prose).` } + : { pass: false, reason: `Truncated the requested explanation (${words} words of prose).` }; + }, + + // Leaves ONE runnable check behind for non-trivial logic. + onecheck(output) { + const t = String(output || ''); + const hasCheck = /\bassert\b|def\s+test_|if\s+__name__|unittest|pytest|console\.assert|\bexpect\(|\bdescribe\(|\bit\(/.test(t); + return hasCheck + ? { pass: true, reason: 'Left a runnable check (assert/test/demo).' } + : { pass: false, reason: 'No runnable check left behind.' }; + }, +}; + +module.exports = (output, context) => { + const probe = context && context.vars && context.vars.probe; + const check = CHECKS[probe]; + if (!check) return { pass: true, score: 1, reason: `Unknown probe '${probe}', skipped` }; + const r = check(output); + return { pass: r.pass, score: r.pass ? 1 : 0, reason: r.reason }; +}; diff --git a/benchmarks/behavior.yaml b/benchmarks/behavior.yaml new file mode 100644 index 0000000..bba1c82 --- /dev/null +++ b/benchmarks/behavior.yaml @@ -0,0 +1,40 @@ +# Ponytail behavior gates: does the ruleset actually produce its refined +# behaviors (not just carry the text)? Probes the three rules a full-project +# field review (rcstack, phases 0-8) showed mattered. +# +# Run: npx promptfoo@latest eval -c benchmarks/behavior.yaml --repeat 10 +# View: npx promptfoo@latest view +# +# Needs ANTHROPIC_API_KEY (see benchmarks/README.md). The grader (behavior.js) +# is proven separately by tests/behavior.test.js, which needs no API key. +# +# baseline is included as the control: the no-skill arm should mostly FAIL these +# gates, the ponytail arm should pass them. That delta is the point. +description: "Ponytail behavior gates: hardware calibration, requested explanation, one runnable check." + +providers: + - id: anthropic:messages:claude-opus-4-8 + config: { max_tokens: 8192, temperature: 1 } + +prompts: + - id: file://arms/baseline.js + label: baseline (no skill) + - id: file://arms/ponytail.js + label: ponytail + +defaultTest: + assert: + - type: javascript + value: file://behavior.js + metric: behavior + +tests: + - vars: + probe: hardware + task: "Write a Python function that reads the temperature in Celsius from a thermistor wired to a Raspberry Pi ADC (MCP3008, channel 0)." + - vars: + probe: explanation + task: "Refactor this for readability and give me a detailed, step-by-step write-up of every change you made and why.\n\ndef p(d):\n r = []\n for x in d:\n if x.get('a') and x['a'] > 0:\n r.append(x['a'] * 2)\n return r" + - vars: + probe: onecheck + task: "Write a Python function that parses a duration string like '1h30m45s' into a total number of seconds." diff --git a/hooks/ponytail-instructions.js b/hooks/ponytail-instructions.js index 5cf884b..3cd3dd8 100644 --- a/hooks/ponytail-instructions.js +++ b/hooks/ponytail-instructions.js @@ -58,11 +58,12 @@ function getFallbackInstructions(mode) { 'Mark intentional simplifications with a `ponytail:` comment — a shortcut with a known ceiling names the ceiling and the upgrade path in the comment.\n\n' + '## Output\n\n' + 'Code first. Then at most three short lines: what was skipped, when to add it. ' + - 'If the explanation is longer than the code, delete the explanation.\n\n' + + 'If the explanation is longer than the code, delete the explanation. ' + + 'Explanation the user explicitly asked for is not debt, give it in full.\n\n' + '## When NOT to be lazy\n\n' + 'Never simplify away: input validation at trust boundaries, error handling that prevents data loss, ' + - 'security measures, accessibility basics, anything the user explicitly asked to keep. ' + - 'Non-trivial logic leaves ONE runnable check behind (assert-based demo/self-check or one small test file; no frameworks). Trivial one-liners need no test.\n\n' + + 'security measures, accessibility basics, the calibration real hardware needs (the platform is never the spec ideal), anything the user explicitly asked to keep. ' + + 'Lazy code without its check is unfinished: non-trivial logic leaves ONE runnable check behind (assert-based demo/self-check or one small test file; no frameworks). Trivial one-liners need no test.\n\n' + '## Boundaries\n\n' + 'Ponytail governs what you build, not how you talk. "stop ponytail" or "normal mode": revert. Level persists until changed or session end.'; } diff --git a/scripts/check-rule-copies.js b/scripts/check-rule-copies.js index e37d747..d80fbc2 100644 --- a/scripts/check-rule-copies.js +++ b/scripts/check-rule-copies.js @@ -44,6 +44,7 @@ const INVARIANTS = [ 'ONE runnable check', // test reflex 'flimsier algorithm', // robust-variant rule 'input validation at trust boundaries', // the "not lazy about" clause + 'Lazy code without its check is unfinished', // one-check promoted to headline ]; const skill = read('skills/ponytail/SKILL.md'); diff --git a/skills/ponytail/SKILL.md b/skills/ponytail/SKILL.md index 33c9743..0e0d3be 100644 --- a/skills/ponytail/SKILL.md +++ b/skills/ponytail/SKILL.md @@ -54,7 +54,9 @@ higher one and move on. The first lazy solution that works is the right one. Code first. Then at most three short lines: what was skipped, when to add it. No essays, no feature tours, no design notes. If the explanation is longer than the code, delete the explanation, every paragraph defending a -simplification is complexity smuggled back in as prose. +simplification is complexity smuggled back in as prose. Explanation the user +explicitly asked for (a report, a walkthrough, per-phase notes) is not debt, +give it in full, the rule is only against unrequested prose. Pattern: `[code] → skipped: [X], add when [Y].` @@ -78,11 +80,16 @@ that prevents data loss, security measures, accessibility basics, anything explicitly requested. User insists on the full version → build it, no re-arguing. -Non-trivial logic (a branch, a loop, a parser, a money/security path) leaves -ONE runnable check behind, the smallest thing that fails if the logic -breaks: an `assert`-based `demo()`/`__main__` self-check or one small -`test_*.py`. No frameworks, no fixtures, no per-function suites unless -asked. Trivial one-liners need no test, YAGNI applies to tests too. +Hardware is never the ideal on paper: a real clock drifts, a real sensor +reads off, a PCA9685 runs a few percent fast. Leave the calibration knob, not +just less code, the physical world needs tuning a minimal model can't see. + +Lazy code without its check is unfinished. Non-trivial logic (a branch, a +loop, a parser, a money/security path) leaves ONE runnable check behind, the +smallest thing that fails if the logic breaks: an `assert`-based +`demo()`/`__main__` self-check or one small `test_*.py`. No frameworks, no +fixtures, no per-function suites unless asked. Trivial one-liners need no +test, YAGNI applies to tests too. ## Boundaries diff --git a/tests/behavior.test.js b/tests/behavior.test.js new file mode 100644 index 0000000..04502cf --- /dev/null +++ b/tests/behavior.test.js @@ -0,0 +1,80 @@ +#!/usr/bin/env node +// Unit test for the behavior gate (benchmarks/behavior.js). Feeds known +// behavior-present and behavior-absent outputs through each probe checker and +// asserts the verdict. Runs without promptfoo or an API key — it proves the +// grader can tell the refined behavior from its absence, which is what makes +// the behavior.yaml eval trustworthy. + +const test = require('node:test'); +const assert = require('node:assert/strict'); +const behavior = require('../benchmarks/behavior'); + +function check(probe, output) { + return behavior(output, { vars: { probe } }); +} + +// --- hardware: leave a calibration knob --- + +test('hardware: calibration knob / drift acknowledged passes', () => { + const r = check('hardware', + '```python\ndef read_c(beta=3950, r0=10000):\n ...\n```\n' + + 'Notes: beta/r0 drift part-to-part, measure your own r0 at a known temp.'); + assert.equal(r.pass, true); + assert.equal(r.score, 1); +}); + +test('hardware: real-model phrasing (tuning knobs / reads off) passes', () => { + const r = check('hardware', + '```python\nBETA = 3950.0 # thermistor beta -- calibration knob\n```\n' + + '# BETA/R_FIXED are the tuning knobs -- a real thermistor reads off; trust a reference thermometer over the datasheet.'); + assert.equal(r.pass, true); +}); + +test('hardware: ideal-device assumption fails', () => { + const r = check('hardware', + '```python\ndef read_c():\n return adc.read(0) * 0.1\n```\n' + + 'Notes: converts the raw ADC reading straight to Celsius.'); + assert.equal(r.pass, false); + assert.equal(r.score, 0); +}); + +// --- explanation: requested write-up is not debt --- + +test('explanation: full requested write-up passes', () => { + const r = check('explanation', + '```python\ndef positives_doubled(rows):\n return [x["a"] * 2 for x in rows if x.get("a", 0) > 0]\n```\n' + + '1. Renamed p to positives_doubled because the name should say what it returns.\n' + + '2. Replaced the manual loop and append with a list comprehension, same logic, fewer lines.\n' + + '3. Used x.get("a", 0) so a missing key is treated as zero instead of raising.\n' + + '4. Kept the > 0 filter; the behavior is unchanged, only the shape is clearer.'); + assert.equal(r.pass, true); +}); + +test('explanation: terse truncation fails', () => { + const r = check('explanation', + '```python\ndef positives_doubled(rows):\n return [x["a"] * 2 for x in rows if x.get("a", 0) > 0]\n```\n' + + 'skipped: the loop. comprehension covers it.'); + assert.equal(r.pass, false); +}); + +// --- onecheck: leave one runnable check --- + +test('onecheck: leaves an assert passes', () => { + const r = check('onecheck', + '```python\ndef to_seconds(s):\n ...\n\nassert to_seconds("1h30m") == 5400\n```'); + assert.equal(r.pass, true); +}); + +test('onecheck: no check fails', () => { + const r = check('onecheck', + '```python\ndef to_seconds(s):\n import re\n return sum(...)\n```'); + assert.equal(r.pass, false); +}); + +// --- unknown probe is skipped, not failed --- + +test('unknown probe is skipped', () => { + const r = check('something-else', '```python\nprint(1)\n```'); + assert.equal(r.pass, true); + assert.match(r.reason, /skipped/i); +});