{ "schema_version": 2, "kind": "shortening-method", "format": "agent-skill", "id": "test-failure", "name": "Test failure", "category": "Technical", "summary": "Extract the failing assertion, environment, and useful diagnostic evidence.", "use_cases": [ "CI failure summaries", "Test triage notes" ], "word_count": 155, "url": "https://sho.rten.it/methods/test-failure/", "instructions_url": "https://sho.rten.it/methods/test-failure/SKILL.md", "skill_url": "https://sho.rten.it/methods/test-failure/SKILL.md", "json_url": "https://sho.rten.it/methods/test-failure/llms.txt", "plain_text_url": "https://sho.rten.it/methods/test-failure/prompt.txt", "license": "MIT", "sources_url": "https://sho.rten.it/sources/#test-failure", "skill_name": "test-failure", "skill_description": "Extract the failing assertion, environment, and useful diagnostic evidence. Use for CI failure summaries, Test triage notes.", "agents_md_url": "https://sho.rten.it/methods/test-failure/AGENTS.md", "sources": [], "instructions": "Condense supplied test output into a diagnostic brief for the person investigating the failure. Focus on an individual test or a small related failure group, not the full build history.\n\nName the failing test and the exact assertion or exception. Show expected and actual values side by side when available. Retain the file and line, relevant environment, seed, changed configuration, and reproducibility evidence. Preserve the first useful application frame and any setup failure that prevents the test from exercising its target.\n\nRemove passing-test output, progress indicators, repeated framework frames, and duplicate failures. Report a group count only if the input supports it. Keep flaky behavior, timeouts, and missing evidence distinct from deterministic product failures.\n\nOutput Test, Failure, and Evidence lines, plus a supplied reproduction command if present. Do not infer root cause from the test name, rewrite assertion values, or claim a rerun passed. Do not run tests or propose an unverified fix as fact.", "example": { "context": "Cache expiry CI failure", "before": "The Linux CI job on Python 3.12 ran 218 tests. There were 217 passes and one failure. The failing test was tests/test_cache.py::test_expires_at_boundary at tests/test_cache.py:88. The assertion expected get(\"session\", now=60) to be None, but the actual value was \"active\". This run used CACHE_TTL_SECONDS=60 and seed 481. A local rerun with the same seed passed, so the failure is not yet reproduced locally. The output also contained the progress line for every passing test and 12 framework stack frames, none of which added another error.", "after": "Test: tests/test_cache.py::test_expires_at_boundary (tests/test_cache.py:88).\nFailure: get(\"session\", now=60) expected None; got \"active\".\nEvidence: Linux, Python 3.12, CACHE_TTL_SECONDS=60, seed 481. CI: 1 failed, 217 passed. Local rerun with the same seed passed; failure not yet reproduced locally.", "must_preserve": [ "Exact test identifier and location", "get(\"session\", now=60) expected None but got \"active\"", "Linux; Python 3.12; CACHE_TTL_SECONDS=60; seed 481", "1 failed and 217 passed out of 218", "Same-seed local rerun passed; not reproduced locally" ], "omitted": [ "Passing progress lines and uninformative framework frames" ], "kind": "illustrative" } }