{"enrichment":{"capability":"SWE-bench is a benchmark for evaluating language models on real-world GitHub software issues, where models generate patches to resolve described problems in codebases.","verdict":"Actively maintained, well-resourced benchmark with no known vulnerabilities and permissive licensing. Suitable for evaluating LM code-generation capabilities, but evaluation is resource-intensive and requires Docker infrastructure setup."},"id":"swebench","links":{"html":"https://skillfed.io/packages/swebench","md":"https://skillfed.io/packages/swebench.md","pypi":"https://pypi.org/project/swebench/"},"maintenance":{"status":"active"},"meta":{"latest_release":"2025-09-11","license_spdx":null,"license_treatment":"permissive","name":"swebench","python_support":"supports_current","summary":"The official SWE-bench package - a benchmark for evaluating LMs on software engineering"},"popularity":{"tier":"top_1000"},"security":{"n_vulnerabilities":0},"version":"4.1.0"}
