{"id":790,"slug":"programbench--programbench-tests","name":"ProgramBench-Tests","author":"programbench","description":"\n\t\n\t\t\n\t\tProgramBench Generated Tests\n\t\n\nThis dataset contains the AI-generated behavioral test suites used to evaluate model solutions in ProgramBench.\nProgramBench is a benchmark that evaluates whether language models can rebuild programs from scratch. Given only a compiled binary and its documentation, AI agents must architect and implement a complete codebase that reproduces the original program's behavior. These test suites are used to assess whether a candidate solution is behaviorally… See the full description on the dataset page: https://huggingface.co/datasets/programbench/ProgramBench-Tests.","tags":"[\"Task_categories:text-Generation\",\"Language:en\",\"Size_categories:n<1K\",\"Code\",\"Software-Engineering\",\"Reverse-Engineering\"]","license":null,"framework":null,"parameters":null,"downloads":130006,"likes":10,"verified":0,"created_at":"2026-08-07 11:23:38","updated_at":"2026-08-07 14:23:28","source_url":"https://huggingface.co/datasets/programbench/ProgramBench-Tests","source_platform":"huggingface","hf_repo_id":"programbench/ProgramBench-Tests","ollama_name":"","category":"dataset","latest_version":"v1.0.0","version_count":1,"signature_count":1,"risk_level":null,"risk_score":null,"versions":[{"id":789,"model_id":790,"version":"v1.0.0","manifest_hash":"26650a38e4180c56e11b53ffe02abcfdda0a140aca5d0b1822a3d65bc0425909","file_count":0,"total_size":0,"r2_manifest_key":"manifests/datasets/programbench--programbench-tests/v1.0.0.json","created_at":"2026-08-07 11:23:38"}],"files":[],"signatures":[{"id":1362,"version_id":789,"signer_did":"did:quantamrkt:registry:shield-v1","algorithm":"ML-DSA-65","signature_hex":"3f46893cf4aa6aec214671bbc5709bd1d7da3faf0d9fd2e92472f4c34aa3a296","attestation_type":"registry","signed_at":"2026-08-07 11:23:38"}],"hndl":null}