{"id":856,"slug":"mlcommons--peoples_speech","name":"peoples_speech","author":"MLCommons","description":"\n\t\n\t\t\n\t\tDataset Card for People's Speech\n\t\n\n\n\t\n\t\t\n\t\tDataset Summary\n\t\n\nThe People's Speech Dataset is among the world's largest English speech recognition corpus today that is licensed for academic and commercial usage under CC-BY-SA and CC-BY 4.0. It includes 30,000+ hours of transcribed speech in English languages with a diverse set of speakers. This open dataset is large enough to train speech-to-text systems and crucially is available with a permissive license.\n\n\t\n\t\t\n\t\n\t\n\t\tSupported Tasks… See the full description on the dataset page: https://huggingface.co/datasets/MLCommons/peoples_speech.","tags":"[\"Task_categories:automatic-Speech-Recognition\",\"Annotations_creators:crowdsourced\",\"Annotations_creators:machine-Generated\",\"Language_creators:crowdsourced\",\"Language_creators:machine-Generated\",\"Multilinguality:monolingual\"]","license":null,"framework":null,"parameters":null,"downloads":75158,"likes":284,"verified":0,"created_at":"2026-08-30 09:23:30","updated_at":"2026-08-30 09:23:30","source_url":"https://huggingface.co/datasets/MLCommons/peoples_speech","source_platform":"huggingface","hf_repo_id":"MLCommons/peoples_speech","ollama_name":"","category":"dataset","latest_version":"v1.0.0","version_count":1,"signature_count":1,"risk_level":null,"risk_score":null,"versions":[{"id":855,"model_id":856,"version":"v1.0.0","manifest_hash":"c6e7c3e4f6e2dff8441650a22e68697e85e936f97e42634e676bd19530460987","file_count":0,"total_size":0,"r2_manifest_key":"manifests/datasets/mlcommons--peoples_speech/v1.0.0.json","created_at":"2026-08-30 09:23:30"}],"files":[],"signatures":[{"id":1428,"version_id":855,"signer_did":"did:quantamrkt:registry:shield-v1","algorithm":"ML-DSA-65","signature_hex":"b80a205535869d26098262c342e8bee3316ec4b905e4b3dcc65c3a64624549ae","attestation_type":"registry","signed_at":"2026-08-30 09:23:30"}],"hndl":null}