{"id":872,"slug":"facebook--voxpopuli","name":"voxpopuli","author":"facebook","description":"\n\t\n\t\t\n\t\n\t\n\t\tDataset Card for Voxpopuli\n\t\n\n\n\t\n\t\t\n\t\n\t\n\t\tDataset Summary\n\t\n\nVoxPopuli is a large-scale multilingual speech corpus for representation learning, semi-supervised learning and interpretation.\nThe raw data is collected from 2009-2020 European Parliament event recordings. We acknowledge the European Parliament for creating and sharing these materials.\nThis implementation contains transcribed speech data for 18 languages.\nIt also contains 29 hours of transcribed speech data of non-native… See the full description on the dataset page: https://huggingface.co/datasets/facebook/voxpopuli.","tags":"[\"Task_categories:automatic-Speech-Recognition\",\"Multilinguality:multilingual\",\"Language:en\",\"Language:de\",\"Language:fr\",\"Language:es\"]","license":null,"framework":null,"parameters":null,"downloads":83925,"likes":162,"verified":0,"created_at":"2026-09-04 09:23:58","updated_at":"2026-09-13 06:23:27","source_url":"https://huggingface.co/datasets/facebook/voxpopuli","source_platform":"huggingface","hf_repo_id":"facebook/voxpopuli","ollama_name":"","category":"dataset","latest_version":"v1.0.0","version_count":1,"signature_count":1,"risk_level":null,"risk_score":null,"versions":[{"id":871,"model_id":872,"version":"v1.0.0","manifest_hash":"3979fdca180ef1f6609ab5947119fbc060c51f64cc8193b475e42a249daa8787","file_count":0,"total_size":0,"r2_manifest_key":"manifests/datasets/facebook--voxpopuli/v1.0.0.json","created_at":"2026-09-04 09:23:58"}],"files":[],"signatures":[{"id":1444,"version_id":871,"signer_did":"did:quantamrkt:registry:shield-v1","algorithm":"ML-DSA-65","signature_hex":"e3b49848a8884afdb25f35cc227f9424b9a55819de2bab29fc5df5540411244e","attestation_type":"registry","signed_at":"2026-09-04 09:23:58"}],"hndl":null}