[{"@type":"PropertyValue","name":"Format","value":"16 kHz, 16 bit, uncompressed wav, mono channel,speaker channel separation;"},{"@type":"PropertyValue","name":"Content category","value":"Dialogue based on given topics"},{"@type":"PropertyValue","name":"Recording condition","value":"Low background noise (indoor)"},{"@type":"PropertyValue","name":"Recording device","value":"Android smartphone, iPhone"},{"@type":"PropertyValue","name":"Country","value":"Republic of the Philippines(PHL)"},{"@type":"PropertyValue","name":"Language(Region) Code","value":"en-PH"},{"@type":"PropertyValue","name":"Language","value":"English"},{"@type":"PropertyValue","name":"Features of annotation","value":"Transcription text, timestamp, speaker ID, gender, noise"},{"@type":"PropertyValue","name":"Accuracy rate","value":"Word accuracy rate(WAR) 98%"}]
{"id":1771,"datatype":"1","titleimg":"https://www.nexdata.ai/shujutang/static/image/index/datatang_yuyin_default.webp","type1":"165","type1str":null,"type2":"166","type2str":null,"dataname":"600 Hours Philippine English Full-Duplex Speech Dataset for Voice Agent Training","datazy":[{"title":"Format","content":"16 kHz, 16 bit, uncompressed wav, mono channel,speaker channel separation;"},{"title":"Content category","content":"Dialogue based on given topics"},{"title":"Recording condition","content":"Low background noise (indoor)"},{"title":"Recording device","content":"Android smartphone, iPhone"},{"title":"Country","content":"Republic of the Philippines(PHL)"},{"title":"Language(Region) Code","content":"en-PH"},{"title":"Language","content":"English"},{"title":"Features of annotation","content":"Transcription text, timestamp, speaker ID, gender, noise"},{"title":"Accuracy rate","content":"Word accuracy rate(WAR) 98%"}],"datatag":"full-duplex,Dialogue","technologydoc":null,"downurl":null,"datainfo":null,"standard":null,"dataylurl":null,"flag":null,"publishtime":null,"createby":null,"createtime":null,"ext1":null,"samplestoreloc":null,"hosturl":null,"datasize":null,"industryPlan":null,"keyInformation":null,"samplePresentation":[],"officialSummary":"This dataset contains 600 hours of Philippine English conversational speech recordings collected from natural dialogues based on predefined topics.The dataset features full-duplex, multi-channel audio recordings. Each audio sample includes accurate transcription and metadata such as speaker ID, gender, age, and other attributes. Quality tested by various AI companies. We strictly adhere to data protection regulations and privacy standards, ensuring the maintenance of user privacy and legal rights throughout the data collection, storage, and usage processes, our datasets are all GDPR, CCPA, PIPL complied.","dataexampl":null,"datakeyword":["full duplex speech dataset","voice agent dataset","conversational AI training data","speech to speech dataset","multi channel speech dataset","Philippine English speech dataset"],"isDelete":null,"ids":null,"idsList":null,"datasetCode":null,"productStatus":null,"tagTypeEn":"Data Type,Language","tagTypeZh":null,"website":null,"samplePresentationList":null,"datazyList":null,"keyInformationList":null,"dataexamplList":null,"bgimg":null,"datazyScriptList":null,"datakeywordListString":null,"sourceShowPage":"speechRec","dataShowType":"[{\"code\":\"0\",\"language\":\"ZH\"},{\"code\":\"1\",\"language\":\"ZH\"},{\"code\":\"2\",\"language\":\"EN,PT,DE,KO,FR,ES,JP\"},{\"code\":\"3\",\"language\":\"EN\"}]","productNameEn":"423 Hours - English(Philippine) Full-Duplex Spontaneous Dialogue Smartphone speech dataset","BGimg":"brightSpot_audio","voiceBg":["/shujutang/static/image/comm/audio_bg.webp","/shujutang/static/image/comm/audio_bg2.webp","/shujutang/static/image/comm/audio_bg3.webp","/shujutang/static/image/comm/audio_bg4.webp","/shujutang/static/image/comm/audio_bg5.webp"]}
600 Hours Philippine English Full-Duplex Speech Dataset for Voice Agent Training
full duplex speech dataset
voice agent dataset
conversational AI training data
speech to speech dataset
multi channel speech dataset
Philippine English speech dataset
This dataset contains 600 hours of Philippine English conversational speech recordings collected from natural dialogues based on predefined topics.The dataset features full-duplex, multi-channel audio recordings. Each audio sample includes accurate transcription and metadata such as speaker ID, gender, age, and other attributes. Quality tested by various AI companies. We strictly adhere to data protection regulations and privacy standards, ensuring the maintenance of user privacy and legal rights throughout the data collection, storage, and usage processes, our datasets are all GDPR, CCPA, PIPL complied.
This is a paid datasets for commercial use, research purpose and more. Licensed ready made datasets help jump-start AI projects.