[{"@type":"PropertyValue","name":"Format","value":"16kHz, 16bit, uncompressed wav, mono channel;"},{"@type":"PropertyValue","name":"Recording condition","value":"Low background noise, without echo;"},{"@type":"PropertyValue","name":"Content category","value":"economy, entertainment, news, informal language, numbers, alphabet;"},{"@type":"PropertyValue","name":"Recording device","value":"Android smartphone: iPhone=3.2:1;"},{"@type":"PropertyValue","name":"Speaker","value":"291 people in total, from South Korea and North Korea; 45% male and 55% female;"},{"@type":"PropertyValue","name":"Country","value":"South Korea(KOR), North Korea(PRK);"},{"@type":"PropertyValue","name":"Language(Region) Code","value":"ko-KR, ko-KP;"},{"@type":"PropertyValue","name":"Language","value":"Korean;"},{"@type":"PropertyValue","name":"Features of annotation","value":"Transcription text, timestamp, 5 noise symbols, special identifiers;"},{"@type":"PropertyValue","name":"Accuracy Rate","value":"Sentence Accuracy Rate (SAR) 95% (noise symbols and special identifiers are excluded)"}]
{"id":60,"datatype":"1","titleimg":"https://res.datatang.com/asset/productNew/APY161101026_R.png?Expires=2007353621&OSSAccessKeyId=LTAI5tQwXnJZbubgVfVa1ep9&Signature=p8l6v5OKHS61eVN5nsHw4sznc3A%3D","type1":"165","type1str":null,"type2":"166","type2str":null,"dataname":"197 Hours Korean Speech Dataset for ASR and AI Model Training","datazy":[{"title":"Format","desc":"Format","content":"16kHz, 16bit, uncompressed wav, mono channel;"},{"title":"Recording condition","desc":"Recording condition","content":"Low background noise, without echo;"},{"title":"Content category","desc":"Content category","content":"economy, entertainment, news, informal language, numbers, alphabet;"},{"title":"Recording device","desc":"Recording device","content":"Android smartphone: iPhone=3.2:1;"},{"title":"Speaker","desc":"Speaker","content":"291 people in total, from South Korea and North Korea; 45% male and 55% female;"},{"title":"Country","desc":"Country","content":"South Korea(KOR), North Korea(PRK);"},{"title":"Language(Region) Code","desc":"Language(Region) Code","content":"ko-KR, ko-KP;"},{"title":"Language","desc":"Language","content":"Korean;"},{"title":"Features of annotation","desc":"Features of annotation","content":"Transcription text, timestamp, 5 noise symbols, special identifiers;"},{"title":"Accuracy Rate","desc":"Accuracy Rate","content":"Sentence Accuracy Rate (SAR) 95% (noise symbols and special identifiers are excluded)"}],"datatag":"Korean,Korea,Smartphone,Reading,Scripted Monologue","technologydoc":null,"downurl":null,"datainfo":null,"standard":null,"dataylurl":null,"flag":null,"publishtime":null,"createby":null,"createtime":null,"ext1":null,"samplestoreloc":null,"hosturl":null,"datasize":null,"industryPlan":null,"keyInformation":null,"samplePresentation":[{"name":"/data/apps/damp/temp/ziptemp/apy161101026_r1695808859355/apy161101026_r/T9137G0046Q0051.wav","url":"https://bj-oss-datatang-03.oss-cn-beijing.aliyuncs.com/filesInfoUpload/data/apps/damp/temp/ziptemp/apy161101026_r1695808859355/apy161101026_r/T9137G0046Q0051.wav?Expires=4102329599&OSSAccessKeyId=LTAI8NWs2pDolLNH&Signature=8fqn5MuKghz7JQzlQp%2FD001NS5I%3D","intro":"성종은 자신의 맏아들을 낳은 아내를 내쫓아 죽였다.","size":0,"progress":100,"type":"mp3"},{"name":"/data/apps/damp/temp/ziptemp/apy161101026_r1695808859355/apy161101026_r/T0140G0177S0001.wav","url":"https://bj-oss-datatang-03.oss-cn-beijing.aliyuncs.com/filesInfoUpload/data/apps/damp/temp/ziptemp/apy161101026_r1695808859355/apy161101026_r/T0140G0177S0001.wav?Expires=4102329599&OSSAccessKeyId=LTAI8NWs2pDolLNH&Signature=mx4ZTvw6%2BzrOGrrh51XXoLwoYgQ%3D","intro":"군[[lipsmack]] 산상고도 신정고를 구 대 칠로 물리치고 준준결승에 진출했습니다.","size":0,"progress":100,"type":"mp3"},{"name":"/data/apps/damp/temp/ziptemp/apy161101026_r1695808859355/apy161101026_r/T0138G0051S0001.wav","url":"https://bj-oss-datatang-03.oss-cn-beijing.aliyuncs.com/filesInfoUpload/data/apps/damp/temp/ziptemp/apy161101026_r1695808859355/apy161101026_r/T0138G0051S0001.wav?Expires=4102329599&OSSAccessKeyId=LTAI8NWs2pDolLNH&Signature=euHCP6TqbkWtWRmLvPR2sIp7GT4%3D","intro":"현재까지 제기된 모든 의혹에 대해서 철저히 조사할 방침입니다.","size":0,"progress":100,"type":"mp3"},{"name":"/data/apps/damp/temp/ziptemp/apy161101026_r1695808859355/apy161101026_r/T0139G0133S0001.wav","url":"https://bj-oss-datatang-03.oss-cn-beijing.aliyuncs.com/filesInfoUpload/data/apps/damp/temp/ziptemp/apy161101026_r1695808859355/apy161101026_r/T0139G0133S0001.wav?Expires=4102329599&OSSAccessKeyId=LTAI8NWs2pDolLNH&Signature=YK5b3OxRzXMNovrqjy7foGrVS%2B0%3D","intro":"굳이 선택해야 한다면 당선가능성과 정체성이 반반이다.","size":0,"progress":100,"type":"mp3"},{"name":"/data/apps/damp/temp/ziptemp/apy161101026_r1695808859355/apy161101026_r/T9137G0008Q0051.wav","url":"https://bj-oss-datatang-03.oss-cn-beijing.aliyuncs.com/filesInfoUpload/data/apps/damp/temp/ziptemp/apy161101026_r1695808859355/apy161101026_r/T9137G0008Q0051.wav?Expires=4102329599&OSSAccessKeyId=LTAI8NWs2pDolLNH&Signature=jO%2FWJPN9uhWEg6wJ9vwdoI255J8%3D","intro":"맨눈으로 가상 현실을 느낄 수 있는 기술입니다.","size":0,"progress":100,"type":"mp3"}],"officialSummary":"This dataset contains 197 hours of scripted Korean speech recordings collected from 291 native Korean speakers based on predefined scripts. The dataset covers diverse content domains, including economy, entertainment, news, informal expressions, numbers, alphabet-related speech, and other general language scenarios. Each audio sample includes accurate transcription and additional metadata attributes. Quality tested by various AI companies. We strictly adhere to data protection regulations and privacy standards, ensuring the maintenance of user privacy and legal rights throughout the data collection, storage, and usage processes, our datasets are all GDPR, CCPA, PIPL complied.","dataexampl":null,"datakeyword":["korean speech dataset","korean ASR dataset","korean speech corpus","korean speech recognition dataset","korean audio dataset","voice AI training data"],"isDelete":null,"ids":null,"idsList":null,"datasetCode":null,"productStatus":null,"tagTypeEn":"Data Type,Language","tagTypeZh":null,"website":null,"samplePresentationList":null,"datazyList":null,"keyInformationList":null,"dataexamplList":null,"bgimg":null,"datazyScriptList":null,"datakeywordListString":null,"sourceShowPage":"speechRec","dataShowType":"[{\"code\":\"0\",\"language\":\"ZH\"},{\"code\":\"1\",\"language\":\"ZH\"},{\"code\":\"2\",\"language\":\"EN,JP,PT,DE,KO,FR,ES\"},{\"code\":\"3\",\"language\":\"EN\"},{\"code\":\"4\",\"language\":\"JP\"}]","productNameEn":"197 Hours - Korean Speech Data by Mobile Phone_Reading","BGimg":"brightSpot_audio","voiceBg":["/shujutang/static/image/comm/audio_bg.webp","/shujutang/static/image/comm/audio_bg2.webp","/shujutang/static/image/comm/audio_bg3.webp","/shujutang/static/image/comm/audio_bg4.webp","/shujutang/static/image/comm/audio_bg5.webp"]}
197 Hours Korean Speech Dataset for ASR and AI Model Training
korean speech dataset
korean ASR dataset
korean speech corpus
korean speech recognition dataset
korean audio dataset
voice AI training data
This dataset contains 197 hours of scripted Korean speech recordings collected from 291 native Korean speakers based on predefined scripts. The dataset covers diverse content domains, including economy, entertainment, news, informal expressions, numbers, alphabet-related speech, and other general language scenarios. Each audio sample includes accurate transcription and additional metadata attributes. Quality tested by various AI companies. We strictly adhere to data protection regulations and privacy standards, ensuring the maintenance of user privacy and legal rights throughout the data collection, storage, and usage processes, our datasets are all GDPR, CCPA, PIPL complied.
This is a paid datasets for commercial use, research purpose and more. Licensed ready made datasets help jump-start AI projects.
What languages and scenarios are covered by Nexdata’s speech recognition datasets?
Nexdata offers speech recognition datasets covering a broad range of languages, dialects, and accents, backed by extensive global language resources. Our datasets include diverse speakers, acoustic environments, and real-world speech scenarios, supporting multilingual ASR, voice assistants, conversational AI, speech-to-text, and other speech-enabled applications.
Can Nexdata customize speech recognition datasets for specific languages or requirements?
Yes. If our off-the-shelf datasets do not fully meet your requirements, Nexdata provides flexible custom data collection, transcription, and annotation services. We can customize datasets based on target languages or dialects, speaker profiles, recording environments, speech scenarios, data volume, and annotation specifications to meet specific ASR development needs.
How does Nexdata ensure the quality and scalability of speech recognition datasets?
Nexdata applies multi-stage quality control throughout speech data collection, transcription, annotation, and validation. Combined with our extensive language resources and scalable collection capabilities, we can support both large-scale multilingual projects and specialized datasets for specific languages, dialects, accents, and speech scenarios.