[{"@type":"PropertyValue","name":"Format","value":"48,000Hz, 24bit, uncompressed wav, mono channel;"},{"@type":"PropertyValue","name":"Recording environment","value":"professional recording studio;"},{"@type":"PropertyValue","name":"Recording content","value":"general corpus;"},{"@type":"PropertyValue","name":"Speaker","value":"Vietnamese, 1 male and 1 female, 10 hours per person;"},{"@type":"PropertyValue","name":"Annotation","value":"word and phoneme transcription, prosodic boundary annotation, phoneme boundary annotation;"},{"@type":"PropertyValue","name":"Device","value":"microphone;"},{"@type":"PropertyValue","name":"Language","value":"Vietnamese;"},{"@type":"PropertyValue","name":"Application scenarios","value":"speech synthesis."}]
{"id":2024,"datatype":"1","titleimg":"https://www.nexdata.ai/shujutang/static/image/index/datatang_yuyin_default.webp","type1":"165","type1str":null,"type2":"219","type2str":null,"dataname":"Vietnamese TTS Dataset - 2 Native Speakers Speech Synthesis Corpus","datazy":[{"title":"Format","content":"48,000Hz, 24bit, uncompressed wav, mono channel;"},{"title":"Recording environment","content":"professional recording studio;"},{"title":"Recording content","content":"general corpus;"},{"title":"Speaker","content":"Vietnamese, 1 male and 1 female, 10 hours per person;"},{"title":"Annotation","content":"word and phoneme transcription, prosodic boundary annotation, phoneme boundary annotation;"},{"title":"Device","content":"microphone;"},{"title":"Language","content":"Vietnamese;"},{"title":"Application scenarios","content":"speech synthesis."}],"datatag":"TTS,Vietnamese","technologydoc":null,"downurl":null,"datainfo":null,"standard":null,"dataylurl":null,"flag":null,"publishtime":null,"createby":null,"createtime":null,"ext1":null,"samplestoreloc":null,"hosturl":null,"datasize":null,"industryPlan":null,"keyInformation":null,"samplePresentation":[{"name":"200919.wav","url":"https://storage-product.datatang.com/damp/product/instructions_zh/20260518135316/200919.wav?Expires=4102415999&OSSAccessKeyId=LTAI5tEBeSWUJiqjXvBMsxEu&Signature=sD4u0YGxfqs0wioa7OgoRYGBH9c%3D","intro":"Nhớ hồi xưa đi học trốn tiết ra sân tập%, bị bác bảo vệ đuổi chạy toát mồ hôi hột%, vui thật đấy%. {ɲ əː3} {h o2 i} {s ɯə1} {ɗ i1} {h ɔ6 k} {c o3 n} {t ie3 t} {z aː1} {s ə1 n} {t ə6 p}, {ɓ i6} {ɓ aː3 k} {ɓ aː4 u} {v e6} {ɗ uo4 i} {c a6 i} {t u aː3 t} {m o2} {h o1 i} {h o6 t}, {v u1 i} {tʰ ə6 t} {ɗ ə3 i}.","size":833168,"progress":100,"type":"mp3"},{"name":"200981.wav","url":"https://storage-product.datatang.com/damp/product/instructions_zh/20260518135316/200981.wav?Expires=4102415999&OSSAccessKeyId=LTAI5tEBeSWUJiqjXvBMsxEu&Signature=TfG83%2BupfhJR17Juv5fZyfeTRSc%3D","intro":"Bà bây giờ cũng sành điệu ghê%. Thế chốt nhé%, ba giờ chiều gặp nhau ở chân đê%. {ɓ aː2} {ɓ ə1 i} {z əː2} {k u5 ŋ} {s aː2 ɲ} {ɗ ie6 u} {ɣ e1}. {tʰ e3} {c o3 t} {ɲ ɛ3}, {ɓ aː1} {z əː2} {c ie2 u} {ɣ a6 p} {ɲ a1 u} {əː4} {c ə1 n} {ɗ e1}.","size":744866,"progress":100,"type":"mp3"},{"name":"101299.wav","url":"https://storage-product.datatang.com/damp/product/instructions_zh/20260518135316/101299.wav?Expires=4102415999&OSSAccessKeyId=LTAI5tEBeSWUJiqjXvBMsxEu&Signature=jYtd1zBt46kol1srQpAOLaCPW%2F0%3D","intro":"Nước sông chuyển sang màu lạ và bốc mùi hôi% là dấu hiệu rõ ràng% của ô nhiễm hữu cơ nặng%. {n ɯə3 k} {s o1 ŋ} {c u ie4 n} {s aː1 ŋ} {m a2 u} {l aː6} {v aː2} {ɓ o3 k} {m u2 i} {h o1 i} {l aː2} {z ə3 u} {h ie6 u} {z ɔ5} {z aː2 ŋ} {k uə4} {o1} {ɲ ie5 m} {h ɯ5 u} {k əː1} {n a6 ŋ}.","size":860346,"progress":100,"type":"mp3"},{"name":"200413.wav","url":"https://storage-product.datatang.com/damp/product/instructions_zh/20260518135316/200413.wav?Expires=4102415999&OSSAccessKeyId=LTAI5tEBeSWUJiqjXvBMsxEu&Signature=ySs0wKLPnV49r8MLW15gosmOSd4%3D","intro":"Đúng rồi%. Mở ra thấy toàn là những kỷ vật của cha mẹ và chúng mình hồi còn bé%. {ɗ u3 ŋ} {z o2 i}. {m əː4} {z aː1} {tʰ ə3 i} {t u aː2 n} {l aː2} {ɲ ɯ5 ŋ} {k i4} {v ə6 t} {k uə4} {c aː1} {m ɛ6} {v aː2} {c u3 ŋ} {m i2 ɲ} {h o2 i} {k ɔ2 n} {ɓ ɛ3}.","size":702762,"progress":100,"type":"mp3"},{"name":"102569.wav","url":"https://storage-product.datatang.com/damp/product/instructions_zh/20260518135316/102569.wav?Expires=4102415999&OSSAccessKeyId=LTAI5tEBeSWUJiqjXvBMsxEu&Signature=%2F9rpbmvLLeEkowJgytyNT3dgABY%3D","intro":"Văn học kinh điển/ vẫn luôn giữ nguyên giá trị nhân văn sâu sắc% bởi phản ánh chân thực khát vọng/ và đấu tranh của con người/ qua nhiều thế kỷ%. {v a1 n} {h ɔ6 k} {k i1 ɲ} {ɗ ie4 n} {v ə5 n} {l uo1 n} {z ɯ5} {ŋ u ie1 n} {z aː3} {c i6} {ɲ ə1 n} {v a1 n} {s ə1 u} {s a3 k} {ɓ əː4 i} {f aː4 n} {aː3 ɲ} {c ə1 n} {tʰ ɯ6 k} {x aː3 t} {v ɔ6 ŋ} {v aː2} {ɗ ə3 u} {c aː1 ɲ} {k uə4} {k ɔ1 n} {ŋ ɯə2 i} {k u aː1} {ɲ ie2 u} {tʰ e3} {k i4}.","size":1302480,"progress":100,"type":"mp3"}],"officialSummary":"This Vietnamese speech synthesis corpus is recorded by 2 native Vietnamese speakers with authentic Vietnamese accents. It features balanced phoneme coverage and was annotated with the participation of professional phoneticians, enabling it to precisely meet the R&D requirements for speech synthesis.","dataexampl":null,"datakeyword":["vietnamese tts dataset","vietnamese speech synthesis corpus","vietnamese speech corpus","vietnamese voice dataset","vietnamese text to speech dataset"],"isDelete":null,"ids":null,"idsList":null,"datasetCode":null,"productStatus":null,"tagTypeEn":"Language,Voice Type","tagTypeZh":null,"website":null,"samplePresentationList":null,"datazyList":null,"keyInformationList":null,"dataexamplList":null,"bgimg":null,"datazyScriptList":null,"datakeywordListString":null,"sourceShowPage":"speechSyn","dataShowType":"[{\"code\":\"0\",\"language\":\"ZH\"},{\"code\":\"1\",\"language\":\"ZH\"},{\"code\":\"2\",\"language\":\"EN\"},{\"code\":\"3\",\"language\":\"EN\"},{\"code\":\"4\",\"language\":\"JP\"}]","productNameEn":"2 People - Vietnamese Average Tone Speech Synthesis Corpus","BGimg":"brightSpot_audio","voiceBg":["/shujutang/static/image/comm/audio_bg.webp","/shujutang/static/image/comm/audio_bg2.webp","/shujutang/static/image/comm/audio_bg3.webp","/shujutang/static/image/comm/audio_bg4.webp","/shujutang/static/image/comm/audio_bg5.webp"]}
https://www.nexdata.ai/shujutang/static/image/index/datatang_yuyin_default.webp
[{"@type":"AudioObject","embedUrl":"https://storage-product.datatang.com/damp/product/instructions_zh/20260518135316/200919.wav?Expires=4102415999&OSSAccessKeyId=LTAI5tEBeSWUJiqjXvBMsxEu&Signature=sD4u0YGxfqs0wioa7OgoRYGBH9c%3D"},{"@type":"AudioObject","embedUrl":"https://storage-product.datatang.com/damp/product/instructions_zh/20260518135316/200981.wav?Expires=4102415999&OSSAccessKeyId=LTAI5tEBeSWUJiqjXvBMsxEu&Signature=TfG83%2BupfhJR17Juv5fZyfeTRSc%3D"},{"@type":"AudioObject","embedUrl":"https://storage-product.datatang.com/damp/product/instructions_zh/20260518135316/101299.wav?Expires=4102415999&OSSAccessKeyId=LTAI5tEBeSWUJiqjXvBMsxEu&Signature=jYtd1zBt46kol1srQpAOLaCPW%2F0%3D"},{"@type":"AudioObject","embedUrl":"https://storage-product.datatang.com/damp/product/instructions_zh/20260518135316/200413.wav?Expires=4102415999&OSSAccessKeyId=LTAI5tEBeSWUJiqjXvBMsxEu&Signature=ySs0wKLPnV49r8MLW15gosmOSd4%3D"},{"@type":"AudioObject","embedUrl":"https://storage-product.datatang.com/damp/product/instructions_zh/20260518135316/102569.wav?Expires=4102415999&OSSAccessKeyId=LTAI5tEBeSWUJiqjXvBMsxEu&Signature=%2F9rpbmvLLeEkowJgytyNT3dgABY%3D"}]
Vietnamese TTS Dataset - 2 Native Speakers Speech Synthesis Corpus vietnamese tts dataset
vietnamese speech synthesis corpus
vietnamese speech corpus
vietnamese voice dataset
vietnamese text to speech dataset
This Vietnamese speech synthesis corpus is recorded by 2 native Vietnamese speakers with authentic Vietnamese accents. It features balanced phoneme coverage and was annotated with the participation of professional phoneticians, enabling it to precisely meet the R&D requirements for speech synthesis. This is a paid dataset licensed for commercial use. Ready-made datasets are available for immediate integration into AI projects.
Specifications
Format
48,000Hz, 24bit, uncompressed wav, mono channel;
Recording environment
professional recording studio;
Recording content
general corpus;
Speaker
Vietnamese, 1 male and 1 female, 10 hours per person;
Annotation
word and phoneme transcription, prosodic boundary annotation, phoneme boundary annotation;
Application scenarios
speech synthesis.
Sample
Audio Nhớ hồi xưa đi học trốn tiết ra sân tập%, bị bác bảo vệ đuổi chạy toát mồ hôi hột%, vui thật đấy%. {ɲ əː3} {h o2 i} {s ɯə1} {ɗ i1} {h ɔ6 k} {c o3 n} {t ie3 t} {z aː1} {s ə1 n} {t ə6 p}, {ɓ i6} {ɓ aː3 k} {ɓ aː4 u} {v e6} {ɗ uo4 i} {c a6 i} {t u aː3 t} {m o2} {h o1 i} {h o6 t}, {v u1 i} {tʰ ə6 t} {ɗ ə3 i}.
Audio Bà bây giờ cũng sành điệu ghê%. Thế chốt nhé%, ba giờ chiều gặp nhau ở chân đê%. {ɓ aː2} {ɓ ə1 i} {z əː2} {k u5 ŋ} {s aː2 ɲ} {ɗ ie6 u} {ɣ e1}. {tʰ e3} {c o3 t} {ɲ ɛ3}, {ɓ aː1} {z əː2} {c ie2 u} {ɣ a6 p} {ɲ a1 u} {əː4} {c ə1 n} {ɗ e1}.
Audio Nước sông chuyển sang màu lạ và bốc mùi hôi% là dấu hiệu rõ ràng% của ô nhiễm hữu cơ nặng%. {n ɯə3 k} {s o1 ŋ} {c u ie4 n} {s aː1 ŋ} {m a2 u} {l aː6} {v aː2} {ɓ o3 k} {m u2 i} {h o1 i} {l aː2} {z ə3 u} {h ie6 u} {z ɔ5} {z aː2 ŋ} {k uə4} {o1} {ɲ ie5 m} {h ɯ5 u} {k əː1} {n a6 ŋ}.
Audio Đúng rồi%. Mở ra thấy toàn là những kỷ vật của cha mẹ và chúng mình hồi còn bé%. {ɗ u3 ŋ} {z o2 i}. {m əː4} {z aː1} {tʰ ə3 i} {t u aː2 n} {l aː2} {ɲ ɯ5 ŋ} {k i4} {v ə6 t} {k uə4} {c aː1} {m ɛ6} {v aː2} {c u3 ŋ} {m i2 ɲ} {h o2 i} {k ɔ2 n} {ɓ ɛ3}.
Audio Văn học kinh điển/ vẫn luôn giữ nguyên giá trị nhân văn sâu sắc% bởi phản ánh chân thực khát vọng/ và đấu tranh của con người/ qua nhiều thế kỷ%. {v a1 n} {h ɔ6 k} {k i1 ɲ} {ɗ ie4 n} {v ə5 n} {l uo1 n} {z ɯ5} {ŋ u ie1 n} {z aː3} {c i6} {ɲ ə1 n} {v a1 n} {s ə1 u} {s a3 k} {ɓ əː4 i} {f aː4 n} {aː3 ɲ} {c ə1 n} {tʰ ɯ6 k} {x aː3 t} {v ɔ6 ŋ} {v aː2} {ɗ ə3 u} {c aː1 ɲ} {k uə4} {k ɔ1 n} {ŋ ɯə2 i} {k u aː1} {ɲ ie2 u} {tʰ e3} {k i4}.
Recommended Dataset
2 People - French Natural Conversation Average Tone Speech Synthesis Corpus
French Natural Conversational Speech Synthesis Corpus. Recorded by two native voice talents (one male, one female), this corpus comprises two modules: free dialogue on specified topics and scripted emotional/paralinguistic text performances. It covers single-emotion, multi-emotion, and paralinguistic expressions (including a variety of interjections). Professional linguists have provided three-tier annotation—on text content, emotion labels, and paralinguistic phenomena. The total duration is approximately 11 hours, with audio specifications of 48 kHz sampling rate, 24‑bit depth, WAV format, mono channel. This dataset is designed to fully meet the diverse requirements of TTS research and development in terms of naturalness, expressiveness, and precise annotation.
French Natural Conversation Average Tone TTS Emotion Multi-level Emotion Paralanguage Interjection Freetalk
Details 2 People - Indonesian Natural Conversation Average Tone Speech Synthesis Corpus
Indonesian Natural Conversational Speech Synthesis Corpus. Recorded by two native voice talents (one male, one female), this corpus comprises two modules: free dialogue on specified topics and scripted emotional/paralinguistic text performances. It covers single-emotion, multi-emotion, and paralinguistic expressions (including a variety of modal particles). Professional linguists have provided three-tier annotation—on text content, emotion labels, and paralinguistic phenomena. The total duration is approximately 11 hours, with audio specifications of 48 kHz sampling rate, 24‑bit depth, WAV format, mono channel. This dataset is designed to fully meet the diverse requirements of TTS research and development in terms of naturalness, expressiveness, and precise annotation.
Indonesian Natural Conversation Freetalk Average Tone TTS Emotion Multi-level Emotion Interjection Paralanguage
Details 2 People - Saudi Arabian Arabic Natural Conversation Average Tone Speech Synthesis Corpus
Saudi Arabian Arabic Natural Conversational Speech Synthesis Corpus. Recorded by two native voice talents (one male, one female), this corpus comprises two modules: free dialogue on specified topics and scripted emotional/paralinguistic text performances. It covers single-emotion, multi-emotion, and paralinguistic expressions (including a variety of interjections). Professional linguists have provided three-tier annotation—on text content, emotion labels, and paralinguistic phenomena. The total duration is approximately 11 hours, with audio specifications of 48 kHz sampling rate, 24‑bit depth, WAV format, mono channel. This dataset is designed to fully meet the diverse requirements of TTS research and development in terms of naturalness, expressiveness, and precise annotation.
Saudi Arabian Arabic Natural Conversation Freetalk Average Tone TTS Emotion Multi-level Emotion Paralanguage Interjection
Details 2 People - Modern Standard Arabic Natural Conversation Average Tone Speech Synthesis Corpus
Modern Standard Arabic Natural Conversational Speech Synthesis Corpus. Recorded by two native voice talents (one male, one female), this corpus comprises two modules: free dialogue on specified topics and scripted emotional/paralinguistic text performances. It covers single-emotion, multi-emotion, and paralinguistic expressions (including a variety of interjections). Professional linguists have provided three-tier annotation—on text content, emotion labels, and paralinguistic phenomena. The total duration is approximately 11 hours, with audio specifications of 48 kHz sampling rate, 24‑bit depth, WAV format, mono channel. This dataset is designed to fully meet the diverse requirements of TTS research and development in terms of naturalness, expressiveness, and precise annotation.
Modern Standard Arabic Natural Conversation Freetalk Average Tone TTS Emotion Multi-level Emotion Paralanguage interjection
Details 2 People - Japanese Natural Conversation Average Tone Speech Synthesis Corpus
Japanese Natural Conversational Speech Synthesis Corpus. Recorded by two native voice talents (one male, one female), this corpus comprises two modules: free dialogue on specified topics and scripted emotional/paralinguistic text performances. It covers single-emotion, multi-emotion, and paralinguistic expressions (including a variety of modal particles). Professional linguists have provided three-tier annotation—on text content, emotion labels, and paralinguistic phenomena. The total duration is approximately 11 hours, with audio specifications of 48 kHz sampling rate, 24‑bit depth, WAV format, mono channel. This dataset is designed to fully meet the diverse requirements of TTS research and development in terms of naturalness, expressiveness, and precise annotation.
Japanese Natural Conversation Freetalk Average Tone TTS Emotion Multi-level Emotion Paralanguage Modal Particle
Details 2 People - Italian Natural Conversation Average Tone Speech Synthesis Corpus
Italian Natural Conversational Speech Synthesis Corpus. Recorded by two native voice talents (one male, one female), this corpus comprises two modules: free dialogue on specified topics and scripted emotional/paralinguistic text performances. It covers single-emotion, multi-emotion, and paralinguistic expressions (including a variety of interjections). Professional linguists have provided three-tier annotation—on text content, emotion labels, and paralinguistic phenomena. The total duration is approximately 11 hours, with audio specifications of 48 kHz sampling rate, 24‑bit depth, WAV format, mono channel. This dataset is designed to fully meet the diverse requirements of TTS research and development in terms of naturalness, expressiveness, and precise annotation.
Italian Natural Conversation Freetalk TTS Emotion Multi-level Emotion Paralanguage Modal Particle Average Tone Interjection
Details 2 People - Indian English Natural Conversation Average Tone Speech Synthesis Corpus
Indian English Natural Conversational Speech Synthesis Corpus. Recorded by two native voice talents (one male, one female), this corpus comprises two modules: free dialogue on specified topics and scripted emotional/paralinguistic text performances. It covers single-emotion, multi-emotion, and paralinguistic expressions (including a variety of interjections). Professional linguists have provided three-tier annotation—on text content, emotion labels, and paralinguistic phenomena. The total duration is approximately 11 hours, with audio specifications of 48 kHz sampling rate, 24‑bit depth, WAV format, mono channel. This dataset is designed to fully meet the diverse requirements of TTS research and development in terms of naturalness, expressiveness, and precise annotation.
Indian English Natural Conversation Average Tone Freetalk TTS Emotion Multi-level Emotion Paralanguage Interjection
Details 2 People - Vietnamese Natural Conversation Average Tone Speech Synthesis Corpus
Vietnamese Natural Conversational Speech Synthesis Corpus. Recorded by two native voice talents (one male, one female), this corpus comprises two modules: free dialogue on specified topics and scripted emotional/paralinguistic text performances. It covers single-emotion, multi-emotion, and paralinguistic expressions (including a variety of interjections). Professional linguists have provided three-tier annotation—on text content, emotion labels, and paralinguistic phenomena. The total duration is approximately 11 hours, with audio specifications of 48 kHz sampling rate, 24‑bit depth, WAV format, mono channel. This dataset is designed to fully meet the diverse requirements of TTS research and development in terms of naturalness, expressiveness, and precise annotation.
Vietnamese Natural Conversation Freetalk Average Tone TTS Emotion Multi-level Emotion Paralanguage Interjection
Details
Tell Us Your Special Needs
Dataset FAQs What languages and voice characteristics are covered by Nexdata’s speech synthesis datasets?
Nexdata offers speech synthesis datasets covering a broad range of languages, dialects, accents, and voice types, supported by extensive global language resources. Our datasets include diverse speakers, speaking styles, emotions, and recording scenarios to support natural and expressive Text-to-Speech (TTS) model development.
Can Nexdata customize speech synthesis datasets for specific languages or requirements?
Yes. If our off-the-shelf TTS datasets do not fully meet your requirements, Nexdata provides flexible custom data collection, transcription, annotation, and quality control services. We can customize datasets based on target languages or dialects, speaker profiles, voice characteristics, emotions, speaking styles, recording environments, and data volume.
How does Nexdata ensure the quality and scalability of speech synthesis datasets?
Nexdata applies multi-stage quality control throughout voice data collection, transcription, annotation, and validation. Combined with our extensive language resources and scalable collection capabilities, we can support both large-scale multilingual TTS projects and specialized datasets for specific voices, accents, emotions, or speech scenarios.
f629c6bb-612f-4227-9d6f-097e23493b85