Dataset Viewer
Auto-converted to Parquet Duplicate
file_name
stringlengths
5
50
id
stringlengths
13
68
source
stringclasses
14 values
source_path
stringlengths
24
100
tags
listlengths
3
15
001_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_001_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/001_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
001_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_001_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/001_nam.wav
[ "distortion", "female", "instantaneous_noise", "microphone_popping", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
002_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_002_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/002_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "wb_noise", "whispered_speech" ]
002_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_002_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/002_nam.wav
[ "distortion", "female", "instantaneous_noise", "microphone_popping", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "wb_noise", "whispered_speech" ]
003_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_003_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/003_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
003_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_003_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/003_nam.wav
[ "distortion", "female", "instantaneous_noise", "low_noise_level", "microphone_popping", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
004_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_004_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/004_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
004_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_004_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/004_nam.wav
[ "distortion", "female", "instantaneous_noise", "microphone_popping", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "wb_noise", "whispered_speech" ]
005_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_005_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/005_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech", "wrong_transcript" ]
005_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_005_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/005_nam.wav
[ "distortion", "female", "instantaneous_noise", "low_noise_level", "microphone_popping", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech", "wrong_transcript" ]
006_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_006_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/006_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
006_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_006_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/006_nam.wav
[ "distortion", "female", "instantaneous_noise", "low_noise_level", "microphone_popping", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
007_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_007_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/007_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
007_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_007_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/007_nam.wav
[ "distortion", "female", "instantaneous_noise", "microphone_popping", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
008_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_008_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/008_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
008_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_008_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/008_nam.wav
[ "distortion", "female", "instantaneous_noise", "low_noise_level", "microphone_popping", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
009_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_009_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/009_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
009_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_009_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/009_nam.wav
[ "distortion", "female", "instantaneous_noise", "low_noise_level", "microphone_popping", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
010_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_010_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/010_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
010_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_010_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/010_nam.wav
[ "distortion", "female", "instantaneous_noise", "low_noise_level", "microphone_popping", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
011_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_011_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/011_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
011_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_011_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/011_nam.wav
[ "distortion", "female", "instantaneous_noise", "microphone_popping", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "wb_noise", "whispered_speech" ]
012_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_012_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/012_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
012_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_012_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/012_nam.wav
[ "distortion", "female", "instantaneous_noise", "low_noise_level", "microphone_popping", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
013_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_013_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/013_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
013_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_013_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/013_nam.wav
[ "distortion", "female", "instantaneous_noise", "low_noise_level", "microphone_popping", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
014_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_014_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/014_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
014_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_014_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/014_nam.wav
[ "distortion", "female", "instantaneous_noise", "microphone_popping", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
015_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_015_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/015_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
015_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_015_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/015_nam.wav
[ "distortion", "female", "instantaneous_noise", "low_noise_level", "microphone_popping", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
016_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_016_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/016_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
016_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_016_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/016_nam.wav
[ "distortion", "female", "instantaneous_noise", "microphone_popping", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
017_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_017_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/017_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
017_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_017_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/017_nam.wav
[ "distortion", "female", "instantaneous_noise", "low_noise_level", "microphone_popping", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
018_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_018_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/018_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
018_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_018_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/018_nam.wav
[ "distortion", "female", "instantaneous_noise", "microphone_popping", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
019_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_019_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/019_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
019_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_019_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/019_nam.wav
[ "distortion", "female", "instantaneous_noise", "microphone_popping", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
020_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_020_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/020_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
020_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_020_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/020_nam.wav
[ "distortion", "female", "instantaneous_noise", "microphone_popping", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
021_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_021_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/021_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
021_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_021_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/021_nam.wav
[ "distortion", "female", "instantaneous_noise", "microphone_popping", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
022_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_022_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/022_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
022_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_022_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/022_nam.wav
[ "distortion", "female", "instantaneous_noise", "low_noise_level", "microphone_popping", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
023_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_023_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/023_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
023_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_023_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/023_nam.wav
[ "distortion", "female", "instantaneous_noise", "microphone_popping", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
024_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_024_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/024_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
024_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_024_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/024_nam.wav
[ "distortion", "female", "instantaneous_noise", "microphone_popping", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
025_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_025_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/025_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech", "wrong_transcript" ]
025_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_025_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/025_nam.wav
[ "distortion", "female", "instantaneous_noise", "low_noise_level", "microphone_popping", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech", "wrong_transcript" ]
026_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_026_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/026_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
026_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_026_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/026_nam.wav
[ "distortion", "female", "instantaneous_noise", "microphone_popping", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
027_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_027_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/027_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
027_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_027_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/027_nam.wav
[ "distortion", "female", "instantaneous_noise", "microphone_popping", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
028_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_028_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/028_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "wb_noise", "whispered_speech" ]
028_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_028_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/028_nam.wav
[ "distortion", "female", "instantaneous_noise", "microphone_popping", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "wb_noise", "whispered_speech" ]
029_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_029_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/029_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
029_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_029_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/029_nam.wav
[ "distortion", "female", "instantaneous_noise", "low_noise_level", "microphone_popping", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
030_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_030_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/030_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
030_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_030_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/030_nam.wav
[ "distortion", "female", "instantaneous_noise", "microphone_popping", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
031_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_031_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/031_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "wb_noise", "whispered_speech" ]
031_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_031_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/031_nam.wav
[ "distortion", "female", "instantaneous_noise", "microphone_popping", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "wb_noise", "whispered_speech" ]
032_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_032_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/032_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "wb_noise", "whispered_speech" ]
032_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_032_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/032_nam.wav
[ "distortion", "female", "instantaneous_noise", "microphone_popping", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "wb_noise", "whispered_speech" ]
033_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_033_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/033_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
033_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_033_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/033_nam.wav
[ "distortion", "female", "instantaneous_noise", "microphone_popping", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
034_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_034_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/034_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
034_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_034_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/034_nam.wav
[ "distortion", "female", "instantaneous_noise", "microphone_popping", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
035_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_035_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/035_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
035_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_035_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/035_nam.wav
[ "distortion", "female", "instantaneous_noise", "microphone_popping", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
036_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_036_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/036_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
036_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_036_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/036_nam.wav
[ "distortion", "female", "instantaneous_noise", "microphone_popping", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
037_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_037_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/037_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "wb_noise", "whispered_speech", "wrong_transcript" ]
037_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_037_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/037_nam.wav
[ "distortion", "female", "instantaneous_noise", "microphone_popping", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "wb_noise", "whispered_speech", "wrong_transcript" ]
038_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_038_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/038_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
038_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_038_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/038_nam.wav
[ "distortion", "female", "instantaneous_noise", "low_noise_level", "microphone_popping", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
039_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_039_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/039_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
039_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_039_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/039_nam.wav
[ "distortion", "female", "instantaneous_noise", "low_noise_level", "microphone_popping", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
040_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_040_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/040_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
040_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_040_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/040_nam.wav
[ "distortion", "female", "instantaneous_noise", "microphone_popping", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
041_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_041_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/041_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
041_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_041_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/041_nam.wav
[ "distortion", "female", "instantaneous_noise", "microphone_popping", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
042_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_042_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/042_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
042_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_042_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/042_nam.wav
[ "distortion", "female", "instantaneous_noise", "microphone_popping", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
043_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_043_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/043_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
043_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_043_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/043_nam.wav
[ "distortion", "female", "instantaneous_noise", "microphone_popping", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
044_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_044_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/044_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
044_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_044_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/044_nam.wav
[ "distortion", "female", "instantaneous_noise", "low_noise_level", "microphone_popping", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
045_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_045_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/045_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
045_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_045_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/045_nam.wav
[ "distortion", "female", "instantaneous_noise", "microphone_popping", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
046_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_046_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/046_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
046_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_046_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/046_nam.wav
[ "distortion", "female", "instantaneous_noise", "microphone_popping", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "wb_noise", "whispered_speech" ]
047_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_047_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/047_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
047_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_047_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/047_nam.wav
[ "distortion", "female", "instantaneous_noise", "microphone_popping", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "wb_noise", "whispered_speech" ]
048_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_048_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/048_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
048_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_048_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/048_nam.wav
[ "distortion", "female", "instantaneous_noise", "microphone_popping", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
049_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_049_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/049_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
049_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_049_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/049_nam.wav
[ "distortion", "female", "instantaneous_noise", "microphone_popping", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
050_headset.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_050_headset
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/050_headset.wav
[ "female", "instantaneous_noise", "mid_high_noise_level", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
050_nam.wav
[CSTR-NAM-TIMIT-Plus]HERALD_CLEAN_050_nam
CSTR-NAM-TIMIT-Plus
CSTR-NAM-TIMIT-Plus/050_nam.wav
[ "distortion", "female", "instantaneous_noise", "low_noise_level", "microphone_popping", "nb_noise", "real_recording", "regular_pitch", "regular_speed", "vocal_sound", "wb_noise", "whispered_speech" ]
End of preview. Expand in Data Studio

Speech Quality Test Labels

It publishes this README, metadata.jsonl, and a tag-to-binary conversion script, because the 14 source datasets have different copyright and access terms.

The manifest contains the 10,728-sample PICSAFEv1 subset. It is not a license to redistribute the underlying recordings.

Metadata fields

  • file_name: audio basename.
  • id: globally unique sample identifier with the [source] prefix.
  • source: source dataset name.
  • source_path: path relative to the origin directory.
  • tags: list of labels.

Source datasets and official download paths

Source dataset Samples License / access note Official download or access path
CSTR-NAM-TIMIT-Plus 842 ODC-By 1.0 Edinburgh DataShare
DNS5_LibriVox 510 LibriVox public domain; jurisdiction restrictions may apply Microsoft DNS5 download instructions
EARS 1,042 CC BY-NC 4.0; non-commercial use Official GitHub repository
MSceneSpeech 1,000 Dataset audio terms are not stated separately; confirm with authors Project page / Google Drive
NISQA (NISQA_TEST_FOR) 724 Original source terms; this subset is restricted to non-commercial research/forensic use NISQA Corpus wiki / DepositOnce archive
SOMOSv2 500 Research and non-commercial use only; redistribution under the same terms Zenodo record
TencentCorpus 1,010 No separate public audio license located; challenge access terms apply ConferencingSpeech2022 repository / challenge plan
VCTK 1,000 CC BY 4.0 for the official VCTK release; verify the exact version Edinburgh DataShare
WenetSpeech 600 CC BY 4.0; access requires the official form/password workflow OpenSLR SLR121
Whisper40 500 No explicit dataset license located in the official repository Official GitHub repository
WSJ 1,000 LDC User Agreement; license required WSJ0 / LDC93S6A and WSJ1 / LDC94S13A
zhvoice 500 Mixed upstream datasets; no unified license Official GitHub repository (Baidu download link is on the page)
LibriTTS-R 500 CC BY 4.0 OpenSLR SLR141
LibriTTS 1,000 CC BY 4.0 OpenSLR SLR60

Label schema

In article, it is described that the 33 labels are:

Nspk, distortion, emotional, enhanced, excessive_sibilance, fast_speed, female, high_pitch, imperceptible_noise_level, instantaneous_noise, low_noise_level, low_pitch, male, microphone_popping, mid_high_noise_level, music_or_effect, nb_noise, non_binary, non_speech, read_speech, real_recording, regular_pitch, regular_speed, reverberation, singing_voice, slow_speed, speech_overlap, spontaneous_speech, synthetic, vocal_sound, wb_noise, whispered_speech, and wrong_transcript.

Converting tags to binary labels

Labels are derived from the human tags, not predictions from a decision tree or metric thresholds.

  • 1: accepted under the selected policy (positive).
  • 0: rejected because at least one negative tag is present (negative).

The label is 0 if any annotation tag belongs to the mode-specific negative set below or the additional --negative_tags set; otherwise it is 1.

Mode Tags that produce label 0 (any match)
strict (default) Nspk, speech_overlap, low_noise_level, mid_high_noise_level, reverberation, music_or_effect, microphone_popping, excessive_sibilance, distortion
mid speech_overlap, mid_high_noise_level, reverberation, music_or_effect, microphone_popping, excessive_sibilance, distortion
loose speech_overlap, mid_high_noise_level, reverberation, music_or_effect, distortion

mid allows Nspk and low_noise_level; loose additionally allows microphone_popping and excessive_sibilance. Other tags do not independently cause rejection. In particular, non_speech, wrong_transcript, singing_voice, vocal_sound, nb_noise, wb_noise, and instantaneous_noise are not default negative tags. Label 1 means policy acceptance, not a universal claim of clean speech or a correct transcript. Add rejection tags explicitly if needed. Names are case-sensitive (Nspk). An empty tag list yields 1, matching the reference function, but does not establish annotation completeness. Missing tag fields and unknown tag names are errors.

Run from the repository directory:

# Default policy.
python3 tags_to_binary.py --input metadata.jsonl --output /tmp/picsafe_strict.jsonl

# Alternative policy (also supports loose).
python3 tags_to_binary.py --input metadata.jsonl --output /tmp/picsafe_mid.jsonl --positive_mode mid

# Custom policy, different from the default labels.
python3 tags_to_binary.py --input metadata.jsonl --output /tmp/picsafe_custom.jsonl --negative_tags non_speech wrong_transcript

# Original annotations: UID and semicolon-separated Tags columns.
python3 tags_to_binary.py --input annotations.tsv --output /tmp/picsafe_from_tsv.jsonl

Output JSONL preserves all original fields and adds binary_label and label_policy (the selected positive_mode and additional negative_tags). TSV input also adds id from UID and tags from Tags. The script prints class counts, validates unique IDs and tag names, and refuses to overwrite existing output files. The source metadata.jsonl is unchanged. Join with metric scores using globally unique id; basenames can collide. To reproduce a threshold-search run, use the same mode, additional negative tags, and sample subset.

Licensing and limitations

This repository does not grant, aggregate, or override any source-dataset license. Obtain the audio directly from the source providers and follow their current terms, attribution requirements, access restrictions, and applicable privacy/publicity rules. Some sources require non-commercial research use, registration, a license, or direct permission from the authors.

Downloads last month
70