| # | uttid | text | ref |
baseline_tgtlen
basket_config_path: quality/tts/tortoise-baskets/cc_20250825_en-US.json
data_meta: null
exp_name: yt4_wavtokenizer_16K_lossent0.15__yt4_wavtokenizer_16K_lossent0.15
lang: en-us
meta:
basket_generation_config:
basket_lang: en-us
basket_path: quality/tts/tortoise-baskets/cc_20250825_en-US.json
batch_size: 1
gpus: 1
inference:
condition_sample_rate: 24000
diff_k: 3
diff_steps: 100
diffusion_exp: /mount/s3/tts-binary-data-nb/eg/exp/yt4_wavtokenizer_16K_lossent0.15
duplicate_reference: true
exp: /mount/s3/tts-binary-data-nb/eg/exp/yt4_wavtokenizer_16K_lossent0.15
gpt_generate_args:
do_sample: true
min_new_tokens: 20
num_return_sequences: 50
use_cache: true
out_sample_rate: 24000
override_conditioning_features:
bad_text_proba: 0.0
c50: 0.0
dmcs_flatness: 100500.0
dmcs_roll_off_0.995: 100500.0
emo2vec: null
emotion: null
is_animation: 0.0
pitch_std: 100.0
snr: 100.0
year: 2025.0
reranking_options:
mode: MBR
top_k: 1
target_len_rate: 1.0
vocoder: bigvgan
voice_samples_preprocessing: []
num_workers: 1
output_dir: cc_20250825/yt4_wavtokenizer_16K_lossent0.15__yt4_wavtokenizer_16K_lossent0.15__2026-01-19_14-20-36
ref_dir: cc_20250825/ref
ticket: QUALITY-41
basket_generation_git_hash: 7ba982d9bb8ddc0cb968d517f583b0227d2624ed
model_data_type: tts-cloning
ticket: QUALITY-41
version: 2026-01-19_14-20-36
|
indextts2 |
literategoggles_idxdistill
basket_config_path: /mount/s3/tts-binary-data-nb/dchebakov/metrics/tortoise-baskets/cc_20250825_en-US.json
data_meta: null
exp_name: yt4_wavtokenizer_16K_lossent0.15_idxdistill_1ref_emo
lang: en-us
meta:
basket_generation_config:
basket_lang: en-us
basket_path: /mount/s3/tts-binary-data-nb/dchebakov/metrics/tortoise-baskets/cc_20250825_en-US.json
batch_size: 1
gpus: 1
inference:
condition_sample_rate: 24000
diff_k: 3
diff_steps: 100
duplicate_reference: false
exp: /mount/s3/tts-binary-data-nb/dchebakov/models/yt4_wavtokenizer_16K_lossent0.15_idxdistill_1ref_emo
gpt_generate_args:
do_sample: true
enforce_silent_start: false
num_return_sequences: 30
use_cache: true
out_sample_rate: 24000
override_conditioning_features:
bad_text_proba: 0.0
c50: 0.0
dmcs_flatness: 100500.0
dmcs_roll_off_0.995: 100500.0
pitch_std: 100.0
snr: 100.0
reranking_options:
mode: MBR
top_k: 1
vocoder: bigvgan
voice_samples_preprocessing: []
num_workers: 1
output_dir: cc_20250825_en-US/yt4_wavtokenizer_16K_lossent0.15_idxdistill_1ref_emo__2026-01-10_07-59-19
ref_dir: cc_20250825_en-US/ref
ticket: QUALITY-000
basket_generation_git_hash: 084ccc0c4313e646e3630e4e5a35b7e04d70fdad
model_data_type: tts-cloning
ticket: QUALITY-000
version: 2026-01-10_07-59-19
|
indexrefencoder_ft_enus
basket_config_path: quality/tts/tortoise-baskets/cc_20250825_en-US.json
data_meta: null
exp_name: yt4_wavtokenizer_16K_lossent0.15_indexrefencoder_separate_self_movies2_finetune_full_v2__yt4_wavtokenizer_16K_lossent0.15_indexrefencoder_separate_self
lang: en-us
meta:
basket_generation_config:
basket_lang: en-us
basket_path: quality/tts/tortoise-baskets/cc_20250825_en-US.json
batch_size: 1
gpus: 1
inference:
condition_sample_rate: 24000
diff_k: 3
diff_steps: 100
diffusion_exp: /mount/s3/tts-binary-data-nb/eg/exp/yt4_wavtokenizer_16K_lossent0.15_indexrefencoder_separate_self
duplicate_reference: true
exp: /mount/s3/tts-binary-data-nb/eg/exp/yt4_wavtokenizer_16K_lossent0.15_indexrefencoder_separate_self_movies2_finetune_full_v2
gpt_generate_args:
do_sample: true
min_new_tokens: 20
num_return_sequences: 50
use_cache: true
out_sample_rate: 24000
override_conditioning_features:
bad_text_proba: 0.0
c50: 0.0
dmcs_flatness: 100500.0
dmcs_roll_off_0.995: 100500.0
emo2vec: null
emotion: null
is_animation: 0.0
pitch_std: 100.0
snr: 100.0
year: 2025.0
reranking_options:
mode: MBR
top_k: 1
target_len_rate: 1.0
vocoder: bigvgan
voice_samples_preprocessing: []
num_workers: 1
output_dir: cc_20250825/yt4_wavtokenizer_16K_lossent0.15_indexrefencoder_separate_self_movies2_finetune_full_v2__yt4_wavtokenizer_16K_lossent0.15_indexrefencoder_separate_self__2026-01-23_13-08-47
ref_dir: cc_20250825/ref
ticket: QUALITY-41
basket_generation_git_hash: 7ba982d9bb8ddc0cb968d517f583b0227d2624ed
model_data_type: tts-cloning
ticket: QUALITY-41
version: 2026-01-23_13-08-47
|
movies2_finetune_closest_anim
basket_config_path: quality/tts/tortoise-baskets/cc_20250825_en-US.json
data_meta: null
exp_name: yt4_wavtokenizer_16K_lossent0.15_movies2_finetune_closest__yt4_wavtokenizer_16K_lossent0.15
lang: en-us
meta:
basket_generation_config:
basket_lang: en-us
basket_path: quality/tts/tortoise-baskets/cc_20250825_en-US.json
batch_size: 1
gpus: 1
inference:
condition_sample_rate: 24000
diff_k: 3
diff_steps: 100
diffusion_exp: /mount/s3/tts-binary-data-nb/eg/exp/yt4_wavtokenizer_16K_lossent0.15
duplicate_reference: true
exp: /mount/s3/tts-binary-data-nb/eg/exp/yt4_wavtokenizer_16K_lossent0.15_movies2_finetune_closest
gpt_generate_args:
do_sample: true
min_new_tokens: 20
num_return_sequences: 50
use_cache: true
out_sample_rate: 24000
override_conditioning_features:
bad_text_proba: 0.0
c50: 0.0
dmcs_flatness: 100500.0
dmcs_roll_off_0.995: 100500.0
emo2vec: null
emotion: null
is_animation: 1.0
pitch_std: 100.0
snr: 100.0
year: 2025.0
reranking_options:
mode: MBR
top_k: 1
target_len_rate: 1.0
vocoder: bigvgan
voice_samples_preprocessing: []
num_workers: 1
output_dir: cc_20250825/yt4_wavtokenizer_16K_lossent0.15_movies2_finetune_closest__yt4_wavtokenizer_16K_lossent0.15__2026-01-14_17-25-18
ref_dir: cc_20250825/ref
ticket: QUALITY-41
basket_generation_git_hash: 7ba982d9bb8ddc0cb968d517f583b0227d2624ed
model_data_type: tts-cloning
ticket: QUALITY-41
version: 2026-01-14_17-25-18
|
baseline
basket_config_path: quality/tts/tortoise-baskets/cc_20250825_en-US.json
data_meta: null
exp_name: yt4_wavtokenizer_16K_lossent0.15__yt4_wavtokenizer_16K_lossent0.15
lang: en-us
meta:
basket_generation_config:
basket_lang: en-us
basket_path: quality/tts/tortoise-baskets/cc_20250825_en-US.json
batch_size: 1
gpus: 1
inference:
condition_sample_rate: 24000
diff_k: 3
diff_steps: 100
diffusion_exp: /mount/s3/tts-binary-data-nb/eg/exp/yt4_wavtokenizer_16K_lossent0.15
duplicate_reference: true
exp: /mount/s3/tts-binary-data-nb/eg/exp/yt4_wavtokenizer_16K_lossent0.15
gpt_generate_args:
do_sample: true
min_new_tokens: 20
num_return_sequences: 50
use_cache: true
out_sample_rate: 24000
override_conditioning_features:
bad_text_proba: 0.0
c50: 0.0
dmcs_flatness: 100500.0
dmcs_roll_off_0.995: 100500.0
emo2vec: null
emotion: null
is_animation: 1.0
pitch_std: 100.0
snr: 100.0
year: 2025.0
reranking_options:
mode: MBR
top_k: 1
vocoder: bigvgan
voice_samples_preprocessing: []
num_workers: 1
output_dir: cc_20250825/yt4_wavtokenizer_16K_lossent0.15__yt4_wavtokenizer_16K_lossent0.15__2026-01-23_18-31-03
ref_dir: cc_20250825/ref
ticket: QUALITY-41
basket_generation_git_hash: 7ba982d9bb8ddc0cb968d517f583b0227d2624ed
model_data_type: tts-cloning
ticket: QUALITY-41
version: 2026-01-23_18-31-03
|
indexrefencoder_ft_enus_notgtlen
basket_config_path: quality/tts/tortoise-baskets/cc_20250825_en-US.json
data_meta: null
exp_name: yt4_wavtokenizer_16K_lossent0.15_indexrefencoder_separate_self_movies2_finetune_full_v2__yt4_wavtokenizer_16K_lossent0.15_indexrefencoder_separate_self
lang: en-us
meta:
basket_generation_config:
basket_lang: en-us
basket_path: quality/tts/tortoise-baskets/cc_20250825_en-US.json
batch_size: 1
gpus: 1
inference:
condition_sample_rate: 24000
diff_k: 3
diff_steps: 100
diffusion_exp: /mount/s3/tts-binary-data-nb/eg/exp/yt4_wavtokenizer_16K_lossent0.15_indexrefencoder_separate_self
duplicate_reference: true
exp: /mount/s3/tts-binary-data-nb/eg/exp/yt4_wavtokenizer_16K_lossent0.15_indexrefencoder_separate_self_movies2_finetune_full_v2
gpt_generate_args:
do_sample: true
min_new_tokens: 20
num_return_sequences: 50
use_cache: true
out_sample_rate: 24000
override_conditioning_features:
bad_text_proba: 0.0
c50: 0.0
dmcs_flatness: 100500.0
dmcs_roll_off_0.995: 100500.0
emo2vec: null
emotion: null
is_animation: 0.0
pitch_std: 100.0
snr: 100.0
year: 2025.0
reranking_options:
mode: MBR
top_k: 1
vocoder: bigvgan
voice_samples_preprocessing: []
num_workers: 1
output_dir: cc_20250825/yt4_wavtokenizer_16K_lossent0.15_indexrefencoder_separate_self_movies2_finetune_full_v2__yt4_wavtokenizer_16K_lossent0.15_indexrefencoder_separate_self__2026-01-23_17-32-54
ref_dir: cc_20250825/ref
ticket: QUALITY-41
basket_generation_git_hash: 7ba982d9bb8ddc0cb968d517f583b0227d2624ed
model_data_type: tts-cloning
ticket: QUALITY-41
version: 2026-01-23_17-32-54
|
movies2_finetune_closest_anim_notgtlen
basket_config_path: quality/tts/tortoise-baskets/cc_20250825_en-US.json
data_meta: null
exp_name: yt4_wavtokenizer_16K_lossent0.15_movies2_finetune_closest__yt4_wavtokenizer_16K_lossent0.15
lang: en-us
meta:
basket_generation_config:
basket_lang: en-us
basket_path: quality/tts/tortoise-baskets/cc_20250825_en-US.json
batch_size: 1
gpus: 1
inference:
condition_sample_rate: 24000
diff_k: 3
diff_steps: 100
diffusion_exp: /mount/s3/tts-binary-data-nb/eg/exp/yt4_wavtokenizer_16K_lossent0.15
duplicate_reference: true
exp: /mount/s3/tts-binary-data-nb/eg/exp/yt4_wavtokenizer_16K_lossent0.15_movies2_finetune_closest
gpt_generate_args:
do_sample: true
min_new_tokens: 20
num_return_sequences: 50
use_cache: true
out_sample_rate: 24000
override_conditioning_features:
bad_text_proba: 0.0
c50: 0.0
dmcs_flatness: 100500.0
dmcs_roll_off_0.995: 100500.0
emo2vec: null
emotion: null
is_animation: 1.0
pitch_std: 100.0
snr: 100.0
year: 2025.0
reranking_options:
mode: MBR
top_k: 1
vocoder: bigvgan
voice_samples_preprocessing: []
num_workers: 1
output_dir: cc_20250825/yt4_wavtokenizer_16K_lossent0.15_movies2_finetune_closest__yt4_wavtokenizer_16K_lossent0.15__2026-01-23_18-01-37
ref_dir: cc_20250825/ref
ticket: QUALITY-41
basket_generation_git_hash: 7ba982d9bb8ddc0cb968d517f583b0227d2624ed
model_data_type: tts-cloning
ticket: QUALITY-41
version: 2026-01-23_18-01-37
|
indexrefencoder_ft_movies3_closest_enus_notgtlen
basket_config_path: quality/tts/tortoise-baskets/cc_20250825_en-US.json
data_meta: null
exp_name: yt4_wavtokenizer_16K_lossent0.15_indexrefencoder_separate_self_movies3_finetune_full_v2_closest__yt4_wavtokenizer_16K_lossent0.15_indexrefencoder_separate_self
lang: en-us
meta:
basket_generation_config:
basket_lang: en-us
basket_path: quality/tts/tortoise-baskets/cc_20250825_en-US.json
batch_size: 1
gpus: 1
inference:
condition_sample_rate: 24000
diff_k: 3
diff_steps: 100
diffusion_exp: /mount/s3/tts-binary-data-nb/eg/exp/yt4_wavtokenizer_16K_lossent0.15_indexrefencoder_separate_self
duplicate_reference: true
exp: /mount/s3/tts-binary-data-nb/eg/exp/yt4_wavtokenizer_16K_lossent0.15_indexrefencoder_separate_self_movies3_finetune_full_v2_closest
gpt_generate_args:
do_sample: true
min_new_tokens: 20
num_return_sequences: 50
use_cache: true
out_sample_rate: 24000
override_conditioning_features:
bad_text_proba: 0.0
c50: 0.0
dmcs_flatness: 100500.0
dmcs_roll_off_0.995: 100500.0
emo2vec: null
emotion: null
is_animation: 1.0
pitch_std: 100.0
snr: 100.0
year: 2025.0
reranking_options:
mode: MBR
top_k: 1
vocoder: bigvgan
voice_samples_preprocessing: []
num_workers: 1
output_dir: cc_20250825/yt4_wavtokenizer_16K_lossent0.15_indexrefencoder_separate_self_movies3_finetune_full_v2_closest__yt4_wavtokenizer_16K_lossent0.15_indexrefencoder_separate_self__2026-01-26_11-26-27
ref_dir: cc_20250825/ref
ticket: QUALITY-41
basket_generation_git_hash: 7ba982d9bb8ddc0cb968d517f583b0227d2624ed
model_data_type: tts-cloning
ticket: QUALITY-41
version: 2026-01-26_11-26-27
|
indexrefencoder_ft_movies3_20Kit_closest_enus_notgtlen
basket_config_path: quality/tts/tortoise-baskets/cc_20250825_en-US.json
data_meta: null
exp_name: yt4_wavtokenizer_16K_lossent0.15_indexrefencoder_separate_self_movies3_finetune_full_v2_closest__yt4_wavtokenizer_16K_lossent0.15_indexrefencoder_separate_self
lang: en-us
meta:
basket_generation_config:
basket_lang: en-us
basket_path: quality/tts/tortoise-baskets/cc_20250825_en-US.json
batch_size: 1
gpus: 1
inference:
condition_sample_rate: 24000
diff_k: 3
diff_steps: 100
diffusion_exp: /mount/s3/tts-binary-data-nb/eg/exp/yt4_wavtokenizer_16K_lossent0.15_indexrefencoder_separate_self
duplicate_reference: true
exp: /mount/s3/tts-binary-data-nb/eg/exp/yt4_wavtokenizer_16K_lossent0.15_indexrefencoder_separate_self_movies3_finetune_full_v2_closest
gpt_generate_args:
do_sample: true
min_new_tokens: 20
num_return_sequences: 50
use_cache: true
out_sample_rate: 24000
override_conditioning_features:
bad_text_proba: 0.0
c50: 0.0
dmcs_flatness: 100500.0
dmcs_roll_off_0.995: 100500.0
emo2vec: null
emotion: null
is_animation: 1.0
pitch_std: 100.0
snr: 100.0
year: 2025.0
reranking_options:
mode: MBR
top_k: 1
vocoder: bigvgan
voice_samples_preprocessing: []
num_workers: 1
output_dir: cc_20250825/yt4_wavtokenizer_16K_lossent0.15_indexrefencoder_separate_self_movies3_finetune_full_v2_closest__yt4_wavtokenizer_16K_lossent0.15_indexrefencoder_separate_self__2026-01-28_13-09-10
ref_dir: cc_20250825/ref
ticket: QUALITY-41
basket_generation_git_hash: 7ba982d9bb8ddc0cb968d517f583b0227d2624ed
model_data_type: tts-cloning
ticket: QUALITY-41
version: 2026-01-28_13-09-10
|
indexrefencoder_noft_enus_notgtlen
basket_config_path: quality/tts/tortoise-baskets/cc_20250825_en-US.json
data_meta: null
exp_name: yt4_wavtokenizer_16K_lossent0.15_indexrefencoder_separate_self__yt4_wavtokenizer_16K_lossent0.15_indexrefencoder_separate_self
lang: en-us
meta:
basket_generation_config:
basket_lang: en-us
basket_path: quality/tts/tortoise-baskets/cc_20250825_en-US.json
batch_size: 1
gpus: 1
inference:
condition_sample_rate: 24000
diff_k: 3
diff_steps: 100
diffusion_exp: /mount/s3/tts-binary-data-nb/eg/exp/yt4_wavtokenizer_16K_lossent0.15_indexrefencoder_separate_self
duplicate_reference: true
exp: /mount/s3/tts-binary-data-nb/eg/exp/yt4_wavtokenizer_16K_lossent0.15_indexrefencoder_separate_self
gpt_generate_args:
do_sample: true
min_new_tokens: 20
num_return_sequences: 50
use_cache: true
out_sample_rate: 24000
override_conditioning_features:
bad_text_proba: 0.0
c50: 0.0
dmcs_flatness: 100500.0
dmcs_roll_off_0.995: 100500.0
emo2vec: null
emotion: null
is_animation: 1.0
pitch_std: 100.0
snr: 100.0
year: 2025.0
reranking_options:
mode: MBR
top_k: 1
vocoder: bigvgan
voice_samples_preprocessing: []
num_workers: 1
output_dir: cc_20250825/yt4_wavtokenizer_16K_lossent0.15_indexrefencoder_separate_self__yt4_wavtokenizer_16K_lossent0.15_indexrefencoder_separate_self__2026-01-28_16-42-41
ref_dir: cc_20250825/ref
ticket: QUALITY-41
basket_generation_git_hash: 7ba982d9bb8ddc0cb968d517f583b0227d2624ed
model_data_type: tts-cloning
ticket: QUALITY-41
version: 2026-01-28_16-42-41
|
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
|
DF-creative-commons-basket/cJvDUasu9pI_hi/M1__9.802-11.260
|
Don't make such faces.
|
|||||||||||||
|
DF-creative-commons-basket/U_1cQLJKt7c_es/M0__0.200-0.570
|
Tell me.
|
|||||||||||||
|
DF-creative-commons-basket/mtohEFEZLBE_pt/F1__17.400-19.630
|
It may sound strange, but I'm a victim.
|
|||||||||||||
|
DF-creative-commons-basket/l-l9x4bLVUY_zh/M0__2.355-3.949
|
I'm still willing to be there with you.
|
|||||||||||||
|
DF-creative-commons-basket/J9RB0FjZwyM_zh/M0__18.194-19.234
|
At this moment.
|
|||||||||||||
|
DF-creative-commons-basket/UXx6zn9y2B8_es/F0__16.050-20.460
|
Because once we identify them this way, we can no longer keep justifying them.
|
|||||||||||||
|
DF-creative-commons-basket/qJaAnEUiO6E_it/F0__8.920-10.925
|
It can happen, but it can be treated.
|
|||||||||||||
|
DF-creative-commons-basket/g7TkB4OzEAg_hi/F0__13.060-15.080
|
Yes, yes, I'll message you when I am there.
|
|||||||||||||
|
DF-creative-commons-basket/x_gaj_ZhDqs_de/F0__13.615-15.695
|
What on earth were you thinking?
|
|||||||||||||
|
DF-creative-commons-basket/jdSyxrG6dfM_it/F0__10.803-11.485
|
Alright.
|
|||||||||||||
|
DF-creative-commons-basket/jdSyxrG6dfM_it/F0__13.783-17.330
|
Riccardo, did you forget something important?
|
|||||||||||||
|
DF-creative-commons-basket/Ll6fcDRKi9k_ru/F1__1.760-4.218
|
Dad, really, I even took a test, want to see the results?
|
|||||||||||||
|
DF-creative-commons-basket/qJaAnEUiO6E_it/M0__10.975-13.170
|
The person with the disorder is not the disorder.
|
|||||||||||||
|
DF-creative-commons-basket/7tY36H7eY58_pt/M0__16.309-18.310
|
One, two, three, four, eighth floor, look there.
|
|||||||||||||
|
DF-creative-commons-basket/HibKRdG8Ie0_es/F0__6.804-8.544
|
That was it, and it's not the first time this has happened.
|
|||||||||||||
|
DF-creative-commons-basket/P6glOSnV2gM_fr/F0__21.290-23.575
|
Exactly — you can see I can’t handle it.
|
|||||||||||||
|
DF-creative-commons-basket/mtohEFEZLBE_pt/M0__13.030-14.790
|
You looked like a victim in the photo.
|
|||||||||||||
|
DF-creative-commons-basket/lTa1VcbYZEE_fr/F0__0.330-4.550
|
Yes, we had a very pleasant and quiet evening.
|
|||||||||||||
|
DF-creative-commons-basket/eN-waporon0_de/M0__7.005-7.805
|
I just couldn't do it.
|
|||||||||||||
|
DF-creative-commons-basket/SdpDal_Tu2s_hi/F1__8.720-12.020
|
Aunty, haven't you decorated your shop for Diwali?
|