overload7015 commited on
Commit
7039efb
·
1 Parent(s): b75291e

Upload 9 files

Browse files
Chtholly_V5Co-1076epoch-80800step-Vec768-Layer12.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b054ad03b0822227050fac368da6602563c1500ee38dcb2c4df5e969801f6abd
3
+ size 627905309
Chtholly_V5Co-1076epoch-80800step-Vec768-Layer12_compressed.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6dd4464efe9f017c44bdfbd77073279bafec9ad1d17839ff402deb89e8cfb3d9
3
+ size 209255614
Chtholly_V5Sp-527epoch-21600step-Vec768-Layer12.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:591324a80369d7a768d93eb448bc059f1577c1277cc0e3644ed20eaa129c3531
3
+ size 627905309
Chtholly_V5Sp-527epoch-21600step-Vec768-Layer12_compressed.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8bef984ab6e7cfa46996e26fb1523705cee692ac0478b8bfc1c0088cc4e18466
3
+ size 209255614
Chtholly_V5_config.json ADDED
@@ -0,0 +1,96 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "train": {
3
+ "log_interval": 200,
4
+ "eval_interval": 800,
5
+ "seed": 1234,
6
+ "epochs": 10000,
7
+ "learning_rate": 0.0001,
8
+ "betas": [
9
+ 0.8,
10
+ 0.99
11
+ ],
12
+ "eps": 1e-09,
13
+ "batch_size": 12,
14
+ "fp16_run": false,
15
+ "lr_decay": 0.999875,
16
+ "segment_size": 10240,
17
+ "init_lr_ratio": 1,
18
+ "warmup_epochs": 0,
19
+ "c_mel": 45,
20
+ "c_kl": 1.0,
21
+ "use_sr": true,
22
+ "max_speclen": 512,
23
+ "port": "8001",
24
+ "keep_ckpts": 10,
25
+ "all_in_mem": false
26
+ },
27
+ "data": {
28
+ "training_files": "filelists/train.txt",
29
+ "validation_files": "filelists/val.txt",
30
+ "max_wav_value": 32768.0,
31
+ "sampling_rate": 44100,
32
+ "filter_length": 2048,
33
+ "hop_length": 512,
34
+ "win_length": 2048,
35
+ "n_mel_channels": 80,
36
+ "mel_fmin": 0.0,
37
+ "mel_fmax": 22050
38
+ },
39
+ "model": {
40
+ "inter_channels": 192,
41
+ "hidden_channels": 192,
42
+ "filter_channels": 768,
43
+ "n_heads": 2,
44
+ "n_layers": 6,
45
+ "kernel_size": 3,
46
+ "p_dropout": 0.1,
47
+ "resblock": "1",
48
+ "resblock_kernel_sizes": [
49
+ 3,
50
+ 7,
51
+ 11
52
+ ],
53
+ "resblock_dilation_sizes": [
54
+ [
55
+ 1,
56
+ 3,
57
+ 5
58
+ ],
59
+ [
60
+ 1,
61
+ 3,
62
+ 5
63
+ ],
64
+ [
65
+ 1,
66
+ 3,
67
+ 5
68
+ ]
69
+ ],
70
+ "upsample_rates": [
71
+ 8,
72
+ 8,
73
+ 2,
74
+ 2,
75
+ 2
76
+ ],
77
+ "upsample_initial_channel": 512,
78
+ "upsample_kernel_sizes": [
79
+ 16,
80
+ 16,
81
+ 4,
82
+ 4,
83
+ 4
84
+ ],
85
+ "n_layers_q": 3,
86
+ "use_spectral_norm": false,
87
+ "gin_channels": 768,
88
+ "ssl_dim": 768,
89
+ "n_speakers": 1,
90
+ "speech_encoder": "vec768l12",
91
+ "speaker_embedding": false
92
+ },
93
+ "spk": {
94
+ "Chtholly_V5": 0
95
+ }
96
+ }
Chtholly_V5_config.yaml ADDED
@@ -0,0 +1,48 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ data:
2
+ block_size: 512
3
+ cnhubertsoft_gate: 10
4
+ duration: 1
5
+ encoder: vec768l12
6
+ encoder_hop_size: 320
7
+ encoder_out_channels: 768
8
+ encoder_sample_rate: 16000
9
+ extensions:
10
+ - wav
11
+ sampling_rate: 44100
12
+ training_files: filelists/train.txt
13
+ validation_files: filelists/val.txt
14
+ device: cuda
15
+ env:
16
+ expdir: logs/44k/diffusion
17
+ gpu_id: 0
18
+ infer:
19
+ method: dpm-solver
20
+ speedup: 10
21
+ model:
22
+ n_chans: 512
23
+ n_hidden: 256
24
+ n_layers: 20
25
+ n_spk: 1
26
+ type: Diffusion
27
+ use_pitch_aug: true
28
+ spk:
29
+ Chtholly_V5: 0
30
+ train:
31
+ amp_dtype: fp32
32
+ batch_size: 48
33
+ cache_all_data: true
34
+ cache_device: cpu
35
+ cache_fp16: true
36
+ decay_step: 100000
37
+ epochs: 100000
38
+ gamma: 0.5
39
+ interval_force_save: 10000
40
+ interval_log: 10
41
+ interval_val: 2000
42
+ lr: 0.0002
43
+ num_workers: 2
44
+ save_opt: false
45
+ weight_decay: 0
46
+ vocoder:
47
+ ckpt: pretrain/nsf_hifigan/model
48
+ type: nsf-hifigan
Chtholly_V5_kmeans_10000.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:788ae6d94f953764ca3605bf4996f2ee61964058d4b1d858624ed50446d08168
3
+ size 32539129
Chtholly_V5_model_52000.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0cfd0b35c8049b380107417597c4029c33c1e81f9593973e1cee078ebd3db3ef
3
+ size 220893767
/345/243/260/347/272/271/347/233/270/344/274/274/345/272/246.png ADDED