End of training

Browse files

Files changed (6) hide show

README.md +76 -0
config.json +92 -0
generation_config.json +9 -0
model.safetensors +3 -0
runs/Apr22_15-15-29_vijaysMacStudio/events.out.tfevents.1745306240.vijaysMacStudio.24978.1 +3 -0
training_args.bin +3 -0

README.md ADDED Viewed

	@@ -0,0 +1,76 @@

+---
+license: mit
+base_model: microsoft/speecht5_tts
+tags:
+- generated_from_trainer
+model-index:
+- name: speecht5_finetuned_k_voice_v3
+  results: []
+---
+<!-- This model card has been generated automatically according to the information the Trainer had access to. You
+should probably proofread and complete it, then remove this comment. -->
+# speecht5_finetuned_k_voice_v3
+This model is a fine-tuned version of [microsoft/speecht5_tts](https://huggingface.co/microsoft/speecht5_tts) on an unknown dataset.
+It achieves the following results on the evaluation set:
+- Loss: 0.4215
+## Model description
+More information needed
+## Intended uses & limitations
+More information needed
+## Training and evaluation data
+More information needed
+## Training procedure
+### Training hyperparameters
+The following hyperparameters were used during training:
+- learning_rate: 5e-06
+- train_batch_size: 2
+- eval_batch_size: 8
+- seed: 42
+- gradient_accumulation_steps: 2
+- total_train_batch_size: 4
+- optimizer: Adam with betas=(0.9,0.999) and epsilon=1e-08
+- lr_scheduler_type: cosine
+- lr_scheduler_warmup_steps: 200
+- num_epochs: 1
+### Training results
+| Training Loss | Epoch  | Step | Validation Loss |
+|:-------------:|:------:|:----:|:---------------:|
+| 0.6081        | 0.2579 | 500  | 0.5853          |
+| 0.5601        | 0.5159 | 1000 | 0.4924          |
+| 0.5331        | 0.7738 | 1500 | 0.4709          |
+| 0.5263        | 1.0317 | 2000 | 0.4698          |
+| 0.5151        | 1.2897 | 2500 | 0.4615          |
+| 0.5172        | 1.5476 | 3000 | 0.4618          |
+| 0.5109        | 1.8055 | 3500 | 0.4486          |
+| 0.5025        | 2.0635 | 4000 | 0.4461          |
+| 0.4822        | 2.3214 | 4500 | 0.4356          |
+| 0.4911        | 2.5793 | 5000 | 0.4439          |
+| 0.4931        | 2.8372 | 5500 | 0.4331          |
+| 0.4904        | 3.0952 | 6000 | 0.4304          |
+| 0.474         | 3.3531 | 6500 | 0.4330          |
+| 0.4716        | 3.6110 | 7000 | 0.4307          |
+| 0.4673        | 3.8690 | 7500 | 0.4274          |
+| 0.4654        | 4.1269 | 8000 | 0.4250          |
+| 0.4609        | 4.3848 | 8500 | 0.4215          |
+### Framework versions
+- Transformers 4.40.1
+- Pytorch 2.6.0
+- Datasets 3.5.0
+- Tokenizers 0.19.1

config.json ADDED Viewed

	@@ -0,0 +1,92 @@

+{
+  "_name_or_path": "microsoft/speecht5_tts",
+  "activation_dropout": 0.1,
+  "apply_spec_augment": true,
+  "architectures": [
+    "SpeechT5ForTextToSpeech"
+  ],
+  "attention_dropout": 0.1,
+  "bos_token_id": 0,
+  "conv_bias": false,
+  "conv_dim": [
+    512,
+    512,
+    512,
+    512,
+    512,
+    512,
+    512
+  ],
+  "conv_kernel": [
+    10,
+    3,
+    3,
+    3,
+    3,
+    2,
+    2
+  ],
+  "conv_stride": [
+    5,
+    2,
+    2,
+    2,
+    2,
+    2,
+    2
+  ],
+  "decoder_attention_heads": 12,
+  "decoder_ffn_dim": 3072,
+  "decoder_layerdrop": 0.1,
+  "decoder_layers": 6,
+  "decoder_start_token_id": 2,
+  "encoder_attention_heads": 12,
+  "encoder_ffn_dim": 3072,
+  "encoder_layerdrop": 0.1,
+  "encoder_layers": 12,
+  "encoder_max_relative_position": 160,
+  "eos_token_id": 2,
+  "feat_extract_activation": "gelu",
+  "feat_extract_norm": "group",
+  "feat_proj_dropout": 0.0,
+  "guided_attention_loss_num_heads": 2,
+  "guided_attention_loss_scale": 10.0,
+  "guided_attention_loss_sigma": 0.4,
+  "hidden_act": "gelu",
+  "hidden_dropout": 0.1,
+  "hidden_size": 768,
+  "initializer_range": 0.02,
+  "is_encoder_decoder": true,
+  "layer_norm_eps": 1e-05,
+  "mask_feature_length": 10,
+  "mask_feature_min_masks": 0,
+  "mask_feature_prob": 0.0,
+  "mask_time_length": 10,
+  "mask_time_min_masks": 2,
+  "mask_time_prob": 0.05,
+  "max_length": 1876,
+  "max_speech_positions": 1876,
+  "max_text_positions": 600,
+  "model_type": "speecht5",
+  "num_conv_pos_embedding_groups": 16,
+  "num_conv_pos_embeddings": 128,
+  "num_feat_extract_layers": 7,
+  "num_mel_bins": 80,
+  "pad_token_id": 1,
+  "positional_dropout": 0.1,
+  "reduction_factor": 2,
+  "scale_embedding": false,
+  "speaker_embedding_dim": 512,
+  "speech_decoder_postnet_dropout": 0.5,
+  "speech_decoder_postnet_kernel": 5,
+  "speech_decoder_postnet_layers": 5,
+  "speech_decoder_postnet_units": 256,
+  "speech_decoder_prenet_dropout": 0.5,
+  "speech_decoder_prenet_layers": 2,
+  "speech_decoder_prenet_units": 256,
+  "torch_dtype": "float32",
+  "transformers_version": "4.40.1",
+  "use_cache": false,
+  "use_guided_attention_loss": true,
+  "vocab_size": 81
+}

generation_config.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "_from_model_config": true,
+  "bos_token_id": 0,
+  "decoder_start_token_id": 2,
+  "eos_token_id": 2,
+  "max_length": 1876,
+  "pad_token_id": 1,
+  "transformers_version": "4.40.1"
+}

model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:64c9f096326a74353ef28e26dc6fcd5059ddb7d29586d290b6d75df6c8281f92
+size 577789320

runs/Apr22_15-15-29_vijaysMacStudio/events.out.tfevents.1745306240.vijaysMacStudio.24978.1 ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:bcce9ec53fd86b145d170f80472806183736c3c698cbc36fec205644d20e6888
+size 6686

training_args.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:c9db70984319350e5da8fdd4bd8b7cbf1f8d1a0a0140edcaa87dacbf831f5df6
+size 5112