Spaces:

the-cramer-project
/

AkylAI_TTS_small

Runtime error

App Files Files Community

Simonlob commited on May 29

Commit

cf1295a

•

1 Parent(s): 7e928b3

Second

Browse files

This view is limited to 50 files because it contains too many changes. See raw diff

Files changed (50) hide show

LICENSE +21 -0
MANIFEST.in +14 -0
Makefile +42 -0
README.md:Zone.Identifier +4 -0
app.py +174 -0
checkpoints/info.txt +1 -0
configs/__init__.py +1 -0
configs/callbacks/default.yaml +5 -0
configs/callbacks/model_checkpoint.yaml +17 -0
configs/callbacks/model_summary.yaml +5 -0
configs/callbacks/none.yaml +0 -0
configs/callbacks/rich_progress_bar.yaml +4 -0
configs/data/akylai.yaml +21 -0
configs/data/akylai_multi.yaml +21 -0
configs/data/hi-fi_en-US_female.yaml +14 -0
configs/data/ljspeech.yaml +22 -0
configs/data/vctk.yaml +14 -0
configs/debug/default.yaml +35 -0
configs/debug/fdr.yaml +9 -0
configs/debug/limit.yaml +12 -0
configs/debug/overfit.yaml +13 -0
configs/debug/profiler.yaml +15 -0
configs/eval.yaml +18 -0
configs/experiment/akylai.yaml +14 -0
configs/experiment/akylai_multi.yaml +14 -0
configs/experiment/hifi_dataset_piper_phonemizer.yaml +14 -0
configs/experiment/ljspeech.yaml +14 -0
configs/experiment/ljspeech_min_memory.yaml +18 -0
configs/experiment/multispeaker.yaml +14 -0
configs/extras/default.yaml +8 -0
configs/hparams_search/mnist_optuna.yaml +52 -0
configs/hydra/default.yaml +19 -0
configs/local/.gitkeep +0 -0
configs/logger/aim.yaml +28 -0
configs/logger/comet.yaml +12 -0
configs/logger/csv.yaml +7 -0
configs/logger/many_loggers.yaml +9 -0
configs/logger/mlflow.yaml +12 -0
configs/logger/neptune.yaml +9 -0
configs/logger/tensorboard.yaml +10 -0
configs/logger/wandb.yaml +16 -0
configs/model/cfm/default.yaml +3 -0
configs/model/decoder/default.yaml +7 -0
configs/model/encoder/default.yaml +18 -0
configs/model/matcha.yaml +15 -0
configs/model/optimizer/adam.yaml +4 -0
configs/paths/default.yaml +18 -0
configs/train.yaml +51 -0
configs/trainer/cpu.yaml +5 -0
configs/trainer/ddp.yaml +9 -0

LICENSE ADDED Viewed

	@@ -0,0 +1,21 @@

+MIT License
+Copyright (c) 2023 Shivam Mehta
+Permission is hereby granted, free of charge, to any person obtaining a copy
+of this software and associated documentation files (the "Software"), to deal
+in the Software without restriction, including without limitation the rights
+to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
+copies of the Software, and to permit persons to whom the Software is
+furnished to do so, subject to the following conditions:
+The above copyright notice and this permission notice shall be included in all
+copies or substantial portions of the Software.
+THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
+IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
+FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
+AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
+LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
+OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
+SOFTWARE.

MANIFEST.in ADDED Viewed

	@@ -0,0 +1,14 @@

+include README.md
+include LICENSE.txt
+include requirements.*.txt
+include *.cff
+include requirements.txt
+include matcha/VERSION
+recursive-include matcha *.json
+recursive-include matcha *.html
+recursive-include matcha *.png
+recursive-include matcha *.md
+recursive-include matcha *.py
+recursive-include matcha *.pyx
+recursive-exclude tests *
+prune tests*

Makefile ADDED Viewed

	@@ -0,0 +1,42 @@

+help:  ## Show help
+	@grep -E '^[.a-zA-Z_-]+:.*?## .*$$' $(MAKEFILE_LIST) | awk 'BEGIN {FS = ":.*?## "}; {printf "\033[36m%-30s\033[0m %s\n", $$1, $$2}'
+clean: ## Clean autogenerated files
+	rm -rf dist
+	find . -type f -name "*.DS_Store" -ls -delete
+	find . | grep -E "(__pycache__|\.pyc|\.pyo)" | xargs rm -rf
+	find . | grep -E ".pytest_cache" | xargs rm -rf
+	find . | grep -E ".ipynb_checkpoints" | xargs rm -rf
+	rm -f .coverage
+clean-logs: ## Clean logs
+	rm -rf logs/**
+create-package: ## Create wheel and tar gz
+	rm -rf dist/
+	python setup.py bdist_wheel --plat-name=manylinux1_x86_64
+	python setup.py sdist
+	python -m twine upload  dist/* --verbose --skip-existing
+format: ## Run pre-commit hooks
+	pre-commit run -a
+sync: ## Merge changes from main branch to your current branch
+	git pull
+	git pull origin main
+test: ## Run not slow tests
+	pytest -k "not slow"
+test-full: ## Run all tests
+	pytest
+train-ljspeech: ## Train the model
+	python matcha/train.py experiment=ljspeech
+train-ljspeech-min: ## Train the model with minimum memory
+	python matcha/train.py experiment=ljspeech_min_memory
+start_app: ## Start the app
+	python matcha/app.py

README.md:Zone.Identifier ADDED Viewed

	@@ -0,0 +1,4 @@

+[ZoneTransfer]
+ZoneId=3
+ReferrerUrl=https://huggingface.co/spaces/the-cramer-project/AkylAI_TTS_small/tree/main
+HostUrl=https://huggingface.co/spaces/the-cramer-project/AkylAI_TTS_small/resolve/main/README.md?download=true

app.py ADDED Viewed

	@@ -0,0 +1,174 @@

+from pathlib import Path
+import argparse
+import soundfile as sf
+import torch
+import io
+import argparse
+from matcha.hifigan.config import v1
+from matcha.hifigan.denoiser import Denoiser
+from matcha.hifigan.env import AttrDict
+from matcha.hifigan.models import Generator as HiFiGAN
+from matcha.models.matcha_tts import MatchaTTS
+from matcha.text import sequence_to_text, text_to_sequence
+from matcha.utils.utils import intersperse
+import gradio as gr
+import requests
+def download_file(url, save_path):
+    response = requests.get(url)
+    with open(save_path, 'wb') as file:
+        file.write(response.content)
+url_checkpoint = 'https://github.com/simonlobgromov/AkylAI_Matcha_Checkpoint/releases/download/Matcha-TTS/checkpoint_epoch.499.ckpt'
+save_checkpoint_path = './checkpoints/checkpoint.ckpt'
+url_generator = 'https://github.com/simonlobgromov/AkylAI_Matcha_HiFiGan/releases/download/Generator/generator_v1'
+save_generator_path = './checkpoints/generator'
+download_file(url_checkpoint, save_checkpoint_path)
+download_file(url_generator, save_generator_path)
+def load_matcha( checkpoint_path, device):
+    model = MatchaTTS.load_from_checkpoint(checkpoint_path, map_location=device)
+    _ = model.eval()
+    return model
+def load_hifigan(checkpoint_path, device):
+    h = AttrDict(v1)
+    hifigan = HiFiGAN(h).to(device)
+    hifigan.load_state_dict(torch.load(checkpoint_path, map_location=device)["generator"])
+    _ = hifigan.eval()
+    hifigan.remove_weight_norm()
+    return hifigan
+def load_vocoder(checkpoint_path, device):
+    vocoder = None
+    vocoder = load_hifigan(checkpoint_path, device)
+    denoiser = Denoiser(vocoder, mode="zeros")
+    return vocoder, denoiser
+def process_text(i: int, text: str, device: torch.device):
+    print(f"[{i}] - Input text: {text}")
+    x = torch.tensor(
+        intersperse(text_to_sequence(text, ["kyrgyz_cleaners"]), 0),
+        dtype=torch.long,
+        device=device,
+    )[None]
+    x_lengths = torch.tensor([x.shape[-1]], dtype=torch.long, device=device)
+    x_phones = sequence_to_text(x.squeeze(0).tolist())
+    print(f"[{i}] - Phonetised text: {x_phones[1::2]}")
+    return {"x_orig": text, "x": x, "x_lengths": x_lengths, "x_phones": x_phones}
+def to_waveform(mel, vocoder, denoiser=None):
+    audio = vocoder(mel).clamp(-1, 1)
+    if denoiser is not None:
+        audio = denoiser(audio.squeeze(), strength=0.00025).cpu().squeeze()
+    return audio.cpu().squeeze()
+@torch.inference_mode()
+def process_text_gradio(text):
+    output = process_text(1, text, device)
+    return output["x_phones"][1::2], output["x"], output["x_lengths"]
+@torch.inference_mode()
+def synthesise_mel(text, text_length, n_timesteps, temperature, length_scale, spk=-1):
+    spk = torch.tensor([spk], device=device, dtype=torch.long) if spk >= 0 else None
+    output = model.synthesise(
+        text,
+        text_length,
+        n_timesteps=n_timesteps,
+        temperature=temperature,
+        spks=spk,
+        length_scale=length_scale,
+    )
+    output["waveform"] = to_waveform(output["mel"], vocoder, denoiser)
+    return output["waveform"].numpy()
+def get_inference(text, n_timesteps=20, mel_temp = 0.667, length_scale=0.8, spk=-1):
+    phones, text, text_lengths = process_text_gradio(text)
+    print(type(synthesise_mel(text, text_lengths, n_timesteps, mel_temp, length_scale, spk)))
+    return synthesise_mel(text, text_lengths, n_timesteps, mel_temp, length_scale, spk)
+device = torch.device("cpu")
+model_path = './checkpoints/checkpoint.ckpt'
+vocoder_path = './checkpoints/generator'
+model = load_matcha(model_path, device)
+vocoder, denoiser = load_vocoder(vocoder_path, device)
+def gen_tts(text, speaking_rate):
+    return 22050, get_inference(text = text, length_scale = speaking_rate)
+default_text = "Баарыңарга салам, менин атым Акылай."
+css = """
+        #share-btn-container {
+            display: flex;
+            padding-left: 0.5rem !important;
+            padding-right: 0.5rem !important;
+            background-color: #000000;
+            justify-content: center;
+            align-items: center;
+            border-radius: 9999px !important;
+            width: 13rem;
+            margin-top: 10px;
+            margin-left: auto;
+            flex: unset !important;
+        }
+        #share-btn {
+            all: initial;
+            color: #ffffff;
+            font-weight: 600;
+            cursor: pointer;
+            font-family: 'IBM Plex Sans', sans-serif;
+            margin-left: 0.5rem !important;
+            padding-top: 0.25rem !important;
+            padding-bottom: 0.25rem !important;
+            right:0;
+        }
+        #share-btn * {
+            all: unset !important;
+        }
+        #share-btn-container div:nth-child(-n+2){
+            width: auto !important;
+            min-height: 0px !important;
+        }
+        #share-btn-container .wrap {
+            display: none !important;
+        }
+"""
+with gr.Blocks(css=css) as block:
+    gr.HTML(
+        """
+            <div style="text-align: center; max-width: 700px; margin: 0 auto;">
+              <div
+                style="
+                  display: inline-flex; align-items: center; gap: 0.8rem; font-size: 1.75rem;
+                "
+              >
+                <h1 style="font-weight: 900; margin-bottom: 7px; line-height: normal;">
+                  Akyl-AI TTS
+                </h1>
+              </div>
+            </div>
+        """
+    )
+    with gr.Row():
+        image_path = "./photo_2024-04-07_15-59-52.png"
+        gr.Image(image_path, label=None, width=660, height=315, show_label=False)
+    with gr.Row():
+        with gr.Column():
+            input_text = gr.Textbox(label="Input Text", lines=2, value=default_text, elem_id="input_text")
+            speaking_rate = gr.Slider(label='Speaking rate', minimum=0.5, maximum=1, step=0.05, value=0.8, interactive=True, show_label=True, elem_id="speaking_rate")
+            run_button = gr.Button("Generate Audio", variant="primary")
+        with gr.Column():
+            audio_out = gr.Audio(label="Parler-TTS generation", type="numpy", elem_id="audio_out")
+    inputs = [input_text, speaking_rate]
+    outputs = [audio_out]
+    run_button.click(fn=gen_tts, inputs=inputs, outputs=outputs, queue=True)
+block.queue()
+block.launch(share=True)

checkpoints/info.txt ADDED Viewed

	@@ -0,0 +1 @@


1	+ Забудь дорогу всяк сюда входящий!

configs/__init__.py ADDED Viewed

	@@ -0,0 +1 @@


1	+ # this file is needed here to include configs when building project as a package

configs/callbacks/default.yaml ADDED Viewed

	@@ -0,0 +1,5 @@

+defaults:
+  - model_checkpoint.yaml
+  - model_summary.yaml
+  - rich_progress_bar.yaml
+  - _self_

configs/callbacks/model_checkpoint.yaml ADDED Viewed

	@@ -0,0 +1,17 @@

+# https://lightning.ai/docs/pytorch/stable/api/lightning.pytorch.callbacks.ModelCheckpoint.html
+model_checkpoint:
+  _target_: lightning.pytorch.callbacks.ModelCheckpoint
+  dirpath: ${paths.output_dir}/checkpoints # directory to save the model file
+  filename: checkpoint_{epoch:03d}  # checkpoint filename
+  monitor: epoch # name of the logged metric which determines when model is improving
+  verbose: False # verbosity mode
+  save_last: true # additionally always save an exact copy of the last checkpoint to a file last.ckpt
+  save_top_k: 5 # save k best models (determined by above metric)
+  mode: "max" # "max" means higher metric value is better, can be also "min"
+  auto_insert_metric_name: True # when True, the checkpoints filenames will contain the metric name
+  save_weights_only: False # if True, then only the model’s weights will be saved
+  every_n_train_steps: null # number of training steps between checkpoints
+  train_time_interval: null # checkpoints are monitored at the specified time interval
+  every_n_epochs: 10 # number of epochs between checkpoints
+  save_on_train_epoch_end: null # whether to run checkpointing at the end of the training epoch or the end of validation

configs/callbacks/model_summary.yaml ADDED Viewed

	@@ -0,0 +1,5 @@

+# https://lightning.ai/docs/pytorch/stable/api/lightning.pytorch.callbacks.RichModelSummary.html
+model_summary:
+  _target_: lightning.pytorch.callbacks.RichModelSummary
+  max_depth: 3 # the maximum depth of layer nesting that the summary will include

configs/callbacks/none.yaml ADDED Viewed

File without changes

configs/callbacks/rich_progress_bar.yaml ADDED Viewed

	@@ -0,0 +1,4 @@

+# https://lightning.ai/docs/pytorch/latest/api/lightning.pytorch.callbacks.RichProgressBar.html
+rich_progress_bar:
+  _target_: lightning.pytorch.callbacks.RichProgressBar

configs/data/akylai.yaml ADDED Viewed

	@@ -0,0 +1,21 @@

+_target_: matcha.data.text_mel_datamodule.TextMelDataModule
+name: akylai
+train_filelist_path: ./Kany_dataset_mk4_v1/Kany_dataset_mk4_v1_filelist_train.txt
+valid_filelist_path: ./Kany_dataset_mk4_v1/Kany_dataset_mk4_v1_filelist_test.txt
+batch_size: 12
+num_workers: 12
+pin_memory: True
+cleaners: [kyrgyz_cleaners]
+add_blank: True
+n_spks: 1
+n_fft: 1024
+n_feats: 80
+sample_rate: 22050
+hop_length: 256
+win_length: 1024
+f_min: 0
+f_max: 8000
+data_statistics:  # Computed for ljspeech dataset
+  mel_mean: -5.638045310974121
+  mel_std: 2.6814498901367188
+seed: ${seed}

configs/data/akylai_multi.yaml ADDED Viewed

	@@ -0,0 +1,21 @@

+_target_: matcha.data.text_mel_datamodule.TextMelDataModule
+name: akylai_multi
+train_filelist_path: ./akylai_multi_dataset/akylai_mlspk_filelist_train.txt
+valid_filelist_path: ./akylai_multi_dataset/akylai_mlspk_filelist_test.txt
+batch_size: 32
+num_workers: 20
+pin_memory: True
+cleaners: [kyrgyz_cleaners]
+add_blank: True
+n_spks: 2
+n_fft: 1024
+n_feats: 80
+sample_rate: 22050
+hop_length: 256
+win_length: 1024
+f_min: 0
+f_max: 8000
+data_statistics:
+  mel_mean: -5.6814561
+  mel_std: 2.7337122
+seed: ${seed}

configs/data/hi-fi_en-US_female.yaml ADDED Viewed

	@@ -0,0 +1,14 @@

+defaults:
+  - ljspeech
+  - _self_
+# Dataset URL: https://ast-astrec.nict.go.jp/en/release/hi-fi-captain/
+_target_: matcha.data.text_mel_datamodule.TextMelDataModule
+name: hi-fi_en-US_female
+train_filelist_path: data/filelists/hi-fi-captain-en-us-female_train.txt
+valid_filelist_path: data/filelists/hi-fi-captain-en-us-female_val.txt
+batch_size: 32
+cleaners: [english_cleaners_piper]
+data_statistics:  # Computed for this dataset
+  mel_mean: -6.38385
+  mel_std: 2.541796

configs/data/ljspeech.yaml ADDED Viewed

	@@ -0,0 +1,22 @@

+_target_: matcha.data.text_mel_datamodule.TextMelDataModule
+name: ljspeech
+train_filelist_path: /content/kany_dataset/kany_filelist_train.txt
+valid_filelist_path: /content/kany_dataset/kany_filelist_test.txt
+batch_size: 16
+num_workers: 20
+pin_memory: True
+cleaners: [kyrgyz_cleaners]
+add_blank: True
+n_spks: 1
+n_fft: 1024
+n_feats: 80
+sample_rate: 22050
+hop_length: 256
+win_length: 1024
+f_min: 0
+f_max: 8000
+data_statistics:  # Computed for ljspeech dataset
+  mel_mean: -5.68145561
+  mel_std: 2.7337122
+seed: ${seed}

configs/data/vctk.yaml ADDED Viewed

	@@ -0,0 +1,14 @@

+defaults:
+  - ljspeech
+  - _self_
+_target_: matcha.data.text_mel_datamodule.TextMelDataModule
+name: vctk
+train_filelist_path: data/filelists/vctk_audio_sid_text_train_filelist.txt
+valid_filelist_path: data/filelists/vctk_audio_sid_text_val_filelist.txt
+batch_size: 32
+add_blank: True
+n_spks: 109
+data_statistics:  # Computed for vctk dataset
+  mel_mean: -6.630575
+  mel_std: 2.482914

configs/debug/default.yaml ADDED Viewed

	@@ -0,0 +1,35 @@

+# @package _global_
+# default debugging setup, runs 1 full epoch
+# other debugging configs can inherit from this one
+# overwrite task name so debugging logs are stored in separate folder
+task_name: "debug"
+# disable callbacks and loggers during debugging
+# callbacks: null
+# logger: null
+extras:
+  ignore_warnings: False
+  enforce_tags: False
+# sets level of all command line loggers to 'DEBUG'
+# https://hydra.cc/docs/tutorials/basic/running_your_app/logging/
+hydra:
+  job_logging:
+    root:
+      level: DEBUG
+  # use this to also set hydra loggers to 'DEBUG'
+  # verbose: True
+trainer:
+  max_epochs: 1
+  accelerator: cpu # debuggers don't like gpus
+  devices: 1 # debuggers don't like multiprocessing
+  detect_anomaly: true # raise exception if NaN or +/-inf is detected in any tensor
+data:
+  num_workers: 0 # debuggers don't like multiprocessing
+  pin_memory: False # disable gpu memory pin

configs/debug/fdr.yaml ADDED Viewed

	@@ -0,0 +1,9 @@

+# @package _global_
+# runs 1 train, 1 validation and 1 test step
+defaults:
+  - default
+trainer:
+  fast_dev_run: true

configs/debug/limit.yaml ADDED Viewed

	@@ -0,0 +1,12 @@

+# @package _global_
+# uses only 1% of the training data and 5% of validation/test data
+defaults:
+  - default
+trainer:
+  max_epochs: 3
+  limit_train_batches: 0.01
+  limit_val_batches: 0.05
+  limit_test_batches: 0.05

configs/debug/overfit.yaml ADDED Viewed

	@@ -0,0 +1,13 @@

+# @package _global_
+# overfits to 3 batches
+defaults:
+  - default
+trainer:
+  max_epochs: 20
+  overfit_batches: 3
+# model ckpt and early stopping need to be disabled during overfitting
+callbacks: null

configs/debug/profiler.yaml ADDED Viewed

	@@ -0,0 +1,15 @@

+# @package _global_
+# runs with execution time profiling
+defaults:
+  - default
+trainer:
+  max_epochs: 1
+  # profiler: "simple"
+  profiler: "advanced"
+  # profiler: "pytorch"
+  accelerator: gpu
+  limit_train_batches: 0.02

configs/eval.yaml ADDED Viewed

	@@ -0,0 +1,18 @@

+# @package _global_
+defaults:
+  - _self_
+  - data: akylai # choose datamodule with `test_dataloader()` for evaluation
+  - model: matcha
+  - logger: null
+  - trainer: default
+  - paths: default
+  - extras: default
+  - hydra: default
+task_name: "eval"
+tags: ["dev"]
+# passing checkpoint path is necessary for evaluation
+ckpt_path: ???

configs/experiment/akylai.yaml ADDED Viewed

	@@ -0,0 +1,14 @@

+# @package _global_
+# to execute this experiment run:
+# python train.py experiment=multispeaker
+defaults:
+  - override /data: akylai.yaml
+# all parameters below will be merged with parameters from default configurations set above
+# this allows you to overwrite only specified parameters
+tags: ["akylai"]
+run_name: akylai

configs/experiment/akylai_multi.yaml ADDED Viewed

	@@ -0,0 +1,14 @@

+# @package _global_
+# to execute this experiment run:
+# python train.py experiment=multispeaker
+defaults:
+  - override /data: akylai_multi.yaml
+# all parameters below will be merged with parameters from default configurations set above
+# this allows you to overwrite only specified parameters
+tags: ["akylai_multi"]
+run_name: akylai_multi

configs/experiment/hifi_dataset_piper_phonemizer.yaml ADDED Viewed

	@@ -0,0 +1,14 @@

+# @package _global_
+# to execute this experiment run:
+# python train.py experiment=multispeaker
+defaults:
+  - override /data: hi-fi_en-US_female.yaml
+# all parameters below will be merged with parameters from default configurations set above
+# this allows you to overwrite only specified parameters
+tags: ["hi-fi", "single_speaker", "piper_phonemizer", "en_US", "female"]
+run_name: hi-fi_en-US_female_piper_phonemizer

configs/experiment/ljspeech.yaml ADDED Viewed

	@@ -0,0 +1,14 @@

+# @package _global_
+# to execute this experiment run:
+# python train.py experiment=multispeaker
+defaults:
+  - override /data: ljspeech.yaml
+# all parameters below will be merged with parameters from default configurations set above
+# this allows you to overwrite only specified parameters
+tags: ["ljspeech"]
+run_name: ljspeech

configs/experiment/ljspeech_min_memory.yaml ADDED Viewed

	@@ -0,0 +1,18 @@

+# @package _global_
+# to execute this experiment run:
+# python train.py experiment=multispeaker
+defaults:
+  - override /data: ljspeech.yaml
+# all parameters below will be merged with parameters from default configurations set above
+# this allows you to overwrite only specified parameters
+tags: ["ljspeech"]
+run_name: ljspeech_min
+model:
+  out_size: 172

configs/experiment/multispeaker.yaml ADDED Viewed

	@@ -0,0 +1,14 @@

+# @package _global_
+# to execute this experiment run:
+# python train.py experiment=multispeaker
+defaults:
+  - override /data: vctk.yaml
+# all parameters below will be merged with parameters from default configurations set above
+# this allows you to overwrite only specified parameters
+tags: ["multispeaker"]
+run_name: multispeaker

configs/extras/default.yaml ADDED Viewed

	@@ -0,0 +1,8 @@

+# disable python warnings if they annoy you
+ignore_warnings: False
+# ask user for tags if none are provided in the config
+enforce_tags: True
+# pretty print config tree at the start of the run using Rich library
+print_config: True

configs/hparams_search/mnist_optuna.yaml ADDED Viewed

	@@ -0,0 +1,52 @@

+# @package _global_
+# example hyperparameter optimization of some experiment with Optuna:
+# python train.py -m hparams_search=mnist_optuna experiment=example
+defaults:
+  - override /hydra/sweeper: optuna
+# choose metric which will be optimized by Optuna
+# make sure this is the correct name of some metric logged in lightning module!
+optimized_metric: "val/acc_best"
+# here we define Optuna hyperparameter search
+# it optimizes for value returned from function with @hydra.main decorator
+# docs: https://hydra.cc/docs/next/plugins/optuna_sweeper
+hydra:
+  mode: "MULTIRUN" # set hydra to multirun by default if this config is attached
+  sweeper:
+    _target_: hydra_plugins.hydra_optuna_sweeper.optuna_sweeper.OptunaSweeper
+    # storage URL to persist optimization results
+    # for example, you can use SQLite if you set 'sqlite:///example.db'
+    storage: null
+    # name of the study to persist optimization results
+    study_name: null
+    # number of parallel workers
+    n_jobs: 1
+    # 'minimize' or 'maximize' the objective
+    direction: maximize
+    # total number of runs that will be executed
+    n_trials: 20
+    # choose Optuna hyperparameter sampler
+    # you can choose bayesian sampler (tpe), random search (without optimization), grid sampler, and others
+    # docs: https://optuna.readthedocs.io/en/stable/reference/samplers.html
+    sampler:
+      _target_: optuna.samplers.TPESampler
+      seed: 1234
+      n_startup_trials: 10 # number of random sampling runs before optimization starts
+    # define hyperparameter search space
+    params:
+      model.optimizer.lr: interval(0.0001, 0.1)
+      data.batch_size: choice(32, 64, 128, 256)
+      model.net.lin1_size: choice(64, 128, 256)
+      model.net.lin2_size: choice(64, 128, 256)
+      model.net.lin3_size: choice(32, 64, 128, 256)

configs/hydra/default.yaml ADDED Viewed

	@@ -0,0 +1,19 @@

+# https://hydra.cc/docs/configure_hydra/intro/
+# enable color logging
+defaults:
+  - override hydra_logging: colorlog
+  - override job_logging: colorlog
+# output directory, generated dynamically on each run
+run:
+  dir: ${paths.log_dir}/${task_name}/${run_name}/runs/${now:%Y-%m-%d}_${now:%H-%M-%S}
+sweep:
+  dir: ${paths.log_dir}/${task_name}/${run_name}/multiruns/${now:%Y-%m-%d}_${now:%H-%M-%S}
+  subdir: ${hydra.job.num}
+job_logging:
+  handlers:
+    file:
+      # Incorporates fix from https://github.com/facebookresearch/hydra/pull/2242
+      filename: ${hydra.runtime.output_dir}/${hydra.job.name}.log

configs/local/.gitkeep ADDED Viewed

File without changes

configs/logger/aim.yaml ADDED Viewed

	@@ -0,0 +1,28 @@

+# https://aimstack.io/
+# example usage in lightning module:
+# https://github.com/aimhubio/aim/blob/main/examples/pytorch_lightning_track.py
+# open the Aim UI with the following command (run in the folder containing the `.aim` folder):
+# `aim up`
+aim:
+  _target_: aim.pytorch_lightning.AimLogger
+  repo: ${paths.root_dir} # .aim folder will be created here
+  # repo: "aim://ip_address:port" # can instead provide IP address pointing to Aim remote tracking server which manages the repo, see https://aimstack.readthedocs.io/en/latest/using/remote_tracking.html#
+  # aim allows to group runs under experiment name
+  experiment: null # any string, set to "default" if not specified
+  train_metric_prefix: "train/"
+  val_metric_prefix: "val/"
+  test_metric_prefix: "test/"
+  # sets the tracking interval in seconds for system usage metrics (CPU, GPU, memory, etc.)
+  system_tracking_interval: 10 # set to null to disable system metrics tracking
+  # enable/disable logging of system params such as installed packages, git info, env vars, etc.
+  log_system_params: true
+  # enable/disable tracking console logs (default value is true)
+  capture_terminal_logs: false # set to false to avoid infinite console log loop issue https://github.com/aimhubio/aim/issues/2550

configs/logger/comet.yaml ADDED Viewed

	@@ -0,0 +1,12 @@

+# https://www.comet.ml
+comet:
+  _target_: lightning.pytorch.loggers.comet.CometLogger
+  api_key: ${oc.env:COMET_API_TOKEN} # api key is loaded from environment variable
+  save_dir: "${paths.output_dir}"
+  project_name: "lightning-hydra-template"
+  rest_api_key: null
+  # experiment_name: ""
+  experiment_key: null # set to resume experiment
+  offline: False
+  prefix: ""

configs/logger/csv.yaml ADDED Viewed

	@@ -0,0 +1,7 @@

+# csv logger built in lightning
+csv:
+  _target_: lightning.pytorch.loggers.csv_logs.CSVLogger
+  save_dir: "${paths.output_dir}"
+  name: "csv/"
+  prefix: ""

configs/logger/many_loggers.yaml ADDED Viewed

	@@ -0,0 +1,9 @@

+# train with many loggers at once
+defaults:
+  # - comet
+  - csv
+  # - mlflow
+  # - neptune
+  - tensorboard
+  - wandb

configs/logger/mlflow.yaml ADDED Viewed

	@@ -0,0 +1,12 @@

+# https://mlflow.org
+mlflow:
+  _target_: lightning.pytorch.loggers.mlflow.MLFlowLogger
+  # experiment_name: ""
+  # run_name: ""
+  tracking_uri: ${paths.log_dir}/mlflow/mlruns # run `mlflow ui` command inside the `logs/mlflow/` dir to open the UI
+  tags: null
+  # save_dir: "./mlruns"
+  prefix: ""
+  artifact_location: null
+  # run_id: ""

configs/logger/neptune.yaml ADDED Viewed

	@@ -0,0 +1,9 @@

+# https://neptune.ai
+neptune:
+  _target_: lightning.pytorch.loggers.neptune.NeptuneLogger
+  api_key: ${oc.env:NEPTUNE_API_TOKEN} # api key is loaded from environment variable
+  project: username/lightning-hydra-template
+  # name: ""
+  log_model_checkpoints: True
+  prefix: ""

configs/logger/tensorboard.yaml ADDED Viewed

	@@ -0,0 +1,10 @@

+# https://www.tensorflow.org/tensorboard/
+tensorboard:
+  _target_: lightning.pytorch.loggers.tensorboard.TensorBoardLogger
+  save_dir: "${paths.output_dir}/tensorboard/"
+  name: null
+  log_graph: False
+  default_hp_metric: True
+  prefix: ""
+  # version: ""

configs/logger/wandb.yaml ADDED Viewed

	@@ -0,0 +1,16 @@

+# https://wandb.ai
+wandb:
+  _target_: lightning.pytorch.loggers.wandb.WandbLogger
+  # name: "" # name of the run (normally generated by wandb)
+  save_dir: "${paths.output_dir}"
+  offline: False
+  id: null # pass correct id to resume experiment!
+  anonymous: null # enable anonymous logging
+  project: "lightning-hydra-template"
+  log_model: False # upload lightning ckpts
+  prefix: "" # a string to put at the beginning of metric keys
+  # entity: "" # set to name of your wandb team
+  group: ""
+  tags: []
+  job_type: ""

configs/model/cfm/default.yaml ADDED Viewed

	@@ -0,0 +1,3 @@

+name: CFM
+solver: euler
+sigma_min: 1e-4

configs/model/decoder/default.yaml ADDED Viewed

	@@ -0,0 +1,7 @@

+channels: [256, 256]
+dropout: 0.05
+attention_head_dim: 64
+n_blocks: 1
+num_mid_blocks: 2
+num_heads: 2
+act_fn: snakebeta

configs/model/encoder/default.yaml ADDED Viewed

	@@ -0,0 +1,18 @@

+encoder_type: RoPE Encoder
+encoder_params:
+  n_feats: ${model.n_feats}
+  n_channels: 192
+  filter_channels: 768
+  filter_channels_dp: 256
+  n_heads: 2
+  n_layers: 6
+  kernel_size: 3
+  p_dropout: 0.1
+  spk_emb_dim: 64
+  n_spks: 1
+  prenet: true
+duration_predictor_params:
+  filter_channels_dp: ${model.encoder.encoder_params.filter_channels_dp}
+  kernel_size: 3
+  p_dropout: ${model.encoder.encoder_params.p_dropout}

configs/model/matcha.yaml ADDED Viewed

	@@ -0,0 +1,15 @@

+defaults:
+  - _self_
+  - encoder: default.yaml
+  - decoder: default.yaml
+  - cfm: default.yaml
+  - optimizer: adam.yaml
+_target_: matcha.models.matcha_tts.MatchaTTS
+n_vocab: 178
+n_spks: ${data.n_spks}
+spk_emb_dim: 64
+n_feats: 80
+data_statistics: ${data.data_statistics}
+out_size: null # Must be divisible by 4
+prior_loss: true

configs/model/optimizer/adam.yaml ADDED Viewed

	@@ -0,0 +1,4 @@

+_target_: torch.optim.Adam
+_partial_: true
+lr: 1e-4
+weight_decay: 0.0

configs/paths/default.yaml ADDED Viewed

	@@ -0,0 +1,18 @@

+# path to root directory
+# this requires PROJECT_ROOT environment variable to exist
+# you can replace it with "." if you want the root to be the current working directory
+root_dir: ${oc.env:PROJECT_ROOT}
+# path to data directory
+data_dir: ${paths.root_dir}/data/
+# path to logging directory
+log_dir: ${paths.root_dir}/logs/
+# path to output directory, created dynamically by hydra
+# path generation pattern is specified in `configs/hydra/default.yaml`
+# use it to store all files generated during the run, like ckpts and metrics
+output_dir: ${hydra:runtime.output_dir}
+# path to working directory
+work_dir: ${hydra:runtime.cwd}

configs/train.yaml ADDED Viewed

	@@ -0,0 +1,51 @@

+# @package _global_
+# specify here default configuration
+# order of defaults determines the order in which configs override each other
+defaults:
+  - _self_
+  - data: akylai
+  - model: matcha
+  - callbacks: default
+  - logger: tensorboard # set logger here or use command line (e.g. `python train.py logger=tensorboard`)
+  - trainer: default
+  - paths: default
+  - extras: default
+  - hydra: default
+  # experiment configs allow for version control of specific hyperparameters
+  # e.g. best hyperparameters for given model and datamodule
+  - experiment: null
+  # config for hyperparameter optimization
+  - hparams_search: null
+  # optional local config for machine/user specific settings
+  # it's optional since it doesn't need to exist and is excluded from version control
+  - optional local: default
+  # debugging config (enable through command line, e.g. `python train.py debug=default)
+  - debug: null
+# task name, determines output directory path
+task_name: "train"
+run_name: ???
+# tags to help you identify your experiments
+# you can overwrite this in experiment configs
+# overwrite from command line with `python train.py tags="[first_tag, second_tag]"`
+tags: ["dev"]
+# set False to skip model training
+train: True
+# evaluate on test set, using best model weights achieved during training
+# lightning chooses best weights based on the metric specified in checkpoint callback
+test: False
+# simply provide checkpoint path to resume training
+ckpt_path: "https://github.com/simonlobgromov/AkylAI_Matcha_Checkpoint/releases/download/Matcha-TTS/checkpoint_epoch.499.ckpt"
+# seed for random number generators in pytorch, numpy and python.random
+seed: 1234

configs/trainer/cpu.yaml ADDED Viewed

	@@ -0,0 +1,5 @@

+defaults:
+  - default
+accelerator: cpu
+devices: 1

configs/trainer/ddp.yaml ADDED Viewed

	@@ -0,0 +1,9 @@

+defaults:
+  - default
+strategy: ddp
+accelerator: gpu
+devices: [0,1]
+num_nodes: 1
+sync_batchnorm: True