File size: 2,866 Bytes
5048002
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
{
  "base_model": "microsoft/VibeVoice-1.5B",
  "checkpoint_dir": "/kaggle/working/vibevoice-finetune/out_tpu/final",
  "dtype": "bfloat16",
  "components_merged": [
    "language_model",
    "prediction_head",
    "acoustic_connector",
    "semantic_connector"
  ],
  "tensors_changed_vs_base": 373,
  "training_state": {
    "step": 2508,
    "args": {
      "model_name_or_path": "microsoft/VibeVoice-1.5B",
      "data_dir": "/kaggle/working/preprocessed/train",
      "eval_data_dir": null,
      "output_dir": "out_tpu",
      "lazy_data": false,
      "max_seq_len": 0,
      "max_latents": 0,
      "max_segments": 0,
      "buckets": null,
      "auto_buckets": 0,
      "per_device_batch_size": 2,
      "gradient_accumulation_steps": 4,
      "learning_rate": 1e-05,
      "head_lr_multiplier": 2.0,
      "weight_decay": 0.01,
      "adam_beta1": 0.9,
      "adam_beta2": 0.95,
      "adam_eps": 1e-08,
      "max_grad_norm": 1.0,
      "no_master_weights": false,
      "num_train_epochs": 6.0,
      "max_steps": 0,
      "warmup_ratio": 0.08,
      "warmup_steps": 0,
      "lr_scheduler": "cosine",
      "min_lr_ratio": 0.05,
      "seed": 42,
      "ce_loss_weight": 1.2,
      "diffusion_loss_weight": 1.0,
      "ddpm_batch_mul": 6,
      "ce_chunk_size": 128,
      "no_semantic": false,
      "no_latent_noise": false,
      "ce_acoustic_supervision": "target",
      "freeze_components": null,
      "freeze_llm_layers": null,
      "train_llm_layers": null,
      "freeze_head_layers": null,
      "freeze_embeddings": false,
      "freeze_lm_head": false,
      "freeze_regex": null,
      "bf16": true,
      "gradient_checkpointing": true,
      "no_spmd": false,
      "logging_steps": 10,
      "save_steps": 2000,
      "save_total_limit": 1,
      "eval_steps": 0,
      "eval_batches": 20,
      "resume_from_checkpoint": null,
      "dataloader_workers": 0,
      "profile_first_steps": 3
    },
    "shapes": {
      "max_seq_len": 512,
      "max_latents": 224,
      "max_segments": 8,
      "acoustic_dim": 64,
      "semantic_dim": 128
    },
    "preprocess_metadata": {
      "model_name_or_path": "microsoft/VibeVoice-1.5B",
      "processor_name_or_path": "vibevoice/processor",
      "compress_ratio": 3200,
      "acoustic_dim": 64,
      "semantic_dim": 128,
      "fix_std": 0.5,
      "latent_dtype": "float16",
      "speech_scaling_factor": 0.1962890625,
      "speech_bias_factor": -0.04931640625,
      "scaling_source": "checkpoint",
      "corpus_latent_mean": -0.16465343947872957,
      "corpus_latent_var": 26.317665766460568,
      "shapes": {
        "max_seq_len": 512,
        "max_latents": 224,
        "max_segments": 8,
        "acoustic_dim": 64,
        "semantic_dim": 128
      },
      "pad_token_id": 151655,
      "augment_target_silence": true,
      "normalize_target_audio": false
    }
  }
}