File size: 3,929 Bytes
cf4cc7a | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 | {
"model_key": "medembed-large",
"use_bidirectional": true,
"use_flash_attention": false,
"use_unsloth": true,
"unsloth_full_finetuning": false,
"unsloth_4bit": false,
"unsloth_qlora": true,
"unsloth_qlora_rank": 32,
"unsloth_qlora_alpha": 64.0,
"unsloth_target_modules": "q_proj,k_proj,v_proj,o_proj,gate_proj,up_proj,down_proj",
"unsloth_pooling_mode": "mean",
"start_training_from_stage": 0,
"enable_lora": true,
"disable_lora": false,
"encoder_only_training": true,
"use_cache": false,
"stage0_cache_mode": "off",
"encoder_tuning_mode": "gradual",
"gradual_unfreeze_initial_layers": 2,
"gradual_unfreeze_every": 1,
"gradual_unfreeze_max_layers": 6,
"gradual_unfreeze_include_pooler": true,
"train_file": "./mimic/train_row_level.json",
"eval_file": "./mimic/val_row_level.json",
"test_file": "./mimic/test_row_level.json",
"output_dir": "./output_best_model/",
"max_train_examples": 10000,
"max_eval_examples": 1000,
"eval_every_n_steps": 0,
"epochs": 20,
"early_stopping_patience": 20,
"early_stopping_min_epochs": 20,
"enable_checkpointing": true,
"lr": 0.0001,
"encoder_lr": 1e-05,
"lr_scheduler_type": "cosine",
"lr_num_cycles": 1,
"min_lr_ratio": 0.1,
"train_batch_size": 16,
"eval_batch_size": 16,
"weight_decay": 0.01,
"warmup_ratio": 0.1,
"max_grad_norm": 1.0,
"gradient_accumulation_steps": 1,
"mix_examples": false,
"triplet_strategy": "limited",
"max_triplets_per_example": 2,
"loss_type": "bidirectional_triplet",
"use_wandb": true,
"use_cross_attention_lora": false,
"lora_rank": 128,
"lora_alpha": 512,
"lora_dropout": 0.1,
"do_visualize": true,
"do_clean_analysis": false,
"visualize_examples": "16",
"normalize_attention": true,
"skip_four_stage_viz": true,
"aggregation_method": "top_k_pairs",
"top_k": 5,
"norm_type": "rmsnorm",
"use_qk_rmsnorm": false,
"use_latent_bottleneck": false,
"latent_num": 64,
"latent_dropout": 0.0,
"pair_topk_mask": false,
"pair_topk_k": 0,
"use_hard_negative_mining": true,
"hard_negative_topk": 1,
"margin": 0.3,
"scale": 10.0,
"pair_margin": 0.3,
"margin_end": 0.5,
"margin_schedule": "none",
"ranking_loss_type": "infonce",
"infonce_tau": 0.2,
"triplet_weight": 0.5,
"attention_loss_weight": 0.0,
"pair_loss_weight": 0.3,
"use_attention_distillation": true,
"distillation_weight": 0.2,
"teacher_temperature": 0.1,
"student_temperature": 0.1,
"distillation_loss_type": "js_div",
"pair_score_method": "cosine",
"share_attention_weights": true,
"extract_join_paths": true,
"join_path_threshold": 0.15,
"use_refinement": false,
"use_self_attention": false,
"self_attention_heads": 1,
"self_attention_dropout": 0.1,
"attention_type": "top_k_sparse",
"use_gated_attention": true,
"gated_attention_mode": "vector",
"gated_attention_hidden_dim": 0,
"gated_attention_dropout": 0.0,
"gated_attention_init_bias": 6.0,
"disable_temperature": true,
"attention_activation": "softmax",
"attention_alpha": 1.5,
"sparse_top_k": 5,
"window_size": 5,
"threshold_base": 0.3,
"init_method": "zeros",
"init_method_params": {
"bias_value": 0.0
},
"init_show_descriptions": false,
"enable_training_curves": true,
"track_batch_losses": true,
"track_val_loss": true,
"auto_plot_curves": false,
"enable_row_sent_eval": true,
"row_sent_test_file": "mimic/test_row_level.json",
"row_sent_annotation_file": "mimic/Annotated_Test.json",
"row_sent_max_examples": null,
"save_best_by_test_metrics": true,
"enable_mimic_eval": true,
"run_post_training_eval": false,
"verbosity": 1,
"quiet": false,
"verbose": false,
"use_compile": false,
"compile_mode": "reduce-overhead",
"verbose_flag": true,
"extra_verbose": false,
"model_name": "abhinand/MedEmbed-large-v0.1",
"architecture": "bidirectional",
"embedding_dim": 1024,
"max_seq_length": 512
} |