{ "model_key": "medembed-large", "use_bidirectional": true, "use_flash_attention": false, "use_unsloth": true, "unsloth_full_finetuning": false, "unsloth_4bit": false, "unsloth_qlora": true, "unsloth_qlora_rank": 32, "unsloth_qlora_alpha": 64.0, "unsloth_target_modules": "q_proj,k_proj,v_proj,o_proj,gate_proj,up_proj,down_proj", "unsloth_pooling_mode": "mean", "start_training_from_stage": 0, "enable_lora": true, "disable_lora": false, "encoder_only_training": true, "use_cache": false, "stage0_cache_mode": "off", "encoder_tuning_mode": "gradual", "gradual_unfreeze_initial_layers": 2, "gradual_unfreeze_every": 1, "gradual_unfreeze_max_layers": 6, "gradual_unfreeze_include_pooler": true, "train_file": "./mimic/train_row_level.json", "eval_file": "./mimic/val_row_level.json", "test_file": "./mimic/test_row_level.json", "output_dir": "./output_best_model/", "max_train_examples": 10000, "max_eval_examples": 1000, "eval_every_n_steps": 0, "epochs": 20, "early_stopping_patience": 20, "early_stopping_min_epochs": 20, "enable_checkpointing": true, "lr": 0.0001, "encoder_lr": 1e-05, "lr_scheduler_type": "cosine", "lr_num_cycles": 1, "min_lr_ratio": 0.1, "train_batch_size": 16, "eval_batch_size": 16, "weight_decay": 0.01, "warmup_ratio": 0.1, "max_grad_norm": 1.0, "gradient_accumulation_steps": 1, "mix_examples": false, "triplet_strategy": "limited", "max_triplets_per_example": 2, "loss_type": "bidirectional_triplet", "use_wandb": true, "use_cross_attention_lora": false, "lora_rank": 128, "lora_alpha": 512, "lora_dropout": 0.1, "do_visualize": true, "do_clean_analysis": false, "visualize_examples": "16", "normalize_attention": true, "skip_four_stage_viz": true, "aggregation_method": "top_k_pairs", "top_k": 5, "norm_type": "rmsnorm", "use_qk_rmsnorm": false, "use_latent_bottleneck": false, "latent_num": 64, "latent_dropout": 0.0, "pair_topk_mask": false, "pair_topk_k": 0, "use_hard_negative_mining": true, "hard_negative_topk": 1, "margin": 0.3, "scale": 10.0, "pair_margin": 0.3, "margin_end": 0.5, "margin_schedule": "none", "ranking_loss_type": "infonce", "infonce_tau": 0.2, "triplet_weight": 0.5, "attention_loss_weight": 0.0, "pair_loss_weight": 0.3, "use_attention_distillation": true, "distillation_weight": 0.2, "teacher_temperature": 0.1, "student_temperature": 0.1, "distillation_loss_type": "js_div", "pair_score_method": "cosine", "share_attention_weights": true, "extract_join_paths": true, "join_path_threshold": 0.15, "use_refinement": false, "use_self_attention": false, "self_attention_heads": 1, "self_attention_dropout": 0.1, "attention_type": "top_k_sparse", "use_gated_attention": true, "gated_attention_mode": "vector", "gated_attention_hidden_dim": 0, "gated_attention_dropout": 0.0, "gated_attention_init_bias": 6.0, "disable_temperature": true, "attention_activation": "softmax", "attention_alpha": 1.5, "sparse_top_k": 5, "window_size": 5, "threshold_base": 0.3, "init_method": "zeros", "init_method_params": { "bias_value": 0.0 }, "init_show_descriptions": false, "enable_training_curves": true, "track_batch_losses": true, "track_val_loss": true, "auto_plot_curves": false, "enable_row_sent_eval": true, "row_sent_test_file": "mimic/test_row_level.json", "row_sent_annotation_file": "mimic/Annotated_Test.json", "row_sent_max_examples": null, "save_best_by_test_metrics": true, "enable_mimic_eval": true, "run_post_training_eval": false, "verbosity": 1, "quiet": false, "verbose": false, "use_compile": false, "compile_mode": "reduce-overhead", "verbose_flag": true, "extra_verbose": false, "model_name": "abhinand/MedEmbed-large-v0.1", "embedding_dim": 1024, "max_seq_length": 512, "architecture": "bidirectional" }