gurumurthy3 commited on
Commit
4669c19
·
verified ·
1 Parent(s): d040430

Upload trainable weights + training curves (epoch 2, step 26216)

Browse files
.gitattributes CHANGED
@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ training_curves.png filter=lfs diff=lfs merge=lfs -text
README.md ADDED
@@ -0,0 +1,29 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ tags:
3
+ - image-captioning
4
+ - gpt2
5
+ - clip
6
+ - vision-language
7
+ ---
8
+
9
+ # GPT2VL-Stackformer v3 (trainable weights only)
10
+
11
+ `model_trainable.safetensors` contains ONLY the trained Perceiver Resampler +
12
+ gated cross-attention weights -- NOT the frozen CLIP / GPT-2 backbone, which
13
+ are pulled from their own pretrained checkpoints (`openai/clip-vit-base-patch16`,
14
+ `gpt2`) at model-build time in the training notebook.
15
+
16
+ ## Training state at this checkpoint
17
+ - epoch: 2
18
+ - step: 26216
19
+
20
+ ## To load
21
+ You need the `GPT2VL` architecture class + GPT-2 weight-transfer code from the
22
+ training notebook. Then:
23
+
24
+ ```python
25
+ model = GPT2VL(json.load(open("config.json")), device="cuda")
26
+ # ... re-run the GPT-2 pretrained weight-transfer step from the training notebook ...
27
+ from safetensors.torch import load_file
28
+ model.load_state_dict(load_file("model_trainable.safetensors"), strict=False)
29
+ ```
checkpoint_state.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3cd3deeee2cd0cea48b585f08787b9ae01343d1bc8334e96af25898c1270b6aa
3
+ size 227365119
config.json ADDED
@@ -0,0 +1,25 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "vocab_size": 50258,
3
+ "embed_dim": 768,
4
+ "num_layers": 12,
5
+ "num_heads": 12,
6
+ "hidden_dim": 3072,
7
+ "context_length": 128,
8
+ "dropout": 0.1,
9
+ "qkv_bias": true,
10
+ "clip_model_name": "openai/clip-vit-base-patch16",
11
+ "vision_dim": 768,
12
+ "freeze_vision_encoder": true,
13
+ "num_visual_tokens": 64,
14
+ "perceiver_depth": 3,
15
+ "perceiver_heads": 8,
16
+ "perceiver_hidden_mult": 4,
17
+ "cross_attention_pos": [
18
+ 3,
19
+ 7,
20
+ 11
21
+ ],
22
+ "cross_attention_heads": 12,
23
+ "gate_init": 0.0,
24
+ "alpha_clamp": 4.0
25
+ }
model_trainable.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e2874d8692b32de9cb99d91827b3f232ad7460c27438ba3f3c1d168565d111d9
3
+ size 113644116
training_curves.png ADDED

Git LFS Details

  • SHA256: 9235fd2c0f81dadc71f1577477fa6caee684b75ae8a970cc6f310e4cac0ccb1e
  • Pointer size: 131 Bytes
  • Size of remote file: 112 kB