SamCodeManMk2 commited on
Commit
53ae83b
·
verified ·
1 Parent(s): f9f6384

Upload folder using huggingface_hub

Browse files
.gitattributes CHANGED
@@ -33,3 +33,6 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ assets/tron-vs-jev-evals.png filter=lfs diff=lfs merge=lfs -text
37
+ assets/tron-vs-jev-index.png filter=lfs diff=lfs merge=lfs -text
38
+ assets/tron-vs-jev-table.png filter=lfs diff=lfs merge=lfs -text
LICENSE ADDED
@@ -0,0 +1,18 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Tron-1B model weights
2
+ Copyright (c) 2026 Samir Sengupta
3
+
4
+ Licensed under the Creative Commons Attribution-NonCommercial 4.0 International License (CC BY-NC 4.0).
5
+
6
+ You are free to share (copy and redistribute in any medium or format) and adapt (remix, transform, and build upon)
7
+ this material, under the following terms:
8
+
9
+ Attribution You must give appropriate credit, provide a link to the license, and indicate if changes were made.
10
+ NonCommercial You may not use the material for commercial purposes.
11
+
12
+ No additional restrictions: you may not apply legal terms or technological measures that legally restrict others
13
+ from doing anything the license permits.
14
+
15
+ Full legal code: https://creativecommons.org/licenses/by-nc/4.0/legalcode
16
+
17
+ This license covers the model weights and configuration in this repository. The troncore runtime used to run them
18
+ is licensed separately under Apache-2.0. Third-party components are listed in NOTICE.
NOTICE ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Tron-1B
2
+ Copyright (c) 2026 Samir Sengupta
3
+
4
+ This model was initialised from the Ettin encoder:
5
+
6
+ jhu-clsp/ettin-encoder-1b (https://huggingface.co/jhu-clsp/ettin-encoder-1b)
7
+ Weller, Ricci, Marone, Chaffin, Lawrie, Van Durme. "Seq vs Seq: An Open Suite of Paired Encoders and Decoders",
8
+ arXiv:2507.11412, 2025.
9
+ Licensed under the MIT License:
10
+
11
+ Permission is hereby granted, free of charge, to any person obtaining a copy of this software and associated
12
+ documentation files (the "Software"), to deal in the Software without restriction, including without limitation
13
+ the rights to use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the Software, and
14
+ to permit persons to whom the Software is furnished to do so, subject to the following conditions:
15
+
16
+ The above copyright notice and this permission notice shall be included in all copies or substantial portions of
17
+ the Software.
18
+
19
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO
20
+ THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
21
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF
22
+ CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
23
+ DEALINGS IN THE SOFTWARE.
README.md ADDED
@@ -0,0 +1,127 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: cc-by-nc-4.0
3
+ language:
4
+ - en
5
+ library_name: troncore
6
+ base_model: jhu-clsp/ettin-encoder-1b
7
+ pipeline_tag: zero-shot-classification
8
+ tags:
9
+ - decision-model
10
+ - zero-shot-classification
11
+ - text-classification
12
+ - intent-classification
13
+ - prompt-injection
14
+ - calibration
15
+ - modernbert
16
+ ---
17
+
18
+ # Tron-1B
19
+
20
+ **Tron-1B answers typed questions about text or JSON in a single pass: choose one option, rate on a scale, or answer yes/no, with calibrated probabilities.** It is built for the fast decisions around an AI application: routing, triage, safety screening, intent detection, and workflow automation.
21
+
22
+ - **Accurate:** beats Jev 1.13.0 on all four of its published benchmarks (below).
23
+ - **Fast:** 16.7 ms per decision (p50) on one GPU.
24
+ - **Calibrated:** its confidence scores track how often it is right, so you can gate actions on them.
25
+ - **Any question, at request time:** you write the question and the options; no retraining for new label sets.
26
+
27
+ ![Decision Index](assets/tron-vs-jev-index.png)
28
+
29
+ ## Quickstart
30
+
31
+ ```bash
32
+ pip install troncore
33
+ ```
34
+
35
+ ```python
36
+ from troncore import Engine
37
+
38
+ eng = Engine("SamCodeManMk2/tron-1b") # downloads once, then runs locally (GPU if available)
39
+
40
+ answers = eng.decide(
41
+ {"subject": "Duplicate charge on invoice #4411",
42
+ "body": "We were billed twice for March. Refund the duplicate today or we cancel our plan."},
43
+ {
44
+ "department": {"type": "choice", "instructions": "Which team should handle this?",
45
+ "criteria": {"billing": "invoices, payments, refunds", "technical": "bugs, outages",
46
+ "sales": "pricing, new contracts", "other": "anything else"}},
47
+ "urgency": {"type": "score", "instructions": "How urgent is this?",
48
+ "criteria": ["not urgent", "this week", "today", "critical"]},
49
+ "churn_risk": {"type": "yesno", "instructions": "Does the customer threaten to cancel?"},
50
+ },
51
+ )
52
+ answers["department"]["choice"] # 'billing'
53
+ answers["churn_risk"]["probability"] # P(yes)
54
+ answers["department"]["confidence"] # calibrated confidence of the top option
55
+ ```
56
+
57
+ Every answer includes `probabilities`, `confidence`, `margin` (top minus second) and `entropy`. Pass
58
+ `min_confidence=0.8` to get `abstain: true` on uncertain answers, so they can be routed to a person or a larger model.
59
+ Questions with more than 128 options are handled automatically by scoring them in rounds. Long inputs can be read
60
+ in sliding windows with `windows="all"`.
61
+
62
+ ### Question types
63
+
64
+ | type | options | answer |
65
+ |---|---|---|
66
+ | `choice` | `criteria`: list of labels, or dict label → description | `choice` + a probability per label |
67
+ | `score` | `criteria`: ordered list of levels | `score` (expected level), `level`, a probability per level |
68
+ | `yesno` (alias `noul`) | fixed no / yes | `answer` (bool) + `probability` of yes |
69
+
70
+ ## Results
71
+
72
+ ![Evaluations](assets/tron-vs-jev-evals.png)
73
+
74
+ | Benchmark | Tron-1B | Jev 1.13.0 |
75
+ |---|---|---|
76
+ | Banking77 (intent, 77 labels)¹ | **94.0** | 87.0 |
77
+ | AG News (topic) | **93.9** | 91.0 |
78
+ | typed-decisions (2,000 business decisions) | **79.6** | 72.7 |
79
+ | DAIR emotion | **92.9** | 48.0 |
80
+ | Latency, p50 per decision² | **16.7 ms** | 236–276 ms |
81
+
82
+ **Tasks Tron-1B never trained on** (whole task families held out of training):
83
+
84
+ | Task | Accuracy |
85
+ |---|---|
86
+ | IMDB sentiment | 94.7 |
87
+ | Prompt-injection detection (deepset) | 90.5 |
88
+ | MASSIVE intent (59 labels) | 88.4 |
89
+ | XNLI entailment (English) | 86.7 |
90
+ | BoolQ reading comprehension | 79.2 |
91
+ | CLINC intent (151 labels) | 62.8 |
92
+ | SST-5 graded sentiment | 54.8 |
93
+
94
+ ¹ Jev's published Banking77 result used 72 labels; Tron's uses all 77. Jev figures are its published numbers, not re-run by us.
95
+ ² Tron: one question on one NVIDIA GB10 GPU. Jev: independently measured p50 through its hosted API, which includes network time.
96
+
97
+ Tron-1B was trained on the training splits of Banking77, AG News and DAIR emotion and evaluated on their official
98
+ test splits. All results measured 2026-09-28 with troncore 1.0.
99
+
100
+ ## How it works
101
+
102
+ Tron-1B is a 1.1B-parameter bidirectional encoder (initialised from
103
+ [ettin-encoder-1b](https://huggingface.co/jhu-clsp/ettin-encoder-1b)) with a decision head. For each question, the
104
+ question, the input and every option are encoded together. Each option is pooled into a vector, and a small
105
+ attention layer compares the options with each other and with the question before they are scored. This is what lets
106
+ it separate close labels such as "card not arrived" and "card delivery estimate". Probabilities are calibrated per
107
+ question type and option count.
108
+
109
+ It was trained on about 1.2 million typed decisions built from public classification, entailment, safety, routing,
110
+ reasoning, preference and business-workflow datasets.
111
+
112
+ ## Limitations
113
+
114
+ - **It decides; it does not write.** Answers are always one of the options you give it.
115
+ - **Very large unseen label sets are its weakest area** (62.8% on CLINC's 151 intents zero-shot). For big label
116
+ sets, short descriptive label names help, and so does a few hundred examples of fine-tuning.
117
+ - **Long inputs:** accuracy is best under about 2,000 tokens of input. Send the relevant section rather than a whole
118
+ document, or use `windows="all"`.
119
+ - **English first.** It handles other languages, but was mostly trained and evaluated in English.
120
+ - **Not a safety guarantee.** Use its safety and injection judgements as one layer of defence, with confidence
121
+ gating, not as the only one.
122
+
123
+ ## License
124
+
125
+ The model weights are released under [CC BY-NC 4.0](https://creativecommons.org/licenses/by-nc/4.0/): free to use,
126
+ share and adapt for non-commercial purposes, with attribution. The `troncore` runtime is Apache-2.0. The base
127
+ encoder is MIT-licensed; see `NOTICE`.
assets/tron-vs-jev-evals.png ADDED

Git LFS Details

  • SHA256: fbc1ec568fd6ca15d9bb085a1bfe4fd61eb2e6489b281af037292caa4ff85f3c
  • Pointer size: 131 Bytes
  • Size of remote file: 185 kB
assets/tron-vs-jev-index.png ADDED

Git LFS Details

  • SHA256: 301a55b1a2de59d07804472c2057b1b6be3cafe689d94086c724f7886875cec4
  • Pointer size: 131 Bytes
  • Size of remote file: 107 kB
assets/tron-vs-jev-table.png ADDED

Git LFS Details

  • SHA256: ee8046f1c34dcf089b486e5eec48cdfaaac98a3272f88933008e4d06cb86bfc4
  • Pointer size: 131 Bytes
  • Size of remote file: 267 kB
config.json ADDED
@@ -0,0 +1,158 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "format": "troncore/1",
3
+ "encoder_config": {
4
+ "transformers_version": "5.17.0",
5
+ "architectures": [
6
+ "ModernBertForMaskedLM"
7
+ ],
8
+ "output_hidden_states": false,
9
+ "return_dict": true,
10
+ "dtype": "float32",
11
+ "chunk_size_feed_forward": 0,
12
+ "is_encoder_decoder": false,
13
+ "id2label": {
14
+ "0": "LABEL_0",
15
+ "1": "LABEL_1"
16
+ },
17
+ "label2id": {
18
+ "LABEL_0": 0,
19
+ "LABEL_1": 1
20
+ },
21
+ "problem_type": null,
22
+ "vocab_size": 50368,
23
+ "hidden_size": 1792,
24
+ "intermediate_size": 3840,
25
+ "num_hidden_layers": 28,
26
+ "num_attention_heads": 28,
27
+ "hidden_activation": "gelu",
28
+ "max_position_embeddings": 7999,
29
+ "initializer_range": 0.02,
30
+ "initializer_cutoff_factor": 2.0,
31
+ "norm_eps": 1e-05,
32
+ "norm_bias": false,
33
+ "pad_token_id": 50283,
34
+ "eos_token_id": 50282,
35
+ "bos_token_id": 50281,
36
+ "cls_token_id": 50281,
37
+ "sep_token_id": 50282,
38
+ "attention_bias": false,
39
+ "attention_dropout": 0.0,
40
+ "layer_types": [
41
+ "full_attention",
42
+ "sliding_attention",
43
+ "sliding_attention",
44
+ "full_attention",
45
+ "sliding_attention",
46
+ "sliding_attention",
47
+ "full_attention",
48
+ "sliding_attention",
49
+ "sliding_attention",
50
+ "full_attention",
51
+ "sliding_attention",
52
+ "sliding_attention",
53
+ "full_attention",
54
+ "sliding_attention",
55
+ "sliding_attention",
56
+ "full_attention",
57
+ "sliding_attention",
58
+ "sliding_attention",
59
+ "full_attention",
60
+ "sliding_attention",
61
+ "sliding_attention",
62
+ "full_attention",
63
+ "sliding_attention",
64
+ "sliding_attention",
65
+ "full_attention",
66
+ "sliding_attention",
67
+ "sliding_attention",
68
+ "full_attention"
69
+ ],
70
+ "rope_parameters": {
71
+ "sliding_attention": {
72
+ "rope_type": "default",
73
+ "rope_theta": 160000.0
74
+ },
75
+ "full_attention": {
76
+ "rope_type": "default",
77
+ "rope_theta": 160000.0
78
+ }
79
+ },
80
+ "local_attention": 128,
81
+ "embedding_dropout": 0.0,
82
+ "mlp_bias": false,
83
+ "mlp_dropout": 0.0,
84
+ "decoder_bias": true,
85
+ "classifier_pooling": "mean",
86
+ "classifier_dropout": 0.0,
87
+ "classifier_bias": false,
88
+ "classifier_activation": "gelu",
89
+ "deterministic_flash_attn": false,
90
+ "sparse_prediction": false,
91
+ "sparse_pred_ignore_index": -100,
92
+ "tie_word_embeddings": true,
93
+ "_name_or_path": "jhu-clsp/ettin-encoder-1b",
94
+ "global_attn_every_n_layers": 3,
95
+ "gradient_checkpointing": false,
96
+ "layer_norm_eps": 1e-05,
97
+ "model_type": "modernbert",
98
+ "position_embedding_type": "sans_pos",
99
+ "is_causal": false,
100
+ "causal_mask": false,
101
+ "output_attentions": false
102
+ },
103
+ "head": {
104
+ "width": 1024,
105
+ "mixer_layers": 2,
106
+ "mixer_heads": 8,
107
+ "dropout": 0.1,
108
+ "n_types": 3
109
+ },
110
+ "layout": {
111
+ "max_len": 4096,
112
+ "question_max_tokens": 128,
113
+ "option_max_tokens": 48,
114
+ "option_budget": 1536,
115
+ "min_state_tokens": 32
116
+ },
117
+ "calibration": {
118
+ "*": 1.0549,
119
+ "yesno": 1.0678,
120
+ "choice": 1.0075,
121
+ "score": 1.2359,
122
+ "yesno:2": 1.0678,
123
+ "choice:2": 1.0955,
124
+ "choice:3-5": 1.0086,
125
+ "choice:6-10": 1.1039,
126
+ "score:3-5": 1.2381,
127
+ "choice:11-30": 0.8095,
128
+ "score:6-10": 1.5212,
129
+ "choice:31+": 0.9561
130
+ },
131
+ "model_name": "tron-1b",
132
+ "val_macro": 0.8492879196203449,
133
+ "val_ece": 0.02179274528040747,
134
+ "val_accuracy": {
135
+ "unsafe_prompt": 0.857707509881423,
136
+ "commonsense_mc": 0.904,
137
+ "injection": 0.9783080260303688,
138
+ "choice": 0.7426666666666667,
139
+ "routing_binary": 0.8611803823773898,
140
+ "noul": 0.8262857142857143,
141
+ "knowledge_mc": 0.6967213114754098,
142
+ "bench_emotion": 0.9447983014861996,
143
+ "zeroshot": 0.9006666666666666,
144
+ "score": 0.6576819407008087,
145
+ "bench_ag_news": 0.9386331938633193,
146
+ "tool_choice": 0.9793103448275862,
147
+ "intent": 0.976063829787234,
148
+ "unsafe_response": 0.8703071672354948,
149
+ "typed_decisions_synth": 0.834,
150
+ "harm_category": 0.8549019607843137,
151
+ "routing": 0.6683291770573566,
152
+ "preference": 0.788135593220339,
153
+ "typed_decisions": 0.8,
154
+ "bench_banking77": 0.906060606060606
155
+ },
156
+ "trained_steps": 18582,
157
+ "base_model": "jhu-clsp/ettin-encoder-1b"
158
+ }
model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:143c32f44f512200ad0e0aa7b0032de199eef7013cc2419ed44cbd6a356f7e55
3
+ size 2123836994
tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
tokenizer_config.json ADDED
@@ -0,0 +1,17 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "backend": "tokenizers",
3
+ "clean_up_tokenization_spaces": true,
4
+ "cls_token": "[CLS]",
5
+ "is_local": true,
6
+ "local_files_only": false,
7
+ "mask_token": "[MASK]",
8
+ "model_input_names": [
9
+ "input_ids",
10
+ "attention_mask"
11
+ ],
12
+ "model_max_length": 8192,
13
+ "pad_token": "[PAD]",
14
+ "sep_token": "[SEP]",
15
+ "tokenizer_class": "TokenizersBackend",
16
+ "unk_token": "[UNK]"
17
+ }