Image-Text-to-Text
Transformers
Safetensors
English
Chinese
qwen3_5_moe
mathematics
proof-verification
advancedmathbench
conversational
Instructions to use internlm/AdvancedMathBench-AutoVerifier with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use internlm/AdvancedMathBench-AutoVerifier with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("image-text-to-text", model="internlm/AdvancedMathBench-AutoVerifier") messages = [ { "role": "user", "content": [ {"type": "image", "url": "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/p-blog/candy.JPG"}, {"type": "text", "text": "What animal is on the candy?"} ] }, ] pipe(text=messages)# Load model directly from transformers import AutoProcessor, AutoModelForMultimodalLM processor = AutoProcessor.from_pretrained("internlm/AdvancedMathBench-AutoVerifier") model = AutoModelForMultimodalLM.from_pretrained("internlm/AdvancedMathBench-AutoVerifier", device_map="auto") messages = [ { "role": "user", "content": [ {"type": "image", "url": "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/p-blog/candy.JPG"}, {"type": "text", "text": "What animal is on the candy?"} ] }, ] inputs = processor.apply_chat_template( messages, add_generation_prompt=True, tokenize=True, return_dict=True, return_tensors="pt", ).to(model.device) outputs = model.generate(**inputs, max_new_tokens=40) print(processor.decode(outputs[0][inputs["input_ids"].shape[-1]:])) - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use internlm/AdvancedMathBench-AutoVerifier with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "internlm/AdvancedMathBench-AutoVerifier" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "internlm/AdvancedMathBench-AutoVerifier", "messages": [ { "role": "user", "content": [ { "type": "text", "text": "Describe this image in one sentence." }, { "type": "image_url", "image_url": { "url": "https://cdn.britannica.com/61/93061-050-99147DCE/Statue-of-Liberty-Island-New-York-Bay.jpg" } } ] } ] }'Use Docker
docker model run hf.co/internlm/AdvancedMathBench-AutoVerifier
- SGLang
How to use internlm/AdvancedMathBench-AutoVerifier with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "internlm/AdvancedMathBench-AutoVerifier" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "internlm/AdvancedMathBench-AutoVerifier", "messages": [ { "role": "user", "content": [ { "type": "text", "text": "Describe this image in one sentence." }, { "type": "image_url", "image_url": { "url": "https://cdn.britannica.com/61/93061-050-99147DCE/Statue-of-Liberty-Island-New-York-Bay.jpg" } } ] } ] }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "internlm/AdvancedMathBench-AutoVerifier" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "internlm/AdvancedMathBench-AutoVerifier", "messages": [ { "role": "user", "content": [ { "type": "text", "text": "Describe this image in one sentence." }, { "type": "image_url", "image_url": { "url": "https://cdn.britannica.com/61/93061-050-99147DCE/Statue-of-Liberty-Island-New-York-Bay.jpg" } } ] } ] }' - Docker Model Runner
How to use internlm/AdvancedMathBench-AutoVerifier with Docker Model Runner:
docker model run hf.co/internlm/AdvancedMathBench-AutoVerifier
Add files using upload-large-folder tool
Browse files- .gitattributes +3 -34
- LICENSE +202 -0
- NOTICE.md +18 -0
- README.md +76 -0
- chat_template.jinja +154 -0
- checkpoint_checksums.sha256 +56 -0
- checkpoint_manifest.json +306 -0
- compatibility.json +44 -0
- config.json +116 -0
- evaluation_settings.json +27 -0
- generation_config.json +13 -0
- merges.txt +0 -0
- model-language-0001-fused-save_rank1.safetensors +3 -0
- model-language-0001-fused-save_rank12.safetensors +3 -0
- model-language-0001-fused-save_rank13.safetensors +3 -0
- model-language-0001-fused-save_rank14.safetensors +3 -0
- model-language-0001-fused-save_rank15.safetensors +3 -0
- model-language-0001-fused-save_rank2.safetensors +3 -0
- model-language-0001-fused-save_rank3.safetensors +3 -0
- model-language-0001-fused-save_rank5.safetensors +3 -0
- model-language-0001-fused-save_rank7.safetensors +3 -0
- model-language-0001-fused-save_rank9.safetensors +3 -0
- model-language-0002-others-save_rank0.safetensors +3 -0
- model-language-0003-others-save_rank0.safetensors +3 -0
- model-language-0004-others-save_rank0.safetensors +3 -0
- model-language-0006-others-save_rank0.safetensors +3 -0
- model-language-0007-others-save_rank0.safetensors +3 -0
- model-language-0008-others-save_rank0.safetensors +3 -0
- model-language-0011-others-save_rank0.safetensors +3 -0
- model-language-0012-others-save_rank0.safetensors +3 -0
- model-language-0013-others-save_rank0.safetensors +3 -0
- model-language-0014-others-save_rank0.safetensors +3 -0
- model-language-0017-others-save_rank0.safetensors +3 -0
- model-language-0018-others-save_rank0.safetensors +3 -0
- model-language-0019-others-save_rank0.safetensors +3 -0
- model-language-0020-others-save_rank0.safetensors +3 -0
- model-language-0021-others-save_rank0.safetensors +3 -0
- model-language-0022-others-save_rank0.safetensors +3 -0
- model-projector-0001-others-save_rank0.safetensors +3 -0
- model.safetensors.index.json +0 -0
- preprocessor_config.json +21 -0
- prompts/proof_verifier.md +61 -0
- special_tokens_map.json +45 -0
- tokenization_interns1.py +1009 -0
- tokenizer_PROT.model +3 -0
- tokenizer_SMILES.model +3 -0
- tokenizer_XNA.model +3 -0
- tokenizer_config.json +506 -0
- video_preprocessor_config.json +21 -0
- vocab.json +0 -0
.gitattributes
CHANGED
|
@@ -1,35 +1,4 @@
|
|
| 1 |
-
*.7z filter=lfs diff=lfs merge=lfs -text
|
| 2 |
-
*.arrow filter=lfs diff=lfs merge=lfs -text
|
| 3 |
-
*.bin filter=lfs diff=lfs merge=lfs -text
|
| 4 |
-
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
| 5 |
-
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
| 6 |
-
*.ftz filter=lfs diff=lfs merge=lfs -text
|
| 7 |
-
*.gz filter=lfs diff=lfs merge=lfs -text
|
| 8 |
-
*.h5 filter=lfs diff=lfs merge=lfs -text
|
| 9 |
-
*.joblib filter=lfs diff=lfs merge=lfs -text
|
| 10 |
-
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
| 11 |
-
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
| 12 |
-
*.model filter=lfs diff=lfs merge=lfs -text
|
| 13 |
-
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
| 14 |
-
*.npy filter=lfs diff=lfs merge=lfs -text
|
| 15 |
-
*.npz filter=lfs diff=lfs merge=lfs -text
|
| 16 |
-
*.onnx filter=lfs diff=lfs merge=lfs -text
|
| 17 |
-
*.ot filter=lfs diff=lfs merge=lfs -text
|
| 18 |
-
*.parquet filter=lfs diff=lfs merge=lfs -text
|
| 19 |
-
*.pb filter=lfs diff=lfs merge=lfs -text
|
| 20 |
-
*.pickle filter=lfs diff=lfs merge=lfs -text
|
| 21 |
-
*.pkl filter=lfs diff=lfs merge=lfs -text
|
| 22 |
-
*.pt filter=lfs diff=lfs merge=lfs -text
|
| 23 |
-
*.pth filter=lfs diff=lfs merge=lfs -text
|
| 24 |
-
*.rar filter=lfs diff=lfs merge=lfs -text
|
| 25 |
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
| 26 |
-
|
| 27 |
-
|
| 28 |
-
|
| 29 |
-
*.tflite filter=lfs diff=lfs merge=lfs -text
|
| 30 |
-
*.tgz filter=lfs diff=lfs merge=lfs -text
|
| 31 |
-
*.wasm filter=lfs diff=lfs merge=lfs -text
|
| 32 |
-
*.xz filter=lfs diff=lfs merge=lfs -text
|
| 33 |
-
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
-
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
-
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
| 2 |
+
tokenizer_XNA.model filter=lfs diff=lfs merge=lfs -text
|
| 3 |
+
tokenizer_PROT.model filter=lfs diff=lfs merge=lfs -text
|
| 4 |
+
tokenizer_SMILES.model filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
LICENSE
ADDED
|
@@ -0,0 +1,202 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
|
| 2 |
+
Apache License
|
| 3 |
+
Version 2.0, January 2004
|
| 4 |
+
http://www.apache.org/licenses/
|
| 5 |
+
|
| 6 |
+
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
| 7 |
+
|
| 8 |
+
1. Definitions.
|
| 9 |
+
|
| 10 |
+
"License" shall mean the terms and conditions for use, reproduction,
|
| 11 |
+
and distribution as defined by Sections 1 through 9 of this document.
|
| 12 |
+
|
| 13 |
+
"Licensor" shall mean the copyright owner or entity authorized by
|
| 14 |
+
the copyright owner that is granting the License.
|
| 15 |
+
|
| 16 |
+
"Legal Entity" shall mean the union of the acting entity and all
|
| 17 |
+
other entities that control, are controlled by, or are under common
|
| 18 |
+
control with that entity. For the purposes of this definition,
|
| 19 |
+
"control" means (i) the power, direct or indirect, to cause the
|
| 20 |
+
direction or management of such entity, whether by contract or
|
| 21 |
+
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
| 22 |
+
outstanding shares, or (iii) beneficial ownership of such entity.
|
| 23 |
+
|
| 24 |
+
"You" (or "Your") shall mean an individual or Legal Entity
|
| 25 |
+
exercising permissions granted by this License.
|
| 26 |
+
|
| 27 |
+
"Source" form shall mean the preferred form for making modifications,
|
| 28 |
+
including but not limited to software source code, documentation
|
| 29 |
+
source, and configuration files.
|
| 30 |
+
|
| 31 |
+
"Object" form shall mean any form resulting from mechanical
|
| 32 |
+
transformation or translation of a Source form, including but
|
| 33 |
+
not limited to compiled object code, generated documentation,
|
| 34 |
+
and conversions to other media types.
|
| 35 |
+
|
| 36 |
+
"Work" shall mean the work of authorship, whether in Source or
|
| 37 |
+
Object form, made available under the License, as indicated by a
|
| 38 |
+
copyright notice that is included in or attached to the work
|
| 39 |
+
(an example is provided in the Appendix below).
|
| 40 |
+
|
| 41 |
+
"Derivative Works" shall mean any work, whether in Source or Object
|
| 42 |
+
form, that is based on (or derived from) the Work and for which the
|
| 43 |
+
editorial revisions, annotations, elaborations, or other modifications
|
| 44 |
+
represent, as a whole, an original work of authorship. For the purposes
|
| 45 |
+
of this License, Derivative Works shall not include works that remain
|
| 46 |
+
separable from, or merely link (or bind by name) to the interfaces of,
|
| 47 |
+
the Work and Derivative Works thereof.
|
| 48 |
+
|
| 49 |
+
"Contribution" shall mean any work of authorship, including
|
| 50 |
+
the original version of the Work and any modifications or additions
|
| 51 |
+
to that Work or Derivative Works thereof, that is intentionally
|
| 52 |
+
submitted to Licensor for inclusion in the Work by the copyright owner
|
| 53 |
+
or by an individual or Legal Entity authorized to submit on behalf of
|
| 54 |
+
the copyright owner. For the purposes of this definition, "submitted"
|
| 55 |
+
means any form of electronic, verbal, or written communication sent
|
| 56 |
+
to the Licensor or its representatives, including but not limited to
|
| 57 |
+
communication on electronic mailing lists, source code control systems,
|
| 58 |
+
and issue tracking systems that are managed by, or on behalf of, the
|
| 59 |
+
Licensor for the purpose of discussing and improving the Work, but
|
| 60 |
+
excluding communication that is conspicuously marked or otherwise
|
| 61 |
+
designated in writing by the copyright owner as "Not a Contribution."
|
| 62 |
+
|
| 63 |
+
"Contributor" shall mean Licensor and any individual or Legal Entity
|
| 64 |
+
on behalf of whom a Contribution has been received by Licensor and
|
| 65 |
+
subsequently incorporated within the Work.
|
| 66 |
+
|
| 67 |
+
2. Grant of Copyright License. Subject to the terms and conditions of
|
| 68 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 69 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 70 |
+
copyright license to reproduce, prepare Derivative Works of,
|
| 71 |
+
publicly display, publicly perform, sublicense, and distribute the
|
| 72 |
+
Work and such Derivative Works in Source or Object form.
|
| 73 |
+
|
| 74 |
+
3. Grant of Patent License. Subject to the terms and conditions of
|
| 75 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 76 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 77 |
+
(except as stated in this section) patent license to make, have made,
|
| 78 |
+
use, offer to sell, sell, import, and otherwise transfer the Work,
|
| 79 |
+
where such license applies only to those patent claims licensable
|
| 80 |
+
by such Contributor that are necessarily infringed by their
|
| 81 |
+
Contribution(s) alone or by combination of their Contribution(s)
|
| 82 |
+
with the Work to which such Contribution(s) was submitted. If You
|
| 83 |
+
institute patent litigation against any entity (including a
|
| 84 |
+
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
| 85 |
+
or a Contribution incorporated within the Work constitutes direct
|
| 86 |
+
or contributory patent infringement, then any patent licenses
|
| 87 |
+
granted to You under this License for that Work shall terminate
|
| 88 |
+
as of the date such litigation is filed.
|
| 89 |
+
|
| 90 |
+
4. Redistribution. You may reproduce and distribute copies of the
|
| 91 |
+
Work or Derivative Works thereof in any medium, with or without
|
| 92 |
+
modifications, and in Source or Object form, provided that You
|
| 93 |
+
meet the following conditions:
|
| 94 |
+
|
| 95 |
+
(a) You must give any other recipients of the Work or
|
| 96 |
+
Derivative Works a copy of this License; and
|
| 97 |
+
|
| 98 |
+
(b) You must cause any modified files to carry prominent notices
|
| 99 |
+
stating that You changed the files; and
|
| 100 |
+
|
| 101 |
+
(c) You must retain, in the Source form of any Derivative Works
|
| 102 |
+
that You distribute, all copyright, patent, trademark, and
|
| 103 |
+
attribution notices from the Source form of the Work,
|
| 104 |
+
excluding those notices that do not pertain to any part of
|
| 105 |
+
the Derivative Works; and
|
| 106 |
+
|
| 107 |
+
(d) If the Work includes a "NOTICE" text file as part of its
|
| 108 |
+
distribution, then any Derivative Works that You distribute must
|
| 109 |
+
include a readable copy of the attribution notices contained
|
| 110 |
+
within such NOTICE file, excluding those notices that do not
|
| 111 |
+
pertain to any part of the Derivative Works, in at least one
|
| 112 |
+
of the following places: within a NOTICE text file distributed
|
| 113 |
+
as part of the Derivative Works; within the Source form or
|
| 114 |
+
documentation, if provided along with the Derivative Works; or,
|
| 115 |
+
within a display generated by the Derivative Works, if and
|
| 116 |
+
wherever such third-party notices normally appear. The contents
|
| 117 |
+
of the NOTICE file are for informational purposes only and
|
| 118 |
+
do not modify the License. You may add Your own attribution
|
| 119 |
+
notices within Derivative Works that You distribute, alongside
|
| 120 |
+
or as an addendum to the NOTICE text from the Work, provided
|
| 121 |
+
that such additional attribution notices cannot be construed
|
| 122 |
+
as modifying the License.
|
| 123 |
+
|
| 124 |
+
You may add Your own copyright statement to Your modifications and
|
| 125 |
+
may provide additional or different license terms and conditions
|
| 126 |
+
for use, reproduction, or distribution of Your modifications, or
|
| 127 |
+
for any such Derivative Works as a whole, provided Your use,
|
| 128 |
+
reproduction, and distribution of the Work otherwise complies with
|
| 129 |
+
the conditions stated in this License.
|
| 130 |
+
|
| 131 |
+
5. Submission of Contributions. Unless You explicitly state otherwise,
|
| 132 |
+
any Contribution intentionally submitted for inclusion in the Work
|
| 133 |
+
by You to the Licensor shall be under the terms and conditions of
|
| 134 |
+
this License, without any additional terms or conditions.
|
| 135 |
+
Notwithstanding the above, nothing herein shall supersede or modify
|
| 136 |
+
the terms of any separate license agreement you may have executed
|
| 137 |
+
with Licensor regarding such Contributions.
|
| 138 |
+
|
| 139 |
+
6. Trademarks. This License does not grant permission to use the trade
|
| 140 |
+
names, trademarks, service marks, or product names of the Licensor,
|
| 141 |
+
except as required for reasonable and customary use in describing the
|
| 142 |
+
origin of the Work and reproducing the content of the NOTICE file.
|
| 143 |
+
|
| 144 |
+
7. Disclaimer of Warranty. Unless required by applicable law or
|
| 145 |
+
agreed to in writing, Licensor provides the Work (and each
|
| 146 |
+
Contributor provides its Contributions) on an "AS IS" BASIS,
|
| 147 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
| 148 |
+
implied, including, without limitation, any warranties or conditions
|
| 149 |
+
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
| 150 |
+
PARTICULAR PURPOSE. You are solely responsible for determining the
|
| 151 |
+
appropriateness of using or redistributing the Work and assume any
|
| 152 |
+
risks associated with Your exercise of permissions under this License.
|
| 153 |
+
|
| 154 |
+
8. Limitation of Liability. In no event and under no legal theory,
|
| 155 |
+
whether in tort (including negligence), contract, or otherwise,
|
| 156 |
+
unless required by applicable law (such as deliberate and grossly
|
| 157 |
+
negligent acts) or agreed to in writing, shall any Contributor be
|
| 158 |
+
liable to You for damages, including any direct, indirect, special,
|
| 159 |
+
incidental, or consequential damages of any character arising as a
|
| 160 |
+
result of this License or out of the use or inability to use the
|
| 161 |
+
Work (including but not limited to damages for loss of goodwill,
|
| 162 |
+
work stoppage, computer failure or malfunction, or any and all
|
| 163 |
+
other commercial damages or losses), even if such Contributor
|
| 164 |
+
has been advised of the possibility of such damages.
|
| 165 |
+
|
| 166 |
+
9. Accepting Warranty or Additional Liability. While redistributing
|
| 167 |
+
the Work or Derivative Works thereof, You may choose to offer,
|
| 168 |
+
and charge a fee for, acceptance of support, warranty, indemnity,
|
| 169 |
+
or other liability obligations and/or rights consistent with this
|
| 170 |
+
License. However, in accepting such obligations, You may act only
|
| 171 |
+
on Your own behalf and on Your sole responsibility, not on behalf
|
| 172 |
+
of any other Contributor, and only if You agree to indemnify,
|
| 173 |
+
defend, and hold each Contributor harmless for any liability
|
| 174 |
+
incurred by, or claims asserted against, such Contributor by reason
|
| 175 |
+
of your accepting any such warranty or additional liability.
|
| 176 |
+
|
| 177 |
+
END OF TERMS AND CONDITIONS
|
| 178 |
+
|
| 179 |
+
APPENDIX: How to apply the Apache License to your work.
|
| 180 |
+
|
| 181 |
+
To apply the Apache License to your work, attach the following
|
| 182 |
+
boilerplate notice, with the fields enclosed by brackets "[]"
|
| 183 |
+
replaced with your own identifying information. (Don't include
|
| 184 |
+
the brackets!) The text should be enclosed in the appropriate
|
| 185 |
+
comment syntax for the file format. We also recommend that a
|
| 186 |
+
file or class name and description of purpose be included on the
|
| 187 |
+
same "printed page" as the copyright notice for easier
|
| 188 |
+
identification within third-party archives.
|
| 189 |
+
|
| 190 |
+
Copyright 2026 Alibaba Cloud
|
| 191 |
+
|
| 192 |
+
Licensed under the Apache License, Version 2.0 (the "License");
|
| 193 |
+
you may not use this file except in compliance with the License.
|
| 194 |
+
You may obtain a copy of the License at
|
| 195 |
+
|
| 196 |
+
http://www.apache.org/licenses/LICENSE-2.0
|
| 197 |
+
|
| 198 |
+
Unless required by applicable law or agreed to in writing, software
|
| 199 |
+
distributed under the License is distributed on an "AS IS" BASIS,
|
| 200 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
| 201 |
+
See the License for the specific language governing permissions and
|
| 202 |
+
limitations under the License.
|
NOTICE.md
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Preserved notices and release provenance
|
| 2 |
+
|
| 3 |
+
The original checkpoint's `LICENSE` file is included without modification.
|
| 4 |
+
It contains Apache License 2.0 and an Alibaba Cloud copyright notice.
|
| 5 |
+
`tokenization_interns1.py` retains the Intern team and Shanghai AI Lab notice
|
| 6 |
+
and all copied-code attributions present in the supplied file.
|
| 7 |
+
|
| 8 |
+
The checkpoint README has been replaced with an AutoVerifier-specific model
|
| 9 |
+
card. Its original general-model description is retained only in the local
|
| 10 |
+
preparation audit directory and is not represented as this verifier's results.
|
| 11 |
+
|
| 12 |
+
All model weights, model/tokenizer configuration files, processor files, and
|
| 13 |
+
the chat template are unmodified. The `merge_ratio.json` preparation metadata
|
| 14 |
+
is retained in the local audit directory, not in this runtime package.
|
| 15 |
+
|
| 16 |
+
Exact upstream model lineage and authorization for the final model release
|
| 17 |
+
must be confirmed by the owners. The inherited license text is not a license
|
| 18 |
+
grant for the separate AdvancedMathBench dataset.
|
README.md
ADDED
|
@@ -0,0 +1,76 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
library_name: transformers
|
| 3 |
+
language:
|
| 4 |
+
- en
|
| 5 |
+
- zh
|
| 6 |
+
tags:
|
| 7 |
+
- mathematics
|
| 8 |
+
- proof-verification
|
| 9 |
+
- advancedmathbench
|
| 10 |
+
- qwen3_5_moe
|
| 11 |
+
---
|
| 12 |
+
# AdvancedMathBench AutoVerifier
|
| 13 |
+
|
| 14 |
+
AutoVerifier evaluates natural-language mathematical proofs, explains errors,
|
| 15 |
+
and identifies the earliest incorrect step. It serves as the automatic grader
|
| 16 |
+
for AdvancedMathBench's ProverBench.
|
| 17 |
+
|
| 18 |
+
## Model
|
| 19 |
+
|
| 20 |
+
- Architecture: `Qwen3_5MoeForConditionalGeneration`.
|
| 21 |
+
- Tokenizer: bundled `InternS1Tokenizer`; requires `sentencepiece` and
|
| 22 |
+
`trust_remote_code=True` after reviewing the tokenizer code.
|
| 23 |
+
- Weights: 40 safetensors shards, approximately 68 GiB.
|
| 24 |
+
|
| 25 |
+
## Input
|
| 26 |
+
|
| 27 |
+
Use [proof_verifier.md](prompts/proof_verifier.md) with a problem, an optional
|
| 28 |
+
reference solution, and a candidate proof split into zero-indexed steps.
|
| 29 |
+
The following constructs the input without loading the model weights:
|
| 30 |
+
|
| 31 |
+
```python
|
| 32 |
+
from pathlib import Path
|
| 33 |
+
from transformers import AutoTokenizer
|
| 34 |
+
|
| 35 |
+
model_dir = "." # Local model repository directory
|
| 36 |
+
tokenizer = AutoTokenizer.from_pretrained(model_dir, trust_remote_code=True)
|
| 37 |
+
|
| 38 |
+
steps = ["A candidate proof step.", "Another candidate proof step."]
|
| 39 |
+
proof = "\n\n".join(
|
| 40 |
+
f"<step{i}>\n\n{step}\n\n</step{i}>" for i, step in enumerate(steps)
|
| 41 |
+
)
|
| 42 |
+
template = Path(model_dir, "prompts/proof_verifier.md").read_text(encoding="utf-8")
|
| 43 |
+
prompt = template.format(
|
| 44 |
+
problem="The mathematical problem.", human_solution="", solution=proof,
|
| 45 |
+
)
|
| 46 |
+
text = tokenizer.apply_chat_template(
|
| 47 |
+
[{"role": "user", "content": prompt}],
|
| 48 |
+
tokenize=False, add_generation_prompt=True, enable_thinking=True,
|
| 49 |
+
)
|
| 50 |
+
```
|
| 51 |
+
|
| 52 |
+
## Output and scoring
|
| 53 |
+
|
| 54 |
+
The final response contains an assessment, identified errors, and the first
|
| 55 |
+
error index. For example, a no-error judgment is:
|
| 56 |
+
|
| 57 |
+
```xml
|
| 58 |
+
<assessment>The proof is correct.</assessment>
|
| 59 |
+
<errors></errors>
|
| 60 |
+
<first_error_step>-1</first_error_step>
|
| 61 |
+
```
|
| 62 |
+
|
| 63 |
+
`-1` means no error was found; nonnegative indices identify the earliest error,
|
| 64 |
+
starting from **0**. Parse the final answer after `</think>` when present.
|
| 65 |
+
|
| 66 |
+
ProverBench checks each proof **8 times** and accepts it only when all eight
|
| 67 |
+
valid judgments report `-1`. Missing or malformed judgments do not count as
|
| 68 |
+
acceptance. Sampling settings are documented in
|
| 69 |
+
[evaluation_settings.json](evaluation_settings.json).
|
| 70 |
+
|
| 71 |
+
## Notes
|
| 72 |
+
|
| 73 |
+
AutoVerifier is a learned grader, not a formal proof checker, and can make errors.
|
| 74 |
+
Tested package versions and validation scope are recorded in
|
| 75 |
+
[compatibility.json](compatibility.json). License notices are provided in [LICENSE](LICENSE) and
|
| 76 |
+
[NOTICE.md](NOTICE.md).
|
chat_template.jinja
ADDED
|
@@ -0,0 +1,154 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{%- set image_count = namespace(value=0) %}
|
| 2 |
+
{%- set video_count = namespace(value=0) %}
|
| 3 |
+
{%- macro render_content(content, do_vision_count, is_system_content=false) %}
|
| 4 |
+
{%- if content is string %}
|
| 5 |
+
{{- content }}
|
| 6 |
+
{%- elif content is iterable and content is not mapping %}
|
| 7 |
+
{%- for item in content %}
|
| 8 |
+
{%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
|
| 9 |
+
{%- if is_system_content %}
|
| 10 |
+
{{- raise_exception('System message cannot contain images.') }}
|
| 11 |
+
{%- endif %}
|
| 12 |
+
{%- if do_vision_count %}
|
| 13 |
+
{%- set image_count.value = image_count.value + 1 %}
|
| 14 |
+
{%- endif %}
|
| 15 |
+
{%- if add_vision_id %}
|
| 16 |
+
{{- 'Picture ' ~ image_count.value ~ ': ' }}
|
| 17 |
+
{%- endif %}
|
| 18 |
+
{{- '<|vision_start|><|image_pad|><|vision_end|>' }}
|
| 19 |
+
{%- elif 'video' in item or item.type == 'video' %}
|
| 20 |
+
{%- if is_system_content %}
|
| 21 |
+
{{- raise_exception('System message cannot contain videos.') }}
|
| 22 |
+
{%- endif %}
|
| 23 |
+
{%- if do_vision_count %}
|
| 24 |
+
{%- set video_count.value = video_count.value + 1 %}
|
| 25 |
+
{%- endif %}
|
| 26 |
+
{%- if add_vision_id %}
|
| 27 |
+
{{- 'Video ' ~ video_count.value ~ ': ' }}
|
| 28 |
+
{%- endif %}
|
| 29 |
+
{{- '<|vision_start|><|video_pad|><|vision_end|>' }}
|
| 30 |
+
{%- elif 'text' in item %}
|
| 31 |
+
{{- item.text }}
|
| 32 |
+
{%- else %}
|
| 33 |
+
{{- raise_exception('Unexpected item type in content.') }}
|
| 34 |
+
{%- endif %}
|
| 35 |
+
{%- endfor %}
|
| 36 |
+
{%- elif content is none or content is undefined %}
|
| 37 |
+
{{- '' }}
|
| 38 |
+
{%- else %}
|
| 39 |
+
{{- raise_exception('Unexpected content type.') }}
|
| 40 |
+
{%- endif %}
|
| 41 |
+
{%- endmacro %}
|
| 42 |
+
{%- if not messages %}
|
| 43 |
+
{{- raise_exception('No messages provided.') }}
|
| 44 |
+
{%- endif %}
|
| 45 |
+
{%- if tools and tools is iterable and tools is not mapping %}
|
| 46 |
+
{{- '<|im_start|>system\n' }}
|
| 47 |
+
{{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
|
| 48 |
+
{%- for tool in tools %}
|
| 49 |
+
{{- "\n" }}
|
| 50 |
+
{{- tool | tojson }}
|
| 51 |
+
{%- endfor %}
|
| 52 |
+
{{- "\n</tools>" }}
|
| 53 |
+
{{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
|
| 54 |
+
{%- if messages[0].role == 'system' %}
|
| 55 |
+
{%- set content = render_content(messages[0].content, false, true)|trim %}
|
| 56 |
+
{%- if content %}
|
| 57 |
+
{{- '\n\n' + content }}
|
| 58 |
+
{%- endif %}
|
| 59 |
+
{%- endif %}
|
| 60 |
+
{{- '<|im_end|>\n' }}
|
| 61 |
+
{%- else %}
|
| 62 |
+
{%- if messages[0].role == 'system' %}
|
| 63 |
+
{%- set content = render_content(messages[0].content, false, true)|trim %}
|
| 64 |
+
{{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
|
| 65 |
+
{%- endif %}
|
| 66 |
+
{%- endif %}
|
| 67 |
+
{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
|
| 68 |
+
{%- for message in messages[::-1] %}
|
| 69 |
+
{%- set index = (messages|length - 1) - loop.index0 %}
|
| 70 |
+
{%- if ns.multi_step_tool and message.role == "user" %}
|
| 71 |
+
{%- set content = render_content(message.content, false)|trim %}
|
| 72 |
+
{%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
|
| 73 |
+
{%- set ns.multi_step_tool = false %}
|
| 74 |
+
{%- set ns.last_query_index = index %}
|
| 75 |
+
{%- endif %}
|
| 76 |
+
{%- endif %}
|
| 77 |
+
{%- endfor %}
|
| 78 |
+
{%- if ns.multi_step_tool %}
|
| 79 |
+
{{- raise_exception('No user query found in messages.') }}
|
| 80 |
+
{%- endif %}
|
| 81 |
+
{%- for message in messages %}
|
| 82 |
+
{%- set content = render_content(message.content, true)|trim %}
|
| 83 |
+
{%- if message.role == "system" %}
|
| 84 |
+
{%- if not loop.first %}
|
| 85 |
+
{{- raise_exception('System message must be at the beginning.') }}
|
| 86 |
+
{%- endif %}
|
| 87 |
+
{%- elif message.role == "user" %}
|
| 88 |
+
{{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
|
| 89 |
+
{%- elif message.role == "assistant" %}
|
| 90 |
+
{%- set reasoning_content = '' %}
|
| 91 |
+
{%- if message.reasoning_content is string %}
|
| 92 |
+
{%- set reasoning_content = message.reasoning_content %}
|
| 93 |
+
{%- else %}
|
| 94 |
+
{%- if '</think>' in content %}
|
| 95 |
+
{%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
|
| 96 |
+
{%- set content = content.split('</think>')[-1].lstrip('\n') %}
|
| 97 |
+
{%- endif %}
|
| 98 |
+
{%- endif %}
|
| 99 |
+
{%- set reasoning_content = reasoning_content|trim %}
|
| 100 |
+
{%- if loop.index0 > ns.last_query_index %}
|
| 101 |
+
{{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
|
| 102 |
+
{%- else %}
|
| 103 |
+
{{- '<|im_start|>' + message.role + '\n' + content }}
|
| 104 |
+
{%- endif %}
|
| 105 |
+
{%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
|
| 106 |
+
{%- for tool_call in message.tool_calls %}
|
| 107 |
+
{%- if tool_call.function is defined %}
|
| 108 |
+
{%- set tool_call = tool_call.function %}
|
| 109 |
+
{%- endif %}
|
| 110 |
+
{%- if loop.first %}
|
| 111 |
+
{%- if content|trim %}
|
| 112 |
+
{{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
|
| 113 |
+
{%- else %}
|
| 114 |
+
{{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
|
| 115 |
+
{%- endif %}
|
| 116 |
+
{%- else %}
|
| 117 |
+
{{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
|
| 118 |
+
{%- endif %}
|
| 119 |
+
{%- if tool_call.arguments is defined %}
|
| 120 |
+
{%- for args_name, args_value in tool_call.arguments|items %}
|
| 121 |
+
{{- '<parameter=' + args_name + '>\n' }}
|
| 122 |
+
{%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
|
| 123 |
+
{{- args_value }}
|
| 124 |
+
{{- '\n</parameter>\n' }}
|
| 125 |
+
{%- endfor %}
|
| 126 |
+
{%- endif %}
|
| 127 |
+
{{- '</function>\n</tool_call>' }}
|
| 128 |
+
{%- endfor %}
|
| 129 |
+
{%- endif %}
|
| 130 |
+
{{- '<|im_end|>\n' }}
|
| 131 |
+
{%- elif message.role == "tool" %}
|
| 132 |
+
{%- if loop.previtem and loop.previtem.role != "tool" %}
|
| 133 |
+
{{- '<|im_start|>user' }}
|
| 134 |
+
{%- endif %}
|
| 135 |
+
{{- '\n<tool_response>\n' }}
|
| 136 |
+
{{- content }}
|
| 137 |
+
{{- '\n</tool_response>' }}
|
| 138 |
+
{%- if not loop.last and loop.nextitem.role != "tool" %}
|
| 139 |
+
{{- '<|im_end|>\n' }}
|
| 140 |
+
{%- elif loop.last %}
|
| 141 |
+
{{- '<|im_end|>\n' }}
|
| 142 |
+
{%- endif %}
|
| 143 |
+
{%- else %}
|
| 144 |
+
{{- raise_exception('Unexpected message role.') }}
|
| 145 |
+
{%- endif %}
|
| 146 |
+
{%- endfor %}
|
| 147 |
+
{%- if add_generation_prompt %}
|
| 148 |
+
{{- '<|im_start|>assistant\n' }}
|
| 149 |
+
{%- if enable_thinking is defined and enable_thinking is false %}
|
| 150 |
+
{{- '<think>\n\n</think>\n\n' }}
|
| 151 |
+
{%- else %}
|
| 152 |
+
{{- '<think>\n' }}
|
| 153 |
+
{%- endif %}
|
| 154 |
+
{%- endif %}
|
checkpoint_checksums.sha256
ADDED
|
@@ -0,0 +1,56 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
bbedc3fda3305820b977265f01b8619d87570a6739de3a5582c3464840f1e57a LICENSE
|
| 2 |
+
a4aee8afcf2e0711942cf848899be66016f8d14a889ff9ede07bca099c28f715 chat_template.jinja
|
| 3 |
+
2848b0e6ac95267352d439a41d6efba7951a5870ce9a0d1daffb4b4d1d079840 config.json
|
| 4 |
+
4f25002776b741773666203dcea8f54619f177ace3ae483d311102092a4658e0 generation_config.json
|
| 5 |
+
3bd640ba6d8da8f5844f3548b7e2184fc664dd670e1447e425a48dbaaad1ef43 merges.txt
|
| 6 |
+
6dc5cb409e57c647a07900539977f4459988da9aae18c713e78c040f9e59e670 model-language-0001-fused-save_rank0.safetensors
|
| 7 |
+
7f74cda2afc0b6cc5f5aee61243d4ea19ec0ef8892f1ce15d3d75f44965ed093 model-language-0001-fused-save_rank1.safetensors
|
| 8 |
+
e7e59ac9144d6ff7234fbfa019f7453a0fd46d5cfa83ea24f1a2d69ea0b69da7 model-language-0001-fused-save_rank10.safetensors
|
| 9 |
+
d308b89f5a235fa5c4b56d54983f27d52f8f41c41896f3e25ed5d10708a901ab model-language-0001-fused-save_rank11.safetensors
|
| 10 |
+
2e65f0d33478f1d87a882acd4f6ee631ffa13a02c04507408777def4117bc791 model-language-0001-fused-save_rank12.safetensors
|
| 11 |
+
5b7616b8ebebcdaafcac6ecf74c3803441a71df622bf688b8e89fd5665f7bac5 model-language-0001-fused-save_rank13.safetensors
|
| 12 |
+
40277357e610e556ba2435737123bca4801ec37888eb1eaff7a95d06723f68ef model-language-0001-fused-save_rank14.safetensors
|
| 13 |
+
5f3238f4e8a27f33ecfcd15452a97c363a895453f7e9c966041102ab878dbbba model-language-0001-fused-save_rank15.safetensors
|
| 14 |
+
ffd0b63509989f3b107572451fb9d721a4a06429d71f80953534d7ecb5dc2a63 model-language-0001-fused-save_rank2.safetensors
|
| 15 |
+
8a8ad9eb11e3686e45c7f075544f2f674c643855bcbed3be1c509976b97ef555 model-language-0001-fused-save_rank3.safetensors
|
| 16 |
+
1607938699926513b4cbc1ebf6d2f6c8a30bc2d0177a683054bd9b15b86e62ad model-language-0001-fused-save_rank4.safetensors
|
| 17 |
+
7884f2544f54dd507ad0303bfe50879fba1633749ef9e29721e3c2a71ad9aac0 model-language-0001-fused-save_rank5.safetensors
|
| 18 |
+
ce2035b780b3fe2f4cc932faab39342053557ba04e0180cae42299c53f34d6cd model-language-0001-fused-save_rank6.safetensors
|
| 19 |
+
ca9f85c24bb6d3946d18ea29b764930297f14fecebe110a449d9a5943a0817ed model-language-0001-fused-save_rank7.safetensors
|
| 20 |
+
bd39af2b019444105f9bf5af150e22b0b4871d7b3c651e36548cd1223af52166 model-language-0001-fused-save_rank8.safetensors
|
| 21 |
+
ee9797ad6fd78d6c81dac053f757b914537297c79eaaaacffb614c2633c2b15a model-language-0001-fused-save_rank9.safetensors
|
| 22 |
+
9981fb59074c1a0d4bb3ae339f769746b2e7e5080b91f30e91a9308d61be659e model-language-0001-others-save_rank0.safetensors
|
| 23 |
+
cb7fa1e4bd79b049d8cb214f5c68a8cb956b3f05c5183a11a806544031b0f235 model-language-0002-others-save_rank0.safetensors
|
| 24 |
+
42b10a711a10996d73560708adee75f198bfa6eefb0451d248a3ec3a5958489e model-language-0003-others-save_rank0.safetensors
|
| 25 |
+
4e580d5dc2e5466279982a26df9398a4e0e1a51e83e0b9cade6c69b95ceb32c7 model-language-0004-others-save_rank0.safetensors
|
| 26 |
+
fc265d902e8c955469c6f71b68379fc73e24b3069ba863dd6fa1a92c6c35d02d model-language-0005-others-save_rank0.safetensors
|
| 27 |
+
94ba85645b6373792cd77878502a0c4f332afe4bea3a0fd110782eaf527044ee model-language-0006-others-save_rank0.safetensors
|
| 28 |
+
74c5a4a4f1a6c7c67b998efcc7a78464f5aa9311a8a2cd31811afa9b0928e5bb model-language-0007-others-save_rank0.safetensors
|
| 29 |
+
9624740aad902bf6e510ce750275057e77bd1e52ea9a7d7736bb8a75ee18ac42 model-language-0008-others-save_rank0.safetensors
|
| 30 |
+
13ab1770ade382328d894f8f4ac00b47d189d9fed83be733731a95d11eb31706 model-language-0009-others-save_rank0.safetensors
|
| 31 |
+
0d158718e05cc09e72e4d81cef437a549abcbc9c13618b75b8912cefadf56c56 model-language-0010-others-save_rank0.safetensors
|
| 32 |
+
f1c4bceffcaaccec6d435636f09fb7d161cc24e6ff8a03f5a57fda1e161e87a0 model-language-0011-others-save_rank0.safetensors
|
| 33 |
+
2bb36f2a6c1854c8166fd3c7d4c49d3601151572ce9d00fa5cbc77aabd5e6b0b model-language-0012-others-save_rank0.safetensors
|
| 34 |
+
0eeb09bf430b2be9db8498c68f7dd00156eda974de87e6edc78477c453e7a6c3 model-language-0013-others-save_rank0.safetensors
|
| 35 |
+
771ec8dcaad6fa681dc64734f6bb28db4b3facd02db456325a99e6467be4d8e9 model-language-0014-others-save_rank0.safetensors
|
| 36 |
+
aa6859905f1e45242a8d274dab76b15aa5c96637fd88aaf959dee6b5b855ddcc model-language-0015-others-save_rank0.safetensors
|
| 37 |
+
c80667bf0680ac6a31e04826f188747f1e348d4f31637d251ec82e7c8cf58880 model-language-0016-others-save_rank0.safetensors
|
| 38 |
+
ffcb3d069d1b92dd0e98ed1150cc1e55dbb6ea73bbd5cfe752cb7eb8886461da model-language-0017-others-save_rank0.safetensors
|
| 39 |
+
98d95974f2e17cdff6575d35f3367fe9618ec56cf384abcde161771769095f83 model-language-0018-others-save_rank0.safetensors
|
| 40 |
+
160b6114d8607073b20000fc665ca6548e9a08facc00c234ed5457056d185175 model-language-0019-others-save_rank0.safetensors
|
| 41 |
+
d075ed8040661d4f9586ab7de09220a6484dec1ad3776210478ea7697577a626 model-language-0020-others-save_rank0.safetensors
|
| 42 |
+
10a6c734ad7ab213d3d12d0404cee6df31f7364b8ce179d1d1c034ef5ec8dd26 model-language-0021-others-save_rank0.safetensors
|
| 43 |
+
0c94ebd2a5ea3a3496253262c5f3c5cf7235d30a0ea84b7b4a8fb0b9018d68df model-language-0022-others-save_rank0.safetensors
|
| 44 |
+
7b38702ff3948f711782b3975d189e3cb67ca1eedbc53e9be13398016afedd50 model-projector-0001-others-save_rank0.safetensors
|
| 45 |
+
1fb153b047f5c3debfffc5efa74e853bbf04278c6268f1eb5b883b6a250100c9 model-vision-0001-others-save_rank0.safetensors
|
| 46 |
+
05124f9e7a8566158eb83934bc2d3a59700640b9296aed654fbe1d2eda6ed39a model.safetensors.index.json
|
| 47 |
+
27225450ac9c6529872ee1924fcb0962ff5634834f817040f444118116f4e516 preprocessor_config.json
|
| 48 |
+
9c6c4c51d1c410cf5563bea65da6a80d06f42ab2abf091a9f7caca5a03fc9aec special_tokens_map.json
|
| 49 |
+
ad0ecddabf936ad205382ebc7dc87755eb711c2833ebd590179188a7fb2716ac tokenization_interns1.py
|
| 50 |
+
5f9e4d4901a92b997e463c1f46055088b6cca5ca61a6522d1b9f64c4bb81cb42 tokenizer.json
|
| 51 |
+
1144f52f86f3ca5a29940d69b037e508c05a89e6eedbe42bea641e226b20dbe0 tokenizer_PROT.model
|
| 52 |
+
fba1c97da0353ccbffd368ae78e311ccbc762aa5ba74f9aff8bf2ab363c4d37d tokenizer_SMILES.model
|
| 53 |
+
58fc8bfb2af3dfe936a13dad8a9cb28dab7850b70b358db19605d867c133fb35 tokenizer_XNA.model
|
| 54 |
+
f2a82034eae0bfaa63d19d9648d88af0b99fdc0914de1bfbbb4621801e2d3e74 tokenizer_config.json
|
| 55 |
+
7768af27c1fafa9cc9011c1dc20067e03f8915e03b63504550e11d5066986d13 video_preprocessor_config.json
|
| 56 |
+
657191b77e1d51b28521773221d4121bfe4844900147f7818499e0347d8c66f6 vocab.json
|
checkpoint_manifest.json
ADDED
|
@@ -0,0 +1,306 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"artifact": "AdvancedMathBench-AutoVerifier",
|
| 3 |
+
"version": "hf-80-release-rc1",
|
| 4 |
+
"prepared_on": "2026-09-29",
|
| 5 |
+
"copy_mode": "independent files; no symlinks or hardlinks",
|
| 6 |
+
"weights_modified": false,
|
| 7 |
+
"files": [
|
| 8 |
+
{
|
| 9 |
+
"file": "LICENSE",
|
| 10 |
+
"bytes": 11544,
|
| 11 |
+
"sha256": "bbedc3fda3305820b977265f01b8619d87570a6739de3a5582c3464840f1e57a"
|
| 12 |
+
},
|
| 13 |
+
{
|
| 14 |
+
"file": "chat_template.jinja",
|
| 15 |
+
"bytes": 7756,
|
| 16 |
+
"sha256": "a4aee8afcf2e0711942cf848899be66016f8d14a889ff9ede07bca099c28f715"
|
| 17 |
+
},
|
| 18 |
+
{
|
| 19 |
+
"file": "config.json",
|
| 20 |
+
"bytes": 3051,
|
| 21 |
+
"sha256": "2848b0e6ac95267352d439a41d6efba7951a5870ce9a0d1daffb4b4d1d079840"
|
| 22 |
+
},
|
| 23 |
+
{
|
| 24 |
+
"file": "generation_config.json",
|
| 25 |
+
"bytes": 244,
|
| 26 |
+
"sha256": "4f25002776b741773666203dcea8f54619f177ace3ae483d311102092a4658e0"
|
| 27 |
+
},
|
| 28 |
+
{
|
| 29 |
+
"file": "merges.txt",
|
| 30 |
+
"bytes": 3353273,
|
| 31 |
+
"sha256": "3bd640ba6d8da8f5844f3548b7e2184fc664dd670e1447e425a48dbaaad1ef43"
|
| 32 |
+
},
|
| 33 |
+
{
|
| 34 |
+
"file": "model-language-0001-fused-save_rank0.safetensors",
|
| 35 |
+
"bytes": 100668896,
|
| 36 |
+
"sha256": "6dc5cb409e57c647a07900539977f4459988da9aae18c713e78c040f9e59e670"
|
| 37 |
+
},
|
| 38 |
+
{
|
| 39 |
+
"file": "model-language-0001-fused-save_rank1.safetensors",
|
| 40 |
+
"bytes": 100668928,
|
| 41 |
+
"sha256": "7f74cda2afc0b6cc5f5aee61243d4ea19ec0ef8892f1ce15d3d75f44965ed093"
|
| 42 |
+
},
|
| 43 |
+
{
|
| 44 |
+
"file": "model-language-0001-fused-save_rank10.safetensors",
|
| 45 |
+
"bytes": 100668976,
|
| 46 |
+
"sha256": "e7e59ac9144d6ff7234fbfa019f7453a0fd46d5cfa83ea24f1a2d69ea0b69da7"
|
| 47 |
+
},
|
| 48 |
+
{
|
| 49 |
+
"file": "model-language-0001-fused-save_rank11.safetensors",
|
| 50 |
+
"bytes": 100668976,
|
| 51 |
+
"sha256": "d308b89f5a235fa5c4b56d54983f27d52f8f41c41896f3e25ed5d10708a901ab"
|
| 52 |
+
},
|
| 53 |
+
{
|
| 54 |
+
"file": "model-language-0001-fused-save_rank12.safetensors",
|
| 55 |
+
"bytes": 100668976,
|
| 56 |
+
"sha256": "2e65f0d33478f1d87a882acd4f6ee631ffa13a02c04507408777def4117bc791"
|
| 57 |
+
},
|
| 58 |
+
{
|
| 59 |
+
"file": "model-language-0001-fused-save_rank13.safetensors",
|
| 60 |
+
"bytes": 100668976,
|
| 61 |
+
"sha256": "5b7616b8ebebcdaafcac6ecf74c3803441a71df622bf688b8e89fd5665f7bac5"
|
| 62 |
+
},
|
| 63 |
+
{
|
| 64 |
+
"file": "model-language-0001-fused-save_rank14.safetensors",
|
| 65 |
+
"bytes": 100668976,
|
| 66 |
+
"sha256": "40277357e610e556ba2435737123bca4801ec37888eb1eaff7a95d06723f68ef"
|
| 67 |
+
},
|
| 68 |
+
{
|
| 69 |
+
"file": "model-language-0001-fused-save_rank15.safetensors",
|
| 70 |
+
"bytes": 100668976,
|
| 71 |
+
"sha256": "5f3238f4e8a27f33ecfcd15452a97c363a895453f7e9c966041102ab878dbbba"
|
| 72 |
+
},
|
| 73 |
+
{
|
| 74 |
+
"file": "model-language-0001-fused-save_rank2.safetensors",
|
| 75 |
+
"bytes": 100668928,
|
| 76 |
+
"sha256": "ffd0b63509989f3b107572451fb9d721a4a06429d71f80953534d7ecb5dc2a63"
|
| 77 |
+
},
|
| 78 |
+
{
|
| 79 |
+
"file": "model-language-0001-fused-save_rank3.safetensors",
|
| 80 |
+
"bytes": 100668928,
|
| 81 |
+
"sha256": "8a8ad9eb11e3686e45c7f075544f2f674c643855bcbed3be1c509976b97ef555"
|
| 82 |
+
},
|
| 83 |
+
{
|
| 84 |
+
"file": "model-language-0001-fused-save_rank4.safetensors",
|
| 85 |
+
"bytes": 100668928,
|
| 86 |
+
"sha256": "1607938699926513b4cbc1ebf6d2f6c8a30bc2d0177a683054bd9b15b86e62ad"
|
| 87 |
+
},
|
| 88 |
+
{
|
| 89 |
+
"file": "model-language-0001-fused-save_rank5.safetensors",
|
| 90 |
+
"bytes": 100668928,
|
| 91 |
+
"sha256": "7884f2544f54dd507ad0303bfe50879fba1633749ef9e29721e3c2a71ad9aac0"
|
| 92 |
+
},
|
| 93 |
+
{
|
| 94 |
+
"file": "model-language-0001-fused-save_rank6.safetensors",
|
| 95 |
+
"bytes": 100668960,
|
| 96 |
+
"sha256": "ce2035b780b3fe2f4cc932faab39342053557ba04e0180cae42299c53f34d6cd"
|
| 97 |
+
},
|
| 98 |
+
{
|
| 99 |
+
"file": "model-language-0001-fused-save_rank7.safetensors",
|
| 100 |
+
"bytes": 100668976,
|
| 101 |
+
"sha256": "ca9f85c24bb6d3946d18ea29b764930297f14fecebe110a449d9a5943a0817ed"
|
| 102 |
+
},
|
| 103 |
+
{
|
| 104 |
+
"file": "model-language-0001-fused-save_rank8.safetensors",
|
| 105 |
+
"bytes": 100668976,
|
| 106 |
+
"sha256": "bd39af2b019444105f9bf5af150e22b0b4871d7b3c651e36548cd1223af52166"
|
| 107 |
+
},
|
| 108 |
+
{
|
| 109 |
+
"file": "model-language-0001-fused-save_rank9.safetensors",
|
| 110 |
+
"bytes": 100668976,
|
| 111 |
+
"sha256": "ee9797ad6fd78d6c81dac053f757b914537297c79eaaaacffb614c2633c2b15a"
|
| 112 |
+
},
|
| 113 |
+
{
|
| 114 |
+
"file": "model-language-0001-others-save_rank0.safetensors",
|
| 115 |
+
"bytes": 2059403360,
|
| 116 |
+
"sha256": "9981fb59074c1a0d4bb3ae339f769746b2e7e5080b91f30e91a9308d61be659e"
|
| 117 |
+
},
|
| 118 |
+
{
|
| 119 |
+
"file": "model-language-0002-others-save_rank0.safetensors",
|
| 120 |
+
"bytes": 3445604256,
|
| 121 |
+
"sha256": "cb7fa1e4bd79b049d8cb214f5c68a8cb956b3f05c5183a11a806544031b0f235"
|
| 122 |
+
},
|
| 123 |
+
{
|
| 124 |
+
"file": "model-language-0003-others-save_rank0.safetensors",
|
| 125 |
+
"bytes": 3357898400,
|
| 126 |
+
"sha256": "42b10a711a10996d73560708adee75f198bfa6eefb0451d248a3ec3a5958489e"
|
| 127 |
+
},
|
| 128 |
+
{
|
| 129 |
+
"file": "model-language-0004-others-save_rank0.safetensors",
|
| 130 |
+
"bytes": 3370808760,
|
| 131 |
+
"sha256": "4e580d5dc2e5466279982a26df9398a4e0e1a51e83e0b9cade6c69b95ceb32c7"
|
| 132 |
+
},
|
| 133 |
+
{
|
| 134 |
+
"file": "model-language-0005-others-save_rank0.safetensors",
|
| 135 |
+
"bytes": 3357898400,
|
| 136 |
+
"sha256": "fc265d902e8c955469c6f71b68379fc73e24b3069ba863dd6fa1a92c6c35d02d"
|
| 137 |
+
},
|
| 138 |
+
{
|
| 139 |
+
"file": "model-language-0006-others-save_rank0.safetensors",
|
| 140 |
+
"bytes": 3370808672,
|
| 141 |
+
"sha256": "94ba85645b6373792cd77878502a0c4f332afe4bea3a0fd110782eaf527044ee"
|
| 142 |
+
},
|
| 143 |
+
{
|
| 144 |
+
"file": "model-language-0007-others-save_rank0.safetensors",
|
| 145 |
+
"bytes": 3357898432,
|
| 146 |
+
"sha256": "74c5a4a4f1a6c7c67b998efcc7a78464f5aa9311a8a2cd31811afa9b0928e5bb"
|
| 147 |
+
},
|
| 148 |
+
{
|
| 149 |
+
"file": "model-language-0008-others-save_rank0.safetensors",
|
| 150 |
+
"bytes": 3370808792,
|
| 151 |
+
"sha256": "9624740aad902bf6e510ce750275057e77bd1e52ea9a7d7736bb8a75ee18ac42"
|
| 152 |
+
},
|
| 153 |
+
{
|
| 154 |
+
"file": "model-language-0009-others-save_rank0.safetensors",
|
| 155 |
+
"bytes": 3357898432,
|
| 156 |
+
"sha256": "13ab1770ade382328d894f8f4ac00b47d189d9fed83be733731a95d11eb31706"
|
| 157 |
+
},
|
| 158 |
+
{
|
| 159 |
+
"file": "model-language-0010-others-save_rank0.safetensors",
|
| 160 |
+
"bytes": 3370808792,
|
| 161 |
+
"sha256": "0d158718e05cc09e72e4d81cef437a549abcbc9c13618b75b8912cefadf56c56"
|
| 162 |
+
},
|
| 163 |
+
{
|
| 164 |
+
"file": "model-language-0011-others-save_rank0.safetensors",
|
| 165 |
+
"bytes": 3357898432,
|
| 166 |
+
"sha256": "f1c4bceffcaaccec6d435636f09fb7d161cc24e6ff8a03f5a57fda1e161e87a0"
|
| 167 |
+
},
|
| 168 |
+
{
|
| 169 |
+
"file": "model-language-0012-others-save_rank0.safetensors",
|
| 170 |
+
"bytes": 3370808792,
|
| 171 |
+
"sha256": "2bb36f2a6c1854c8166fd3c7d4c49d3601151572ce9d00fa5cbc77aabd5e6b0b"
|
| 172 |
+
},
|
| 173 |
+
{
|
| 174 |
+
"file": "model-language-0013-others-save_rank0.safetensors",
|
| 175 |
+
"bytes": 3357898432,
|
| 176 |
+
"sha256": "0eeb09bf430b2be9db8498c68f7dd00156eda974de87e6edc78477c453e7a6c3"
|
| 177 |
+
},
|
| 178 |
+
{
|
| 179 |
+
"file": "model-language-0014-others-save_rank0.safetensors",
|
| 180 |
+
"bytes": 3370808792,
|
| 181 |
+
"sha256": "771ec8dcaad6fa681dc64734f6bb28db4b3facd02db456325a99e6467be4d8e9"
|
| 182 |
+
},
|
| 183 |
+
{
|
| 184 |
+
"file": "model-language-0015-others-save_rank0.safetensors",
|
| 185 |
+
"bytes": 3357898432,
|
| 186 |
+
"sha256": "aa6859905f1e45242a8d274dab76b15aa5c96637fd88aaf959dee6b5b855ddcc"
|
| 187 |
+
},
|
| 188 |
+
{
|
| 189 |
+
"file": "model-language-0016-others-save_rank0.safetensors",
|
| 190 |
+
"bytes": 3370808792,
|
| 191 |
+
"sha256": "c80667bf0680ac6a31e04826f188747f1e348d4f31637d251ec82e7c8cf58880"
|
| 192 |
+
},
|
| 193 |
+
{
|
| 194 |
+
"file": "model-language-0017-others-save_rank0.safetensors",
|
| 195 |
+
"bytes": 3357898432,
|
| 196 |
+
"sha256": "ffcb3d069d1b92dd0e98ed1150cc1e55dbb6ea73bbd5cfe752cb7eb8886461da"
|
| 197 |
+
},
|
| 198 |
+
{
|
| 199 |
+
"file": "model-language-0018-others-save_rank0.safetensors",
|
| 200 |
+
"bytes": 3370808792,
|
| 201 |
+
"sha256": "98d95974f2e17cdff6575d35f3367fe9618ec56cf384abcde161771769095f83"
|
| 202 |
+
},
|
| 203 |
+
{
|
| 204 |
+
"file": "model-language-0019-others-save_rank0.safetensors",
|
| 205 |
+
"bytes": 3357898432,
|
| 206 |
+
"sha256": "160b6114d8607073b20000fc665ca6548e9a08facc00c234ed5457056d185175"
|
| 207 |
+
},
|
| 208 |
+
{
|
| 209 |
+
"file": "model-language-0020-others-save_rank0.safetensors",
|
| 210 |
+
"bytes": 3370808792,
|
| 211 |
+
"sha256": "d075ed8040661d4f9586ab7de09220a6484dec1ad3776210478ea7697577a626"
|
| 212 |
+
},
|
| 213 |
+
{
|
| 214 |
+
"file": "model-language-0021-others-save_rank0.safetensors",
|
| 215 |
+
"bytes": 3283107048,
|
| 216 |
+
"sha256": "10a6c734ad7ab213d3d12d0404cee6df31f7364b8ce179d1d1c034ef5ec8dd26"
|
| 217 |
+
},
|
| 218 |
+
{
|
| 219 |
+
"file": "model-language-0022-others-save_rank0.safetensors",
|
| 220 |
+
"bytes": 1108372448,
|
| 221 |
+
"sha256": "0c94ebd2a5ea3a3496253262c5f3c5cf7235d30a0ea84b7b4a8fb0b9018d68df"
|
| 222 |
+
},
|
| 223 |
+
{
|
| 224 |
+
"file": "model-projector-0001-others-save_rank0.safetensors",
|
| 225 |
+
"bytes": 61360248,
|
| 226 |
+
"sha256": "7b38702ff3948f711782b3975d189e3cb67ca1eedbc53e9be13398016afedd50"
|
| 227 |
+
},
|
| 228 |
+
{
|
| 229 |
+
"file": "model-vision-0001-others-save_rank0.safetensors",
|
| 230 |
+
"bytes": 831819240,
|
| 231 |
+
"sha256": "1fb153b047f5c3debfffc5efa74e853bbf04278c6268f1eb5b883b6a250100c9"
|
| 232 |
+
},
|
| 233 |
+
{
|
| 234 |
+
"file": "model.safetensors.index.json",
|
| 235 |
+
"bytes": 202607,
|
| 236 |
+
"sha256": "05124f9e7a8566158eb83934bc2d3a59700640b9296aed654fbe1d2eda6ed39a"
|
| 237 |
+
},
|
| 238 |
+
{
|
| 239 |
+
"file": "preprocessor_config.json",
|
| 240 |
+
"bytes": 390,
|
| 241 |
+
"sha256": "27225450ac9c6529872ee1924fcb0962ff5634834f817040f444118116f4e516"
|
| 242 |
+
},
|
| 243 |
+
{
|
| 244 |
+
"file": "special_tokens_map.json",
|
| 245 |
+
"bytes": 1020,
|
| 246 |
+
"sha256": "9c6c4c51d1c410cf5563bea65da6a80d06f42ab2abf091a9f7caca5a03fc9aec"
|
| 247 |
+
},
|
| 248 |
+
{
|
| 249 |
+
"file": "tokenization_interns1.py",
|
| 250 |
+
"bytes": 42042,
|
| 251 |
+
"sha256": "ad0ecddabf936ad205382ebc7dc87755eb711c2833ebd590179188a7fb2716ac"
|
| 252 |
+
},
|
| 253 |
+
{
|
| 254 |
+
"file": "tokenizer.json",
|
| 255 |
+
"bytes": 12807982,
|
| 256 |
+
"sha256": "5f9e4d4901a92b997e463c1f46055088b6cca5ca61a6522d1b9f64c4bb81cb42"
|
| 257 |
+
},
|
| 258 |
+
{
|
| 259 |
+
"file": "tokenizer_PROT.model",
|
| 260 |
+
"bytes": 12118,
|
| 261 |
+
"sha256": "1144f52f86f3ca5a29940d69b037e508c05a89e6eedbe42bea641e226b20dbe0"
|
| 262 |
+
},
|
| 263 |
+
{
|
| 264 |
+
"file": "tokenizer_SMILES.model",
|
| 265 |
+
"bytes": 14775,
|
| 266 |
+
"sha256": "fba1c97da0353ccbffd368ae78e311ccbc762aa5ba74f9aff8bf2ab363c4d37d"
|
| 267 |
+
},
|
| 268 |
+
{
|
| 269 |
+
"file": "tokenizer_XNA.model",
|
| 270 |
+
"bytes": 15451,
|
| 271 |
+
"sha256": "58fc8bfb2af3dfe936a13dad8a9cb28dab7850b70b358db19605d867c133fb35"
|
| 272 |
+
},
|
| 273 |
+
{
|
| 274 |
+
"file": "tokenizer_config.json",
|
| 275 |
+
"bytes": 11671,
|
| 276 |
+
"sha256": "f2a82034eae0bfaa63d19d9648d88af0b99fdc0914de1bfbbb4621801e2d3e74"
|
| 277 |
+
},
|
| 278 |
+
{
|
| 279 |
+
"file": "video_preprocessor_config.json",
|
| 280 |
+
"bytes": 385,
|
| 281 |
+
"sha256": "7768af27c1fafa9cc9011c1dc20067e03f8915e03b63504550e11d5066986d13"
|
| 282 |
+
},
|
| 283 |
+
{
|
| 284 |
+
"file": "vocab.json",
|
| 285 |
+
"bytes": 6312414,
|
| 286 |
+
"sha256": "657191b77e1d51b28521773221d4121bfe4844900147f7818499e0347d8c66f6"
|
| 287 |
+
}
|
| 288 |
+
],
|
| 289 |
+
"structure": {
|
| 290 |
+
"shards": 40,
|
| 291 |
+
"tensors": 1811,
|
| 292 |
+
"tensor_bytes": 72958512864,
|
| 293 |
+
"stored_elements_by_dtype": {
|
| 294 |
+
"BF16": 35449554800,
|
| 295 |
+
"F32": 514850816
|
| 296 |
+
},
|
| 297 |
+
"stored_elements_by_prefix": {
|
| 298 |
+
"mtp": 844640768,
|
| 299 |
+
"model": 35119764848
|
| 300 |
+
},
|
| 301 |
+
"index_and_all_headers_valid": true
|
| 302 |
+
},
|
| 303 |
+
"generation_validation": "not_run",
|
| 304 |
+
"upstream_license_file_preserved": true,
|
| 305 |
+
"note": "Checkpoint includes MTP tensors, vision tensors, a custom tokenizer and mixed dtypes; no weights were removed or converted."
|
| 306 |
+
}
|
compatibility.json
ADDED
|
@@ -0,0 +1,44 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"validation_kind": "offline CPU metadata, tokenizer, and optional meta-device shape checks",
|
| 3 |
+
"python": "3.10.18",
|
| 4 |
+
"versions": {
|
| 5 |
+
"transformers": "5.8.0",
|
| 6 |
+
"torch": "2.8.0",
|
| 7 |
+
"datasets": "3.6.0",
|
| 8 |
+
"huggingface_hub": "1.14.0",
|
| 9 |
+
"sentencepiece": "0.2.1",
|
| 10 |
+
"tokenizers": "0.22.1",
|
| 11 |
+
"safetensors": "0.6.2",
|
| 12 |
+
"regex": "2026.4.4",
|
| 13 |
+
"packaging": "25.0",
|
| 14 |
+
"jsonschema": "4.25.1"
|
| 15 |
+
},
|
| 16 |
+
"config": {
|
| 17 |
+
"class": "Qwen3_5MoeConfig",
|
| 18 |
+
"model_type": "qwen3_5_moe",
|
| 19 |
+
"architecture": [
|
| 20 |
+
"Qwen3_5MoeForConditionalGeneration"
|
| 21 |
+
],
|
| 22 |
+
"loaded_without_internal_packages": true
|
| 23 |
+
},
|
| 24 |
+
"tokenizer": {
|
| 25 |
+
"class": "InternS1Tokenizer",
|
| 26 |
+
"vocab_size": 251174,
|
| 27 |
+
"roundtrip": "passed",
|
| 28 |
+
"chat_template": "passed",
|
| 29 |
+
"chat_tokens": 20,
|
| 30 |
+
"prompt_tail": "<|im_start|>user\n证明群元素与其逆元具有相同的阶。<|im_end|>\n<|im_start|>assistant\n<think>\n"
|
| 31 |
+
},
|
| 32 |
+
"public_architecture_shape_check": {
|
| 33 |
+
"expected_core_tensors": 1026,
|
| 34 |
+
"checkpoint_tensors": 1811,
|
| 35 |
+
"missing_core_keys": 0,
|
| 36 |
+
"shape_mismatches": 0,
|
| 37 |
+
"unexpected_keys": 785,
|
| 38 |
+
"all_unexpected_keys_are_mtp": true,
|
| 39 |
+
"core_parameters_match": true
|
| 40 |
+
},
|
| 41 |
+
"full_weight_loading": "not_run",
|
| 42 |
+
"generation": "not_run",
|
| 43 |
+
"mtp_note": "All original MTP tensors are retained; the tested public architecture has no matching parameters for them. Shape compatibility alone does not validate from_pretrained loading or inference."
|
| 44 |
+
}
|
config.json
ADDED
|
@@ -0,0 +1,116 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"architectures": [
|
| 3 |
+
"Qwen3_5MoeForConditionalGeneration"
|
| 4 |
+
],
|
| 5 |
+
"image_token_id": 248056,
|
| 6 |
+
"model_type": "qwen3_5_moe",
|
| 7 |
+
"text_config": {
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"attn_output_gate": true,
|
| 11 |
+
"dtype": "bfloat16",
|
| 12 |
+
"eos_token_id": 248044,
|
| 13 |
+
"full_attention_interval": 4,
|
| 14 |
+
"head_dim": 256,
|
| 15 |
+
"hidden_act": "silu",
|
| 16 |
+
"hidden_size": 2048,
|
| 17 |
+
"initializer_range": 0.02,
|
| 18 |
+
"layer_types": [
|
| 19 |
+
"linear_attention",
|
| 20 |
+
"linear_attention",
|
| 21 |
+
"linear_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"linear_attention",
|
| 24 |
+
"linear_attention",
|
| 25 |
+
"linear_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"linear_attention",
|
| 28 |
+
"linear_attention",
|
| 29 |
+
"linear_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"linear_attention",
|
| 32 |
+
"linear_attention",
|
| 33 |
+
"linear_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"linear_attention",
|
| 36 |
+
"linear_attention",
|
| 37 |
+
"linear_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"linear_attention",
|
| 40 |
+
"linear_attention",
|
| 41 |
+
"linear_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"linear_attention",
|
| 44 |
+
"linear_attention",
|
| 45 |
+
"linear_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"linear_attention",
|
| 48 |
+
"linear_attention",
|
| 49 |
+
"linear_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"linear_attention",
|
| 52 |
+
"linear_attention",
|
| 53 |
+
"linear_attention",
|
| 54 |
+
"full_attention",
|
| 55 |
+
"linear_attention",
|
| 56 |
+
"linear_attention",
|
| 57 |
+
"linear_attention",
|
| 58 |
+
"full_attention"
|
| 59 |
+
],
|
| 60 |
+
"linear_conv_kernel_dim": 4,
|
| 61 |
+
"linear_key_head_dim": 128,
|
| 62 |
+
"linear_num_key_heads": 16,
|
| 63 |
+
"linear_num_value_heads": 32,
|
| 64 |
+
"linear_value_head_dim": 128,
|
| 65 |
+
"max_position_embeddings": 262144,
|
| 66 |
+
"mlp_only_layers": [],
|
| 67 |
+
"model_type": "qwen3_5_moe_text",
|
| 68 |
+
"moe_intermediate_size": 512,
|
| 69 |
+
"mtp_num_hidden_layers": 1,
|
| 70 |
+
"mtp_use_dedicated_embeddings": false,
|
| 71 |
+
"num_attention_heads": 16,
|
| 72 |
+
"num_experts": 256,
|
| 73 |
+
"num_experts_per_tok": 8,
|
| 74 |
+
"num_hidden_layers": 40,
|
| 75 |
+
"num_key_value_heads": 2,
|
| 76 |
+
"rms_norm_eps": 1e-06,
|
| 77 |
+
"router_aux_loss_coef": 0.001,
|
| 78 |
+
"shared_expert_intermediate_size": 512,
|
| 79 |
+
"use_cache": true,
|
| 80 |
+
"vocab_size": 251392,
|
| 81 |
+
"mamba_ssm_dtype": "float32",
|
| 82 |
+
"rope_parameters": {
|
| 83 |
+
"mrope_interleaved": true,
|
| 84 |
+
"mrope_section": [
|
| 85 |
+
11,
|
| 86 |
+
11,
|
| 87 |
+
10
|
| 88 |
+
],
|
| 89 |
+
"rope_type": "default",
|
| 90 |
+
"rope_theta": 10000000,
|
| 91 |
+
"partial_rotary_factor": 0.25
|
| 92 |
+
},
|
| 93 |
+
"sci_mtp_num_hidden_layers": 1
|
| 94 |
+
},
|
| 95 |
+
"tie_word_embeddings": false,
|
| 96 |
+
"transformers_version": "4.57.0.dev0",
|
| 97 |
+
"video_token_id": 248057,
|
| 98 |
+
"vision_config": {
|
| 99 |
+
"deepstack_visual_indexes": [],
|
| 100 |
+
"depth": 27,
|
| 101 |
+
"hidden_act": "gelu_pytorch_tanh",
|
| 102 |
+
"hidden_size": 1152,
|
| 103 |
+
"in_channels": 3,
|
| 104 |
+
"initializer_range": 0.02,
|
| 105 |
+
"intermediate_size": 4304,
|
| 106 |
+
"model_type": "qwen3_5_moe",
|
| 107 |
+
"num_heads": 16,
|
| 108 |
+
"num_position_embeddings": 2304,
|
| 109 |
+
"out_hidden_size": 2048,
|
| 110 |
+
"patch_size": 16,
|
| 111 |
+
"spatial_merge_size": 2,
|
| 112 |
+
"temporal_patch_size": 2
|
| 113 |
+
},
|
| 114 |
+
"vision_end_token_id": 248054,
|
| 115 |
+
"vision_start_token_id": 248053
|
| 116 |
+
}
|
evaluation_settings.json
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"purpose": "Describe settings separately; not an executable evaluation config",
|
| 3 |
+
"explicit_historical_evaluation_settings": {
|
| 4 |
+
"verifier_repeats_per_proof": 8,
|
| 5 |
+
"max_new_tokens": 65536,
|
| 6 |
+
"temperature": 1.0,
|
| 7 |
+
"top_p": null,
|
| 8 |
+
"top_k": null
|
| 9 |
+
},
|
| 10 |
+
"unchanged_checkpoint_generation_defaults": {
|
| 11 |
+
"do_sample": true,
|
| 12 |
+
"temperature": 1.0,
|
| 13 |
+
"top_p": 0.95,
|
| 14 |
+
"top_k": 20,
|
| 15 |
+
"eos_token_id": [248046, 248044],
|
| 16 |
+
"pad_token_id": 248044
|
| 17 |
+
},
|
| 18 |
+
"step_index_base": 0,
|
| 19 |
+
"no_error_sentinel": -1,
|
| 20 |
+
"verifier_prompt": "prompts/proof_verifier.md",
|
| 21 |
+
"notes": [
|
| 22 |
+
"Historical API configuration did not explicitly fix top_p/top_k; runtime defaults require confirmation for exact reproduction.",
|
| 23 |
+
"Checkpoint default sampling values are not asserted to be the historical serving parameters.",
|
| 24 |
+
"The bundled full model includes MTP and vision tensors; this staging step does not convert weights or strip model components.",
|
| 25 |
+
"The intended pessimistic decision requires eight valid no-error judgments; legacy parse-failure handling belongs to the separately versioned evaluator."
|
| 26 |
+
]
|
| 27 |
+
}
|
generation_config.json
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"bos_token_id": 248044,
|
| 3 |
+
"do_sample": true,
|
| 4 |
+
"eos_token_id": [
|
| 5 |
+
248046,
|
| 6 |
+
248044
|
| 7 |
+
],
|
| 8 |
+
"pad_token_id": 248044,
|
| 9 |
+
"temperature": 1.0,
|
| 10 |
+
"top_k": 20,
|
| 11 |
+
"top_p": 0.95,
|
| 12 |
+
"transformers_version": "4.57.0.dev0"
|
| 13 |
+
}
|
merges.txt
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
model-language-0001-fused-save_rank1.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:7f74cda2afc0b6cc5f5aee61243d4ea19ec0ef8892f1ce15d3d75f44965ed093
|
| 3 |
+
size 100668928
|
model-language-0001-fused-save_rank12.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:2e65f0d33478f1d87a882acd4f6ee631ffa13a02c04507408777def4117bc791
|
| 3 |
+
size 100668976
|
model-language-0001-fused-save_rank13.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:5b7616b8ebebcdaafcac6ecf74c3803441a71df622bf688b8e89fd5665f7bac5
|
| 3 |
+
size 100668976
|
model-language-0001-fused-save_rank14.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:40277357e610e556ba2435737123bca4801ec37888eb1eaff7a95d06723f68ef
|
| 3 |
+
size 100668976
|
model-language-0001-fused-save_rank15.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:5f3238f4e8a27f33ecfcd15452a97c363a895453f7e9c966041102ab878dbbba
|
| 3 |
+
size 100668976
|
model-language-0001-fused-save_rank2.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:ffd0b63509989f3b107572451fb9d721a4a06429d71f80953534d7ecb5dc2a63
|
| 3 |
+
size 100668928
|
model-language-0001-fused-save_rank3.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:8a8ad9eb11e3686e45c7f075544f2f674c643855bcbed3be1c509976b97ef555
|
| 3 |
+
size 100668928
|
model-language-0001-fused-save_rank5.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:7884f2544f54dd507ad0303bfe50879fba1633749ef9e29721e3c2a71ad9aac0
|
| 3 |
+
size 100668928
|
model-language-0001-fused-save_rank7.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:ca9f85c24bb6d3946d18ea29b764930297f14fecebe110a449d9a5943a0817ed
|
| 3 |
+
size 100668976
|
model-language-0001-fused-save_rank9.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:ee9797ad6fd78d6c81dac053f757b914537297c79eaaaacffb614c2633c2b15a
|
| 3 |
+
size 100668976
|
model-language-0002-others-save_rank0.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:cb7fa1e4bd79b049d8cb214f5c68a8cb956b3f05c5183a11a806544031b0f235
|
| 3 |
+
size 3445604256
|
model-language-0003-others-save_rank0.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:42b10a711a10996d73560708adee75f198bfa6eefb0451d248a3ec3a5958489e
|
| 3 |
+
size 3357898400
|
model-language-0004-others-save_rank0.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:4e580d5dc2e5466279982a26df9398a4e0e1a51e83e0b9cade6c69b95ceb32c7
|
| 3 |
+
size 3370808760
|
model-language-0006-others-save_rank0.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:94ba85645b6373792cd77878502a0c4f332afe4bea3a0fd110782eaf527044ee
|
| 3 |
+
size 3370808672
|
model-language-0007-others-save_rank0.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:74c5a4a4f1a6c7c67b998efcc7a78464f5aa9311a8a2cd31811afa9b0928e5bb
|
| 3 |
+
size 3357898432
|
model-language-0008-others-save_rank0.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:9624740aad902bf6e510ce750275057e77bd1e52ea9a7d7736bb8a75ee18ac42
|
| 3 |
+
size 3370808792
|
model-language-0011-others-save_rank0.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f1c4bceffcaaccec6d435636f09fb7d161cc24e6ff8a03f5a57fda1e161e87a0
|
| 3 |
+
size 3357898432
|
model-language-0012-others-save_rank0.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:2bb36f2a6c1854c8166fd3c7d4c49d3601151572ce9d00fa5cbc77aabd5e6b0b
|
| 3 |
+
size 3370808792
|
model-language-0013-others-save_rank0.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:0eeb09bf430b2be9db8498c68f7dd00156eda974de87e6edc78477c453e7a6c3
|
| 3 |
+
size 3357898432
|
model-language-0014-others-save_rank0.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:771ec8dcaad6fa681dc64734f6bb28db4b3facd02db456325a99e6467be4d8e9
|
| 3 |
+
size 3370808792
|
model-language-0017-others-save_rank0.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:ffcb3d069d1b92dd0e98ed1150cc1e55dbb6ea73bbd5cfe752cb7eb8886461da
|
| 3 |
+
size 3357898432
|
model-language-0018-others-save_rank0.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:98d95974f2e17cdff6575d35f3367fe9618ec56cf384abcde161771769095f83
|
| 3 |
+
size 3370808792
|
model-language-0019-others-save_rank0.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:160b6114d8607073b20000fc665ca6548e9a08facc00c234ed5457056d185175
|
| 3 |
+
size 3357898432
|
model-language-0020-others-save_rank0.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:d075ed8040661d4f9586ab7de09220a6484dec1ad3776210478ea7697577a626
|
| 3 |
+
size 3370808792
|
model-language-0021-others-save_rank0.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:10a6c734ad7ab213d3d12d0404cee6df31f7364b8ce179d1d1c034ef5ec8dd26
|
| 3 |
+
size 3283107048
|
model-language-0022-others-save_rank0.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:0c94ebd2a5ea3a3496253262c5f3c5cf7235d30a0ea84b7b4a8fb0b9018d68df
|
| 3 |
+
size 1108372448
|
model-projector-0001-others-save_rank0.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:7b38702ff3948f711782b3975d189e3cb67ca1eedbc53e9be13398016afedd50
|
| 3 |
+
size 61360248
|
model.safetensors.index.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
preprocessor_config.json
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"size": {
|
| 3 |
+
"longest_edge": 16777216,
|
| 4 |
+
"shortest_edge": 65536
|
| 5 |
+
},
|
| 6 |
+
"patch_size": 16,
|
| 7 |
+
"temporal_patch_size": 2,
|
| 8 |
+
"merge_size": 2,
|
| 9 |
+
"image_mean": [
|
| 10 |
+
0.5,
|
| 11 |
+
0.5,
|
| 12 |
+
0.5
|
| 13 |
+
],
|
| 14 |
+
"image_std": [
|
| 15 |
+
0.5,
|
| 16 |
+
0.5,
|
| 17 |
+
0.5
|
| 18 |
+
],
|
| 19 |
+
"processor_class": "Qwen3VLProcessor",
|
| 20 |
+
"image_processor_type": "Qwen2VLImageProcessorFast"
|
| 21 |
+
}
|
prompts/proof_verifier.md
ADDED
|
@@ -0,0 +1,61 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
You are an **expert math proof grader**. You are judging the correctness of an LLM-generated proof for a math problem.
|
| 2 |
+
|
| 3 |
+
### Input
|
| 4 |
+
|
| 5 |
+
Your input will consist of:
|
| 6 |
+
|
| 7 |
+
* **Problem Statement**: A mathematical problem that the proof is attempting to solve.
|
| 8 |
+
* **Reference Solution (optional)**: When present, a correct solution or proof for reference. This is **not necessarily the only valid solution**. If the problem requires a final numeric or algebraic answer, this section contains the correct answer, which should be the only accepted final answer (though alternative reasoning paths are valid). If it is missing or empty, judge using only the problem statement and the proof.
|
| 9 |
+
* **Proof Solution**: The proof that you need to evaluate. This proof may contain errors, omissions, or unclear steps. The proof was generated by another language model. The proof has a clear step-wise structure: each step is wrapped as `<step idx> ... </step idx>`, where `idx` is a zero-based step index.
|
| 10 |
+
|
| 11 |
+
### Task
|
| 12 |
+
|
| 13 |
+
Analyze the proof carefully.
|
| 14 |
+
|
| 15 |
+
**Core principles (in order of precedence):**
|
| 16 |
+
1) **Mathematical validity** of the proof’s reasoning and conclusion.
|
| 17 |
+
2) **Problem constraints** (e.g., unique required final value; forbidden tools if stated).
|
| 18 |
+
3) **Reference solution** (when present) as an anchor for sufficiency, not exclusivity.
|
| 19 |
+
|
| 20 |
+
**Alternative-approach policy:**
|
| 21 |
+
- If the proof uses a different but valid method, accept it as long as the reasoning is mathematically sound and satisfies the problem constraints.
|
| 22 |
+
- **Do not penalize** solely for re-ordering steps, using different lemmas, or giving a correct shortcut, **unless** the problem forbids it.
|
| 23 |
+
|
| 24 |
+
**Rigor and evidence:**
|
| 25 |
+
- Treat a claim as correct **only if it is adequately justified** within the proof (not merely asserted).
|
| 26 |
+
- If a step is plausible but under-justified, note the gap explicitly and judge conservatively.
|
| 27 |
+
|
| 28 |
+
**What to produce:**
|
| 29 |
+
- Identify logical errors, incorrect steps, or unjustified leaps.
|
| 30 |
+
- Give a **detailed assessment** of the proof’s correctness and rigor.
|
| 31 |
+
- Determine whether the proof is **fully correct**, **partially correct**, or **incorrect**, and justify this judgment clearly.
|
| 32 |
+
|
| 33 |
+
### Output Format
|
| 34 |
+
|
| 35 |
+
Respond with **only** well-formed XML using the structure below. Do not include any extra text or Markdown.
|
| 36 |
+
|
| 37 |
+
**Requirements:**
|
| 38 |
+
- `<assessment>` must be a **detailed analysis** explaining your reasoning step-by-step. Reference specific steps (`idx`) where relevant.
|
| 39 |
+
- `<errors>` must be a list of specific issues (empty if the proof is fully correct).
|
| 40 |
+
- `<first_error_step>` must be the **index of the earliest step (`idx`) where a mathematical error or unjustified leap first occurs**.
|
| 41 |
+
- If no error exists, set `<first_error_step>` to `-1`.
|
| 42 |
+
|
| 43 |
+
Example output:
|
| 44 |
+
|
| 45 |
+
<assessment>The proof shows a good understanding of the main idea, but has some unclear reasoning and minor mistakes...</assessment>
|
| 46 |
+
<errors>
|
| 47 |
+
1. specific error 1,
|
| 48 |
+
2. specific error 2,
|
| 49 |
+
...
|
| 50 |
+
</errors>
|
| 51 |
+
<first_error_step>2</first_error_step>
|
| 52 |
+
|
| 53 |
+
--------------------------------------------------
|
| 54 |
+
**Problem Statement**
|
| 55 |
+
{problem}
|
| 56 |
+
|
| 57 |
+
**Reference Solution (optional)**
|
| 58 |
+
{human_solution}
|
| 59 |
+
|
| 60 |
+
**Proof Solution**
|
| 61 |
+
{solution}
|
special_tokens_map.json
ADDED
|
@@ -0,0 +1,45 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"additional_special_tokens": [
|
| 3 |
+
"<|im_start|>",
|
| 4 |
+
"<|im_end|>",
|
| 5 |
+
"<|object_ref_start|>",
|
| 6 |
+
"<|object_ref_end|>",
|
| 7 |
+
"<|box_start|>",
|
| 8 |
+
"<|box_end|>",
|
| 9 |
+
"<|quad_start|>",
|
| 10 |
+
"<|quad_end|>",
|
| 11 |
+
"<|vision_start|>",
|
| 12 |
+
"<|vision_end|>",
|
| 13 |
+
"<|vision_pad|>",
|
| 14 |
+
"<|image_pad|>",
|
| 15 |
+
"<|video_pad|>"
|
| 16 |
+
],
|
| 17 |
+
"audio_bos_token": "<|audio_start|>",
|
| 18 |
+
"audio_eos_token": "<|audio_end|>",
|
| 19 |
+
"audio_token": "<|audio_pad|>",
|
| 20 |
+
"bos_token": {
|
| 21 |
+
"content": "<|im_start|>",
|
| 22 |
+
"lstrip": false,
|
| 23 |
+
"normalized": false,
|
| 24 |
+
"rstrip": false,
|
| 25 |
+
"single_word": false
|
| 26 |
+
},
|
| 27 |
+
"eos_token": {
|
| 28 |
+
"content": "<|im_end|>",
|
| 29 |
+
"lstrip": false,
|
| 30 |
+
"normalized": false,
|
| 31 |
+
"rstrip": false,
|
| 32 |
+
"single_word": false
|
| 33 |
+
},
|
| 34 |
+
"image_token": "<|image_pad|>",
|
| 35 |
+
"pad_token": {
|
| 36 |
+
"content": "<|endoftext|>",
|
| 37 |
+
"lstrip": false,
|
| 38 |
+
"normalized": false,
|
| 39 |
+
"rstrip": false,
|
| 40 |
+
"single_word": false
|
| 41 |
+
},
|
| 42 |
+
"video_token": "<|video_pad|>",
|
| 43 |
+
"vision_bos_token": "<|vision_start|>",
|
| 44 |
+
"vision_eos_token": "<|vision_end|>"
|
| 45 |
+
}
|
tokenization_interns1.py
ADDED
|
@@ -0,0 +1,1009 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# coding=utf-8
|
| 2 |
+
# Copyright 2025 The Intern team and Shanghai AI Lab team. All rights reserved.
|
| 3 |
+
#
|
| 4 |
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
| 5 |
+
# you may not use this file except in compliance with the License.
|
| 6 |
+
# You may obtain a copy of the License at
|
| 7 |
+
#
|
| 8 |
+
# http://www.apache.org/licenses/LICENSE-2.0
|
| 9 |
+
#
|
| 10 |
+
# Unless required by applicable law or agreed to in writing, software
|
| 11 |
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
| 12 |
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
| 13 |
+
# See the License for the specific language governing permissions and
|
| 14 |
+
# limitations under the License.
|
| 15 |
+
"""Tokenization classes for InternS1."""
|
| 16 |
+
|
| 17 |
+
import json
|
| 18 |
+
import os
|
| 19 |
+
import unicodedata
|
| 20 |
+
from abc import ABC, abstractmethod
|
| 21 |
+
from typing import Optional, Union
|
| 22 |
+
from functools import lru_cache
|
| 23 |
+
|
| 24 |
+
import regex as re
|
| 25 |
+
import sentencepiece as spm
|
| 26 |
+
|
| 27 |
+
from transformers.tokenization_utils_base import AddedToken, TextInput
|
| 28 |
+
from transformers.utils import logging
|
| 29 |
+
from packaging import version
|
| 30 |
+
import transformers
|
| 31 |
+
if version.parse(transformers.__version__) >= version.parse("5.0.0"):
|
| 32 |
+
from transformers.tokenization_python import PreTrainedTokenizer
|
| 33 |
+
else:
|
| 34 |
+
from transformers.tokenization_utils import PreTrainedTokenizer
|
| 35 |
+
|
| 36 |
+
logger = logging.get_logger(__name__)
|
| 37 |
+
|
| 38 |
+
try:
|
| 39 |
+
from rdkit import Chem, RDLogger
|
| 40 |
+
|
| 41 |
+
RDLogger.DisableLog("rdApp.error")
|
| 42 |
+
RDLogger.DisableLog("rdApp.*")
|
| 43 |
+
RDKIT_AVAILABLE = True
|
| 44 |
+
except ImportError:
|
| 45 |
+
logger.warning_once(
|
| 46 |
+
"If tokenization with SMILES formula is of necessity, please 'pip install RDKit' for better tokenization quality."
|
| 47 |
+
)
|
| 48 |
+
RDKIT_AVAILABLE = False
|
| 49 |
+
|
| 50 |
+
VOCAB_FILES_NAMES = {
|
| 51 |
+
"vocab_file": "vocab.json",
|
| 52 |
+
"merges_file": "merges.txt",
|
| 53 |
+
"sp_model_SMILES": "tokenizer_SMILES.model",
|
| 54 |
+
"sp_model_PROT": "tokenizer_PROT.model",
|
| 55 |
+
"sp_model_XNA": "tokenizer_XNA.model",
|
| 56 |
+
}
|
| 57 |
+
|
| 58 |
+
PRETOKENIZE_REGEX = r"""(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\r\n\p{L}\p{N}]?\p{L}+|\p{N}| ?[^\s\p{L}\p{N}]+[\r\n]*|\s*[\r\n]+|\s+(?!\S)|\s+"""
|
| 59 |
+
|
| 60 |
+
|
| 61 |
+
class InternS1CheckModuleMixin(ABC):
|
| 62 |
+
"""
|
| 63 |
+
Basic auto-detection module.
|
| 64 |
+
|
| 65 |
+
Note that short strings are ignored by this module.
|
| 66 |
+
"""
|
| 67 |
+
|
| 68 |
+
def __init__(self, *, min_length: int):
|
| 69 |
+
self.min_length = min_length
|
| 70 |
+
self.REGEX = self._build_regex()
|
| 71 |
+
self.all_auto_detect_token_start = ["<SMILES_AUTO_DETECT>", "<PROT_AUTO_DETECT>", "<XNA_AUTO_DETECT>"]
|
| 72 |
+
self.all_auto_detect_token_end = ["</SMILES_AUTO_DETECT>", "</PROT_AUTO_DETECT>", "</XNA_AUTO_DETECT>"]
|
| 73 |
+
self.auto_detect_token = []
|
| 74 |
+
self.truncation = False
|
| 75 |
+
|
| 76 |
+
@abstractmethod
|
| 77 |
+
def _build_regex(self):
|
| 78 |
+
pass
|
| 79 |
+
|
| 80 |
+
@abstractmethod
|
| 81 |
+
def check_legitimacy(self, candidate: str) -> bool:
|
| 82 |
+
pass
|
| 83 |
+
|
| 84 |
+
def re_split(self, texts: Union[str, list[str]]) -> list[str]:
|
| 85 |
+
if isinstance(texts, str):
|
| 86 |
+
texts = [texts]
|
| 87 |
+
|
| 88 |
+
total_results = []
|
| 89 |
+
|
| 90 |
+
no_split_flag = 0
|
| 91 |
+
|
| 92 |
+
for text in texts:
|
| 93 |
+
if text in self.all_auto_detect_token_start:
|
| 94 |
+
total_results.append(text)
|
| 95 |
+
no_split_flag += 1
|
| 96 |
+
continue
|
| 97 |
+
elif text in self.all_auto_detect_token_end:
|
| 98 |
+
total_results.append(text)
|
| 99 |
+
no_split_flag = max(0, no_split_flag - 1)
|
| 100 |
+
continue
|
| 101 |
+
|
| 102 |
+
if no_split_flag > 0:
|
| 103 |
+
total_results.append(text)
|
| 104 |
+
continue
|
| 105 |
+
|
| 106 |
+
results = []
|
| 107 |
+
current_pos = 0
|
| 108 |
+
for match in self.REGEX.finditer(text):
|
| 109 |
+
candidate = match.group(1)
|
| 110 |
+
|
| 111 |
+
if len(candidate) >= self.min_length:
|
| 112 |
+
match_start, match_end = match.span(1)
|
| 113 |
+
|
| 114 |
+
if not self.check_legitimacy(candidate):
|
| 115 |
+
continue
|
| 116 |
+
|
| 117 |
+
if not self.truncation:
|
| 118 |
+
if match_start > 0 and text[match_start - 1].encode("UTF-8").isalpha():
|
| 119 |
+
continue
|
| 120 |
+
if match_end < len(text) and text[match_end].encode("UTF-8").isalpha():
|
| 121 |
+
continue
|
| 122 |
+
|
| 123 |
+
if match_start > current_pos:
|
| 124 |
+
non_candidate_part = text[current_pos:match_start]
|
| 125 |
+
results.append(non_candidate_part)
|
| 126 |
+
else:
|
| 127 |
+
continue
|
| 128 |
+
|
| 129 |
+
results.extend([self.auto_detect_token[0], candidate, self.auto_detect_token[1]])
|
| 130 |
+
current_pos = match_end
|
| 131 |
+
|
| 132 |
+
if current_pos < len(text):
|
| 133 |
+
remaining_part = text[current_pos:]
|
| 134 |
+
results.append(remaining_part)
|
| 135 |
+
|
| 136 |
+
total_results.extend(results)
|
| 137 |
+
|
| 138 |
+
return total_results
|
| 139 |
+
|
| 140 |
+
|
| 141 |
+
class XnaCheckModule(InternS1CheckModuleMixin):
|
| 142 |
+
"""
|
| 143 |
+
XNA sequence auto-detection module.
|
| 144 |
+
|
| 145 |
+
Automatically detects XNA sequence using regex patterns.
|
| 146 |
+
"""
|
| 147 |
+
def __init__(self, *, min_length: int = 27):
|
| 148 |
+
super().__init__(min_length=min_length)
|
| 149 |
+
self.auto_detect_token = ["<XNA_AUTO_DETECT>", "</XNA_AUTO_DETECT>"]
|
| 150 |
+
self.truncation = True
|
| 151 |
+
|
| 152 |
+
def _build_regex(self):
|
| 153 |
+
return re.compile(r"([ATCGU]{" + str(self.min_length) + r",})")
|
| 154 |
+
|
| 155 |
+
def check_legitimacy(self, candidate: str):
|
| 156 |
+
return True
|
| 157 |
+
|
| 158 |
+
|
| 159 |
+
class ProtCheckModule(InternS1CheckModuleMixin):
|
| 160 |
+
"""
|
| 161 |
+
Protein sequence auto-detection module.
|
| 162 |
+
|
| 163 |
+
Automatically detects protein sequence using regex patterns.
|
| 164 |
+
"""
|
| 165 |
+
def __init__(self, *, min_length: int = 27):
|
| 166 |
+
super().__init__(min_length=min_length)
|
| 167 |
+
self.auto_detect_token = ["<PROT_AUTO_DETECT>", "</PROT_AUTO_DETECT>"]
|
| 168 |
+
self.truncation = True
|
| 169 |
+
self._xna_pattern = re.compile(r"^[ATCGU]+$")
|
| 170 |
+
|
| 171 |
+
def _build_regex(self):
|
| 172 |
+
return re.compile(r"([A-Z]{" + str(self.min_length) + r",})")
|
| 173 |
+
|
| 174 |
+
def check_legitimacy(self, candidate: str):
|
| 175 |
+
if self._xna_pattern.match(candidate):
|
| 176 |
+
return False
|
| 177 |
+
return True
|
| 178 |
+
|
| 179 |
+
|
| 180 |
+
# fmt: off
|
| 181 |
+
bonds = ["-", "=", "#", ":", "/", "\\", ".", "$"]
|
| 182 |
+
organic_symbols = ["B", "C", "N", "O", "P", "S", "F", "Cl", "Br", "I"]
|
| 183 |
+
other_allows = bonds + ["[", "]", "(", ")", ";"]
|
| 184 |
+
aromatic_symbols = ["b", "c", "n", "o", "s", "p"]
|
| 185 |
+
elements = [
|
| 186 |
+
"H", "He", "Li", "Be", "B", "C", "N", "O", "F", "Ne",
|
| 187 |
+
"Na", "Mg", "Al", "Si", "P", "S", "Cl", "Ar", "K", "Ca",
|
| 188 |
+
"Sc", "Ti", "V", "Cr", "Mn", "Fe", "Co", "Ni", "Cu", "Zn",
|
| 189 |
+
"Ga", "Ge", "As", "Se", "Br", "Kr", "Rb", "Sr", "Y", "Zr",
|
| 190 |
+
"Nb", "Mo", "Tc", "Ru", "Rh", "Pd", "Ag", "Cd", "In", "Sn",
|
| 191 |
+
"Sb", "Te", "I", "Xe", "Cs", "Ba", "La", "Ce", "Pr", "Nd",
|
| 192 |
+
"Pm", "Sm", "Eu", "Gd", "Tb", "Dy", "Ho", "Er", "Tm", "Yb",
|
| 193 |
+
"Lu", "Hf", "Ta", "W", "Re", "Os", "Ir", "Pt", "Au", "Hg",
|
| 194 |
+
"Tl", "Pb", "Bi", "Po", "At", "Rn", "Fr", "Ra", "Ac", "Th",
|
| 195 |
+
"Pa", "U", "Np", "Pu", "Am", "Cm", "Bk", "Cf", "Es", "Fm",
|
| 196 |
+
"Md", "No", "Lr", "Rf", "Db", "Sg", "Bh", "Hs", "Mt", "Ds",
|
| 197 |
+
"Rg", "Cn", "Nh", "Fl", "Mc", "Lv", "Ts", "Og"
|
| 198 |
+
]
|
| 199 |
+
# fmt: on
|
| 200 |
+
|
| 201 |
+
|
| 202 |
+
class SmilesCheckModule(InternS1CheckModuleMixin):
|
| 203 |
+
"""
|
| 204 |
+
SMILES molecular sequence auto-detection module.
|
| 205 |
+
|
| 206 |
+
Automatically detects and validates SMILES strings in text using regex patterns
|
| 207 |
+
or chemical syntax rules. Uses RDKit for precise validation when available,
|
| 208 |
+
otherwise falls back to rule-based validation.
|
| 209 |
+
"""
|
| 210 |
+
|
| 211 |
+
def __init__(self, *, min_length: int = 10):
|
| 212 |
+
super().__init__(min_length=min_length)
|
| 213 |
+
self.auto_detect_token = ["<SMILES_AUTO_DETECT>", "</SMILES_AUTO_DETECT>"]
|
| 214 |
+
self._SQ_BRACKET_BAN_1 = re.compile(r"(?:[A-GI-Z]|[a-z]){3,}")
|
| 215 |
+
self._SQ_BRACKET_BAN_2 = re.compile(r"\d{4,}")
|
| 216 |
+
|
| 217 |
+
def _build_regex(self):
|
| 218 |
+
# fmt: off
|
| 219 |
+
_two_letter_elements = [
|
| 220 |
+
'Ac', 'Ag', 'Al', 'Am', 'Ar', 'As', 'At', 'Au', 'Ba', 'Be', 'Bh', 'Bi', 'Bk', 'Br', 'Ca', 'Cd',
|
| 221 |
+
'Ce', 'Cf', 'Cl', 'Cm', 'Cn', 'Co', 'Cr', 'Cs', 'Cu', 'Db', 'Ds', 'Dy', 'Er', 'Es', 'Eu', 'Fe',
|
| 222 |
+
'Fl', 'Fm', 'Fr', 'Ga', 'Gd', 'Ge', 'He', 'Hf', 'Hg', 'Ho', 'Hs', 'In', 'Ir', 'Kr', 'La', 'Li',
|
| 223 |
+
'Lr', 'Lu', 'Lv', 'Mc', 'Md', 'Mg', 'Mn', 'Mo', 'Mt', 'Na', 'Nb', 'Nd', 'Ne', 'Nh', 'Ni', 'No',
|
| 224 |
+
'Np', 'Og', 'Os', 'Pa', 'Pb', 'Pd', 'Pm', 'Po', 'Pr', 'Pt', 'Pu', 'Ra', 'Rb', 'Re', 'Rf', 'Rg',
|
| 225 |
+
'Rh', 'Rn', 'Ru', 'Sb', 'Sc', 'Se', 'Sg', 'Si', 'Sm', 'Sn', 'Sr', 'Ta', 'Tb', 'Tc', 'Te', 'Th',
|
| 226 |
+
'Ti', 'Tl', 'Tm', 'Ts', 'Xe', 'Yb', 'Zn', 'Zr'
|
| 227 |
+
]
|
| 228 |
+
_single_letter_elements = [
|
| 229 |
+
"B", "C", "F", "H", "I", "K", "N", "O", "P", "S", "U", "V", "W", "Y", 'b', 'c', 'n', 'o', 'p', 's'
|
| 230 |
+
]
|
| 231 |
+
# fmt: on
|
| 232 |
+
all_elements_sorted = sorted(_two_letter_elements + _single_letter_elements, key=lambda x: (-len(x), x))
|
| 233 |
+
elements_pattern_str = "|".join(all_elements_sorted)
|
| 234 |
+
|
| 235 |
+
bracket_atom_pattern_str = r"\[[^\]]+\]"
|
| 236 |
+
other_single_chars_pattern_str = r"[\(\)\.=\-#@\d\$\%\*:\+\-\/\\]"
|
| 237 |
+
smiles_unit_pattern = (
|
| 238 |
+
r"(?:"
|
| 239 |
+
+ bracket_atom_pattern_str
|
| 240 |
+
+ r"|"
|
| 241 |
+
+ elements_pattern_str
|
| 242 |
+
+ r"|"
|
| 243 |
+
+ other_single_chars_pattern_str
|
| 244 |
+
+ r")"
|
| 245 |
+
)
|
| 246 |
+
core_sequence_pattern = rf"(?>{smiles_unit_pattern}){{10,}}"
|
| 247 |
+
constrained_core_sequence_pattern = rf"(?![:.=]){core_sequence_pattern}(?<![:.=])"
|
| 248 |
+
|
| 249 |
+
final_regex_str = rf"({constrained_core_sequence_pattern})"
|
| 250 |
+
|
| 251 |
+
COMPILED_REGEX = re.compile(final_regex_str)
|
| 252 |
+
return COMPILED_REGEX
|
| 253 |
+
|
| 254 |
+
def check_legitimacy_slow(self, candidate: str) -> bool:
|
| 255 |
+
"""Check legitimacy with RDKit"""
|
| 256 |
+
if sum(1 for char in candidate if char.encode("UTF-8").isalpha()) < 5:
|
| 257 |
+
return False
|
| 258 |
+
|
| 259 |
+
mol = Chem.MolFromSmiles(candidate)
|
| 260 |
+
if mol is None:
|
| 261 |
+
return False
|
| 262 |
+
else:
|
| 263 |
+
return True
|
| 264 |
+
|
| 265 |
+
def check_legitimacy_fast(self, candidate: str) -> bool:
|
| 266 |
+
"""Check legitimacy with hard rules"""
|
| 267 |
+
if sum(1 for char in candidate if char.encode("UTF-8").isalpha()) < 5:
|
| 268 |
+
return False
|
| 269 |
+
|
| 270 |
+
if not self.check_rings_and_brackets(candidate):
|
| 271 |
+
return False
|
| 272 |
+
else:
|
| 273 |
+
return True
|
| 274 |
+
|
| 275 |
+
def check_legitimacy(self, candidate: str) -> bool:
|
| 276 |
+
if RDKIT_AVAILABLE:
|
| 277 |
+
return self.check_legitimacy_slow(candidate)
|
| 278 |
+
else:
|
| 279 |
+
return self.check_legitimacy_fast(candidate)
|
| 280 |
+
|
| 281 |
+
def check_brackets(self, text):
|
| 282 |
+
matches = re.findall(r"\[([^\[\]]*)\]", text)
|
| 283 |
+
for part in matches:
|
| 284 |
+
if "(" in part or ")" in part:
|
| 285 |
+
return False
|
| 286 |
+
if len(part) == 0:
|
| 287 |
+
return False
|
| 288 |
+
if part[0] in elements or part[0] in aromatic_symbols or part[:2] in elements:
|
| 289 |
+
return True
|
| 290 |
+
return True
|
| 291 |
+
|
| 292 |
+
def check_rings_and_brackets(self, text):
|
| 293 |
+
rings = {}
|
| 294 |
+
left_sq_bracket, right_sq_bracket = 0, 0
|
| 295 |
+
left_pt_bracket, right_pt_bracket = 0, 0
|
| 296 |
+
all_lower = True
|
| 297 |
+
digits_cnt = 0
|
| 298 |
+
pos = 0
|
| 299 |
+
while pos < len(text):
|
| 300 |
+
step = 0
|
| 301 |
+
c = text[pos]
|
| 302 |
+
if ord(c) >= 65 and ord(c) <= 90:
|
| 303 |
+
all_lower = False
|
| 304 |
+
if (pos == len(text) - 1 or pos == 0) and c in bonds:
|
| 305 |
+
return False
|
| 306 |
+
if pos > 0 and text[pos - 1] in bonds and text[pos] in bonds:
|
| 307 |
+
return False
|
| 308 |
+
if c == "[":
|
| 309 |
+
step = 1
|
| 310 |
+
left_sq_bracket += 1
|
| 311 |
+
if left_sq_bracket > right_sq_bracket + 1:
|
| 312 |
+
return False
|
| 313 |
+
if pos == len(text) - 1:
|
| 314 |
+
return False
|
| 315 |
+
if "]" not in text[pos + 1 :]:
|
| 316 |
+
return False
|
| 317 |
+
bracket_span = text[pos + 1 : text.find("]")]
|
| 318 |
+
|
| 319 |
+
if self._SQ_BRACKET_BAN_1.search(bracket_span) or self._SQ_BRACKET_BAN_2.search(bracket_span):
|
| 320 |
+
return False
|
| 321 |
+
|
| 322 |
+
matches = re.findall(r"\d+", bracket_span)
|
| 323 |
+
if len(matches) > 2:
|
| 324 |
+
return False
|
| 325 |
+
if c == "]":
|
| 326 |
+
step = 1
|
| 327 |
+
right_sq_bracket += 1
|
| 328 |
+
if right_sq_bracket > left_sq_bracket:
|
| 329 |
+
return False
|
| 330 |
+
|
| 331 |
+
if c == "(":
|
| 332 |
+
step = 1
|
| 333 |
+
left_pt_bracket += 1
|
| 334 |
+
if c == ")":
|
| 335 |
+
step = 1
|
| 336 |
+
right_pt_bracket += 1
|
| 337 |
+
if right_pt_bracket > left_pt_bracket:
|
| 338 |
+
return False
|
| 339 |
+
|
| 340 |
+
if left_sq_bracket == right_sq_bracket:
|
| 341 |
+
if c.isdigit():
|
| 342 |
+
digits_cnt += 1
|
| 343 |
+
step = 1
|
| 344 |
+
if (
|
| 345 |
+
pos == 0
|
| 346 |
+
or (pos == 1 and text[pos - 1] != "%")
|
| 347 |
+
or (pos > 1 and text[pos - 1] != "%" and text[pos - 2] != "%")
|
| 348 |
+
):
|
| 349 |
+
if c in rings:
|
| 350 |
+
if rings[c] == "unclosed":
|
| 351 |
+
rings[c] = "closed"
|
| 352 |
+
else:
|
| 353 |
+
rings[c] = "unclosed"
|
| 354 |
+
else:
|
| 355 |
+
rings[c] = "unclosed"
|
| 356 |
+
if c == "%":
|
| 357 |
+
if pos >= len(text) - 2 or not text[pos + 1].isdigit() or not text[pos + 2].isdigit():
|
| 358 |
+
return False
|
| 359 |
+
step = 3
|
| 360 |
+
digits_cnt += 1
|
| 361 |
+
num = text[pos + 1 : pos + 3]
|
| 362 |
+
if num in rings:
|
| 363 |
+
if rings[num] == "unclosed":
|
| 364 |
+
rings[num] = "closed"
|
| 365 |
+
else:
|
| 366 |
+
rings[num] = "unclosed"
|
| 367 |
+
else:
|
| 368 |
+
rings[num] = "unclosed"
|
| 369 |
+
if step == 0:
|
| 370 |
+
if (
|
| 371 |
+
pos < len(text) - 1
|
| 372 |
+
and text[pos : pos + 2] in organic_symbols + aromatic_symbols + other_allows
|
| 373 |
+
):
|
| 374 |
+
step = 2
|
| 375 |
+
elif c in organic_symbols + aromatic_symbols + other_allows:
|
| 376 |
+
step = 1
|
| 377 |
+
else:
|
| 378 |
+
return False
|
| 379 |
+
|
| 380 |
+
if step == 0:
|
| 381 |
+
step = 1
|
| 382 |
+
pos += step
|
| 383 |
+
|
| 384 |
+
if left_sq_bracket != right_sq_bracket or any(v == "unclosed" for v in rings.values()):
|
| 385 |
+
return False
|
| 386 |
+
if all_lower and digits_cnt < 2:
|
| 387 |
+
return False
|
| 388 |
+
return self.check_brackets(text)
|
| 389 |
+
|
| 390 |
+
|
| 391 |
+
@lru_cache
|
| 392 |
+
# Copied from transformers.models.gpt2.tokenization_gpt2.bytes_to_unicode
|
| 393 |
+
def bytes_to_unicode():
|
| 394 |
+
"""
|
| 395 |
+
Returns list of utf-8 byte and a mapping to unicode strings. We specifically avoids mapping to whitespace/control
|
| 396 |
+
characters the bpe code barfs on.
|
| 397 |
+
|
| 398 |
+
The reversible bpe codes work on unicode strings. This means you need a large # of unicode characters in your vocab
|
| 399 |
+
if you want to avoid UNKs. When you're at something like a 10B token dataset you end up needing around 5K for
|
| 400 |
+
decent coverage. This is a significant percentage of your normal, say, 32K bpe vocab. To avoid that, we want lookup
|
| 401 |
+
tables between utf-8 bytes and unicode strings.
|
| 402 |
+
"""
|
| 403 |
+
bs = (
|
| 404 |
+
list(range(ord("!"), ord("~") + 1)) + list(range(ord("¡"), ord("¬") + 1)) + list(range(ord("®"), ord("ÿ") + 1))
|
| 405 |
+
)
|
| 406 |
+
cs = bs[:]
|
| 407 |
+
n = 0
|
| 408 |
+
for b in range(2**8):
|
| 409 |
+
if b not in bs:
|
| 410 |
+
bs.append(b)
|
| 411 |
+
cs.append(2**8 + n)
|
| 412 |
+
n += 1
|
| 413 |
+
cs = [chr(n) for n in cs]
|
| 414 |
+
return dict(zip(bs, cs))
|
| 415 |
+
|
| 416 |
+
|
| 417 |
+
# Copied from transformers.models.gpt2.tokenization_gpt2.get_pairs
|
| 418 |
+
def get_pairs(word):
|
| 419 |
+
"""
|
| 420 |
+
Return set of symbol pairs in a word.
|
| 421 |
+
|
| 422 |
+
Word is represented as tuple of symbols (symbols being variable-length strings).
|
| 423 |
+
"""
|
| 424 |
+
pairs = set()
|
| 425 |
+
prev_char = word[0]
|
| 426 |
+
for char in word[1:]:
|
| 427 |
+
pairs.add((prev_char, char))
|
| 428 |
+
prev_char = char
|
| 429 |
+
return pairs
|
| 430 |
+
|
| 431 |
+
|
| 432 |
+
# @requires(backends=("sentencepiece",))
|
| 433 |
+
class InternS1Tokenizer(PreTrainedTokenizer):
|
| 434 |
+
"""
|
| 435 |
+
Construct an InternS1 tokenizer. Based on byte-level Byte-Pair-Encoding.
|
| 436 |
+
|
| 437 |
+
Same with GPT2Tokenizer, this tokenizer has been trained to treat spaces like parts of the tokens so a word will
|
| 438 |
+
be encoded differently whether it is at the beginning of the sentence (without space) or not:
|
| 439 |
+
|
| 440 |
+
```python
|
| 441 |
+
>>> from transformers import AutoTokenizer
|
| 442 |
+
|
| 443 |
+
>>> tokenizer = AutoTokenizer.from_pretrained("InternS1Tokenizer", trust_remote_code=True)
|
| 444 |
+
>>> tokenizer("Hello world")["input_ids"]
|
| 445 |
+
[9707, 1879]
|
| 446 |
+
|
| 447 |
+
>>> tokenizer(" Hello world")["input_ids"]
|
| 448 |
+
[21927, 1879]
|
| 449 |
+
```
|
| 450 |
+
This is expected.
|
| 451 |
+
|
| 452 |
+
Include custom extension to support better domain-specific text tokenization, leveraging a separately trained tokenizer model.
|
| 453 |
+
|
| 454 |
+
```python
|
| 455 |
+
>>> from transformers import AutoTokenizer
|
| 456 |
+
|
| 457 |
+
>>> tokenizer = AutoTokenizer.from_pretrained("InternS1Tokenizer", trust_remote_code=True)
|
| 458 |
+
>>> tokenizer.tokenize("Describe <SMILES>C1=CC=C(C=C1)C=O</SMILES> and CC1=CC=CC=C1C=O")
|
| 459 |
+
["Describe ", "<SMILES>", "C1=CC=C(C=C1)C=O", "</SMILES>", " and ", "<SMILES_AUTO_DETECT>",
|
| 460 |
+
"CC1=CC=CC=C1C=O", "</SMILES_AUTO_DETECT>"]
|
| 461 |
+
>>> token_ids = tokenizer("Describe <SMILES>C1=CC=C(C=C1)C=O</SMILES> and CC1=CC=CC=C1C=O")["input_ids"]
|
| 462 |
+
>>> token_ids
|
| 463 |
+
[74785, 220, 151925, 151854, 151860, 151698, 151707, 151860, 151690, 151726, 151926, 323, 220, 151672, 151860, 151701, 151860, 151854, 151726]
|
| 464 |
+
|
| 465 |
+
>>> tokenizer.convert_ids_to_tokens(token_ids)
|
| 466 |
+
['Describe', 'Ġ', '<SMILES>', 'C', '1', '=CC=C(', 'C=C', '1', ')C', '=O', '</SMILES>', 'Ġand', 'Ġ', 'CC', '1', '=CC=CC=C', '1', 'C', '=O']
|
| 467 |
+
```
|
| 468 |
+
|
| 469 |
+
Users should refer to this superclass [`PreTrainedTokenizer`] for more information regarding those overloaded methods
|
| 470 |
+
|
| 471 |
+
Args:
|
| 472 |
+
vocab_file (`str`):
|
| 473 |
+
Path to the vocabulary file.
|
| 474 |
+
merges_file (`str`):
|
| 475 |
+
Path to the merges file.
|
| 476 |
+
errors (`str`, *optional*, defaults to `"replace"`):
|
| 477 |
+
Paradigm to follow when decoding bytes to UTF-8. See
|
| 478 |
+
[bytes.decode](https://docs.python.org/3/library/stdtypes.html#bytes.decode) for more information.
|
| 479 |
+
unk_token (`str`, *optional*, defaults to `"<|endoftext|>"`):
|
| 480 |
+
The unknown token. A token that is not in the vocabulary cannot be converted to an ID and is set to be this
|
| 481 |
+
token instead.
|
| 482 |
+
bos_token (`str`, *optional*):
|
| 483 |
+
The beginning of sequence token. Not applicable for this tokenizer.
|
| 484 |
+
eos_token (`str`, *optional*, defaults to `"<|endoftext|>"`):
|
| 485 |
+
The end of sequence token.
|
| 486 |
+
pad_token (`str`, *optional*, defaults to `"<|endoftext|>"`):
|
| 487 |
+
The token used for padding, for example when batching sequences of different lengths.
|
| 488 |
+
clean_up_tokenization_spaces (`bool`, *optional*, defaults to `False`):
|
| 489 |
+
Whether or not the model should cleanup the spaces that were added when splitting the input text during the
|
| 490 |
+
tokenization process. Not applicable to this tokenizer, since tokenization does not add spaces.
|
| 491 |
+
split_special_tokens (`bool`, *optional*, defaults to `False`):
|
| 492 |
+
Whether or not the special tokens should be split during the tokenization process. The default behavior is
|
| 493 |
+
to not split special tokens. This means that if `<|endoftext|>` is the `eos_token`, then `tokenizer.tokenize("<|endoftext|>") =
|
| 494 |
+
['<|endoftext|>`]. Otherwise, if `split_special_tokens=True`, then `tokenizer.tokenize("<|endoftext|>")` will be give `['<',
|
| 495 |
+
'|', 'endo', 'ft', 'ext', '|', '>']`. This argument is only supported for `slow` tokenizers for the moment.
|
| 496 |
+
"""
|
| 497 |
+
|
| 498 |
+
vocab_files_names = VOCAB_FILES_NAMES
|
| 499 |
+
model_input_names = ["input_ids", "attention_mask"]
|
| 500 |
+
|
| 501 |
+
def __init__(
|
| 502 |
+
self,
|
| 503 |
+
vocab_file,
|
| 504 |
+
merges_file,
|
| 505 |
+
errors="replace",
|
| 506 |
+
unk_token="<|endoftext|>",
|
| 507 |
+
bos_token=None,
|
| 508 |
+
eos_token="<|endoftext|>",
|
| 509 |
+
pad_token="<|endoftext|>",
|
| 510 |
+
clean_up_tokenization_spaces=False,
|
| 511 |
+
split_special_tokens=False,
|
| 512 |
+
special_tokens_pattern="none",
|
| 513 |
+
**kwargs,
|
| 514 |
+
):
|
| 515 |
+
bos_token = (
|
| 516 |
+
AddedToken(bos_token, lstrip=False, rstrip=False, special=True, normalized=False)
|
| 517 |
+
if isinstance(bos_token, str)
|
| 518 |
+
else bos_token
|
| 519 |
+
)
|
| 520 |
+
eos_token = (
|
| 521 |
+
AddedToken(eos_token, lstrip=False, rstrip=False, special=True, normalized=False)
|
| 522 |
+
if isinstance(eos_token, str)
|
| 523 |
+
else eos_token
|
| 524 |
+
)
|
| 525 |
+
unk_token = (
|
| 526 |
+
AddedToken(unk_token, lstrip=False, rstrip=False, special=True, normalized=False)
|
| 527 |
+
if isinstance(unk_token, str)
|
| 528 |
+
else unk_token
|
| 529 |
+
)
|
| 530 |
+
pad_token = (
|
| 531 |
+
AddedToken(pad_token, lstrip=False, rstrip=False, special=True, normalized=False)
|
| 532 |
+
if isinstance(pad_token, str)
|
| 533 |
+
else pad_token
|
| 534 |
+
)
|
| 535 |
+
|
| 536 |
+
with open(vocab_file, encoding="utf-8") as vocab_handle:
|
| 537 |
+
self.encoder = json.load(vocab_handle)
|
| 538 |
+
self.decoder = {v: k for k, v in self.encoder.items()}
|
| 539 |
+
self.errors = errors # how to handle errors in decoding
|
| 540 |
+
self.byte_encoder = bytes_to_unicode()
|
| 541 |
+
self.byte_decoder = {v: k for k, v in self.byte_encoder.items()}
|
| 542 |
+
bpe_merges = []
|
| 543 |
+
with open(merges_file, encoding="utf-8") as merges_handle:
|
| 544 |
+
for i, line in enumerate(merges_handle):
|
| 545 |
+
line = line.strip()
|
| 546 |
+
if (i == 0 and line.startswith("#version:")) or not line:
|
| 547 |
+
continue
|
| 548 |
+
bpe_merges.append(tuple(line.split()))
|
| 549 |
+
self.bpe_ranks = dict(zip(bpe_merges, range(len(bpe_merges))))
|
| 550 |
+
# NOTE: the cache can grow without bound and will get really large for long running processes
|
| 551 |
+
# (esp. for texts of language that do not use space between word, e.g. Chinese); technically
|
| 552 |
+
# not a memory leak but appears as one.
|
| 553 |
+
# GPT2Tokenizer has the same problem, so let's be consistent.
|
| 554 |
+
self.cache = {}
|
| 555 |
+
|
| 556 |
+
self.pat = re.compile(PRETOKENIZE_REGEX)
|
| 557 |
+
|
| 558 |
+
if kwargs.get("add_prefix_space", False):
|
| 559 |
+
logger.warning_once(
|
| 560 |
+
f"{self.__class__.__name} does not support `add_prefix_space`, setting it to True has no effect."
|
| 561 |
+
)
|
| 562 |
+
|
| 563 |
+
super().__init__(
|
| 564 |
+
vocab_file=vocab_file,
|
| 565 |
+
merges_file=merges_file,
|
| 566 |
+
errors=errors,
|
| 567 |
+
unk_token=unk_token,
|
| 568 |
+
bos_token=bos_token,
|
| 569 |
+
eos_token=eos_token,
|
| 570 |
+
pad_token=pad_token,
|
| 571 |
+
clean_up_tokenization_spaces=clean_up_tokenization_spaces,
|
| 572 |
+
split_special_tokens=split_special_tokens,
|
| 573 |
+
special_tokens_pattern=special_tokens_pattern,
|
| 574 |
+
**kwargs,
|
| 575 |
+
)
|
| 576 |
+
|
| 577 |
+
self.prepare_extra_tokenizers(vocab_file)
|
| 578 |
+
|
| 579 |
+
@property
|
| 580 |
+
def vocab_size(self) -> int:
|
| 581 |
+
return len(self.encoder)
|
| 582 |
+
|
| 583 |
+
# Copied from transformers.models.gpt2.tokenization_gpt2.GPT2Tokenizer.get_vocab
|
| 584 |
+
def get_vocab(self):
|
| 585 |
+
return dict(self.encoder, **self.added_tokens_encoder)
|
| 586 |
+
|
| 587 |
+
# Copied from transformers.models.gpt2.tokenization_gpt2.GPT2Tokenizer.bpe
|
| 588 |
+
def bpe(self, token):
|
| 589 |
+
if token in self.cache:
|
| 590 |
+
return self.cache[token]
|
| 591 |
+
word = tuple(token)
|
| 592 |
+
pairs = get_pairs(word)
|
| 593 |
+
|
| 594 |
+
if not pairs:
|
| 595 |
+
return token
|
| 596 |
+
|
| 597 |
+
while True:
|
| 598 |
+
bigram = min(pairs, key=lambda pair: self.bpe_ranks.get(pair, float("inf")))
|
| 599 |
+
if bigram not in self.bpe_ranks:
|
| 600 |
+
break
|
| 601 |
+
first, second = bigram
|
| 602 |
+
new_word = []
|
| 603 |
+
i = 0
|
| 604 |
+
while i < len(word):
|
| 605 |
+
try:
|
| 606 |
+
j = word.index(first, i)
|
| 607 |
+
except ValueError:
|
| 608 |
+
new_word.extend(word[i:])
|
| 609 |
+
break
|
| 610 |
+
else:
|
| 611 |
+
new_word.extend(word[i:j])
|
| 612 |
+
i = j
|
| 613 |
+
|
| 614 |
+
if word[i] == first and i < len(word) - 1 and word[i + 1] == second:
|
| 615 |
+
new_word.append(first + second)
|
| 616 |
+
i += 2
|
| 617 |
+
else:
|
| 618 |
+
new_word.append(word[i])
|
| 619 |
+
i += 1
|
| 620 |
+
new_word = tuple(new_word)
|
| 621 |
+
word = new_word
|
| 622 |
+
if len(word) == 1:
|
| 623 |
+
break
|
| 624 |
+
else:
|
| 625 |
+
pairs = get_pairs(word)
|
| 626 |
+
word = " ".join(word)
|
| 627 |
+
self.cache[token] = word
|
| 628 |
+
return word
|
| 629 |
+
|
| 630 |
+
def prepare_extra_tokenizers(self, vocab_file: str) -> None:
|
| 631 |
+
"""
|
| 632 |
+
Prepare domain-specific tokenizers.
|
| 633 |
+
|
| 634 |
+
Define variables/maps here which guide domain-specific tokenization later.
|
| 635 |
+
"""
|
| 636 |
+
# Load extra tokenizers with SentencePiece model
|
| 637 |
+
dir_name = os.path.dirname(vocab_file)
|
| 638 |
+
|
| 639 |
+
self.sp_model_SMILES = spm.SentencePieceProcessor()
|
| 640 |
+
self.sp_model_SMILES.Load(os.path.join(dir_name, "tokenizer_SMILES.model"))
|
| 641 |
+
self.sp_model_SMILES.offset = self.init_kwargs["offset_SMILES"]
|
| 642 |
+
|
| 643 |
+
self.sp_model_PROT = spm.SentencePieceProcessor()
|
| 644 |
+
self.sp_model_PROT.Load(os.path.join(dir_name, "tokenizer_PROT.model"))
|
| 645 |
+
self.sp_model_PROT.offset = self.init_kwargs["offset_PROT"]
|
| 646 |
+
|
| 647 |
+
self.sp_model_XNA = spm.SentencePieceProcessor()
|
| 648 |
+
self.sp_model_XNA.Load(os.path.join(dir_name, "tokenizer_XNA.model"))
|
| 649 |
+
self.sp_model_XNA.offset = self.init_kwargs["offset_XNA"]
|
| 650 |
+
|
| 651 |
+
base_mapping = {
|
| 652 |
+
"SMILES": self.sp_model_SMILES,
|
| 653 |
+
"protein": self.sp_model_PROT,
|
| 654 |
+
"dna": self.sp_model_XNA,
|
| 655 |
+
"rna": self.sp_model_XNA,
|
| 656 |
+
}
|
| 657 |
+
auto_detect_mapping = {
|
| 658 |
+
"SMILES": self.sp_model_SMILES,
|
| 659 |
+
"PROT": self.sp_model_PROT,
|
| 660 |
+
"XNA": self.sp_model_XNA,
|
| 661 |
+
}
|
| 662 |
+
# Guiding tokens of domain-specific tokenization
|
| 663 |
+
self.ex_begin_mapping = {f"<{key}>": value for key, value in base_mapping.items()}
|
| 664 |
+
self.ex_end_mapping = {f"</{key}>": value for key, value in base_mapping.items()}
|
| 665 |
+
# Transient markers for auto-detection, these tokens will not be assigned token ids
|
| 666 |
+
self.ex_auto_begin_mapping = {f"<{key}_AUTO_DETECT>": value for key, value in auto_detect_mapping.items()}
|
| 667 |
+
self.ex_auto_end_mapping = {f"</{key}_AUTO_DETECT>": value for key, value in auto_detect_mapping.items()}
|
| 668 |
+
# Token markers to prevent unwanted auto-detection
|
| 669 |
+
self.ex_protect_begin_tokens = ["<MOLFORMULA>"]
|
| 670 |
+
self.ex_protect_end_tokens = ["</MOLFORMULA>"]
|
| 671 |
+
# For simplicity
|
| 672 |
+
self.ex_protect_tokens = self.ex_protect_begin_tokens + self.ex_protect_end_tokens
|
| 673 |
+
self.ex_all_begin_mapping = self.ex_begin_mapping | self.ex_auto_begin_mapping
|
| 674 |
+
self.ex_all_end_mapping = self.ex_end_mapping | self.ex_auto_end_mapping
|
| 675 |
+
|
| 676 |
+
# Update encoder & decoder with extra tokenizers
|
| 677 |
+
for tokenizer_name, sp_model in [
|
| 678 |
+
("SMILES", self.sp_model_SMILES),
|
| 679 |
+
("PROT", self.sp_model_PROT),
|
| 680 |
+
("XNA", self.sp_model_XNA),
|
| 681 |
+
]:
|
| 682 |
+
self.decoder.update(
|
| 683 |
+
{i + sp_model.offset: sp_model.id_to_piece(i) for i in range(sp_model.get_piece_size())}
|
| 684 |
+
)
|
| 685 |
+
# Not really used, only to fill holes in encoder, to keep methods like `add_tokens` working
|
| 686 |
+
self.encoder.update(
|
| 687 |
+
{
|
| 688 |
+
f"<|{tokenizer_name}_{sp_model.id_to_piece(i)}|>": i + sp_model.offset
|
| 689 |
+
for i in range(sp_model.get_piece_size())
|
| 690 |
+
}
|
| 691 |
+
)
|
| 692 |
+
|
| 693 |
+
# protect-tokens should keep complete temporarily to guide later tokenization
|
| 694 |
+
# it will be segmented later
|
| 695 |
+
for token in self.ex_protect_tokens:
|
| 696 |
+
self.tokens_trie.add(token)
|
| 697 |
+
|
| 698 |
+
self._unk_token = "<unk>" # Fall-back
|
| 699 |
+
self.check_module_list = [SmilesCheckModule(), ProtCheckModule(), XnaCheckModule()]
|
| 700 |
+
|
| 701 |
+
def _pop_logical_sp_token(self, extra_tokenizer_stack: list, mapping_name: str) -> None:
|
| 702 |
+
"""Switch tokenizer when it comes to an end sp token"""
|
| 703 |
+
extra_tokenizer = extra_tokenizer_stack.pop()
|
| 704 |
+
if extra_tokenizer != self.ex_all_end_mapping[mapping_name]:
|
| 705 |
+
logger.warning_once(
|
| 706 |
+
f"Encounter incorrect nesting of extra tokenizer: {self.ex_all_end_mapping[mapping_name]} and {extra_tokenizer}"
|
| 707 |
+
)
|
| 708 |
+
logger.warning_once("This may lead to unexpected behaviour of the tokenizer, please check your input.")
|
| 709 |
+
|
| 710 |
+
def tokenize(self, text: TextInput, **kwargs) -> list[str]:
|
| 711 |
+
"""
|
| 712 |
+
Converts a string into a sequence of tokens, using the tokenizer.
|
| 713 |
+
|
| 714 |
+
It will switch to domain-specific tokenizer once encountering extra/logical sp tokens.
|
| 715 |
+
|
| 716 |
+
Args:
|
| 717 |
+
text: TextInput
|
| 718 |
+
"""
|
| 719 |
+
split_special_tokens = kwargs.pop("split_special_tokens", self.split_special_tokens)
|
| 720 |
+
|
| 721 |
+
text, kwargs = self.prepare_for_tokenization(text, **kwargs)
|
| 722 |
+
|
| 723 |
+
if hasattr(self, "do_lower_case") and self.do_lower_case:
|
| 724 |
+
# convert non-special tokens to lowercase. Might be super slow as well?
|
| 725 |
+
escaped_special_toks = [re.escape(s_tok) for s_tok in (self.all_special_tokens)]
|
| 726 |
+
escaped_special_toks += [
|
| 727 |
+
re.escape(s_tok.content)
|
| 728 |
+
for s_tok in (self._added_tokens_decoder.values())
|
| 729 |
+
if not s_tok.special and s_tok.normalized
|
| 730 |
+
]
|
| 731 |
+
pattern = r"(" + r"|".join(escaped_special_toks) + r")|" + r"(.+?)"
|
| 732 |
+
text = re.sub(pattern, lambda m: m.groups()[0] or m.groups()[1].lower(), text)
|
| 733 |
+
|
| 734 |
+
if split_special_tokens:
|
| 735 |
+
no_split_token = []
|
| 736 |
+
tokens = [text]
|
| 737 |
+
else:
|
| 738 |
+
no_split_token = self._added_tokens_encoder.keys() # don't split on any of the added tokens
|
| 739 |
+
# "This is something<special_token_1> else"
|
| 740 |
+
tokens = self.tokens_trie.split(text)
|
| 741 |
+
|
| 742 |
+
# ["This is something", "<special_token_1>", " else"]
|
| 743 |
+
for i, token in enumerate(tokens):
|
| 744 |
+
if token in no_split_token:
|
| 745 |
+
tok_extended = self._added_tokens_decoder.get(self._added_tokens_encoder[token], None)
|
| 746 |
+
left = tokens[i - 1] if i > 0 else None
|
| 747 |
+
right = tokens[i + 1] if i < len(tokens) - 1 else None
|
| 748 |
+
if isinstance(tok_extended, AddedToken):
|
| 749 |
+
if tok_extended.rstrip and right:
|
| 750 |
+
# A bit counter-intuitive but we strip the left of the string
|
| 751 |
+
# since tok_extended.rstrip means the special token is eating all white spaces on its right
|
| 752 |
+
tokens[i + 1] = right.lstrip()
|
| 753 |
+
# Strip white spaces on the left
|
| 754 |
+
if tok_extended.lstrip and left:
|
| 755 |
+
tokens[i - 1] = left.rstrip() # Opposite here
|
| 756 |
+
if tok_extended.single_word and left and left[-1] != " ":
|
| 757 |
+
tokens[i - 1] += token
|
| 758 |
+
tokens[i] = ""
|
| 759 |
+
elif tok_extended.single_word and right and right[0] != " ":
|
| 760 |
+
tokens[i + 1] = token + tokens[i + 1]
|
| 761 |
+
tokens[i] = ""
|
| 762 |
+
else:
|
| 763 |
+
raise ValueError(
|
| 764 |
+
f"{tok_extended} cannot be tokenized because it was not properly added"
|
| 765 |
+
f" to the tokenizer. This means that it is not an `AddedToken` but a {type(tok_extended)}"
|
| 766 |
+
)
|
| 767 |
+
|
| 768 |
+
# ["This is something", "<special_token_1>", "else"]
|
| 769 |
+
tokenized_text = []
|
| 770 |
+
|
| 771 |
+
# Codes for automatically detecting domain-specific content
|
| 772 |
+
# All parts that have been marked by domain-specific or protection tokens will not be subject to auto detection
|
| 773 |
+
# See transformers/tests/models/intern_s1/test_tokenization_intern_s1.py::test_auto_detection() for more details
|
| 774 |
+
new_tokens = []
|
| 775 |
+
not_split_flag = 0
|
| 776 |
+
for token in tokens:
|
| 777 |
+
if not token:
|
| 778 |
+
continue
|
| 779 |
+
if token in no_split_token or token in self.ex_protect_tokens:
|
| 780 |
+
new_tokens.append(token)
|
| 781 |
+
if token in self.ex_begin_mapping or token in self.ex_protect_begin_tokens:
|
| 782 |
+
not_split_flag += 1 # In case nested sp tokens
|
| 783 |
+
elif token in self.ex_end_mapping or token in self.ex_protect_end_tokens:
|
| 784 |
+
not_split_flag = max(0, not_split_flag - 1)
|
| 785 |
+
else:
|
| 786 |
+
if not_split_flag:
|
| 787 |
+
new_tokens.append(token)
|
| 788 |
+
else:
|
| 789 |
+
for check_module in self.check_module_list:
|
| 790 |
+
token = check_module.re_split(token)
|
| 791 |
+
|
| 792 |
+
new_tokens.extend(token)
|
| 793 |
+
tokens = new_tokens
|
| 794 |
+
|
| 795 |
+
# Use stack to maintain which tokenizer should be used, considering the possibility of nested extra tokenizer
|
| 796 |
+
extra_tokenizer_stack = []
|
| 797 |
+
for token in tokens:
|
| 798 |
+
# Need to skip eventual empty (fully stripped) tokens
|
| 799 |
+
if not token:
|
| 800 |
+
continue
|
| 801 |
+
# protect-tokens are not assigned token ids, should be segmented here
|
| 802 |
+
if token in self.ex_protect_tokens:
|
| 803 |
+
tokenized_text.extend(self._tokenize(token))
|
| 804 |
+
# push tokenizer to stack when encountering begin token
|
| 805 |
+
elif token in self.ex_all_begin_mapping:
|
| 806 |
+
tokenized_text.append(token)
|
| 807 |
+
extra_tokenizer_stack.append(self.ex_all_begin_mapping[token])
|
| 808 |
+
# pop tokenizer from stack when encountering end token
|
| 809 |
+
elif token in self.ex_all_end_mapping:
|
| 810 |
+
tokenized_text.append(token)
|
| 811 |
+
if extra_tokenizer_stack:
|
| 812 |
+
self._pop_logical_sp_token(extra_tokenizer_stack, token)
|
| 813 |
+
# other special tokens
|
| 814 |
+
elif token in no_split_token:
|
| 815 |
+
tokenized_text.append(token)
|
| 816 |
+
else:
|
| 817 |
+
tokenized_text.extend(self._tokenize(token, extra_tokenizer_stack=extra_tokenizer_stack))
|
| 818 |
+
|
| 819 |
+
# ["This", " is", " something", "<special_token_1>", "else"]
|
| 820 |
+
return tokenized_text
|
| 821 |
+
|
| 822 |
+
def _tokenize(self, text, **kwargs):
|
| 823 |
+
"""
|
| 824 |
+
Modified from `transformers.models.gpt2.tokenization_gpt2.GPT2Tokenizer._tokenize`.
|
| 825 |
+
|
| 826 |
+
This adaptation supports domain-specific tokenizers.
|
| 827 |
+
"""
|
| 828 |
+
extra_tokenizer_stack = kwargs.pop("extra_tokenizer_stack", False)
|
| 829 |
+
if extra_tokenizer_stack:
|
| 830 |
+
tokenized_text = extra_tokenizer_stack[-1].encode(text, out_type=str)
|
| 831 |
+
tokenized_id = extra_tokenizer_stack[-1].encode(text, out_type=int)
|
| 832 |
+
final_tokenized_text = []
|
| 833 |
+
for text_piece, id_piece in zip(tokenized_text, tokenized_id):
|
| 834 |
+
if id_piece == 0:
|
| 835 |
+
final_tokenized_text.extend(self._bpe_tokenize(text_piece))
|
| 836 |
+
else:
|
| 837 |
+
final_tokenized_text.append(text_piece)
|
| 838 |
+
return final_tokenized_text
|
| 839 |
+
else:
|
| 840 |
+
return self._bpe_tokenize(text)
|
| 841 |
+
|
| 842 |
+
def _bpe_tokenize(self, text, **kwargs):
|
| 843 |
+
text = text.replace(
|
| 844 |
+
"▁", " "
|
| 845 |
+
) # This discrepancy stems from differing whitespace treatment in SentencePiece versus BPE tokenization.
|
| 846 |
+
bpe_tokens = []
|
| 847 |
+
for token in re.findall(self.pat, text):
|
| 848 |
+
token = "".join(
|
| 849 |
+
self.byte_encoder[b] for b in token.encode("utf-8")
|
| 850 |
+
) # Maps all our bytes to unicode strings, avoiding control tokens of the BPE (spaces in our case)
|
| 851 |
+
bpe_tokens.extend(bpe_token for bpe_token in self.bpe(token).split(" "))
|
| 852 |
+
return bpe_tokens
|
| 853 |
+
|
| 854 |
+
def convert_tokens_to_ids(self, tokens: Union[str, list[str]]) -> Union[int, list[int]]:
|
| 855 |
+
"""
|
| 856 |
+
Modified from `transformers.tokenization_utils.PreTrainedTokenzier.convert_tokens_to_ids`.
|
| 857 |
+
|
| 858 |
+
Converts a token string (or a sequence of tokens) in a single integer id (or a sequence of ids), using the
|
| 859 |
+
vocabulary.
|
| 860 |
+
|
| 861 |
+
This adaptation supports domain-specific tokenizers.
|
| 862 |
+
|
| 863 |
+
Args:
|
| 864 |
+
tokens (`str` or `List[str]`): One or several token(s) to convert to token id(s).
|
| 865 |
+
|
| 866 |
+
Returns:
|
| 867 |
+
`int` or `List[int]`: The token id or list of token ids.
|
| 868 |
+
"""
|
| 869 |
+
if tokens is None:
|
| 870 |
+
return None
|
| 871 |
+
|
| 872 |
+
if isinstance(tokens, str):
|
| 873 |
+
return self._convert_token_to_id_with_added_voc(tokens)
|
| 874 |
+
|
| 875 |
+
ids = []
|
| 876 |
+
extra_tokenizer_stack = []
|
| 877 |
+
|
| 878 |
+
for token in tokens:
|
| 879 |
+
if token not in self.ex_auto_begin_mapping and token not in self.ex_auto_end_mapping:
|
| 880 |
+
ids.append(
|
| 881 |
+
self._convert_token_to_id_with_added_voc(token, extra_tokenizer_stack=extra_tokenizer_stack)
|
| 882 |
+
)
|
| 883 |
+
if token in self.ex_all_begin_mapping:
|
| 884 |
+
extra_tokenizer_stack.append(self.ex_all_begin_mapping[token])
|
| 885 |
+
elif token in self.ex_all_end_mapping:
|
| 886 |
+
if extra_tokenizer_stack:
|
| 887 |
+
self._pop_logical_sp_token(extra_tokenizer_stack, token)
|
| 888 |
+
return ids
|
| 889 |
+
|
| 890 |
+
def _convert_token_to_id_with_added_voc(self, token, **kwargs):
|
| 891 |
+
"""
|
| 892 |
+
Modified from `transformers.tokenization_utils.PreTrainedTokenzier._convert_token_to_id_with_added_voc`.
|
| 893 |
+
|
| 894 |
+
This adaptation supports domain-specific tokenizers.
|
| 895 |
+
"""
|
| 896 |
+
if token is None:
|
| 897 |
+
return None
|
| 898 |
+
|
| 899 |
+
if token in self._added_tokens_encoder:
|
| 900 |
+
return self._added_tokens_encoder[token]
|
| 901 |
+
return self._convert_token_to_id(token, **kwargs)
|
| 902 |
+
|
| 903 |
+
def _convert_token_to_id(self, token, **kwargs):
|
| 904 |
+
"""
|
| 905 |
+
Modified from `transformers.tokenization_utils.PreTrainedTokenzier._convert_token_to_id`.
|
| 906 |
+
|
| 907 |
+
Converts a token (str) in an id using the vocab.
|
| 908 |
+
|
| 909 |
+
Fall back to original tokenizer once OOV.
|
| 910 |
+
"""
|
| 911 |
+
extra_tokenizer_stack = kwargs.pop("extra_tokenizer_stack", False)
|
| 912 |
+
if extra_tokenizer_stack:
|
| 913 |
+
token_id = extra_tokenizer_stack[-1].piece_to_id(token)
|
| 914 |
+
if token_id == extra_tokenizer_stack[-1].unk_id():
|
| 915 |
+
return self.encoder.get(token, self.encoder.get(self._unk_token))
|
| 916 |
+
else:
|
| 917 |
+
return token_id + extra_tokenizer_stack[-1].offset
|
| 918 |
+
else:
|
| 919 |
+
return self.encoder.get(token, self.encoder.get(self._unk_token))
|
| 920 |
+
|
| 921 |
+
# Copied from transformers.models.gpt2.tokenization_gpt2.GPT2Tokenizer._convert_id_to_token
|
| 922 |
+
def _convert_id_to_token(self, index):
|
| 923 |
+
"""Converts an index (integer) in a token (str) using the vocab."""
|
| 924 |
+
return self.decoder.get(index)
|
| 925 |
+
|
| 926 |
+
def convert_tokens_to_string(self, tokens):
|
| 927 |
+
"""Converts a sequence of tokens (string) in a single string."""
|
| 928 |
+
text = "".join(tokens)
|
| 929 |
+
text = text.replace(
|
| 930 |
+
"▁", "Ġ"
|
| 931 |
+
) # This discrepancy stems from differing whitespace treatment in SentencePiece versus BPE tokenization.
|
| 932 |
+
text = text.replace("\n", "Ċ")
|
| 933 |
+
text = bytearray([self.byte_decoder[c] for c in text]).decode("utf-8", errors=self.errors)
|
| 934 |
+
return text
|
| 935 |
+
|
| 936 |
+
def decode(
|
| 937 |
+
self,
|
| 938 |
+
token_ids,
|
| 939 |
+
skip_special_tokens: bool = False,
|
| 940 |
+
clean_up_tokenization_spaces: Optional[bool] = False,
|
| 941 |
+
spaces_between_special_tokens: bool = False,
|
| 942 |
+
**kwargs,
|
| 943 |
+
) -> str:
|
| 944 |
+
# `spaces_between_special_tokens` defaults to True for _decode in slow tokenizers
|
| 945 |
+
# and cannot be configured elsewhere, but it should default to False for InternS1Tokenizer
|
| 946 |
+
return super().decode(
|
| 947 |
+
token_ids,
|
| 948 |
+
skip_special_tokens=skip_special_tokens,
|
| 949 |
+
clean_up_tokenization_spaces=clean_up_tokenization_spaces,
|
| 950 |
+
spaces_between_special_tokens=spaces_between_special_tokens,
|
| 951 |
+
**kwargs,
|
| 952 |
+
)
|
| 953 |
+
|
| 954 |
+
def save_vocabulary(self, save_directory: str, filename_prefix: Optional[str] = None) -> tuple[str]:
|
| 955 |
+
"""
|
| 956 |
+
Modified from `transformers.models.gpt2.tokenization_gpt2.GPT2Tokenizer.save_vocabulary` to support saving custom extension.
|
| 957 |
+
"""
|
| 958 |
+
if not os.path.isdir(save_directory):
|
| 959 |
+
logger.error(f"Vocabulary path ({save_directory}) should be a directory")
|
| 960 |
+
return
|
| 961 |
+
vocab_file = os.path.join(
|
| 962 |
+
save_directory, (filename_prefix + "-" if filename_prefix else "") + VOCAB_FILES_NAMES["vocab_file"]
|
| 963 |
+
)
|
| 964 |
+
merge_file = os.path.join(
|
| 965 |
+
save_directory, (filename_prefix + "-" if filename_prefix else "") + VOCAB_FILES_NAMES["merges_file"]
|
| 966 |
+
)
|
| 967 |
+
sp_model_smiles = os.path.join(
|
| 968 |
+
save_directory, (filename_prefix + "-" if filename_prefix else "") + VOCAB_FILES_NAMES["sp_model_SMILES"]
|
| 969 |
+
)
|
| 970 |
+
sp_model_prot = os.path.join(
|
| 971 |
+
save_directory, (filename_prefix + "-" if filename_prefix else "") + VOCAB_FILES_NAMES["sp_model_PROT"]
|
| 972 |
+
)
|
| 973 |
+
sp_model_xna = os.path.join(
|
| 974 |
+
save_directory, (filename_prefix + "-" if filename_prefix else "") + VOCAB_FILES_NAMES["sp_model_XNA"]
|
| 975 |
+
)
|
| 976 |
+
|
| 977 |
+
with open(vocab_file, "w", encoding="utf-8") as f:
|
| 978 |
+
f.write(json.dumps(self.encoder, indent=2, sort_keys=True, ensure_ascii=False) + "\n")
|
| 979 |
+
|
| 980 |
+
index = 0
|
| 981 |
+
with open(merge_file, "w", encoding="utf-8") as writer:
|
| 982 |
+
writer.write("#version: 0.2\n")
|
| 983 |
+
for bpe_tokens, token_index in sorted(self.bpe_ranks.items(), key=lambda kv: kv[1]):
|
| 984 |
+
if index != token_index:
|
| 985 |
+
logger.warning(
|
| 986 |
+
f"Saving vocabulary to {merge_file}: BPE merge indices are not consecutive."
|
| 987 |
+
" Please check that the tokenizer is not corrupted!"
|
| 988 |
+
)
|
| 989 |
+
index = token_index
|
| 990 |
+
writer.write(" ".join(bpe_tokens) + "\n")
|
| 991 |
+
index += 1
|
| 992 |
+
|
| 993 |
+
with open(sp_model_smiles, "wb") as f:
|
| 994 |
+
f.write(self.sp_model_SMILES.serialized_model_proto())
|
| 995 |
+
|
| 996 |
+
with open(sp_model_prot, "wb") as f:
|
| 997 |
+
f.write(self.sp_model_PROT.serialized_model_proto())
|
| 998 |
+
|
| 999 |
+
with open(sp_model_xna, "wb") as f:
|
| 1000 |
+
f.write(self.sp_model_XNA.serialized_model_proto())
|
| 1001 |
+
|
| 1002 |
+
return vocab_file, merge_file
|
| 1003 |
+
|
| 1004 |
+
def prepare_for_tokenization(self, text, **kwargs):
|
| 1005 |
+
text = unicodedata.normalize("NFC", text)
|
| 1006 |
+
return (text, kwargs)
|
| 1007 |
+
|
| 1008 |
+
|
| 1009 |
+
__all__ = ["InternS1Tokenizer"]
|
tokenizer_PROT.model
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1144f52f86f3ca5a29940d69b037e508c05a89e6eedbe42bea641e226b20dbe0
|
| 3 |
+
size 12118
|
tokenizer_SMILES.model
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:fba1c97da0353ccbffd368ae78e311ccbc762aa5ba74f9aff8bf2ab363c4d37d
|
| 3 |
+
size 14775
|
tokenizer_XNA.model
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:58fc8bfb2af3dfe936a13dad8a9cb28dab7850b70b358db19605d867c133fb35
|
| 3 |
+
size 15451
|
tokenizer_config.json
ADDED
|
@@ -0,0 +1,506 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"add_prefix_space": false,
|
| 3 |
+
"added_tokens_decoder": {
|
| 4 |
+
"248044": {
|
| 5 |
+
"content": "<|endoftext|>",
|
| 6 |
+
"lstrip": false,
|
| 7 |
+
"normalized": false,
|
| 8 |
+
"rstrip": false,
|
| 9 |
+
"single_word": false,
|
| 10 |
+
"special": true
|
| 11 |
+
},
|
| 12 |
+
"248045": {
|
| 13 |
+
"content": "<|im_start|>",
|
| 14 |
+
"lstrip": false,
|
| 15 |
+
"normalized": false,
|
| 16 |
+
"rstrip": false,
|
| 17 |
+
"single_word": false,
|
| 18 |
+
"special": true
|
| 19 |
+
},
|
| 20 |
+
"248046": {
|
| 21 |
+
"content": "<|im_end|>",
|
| 22 |
+
"lstrip": false,
|
| 23 |
+
"normalized": false,
|
| 24 |
+
"rstrip": false,
|
| 25 |
+
"single_word": false,
|
| 26 |
+
"special": true
|
| 27 |
+
},
|
| 28 |
+
"248047": {
|
| 29 |
+
"content": "<|object_ref_start|>",
|
| 30 |
+
"lstrip": false,
|
| 31 |
+
"normalized": false,
|
| 32 |
+
"rstrip": false,
|
| 33 |
+
"single_word": false,
|
| 34 |
+
"special": true
|
| 35 |
+
},
|
| 36 |
+
"248048": {
|
| 37 |
+
"content": "<|object_ref_end|>",
|
| 38 |
+
"lstrip": false,
|
| 39 |
+
"normalized": false,
|
| 40 |
+
"rstrip": false,
|
| 41 |
+
"single_word": false,
|
| 42 |
+
"special": true
|
| 43 |
+
},
|
| 44 |
+
"248049": {
|
| 45 |
+
"content": "<|box_start|>",
|
| 46 |
+
"lstrip": false,
|
| 47 |
+
"normalized": false,
|
| 48 |
+
"rstrip": false,
|
| 49 |
+
"single_word": false,
|
| 50 |
+
"special": true
|
| 51 |
+
},
|
| 52 |
+
"248050": {
|
| 53 |
+
"content": "<|box_end|>",
|
| 54 |
+
"lstrip": false,
|
| 55 |
+
"normalized": false,
|
| 56 |
+
"rstrip": false,
|
| 57 |
+
"single_word": false,
|
| 58 |
+
"special": true
|
| 59 |
+
},
|
| 60 |
+
"248051": {
|
| 61 |
+
"content": "<|quad_start|>",
|
| 62 |
+
"lstrip": false,
|
| 63 |
+
"normalized": false,
|
| 64 |
+
"rstrip": false,
|
| 65 |
+
"single_word": false,
|
| 66 |
+
"special": true
|
| 67 |
+
},
|
| 68 |
+
"248052": {
|
| 69 |
+
"content": "<|quad_end|>",
|
| 70 |
+
"lstrip": false,
|
| 71 |
+
"normalized": false,
|
| 72 |
+
"rstrip": false,
|
| 73 |
+
"single_word": false,
|
| 74 |
+
"special": true
|
| 75 |
+
},
|
| 76 |
+
"248053": {
|
| 77 |
+
"content": "<|vision_start|>",
|
| 78 |
+
"lstrip": false,
|
| 79 |
+
"normalized": false,
|
| 80 |
+
"rstrip": false,
|
| 81 |
+
"single_word": false,
|
| 82 |
+
"special": true
|
| 83 |
+
},
|
| 84 |
+
"248054": {
|
| 85 |
+
"content": "<|vision_end|>",
|
| 86 |
+
"lstrip": false,
|
| 87 |
+
"normalized": false,
|
| 88 |
+
"rstrip": false,
|
| 89 |
+
"single_word": false,
|
| 90 |
+
"special": true
|
| 91 |
+
},
|
| 92 |
+
"248055": {
|
| 93 |
+
"content": "<|vision_pad|>",
|
| 94 |
+
"lstrip": false,
|
| 95 |
+
"normalized": false,
|
| 96 |
+
"rstrip": false,
|
| 97 |
+
"single_word": false,
|
| 98 |
+
"special": true
|
| 99 |
+
},
|
| 100 |
+
"248056": {
|
| 101 |
+
"content": "<|image_pad|>",
|
| 102 |
+
"lstrip": false,
|
| 103 |
+
"normalized": false,
|
| 104 |
+
"rstrip": false,
|
| 105 |
+
"single_word": false,
|
| 106 |
+
"special": true
|
| 107 |
+
},
|
| 108 |
+
"248057": {
|
| 109 |
+
"content": "<|video_pad|>",
|
| 110 |
+
"lstrip": false,
|
| 111 |
+
"normalized": false,
|
| 112 |
+
"rstrip": false,
|
| 113 |
+
"single_word": false,
|
| 114 |
+
"special": true
|
| 115 |
+
},
|
| 116 |
+
"248058": {
|
| 117 |
+
"content": "<tool_call>",
|
| 118 |
+
"lstrip": false,
|
| 119 |
+
"normalized": false,
|
| 120 |
+
"rstrip": false,
|
| 121 |
+
"single_word": false,
|
| 122 |
+
"special": false
|
| 123 |
+
},
|
| 124 |
+
"248059": {
|
| 125 |
+
"content": "</tool_call>",
|
| 126 |
+
"lstrip": false,
|
| 127 |
+
"normalized": false,
|
| 128 |
+
"rstrip": false,
|
| 129 |
+
"single_word": false,
|
| 130 |
+
"special": false
|
| 131 |
+
},
|
| 132 |
+
"248060": {
|
| 133 |
+
"content": "<|fim_prefix|>",
|
| 134 |
+
"lstrip": false,
|
| 135 |
+
"normalized": false,
|
| 136 |
+
"rstrip": false,
|
| 137 |
+
"single_word": false,
|
| 138 |
+
"special": false
|
| 139 |
+
},
|
| 140 |
+
"248061": {
|
| 141 |
+
"content": "<|fim_middle|>",
|
| 142 |
+
"lstrip": false,
|
| 143 |
+
"normalized": false,
|
| 144 |
+
"rstrip": false,
|
| 145 |
+
"single_word": false,
|
| 146 |
+
"special": false
|
| 147 |
+
},
|
| 148 |
+
"248062": {
|
| 149 |
+
"content": "<|fim_suffix|>",
|
| 150 |
+
"lstrip": false,
|
| 151 |
+
"normalized": false,
|
| 152 |
+
"rstrip": false,
|
| 153 |
+
"single_word": false,
|
| 154 |
+
"special": false
|
| 155 |
+
},
|
| 156 |
+
"248063": {
|
| 157 |
+
"content": "<|fim_pad|>",
|
| 158 |
+
"lstrip": false,
|
| 159 |
+
"normalized": false,
|
| 160 |
+
"rstrip": false,
|
| 161 |
+
"single_word": false,
|
| 162 |
+
"special": false
|
| 163 |
+
},
|
| 164 |
+
"248064": {
|
| 165 |
+
"content": "<|repo_name|>",
|
| 166 |
+
"lstrip": false,
|
| 167 |
+
"normalized": false,
|
| 168 |
+
"rstrip": false,
|
| 169 |
+
"single_word": false,
|
| 170 |
+
"special": false
|
| 171 |
+
},
|
| 172 |
+
"248065": {
|
| 173 |
+
"content": "<|file_sep|>",
|
| 174 |
+
"lstrip": false,
|
| 175 |
+
"normalized": false,
|
| 176 |
+
"rstrip": false,
|
| 177 |
+
"single_word": false,
|
| 178 |
+
"special": false
|
| 179 |
+
},
|
| 180 |
+
"248066": {
|
| 181 |
+
"content": "<tool_response>",
|
| 182 |
+
"lstrip": false,
|
| 183 |
+
"normalized": false,
|
| 184 |
+
"rstrip": false,
|
| 185 |
+
"single_word": false,
|
| 186 |
+
"special": false
|
| 187 |
+
},
|
| 188 |
+
"248067": {
|
| 189 |
+
"content": "</tool_response>",
|
| 190 |
+
"lstrip": false,
|
| 191 |
+
"normalized": false,
|
| 192 |
+
"rstrip": false,
|
| 193 |
+
"single_word": false,
|
| 194 |
+
"special": false
|
| 195 |
+
},
|
| 196 |
+
"248068": {
|
| 197 |
+
"content": "<think>",
|
| 198 |
+
"lstrip": false,
|
| 199 |
+
"normalized": false,
|
| 200 |
+
"rstrip": false,
|
| 201 |
+
"single_word": false,
|
| 202 |
+
"special": false
|
| 203 |
+
},
|
| 204 |
+
"248069": {
|
| 205 |
+
"content": "</think>",
|
| 206 |
+
"lstrip": false,
|
| 207 |
+
"normalized": false,
|
| 208 |
+
"rstrip": false,
|
| 209 |
+
"single_word": false,
|
| 210 |
+
"special": false
|
| 211 |
+
},
|
| 212 |
+
"248070": {
|
| 213 |
+
"content": "<|audio_start|>",
|
| 214 |
+
"lstrip": false,
|
| 215 |
+
"normalized": false,
|
| 216 |
+
"rstrip": false,
|
| 217 |
+
"single_word": false,
|
| 218 |
+
"special": true
|
| 219 |
+
},
|
| 220 |
+
"248071": {
|
| 221 |
+
"content": "<|audio_end|>",
|
| 222 |
+
"lstrip": false,
|
| 223 |
+
"normalized": false,
|
| 224 |
+
"rstrip": false,
|
| 225 |
+
"single_word": false,
|
| 226 |
+
"special": true
|
| 227 |
+
},
|
| 228 |
+
"248072": {
|
| 229 |
+
"content": "<tts_pad>",
|
| 230 |
+
"lstrip": false,
|
| 231 |
+
"normalized": false,
|
| 232 |
+
"rstrip": false,
|
| 233 |
+
"single_word": false,
|
| 234 |
+
"special": true
|
| 235 |
+
},
|
| 236 |
+
"248073": {
|
| 237 |
+
"content": "<tts_text_bos>",
|
| 238 |
+
"lstrip": false,
|
| 239 |
+
"normalized": false,
|
| 240 |
+
"rstrip": false,
|
| 241 |
+
"single_word": false,
|
| 242 |
+
"special": true
|
| 243 |
+
},
|
| 244 |
+
"248074": {
|
| 245 |
+
"content": "<tts_text_eod>",
|
| 246 |
+
"lstrip": false,
|
| 247 |
+
"normalized": false,
|
| 248 |
+
"rstrip": false,
|
| 249 |
+
"single_word": false,
|
| 250 |
+
"special": true
|
| 251 |
+
},
|
| 252 |
+
"248075": {
|
| 253 |
+
"content": "<tts_text_bos_single>",
|
| 254 |
+
"lstrip": false,
|
| 255 |
+
"normalized": false,
|
| 256 |
+
"rstrip": false,
|
| 257 |
+
"single_word": false,
|
| 258 |
+
"special": true
|
| 259 |
+
},
|
| 260 |
+
"248076": {
|
| 261 |
+
"content": "<|audio_pad|>",
|
| 262 |
+
"lstrip": false,
|
| 263 |
+
"normalized": false,
|
| 264 |
+
"rstrip": false,
|
| 265 |
+
"single_word": false,
|
| 266 |
+
"special": true
|
| 267 |
+
},
|
| 268 |
+
"248077": {
|
| 269 |
+
"content": "<IMG_CONTEXT>",
|
| 270 |
+
"lstrip": false,
|
| 271 |
+
"normalized": false,
|
| 272 |
+
"rstrip": false,
|
| 273 |
+
"single_word": false,
|
| 274 |
+
"special": true
|
| 275 |
+
},
|
| 276 |
+
"248078": {
|
| 277 |
+
"content": "<img>",
|
| 278 |
+
"lstrip": false,
|
| 279 |
+
"normalized": false,
|
| 280 |
+
"rstrip": false,
|
| 281 |
+
"single_word": false,
|
| 282 |
+
"special": true
|
| 283 |
+
},
|
| 284 |
+
"248079": {
|
| 285 |
+
"content": "</img>",
|
| 286 |
+
"lstrip": false,
|
| 287 |
+
"normalized": false,
|
| 288 |
+
"rstrip": false,
|
| 289 |
+
"single_word": false,
|
| 290 |
+
"special": true
|
| 291 |
+
},
|
| 292 |
+
"248080": {
|
| 293 |
+
"content": "<quad>",
|
| 294 |
+
"lstrip": false,
|
| 295 |
+
"normalized": false,
|
| 296 |
+
"rstrip": false,
|
| 297 |
+
"single_word": false,
|
| 298 |
+
"special": true
|
| 299 |
+
},
|
| 300 |
+
"248081": {
|
| 301 |
+
"content": "</quad>",
|
| 302 |
+
"lstrip": false,
|
| 303 |
+
"normalized": false,
|
| 304 |
+
"rstrip": false,
|
| 305 |
+
"single_word": false,
|
| 306 |
+
"special": true
|
| 307 |
+
},
|
| 308 |
+
"248082": {
|
| 309 |
+
"content": "<ref>",
|
| 310 |
+
"lstrip": false,
|
| 311 |
+
"normalized": false,
|
| 312 |
+
"rstrip": false,
|
| 313 |
+
"single_word": false,
|
| 314 |
+
"special": true
|
| 315 |
+
},
|
| 316 |
+
"248083": {
|
| 317 |
+
"content": "</ref>",
|
| 318 |
+
"lstrip": false,
|
| 319 |
+
"normalized": false,
|
| 320 |
+
"rstrip": false,
|
| 321 |
+
"single_word": false,
|
| 322 |
+
"special": true
|
| 323 |
+
},
|
| 324 |
+
"248084": {
|
| 325 |
+
"content": "<box>",
|
| 326 |
+
"lstrip": false,
|
| 327 |
+
"normalized": false,
|
| 328 |
+
"rstrip": false,
|
| 329 |
+
"single_word": false,
|
| 330 |
+
"special": true
|
| 331 |
+
},
|
| 332 |
+
"248085": {
|
| 333 |
+
"content": "</box>",
|
| 334 |
+
"lstrip": false,
|
| 335 |
+
"normalized": false,
|
| 336 |
+
"rstrip": false,
|
| 337 |
+
"single_word": false,
|
| 338 |
+
"special": true
|
| 339 |
+
},
|
| 340 |
+
"248086": {
|
| 341 |
+
"content": "<|action_start|>",
|
| 342 |
+
"lstrip": false,
|
| 343 |
+
"normalized": false,
|
| 344 |
+
"rstrip": false,
|
| 345 |
+
"single_word": false,
|
| 346 |
+
"special": true
|
| 347 |
+
},
|
| 348 |
+
"248087": {
|
| 349 |
+
"content": "<|action_end|>",
|
| 350 |
+
"lstrip": false,
|
| 351 |
+
"normalized": false,
|
| 352 |
+
"rstrip": false,
|
| 353 |
+
"single_word": false,
|
| 354 |
+
"special": true
|
| 355 |
+
},
|
| 356 |
+
"248088": {
|
| 357 |
+
"content": "<|interpreter|>",
|
| 358 |
+
"lstrip": false,
|
| 359 |
+
"normalized": false,
|
| 360 |
+
"rstrip": false,
|
| 361 |
+
"single_word": false,
|
| 362 |
+
"special": true
|
| 363 |
+
},
|
| 364 |
+
"248089": {
|
| 365 |
+
"content": "<|plugin|>",
|
| 366 |
+
"lstrip": false,
|
| 367 |
+
"normalized": false,
|
| 368 |
+
"rstrip": false,
|
| 369 |
+
"single_word": false,
|
| 370 |
+
"special": true
|
| 371 |
+
},
|
| 372 |
+
"248090": {
|
| 373 |
+
"content": "<video>",
|
| 374 |
+
"lstrip": false,
|
| 375 |
+
"normalized": false,
|
| 376 |
+
"rstrip": false,
|
| 377 |
+
"single_word": false,
|
| 378 |
+
"special": true
|
| 379 |
+
},
|
| 380 |
+
"248091": {
|
| 381 |
+
"content": "<|ts|>",
|
| 382 |
+
"lstrip": false,
|
| 383 |
+
"normalized": false,
|
| 384 |
+
"rstrip": false,
|
| 385 |
+
"single_word": false,
|
| 386 |
+
"special": true
|
| 387 |
+
},
|
| 388 |
+
"248092": {
|
| 389 |
+
"content": "<|/ts|>",
|
| 390 |
+
"lstrip": false,
|
| 391 |
+
"normalized": false,
|
| 392 |
+
"rstrip": false,
|
| 393 |
+
"single_word": false,
|
| 394 |
+
"special": true
|
| 395 |
+
},
|
| 396 |
+
"248093": {
|
| 397 |
+
"content": "<TS_CONTEXT>",
|
| 398 |
+
"lstrip": false,
|
| 399 |
+
"normalized": false,
|
| 400 |
+
"rstrip": false,
|
| 401 |
+
"single_word": false,
|
| 402 |
+
"special": true
|
| 403 |
+
},
|
| 404 |
+
"248094": {
|
| 405 |
+
"content": "<SMILES>",
|
| 406 |
+
"lstrip": false,
|
| 407 |
+
"normalized": false,
|
| 408 |
+
"rstrip": false,
|
| 409 |
+
"single_word": false,
|
| 410 |
+
"special": false
|
| 411 |
+
},
|
| 412 |
+
"248095": {
|
| 413 |
+
"content": "</SMILES>",
|
| 414 |
+
"lstrip": false,
|
| 415 |
+
"normalized": false,
|
| 416 |
+
"rstrip": false,
|
| 417 |
+
"single_word": false,
|
| 418 |
+
"special": false
|
| 419 |
+
},
|
| 420 |
+
"248096": {
|
| 421 |
+
"content": "<protein>",
|
| 422 |
+
"lstrip": false,
|
| 423 |
+
"normalized": false,
|
| 424 |
+
"rstrip": false,
|
| 425 |
+
"single_word": false,
|
| 426 |
+
"special": false
|
| 427 |
+
},
|
| 428 |
+
"248097": {
|
| 429 |
+
"content": "</protein>",
|
| 430 |
+
"lstrip": false,
|
| 431 |
+
"normalized": false,
|
| 432 |
+
"rstrip": false,
|
| 433 |
+
"single_word": false,
|
| 434 |
+
"special": false
|
| 435 |
+
},
|
| 436 |
+
"248098": {
|
| 437 |
+
"content": "<dna>",
|
| 438 |
+
"lstrip": false,
|
| 439 |
+
"normalized": false,
|
| 440 |
+
"rstrip": false,
|
| 441 |
+
"single_word": false,
|
| 442 |
+
"special": false
|
| 443 |
+
},
|
| 444 |
+
"248099": {
|
| 445 |
+
"content": "</dna>",
|
| 446 |
+
"lstrip": false,
|
| 447 |
+
"normalized": false,
|
| 448 |
+
"rstrip": false,
|
| 449 |
+
"single_word": false,
|
| 450 |
+
"special": false
|
| 451 |
+
},
|
| 452 |
+
"248100": {
|
| 453 |
+
"content": "<rna>",
|
| 454 |
+
"lstrip": false,
|
| 455 |
+
"normalized": false,
|
| 456 |
+
"rstrip": false,
|
| 457 |
+
"single_word": false,
|
| 458 |
+
"special": false
|
| 459 |
+
},
|
| 460 |
+
"248101": {
|
| 461 |
+
"content": "</rna>",
|
| 462 |
+
"lstrip": false,
|
| 463 |
+
"normalized": false,
|
| 464 |
+
"rstrip": false,
|
| 465 |
+
"single_word": false,
|
| 466 |
+
"special": false
|
| 467 |
+
}
|
| 468 |
+
},
|
| 469 |
+
"audio_bos_token": "<|audio_start|>",
|
| 470 |
+
"audio_eos_token": "<|audio_end|>",
|
| 471 |
+
"audio_token": "<|audio_pad|>",
|
| 472 |
+
"auto_map": {
|
| 473 |
+
"AutoTokenizer": [
|
| 474 |
+
"tokenization_interns1.InternS1Tokenizer",
|
| 475 |
+
null
|
| 476 |
+
]
|
| 477 |
+
},
|
| 478 |
+
"backend": "custom",
|
| 479 |
+
"bos_token": "<|im_start|>",
|
| 480 |
+
"clean_up_tokenization_spaces": false,
|
| 481 |
+
"eos_token": "<|im_end|>",
|
| 482 |
+
"errors": "replace",
|
| 483 |
+
"image_token": "<|image_pad|>",
|
| 484 |
+
"is_local": true,
|
| 485 |
+
"model_max_length": 262144,
|
| 486 |
+
"model_specific_special_tokens": {
|
| 487 |
+
"audio_bos_token": "<|audio_start|>",
|
| 488 |
+
"audio_eos_token": "<|audio_end|>",
|
| 489 |
+
"audio_token": "<|audio_pad|>",
|
| 490 |
+
"image_token": "<|image_pad|>",
|
| 491 |
+
"video_token": "<|video_pad|>",
|
| 492 |
+
"vision_bos_token": "<|vision_start|>",
|
| 493 |
+
"vision_eos_token": "<|vision_end|>"
|
| 494 |
+
},
|
| 495 |
+
"offset_PROT": 249126,
|
| 496 |
+
"offset_SMILES": 248102,
|
| 497 |
+
"offset_XNA": 250150,
|
| 498 |
+
"pad_token": "<|endoftext|>",
|
| 499 |
+
"pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
|
| 500 |
+
"split_special_tokens": false,
|
| 501 |
+
"tokenizer_class": "InternS1Tokenizer",
|
| 502 |
+
"unk_token": null,
|
| 503 |
+
"video_token": "<|video_pad|>",
|
| 504 |
+
"vision_bos_token": "<|vision_start|>",
|
| 505 |
+
"vision_eos_token": "<|vision_end|>"
|
| 506 |
+
}
|
video_preprocessor_config.json
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"size": {
|
| 3 |
+
"longest_edge": 25165824,
|
| 4 |
+
"shortest_edge": 4096
|
| 5 |
+
},
|
| 6 |
+
"patch_size": 16,
|
| 7 |
+
"temporal_patch_size": 2,
|
| 8 |
+
"merge_size": 2,
|
| 9 |
+
"image_mean": [
|
| 10 |
+
0.5,
|
| 11 |
+
0.5,
|
| 12 |
+
0.5
|
| 13 |
+
],
|
| 14 |
+
"image_std": [
|
| 15 |
+
0.5,
|
| 16 |
+
0.5,
|
| 17 |
+
0.5
|
| 18 |
+
],
|
| 19 |
+
"processor_class": "Qwen3VLProcessor",
|
| 20 |
+
"video_processor_type": "Qwen3VLVideoProcessor"
|
| 21 |
+
}
|
vocab.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|