jvogan commited on
Commit ·
6ca9813
0
Parent(s):
Prepare VibeThinker-3B J Lens model release
Browse files- .gitattributes +3 -0
- .gitignore +36 -0
- LICENSES/APACHE-2.0.txt +201 -0
- LICENSES/QWEN-RESEARCH.txt +54 -0
- LICENSES/VIBETHINKER-LICENSE-NOTE.txt +21 -0
- NOTICE +25 -0
- README.md +322 -0
- SHA256SUMS +2 -0
- THIRD_PARTY_NOTICES.md +53 -0
- assets/jlens-model-banner.png +3 -0
- assets/two-lens-files.png +3 -0
- assets/two-lens-files.svg +7 -0
- evaluation.json +144 -0
- evaluation.safetensors +3 -0
- evaluation_compatibility.json +105 -0
- evaluation_provenance.json +90 -0
- evaluation_tensor_manifest.json +209 -0
- evaluation_validation.json +29 -0
- lens_config.json +62 -0
- model.safetensors +3 -0
- provenance.json +128 -0
- requirements.txt +2 -0
- scripts/convert_checkpoint.py +275 -0
- scripts/convert_evaluation_checkpoint.py +439 -0
- scripts/validate_artifact.py +1395 -0
- tensor_manifest.json +208 -0
- validation.json +28 -0
.gitattributes
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
| 2 |
+
*.png filter=lfs diff=lfs merge=lfs -text
|
| 3 |
+
LICENSES/QWEN-RESEARCH.txt text eol=lf whitespace=-trailing-space
|
.gitignore
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Public release allowlist. Add each new intended file explicitly.
|
| 2 |
+
*
|
| 3 |
+
|
| 4 |
+
!/.gitignore
|
| 5 |
+
!/.gitattributes
|
| 6 |
+
!/README.md
|
| 7 |
+
!/NOTICE
|
| 8 |
+
!/THIRD_PARTY_NOTICES.md
|
| 9 |
+
!/SHA256SUMS
|
| 10 |
+
!/model.safetensors
|
| 11 |
+
!/evaluation.safetensors
|
| 12 |
+
!/lens_config.json
|
| 13 |
+
!/tensor_manifest.json
|
| 14 |
+
!/evaluation_tensor_manifest.json
|
| 15 |
+
!/provenance.json
|
| 16 |
+
!/evaluation_provenance.json
|
| 17 |
+
!/validation.json
|
| 18 |
+
!/evaluation_validation.json
|
| 19 |
+
!/evaluation.json
|
| 20 |
+
!/evaluation_compatibility.json
|
| 21 |
+
!/requirements.txt
|
| 22 |
+
|
| 23 |
+
!/assets/
|
| 24 |
+
!/assets/jlens-model-banner.png
|
| 25 |
+
!/assets/two-lens-files.png
|
| 26 |
+
!/assets/two-lens-files.svg
|
| 27 |
+
|
| 28 |
+
!/LICENSES/
|
| 29 |
+
!/LICENSES/APACHE-2.0.txt
|
| 30 |
+
!/LICENSES/QWEN-RESEARCH.txt
|
| 31 |
+
!/LICENSES/VIBETHINKER-LICENSE-NOTE.txt
|
| 32 |
+
|
| 33 |
+
!/scripts/
|
| 34 |
+
!/scripts/convert_checkpoint.py
|
| 35 |
+
!/scripts/convert_evaluation_checkpoint.py
|
| 36 |
+
!/scripts/validate_artifact.py
|
LICENSES/APACHE-2.0.txt
ADDED
|
@@ -0,0 +1,201 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Apache License
|
| 2 |
+
Version 2.0, January 2004
|
| 3 |
+
http://www.apache.org/licenses/
|
| 4 |
+
|
| 5 |
+
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
| 6 |
+
|
| 7 |
+
1. Definitions.
|
| 8 |
+
|
| 9 |
+
"License" shall mean the terms and conditions for use, reproduction,
|
| 10 |
+
and distribution as defined by Sections 1 through 9 of this document.
|
| 11 |
+
|
| 12 |
+
"Licensor" shall mean the copyright owner or entity authorized by
|
| 13 |
+
the copyright owner that is granting the License.
|
| 14 |
+
|
| 15 |
+
"Legal Entity" shall mean the union of the acting entity and all
|
| 16 |
+
other entities that control, are controlled by, or are under common
|
| 17 |
+
control with that entity. For the purposes of this definition,
|
| 18 |
+
"control" means (i) the power, direct or indirect, to cause the
|
| 19 |
+
direction or management of such entity, whether by contract or
|
| 20 |
+
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
| 21 |
+
outstanding shares, or (iii) beneficial ownership of such entity.
|
| 22 |
+
|
| 23 |
+
"You" (or "Your") shall mean an individual or Legal Entity
|
| 24 |
+
exercising permissions granted by this License.
|
| 25 |
+
|
| 26 |
+
"Source" form shall mean the preferred form for making modifications,
|
| 27 |
+
including but not limited to software source code, documentation
|
| 28 |
+
source, and configuration files.
|
| 29 |
+
|
| 30 |
+
"Object" form shall mean any form resulting from mechanical
|
| 31 |
+
transformation or translation of a Source form, including but
|
| 32 |
+
not limited to compiled object code, generated documentation,
|
| 33 |
+
and conversions to other media types.
|
| 34 |
+
|
| 35 |
+
"Work" shall mean the work of authorship, whether in Source or
|
| 36 |
+
Object form, made available under the License, as indicated by a
|
| 37 |
+
copyright notice that is included in or attached to the work
|
| 38 |
+
(an example is provided in the Appendix below).
|
| 39 |
+
|
| 40 |
+
"Derivative Works" shall mean any work, whether in Source or Object
|
| 41 |
+
form, that is based on (or derived from) the Work and for which the
|
| 42 |
+
editorial revisions, annotations, elaborations, or other modifications
|
| 43 |
+
represent, as a whole, an original work of authorship. For the purposes
|
| 44 |
+
of this License, Derivative Works shall not include works that remain
|
| 45 |
+
separable from, or merely link (or bind by name) to the interfaces of,
|
| 46 |
+
the Work and Derivative Works thereof.
|
| 47 |
+
|
| 48 |
+
"Contribution" shall mean any work of authorship, including
|
| 49 |
+
the original version of the Work and any modifications or additions
|
| 50 |
+
to that Work or Derivative Works thereof, that is intentionally
|
| 51 |
+
submitted to Licensor for inclusion in the Work by the copyright owner
|
| 52 |
+
or by an individual or Legal Entity authorized to submit on behalf of
|
| 53 |
+
the copyright owner. For the purposes of this definition, "submitted"
|
| 54 |
+
means any form of electronic, verbal, or written communication sent
|
| 55 |
+
to the Licensor or its representatives, including but not limited to
|
| 56 |
+
communication on electronic mailing lists, source code control systems,
|
| 57 |
+
and issue tracking systems that are managed by, or on behalf of, the
|
| 58 |
+
Licensor for the purpose of discussing and improving the Work, but
|
| 59 |
+
excluding communication that is conspicuously marked or otherwise
|
| 60 |
+
designated in writing by the copyright owner as "Not a Contribution."
|
| 61 |
+
|
| 62 |
+
"Contributor" shall mean Licensor and any individual or Legal Entity
|
| 63 |
+
on behalf of whom a Contribution has been received by Licensor and
|
| 64 |
+
subsequently incorporated within the Work.
|
| 65 |
+
|
| 66 |
+
2. Grant of Copyright License. Subject to the terms and conditions of
|
| 67 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 68 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 69 |
+
copyright license to reproduce, prepare Derivative Works of,
|
| 70 |
+
publicly display, publicly perform, sublicense, and distribute the
|
| 71 |
+
Work and such Derivative Works in Source or Object form.
|
| 72 |
+
|
| 73 |
+
3. Grant of Patent License. Subject to the terms and conditions of
|
| 74 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 75 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 76 |
+
(except as stated in this section) patent license to make, have made,
|
| 77 |
+
use, offer to sell, sell, import, and otherwise transfer the Work,
|
| 78 |
+
where such license applies only to those patent claims licensable
|
| 79 |
+
by such Contributor that are necessarily infringed by their
|
| 80 |
+
Contribution(s) alone or by combination of their Contribution(s)
|
| 81 |
+
with the Work to which such Contribution(s) was submitted. If You
|
| 82 |
+
institute patent litigation against any entity (including a
|
| 83 |
+
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
| 84 |
+
or a Contribution incorporated within the Work constitutes direct
|
| 85 |
+
or contributory patent infringement, then any patent licenses
|
| 86 |
+
granted to You under this License for that Work shall terminate as
|
| 87 |
+
of the date such litigation is filed.
|
| 88 |
+
|
| 89 |
+
4. Redistribution. You may reproduce and distribute copies of the
|
| 90 |
+
Work or Derivative Works thereof in any medium, with or without
|
| 91 |
+
modifications, and in Source or Object form, provided that You
|
| 92 |
+
meet the following conditions:
|
| 93 |
+
|
| 94 |
+
(a) You must give any other recipients of the Work or
|
| 95 |
+
Derivative Works a copy of this License; and
|
| 96 |
+
|
| 97 |
+
(b) You must cause any modified files to carry prominent notices
|
| 98 |
+
stating that You changed the files; and
|
| 99 |
+
|
| 100 |
+
(c) You must retain, in the Source form of any Derivative Works
|
| 101 |
+
that You distribute, all copyright, patent, trademark, and
|
| 102 |
+
attribution notices from the Source form of the Work,
|
| 103 |
+
excluding those notices that do not pertain to any part of
|
| 104 |
+
the Derivative Works; and
|
| 105 |
+
|
| 106 |
+
(d) If the Work includes a "NOTICE" text file as part of its
|
| 107 |
+
distribution, then any Derivative Works that You distribute must
|
| 108 |
+
include a readable copy of the attribution notices contained
|
| 109 |
+
within such NOTICE file, excluding those notices that do not
|
| 110 |
+
pertain to any part of the Derivative Works, in at least one
|
| 111 |
+
of the following places: within a NOTICE text file distributed
|
| 112 |
+
as part of the Derivative Works; within the Source form or
|
| 113 |
+
documentation, if provided along with the Derivative Works; or,
|
| 114 |
+
within a display generated by the Derivative Works, if and
|
| 115 |
+
wherever such third-party notices normally appear. The contents
|
| 116 |
+
of the NOTICE file are for informational purposes only and
|
| 117 |
+
do not modify the License. You may add Your own attribution
|
| 118 |
+
notices within Derivative Works that You distribute, alongside
|
| 119 |
+
or as an addendum to the NOTICE text from the Work, provided
|
| 120 |
+
that such additional attribution notices cannot be construed
|
| 121 |
+
as modifying the License.
|
| 122 |
+
|
| 123 |
+
You may add Your own copyright statement to Your modifications and
|
| 124 |
+
may provide additional or different license terms and conditions
|
| 125 |
+
for use, reproduction, or distribution of Your modifications, or
|
| 126 |
+
for any such Derivative Works as a whole, provided Your use,
|
| 127 |
+
reproduction, and distribution of the Work otherwise complies with
|
| 128 |
+
the conditions stated in this License.
|
| 129 |
+
|
| 130 |
+
5. Submission of Contributions. Unless You explicitly state otherwise,
|
| 131 |
+
any Contribution intentionally submitted for inclusion in the Work
|
| 132 |
+
by You to the Licensor shall be under the terms and conditions of
|
| 133 |
+
this License, without any additional terms or conditions.
|
| 134 |
+
Notwithstanding the above, nothing herein shall supersede or modify
|
| 135 |
+
the terms of any separate license agreement you may have executed
|
| 136 |
+
with Licensor regarding such Contributions.
|
| 137 |
+
|
| 138 |
+
6. Trademarks. This License does not grant permission to use the trade
|
| 139 |
+
names, trademarks, service marks, or product names of the Licensor,
|
| 140 |
+
except as required for reasonable and customary use in describing the
|
| 141 |
+
origin of the Work and reproducing the content of the NOTICE file.
|
| 142 |
+
|
| 143 |
+
7. Disclaimer of Warranty. Unless required by applicable law or
|
| 144 |
+
agreed to in writing, Licensor provides the Work (and each
|
| 145 |
+
Contributor provides its Contributions) on an "AS IS" BASIS,
|
| 146 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
| 147 |
+
implied, including, without limitation, any warranties or conditions
|
| 148 |
+
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
| 149 |
+
PARTICULAR PURPOSE. You are solely responsible for determining the
|
| 150 |
+
appropriateness of using or redistributing the Work and assume any
|
| 151 |
+
risks associated with Your exercise of permissions under this License.
|
| 152 |
+
|
| 153 |
+
8. Limitation of Liability. In no event and under no legal theory,
|
| 154 |
+
whether in tort (including negligence), contract, or otherwise,
|
| 155 |
+
unless required by applicable law (such as deliberate and grossly
|
| 156 |
+
negligent acts) or agreed to in writing, shall any Contributor be
|
| 157 |
+
liable to You for damages, including any direct, indirect, special,
|
| 158 |
+
incidental, or consequential damages of any character arising as a
|
| 159 |
+
result of this License or out of the use or inability to use the
|
| 160 |
+
Work (including but not limited to damages for loss of goodwill,
|
| 161 |
+
work stoppage, computer failure or malfunction, or any and all other
|
| 162 |
+
commercial damages or losses), even if such Contributor has been
|
| 163 |
+
advised of the possibility of such damages.
|
| 164 |
+
|
| 165 |
+
9. Accepting Warranty or Additional Liability. While redistributing
|
| 166 |
+
the Work or Derivative Works thereof, You may choose to offer,
|
| 167 |
+
and charge a fee for, acceptance of support, warranty, indemnity,
|
| 168 |
+
or other liability obligations and/or rights consistent with this
|
| 169 |
+
License. However, in accepting such obligations, You may act only
|
| 170 |
+
on Your own behalf and on Your sole responsibility, not on behalf
|
| 171 |
+
of any other Contributor, and only if You agree to indemnify,
|
| 172 |
+
defend, and hold each Contributor harmless for any liability
|
| 173 |
+
incurred by, or claims asserted against, such Contributor by reason
|
| 174 |
+
of your accepting any such warranty or additional liability.
|
| 175 |
+
|
| 176 |
+
END OF TERMS AND CONDITIONS
|
| 177 |
+
|
| 178 |
+
APPENDIX: How to apply the Apache License to your work.
|
| 179 |
+
|
| 180 |
+
To apply the Apache License to your work, attach the following
|
| 181 |
+
boilerplate notice, with the fields enclosed by brackets "[]"
|
| 182 |
+
replaced with your own identifying information. (Don't include
|
| 183 |
+
the brackets!) The text should be enclosed in the appropriate
|
| 184 |
+
comment syntax for the file format. We also recommend that a
|
| 185 |
+
file or class name and description of purpose be included on the
|
| 186 |
+
same "printed page" as the copyright notice for easier
|
| 187 |
+
identification within third-party archives.
|
| 188 |
+
|
| 189 |
+
Copyright [yyyy] [name of copyright owner]
|
| 190 |
+
|
| 191 |
+
Licensed under the Apache License, Version 2.0 (the "License");
|
| 192 |
+
you may not use this file except in compliance with the License.
|
| 193 |
+
You may obtain a copy of the License at
|
| 194 |
+
|
| 195 |
+
http://www.apache.org/licenses/LICENSE-2.0
|
| 196 |
+
|
| 197 |
+
Unless required by applicable law or agreed to in writing, software
|
| 198 |
+
distributed under the License is distributed on an "AS IS" BASIS,
|
| 199 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
| 200 |
+
See the License for the specific language governing permissions and
|
| 201 |
+
limitations under the License.
|
LICENSES/QWEN-RESEARCH.txt
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Qwen RESEARCH LICENSE AGREEMENT
|
| 2 |
+
|
| 3 |
+
Qwen RESEARCH LICENSE AGREEMENT Release Date: September 19, 2024
|
| 4 |
+
|
| 5 |
+
By clicking to agree or by using or distributing any portion or element of the Qwen Materials, you will be deemed to have recognized and accepted the content of this Agreement, which is effective immediately.
|
| 6 |
+
|
| 7 |
+
1. Definitions
|
| 8 |
+
a. This Qwen RESEARCH LICENSE AGREEMENT (this "Agreement") shall mean the terms and conditions for use, reproduction, distribution and modification of the Materials as defined by this Agreement.
|
| 9 |
+
b. "We" (or "Us") shall mean Alibaba Cloud.
|
| 10 |
+
c. "You" (or "Your") shall mean a natural person or legal entity exercising the rights granted by this Agreement and/or using the Materials for any purpose and in any field of use.
|
| 11 |
+
d. "Third Parties" shall mean individuals or legal entities that are not under common control with us or you.
|
| 12 |
+
e. "Qwen" shall mean the large language models, and software and algorithms, consisting of trained model weights, parameters (including optimizer states), machine-learning model code, inference-enabling code, training-enabling code, fine-tuning enabling code and other elements of the foregoing distributed by us.
|
| 13 |
+
f. "Materials" shall mean, collectively, Alibaba Cloud's proprietary Qwen and Documentation (and any portion thereof) made available under this Agreement.
|
| 14 |
+
g. "Source" form shall mean the preferred form for making modifications, including but not limited to model source code, documentation source, and configuration files.
|
| 15 |
+
h. "Object" form shall mean any form resulting from mechanical transformation or translation of a Source form, including but not limited to compiled object code, generated documentation, and conversions to other media types.
|
| 16 |
+
i. "Non-Commercial" shall mean for research or evaluation purposes only.
|
| 17 |
+
|
| 18 |
+
2. Grant of Rights
|
| 19 |
+
a. You are granted a non-exclusive, worldwide, non-transferable and royalty-free limited license under Alibaba Cloud's intellectual property or other rights owned by us embodied in the Materials to use, reproduce, distribute, copy, create derivative works of, and make modifications to the Materials FOR NON-COMMERCIAL PURPOSES ONLY.
|
| 20 |
+
b. If you are commercially using the Materials, you shall request a license from us.
|
| 21 |
+
|
| 22 |
+
3. Redistribution
|
| 23 |
+
You may distribute copies or make the Materials, or derivative works thereof, available as part of a product or service that contains any of them, with or without modifications, and in Source or Object form, provided that you meet the following conditions:
|
| 24 |
+
a. You shall give any other recipients of the Materials or derivative works a copy of this Agreement;
|
| 25 |
+
b. You shall cause any modified files to carry prominent notices stating that you changed the files;
|
| 26 |
+
c. You shall retain in all copies of the Materials that you distribute the following attribution notices within a "Notice" text file distributed as a part of such copies: "Qwen is licensed under the Qwen RESEARCH LICENSE AGREEMENT, Copyright (c) Alibaba Cloud. All Rights Reserved."; and
|
| 27 |
+
d. You may add your own copyright statement to your modifications and may provide additional or different license terms and conditions for use, reproduction, or distribution of your modifications, or for any such derivative works as a whole, provided your use, reproduction, and distribution of the work otherwise complies with the terms and conditions of this Agreement.
|
| 28 |
+
|
| 29 |
+
4. Rules of use
|
| 30 |
+
a. The Materials may be subject to export controls or restrictions in China, the United States or other countries or regions. You shall comply with applicable laws and regulations in your use of the Materials.
|
| 31 |
+
b. If you use the Materials or any outputs or results therefrom to create, train, fine-tune, or improve an AI model that is distributed or made available, you shall prominently display “Built with Qwen” or “Improved using Qwen” in the related product documentation.
|
| 32 |
+
|
| 33 |
+
5. Intellectual Property
|
| 34 |
+
a. We retain ownership of all intellectual property rights in and to the Materials and derivatives made by or for us. Conditioned upon compliance with the terms and conditions of this Agreement, with respect to any derivative works and modifications of the Materials that are made by you, you are and will be the owner of such derivative works and modifications.
|
| 35 |
+
b. No trademark license is granted to use the trade names, trademarks, service marks, or product names of us, except as required to fulfill notice requirements under this Agreement or as required for reasonable and customary use in describing and redistributing the Materials.
|
| 36 |
+
c. If you commence a lawsuit or other proceedings (including a cross-claim or counterclaim in a lawsuit) against us or any entity alleging that the Materials or any output therefrom, or any part of the foregoing, infringe any intellectual property or other right owned or licensable by you, then all licenses granted to you under this Agreement shall terminate as of the date such lawsuit or other proceeding is commenced or brought.
|
| 37 |
+
|
| 38 |
+
6. Disclaimer of Warranty and Limitation of Liability
|
| 39 |
+
a. We are not obligated to support, update, provide training for, or develop any further version of the Qwen Materials or to grant any license thereto.
|
| 40 |
+
b. THE MATERIALS ARE PROVIDED "AS IS" WITHOUT ANY EXPRESS OR IMPLIED WARRANTY OF ANY KIND INCLUDING WARRANTIES OF MERCHANTABILITY, NONINFRINGEMENT, OR FITNESS FOR A PARTICULAR PURPOSE. WE MAKE NO WARRANTY AND ASSUME NO RESPONSIBILITY FOR THE SAFETY OR STABILITY OF THE MATERIALS AND ANY OUTPUT THEREFROM.
|
| 41 |
+
c. IN NO EVENT SHALL WE BE LIABLE TO YOU FOR ANY DAMAGES, INCLUDING, BUT NOT LIMITED TO ANY DIRECT, OR INDIRECT, SPECIAL OR CONSEQUENTIAL DAMAGES ARISING FROM YOUR USE OR INABILITY TO USE THE MATERIALS OR ANY OUTPUT OF IT, NO MATTER HOW IT’S CAUSED.
|
| 42 |
+
d. You will defend, indemnify and hold harmless us from and against any claim by any third party arising out of or related to your use or distribution of the Materials.
|
| 43 |
+
|
| 44 |
+
7. Survival and Termination.
|
| 45 |
+
a. The term of this Agreement shall commence upon your acceptance of this Agreement or access to the Materials and will continue in full force and effect until terminated in accordance with the terms and conditions herein.
|
| 46 |
+
b. We may terminate this Agreement if you breach any of the terms or conditions of this Agreement. Upon termination of this Agreement, you must delete and cease use of the Materials. Sections 6 and 8 shall survive the termination of this Agreement.
|
| 47 |
+
|
| 48 |
+
8. Governing Law and Jurisdiction.
|
| 49 |
+
a. This Agreement and any dispute arising out of or relating to it will be governed by the laws of China, without regard to conflict of law principles, and the UN Convention on Contracts for the International Sale of Goods does not apply to this Agreement.
|
| 50 |
+
b. The People's Courts in Hangzhou City shall have exclusive jurisdiction over any dispute arising out of this Agreement.
|
| 51 |
+
|
| 52 |
+
9. Other Terms and Conditions.
|
| 53 |
+
a. Any arrangements, understandings, or agreements regarding the Material not stated herein are separate from and independent of the terms and conditions of this Agreement. You shall request a separate license from us, if you use the Materials in ways not expressly agreed to in this Agreement.
|
| 54 |
+
b. We shall not be bound by any additional or different terms or conditions communicated by you unless expressly agreed.
|
LICENSES/VIBETHINKER-LICENSE-NOTE.txt
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
VibeThinker-3B license metadata note
|
| 2 |
+
|
| 3 |
+
Upstream repository: WeiboAI/VibeThinker-3B
|
| 4 |
+
Pinned revision: 77bd2cced09193c8b9a59a32bd8577bbd1f3e01c
|
| 5 |
+
Pinned revision URL:
|
| 6 |
+
https://huggingface.co/WeiboAI/VibeThinker-3B/tree/77bd2cced09193c8b9a59a32bd8577bbd1f3e01c
|
| 7 |
+
|
| 8 |
+
The README model-card metadata at this revision declares:
|
| 9 |
+
|
| 10 |
+
license: mit
|
| 11 |
+
|
| 12 |
+
The pinned repository tree does not contain a standalone LICENSE file or
|
| 13 |
+
NOTICE file. It does not contain a standalone copyright notice that identifies
|
| 14 |
+
a holder and year.
|
| 15 |
+
|
| 16 |
+
This file records the upstream metadata and file inventory. It is not an
|
| 17 |
+
upstream license text. This release does not assign a VibeThinker copyright
|
| 18 |
+
holder or year.
|
| 19 |
+
|
| 20 |
+
The Qwen Research License Agreement used for this distribution is included
|
| 21 |
+
separately at LICENSES/QWEN-RESEARCH.txt.
|
NOTICE
ADDED
|
@@ -0,0 +1,25 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
VibeThinker-3B Jacobian Lens
|
| 2 |
+
|
| 3 |
+
This package contains a fitted average-Jacobian readout artifact derived from
|
| 4 |
+
activations of WeiboAI/VibeThinker-3B. The upstream model card identifies
|
| 5 |
+
Qwen/Qwen2.5-Coder-3B as its base model.
|
| 6 |
+
|
| 7 |
+
Qwen is licensed under the Qwen RESEARCH LICENSE AGREEMENT, Copyright (c)
|
| 8 |
+
Alibaba Cloud. All Rights Reserved.
|
| 9 |
+
|
| 10 |
+
Built with Qwen.
|
| 11 |
+
|
| 12 |
+
The pinned VibeThinker model-card metadata declares `license: mit`. The pinned
|
| 13 |
+
repository tree contains no standalone LICENSE file, NOTICE file, or copyright
|
| 14 |
+
notice that identifies a holder and year. The metadata record and pinned
|
| 15 |
+
revision URL are in LICENSES/VIBETHINKER-LICENSE-NOTE.txt.
|
| 16 |
+
|
| 17 |
+
Modification notice: model.safetensors is a derived Jacobian lens tensor
|
| 18 |
+
artifact, not a copy of the upstream language-model weights. It was converted
|
| 19 |
+
losslessly from the fitted PyTorch serialization to safetensors; every FP16
|
| 20 |
+
tensor bit pattern was preserved. No VibeThinker or Qwen base-model weights are
|
| 21 |
+
included in this repository.
|
| 22 |
+
|
| 23 |
+
The lens was fit using Anthropic PBC's jacobian-lens reference implementation,
|
| 24 |
+
licensed under the Apache License 2.0. The fitting corpus was sourced from
|
| 25 |
+
WikiText-103; no source prompt text is redistributed here.
|
README.md
ADDED
|
@@ -0,0 +1,322 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
license: other
|
| 3 |
+
license_name: qwen-research-license
|
| 4 |
+
license_link: https://huggingface.co/JacobMolBio/vibethinker-3b-jlens-model/blob/main/LICENSES/QWEN-RESEARCH.txt
|
| 5 |
+
tags:
|
| 6 |
+
- vibethinker-3b
|
| 7 |
+
- jacobian-lens
|
| 8 |
+
- mechanistic-interpretability
|
| 9 |
+
- interpretability
|
| 10 |
+
- qwen2
|
| 11 |
+
- safetensors
|
| 12 |
+
---
|
| 13 |
+
|
| 14 |
+

|
| 15 |
+
|
| 16 |
+
# VibeThinker-3B J Lens
|
| 17 |
+
|
| 18 |
+
The J Lens is a Jacobian lens fitted to
|
| 19 |
+
[`WeiboAI/VibeThinker-3B`](https://huggingface.co/WeiboAI/VibeThinker-3B/tree/77bd2cced09193c8b9a59a32bd8577bbd1f3e01c).
|
| 20 |
+
Pick one of the 18 fitted source layers and a token position, and the lens
|
| 21 |
+
decodes that residual-stream activation into a ranked list of vocabulary
|
| 22 |
+
tokens. Following one position across layers shows how the decoded ranking
|
| 23 |
+
changes on the way to the model's final output.
|
| 24 |
+
|
| 25 |
+
The lens contains 18 matrices, one for each even-numbered source layer from 0
|
| 26 |
+
through 34. Each matrix maps its source-layer residual-stream activation into
|
| 27 |
+
layer-35 coordinates. VibeThinker-3B's final normalization and vocabulary
|
| 28 |
+
projection produce the ranked token scores.
|
| 29 |
+
|
| 30 |
+
This repository contains the fitted lens in two Safetensors precisions.
|
| 31 |
+
`model.safetensors` is the FP16 lens used to capture the released traces.
|
| 32 |
+
`evaluation.safetensors` is the FP32 lens used for the recorded readout
|
| 33 |
+
evaluation. Casting each FP32 matrix to FP16 reproduces the FP16 file exactly;
|
| 34 |
+
both are included so every published result stays paired with the tensor
|
| 35 |
+
values that produced it.
|
| 36 |
+
|
| 37 |
+
The companion trace repository contains six saved traces. Each trace pairs a
|
| 38 |
+
prompt and its generated response with the top 12 decoded tokens at each
|
| 39 |
+
captured position for the 18 source layers and the final model layer.
|
| 40 |
+
|
| 41 |
+
## Choose a path
|
| 42 |
+
|
| 43 |
+
| Goal | Use |
|
| 44 |
+
|---|---|
|
| 45 |
+
| Browse saved traces | Open the [static viewer](https://jvogan.github.io/vibethinker-3b-jlens). It reads released trace JSON in the browser; it does not download model weights or run inference. |
|
| 46 |
+
| Run a new prompt locally | Use `scripts/render_slice.py` in the [source repository](https://github.com/jvogan/vibethinker-3b-jlens). It uses the pinned VibeThinker-3B base-model weights and `model.safetensors`, the FP16 J Lens. |
|
| 47 |
+
| Load the released lens in Python | Use `model.safetensors`, the FP16 trace lens. |
|
| 48 |
+
| Recalculate the recorded readout statistics | Use `evaluation.safetensors`, the FP32 evaluation lens, with the released rank rows and `scripts/recalculate_readout.py` in the trace repository. |
|
| 49 |
+
|
| 50 |
+
## Companion repositories
|
| 51 |
+
|
| 52 |
+
| Release component | Location |
|
| 53 |
+
|---|---|
|
| 54 |
+
| Source code and Pages source | [https://github.com/jvogan/vibethinker-3b-jlens](https://github.com/jvogan/vibethinker-3b-jlens) |
|
| 55 |
+
| Captured trace dataset | [https://huggingface.co/datasets/JacobMolBio/vibethinker-3b-jlens-traces](https://huggingface.co/datasets/JacobMolBio/vibethinker-3b-jlens-traces) |
|
| 56 |
+
| Static Pages site | [https://jvogan.github.io/vibethinker-3b-jlens](https://jvogan.github.io/vibethinker-3b-jlens) |
|
| 57 |
+
| Hugging Face model repository ID | [JacobMolBio/vibethinker-3b-jlens-model](https://huggingface.co/JacobMolBio/vibethinker-3b-jlens-model) |
|
| 58 |
+
|
| 59 |
+
## Artifact identity
|
| 60 |
+
|
| 61 |
+
| Field | Trace artifact | Evaluation artifact |
|
| 62 |
+
|---|---|---|
|
| 63 |
+
| File | `model.safetensors` | `evaluation.safetensors` |
|
| 64 |
+
| SHA-256 | `089d776979408f23e5377539c15aa8025d171633718ccdce709bcd3372e7942c` | `0cc184eb65d273ac8bfee5450a141c8cdf8dd6ec8caa68d7d47260ef621777d1` |
|
| 65 |
+
| Size | 150,996,824 bytes | 301,991,904 bytes |
|
| 66 |
+
| Dtype | FP16 | FP32 |
|
| 67 |
+
| Role | Captured trace readouts | Recorded readout evaluation |
|
| 68 |
+
|
| 69 |
+
[](assets/two-lens-files.svg)
|
| 70 |
+
|
| 71 |
+
Both artifacts contain 18 matrices with shape 2,048 × 2,048. Their source
|
| 72 |
+
layers are every even layer from 0 through 34, and their target layer is 35.
|
| 73 |
+
The lens was fit on 1,000 prompts for model revision
|
| 74 |
+
`77bd2cced09193c8b9a59a32bd8577bbd1f3e01c`. The revision binding was
|
| 75 |
+
reconstructed from the Hub head after the fit;
|
| 76 |
+
[`provenance.json`](provenance.json) records that binding and its limits.
|
| 77 |
+
|
| 78 |
+
## Files
|
| 79 |
+
|
| 80 |
+
- `model.safetensors` — the 18 losslessly reserialized FP16 lens matrices.
|
| 81 |
+
- `evaluation.safetensors` — the 18 losslessly reserialized FP32 evaluation matrices.
|
| 82 |
+
- `lens_config.json` — model binding, layer layout, fit parameters, and artifact ID.
|
| 83 |
+
- `tensor_manifest.json` — shape, dtype, size, and raw-byte SHA-256 for every tensor.
|
| 84 |
+
- `evaluation_tensor_manifest.json` — the corresponding FP32 tensor manifest.
|
| 85 |
+
- `evaluation_provenance.json` — derivation and source-hash bindings.
|
| 86 |
+
- `evaluation_compatibility.json` — the FP32-to-FP16 comparison for every layer.
|
| 87 |
+
- `evaluation_validation.json` — the exact FP32 conversion checks.
|
| 88 |
+
- `provenance.json` — fit, corpus, software, and conversion provenance.
|
| 89 |
+
- `validation.json` — exhaustive source-to-safetensors comparison result.
|
| 90 |
+
- `evaluation.json` — recorded readout-only evaluation and its scope.
|
| 91 |
+
- `SHA256SUMS` — checksums for both Safetensors files.
|
| 92 |
+
- `requirements.txt` — pinned packages for loading and validation.
|
| 93 |
+
- `scripts/validate_artifact.py` — validates the artifact without pickle.
|
| 94 |
+
- `scripts/convert_checkpoint.py` — reproducible checkpoint conversion for a user-provided source file.
|
| 95 |
+
- `scripts/convert_evaluation_checkpoint.py` — reproducible FP32 conversion.
|
| 96 |
+
- `LICENSES/QWEN-RESEARCH.txt` — complete distribution terms for the artifacts.
|
| 97 |
+
- `NOTICE` and `THIRD_PARTY_NOTICES.md` — required attribution and license context.
|
| 98 |
+
|
| 99 |
+
## Tensor layout
|
| 100 |
+
|
| 101 |
+
The keys are `J.<source_layer>`:
|
| 102 |
+
|
| 103 |
+
```text
|
| 104 |
+
J.0, J.2, J.4, J.6, J.8, J.10, J.12, J.14, J.16,
|
| 105 |
+
J.18, J.20, J.22, J.24, J.26, J.28, J.30, J.32, J.34
|
| 106 |
+
```
|
| 107 |
+
|
| 108 |
+
Each matrix maps the post-transformer-block residual at its source layer into
|
| 109 |
+
the layer-35 residual basis. With row-major batches of residual vectors, the
|
| 110 |
+
transport used by the reference implementation is:
|
| 111 |
+
|
| 112 |
+
```python
|
| 113 |
+
transported = residual.float() @ J.float().T
|
| 114 |
+
logits = unembed(transported)
|
| 115 |
+
```
|
| 116 |
+
|
| 117 |
+
Both Safetensors headers contain the model ID and revision, source and target
|
| 118 |
+
layers, width, prompt count, source-checkpoint hash, and key pattern.
|
| 119 |
+
|
| 120 |
+
`requirements.txt` pins the runtime used to create and validate
|
| 121 |
+
`evaluation.safetensors`. `provenance.json` records the earlier runtime used
|
| 122 |
+
to convert `model.safetensors`.
|
| 123 |
+
|
| 124 |
+
## Install and load
|
| 125 |
+
|
| 126 |
+
The `jlens` Python package comes from the companion source repository. This
|
| 127 |
+
repository contains the fitted tensors and the dependencies needed to inspect
|
| 128 |
+
and validate them.
|
| 129 |
+
|
| 130 |
+
From this repository checkout, install both parts:
|
| 131 |
+
|
| 132 |
+
```bash
|
| 133 |
+
export JLENS_CODE_REPO_URL="https://github.com/jvogan/vibethinker-3b-jlens"
|
| 134 |
+
git clone "$JLENS_CODE_REPO_URL" ../vibethinker-3b-jlens
|
| 135 |
+
python3 -m pip install -r requirements.txt
|
| 136 |
+
python3 -m pip install ../vibethinker-3b-jlens
|
| 137 |
+
```
|
| 138 |
+
|
| 139 |
+
Load the local artifact:
|
| 140 |
+
|
| 141 |
+
```python
|
| 142 |
+
from jlens import JacobianLens
|
| 143 |
+
|
| 144 |
+
lens = JacobianLens.load("model.safetensors")
|
| 145 |
+
print(lens.source_layers)
|
| 146 |
+
```
|
| 147 |
+
|
| 148 |
+
Or load an immutable Hugging Face revision, setting `JLENS_MODEL_REVISION`
|
| 149 |
+
to the model-repository commit recorded in the source repository's
|
| 150 |
+
`release-manifest.json`:
|
| 151 |
+
|
| 152 |
+
```bash
|
| 153 |
+
export JLENS_MODEL_REPO_ID="JacobMolBio/vibethinker-3b-jlens-model"
|
| 154 |
+
export JLENS_MODEL_REVISION="<commit from the source repository's release-manifest.json>"
|
| 155 |
+
```
|
| 156 |
+
|
| 157 |
+
```python
|
| 158 |
+
import os
|
| 159 |
+
|
| 160 |
+
from jlens import JacobianLens
|
| 161 |
+
|
| 162 |
+
lens = JacobianLens.from_pretrained(
|
| 163 |
+
os.environ["JLENS_MODEL_REPO_ID"],
|
| 164 |
+
filename="model.safetensors",
|
| 165 |
+
revision=os.environ["JLENS_MODEL_REVISION"],
|
| 166 |
+
)
|
| 167 |
+
```
|
| 168 |
+
|
| 169 |
+
To inspect tensors without the companion package, install `requirements.txt`
|
| 170 |
+
and use Safetensors directly:
|
| 171 |
+
|
| 172 |
+
```python
|
| 173 |
+
from safetensors import safe_open
|
| 174 |
+
|
| 175 |
+
with safe_open("model.safetensors", framework="pt", device="cpu") as handle:
|
| 176 |
+
metadata = handle.metadata()
|
| 177 |
+
jacobians = {
|
| 178 |
+
int(key.split(".", 1)[1]): handle.get_tensor(key).clone()
|
| 179 |
+
for key in handle.keys()
|
| 180 |
+
}
|
| 181 |
+
|
| 182 |
+
assert metadata["model_revision"] == (
|
| 183 |
+
"77bd2cced09193c8b9a59a32bd8577bbd1f3e01c"
|
| 184 |
+
)
|
| 185 |
+
assert sorted(jacobians) == list(range(0, 36, 2))
|
| 186 |
+
```
|
| 187 |
+
|
| 188 |
+
Load VibeThinker itself from the upstream repository at the exact bound
|
| 189 |
+
revision; this repository does not include VibeThinker or Qwen weights:
|
| 190 |
+
|
| 191 |
+
```python
|
| 192 |
+
from transformers import AutoModelForCausalLM, AutoTokenizer
|
| 193 |
+
|
| 194 |
+
model_id = "WeiboAI/VibeThinker-3B"
|
| 195 |
+
revision = "77bd2cced09193c8b9a59a32bd8577bbd1f3e01c"
|
| 196 |
+
tokenizer = AutoTokenizer.from_pretrained(model_id, revision=revision)
|
| 197 |
+
model = AutoModelForCausalLM.from_pretrained(
|
| 198 |
+
model_id,
|
| 199 |
+
revision=revision,
|
| 200 |
+
dtype="auto",
|
| 201 |
+
)
|
| 202 |
+
```
|
| 203 |
+
|
| 204 |
+
Review and accept the upstream terms before downloading or using the model.
|
| 205 |
+
|
| 206 |
+
## Viewer and new-prompt compute
|
| 207 |
+
|
| 208 |
+
The companion Pages site is a static viewer for captured trace JSON. It
|
| 209 |
+
parses, filters, and renders those records in the browser; loading a local
|
| 210 |
+
JSON file uses the browser's file reader and does not upload the file. The
|
| 211 |
+
static site cannot analyze a new prompt. That requires the companion source
|
| 212 |
+
package, this lens artifact, the pinned VibeThinker weights, and local Python
|
| 213 |
+
compute.
|
| 214 |
+
|
| 215 |
+
## Fit method
|
| 216 |
+
|
| 217 |
+
The lens was fit with the Apache-2.0 Anthropic Jacobian Lens reference
|
| 218 |
+
implementation at commit
|
| 219 |
+
[`581d398613e5602a5af361e1c34d3a92ea82ba8e`](https://github.com/anthropics/jacobian-lens/tree/581d398613e5602a5af361e1c34d3a92ea82ba8e).
|
| 220 |
+
For each fitted source layer, the estimator sums cotangents over valid causal
|
| 221 |
+
targets at or after a source position, averages over valid source positions,
|
| 222 |
+
and gives each prompt equal weight.
|
| 223 |
+
|
| 224 |
+
- Hook point: post-transformer-block output residual
|
| 225 |
+
- Source layers: every even layer from 0 through 34
|
| 226 |
+
- Target layer: 35
|
| 227 |
+
- Sequence cap: 128 tokens
|
| 228 |
+
- Leading positions skipped: 16
|
| 229 |
+
- Final position excluded from fitting
|
| 230 |
+
- Dimension batch: 8
|
| 231 |
+
- Fit runtime dtype: BF16
|
| 232 |
+
- Trace artifact dtype: FP16
|
| 233 |
+
- Evaluation artifact dtype: FP32
|
| 234 |
+
- Fit manifest: 1,000 examples from WikiText-103 raw train, minimum 600
|
| 235 |
+
characters
|
| 236 |
+
|
| 237 |
+
[`provenance.json`](provenance.json) records the fit manifest, a later local
|
| 238 |
+
rematerialization of the prompt set, and the hash checks that bind them.
|
| 239 |
+
Prompt text is not included in this repository.
|
| 240 |
+
|
| 241 |
+
## Validation
|
| 242 |
+
|
| 243 |
+
Validation checked the following for every tensor in both artifacts:
|
| 244 |
+
|
| 245 |
+
- exact key set;
|
| 246 |
+
- exact shape and each artifact's declared FP16 or FP32 dtype;
|
| 247 |
+
- exact numerical equality with the recorded source checkpoint;
|
| 248 |
+
- exact FP16 or FP32 bit-pattern equality;
|
| 249 |
+
- exact raw tensor-byte SHA-256;
|
| 250 |
+
- contiguity and full finiteness; and
|
| 251 |
+
- a Safetensors header free of private source paths.
|
| 252 |
+
|
| 253 |
+
All 18 tensors in each artifact passed, and zero values changed. The
|
| 254 |
+
FP32-to-FP16 cast also matches `model.safetensors` exactly for all 18 layers.
|
| 255 |
+
Recheck both artifacts with:
|
| 256 |
+
|
| 257 |
+
```bash
|
| 258 |
+
python3 scripts/validate_artifact.py
|
| 259 |
+
```
|
| 260 |
+
|
| 261 |
+
The validator verifies both file checksums, cross-file model and artifact
|
| 262 |
+
bindings, header metadata, tensor shapes and hashes, license and notice files,
|
| 263 |
+
and FP32-to-FP16 compatibility.
|
| 264 |
+
|
| 265 |
+
## Evaluation
|
| 266 |
+
|
| 267 |
+
The recorded evaluation measures ranked-token readouts: for each prompt in a
|
| 268 |
+
551-item suite, does a known target term rank highly in the decoded tokens?
|
| 269 |
+
The paired token-target and shuffled-layer mapping checks passed for the
|
| 270 |
+
selected band at layers 24, 26, 28, 30, 32, and 34. The comparison with the
|
| 271 |
+
ordinary logit lens did not establish improvement.
|
| 272 |
+
[`evaluation.json`](evaluation.json) records this readout-only scope as
|
| 273 |
+
`validated_readout_only`.
|
| 274 |
+
|
| 275 |
+
The 551 items cover association, multihop, multilingual,
|
| 276 |
+
order-of-operations, poetry, and typo prompts, each pairing a prompt with one
|
| 277 |
+
or more target terms. A deterministic split assigned 377 eligible items to
|
| 278 |
+
the test set; the band was selected on development items. The companion trace
|
| 279 |
+
repository contains all evaluation inputs, the 50,050 readout rows, the
|
| 280 |
+
aggregate metrics, the bootstrap intervals, and the exact method, and its
|
| 281 |
+
`scripts/recalculate_readout.py` recomputes every recorded statistic from the
|
| 282 |
+
released rows. Regenerating the rows themselves requires the pinned base
|
| 283 |
+
model, `evaluation.safetensors`, and a separate evaluator implementation.
|
| 284 |
+
|
| 285 |
+
The recorded metrics bind to `evaluation.safetensors`
|
| 286 |
+
(SHA-256 `0cc184eb65d273ac8bfee5450a141c8cdf8dd6ec8caa68d7d47260ef621777d1`,
|
| 287 |
+
from source FP32 checkpoint SHA-256
|
| 288 |
+
`8f752032a26a5196c1cb447ef63f01e8a29820ff0c57178dd80c6e26d32b12a9`);
|
| 289 |
+
they do not evaluate `model.safetensors`. The evaluation does not measure
|
| 290 |
+
general model accuracy or test causal steering or free-generation behavior.
|
| 291 |
+
|
| 292 |
+
The lens is stride-2, was fit on 1,000 prompts, and is model-revision
|
| 293 |
+
specific. Single-cell readouts can be noisy; compare patterns across nearby
|
| 294 |
+
layers and positions before interpreting an isolated token.
|
| 295 |
+
|
| 296 |
+
## License and attribution
|
| 297 |
+
|
| 298 |
+
The Qwen Research License applies to this distribution and limits use to
|
| 299 |
+
non-commercial research and evaluation. Commercial use requires a separate
|
| 300 |
+
license from the upstream rights holder. Redistribution must include the
|
| 301 |
+
complete [Qwen Research License](LICENSES/QWEN-RESEARCH.txt) and retain the
|
| 302 |
+
required notice in [`NOTICE`](NOTICE). Built with Qwen. No VibeThinker or
|
| 303 |
+
Qwen base-model weights are redistributed here.
|
| 304 |
+
|
| 305 |
+
The pinned VibeThinker model-card metadata declares `license: mit` and names
|
| 306 |
+
`Qwen/Qwen2.5-Coder-3B` as its base model; the pinned tree contains no
|
| 307 |
+
standalone license file.
|
| 308 |
+
[`LICENSES/VIBETHINKER-LICENSE-NOTE.txt`](LICENSES/VIBETHINKER-LICENSE-NOTE.txt)
|
| 309 |
+
records that metadata and the pinned revision URL.
|
| 310 |
+
|
| 311 |
+
The fitting implementation is Anthropic's Apache-2.0
|
| 312 |
+
[`jacobian-lens`](https://github.com/anthropics/jacobian-lens). WikiText-103
|
| 313 |
+
is available under CC BY-SA 4.0; no source text is redistributed here. See
|
| 314 |
+
[`THIRD_PARTY_NOTICES.md`](THIRD_PARTY_NOTICES.md) for details and links.
|
| 315 |
+
|
| 316 |
+
## References
|
| 317 |
+
|
| 318 |
+
- Anthropic, [*Verbalizable Representations Form a Global Workspace in Language Models*](https://transformer-circuits.pub/2026/workspace/index.html)
|
| 319 |
+
- Anthropic, [`jacobian-lens`](https://github.com/anthropics/jacobian-lens)
|
| 320 |
+
- [`WeiboAI/VibeThinker-3B`](https://huggingface.co/WeiboAI/VibeThinker-3B/tree/77bd2cced09193c8b9a59a32bd8577bbd1f3e01c)
|
| 321 |
+
- [`Qwen/Qwen2.5-Coder-3B`](https://huggingface.co/Qwen/Qwen2.5-Coder-3B)
|
| 322 |
+
- [`Salesforce/wikitext`](https://huggingface.co/datasets/Salesforce/wikitext)
|
SHA256SUMS
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
|
|
|
|
|
|
|
| 1 |
+
089d776979408f23e5377539c15aa8025d171633718ccdce709bcd3372e7942c model.safetensors
|
| 2 |
+
0cc184eb65d273ac8bfee5450a141c8cdf8dd6ec8caa68d7d47260ef621777d1 evaluation.safetensors
|
THIRD_PARTY_NOTICES.md
ADDED
|
@@ -0,0 +1,53 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Third-party notices
|
| 2 |
+
|
| 3 |
+
This document records upstream materials associated with the artifact. It does
|
| 4 |
+
not replace their license texts or expand the rights they grant.
|
| 5 |
+
|
| 6 |
+
## VibeThinker-3B and Qwen2.5-Coder-3B
|
| 7 |
+
|
| 8 |
+
The lens was fit from activations of
|
| 9 |
+
[`WeiboAI/VibeThinker-3B`](https://huggingface.co/WeiboAI/VibeThinker-3B/tree/77bd2cced09193c8b9a59a32bd8577bbd1f3e01c), bound
|
| 10 |
+
for reproducibility to revision
|
| 11 |
+
`77bd2cced09193c8b9a59a32bd8577bbd1f3e01c`. The VibeThinker card identifies
|
| 12 |
+
[`Qwen/Qwen2.5-Coder-3B`](https://huggingface.co/Qwen/Qwen2.5-Coder-3B) as its
|
| 13 |
+
base model.
|
| 14 |
+
|
| 15 |
+
The pinned VibeThinker model-card metadata declares `license: mit`. The pinned
|
| 16 |
+
tree contains no standalone LICENSE file, NOTICE file, or copyright notice
|
| 17 |
+
that identifies a holder and year.
|
| 18 |
+
[`LICENSES/VIBETHINKER-LICENSE-NOTE.txt`](LICENSES/VIBETHINKER-LICENSE-NOTE.txt)
|
| 19 |
+
records the declaration, pinned revision, and file-availability boundary. It
|
| 20 |
+
is not an upstream license text.
|
| 21 |
+
|
| 22 |
+
The Qwen base model is subject to the Qwen Research License. This repository
|
| 23 |
+
therefore uses `license: other` and applies the non-commercial
|
| 24 |
+
research/evaluation restriction. The complete Qwen terms are included at
|
| 25 |
+
[`LICENSES/QWEN-RESEARCH.txt`](LICENSES/QWEN-RESEARCH.txt).
|
| 26 |
+
|
| 27 |
+
Required notice:
|
| 28 |
+
|
| 29 |
+
> Qwen is licensed under the Qwen RESEARCH LICENSE AGREEMENT, Copyright (c)
|
| 30 |
+
> Alibaba Cloud. All Rights Reserved.
|
| 31 |
+
|
| 32 |
+
Built with Qwen.
|
| 33 |
+
|
| 34 |
+
No upstream VibeThinker or Qwen model weights are included.
|
| 35 |
+
|
| 36 |
+
## Anthropic Jacobian Lens
|
| 37 |
+
|
| 38 |
+
The artifact was fit using Anthropic PBC's
|
| 39 |
+
[`jacobian-lens`](https://github.com/anthropics/jacobian-lens) reference
|
| 40 |
+
implementation at commit
|
| 41 |
+
`581d398613e5602a5af361e1c34d3a92ea82ba8e`.
|
| 42 |
+
|
| 43 |
+
Copyright 2026 Anthropic PBC. Licensed under the Apache License, Version 2.0.
|
| 44 |
+
The full license is included at
|
| 45 |
+
[`LICENSES/APACHE-2.0.txt`](LICENSES/APACHE-2.0.txt).
|
| 46 |
+
|
| 47 |
+
## WikiText-103
|
| 48 |
+
|
| 49 |
+
The fitting corpus was selected from the WikiText-103 raw train split in
|
| 50 |
+
[`Salesforce/wikitext`](https://huggingface.co/datasets/Salesforce/wikitext).
|
| 51 |
+
The dataset card states that WikiText is available under the Creative Commons
|
| 52 |
+
Attribution-ShareAlike 4.0 license. This repository records only corpus
|
| 53 |
+
provenance and cryptographic digests; it does not redistribute the prompt text.
|
assets/jlens-model-banner.png
ADDED
|
Git LFS Details
|
assets/two-lens-files.png
ADDED
|
Git LFS Details
|
assets/two-lens-files.svg
ADDED
|
|
evaluation.json
ADDED
|
@@ -0,0 +1,144 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"artifact_kind": "frozen_vibethinker_v1_readout_reference",
|
| 3 |
+
"claim_boundary": "This reference records the V1 readout results. It excludes causal-assay results and does not establish free-generation steering or a global workspace.",
|
| 4 |
+
"classification": "validated_readout_only",
|
| 5 |
+
"coverage": {
|
| 6 |
+
"locked_test_items": 377,
|
| 7 |
+
"readout_items_completed": 539,
|
| 8 |
+
"readout_items_no_eligible_targets": 12,
|
| 9 |
+
"released_prompts": 551
|
| 10 |
+
},
|
| 11 |
+
"coverage_scope": "frozen_readout_source_run_not_the_100_prompt_ui_test_pack",
|
| 12 |
+
"task_definition": {
|
| 13 |
+
"aggregate_mean_reciprocal_rank": "mean_across_eligible_target_terms_of_reciprocal_best_rank",
|
| 14 |
+
"eligible_target": "at_least_one_candidate_form_tokenizes_to_exactly_one_token",
|
| 15 |
+
"final_model": "rank_target_terms_in_the_model_next_token_logits_at_the_score_position",
|
| 16 |
+
"item": "one_prompt_with_one_or_more_target_terms",
|
| 17 |
+
"layer_scope_reduction": "best_target_rank_across_layers_in_the_reported_scope",
|
| 18 |
+
"no_eligible_target_item": "none_of_the_item_target_terms_has_an_eligible_single_token_form",
|
| 19 |
+
"paired_bootstrap_mean_reciprocal_rank": "within_item_mean_of_reciprocal_best_rank_across_eligible_target_terms",
|
| 20 |
+
"pass_at_k": "mean_across_items_of_the_fraction_of_item_target_terms_with_best_rank_at_most_k",
|
| 21 |
+
"score_position": {
|
| 22 |
+
"default": "final_prompt_token",
|
| 23 |
+
"poetry": "last_newline_token"
|
| 24 |
+
},
|
| 25 |
+
"split": {
|
| 26 |
+
"dev_fraction": 0.3,
|
| 27 |
+
"method": "sha256_stable_split",
|
| 28 |
+
"seed": "vibethinker-jlens-v1"
|
| 29 |
+
},
|
| 30 |
+
"suites": ["association", "multihop", "multilingual", "order-ops", "poetry", "typo"],
|
| 31 |
+
"target_candidate_forms": [
|
| 32 |
+
"original_lowercase_and_capitalized_forms_with_and_without_leading_space",
|
| 33 |
+
"order_ops_also_adds_configured_operation_synonyms_and_number_word_digit_forms"
|
| 34 |
+
],
|
| 35 |
+
"target_rank": "best_one_based_vocabulary_rank_among_eligible_target_token_ids"
|
| 36 |
+
},
|
| 37 |
+
"evidence": {
|
| 38 |
+
"incremental_over_ordinary_logit_lens": false,
|
| 39 |
+
"layer_mapping_specific_vs_shuffled_jacobian": true,
|
| 40 |
+
"paired_bootstrap": {
|
| 41 |
+
"confidence": 0.95,
|
| 42 |
+
"input": "paired_item_metric_differences",
|
| 43 |
+
"interval": "percentile",
|
| 44 |
+
"metrics": ["pass@10", "mean_reciprocal_rank"],
|
| 45 |
+
"samples": 2000,
|
| 46 |
+
"scope": {
|
| 47 |
+
"kind": "selected_band",
|
| 48 |
+
"source_layers": [24, 26, 28, 30, 32, 34]
|
| 49 |
+
},
|
| 50 |
+
"split": "test"
|
| 51 |
+
},
|
| 52 |
+
"paired_bootstrap_decisions": {
|
| 53 |
+
"incremental_over_ordinary_logit_lens": {
|
| 54 |
+
"comparison": "jlens_band_minus_logit_lens_band",
|
| 55 |
+
"criterion": "at_least_one_lower_bound_greater_than_zero",
|
| 56 |
+
"met": false
|
| 57 |
+
},
|
| 58 |
+
"layer_mapping_specific_vs_shuffled_jacobian": {
|
| 59 |
+
"comparison": "jlens_band_minus_shuffled_layer_band",
|
| 60 |
+
"criterion": "both_lower_bounds_greater_than_zero",
|
| 61 |
+
"met": true
|
| 62 |
+
},
|
| 63 |
+
"token_specific_vs_shuffled_target": {
|
| 64 |
+
"comparison": "jlens_band_minus_shuffled_token_band",
|
| 65 |
+
"criterion": "both_lower_bounds_greater_than_zero",
|
| 66 |
+
"met": true
|
| 67 |
+
}
|
| 68 |
+
},
|
| 69 |
+
"specificity_scope": {
|
| 70 |
+
"kind": "selected_band",
|
| 71 |
+
"source_layers": [24, 26, 28, 30, 32, 34]
|
| 72 |
+
},
|
| 73 |
+
"token_specific_vs_shuffled_target": true,
|
| 74 |
+
"validated_readout_signal": true
|
| 75 |
+
},
|
| 76 |
+
"lens_sha256": "8f752032a26a5196c1cb447ef63f01e8a29820ff0c57178dd80c6e26d32b12a9",
|
| 77 |
+
"lens_variant": "fp32_evaluation_safetensors_included",
|
| 78 |
+
"released_lens_artifact": {
|
| 79 |
+
"filename": "evaluation.safetensors",
|
| 80 |
+
"format": "safetensors",
|
| 81 |
+
"sha256": "0cc184eb65d273ac8bfee5450a141c8cdf8dd6ec8caa68d7d47260ef621777d1",
|
| 82 |
+
"size_bytes": 301991904,
|
| 83 |
+
"source_checkpoint_sha256": "8f752032a26a5196c1cb447ef63f01e8a29820ff0c57178dd80c6e26d32b12a9",
|
| 84 |
+
"tensor_conversion": "lossless_fp32_reserialization"
|
| 85 |
+
},
|
| 86 |
+
"source_artifact_availability": {
|
| 87 |
+
"fp32_compatibility_check_output_included": true,
|
| 88 |
+
"fp32_compatibility_check_output_path": "evaluation_compatibility.json",
|
| 89 |
+
"fp32_derivation_record_included": true,
|
| 90 |
+
"fp32_derivation_record_path": "evaluation_provenance.json",
|
| 91 |
+
"fp32_evaluation_lens_included": true,
|
| 92 |
+
"fp32_evaluation_lens_path": "evaluation.safetensors",
|
| 93 |
+
"item_level_evaluation_rows_included": false,
|
| 94 |
+
"item_level_evaluation_rows_location": "companion_trace_repository:data/evaluation-results/readout-trials.jsonl",
|
| 95 |
+
"paired_bootstrap_interval_bounds_included": false,
|
| 96 |
+
"paired_bootstrap_interval_bounds_location": "companion_trace_repository:data/evaluation-results/readout-bootstrap-intervals.json",
|
| 97 |
+
"source_evaluation_bundle_included": false,
|
| 98 |
+
"source_hashes_included": true,
|
| 99 |
+
"source_hashes_location": "evaluation_provenance.json_and_companion_trace_repository:data/readout-reference.json",
|
| 100 |
+
"release_content": "fp16_trace_lens_fp32_evaluation_lens_aggregate_reference_provenance_and_compatibility"
|
| 101 |
+
},
|
| 102 |
+
"metrics": {
|
| 103 |
+
"jlens_all_layers": {
|
| 104 |
+
"mean_reciprocal_rank": 0.027840488103301416,
|
| 105 |
+
"n_items": 377,
|
| 106 |
+
"pass_at_1": 0.009283819628647215,
|
| 107 |
+
"pass_at_10": 0.08819628647214854,
|
| 108 |
+
"pass_at_50": 0.1655614500442087
|
| 109 |
+
},
|
| 110 |
+
"jlens_selected_band": {
|
| 111 |
+
"mean_reciprocal_rank": 0.01888596779741093,
|
| 112 |
+
"n_items": 377,
|
| 113 |
+
"pass_at_1": 0.003978779840848806,
|
| 114 |
+
"pass_at_10": 0.0629973474801061,
|
| 115 |
+
"pass_at_50": 0.11914235190097258
|
| 116 |
+
},
|
| 117 |
+
"logit_lens_all_layers": {
|
| 118 |
+
"mean_reciprocal_rank": 0.027249274540932285,
|
| 119 |
+
"n_items": 377,
|
| 120 |
+
"pass_at_1": 0.007294429708222812,
|
| 121 |
+
"pass_at_10": 0.08377541998231654,
|
| 122 |
+
"pass_at_50": 0.16114058355437666
|
| 123 |
+
},
|
| 124 |
+
"shuffled_layer_all_layers": {
|
| 125 |
+
"mean_reciprocal_rank": 0.042073366910739256,
|
| 126 |
+
"n_items": 377,
|
| 127 |
+
"pass_at_1": 0.04509283819628647,
|
| 128 |
+
"pass_at_10": 0.08819628647214854,
|
| 129 |
+
"pass_at_50": 0.1823607427055703
|
| 130 |
+
},
|
| 131 |
+
"final_model": {
|
| 132 |
+
"mean_reciprocal_rank": 0.019635164837169684,
|
| 133 |
+
"n_items": 377,
|
| 134 |
+
"pass_at_1": 0.003978779840848806,
|
| 135 |
+
"pass_at_10": 0.0665340406719717,
|
| 136 |
+
"pass_at_50": 0.10587975243147656
|
| 137 |
+
}
|
| 138 |
+
},
|
| 139 |
+
"model": "WeiboAI/VibeThinker-3B",
|
| 140 |
+
"model_revision": "77bd2cced09193c8b9a59a32bd8577bbd1f3e01c",
|
| 141 |
+
"note": "The recorded readout metrics bind to evaluation.safetensors, the FP32 evaluation lens in this repository. They do not evaluate model.safetensors, the FP16 lens used for the captured traces.",
|
| 142 |
+
"selected_band": [24, 26, 28, 30, 32, 34],
|
| 143 |
+
"schema_version": 1
|
| 144 |
+
}
|
evaluation.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:0cc184eb65d273ac8bfee5450a141c8cdf8dd6ec8caa68d7d47260ef621777d1
|
| 3 |
+
size 301991904
|
evaluation_compatibility.json
ADDED
|
@@ -0,0 +1,105 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"all_layer_casts_match_exactly": true,
|
| 3 |
+
"artifact_kind": "fp32_to_fp16_lens_compatibility",
|
| 4 |
+
"conversion": "IEEE_FP32_to_FP16_round_to_nearest_even",
|
| 5 |
+
"fp16_artifact": "model.safetensors",
|
| 6 |
+
"fp16_artifact_sha256": "089d776979408f23e5377539c15aa8025d171633718ccdce709bcd3372e7942c",
|
| 7 |
+
"fp32_artifact": "evaluation.safetensors",
|
| 8 |
+
"fp32_artifact_sha256": "0cc184eb65d273ac8bfee5450a141c8cdf8dd6ec8caa68d7d47260ef621777d1",
|
| 9 |
+
"fp32_source_checkpoint_sha256": "8f752032a26a5196c1cb447ef63f01e8a29820ff0c57178dd80c6e26d32b12a9",
|
| 10 |
+
"max_absolute_error": 0.00048828125,
|
| 11 |
+
"per_layer": {
|
| 12 |
+
"0": {
|
| 13 |
+
"cast_matches_model_safetensors_exactly": true,
|
| 14 |
+
"max_absolute_error": 0.00037920475006103516,
|
| 15 |
+
"relative_frobenius_error": 0.000206997521647552
|
| 16 |
+
},
|
| 17 |
+
"10": {
|
| 18 |
+
"cast_matches_model_safetensors_exactly": true,
|
| 19 |
+
"max_absolute_error": 0.0002696514129638672,
|
| 20 |
+
"relative_frobenius_error": 0.00020761765345058065
|
| 21 |
+
},
|
| 22 |
+
"12": {
|
| 23 |
+
"cast_matches_model_safetensors_exactly": true,
|
| 24 |
+
"max_absolute_error": 0.00046837329864501953,
|
| 25 |
+
"relative_frobenius_error": 0.00020767621517046658
|
| 26 |
+
},
|
| 27 |
+
"14": {
|
| 28 |
+
"cast_matches_model_safetensors_exactly": true,
|
| 29 |
+
"max_absolute_error": 0.00048673152923583984,
|
| 30 |
+
"relative_frobenius_error": 0.00020743575797500317
|
| 31 |
+
},
|
| 32 |
+
"16": {
|
| 33 |
+
"cast_matches_model_safetensors_exactly": true,
|
| 34 |
+
"max_absolute_error": 0.0004863739013671875,
|
| 35 |
+
"relative_frobenius_error": 0.00020778469258608042
|
| 36 |
+
},
|
| 37 |
+
"18": {
|
| 38 |
+
"cast_matches_model_safetensors_exactly": true,
|
| 39 |
+
"max_absolute_error": 0.0004839897155761719,
|
| 40 |
+
"relative_frobenius_error": 0.00020779679841448286
|
| 41 |
+
},
|
| 42 |
+
"2": {
|
| 43 |
+
"cast_matches_model_safetensors_exactly": true,
|
| 44 |
+
"max_absolute_error": 0.0004121065139770508,
|
| 45 |
+
"relative_frobenius_error": 0.00020722711230071437
|
| 46 |
+
},
|
| 47 |
+
"20": {
|
| 48 |
+
"cast_matches_model_safetensors_exactly": true,
|
| 49 |
+
"max_absolute_error": 0.0004881620407104492,
|
| 50 |
+
"relative_frobenius_error": 0.00020744978906595194
|
| 51 |
+
},
|
| 52 |
+
"22": {
|
| 53 |
+
"cast_matches_model_safetensors_exactly": true,
|
| 54 |
+
"max_absolute_error": 0.0004878044128417969,
|
| 55 |
+
"relative_frobenius_error": 0.00020796356930850946
|
| 56 |
+
},
|
| 57 |
+
"24": {
|
| 58 |
+
"cast_matches_model_safetensors_exactly": true,
|
| 59 |
+
"max_absolute_error": 0.0004870891571044922,
|
| 60 |
+
"relative_frobenius_error": 0.00020680286914392338
|
| 61 |
+
},
|
| 62 |
+
"26": {
|
| 63 |
+
"cast_matches_model_safetensors_exactly": true,
|
| 64 |
+
"max_absolute_error": 0.0004875659942626953,
|
| 65 |
+
"relative_frobenius_error": 0.00020638797987813289
|
| 66 |
+
},
|
| 67 |
+
"28": {
|
| 68 |
+
"cast_matches_model_safetensors_exactly": true,
|
| 69 |
+
"max_absolute_error": 0.0004818439483642578,
|
| 70 |
+
"relative_frobenius_error": 0.00020315668925901246
|
| 71 |
+
},
|
| 72 |
+
"30": {
|
| 73 |
+
"cast_matches_model_safetensors_exactly": true,
|
| 74 |
+
"max_absolute_error": 0.00048804283142089844,
|
| 75 |
+
"relative_frobenius_error": 0.000219304047701117
|
| 76 |
+
},
|
| 77 |
+
"32": {
|
| 78 |
+
"cast_matches_model_safetensors_exactly": true,
|
| 79 |
+
"max_absolute_error": 0.00048828125,
|
| 80 |
+
"relative_frobenius_error": 0.0002275111331836927
|
| 81 |
+
},
|
| 82 |
+
"34": {
|
| 83 |
+
"cast_matches_model_safetensors_exactly": true,
|
| 84 |
+
"max_absolute_error": 0.00048804283142089844,
|
| 85 |
+
"relative_frobenius_error": 0.00025735070181156144
|
| 86 |
+
},
|
| 87 |
+
"4": {
|
| 88 |
+
"cast_matches_model_safetensors_exactly": true,
|
| 89 |
+
"max_absolute_error": 0.000270843505859375,
|
| 90 |
+
"relative_frobenius_error": 0.00020699111471641674
|
| 91 |
+
},
|
| 92 |
+
"6": {
|
| 93 |
+
"cast_matches_model_safetensors_exactly": true,
|
| 94 |
+
"max_absolute_error": 0.00024312734603881836,
|
| 95 |
+
"relative_frobenius_error": 0.00020716959804749024
|
| 96 |
+
},
|
| 97 |
+
"8": {
|
| 98 |
+
"cast_matches_model_safetensors_exactly": true,
|
| 99 |
+
"max_absolute_error": 0.00040781497955322266,
|
| 100 |
+
"relative_frobenius_error": 0.00020697079418794512
|
| 101 |
+
}
|
| 102 |
+
},
|
| 103 |
+
"relative_frobenius_error": 0.00020901276774552775,
|
| 104 |
+
"schema_version": 1
|
| 105 |
+
}
|
evaluation_provenance.json
ADDED
|
@@ -0,0 +1,90 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"artifact": {
|
| 3 |
+
"filename": "evaluation.safetensors",
|
| 4 |
+
"format": "safetensors",
|
| 5 |
+
"sha256": "0cc184eb65d273ac8bfee5450a141c8cdf8dd6ec8caa68d7d47260ef621777d1",
|
| 6 |
+
"size_bytes": 301991904,
|
| 7 |
+
"tensor_conversion": "lossless_fp32_reserialization"
|
| 8 |
+
},
|
| 9 |
+
"artifact_kind": "jacobian_lens_evaluation_fp32_provenance",
|
| 10 |
+
"compatibility": {
|
| 11 |
+
"all_layer_casts_match_exactly": true,
|
| 12 |
+
"file": "evaluation_compatibility.json",
|
| 13 |
+
"fp16_artifact": "model.safetensors",
|
| 14 |
+
"fp16_artifact_sha256": "089d776979408f23e5377539c15aa8025d171633718ccdce709bcd3372e7942c"
|
| 15 |
+
},
|
| 16 |
+
"derivation": {
|
| 17 |
+
"formula": "jacobian_sum[layer] / n_done",
|
| 18 |
+
"n_done": 1000,
|
| 19 |
+
"source_export_script_sha256": "d1d9e0b7afd1d69771907a839f62b30936995fdd63b17ea2ac80f724b3cdce17",
|
| 20 |
+
"source_fit_checkpoint_sha256": "a1236cfe5d04601575b3de150ffe50e3a67e755ede1197ecf74c205a31bdc258",
|
| 21 |
+
"source_matrix_stats_sha256": "a21fe7c1f661fa9b5455d89c0a415beae794012eab38f782b667b27fac942401",
|
| 22 |
+
"source_provenance_sha256": "b6e5764fa1580a142403a425ecd03cafe55d4cc36ca1e6b90da7fc19a35aad36",
|
| 23 |
+
"source_validation_sha256": "4aea71008a10ef2d043129f7767e5d387372fe1fd5022550b59017a44e0b965d"
|
| 24 |
+
},
|
| 25 |
+
"lens": {
|
| 26 |
+
"d_model": 2048,
|
| 27 |
+
"dtype": "float32",
|
| 28 |
+
"estimator": {
|
| 29 |
+
"dim_batch": 8,
|
| 30 |
+
"exclude_final_position": true,
|
| 31 |
+
"fit_dtype": "bfloat16",
|
| 32 |
+
"max_seq_len": 128,
|
| 33 |
+
"name": "causal_all_current_and_future_targets_mean_jacobian",
|
| 34 |
+
"prompt_aggregation": "equal_weight_mean_over_prompts",
|
| 35 |
+
"skip_first": 16,
|
| 36 |
+
"source_position_aggregation": "mean_over_valid_source_positions",
|
| 37 |
+
"target_position_aggregation": "sum_over_valid_targets_at_or_after_source"
|
| 38 |
+
},
|
| 39 |
+
"hook_convention": "post_transformer_block_output_residual",
|
| 40 |
+
"n_prompts": 1000,
|
| 41 |
+
"source_layers": [
|
| 42 |
+
0,
|
| 43 |
+
2,
|
| 44 |
+
4,
|
| 45 |
+
6,
|
| 46 |
+
8,
|
| 47 |
+
10,
|
| 48 |
+
12,
|
| 49 |
+
14,
|
| 50 |
+
16,
|
| 51 |
+
18,
|
| 52 |
+
20,
|
| 53 |
+
22,
|
| 54 |
+
24,
|
| 55 |
+
26,
|
| 56 |
+
28,
|
| 57 |
+
30,
|
| 58 |
+
32,
|
| 59 |
+
34
|
| 60 |
+
],
|
| 61 |
+
"target_layer": 35
|
| 62 |
+
},
|
| 63 |
+
"limitations": [
|
| 64 |
+
"The original fit did not store the resolved Hugging Face commit; the revision binding was reconstructed from the Hub head and its last-modified timestamp.",
|
| 65 |
+
"The recorded evaluation validates token readout and does not establish causal steering or a global workspace."
|
| 66 |
+
],
|
| 67 |
+
"model": {
|
| 68 |
+
"architecture": "Qwen2ForCausalLM",
|
| 69 |
+
"d_model": 2048,
|
| 70 |
+
"id": "WeiboAI/VibeThinker-3B",
|
| 71 |
+
"n_layers": 36,
|
| 72 |
+
"revision": "77bd2cced09193c8b9a59a32bd8577bbd1f3e01c",
|
| 73 |
+
"revision_binding": "inferred_hub_head_unchanged_since_before_fit",
|
| 74 |
+
"revision_last_modified": "2026-06-30T11:35:41+00:00",
|
| 75 |
+
"tied_embeddings": true
|
| 76 |
+
},
|
| 77 |
+
"schema_version": 1,
|
| 78 |
+
"software": {
|
| 79 |
+
"anthropic_jacobian_lens_commit": "581d398613e5602a5af361e1c34d3a92ea82ba8e",
|
| 80 |
+
"conversion_runtime": {
|
| 81 |
+
"python_implementation": "CPython",
|
| 82 |
+
"safetensors": "0.8.0",
|
| 83 |
+
"torch": "2.13.0"
|
| 84 |
+
}
|
| 85 |
+
},
|
| 86 |
+
"source_checkpoint": {
|
| 87 |
+
"format": "pytorch",
|
| 88 |
+
"sha256": "8f752032a26a5196c1cb447ef63f01e8a29820ff0c57178dd80c6e26d32b12a9"
|
| 89 |
+
}
|
| 90 |
+
}
|
evaluation_tensor_manifest.json
ADDED
|
@@ -0,0 +1,209 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"artifact": "evaluation.safetensors",
|
| 3 |
+
"artifact_sha256": "0cc184eb65d273ac8bfee5450a141c8cdf8dd6ec8caa68d7d47260ef621777d1",
|
| 4 |
+
"artifact_size_bytes": 301991904,
|
| 5 |
+
"schema_version": 1,
|
| 6 |
+
"source_checkpoint_sha256": "8f752032a26a5196c1cb447ef63f01e8a29820ff0c57178dd80c6e26d32b12a9",
|
| 7 |
+
"tensor_count": 18,
|
| 8 |
+
"tensor_storage_bytes": 301989888,
|
| 9 |
+
"tensors": {
|
| 10 |
+
"J.0": {
|
| 11 |
+
"dtype": "float32",
|
| 12 |
+
"nbytes": 16777216,
|
| 13 |
+
"numel": 4194304,
|
| 14 |
+
"sha256_c_contiguous_little_endian_bytes": "4a77bed8711fd26317619e7265da5c9ce73aadd2d4851800715c8c5ec6012c6e",
|
| 15 |
+
"shape": [
|
| 16 |
+
2048,
|
| 17 |
+
2048
|
| 18 |
+
],
|
| 19 |
+
"source_layer": 0
|
| 20 |
+
},
|
| 21 |
+
"J.10": {
|
| 22 |
+
"dtype": "float32",
|
| 23 |
+
"nbytes": 16777216,
|
| 24 |
+
"numel": 4194304,
|
| 25 |
+
"sha256_c_contiguous_little_endian_bytes": "e2856885c149fa8244e327369c3bb081cfe651e1a6958bda317f8691e2db3c3b",
|
| 26 |
+
"shape": [
|
| 27 |
+
2048,
|
| 28 |
+
2048
|
| 29 |
+
],
|
| 30 |
+
"source_layer": 10
|
| 31 |
+
},
|
| 32 |
+
"J.12": {
|
| 33 |
+
"dtype": "float32",
|
| 34 |
+
"nbytes": 16777216,
|
| 35 |
+
"numel": 4194304,
|
| 36 |
+
"sha256_c_contiguous_little_endian_bytes": "e6a50d9c6495acebeac769533bd6900ce94443fc977506f836881e5dcb0335a0",
|
| 37 |
+
"shape": [
|
| 38 |
+
2048,
|
| 39 |
+
2048
|
| 40 |
+
],
|
| 41 |
+
"source_layer": 12
|
| 42 |
+
},
|
| 43 |
+
"J.14": {
|
| 44 |
+
"dtype": "float32",
|
| 45 |
+
"nbytes": 16777216,
|
| 46 |
+
"numel": 4194304,
|
| 47 |
+
"sha256_c_contiguous_little_endian_bytes": "b27ade6bee39486cad9b0cdf091f0690528509b0f4936d67e30ce3dc6eeeb6b8",
|
| 48 |
+
"shape": [
|
| 49 |
+
2048,
|
| 50 |
+
2048
|
| 51 |
+
],
|
| 52 |
+
"source_layer": 14
|
| 53 |
+
},
|
| 54 |
+
"J.16": {
|
| 55 |
+
"dtype": "float32",
|
| 56 |
+
"nbytes": 16777216,
|
| 57 |
+
"numel": 4194304,
|
| 58 |
+
"sha256_c_contiguous_little_endian_bytes": "e80280578bd50abdb0c1610d453eb11495be80a8580fe8f490ff63d6ab1e2e1b",
|
| 59 |
+
"shape": [
|
| 60 |
+
2048,
|
| 61 |
+
2048
|
| 62 |
+
],
|
| 63 |
+
"source_layer": 16
|
| 64 |
+
},
|
| 65 |
+
"J.18": {
|
| 66 |
+
"dtype": "float32",
|
| 67 |
+
"nbytes": 16777216,
|
| 68 |
+
"numel": 4194304,
|
| 69 |
+
"sha256_c_contiguous_little_endian_bytes": "23a2a10e7961625232b257e0943ad10893a9e2e051bf65b9f3e57778e293397f",
|
| 70 |
+
"shape": [
|
| 71 |
+
2048,
|
| 72 |
+
2048
|
| 73 |
+
],
|
| 74 |
+
"source_layer": 18
|
| 75 |
+
},
|
| 76 |
+
"J.2": {
|
| 77 |
+
"dtype": "float32",
|
| 78 |
+
"nbytes": 16777216,
|
| 79 |
+
"numel": 4194304,
|
| 80 |
+
"sha256_c_contiguous_little_endian_bytes": "f90ad9dae2c7b8701cc5b6623ddb0a77cb072c5b95a8a45fff1eca8680e796e3",
|
| 81 |
+
"shape": [
|
| 82 |
+
2048,
|
| 83 |
+
2048
|
| 84 |
+
],
|
| 85 |
+
"source_layer": 2
|
| 86 |
+
},
|
| 87 |
+
"J.20": {
|
| 88 |
+
"dtype": "float32",
|
| 89 |
+
"nbytes": 16777216,
|
| 90 |
+
"numel": 4194304,
|
| 91 |
+
"sha256_c_contiguous_little_endian_bytes": "c927effa01101f88b8c06023db5b9f727416ffde7f40be1515d48fa8215389cf",
|
| 92 |
+
"shape": [
|
| 93 |
+
2048,
|
| 94 |
+
2048
|
| 95 |
+
],
|
| 96 |
+
"source_layer": 20
|
| 97 |
+
},
|
| 98 |
+
"J.22": {
|
| 99 |
+
"dtype": "float32",
|
| 100 |
+
"nbytes": 16777216,
|
| 101 |
+
"numel": 4194304,
|
| 102 |
+
"sha256_c_contiguous_little_endian_bytes": "bb319fcc2484b93bc7a7754e508e807cfcd5387474ad7fb97c786da57b22aeea",
|
| 103 |
+
"shape": [
|
| 104 |
+
2048,
|
| 105 |
+
2048
|
| 106 |
+
],
|
| 107 |
+
"source_layer": 22
|
| 108 |
+
},
|
| 109 |
+
"J.24": {
|
| 110 |
+
"dtype": "float32",
|
| 111 |
+
"nbytes": 16777216,
|
| 112 |
+
"numel": 4194304,
|
| 113 |
+
"sha256_c_contiguous_little_endian_bytes": "a34fb656e93ee78c27bb88b42895aa11ecfc525d790ef54b1a18f72b7cf583e7",
|
| 114 |
+
"shape": [
|
| 115 |
+
2048,
|
| 116 |
+
2048
|
| 117 |
+
],
|
| 118 |
+
"source_layer": 24
|
| 119 |
+
},
|
| 120 |
+
"J.26": {
|
| 121 |
+
"dtype": "float32",
|
| 122 |
+
"nbytes": 16777216,
|
| 123 |
+
"numel": 4194304,
|
| 124 |
+
"sha256_c_contiguous_little_endian_bytes": "1865fccec5a832ca6b0766324bd078d5ea12e2715c7d25664dfd56da8cd82b21",
|
| 125 |
+
"shape": [
|
| 126 |
+
2048,
|
| 127 |
+
2048
|
| 128 |
+
],
|
| 129 |
+
"source_layer": 26
|
| 130 |
+
},
|
| 131 |
+
"J.28": {
|
| 132 |
+
"dtype": "float32",
|
| 133 |
+
"nbytes": 16777216,
|
| 134 |
+
"numel": 4194304,
|
| 135 |
+
"sha256_c_contiguous_little_endian_bytes": "9022027a77af3525b4c602723fedbd3871ffc0b4cf812081cdf6739ffb0f778b",
|
| 136 |
+
"shape": [
|
| 137 |
+
2048,
|
| 138 |
+
2048
|
| 139 |
+
],
|
| 140 |
+
"source_layer": 28
|
| 141 |
+
},
|
| 142 |
+
"J.30": {
|
| 143 |
+
"dtype": "float32",
|
| 144 |
+
"nbytes": 16777216,
|
| 145 |
+
"numel": 4194304,
|
| 146 |
+
"sha256_c_contiguous_little_endian_bytes": "ed63bd90333f5feb4441f1887bef102e3c5b95776a6f9c3c88a657ba1a92845f",
|
| 147 |
+
"shape": [
|
| 148 |
+
2048,
|
| 149 |
+
2048
|
| 150 |
+
],
|
| 151 |
+
"source_layer": 30
|
| 152 |
+
},
|
| 153 |
+
"J.32": {
|
| 154 |
+
"dtype": "float32",
|
| 155 |
+
"nbytes": 16777216,
|
| 156 |
+
"numel": 4194304,
|
| 157 |
+
"sha256_c_contiguous_little_endian_bytes": "c16a4dc0bd1af68302094d81cc898fa7fa30dca7f6ca0397f765c2b470716852",
|
| 158 |
+
"shape": [
|
| 159 |
+
2048,
|
| 160 |
+
2048
|
| 161 |
+
],
|
| 162 |
+
"source_layer": 32
|
| 163 |
+
},
|
| 164 |
+
"J.34": {
|
| 165 |
+
"dtype": "float32",
|
| 166 |
+
"nbytes": 16777216,
|
| 167 |
+
"numel": 4194304,
|
| 168 |
+
"sha256_c_contiguous_little_endian_bytes": "92c6ab150ddf0ece4eeb3a9998d9a87a9d8fabbfdad35f15b18f27a80153ecc5",
|
| 169 |
+
"shape": [
|
| 170 |
+
2048,
|
| 171 |
+
2048
|
| 172 |
+
],
|
| 173 |
+
"source_layer": 34
|
| 174 |
+
},
|
| 175 |
+
"J.4": {
|
| 176 |
+
"dtype": "float32",
|
| 177 |
+
"nbytes": 16777216,
|
| 178 |
+
"numel": 4194304,
|
| 179 |
+
"sha256_c_contiguous_little_endian_bytes": "59beda69c9baf4392cc55731fda3e3527c79948651fe9f38a9ce21669fe2aefa",
|
| 180 |
+
"shape": [
|
| 181 |
+
2048,
|
| 182 |
+
2048
|
| 183 |
+
],
|
| 184 |
+
"source_layer": 4
|
| 185 |
+
},
|
| 186 |
+
"J.6": {
|
| 187 |
+
"dtype": "float32",
|
| 188 |
+
"nbytes": 16777216,
|
| 189 |
+
"numel": 4194304,
|
| 190 |
+
"sha256_c_contiguous_little_endian_bytes": "faca3b5a874af143d133a705ca92d8bcbb71fe862f17559464d11d9bf3a5015c",
|
| 191 |
+
"shape": [
|
| 192 |
+
2048,
|
| 193 |
+
2048
|
| 194 |
+
],
|
| 195 |
+
"source_layer": 6
|
| 196 |
+
},
|
| 197 |
+
"J.8": {
|
| 198 |
+
"dtype": "float32",
|
| 199 |
+
"nbytes": 16777216,
|
| 200 |
+
"numel": 4194304,
|
| 201 |
+
"sha256_c_contiguous_little_endian_bytes": "2e9e5f1d2773f627ed231a0c5a6697077761354574350a63ad82035540a1944b",
|
| 202 |
+
"shape": [
|
| 203 |
+
2048,
|
| 204 |
+
2048
|
| 205 |
+
],
|
| 206 |
+
"source_layer": 8
|
| 207 |
+
}
|
| 208 |
+
}
|
| 209 |
+
}
|
evaluation_validation.json
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"artifact": "evaluation.safetensors",
|
| 3 |
+
"artifact_sha256": "0cc184eb65d273ac8bfee5450a141c8cdf8dd6ec8caa68d7d47260ef621777d1",
|
| 4 |
+
"artifact_size_bytes": 301991904,
|
| 5 |
+
"checks": {
|
| 6 |
+
"all_source_tensors_contiguous": true,
|
| 7 |
+
"all_source_tensors_finite": true,
|
| 8 |
+
"all_source_tensors_fp32": true,
|
| 9 |
+
"all_source_tensors_shape_2048x2048": true,
|
| 10 |
+
"fp16_cast_matches_companion_artifact": true,
|
| 11 |
+
"roundtrip_all_tensor_bit_patterns_equal": true,
|
| 12 |
+
"roundtrip_all_tensor_byte_hashes_equal": true,
|
| 13 |
+
"roundtrip_all_tensor_dtypes_equal": true,
|
| 14 |
+
"roundtrip_all_tensor_shapes_equal": true,
|
| 15 |
+
"roundtrip_all_tensor_values_equal": true,
|
| 16 |
+
"roundtrip_key_set_exact": true,
|
| 17 |
+
"safetensors_header_public_safe": true,
|
| 18 |
+
"safetensors_header_roundtrip_exact": true,
|
| 19 |
+
"source_checkpoint_sha256_exact": true,
|
| 20 |
+
"source_metadata_exact": true,
|
| 21 |
+
"source_top_level_key_set_exact": true
|
| 22 |
+
},
|
| 23 |
+
"exact_tensor_matches": 18,
|
| 24 |
+
"expected_tensor_matches": 18,
|
| 25 |
+
"ok": true,
|
| 26 |
+
"schema_version": 1,
|
| 27 |
+
"source_checkpoint_sha256": "8f752032a26a5196c1cb447ef63f01e8a29820ff0c57178dd80c6e26d32b12a9",
|
| 28 |
+
"tensor_values_changed": 0
|
| 29 |
+
}
|
lens_config.json
ADDED
|
@@ -0,0 +1,62 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"artifact": {
|
| 3 |
+
"filename": "model.safetensors",
|
| 4 |
+
"format": "safetensors",
|
| 5 |
+
"sha256": "089d776979408f23e5377539c15aa8025d171633718ccdce709bcd3372e7942c",
|
| 6 |
+
"size_bytes": 150996824
|
| 7 |
+
},
|
| 8 |
+
"artifact_kind": "jacobian_lens",
|
| 9 |
+
"evaluation_artifact": {
|
| 10 |
+
"filename": "evaluation.safetensors",
|
| 11 |
+
"format": "safetensors",
|
| 12 |
+
"sha256": "0cc184eb65d273ac8bfee5450a141c8cdf8dd6ec8caa68d7d47260ef621777d1",
|
| 13 |
+
"size_bytes": 301991904,
|
| 14 |
+
"source_checkpoint_sha256": "8f752032a26a5196c1cb447ef63f01e8a29820ff0c57178dd80c6e26d32b12a9",
|
| 15 |
+
"tensor_conversion": "lossless_fp32_reserialization",
|
| 16 |
+
"tensor_dtype": "float32"
|
| 17 |
+
},
|
| 18 |
+
"d_model": 2048,
|
| 19 |
+
"fit": {
|
| 20 |
+
"dim_batch": 8,
|
| 21 |
+
"exclude_final_position": true,
|
| 22 |
+
"fit_dtype": "bfloat16",
|
| 23 |
+
"max_seq_len": 128,
|
| 24 |
+
"n_prompts": 1000,
|
| 25 |
+
"skip_first": 16
|
| 26 |
+
},
|
| 27 |
+
"model": {
|
| 28 |
+
"architecture": "Qwen2ForCausalLM",
|
| 29 |
+
"id": "WeiboAI/VibeThinker-3B",
|
| 30 |
+
"n_layers": 36,
|
| 31 |
+
"revision": "77bd2cced09193c8b9a59a32bd8577bbd1f3e01c"
|
| 32 |
+
},
|
| 33 |
+
"n_prompts": 1000,
|
| 34 |
+
"schema_version": 1,
|
| 35 |
+
"source_checkpoint": {
|
| 36 |
+
"format": "pytorch",
|
| 37 |
+
"sha256": "f36a99447623e0d777c70951a9148a7a52e42e0df82942e22ce6326f63d8d664"
|
| 38 |
+
},
|
| 39 |
+
"source_layers": [
|
| 40 |
+
0,
|
| 41 |
+
2,
|
| 42 |
+
4,
|
| 43 |
+
6,
|
| 44 |
+
8,
|
| 45 |
+
10,
|
| 46 |
+
12,
|
| 47 |
+
14,
|
| 48 |
+
16,
|
| 49 |
+
18,
|
| 50 |
+
20,
|
| 51 |
+
22,
|
| 52 |
+
24,
|
| 53 |
+
26,
|
| 54 |
+
28,
|
| 55 |
+
30,
|
| 56 |
+
32,
|
| 57 |
+
34
|
| 58 |
+
],
|
| 59 |
+
"target_layer": 35,
|
| 60 |
+
"tensor_dtype": "float16",
|
| 61 |
+
"tensor_key_pattern": "J.{source_layer}"
|
| 62 |
+
}
|
model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:089d776979408f23e5377539c15aa8025d171633718ccdce709bcd3372e7942c
|
| 3 |
+
size 150996824
|
provenance.json
ADDED
|
@@ -0,0 +1,128 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"artifact": {
|
| 3 |
+
"filename": "model.safetensors",
|
| 4 |
+
"format": "safetensors",
|
| 5 |
+
"sha256": "089d776979408f23e5377539c15aa8025d171633718ccdce709bcd3372e7942c",
|
| 6 |
+
"size_bytes": 150996824,
|
| 7 |
+
"tensor_conversion": "lossless_fp16_reserialization"
|
| 8 |
+
},
|
| 9 |
+
"artifact_kind": "jacobian_lens_provenance",
|
| 10 |
+
"conversion": {
|
| 11 |
+
"safetensors": "0.7.0",
|
| 12 |
+
"source_and_output_tensor_bit_patterns_equal": true,
|
| 13 |
+
"source_and_output_tensor_dtypes_equal": true,
|
| 14 |
+
"source_and_output_tensor_keys_equal": true,
|
| 15 |
+
"source_and_output_tensor_shapes_equal": true,
|
| 16 |
+
"torch": "2.12.0"
|
| 17 |
+
},
|
| 18 |
+
"fit": {
|
| 19 |
+
"d_model": 2048,
|
| 20 |
+
"dim_batch": 8,
|
| 21 |
+
"fit_date": null,
|
| 22 |
+
"fit_date_available": false,
|
| 23 |
+
"estimator": {
|
| 24 |
+
"exclude_final_position": true,
|
| 25 |
+
"name": "causal_all_current_and_future_targets_mean_jacobian",
|
| 26 |
+
"prompt_aggregation": "equal_weight_mean_over_prompts",
|
| 27 |
+
"source_position_aggregation": "mean_over_valid_source_positions",
|
| 28 |
+
"target_position_aggregation": "sum_over_valid_targets_at_or_after_source"
|
| 29 |
+
},
|
| 30 |
+
"fit_dtype": "bfloat16",
|
| 31 |
+
"hook_convention": "post_transformer_block_output_residual",
|
| 32 |
+
"max_seq_len": 128,
|
| 33 |
+
"n_prompts": 1000,
|
| 34 |
+
"skip_first": 16,
|
| 35 |
+
"source_layers": [
|
| 36 |
+
0,
|
| 37 |
+
2,
|
| 38 |
+
4,
|
| 39 |
+
6,
|
| 40 |
+
8,
|
| 41 |
+
10,
|
| 42 |
+
12,
|
| 43 |
+
14,
|
| 44 |
+
16,
|
| 45 |
+
18,
|
| 46 |
+
20,
|
| 47 |
+
22,
|
| 48 |
+
24,
|
| 49 |
+
26,
|
| 50 |
+
28,
|
| 51 |
+
30,
|
| 52 |
+
32,
|
| 53 |
+
34
|
| 54 |
+
],
|
| 55 |
+
"stored_dtype": "float16",
|
| 56 |
+
"target_layer": 35
|
| 57 |
+
},
|
| 58 |
+
"limitations": [
|
| 59 |
+
"The original fit did not persist the resolved Hugging Face commit. The revision binding was reconstructed from the Hub head and its last-modified timestamp after the fit.",
|
| 60 |
+
"Artifact integrity and provenance do not establish causal intervention validity.",
|
| 61 |
+
"The fitting prompt text is not redistributed in this repository.",
|
| 62 |
+
"The original remote prompt JSONL is absent from the fetched fit artifacts. A later local materialization matches the remote manifest's aggregate row-hash chain and summary fields, but exact remote JSONL byte identity was not verified."
|
| 63 |
+
],
|
| 64 |
+
"model": {
|
| 65 |
+
"architecture": "Qwen2ForCausalLM",
|
| 66 |
+
"d_model": 2048,
|
| 67 |
+
"id": "WeiboAI/VibeThinker-3B",
|
| 68 |
+
"n_layers": 36,
|
| 69 |
+
"revision": "77bd2cced09193c8b9a59a32bd8577bbd1f3e01c",
|
| 70 |
+
"revision_binding": "inferred_hub_head_unchanged_since_before_fit",
|
| 71 |
+
"revision_last_modified": "2026-06-30T11:35:41+00:00",
|
| 72 |
+
"tied_embeddings": true
|
| 73 |
+
},
|
| 74 |
+
"prompt_corpus": {
|
| 75 |
+
"dataset": "Salesforce/wikitext",
|
| 76 |
+
"dataset_config": "wikitext-103-raw-v1",
|
| 77 |
+
"dataset_revision": null,
|
| 78 |
+
"fit_manifest": {
|
| 79 |
+
"first_id": "wikitext-0000",
|
| 80 |
+
"last_id": "wikitext-0999",
|
| 81 |
+
"manifest_sha256": "96593961cce7f84f72698cddfe67f2c0d2ea334083d93ca56aaa150ad42923c2",
|
| 82 |
+
"min_chars": 600,
|
| 83 |
+
"n_prompts": 1000,
|
| 84 |
+
"prompt_sha256_chain": "d49717ca76e73c570e18d0229c735e7049d636b7d5fdf3cc371e4877351cf368",
|
| 85 |
+
"prompt_sha256_chain_algorithm": "sha256(concatenated_per_row_sha256_hex)",
|
| 86 |
+
"total_chars": 982826
|
| 87 |
+
},
|
| 88 |
+
"later_local_materialization": {
|
| 89 |
+
"concatenated_per_row_sha256_hex": "d49717ca76e73c570e18d0229c735e7049d636b7d5fdf3cc371e4877351cf368",
|
| 90 |
+
"file_sha256": "1bf8cac8f1fae950201eb4161eb76a75bc5f7b4b9665c1a9b16e17aee3d37334",
|
| 91 |
+
"first_id": "wikitext-0000",
|
| 92 |
+
"last_id": "wikitext-0999",
|
| 93 |
+
"local_manifest_sha256": "a3aac6dd98f456bb2a95843f10ca71241d8f40e794bf1f1c83c71bcafd7c06ed",
|
| 94 |
+
"n_prompts": 1000,
|
| 95 |
+
"newline_joined_per_row_sha256_hex": "993212d9409cddaf3820330095b656b50231cb15578007ec2ea97229ca9792ad",
|
| 96 |
+
"row_hash_validation": "all_local_rows_match_embedded_sha256_utf8_text",
|
| 97 |
+
"total_chars": 982826
|
| 98 |
+
},
|
| 99 |
+
"reconciliation": {
|
| 100 |
+
"aggregate_row_hash_chain_matches": true,
|
| 101 |
+
"exact_remote_jsonl_byte_identity_verified": false,
|
| 102 |
+
"original_remote_jsonl_available": false,
|
| 103 |
+
"row_by_row_remote_comparison_performed": false,
|
| 104 |
+
"status": "later_local_materialization_matches_remote_aggregate_manifest",
|
| 105 |
+
"summary_fields_match": [
|
| 106 |
+
"n_prompts",
|
| 107 |
+
"total_chars",
|
| 108 |
+
"first_id",
|
| 109 |
+
"last_id"
|
| 110 |
+
]
|
| 111 |
+
},
|
| 112 |
+
"selection_rule_reported": "first 1000 rows with at least 600 characters",
|
| 113 |
+
"split": "train"
|
| 114 |
+
},
|
| 115 |
+
"schema_version": 1,
|
| 116 |
+
"software": {
|
| 117 |
+
"anthropic_jacobian_lens_commit": "581d398613e5602a5af361e1c34d3a92ea82ba8e",
|
| 118 |
+
"fit_runtime": {
|
| 119 |
+
"python": "3.10.12",
|
| 120 |
+
"torch": "2.1.0+cu118",
|
| 121 |
+
"transformers": "4.44.2"
|
| 122 |
+
}
|
| 123 |
+
},
|
| 124 |
+
"source_checkpoint": {
|
| 125 |
+
"format": "pytorch",
|
| 126 |
+
"sha256": "f36a99447623e0d777c70951a9148a7a52e42e0df82942e22ce6326f63d8d664"
|
| 127 |
+
}
|
| 128 |
+
}
|
requirements.txt
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
|
|
|
|
|
|
|
| 1 |
+
safetensors==0.8.0
|
| 2 |
+
torch==2.13.0
|
scripts/convert_checkpoint.py
ADDED
|
@@ -0,0 +1,275 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/usr/bin/env python3
|
| 2 |
+
"""Convert the frozen VibeThinker J Lens checkpoint to Safetensors.
|
| 3 |
+
|
| 4 |
+
The input path is intentionally required at runtime and is never copied into
|
| 5 |
+
the safetensors header or any generated public metadata. The conversion is
|
| 6 |
+
lossless: every FP16 bit pattern is compared after a safetensors round trip.
|
| 7 |
+
"""
|
| 8 |
+
|
| 9 |
+
from __future__ import annotations
|
| 10 |
+
|
| 11 |
+
import argparse
|
| 12 |
+
import hashlib
|
| 13 |
+
import json
|
| 14 |
+
import os
|
| 15 |
+
from collections.abc import Mapping
|
| 16 |
+
from pathlib import Path
|
| 17 |
+
from typing import Any
|
| 18 |
+
|
| 19 |
+
import torch
|
| 20 |
+
from safetensors import safe_open
|
| 21 |
+
|
| 22 |
+
SOURCE_SHA256 = "f36a99447623e0d777c70951a9148a7a52e42e0df82942e22ce6326f63d8d664"
|
| 23 |
+
MODEL_ID = "WeiboAI/VibeThinker-3B"
|
| 24 |
+
MODEL_REVISION = "77bd2cced09193c8b9a59a32bd8577bbd1f3e01c"
|
| 25 |
+
SOURCE_LAYERS = tuple(range(0, 36, 2))
|
| 26 |
+
TARGET_LAYER = 35
|
| 27 |
+
D_MODEL = 2048
|
| 28 |
+
N_PROMPTS = 1000
|
| 29 |
+
EXPECTED_TOP_LEVEL_KEYS = {"J", "n_prompts", "source_layers", "d_model"}
|
| 30 |
+
FORBIDDEN_PUBLIC_FRAGMENTS = (
|
| 31 |
+
os.sep.join(("", "Users", "")),
|
| 32 |
+
os.sep.join(("", "Volumes", "")),
|
| 33 |
+
os.sep.join(("", "workspace")),
|
| 34 |
+
)
|
| 35 |
+
|
| 36 |
+
|
| 37 |
+
def sha256_file(path: Path) -> str:
|
| 38 |
+
digest = hashlib.sha256()
|
| 39 |
+
with path.open("rb") as handle:
|
| 40 |
+
for chunk in iter(lambda: handle.read(8 * 1024 * 1024), b""):
|
| 41 |
+
digest.update(chunk)
|
| 42 |
+
return digest.hexdigest()
|
| 43 |
+
|
| 44 |
+
|
| 45 |
+
def tensor_storage_bytes(tensor: torch.Tensor) -> bytes:
|
| 46 |
+
"""Return C-contiguous little-endian bytes, including FP16 bit patterns."""
|
| 47 |
+
|
| 48 |
+
array = tensor.detach().cpu().contiguous().view(torch.int16).numpy()
|
| 49 |
+
return array.astype("<i2", copy=False).tobytes(order="C")
|
| 50 |
+
|
| 51 |
+
|
| 52 |
+
def tensor_sha256(tensor: torch.Tensor) -> str:
|
| 53 |
+
return hashlib.sha256(tensor_storage_bytes(tensor)).hexdigest()
|
| 54 |
+
|
| 55 |
+
|
| 56 |
+
def write_json(path: Path, value: Any) -> None:
|
| 57 |
+
path.write_text(
|
| 58 |
+
json.dumps(value, indent=2, sort_keys=True, ensure_ascii=True) + "\n",
|
| 59 |
+
encoding="utf-8",
|
| 60 |
+
)
|
| 61 |
+
|
| 62 |
+
|
| 63 |
+
def require(condition: bool, message: str) -> None:
|
| 64 |
+
if not condition:
|
| 65 |
+
raise ValueError(message)
|
| 66 |
+
|
| 67 |
+
|
| 68 |
+
def validate_source(checkpoint: Any) -> dict[int, torch.Tensor]:
|
| 69 |
+
require(isinstance(checkpoint, Mapping), "checkpoint must be a mapping")
|
| 70 |
+
require(
|
| 71 |
+
set(checkpoint) == EXPECTED_TOP_LEVEL_KEYS,
|
| 72 |
+
f"unexpected checkpoint keys: {sorted(checkpoint)}",
|
| 73 |
+
)
|
| 74 |
+
require(checkpoint["n_prompts"] == N_PROMPTS, "unexpected n_prompts")
|
| 75 |
+
require(checkpoint["d_model"] == D_MODEL, "unexpected d_model")
|
| 76 |
+
require(
|
| 77 |
+
tuple(checkpoint["source_layers"]) == SOURCE_LAYERS,
|
| 78 |
+
"unexpected source_layers",
|
| 79 |
+
)
|
| 80 |
+
|
| 81 |
+
matrices = checkpoint["J"]
|
| 82 |
+
require(isinstance(matrices, Mapping), "J must be a layer-to-tensor mapping")
|
| 83 |
+
require(set(matrices) == set(SOURCE_LAYERS), "unexpected J layer keys")
|
| 84 |
+
|
| 85 |
+
validated: dict[int, torch.Tensor] = {}
|
| 86 |
+
for layer in SOURCE_LAYERS:
|
| 87 |
+
tensor = matrices[layer]
|
| 88 |
+
require(isinstance(tensor, torch.Tensor), f"J[{layer}] is not a tensor")
|
| 89 |
+
require(tensor.device.type == "cpu", f"J[{layer}] is not on CPU")
|
| 90 |
+
require(tensor.dtype == torch.float16, f"J[{layer}] is not FP16")
|
| 91 |
+
require(tuple(tensor.shape) == (D_MODEL, D_MODEL), f"J[{layer}] shape mismatch")
|
| 92 |
+
require(tensor.is_contiguous(), f"J[{layer}] is not contiguous")
|
| 93 |
+
require(bool(torch.isfinite(tensor).all()), f"J[{layer}] contains non-finite values")
|
| 94 |
+
validated[layer] = tensor
|
| 95 |
+
return validated
|
| 96 |
+
|
| 97 |
+
|
| 98 |
+
def public_header() -> dict[str, str]:
|
| 99 |
+
return {
|
| 100 |
+
"artifact_kind": "jacobian_lens",
|
| 101 |
+
"d_model": str(D_MODEL),
|
| 102 |
+
"format": "pt",
|
| 103 |
+
"model_id": MODEL_ID,
|
| 104 |
+
"model_revision": MODEL_REVISION,
|
| 105 |
+
"n_prompts": str(N_PROMPTS),
|
| 106 |
+
"schema_version": "1",
|
| 107 |
+
"source_checkpoint_sha256": SOURCE_SHA256,
|
| 108 |
+
"source_layers": json.dumps(SOURCE_LAYERS, separators=(",", ":")),
|
| 109 |
+
"target_layer": str(TARGET_LAYER),
|
| 110 |
+
"tensor_dtype": "float16",
|
| 111 |
+
"tensor_key_pattern": "J.{source_layer}",
|
| 112 |
+
}
|
| 113 |
+
|
| 114 |
+
|
| 115 |
+
def assert_public_header(metadata: Mapping[str, str]) -> None:
|
| 116 |
+
encoded = json.dumps(dict(metadata), sort_keys=True)
|
| 117 |
+
for fragment in FORBIDDEN_PUBLIC_FRAGMENTS:
|
| 118 |
+
require(fragment not in encoded, f"private fragment found in safetensors header: {fragment}")
|
| 119 |
+
|
| 120 |
+
|
| 121 |
+
def save_deterministic_safetensors(
|
| 122 |
+
tensors: Mapping[str, torch.Tensor],
|
| 123 |
+
path: Path,
|
| 124 |
+
metadata: Mapping[str, str],
|
| 125 |
+
) -> None:
|
| 126 |
+
"""Write the documented safetensors format with canonical key ordering.
|
| 127 |
+
|
| 128 |
+
The upstream writer preserves tensor data exactly, but its Rust metadata
|
| 129 |
+
map can serialize keys in a process-random order. Canonical JSON ordering
|
| 130 |
+
makes the complete artifact reproducible byte for byte across runs.
|
| 131 |
+
"""
|
| 132 |
+
|
| 133 |
+
offset = 0
|
| 134 |
+
header: dict[str, Any] = {
|
| 135 |
+
"__metadata__": {key: metadata[key] for key in sorted(metadata)}
|
| 136 |
+
}
|
| 137 |
+
for key in sorted(tensors):
|
| 138 |
+
tensor = tensors[key]
|
| 139 |
+
require(tensor.dtype == torch.float16, f"{key} is not FP16")
|
| 140 |
+
nbytes = tensor.numel() * tensor.element_size()
|
| 141 |
+
header[key] = {
|
| 142 |
+
"dtype": "F16",
|
| 143 |
+
"shape": list(tensor.shape),
|
| 144 |
+
"data_offsets": [offset, offset + nbytes],
|
| 145 |
+
}
|
| 146 |
+
offset += nbytes
|
| 147 |
+
|
| 148 |
+
encoded_header = json.dumps(
|
| 149 |
+
header,
|
| 150 |
+
ensure_ascii=False,
|
| 151 |
+
separators=(",", ":"),
|
| 152 |
+
).encode("utf-8")
|
| 153 |
+
padding = (-len(encoded_header)) % 8
|
| 154 |
+
encoded_header += b" " * padding
|
| 155 |
+
|
| 156 |
+
with path.open("wb") as handle:
|
| 157 |
+
handle.write(len(encoded_header).to_bytes(8, byteorder="little", signed=False))
|
| 158 |
+
handle.write(encoded_header)
|
| 159 |
+
for key in sorted(tensors):
|
| 160 |
+
handle.write(tensor_storage_bytes(tensors[key]))
|
| 161 |
+
|
| 162 |
+
|
| 163 |
+
def convert(source: Path, output_dir: Path, overwrite: bool) -> dict[str, Any]:
|
| 164 |
+
source_digest = sha256_file(source)
|
| 165 |
+
require(source_digest == SOURCE_SHA256, "source checkpoint SHA-256 mismatch")
|
| 166 |
+
|
| 167 |
+
checkpoint = torch.load(source, map_location="cpu", weights_only=True)
|
| 168 |
+
matrices = validate_source(checkpoint)
|
| 169 |
+
tensors = {f"J.{layer}": matrices[layer] for layer in SOURCE_LAYERS}
|
| 170 |
+
|
| 171 |
+
output_dir.mkdir(parents=True, exist_ok=True)
|
| 172 |
+
output_path = output_dir / "model.safetensors"
|
| 173 |
+
if output_path.exists() and not overwrite:
|
| 174 |
+
raise FileExistsError(f"refusing to overwrite {output_path.name}; pass --overwrite")
|
| 175 |
+
|
| 176 |
+
metadata = public_header()
|
| 177 |
+
assert_public_header(metadata)
|
| 178 |
+
temporary_path = output_dir / ".model.safetensors.tmp"
|
| 179 |
+
save_deterministic_safetensors(tensors, temporary_path, metadata)
|
| 180 |
+
os.replace(temporary_path, output_path)
|
| 181 |
+
|
| 182 |
+
manifest_tensors: dict[str, Any] = {}
|
| 183 |
+
exact_matches = 0
|
| 184 |
+
with safe_open(output_path, framework="pt", device="cpu") as artifact:
|
| 185 |
+
stored_metadata = artifact.metadata() or {}
|
| 186 |
+
require(stored_metadata == metadata, "safetensors metadata changed during serialization")
|
| 187 |
+
assert_public_header(stored_metadata)
|
| 188 |
+
require(set(artifact.keys()) == set(tensors), "safetensors key set mismatch")
|
| 189 |
+
|
| 190 |
+
for layer in SOURCE_LAYERS:
|
| 191 |
+
key = f"J.{layer}"
|
| 192 |
+
source_tensor = matrices[layer]
|
| 193 |
+
output_tensor = artifact.get_tensor(key)
|
| 194 |
+
require(output_tensor.dtype == source_tensor.dtype, f"{key} dtype mismatch")
|
| 195 |
+
require(tuple(output_tensor.shape) == tuple(source_tensor.shape), f"{key} shape mismatch")
|
| 196 |
+
require(output_tensor.is_contiguous(), f"{key} is not contiguous")
|
| 197 |
+
require(torch.equal(output_tensor, source_tensor), f"{key} value mismatch")
|
| 198 |
+
require(
|
| 199 |
+
torch.equal(output_tensor.view(torch.int16), source_tensor.view(torch.int16)),
|
| 200 |
+
f"{key} FP16 bit-pattern mismatch",
|
| 201 |
+
)
|
| 202 |
+
source_tensor_digest = tensor_sha256(source_tensor)
|
| 203 |
+
output_tensor_digest = tensor_sha256(output_tensor)
|
| 204 |
+
require(source_tensor_digest == output_tensor_digest, f"{key} byte hash mismatch")
|
| 205 |
+
exact_matches += 1
|
| 206 |
+
manifest_tensors[key] = {
|
| 207 |
+
"dtype": "float16",
|
| 208 |
+
"nbytes": source_tensor.numel() * source_tensor.element_size(),
|
| 209 |
+
"numel": source_tensor.numel(),
|
| 210 |
+
"sha256_c_contiguous_little_endian_bytes": source_tensor_digest,
|
| 211 |
+
"shape": list(source_tensor.shape),
|
| 212 |
+
"source_layer": layer,
|
| 213 |
+
}
|
| 214 |
+
|
| 215 |
+
output_digest = sha256_file(output_path)
|
| 216 |
+
output_size = output_path.stat().st_size
|
| 217 |
+
tensor_manifest = {
|
| 218 |
+
"artifact": "model.safetensors",
|
| 219 |
+
"artifact_sha256": output_digest,
|
| 220 |
+
"artifact_size_bytes": output_size,
|
| 221 |
+
"schema_version": 1,
|
| 222 |
+
"tensor_count": len(manifest_tensors),
|
| 223 |
+
"tensor_storage_bytes": sum(item["nbytes"] for item in manifest_tensors.values()),
|
| 224 |
+
"tensors": manifest_tensors,
|
| 225 |
+
}
|
| 226 |
+
validation = {
|
| 227 |
+
"artifact": "model.safetensors",
|
| 228 |
+
"artifact_sha256": output_digest,
|
| 229 |
+
"artifact_size_bytes": output_size,
|
| 230 |
+
"checks": {
|
| 231 |
+
"all_source_tensors_contiguous": True,
|
| 232 |
+
"all_source_tensors_finite": True,
|
| 233 |
+
"all_source_tensors_fp16": True,
|
| 234 |
+
"all_source_tensors_shape_2048x2048": True,
|
| 235 |
+
"roundtrip_all_tensor_byte_hashes_equal": True,
|
| 236 |
+
"roundtrip_all_tensor_dtypes_equal": True,
|
| 237 |
+
"roundtrip_all_tensor_shapes_equal": True,
|
| 238 |
+
"roundtrip_all_tensor_values_equal": True,
|
| 239 |
+
"roundtrip_all_tensor_bit_patterns_equal": True,
|
| 240 |
+
"roundtrip_key_set_exact": True,
|
| 241 |
+
"safetensors_header_public_safe": True,
|
| 242 |
+
"safetensors_header_roundtrip_exact": True,
|
| 243 |
+
"source_checkpoint_sha256_exact": True,
|
| 244 |
+
"source_metadata_exact": True,
|
| 245 |
+
"source_top_level_key_set_exact": True,
|
| 246 |
+
},
|
| 247 |
+
"exact_tensor_matches": exact_matches,
|
| 248 |
+
"expected_tensor_matches": len(SOURCE_LAYERS),
|
| 249 |
+
"ok": exact_matches == len(SOURCE_LAYERS),
|
| 250 |
+
"schema_version": 1,
|
| 251 |
+
"source_checkpoint_sha256": source_digest,
|
| 252 |
+
"tensor_values_changed": 0,
|
| 253 |
+
}
|
| 254 |
+
write_json(output_dir / "tensor_manifest.json", tensor_manifest)
|
| 255 |
+
write_json(output_dir / "validation.json", validation)
|
| 256 |
+
(output_dir / "SHA256SUMS").write_text(
|
| 257 |
+
f"{output_digest} model.safetensors\n",
|
| 258 |
+
encoding="ascii",
|
| 259 |
+
)
|
| 260 |
+
return validation
|
| 261 |
+
|
| 262 |
+
|
| 263 |
+
def main() -> None:
|
| 264 |
+
parser = argparse.ArgumentParser(description=__doc__)
|
| 265 |
+
parser.add_argument("--source", required=True, type=Path, help="Private source .pt checkpoint")
|
| 266 |
+
parser.add_argument("--output-dir", required=True, type=Path, help="Public artifact directory")
|
| 267 |
+
parser.add_argument("--overwrite", action="store_true")
|
| 268 |
+
args = parser.parse_args()
|
| 269 |
+
|
| 270 |
+
result = convert(args.source, args.output_dir, args.overwrite)
|
| 271 |
+
print(json.dumps(result, indent=2, sort_keys=True))
|
| 272 |
+
|
| 273 |
+
|
| 274 |
+
if __name__ == "__main__":
|
| 275 |
+
main()
|
scripts/convert_evaluation_checkpoint.py
ADDED
|
@@ -0,0 +1,439 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/usr/bin/env python3
|
| 2 |
+
"""Convert the frozen FP32 evaluation lens to deterministic Safetensors."""
|
| 3 |
+
|
| 4 |
+
from __future__ import annotations
|
| 5 |
+
|
| 6 |
+
import argparse
|
| 7 |
+
import hashlib
|
| 8 |
+
import json
|
| 9 |
+
import os
|
| 10 |
+
from collections.abc import Mapping
|
| 11 |
+
from pathlib import Path
|
| 12 |
+
from typing import Any
|
| 13 |
+
|
| 14 |
+
import safetensors
|
| 15 |
+
import torch
|
| 16 |
+
from safetensors import safe_open
|
| 17 |
+
|
| 18 |
+
SOURCE_SHA256 = "8f752032a26a5196c1cb447ef63f01e8a29820ff0c57178dd80c6e26d32b12a9"
|
| 19 |
+
SOURCE_PROVENANCE_SHA256 = (
|
| 20 |
+
"b6e5764fa1580a142403a425ecd03cafe55d4cc36ca1e6b90da7fc19a35aad36"
|
| 21 |
+
)
|
| 22 |
+
SOURCE_VALIDATION_SHA256 = (
|
| 23 |
+
"4aea71008a10ef2d043129f7767e5d387372fe1fd5022550b59017a44e0b965d"
|
| 24 |
+
)
|
| 25 |
+
SOURCE_MATRIX_STATS_SHA256 = (
|
| 26 |
+
"a21fe7c1f661fa9b5455d89c0a415beae794012eab38f782b667b27fac942401"
|
| 27 |
+
)
|
| 28 |
+
SOURCE_FIT_CHECKPOINT_SHA256 = (
|
| 29 |
+
"a1236cfe5d04601575b3de150ffe50e3a67e755ede1197ecf74c205a31bdc258"
|
| 30 |
+
)
|
| 31 |
+
SOURCE_EXPORT_SCRIPT_SHA256 = (
|
| 32 |
+
"d1d9e0b7afd1d69771907a839f62b30936995fdd63b17ea2ac80f724b3cdce17"
|
| 33 |
+
)
|
| 34 |
+
FP16_ARTIFACT_SHA256 = (
|
| 35 |
+
"089d776979408f23e5377539c15aa8025d171633718ccdce709bcd3372e7942c"
|
| 36 |
+
)
|
| 37 |
+
MODEL_ID = "WeiboAI/VibeThinker-3B"
|
| 38 |
+
MODEL_REVISION = "77bd2cced09193c8b9a59a32bd8577bbd1f3e01c"
|
| 39 |
+
SOURCE_LAYERS = tuple(range(0, 36, 2))
|
| 40 |
+
TARGET_LAYER = 35
|
| 41 |
+
D_MODEL = 2048
|
| 42 |
+
N_PROMPTS = 1000
|
| 43 |
+
EXPECTED_TOP_LEVEL_KEYS = {"J", "n_prompts", "source_layers", "d_model"}
|
| 44 |
+
FORBIDDEN_PUBLIC_FRAGMENTS = (
|
| 45 |
+
os.sep.join(("", "Users", "")),
|
| 46 |
+
os.sep.join(("", "Volumes", "")),
|
| 47 |
+
os.sep.join(("", "workspace")),
|
| 48 |
+
)
|
| 49 |
+
|
| 50 |
+
|
| 51 |
+
def sha256_file(path: Path) -> str:
|
| 52 |
+
digest = hashlib.sha256()
|
| 53 |
+
with path.open("rb") as handle:
|
| 54 |
+
for chunk in iter(lambda: handle.read(8 * 1024 * 1024), b""):
|
| 55 |
+
digest.update(chunk)
|
| 56 |
+
return digest.hexdigest()
|
| 57 |
+
|
| 58 |
+
|
| 59 |
+
def tensor_storage_bytes(tensor: torch.Tensor) -> bytes:
|
| 60 |
+
"""Return C-contiguous little-endian FP32 bytes without value conversion."""
|
| 61 |
+
|
| 62 |
+
array = tensor.detach().cpu().contiguous().view(torch.int32).numpy()
|
| 63 |
+
return array.astype("<i4", copy=False).tobytes(order="C")
|
| 64 |
+
|
| 65 |
+
|
| 66 |
+
def tensor_sha256(tensor: torch.Tensor) -> str:
|
| 67 |
+
return hashlib.sha256(tensor_storage_bytes(tensor)).hexdigest()
|
| 68 |
+
|
| 69 |
+
|
| 70 |
+
def write_json(path: Path, value: Any) -> None:
|
| 71 |
+
path.write_text(
|
| 72 |
+
json.dumps(value, indent=2, sort_keys=True, ensure_ascii=True) + "\n",
|
| 73 |
+
encoding="utf-8",
|
| 74 |
+
)
|
| 75 |
+
|
| 76 |
+
|
| 77 |
+
def require(condition: bool, message: str) -> None:
|
| 78 |
+
if not condition:
|
| 79 |
+
raise ValueError(message)
|
| 80 |
+
|
| 81 |
+
|
| 82 |
+
def validate_source(checkpoint: Any) -> dict[int, torch.Tensor]:
|
| 83 |
+
require(isinstance(checkpoint, Mapping), "checkpoint must be a mapping")
|
| 84 |
+
require(
|
| 85 |
+
set(checkpoint) == EXPECTED_TOP_LEVEL_KEYS,
|
| 86 |
+
f"unexpected checkpoint keys: {sorted(checkpoint)}",
|
| 87 |
+
)
|
| 88 |
+
require(checkpoint["n_prompts"] == N_PROMPTS, "unexpected n_prompts")
|
| 89 |
+
require(checkpoint["d_model"] == D_MODEL, "unexpected d_model")
|
| 90 |
+
require(
|
| 91 |
+
tuple(checkpoint["source_layers"]) == SOURCE_LAYERS,
|
| 92 |
+
"unexpected source_layers",
|
| 93 |
+
)
|
| 94 |
+
matrices = checkpoint["J"]
|
| 95 |
+
require(isinstance(matrices, Mapping), "J must be a layer-to-tensor mapping")
|
| 96 |
+
require(set(matrices) == set(SOURCE_LAYERS), "unexpected J layer keys")
|
| 97 |
+
|
| 98 |
+
validated: dict[int, torch.Tensor] = {}
|
| 99 |
+
for layer in SOURCE_LAYERS:
|
| 100 |
+
tensor = matrices[layer]
|
| 101 |
+
require(isinstance(tensor, torch.Tensor), f"J[{layer}] is not a tensor")
|
| 102 |
+
require(tensor.device.type == "cpu", f"J[{layer}] is not on CPU")
|
| 103 |
+
require(tensor.dtype == torch.float32, f"J[{layer}] is not FP32")
|
| 104 |
+
require(
|
| 105 |
+
tuple(tensor.shape) == (D_MODEL, D_MODEL),
|
| 106 |
+
f"J[{layer}] shape mismatch",
|
| 107 |
+
)
|
| 108 |
+
require(tensor.is_contiguous(), f"J[{layer}] is not contiguous")
|
| 109 |
+
require(
|
| 110 |
+
bool(torch.isfinite(tensor).all()),
|
| 111 |
+
f"J[{layer}] contains non-finite values",
|
| 112 |
+
)
|
| 113 |
+
validated[layer] = tensor
|
| 114 |
+
return validated
|
| 115 |
+
|
| 116 |
+
|
| 117 |
+
def public_header() -> dict[str, str]:
|
| 118 |
+
return {
|
| 119 |
+
"artifact_kind": "jacobian_lens_evaluation_fp32",
|
| 120 |
+
"d_model": str(D_MODEL),
|
| 121 |
+
"format": "pt",
|
| 122 |
+
"model_id": MODEL_ID,
|
| 123 |
+
"model_revision": MODEL_REVISION,
|
| 124 |
+
"n_prompts": str(N_PROMPTS),
|
| 125 |
+
"schema_version": "1",
|
| 126 |
+
"source_fit_checkpoint_sha256": SOURCE_FIT_CHECKPOINT_SHA256,
|
| 127 |
+
"source_fp32_checkpoint_sha256": SOURCE_SHA256,
|
| 128 |
+
"source_layers": json.dumps(SOURCE_LAYERS, separators=(",", ":")),
|
| 129 |
+
"target_layer": str(TARGET_LAYER),
|
| 130 |
+
"tensor_dtype": "float32",
|
| 131 |
+
"tensor_key_pattern": "J.{source_layer}",
|
| 132 |
+
}
|
| 133 |
+
|
| 134 |
+
|
| 135 |
+
def assert_public_header(metadata: Mapping[str, str]) -> None:
|
| 136 |
+
encoded = json.dumps(dict(metadata), sort_keys=True)
|
| 137 |
+
for fragment in FORBIDDEN_PUBLIC_FRAGMENTS:
|
| 138 |
+
require(
|
| 139 |
+
fragment not in encoded,
|
| 140 |
+
f"private fragment found in Safetensors header: {fragment}",
|
| 141 |
+
)
|
| 142 |
+
|
| 143 |
+
|
| 144 |
+
def save_deterministic_safetensors(
|
| 145 |
+
tensors: Mapping[str, torch.Tensor],
|
| 146 |
+
path: Path,
|
| 147 |
+
metadata: Mapping[str, str],
|
| 148 |
+
) -> None:
|
| 149 |
+
offset = 0
|
| 150 |
+
header: dict[str, Any] = {
|
| 151 |
+
"__metadata__": {key: metadata[key] for key in sorted(metadata)}
|
| 152 |
+
}
|
| 153 |
+
for key in sorted(tensors):
|
| 154 |
+
tensor = tensors[key]
|
| 155 |
+
require(tensor.dtype == torch.float32, f"{key} is not FP32")
|
| 156 |
+
nbytes = tensor.numel() * tensor.element_size()
|
| 157 |
+
header[key] = {
|
| 158 |
+
"dtype": "F32",
|
| 159 |
+
"shape": list(tensor.shape),
|
| 160 |
+
"data_offsets": [offset, offset + nbytes],
|
| 161 |
+
}
|
| 162 |
+
offset += nbytes
|
| 163 |
+
|
| 164 |
+
encoded_header = json.dumps(
|
| 165 |
+
header,
|
| 166 |
+
ensure_ascii=False,
|
| 167 |
+
separators=(",", ":"),
|
| 168 |
+
).encode("utf-8")
|
| 169 |
+
encoded_header += b" " * ((-len(encoded_header)) % 8)
|
| 170 |
+
with path.open("wb") as handle:
|
| 171 |
+
handle.write(len(encoded_header).to_bytes(8, "little", signed=False))
|
| 172 |
+
handle.write(encoded_header)
|
| 173 |
+
for key in sorted(tensors):
|
| 174 |
+
handle.write(tensor_storage_bytes(tensors[key]))
|
| 175 |
+
|
| 176 |
+
|
| 177 |
+
def compare_fp16(
|
| 178 |
+
matrices: Mapping[int, torch.Tensor],
|
| 179 |
+
fp16_path: Path,
|
| 180 |
+
) -> dict[str, Any]:
|
| 181 |
+
require(
|
| 182 |
+
sha256_file(fp16_path) == FP16_ARTIFACT_SHA256,
|
| 183 |
+
"FP16 companion artifact SHA-256 mismatch",
|
| 184 |
+
)
|
| 185 |
+
per_layer: dict[str, Any] = {}
|
| 186 |
+
error_squared = 0.0
|
| 187 |
+
reference_squared = 0.0
|
| 188 |
+
with safe_open(fp16_path, framework="pt", device="cpu") as fp16_artifact:
|
| 189 |
+
require(
|
| 190 |
+
set(fp16_artifact.keys()) == {f"J.{layer}" for layer in SOURCE_LAYERS},
|
| 191 |
+
"FP16 companion key set mismatch",
|
| 192 |
+
)
|
| 193 |
+
for layer in SOURCE_LAYERS:
|
| 194 |
+
fp32 = matrices[layer]
|
| 195 |
+
stored_fp16 = fp16_artifact.get_tensor(f"J.{layer}")
|
| 196 |
+
cast_fp16 = fp32.to(torch.float16)
|
| 197 |
+
require(
|
| 198 |
+
torch.equal(cast_fp16.view(torch.int16), stored_fp16.view(torch.int16)),
|
| 199 |
+
f"FP32-to-FP16 cast differs at J.{layer}",
|
| 200 |
+
)
|
| 201 |
+
error = cast_fp16.float() - fp32
|
| 202 |
+
error_norm = float(torch.linalg.vector_norm(error))
|
| 203 |
+
reference_norm = float(torch.linalg.vector_norm(fp32))
|
| 204 |
+
layer_error_squared = error_norm**2
|
| 205 |
+
layer_reference_squared = reference_norm**2
|
| 206 |
+
error_squared += layer_error_squared
|
| 207 |
+
reference_squared += layer_reference_squared
|
| 208 |
+
per_layer[str(layer)] = {
|
| 209 |
+
"cast_matches_model_safetensors_exactly": True,
|
| 210 |
+
"max_absolute_error": float(error.abs().max().item()),
|
| 211 |
+
"relative_frobenius_error": error_norm / reference_norm,
|
| 212 |
+
}
|
| 213 |
+
return {
|
| 214 |
+
"schema_version": 1,
|
| 215 |
+
"artifact_kind": "fp32_to_fp16_lens_compatibility",
|
| 216 |
+
"fp32_source_checkpoint_sha256": SOURCE_SHA256,
|
| 217 |
+
"fp16_artifact": "model.safetensors",
|
| 218 |
+
"fp16_artifact_sha256": FP16_ARTIFACT_SHA256,
|
| 219 |
+
"conversion": "IEEE_FP32_to_FP16_round_to_nearest_even",
|
| 220 |
+
"all_layer_casts_match_exactly": True,
|
| 221 |
+
"relative_frobenius_error": (error_squared / reference_squared) ** 0.5,
|
| 222 |
+
"max_absolute_error": max(
|
| 223 |
+
record["max_absolute_error"] for record in per_layer.values()
|
| 224 |
+
),
|
| 225 |
+
"per_layer": per_layer,
|
| 226 |
+
}
|
| 227 |
+
|
| 228 |
+
|
| 229 |
+
def convert(source: Path, output_dir: Path, overwrite: bool) -> dict[str, Any]:
|
| 230 |
+
source_digest = sha256_file(source)
|
| 231 |
+
require(source_digest == SOURCE_SHA256, "source checkpoint SHA-256 mismatch")
|
| 232 |
+
checkpoint = torch.load(source, map_location="cpu", weights_only=True)
|
| 233 |
+
matrices = validate_source(checkpoint)
|
| 234 |
+
tensors = {f"J.{layer}": matrices[layer] for layer in SOURCE_LAYERS}
|
| 235 |
+
|
| 236 |
+
output_dir.mkdir(parents=True, exist_ok=True)
|
| 237 |
+
output_path = output_dir / "evaluation.safetensors"
|
| 238 |
+
if output_path.exists() and not overwrite:
|
| 239 |
+
raise FileExistsError(
|
| 240 |
+
f"refusing to overwrite {output_path.name}; pass --overwrite"
|
| 241 |
+
)
|
| 242 |
+
metadata = public_header()
|
| 243 |
+
assert_public_header(metadata)
|
| 244 |
+
temporary_path = output_dir / ".evaluation.safetensors.tmp"
|
| 245 |
+
save_deterministic_safetensors(tensors, temporary_path, metadata)
|
| 246 |
+
os.replace(temporary_path, output_path)
|
| 247 |
+
|
| 248 |
+
manifest_tensors: dict[str, Any] = {}
|
| 249 |
+
exact_matches = 0
|
| 250 |
+
with safe_open(output_path, framework="pt", device="cpu") as artifact:
|
| 251 |
+
stored_metadata = artifact.metadata() or {}
|
| 252 |
+
require(stored_metadata == metadata, "Safetensors metadata changed")
|
| 253 |
+
assert_public_header(stored_metadata)
|
| 254 |
+
require(set(artifact.keys()) == set(tensors), "Safetensors key set mismatch")
|
| 255 |
+
for layer in SOURCE_LAYERS:
|
| 256 |
+
key = f"J.{layer}"
|
| 257 |
+
source_tensor = matrices[layer]
|
| 258 |
+
output_tensor = artifact.get_tensor(key)
|
| 259 |
+
require(output_tensor.dtype == torch.float32, f"{key} dtype mismatch")
|
| 260 |
+
require(
|
| 261 |
+
tuple(output_tensor.shape) == (D_MODEL, D_MODEL),
|
| 262 |
+
f"{key} shape mismatch",
|
| 263 |
+
)
|
| 264 |
+
require(output_tensor.is_contiguous(), f"{key} is not contiguous")
|
| 265 |
+
require(torch.equal(output_tensor, source_tensor), f"{key} value mismatch")
|
| 266 |
+
require(
|
| 267 |
+
torch.equal(
|
| 268 |
+
output_tensor.view(torch.int32),
|
| 269 |
+
source_tensor.view(torch.int32),
|
| 270 |
+
),
|
| 271 |
+
f"{key} FP32 bit-pattern mismatch",
|
| 272 |
+
)
|
| 273 |
+
source_tensor_digest = tensor_sha256(source_tensor)
|
| 274 |
+
require(
|
| 275 |
+
tensor_sha256(output_tensor) == source_tensor_digest,
|
| 276 |
+
f"{key} raw byte hash mismatch",
|
| 277 |
+
)
|
| 278 |
+
exact_matches += 1
|
| 279 |
+
manifest_tensors[key] = {
|
| 280 |
+
"dtype": "float32",
|
| 281 |
+
"nbytes": source_tensor.numel() * source_tensor.element_size(),
|
| 282 |
+
"numel": source_tensor.numel(),
|
| 283 |
+
"sha256_c_contiguous_little_endian_bytes": source_tensor_digest,
|
| 284 |
+
"shape": list(source_tensor.shape),
|
| 285 |
+
"source_layer": layer,
|
| 286 |
+
}
|
| 287 |
+
|
| 288 |
+
output_digest = sha256_file(output_path)
|
| 289 |
+
output_size = output_path.stat().st_size
|
| 290 |
+
manifest = {
|
| 291 |
+
"schema_version": 1,
|
| 292 |
+
"artifact": "evaluation.safetensors",
|
| 293 |
+
"artifact_sha256": output_digest,
|
| 294 |
+
"artifact_size_bytes": output_size,
|
| 295 |
+
"source_checkpoint_sha256": source_digest,
|
| 296 |
+
"tensor_count": len(manifest_tensors),
|
| 297 |
+
"tensor_storage_bytes": sum(
|
| 298 |
+
record["nbytes"] for record in manifest_tensors.values()
|
| 299 |
+
),
|
| 300 |
+
"tensors": manifest_tensors,
|
| 301 |
+
}
|
| 302 |
+
compatibility = compare_fp16(matrices, output_dir / "model.safetensors")
|
| 303 |
+
compatibility["fp32_artifact"] = "evaluation.safetensors"
|
| 304 |
+
compatibility["fp32_artifact_sha256"] = output_digest
|
| 305 |
+
validation = {
|
| 306 |
+
"schema_version": 1,
|
| 307 |
+
"artifact": "evaluation.safetensors",
|
| 308 |
+
"artifact_sha256": output_digest,
|
| 309 |
+
"artifact_size_bytes": output_size,
|
| 310 |
+
"source_checkpoint_sha256": source_digest,
|
| 311 |
+
"exact_tensor_matches": exact_matches,
|
| 312 |
+
"expected_tensor_matches": len(SOURCE_LAYERS),
|
| 313 |
+
"tensor_values_changed": 0,
|
| 314 |
+
"checks": {
|
| 315 |
+
"all_source_tensors_contiguous": True,
|
| 316 |
+
"all_source_tensors_finite": True,
|
| 317 |
+
"all_source_tensors_fp32": True,
|
| 318 |
+
"all_source_tensors_shape_2048x2048": True,
|
| 319 |
+
"fp16_cast_matches_companion_artifact": True,
|
| 320 |
+
"roundtrip_all_tensor_byte_hashes_equal": True,
|
| 321 |
+
"roundtrip_all_tensor_dtypes_equal": True,
|
| 322 |
+
"roundtrip_all_tensor_shapes_equal": True,
|
| 323 |
+
"roundtrip_all_tensor_values_equal": True,
|
| 324 |
+
"roundtrip_all_tensor_bit_patterns_equal": True,
|
| 325 |
+
"roundtrip_key_set_exact": True,
|
| 326 |
+
"safetensors_header_public_safe": True,
|
| 327 |
+
"safetensors_header_roundtrip_exact": True,
|
| 328 |
+
"source_checkpoint_sha256_exact": True,
|
| 329 |
+
"source_metadata_exact": True,
|
| 330 |
+
"source_top_level_key_set_exact": True,
|
| 331 |
+
},
|
| 332 |
+
"ok": exact_matches == len(SOURCE_LAYERS),
|
| 333 |
+
}
|
| 334 |
+
provenance = {
|
| 335 |
+
"schema_version": 1,
|
| 336 |
+
"artifact_kind": "jacobian_lens_evaluation_fp32_provenance",
|
| 337 |
+
"artifact": {
|
| 338 |
+
"filename": "evaluation.safetensors",
|
| 339 |
+
"format": "safetensors",
|
| 340 |
+
"sha256": output_digest,
|
| 341 |
+
"size_bytes": output_size,
|
| 342 |
+
"tensor_conversion": "lossless_fp32_reserialization",
|
| 343 |
+
},
|
| 344 |
+
"source_checkpoint": {
|
| 345 |
+
"format": "pytorch",
|
| 346 |
+
"sha256": source_digest,
|
| 347 |
+
},
|
| 348 |
+
"derivation": {
|
| 349 |
+
"formula": "jacobian_sum[layer] / n_done",
|
| 350 |
+
"n_done": N_PROMPTS,
|
| 351 |
+
"source_fit_checkpoint_sha256": SOURCE_FIT_CHECKPOINT_SHA256,
|
| 352 |
+
"source_export_script_sha256": SOURCE_EXPORT_SCRIPT_SHA256,
|
| 353 |
+
"source_matrix_stats_sha256": SOURCE_MATRIX_STATS_SHA256,
|
| 354 |
+
"source_provenance_sha256": SOURCE_PROVENANCE_SHA256,
|
| 355 |
+
"source_validation_sha256": SOURCE_VALIDATION_SHA256,
|
| 356 |
+
},
|
| 357 |
+
"model": {
|
| 358 |
+
"architecture": "Qwen2ForCausalLM",
|
| 359 |
+
"d_model": D_MODEL,
|
| 360 |
+
"id": MODEL_ID,
|
| 361 |
+
"n_layers": 36,
|
| 362 |
+
"revision": MODEL_REVISION,
|
| 363 |
+
"revision_binding": "inferred_hub_head_unchanged_since_before_fit",
|
| 364 |
+
"revision_last_modified": "2026-06-30T11:35:41+00:00",
|
| 365 |
+
"tied_embeddings": True,
|
| 366 |
+
},
|
| 367 |
+
"lens": {
|
| 368 |
+
"d_model": D_MODEL,
|
| 369 |
+
"dtype": "float32",
|
| 370 |
+
"hook_convention": "post_transformer_block_output_residual",
|
| 371 |
+
"n_prompts": N_PROMPTS,
|
| 372 |
+
"source_layers": list(SOURCE_LAYERS),
|
| 373 |
+
"target_layer": TARGET_LAYER,
|
| 374 |
+
"estimator": {
|
| 375 |
+
"dim_batch": 8,
|
| 376 |
+
"exclude_final_position": True,
|
| 377 |
+
"fit_dtype": "bfloat16",
|
| 378 |
+
"max_seq_len": 128,
|
| 379 |
+
"name": "causal_all_current_and_future_targets_mean_jacobian",
|
| 380 |
+
"prompt_aggregation": "equal_weight_mean_over_prompts",
|
| 381 |
+
"skip_first": 16,
|
| 382 |
+
"source_position_aggregation": "mean_over_valid_source_positions",
|
| 383 |
+
"target_position_aggregation": (
|
| 384 |
+
"sum_over_valid_targets_at_or_after_source"
|
| 385 |
+
),
|
| 386 |
+
},
|
| 387 |
+
},
|
| 388 |
+
"compatibility": {
|
| 389 |
+
"file": "evaluation_compatibility.json",
|
| 390 |
+
"fp16_artifact": "model.safetensors",
|
| 391 |
+
"fp16_artifact_sha256": FP16_ARTIFACT_SHA256,
|
| 392 |
+
"all_layer_casts_match_exactly": True,
|
| 393 |
+
},
|
| 394 |
+
"software": {
|
| 395 |
+
"conversion_runtime": {
|
| 396 |
+
"python_implementation": "CPython",
|
| 397 |
+
"safetensors": safetensors.__version__,
|
| 398 |
+
"torch": torch.__version__,
|
| 399 |
+
},
|
| 400 |
+
"anthropic_jacobian_lens_commit": (
|
| 401 |
+
"581d398613e5602a5af361e1c34d3a92ea82ba8e"
|
| 402 |
+
),
|
| 403 |
+
},
|
| 404 |
+
"limitations": [
|
| 405 |
+
"The original fit did not store the resolved Hugging Face commit; the revision binding was reconstructed from the Hub head and its last-modified timestamp.",
|
| 406 |
+
"The recorded evaluation validates token readout and does not establish causal steering or a global workspace.",
|
| 407 |
+
],
|
| 408 |
+
}
|
| 409 |
+
|
| 410 |
+
write_json(output_dir / "evaluation_tensor_manifest.json", manifest)
|
| 411 |
+
write_json(output_dir / "evaluation_compatibility.json", compatibility)
|
| 412 |
+
write_json(output_dir / "evaluation_validation.json", validation)
|
| 413 |
+
write_json(output_dir / "evaluation_provenance.json", provenance)
|
| 414 |
+
return {
|
| 415 |
+
"artifact": output_path.name,
|
| 416 |
+
"artifact_sha256": output_digest,
|
| 417 |
+
"artifact_size_bytes": output_size,
|
| 418 |
+
"exact_tensor_matches": exact_matches,
|
| 419 |
+
"fp16_cast_matches": compatibility["all_layer_casts_match_exactly"],
|
| 420 |
+
"source_checkpoint_sha256": source_digest,
|
| 421 |
+
}
|
| 422 |
+
|
| 423 |
+
|
| 424 |
+
def main() -> None:
|
| 425 |
+
parser = argparse.ArgumentParser(description=__doc__)
|
| 426 |
+
parser.add_argument("--source", required=True, type=Path)
|
| 427 |
+
parser.add_argument(
|
| 428 |
+
"--output-dir",
|
| 429 |
+
type=Path,
|
| 430 |
+
default=Path(__file__).resolve().parents[1],
|
| 431 |
+
)
|
| 432 |
+
parser.add_argument("--overwrite", action="store_true")
|
| 433 |
+
args = parser.parse_args()
|
| 434 |
+
result = convert(args.source, args.output_dir.resolve(), args.overwrite)
|
| 435 |
+
print(json.dumps(result, indent=2, sort_keys=True))
|
| 436 |
+
|
| 437 |
+
|
| 438 |
+
if __name__ == "__main__":
|
| 439 |
+
main()
|
scripts/validate_artifact.py
ADDED
|
@@ -0,0 +1,1395 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/usr/bin/env python3
|
| 2 |
+
"""Validate the public model-release candidate without the private source file."""
|
| 3 |
+
|
| 4 |
+
from __future__ import annotations
|
| 5 |
+
|
| 6 |
+
import argparse
|
| 7 |
+
import hashlib
|
| 8 |
+
import json
|
| 9 |
+
import math
|
| 10 |
+
import re
|
| 11 |
+
import subprocess
|
| 12 |
+
from pathlib import Path
|
| 13 |
+
from typing import Any
|
| 14 |
+
from urllib.parse import urlsplit
|
| 15 |
+
|
| 16 |
+
import torch
|
| 17 |
+
from safetensors import safe_open
|
| 18 |
+
|
| 19 |
+
MODEL_ID = "WeiboAI/VibeThinker-3B"
|
| 20 |
+
MODEL_REVISION = "77bd2cced09193c8b9a59a32bd8577bbd1f3e01c"
|
| 21 |
+
SOURCE_CHECKPOINT_SHA256 = (
|
| 22 |
+
"f36a99447623e0d777c70951a9148a7a52e42e0df82942e22ce6326f63d8d664"
|
| 23 |
+
)
|
| 24 |
+
ARTIFACT_SHA256 = "089d776979408f23e5377539c15aa8025d171633718ccdce709bcd3372e7942c"
|
| 25 |
+
ARTIFACT_SIZE_BYTES = 150_996_824
|
| 26 |
+
EVALUATION_LENS_SHA256 = (
|
| 27 |
+
"8f752032a26a5196c1cb447ef63f01e8a29820ff0c57178dd80c6e26d32b12a9"
|
| 28 |
+
)
|
| 29 |
+
EVALUATION_ARTIFACT_SHA256 = (
|
| 30 |
+
"0cc184eb65d273ac8bfee5450a141c8cdf8dd6ec8caa68d7d47260ef621777d1"
|
| 31 |
+
)
|
| 32 |
+
EVALUATION_ARTIFACT_SIZE_BYTES = 301_991_904
|
| 33 |
+
EVALUATION_FILE_SHA256 = (
|
| 34 |
+
"073f2886e370acec7dd1564f2d7e834b4253493e84b026ec0cc017a6069a5f78"
|
| 35 |
+
)
|
| 36 |
+
SOURCE_LAYERS = tuple(range(0, 36, 2))
|
| 37 |
+
SELECTED_BAND = (24, 26, 28, 30, 32, 34)
|
| 38 |
+
D_MODEL = 2048
|
| 39 |
+
TENSOR_NBYTES = D_MODEL * D_MODEL * 2
|
| 40 |
+
TENSOR_STORAGE_BYTES = len(SOURCE_LAYERS) * TENSOR_NBYTES
|
| 41 |
+
EVALUATION_TENSOR_NBYTES = D_MODEL * D_MODEL * 4
|
| 42 |
+
EVALUATION_TENSOR_STORAGE_BYTES = len(SOURCE_LAYERS) * EVALUATION_TENSOR_NBYTES
|
| 43 |
+
EXPECTED_KEYS = {f"J.{layer}" for layer in SOURCE_LAYERS}
|
| 44 |
+
EXPECTED_TENSOR_RECORD_KEYS = {
|
| 45 |
+
"dtype",
|
| 46 |
+
"nbytes",
|
| 47 |
+
"numel",
|
| 48 |
+
"sha256_c_contiguous_little_endian_bytes",
|
| 49 |
+
"shape",
|
| 50 |
+
"source_layer",
|
| 51 |
+
}
|
| 52 |
+
EXPECTED_METADATA = {
|
| 53 |
+
"artifact_kind": "jacobian_lens",
|
| 54 |
+
"d_model": str(D_MODEL),
|
| 55 |
+
"format": "pt",
|
| 56 |
+
"model_id": MODEL_ID,
|
| 57 |
+
"model_revision": MODEL_REVISION,
|
| 58 |
+
"n_prompts": "1000",
|
| 59 |
+
"schema_version": "1",
|
| 60 |
+
"source_checkpoint_sha256": SOURCE_CHECKPOINT_SHA256,
|
| 61 |
+
"source_layers": "[0,2,4,6,8,10,12,14,16,18,20,22,24,26,28,30,32,34]",
|
| 62 |
+
"target_layer": "35",
|
| 63 |
+
"tensor_dtype": "float16",
|
| 64 |
+
"tensor_key_pattern": "J.{source_layer}",
|
| 65 |
+
}
|
| 66 |
+
EXPECTED_EVALUATION_METADATA = {
|
| 67 |
+
"artifact_kind": "jacobian_lens_evaluation_fp32",
|
| 68 |
+
"d_model": str(D_MODEL),
|
| 69 |
+
"format": "pt",
|
| 70 |
+
"model_id": MODEL_ID,
|
| 71 |
+
"model_revision": MODEL_REVISION,
|
| 72 |
+
"n_prompts": "1000",
|
| 73 |
+
"schema_version": "1",
|
| 74 |
+
"source_fit_checkpoint_sha256": (
|
| 75 |
+
"a1236cfe5d04601575b3de150ffe50e3a67e755ede1197ecf74c205a31bdc258"
|
| 76 |
+
),
|
| 77 |
+
"source_fp32_checkpoint_sha256": EVALUATION_LENS_SHA256,
|
| 78 |
+
"source_layers": "[0,2,4,6,8,10,12,14,16,18,20,22,24,26,28,30,32,34]",
|
| 79 |
+
"target_layer": "35",
|
| 80 |
+
"tensor_dtype": "float32",
|
| 81 |
+
"tensor_key_pattern": "J.{source_layer}",
|
| 82 |
+
}
|
| 83 |
+
EXPECTED_CARD_FRONT_MATTER = """license: other
|
| 84 |
+
license_name: qwen-research-license
|
| 85 |
+
license_link: https://huggingface.co/JacobMolBio/vibethinker-3b-jlens-model/blob/main/LICENSES/QWEN-RESEARCH.txt
|
| 86 |
+
tags:
|
| 87 |
+
- vibethinker-3b
|
| 88 |
+
- jacobian-lens
|
| 89 |
+
- mechanistic-interpretability
|
| 90 |
+
- interpretability
|
| 91 |
+
- qwen2
|
| 92 |
+
- safetensors"""
|
| 93 |
+
EXPECTED_RELEASE_FILES = {
|
| 94 |
+
".gitattributes",
|
| 95 |
+
".gitignore",
|
| 96 |
+
"assets/jlens-model-banner.png",
|
| 97 |
+
"assets/two-lens-files.png",
|
| 98 |
+
"assets/two-lens-files.svg",
|
| 99 |
+
"LICENSES/APACHE-2.0.txt",
|
| 100 |
+
"LICENSES/QWEN-RESEARCH.txt",
|
| 101 |
+
"LICENSES/VIBETHINKER-LICENSE-NOTE.txt",
|
| 102 |
+
"NOTICE",
|
| 103 |
+
"README.md",
|
| 104 |
+
"SHA256SUMS",
|
| 105 |
+
"THIRD_PARTY_NOTICES.md",
|
| 106 |
+
"evaluation.json",
|
| 107 |
+
"evaluation.safetensors",
|
| 108 |
+
"evaluation_compatibility.json",
|
| 109 |
+
"evaluation_provenance.json",
|
| 110 |
+
"evaluation_tensor_manifest.json",
|
| 111 |
+
"evaluation_validation.json",
|
| 112 |
+
"lens_config.json",
|
| 113 |
+
"model.safetensors",
|
| 114 |
+
"provenance.json",
|
| 115 |
+
"requirements.txt",
|
| 116 |
+
"scripts/convert_checkpoint.py",
|
| 117 |
+
"scripts/convert_evaluation_checkpoint.py",
|
| 118 |
+
"scripts/validate_artifact.py",
|
| 119 |
+
"tensor_manifest.json",
|
| 120 |
+
"validation.json",
|
| 121 |
+
}
|
| 122 |
+
PINNED_PUBLIC_FILE_SHA256 = {
|
| 123 |
+
".gitattributes": "cd0273298656ca90cb8d08fd25b7601483f7e13417e3115ab809856c42e794f7",
|
| 124 |
+
".gitignore": "4c7486a4b7c5ad04e0e62225b9a55c9ae00b039168d34614d402ac1f73acc459",
|
| 125 |
+
"assets/jlens-model-banner.png": "f27c4abb0c84481c2b0e67ed0c390716905bd1fe6fbcb09f2a65bbe15ecdc935",
|
| 126 |
+
"LICENSES/APACHE-2.0.txt": "ec01a6d25ea6a6b50430eda7af9e23e1c510048502d0a3adb5b54559e687d4fa",
|
| 127 |
+
"LICENSES/QWEN-RESEARCH.txt": "ef52482bb785733093dc9a2e8edd8e764c77d12d8e9d8f10a80c9b547d32d0f9",
|
| 128 |
+
"LICENSES/VIBETHINKER-LICENSE-NOTE.txt": "a74ea19436fbab1210ff16aa9037337d95ca9f93675406da4383bd8de4f4d25b",
|
| 129 |
+
"NOTICE": "35d9d57e97593a99aff4434875b43943d7627258495466bd5f8f81cdd8d40289",
|
| 130 |
+
"THIRD_PARTY_NOTICES.md": "d68e3b08646b56bf8d8c80d1d1acd98edfb40e72cb287762a3150d9b9d75a18f",
|
| 131 |
+
}
|
| 132 |
+
PUBLICATION_ROWS = {
|
| 133 |
+
"CODE_REPOSITORY_URL": "Source code and Pages source",
|
| 134 |
+
"TRACE_REPOSITORY_URL": "Captured trace dataset",
|
| 135 |
+
"PUBLIC_SITE_URL": "Static Pages site",
|
| 136 |
+
"MODEL_REPOSITORY_ID": "Hugging Face model repository ID",
|
| 137 |
+
}
|
| 138 |
+
PUBLICATION_PLACEHOLDERS = {key: "{{" + key + "}}" for key in PUBLICATION_ROWS}
|
| 139 |
+
CODE_REPOSITORY_NAME = "vibethinker-3b-jlens"
|
| 140 |
+
TRACE_REPOSITORY_NAME = "vibethinker-3b-jlens-traces"
|
| 141 |
+
MODEL_REPOSITORY_NAME = "vibethinker-3b-jlens-model"
|
| 142 |
+
REPOSITORY_COMPONENT = re.compile(r"[A-Za-z0-9][A-Za-z0-9_.-]*")
|
| 143 |
+
SHA256_PATTERN = re.compile(r"[0-9a-f]{64}")
|
| 144 |
+
PUBLICATION_SENTINELS = {
|
| 145 |
+
"",
|
| 146 |
+
"example",
|
| 147 |
+
"local",
|
| 148 |
+
"none",
|
| 149 |
+
"null",
|
| 150 |
+
"org",
|
| 151 |
+
"organization",
|
| 152 |
+
"placeholder",
|
| 153 |
+
"repo",
|
| 154 |
+
"repository",
|
| 155 |
+
"tbd",
|
| 156 |
+
"todo",
|
| 157 |
+
"user",
|
| 158 |
+
"username",
|
| 159 |
+
}
|
| 160 |
+
FORBIDDEN_PUBLIC_PATTERNS = {
|
| 161 |
+
"email address": re.compile(
|
| 162 |
+
r"\b[A-Za-z0-9.!#$%&'*+/=?^_`{|}~-]+"
|
| 163 |
+
+ chr(64)
|
| 164 |
+
+ r"[A-Za-z0-9](?:[A-Za-z0-9.-]{0,61}[A-Za-z0-9])?"
|
| 165 |
+
+ r"\.[A-Za-z]{2,}\b"
|
| 166 |
+
),
|
| 167 |
+
"AWS access key": re.compile(r"\b" + "AK" + r"IA[0-9A-Z]{16}\b"),
|
| 168 |
+
"AWS temporary access key": re.compile(r"\b" + "AS" + r"IA[0-9A-Z]{16}\b"),
|
| 169 |
+
"GitHub access token": re.compile(r"\b" + "gh" + r"[pousr]_[A-Za-z0-9]{20,}\b"),
|
| 170 |
+
"GitHub fine-grained token": re.compile(
|
| 171 |
+
r"\b" + "github" + r"_pat_[A-Za-z0-9_]{20,}\b"
|
| 172 |
+
),
|
| 173 |
+
"GitLab access token": re.compile(r"\b" + "gl" + r"pat-[A-Za-z0-9_-]{20,}\b"),
|
| 174 |
+
"Hugging Face access token": re.compile(r"\b" + "hf" + r"_[A-Za-z0-9]{20,}\b"),
|
| 175 |
+
"OpenAI-style access token": re.compile(r"\b" + "sk" + r"-[A-Za-z0-9_-]{20,}\b"),
|
| 176 |
+
"Google API key": re.compile(r"\b" + "AI" + r"za[0-9A-Za-z_-]{30,}\b"),
|
| 177 |
+
"Slack access token": re.compile(r"\b" + "xo" + r"[abprs]-[A-Za-z0-9-]{20,}\b"),
|
| 178 |
+
"private key": re.compile(
|
| 179 |
+
"-----BEGIN " + r"(?:DSA |EC |OPENSSH |RSA )?PRIVATE KEY-----"
|
| 180 |
+
),
|
| 181 |
+
"bearer credential": re.compile(
|
| 182 |
+
r"\b" + "Bearer" + r"\s+[A-Za-z0-9._~+/=-]{20,}", re.IGNORECASE
|
| 183 |
+
),
|
| 184 |
+
"local file URI": re.compile(r"\b" + "file:" + r"//", re.IGNORECASE),
|
| 185 |
+
"macOS user path": re.compile(r"(?<![A-Za-z0-9:])/" + r"Users/[^/\s]+/"),
|
| 186 |
+
"mounted volume path": re.compile(r"(?<![A-Za-z0-9:])/" + r"Volumes/[^/\s]+/"),
|
| 187 |
+
"Unix home path": re.compile(r"(?<![A-Za-z0-9:])/" + r"home/[^/\s]+/"),
|
| 188 |
+
"root home path": re.compile(r"(?<![A-Za-z0-9:])/" + r"root(?:/|\b)"),
|
| 189 |
+
"temporary path": re.compile(
|
| 190 |
+
r"(?<![A-Za-z0-9:])/" + r"(?:tmp|private/tmp|var/folders)/"
|
| 191 |
+
),
|
| 192 |
+
"workspace path": re.compile(r"(?<![A-Za-z0-9:])/" + r"workspaces?/[^\s]+"),
|
| 193 |
+
"mounted data path": re.compile(r"(?<![A-Za-z0-9:])/" + r"mnt/[^\s]+"),
|
| 194 |
+
"Windows user path": re.compile(r"[A-Za-z]:\\" + r"Users\\[^\\\s]+\\"),
|
| 195 |
+
"loopback hostname": re.compile(r"\b" + "local" + r"host\b", re.IGNORECASE),
|
| 196 |
+
"loopback IPv4 address": re.compile(r"\b127(?:\.[0-9]{1,3}){3}\b"),
|
| 197 |
+
"unspecified IPv4 address": re.compile(r"\b0\.0\.0\.0\b"),
|
| 198 |
+
}
|
| 199 |
+
FORBIDDEN_PUBLIC_LITERALS = {
|
| 200 |
+
"local account name": "jacob" + "vogan",
|
| 201 |
+
"local account alias": "jaco" + "vogan",
|
| 202 |
+
"invented VibeThinker copyright": "Copyright (c) 2025 " + "WeiboAI",
|
| 203 |
+
"removed VibeThinker license filename": "VIBETHINKER-" + "MIT.txt",
|
| 204 |
+
}
|
| 205 |
+
EXPECTED_EVALUATION_CLAIM_BOUNDARY = (
|
| 206 |
+
"This reference records the V1 readout results. It excludes causal-assay "
|
| 207 |
+
"results and does not establish free-generation steering or a global "
|
| 208 |
+
"workspace."
|
| 209 |
+
)
|
| 210 |
+
EXPECTED_EVALUATION_NOTE = (
|
| 211 |
+
"The recorded readout metrics bind to evaluation.safetensors, the FP32 "
|
| 212 |
+
"evaluation lens in this repository. They do not evaluate "
|
| 213 |
+
"model.safetensors, the FP16 lens used for the captured traces."
|
| 214 |
+
)
|
| 215 |
+
EXPECTED_EVALUATION_TASK = {
|
| 216 |
+
"aggregate_mean_reciprocal_rank": (
|
| 217 |
+
"mean_across_eligible_target_terms_of_reciprocal_best_rank"
|
| 218 |
+
),
|
| 219 |
+
"eligible_target": ("at_least_one_candidate_form_tokenizes_to_exactly_one_token"),
|
| 220 |
+
"final_model": (
|
| 221 |
+
"rank_target_terms_in_the_model_next_token_logits_at_the_score_position"
|
| 222 |
+
),
|
| 223 |
+
"item": "one_prompt_with_one_or_more_target_terms",
|
| 224 |
+
"layer_scope_reduction": ("best_target_rank_across_layers_in_the_reported_scope"),
|
| 225 |
+
"no_eligible_target_item": (
|
| 226 |
+
"none_of_the_item_target_terms_has_an_eligible_single_token_form"
|
| 227 |
+
),
|
| 228 |
+
"paired_bootstrap_mean_reciprocal_rank": (
|
| 229 |
+
"within_item_mean_of_reciprocal_best_rank_across_eligible_target_terms"
|
| 230 |
+
),
|
| 231 |
+
"pass_at_k": (
|
| 232 |
+
"mean_across_items_of_the_fraction_of_item_target_terms_with_best_rank_at_most_k"
|
| 233 |
+
),
|
| 234 |
+
"score_position": {
|
| 235 |
+
"default": "final_prompt_token",
|
| 236 |
+
"poetry": "last_newline_token",
|
| 237 |
+
},
|
| 238 |
+
"split": {
|
| 239 |
+
"dev_fraction": 0.3,
|
| 240 |
+
"method": "sha256_stable_split",
|
| 241 |
+
"seed": "vibethinker-jlens-v1",
|
| 242 |
+
},
|
| 243 |
+
"suites": [
|
| 244 |
+
"association",
|
| 245 |
+
"multihop",
|
| 246 |
+
"multilingual",
|
| 247 |
+
"order-ops",
|
| 248 |
+
"poetry",
|
| 249 |
+
"typo",
|
| 250 |
+
],
|
| 251 |
+
"target_candidate_forms": [
|
| 252 |
+
"original_lowercase_and_capitalized_forms_with_and_without_leading_space",
|
| 253 |
+
"order_ops_also_adds_configured_operation_synonyms_and_number_word_digit_forms",
|
| 254 |
+
],
|
| 255 |
+
"target_rank": ("best_one_based_vocabulary_rank_among_eligible_target_token_ids"),
|
| 256 |
+
}
|
| 257 |
+
|
| 258 |
+
|
| 259 |
+
def require(condition: bool, message: str) -> None:
|
| 260 |
+
if not condition:
|
| 261 |
+
raise ValueError(message)
|
| 262 |
+
|
| 263 |
+
|
| 264 |
+
def sha256_file(path: Path) -> str:
|
| 265 |
+
digest = hashlib.sha256()
|
| 266 |
+
with path.open("rb") as handle:
|
| 267 |
+
for chunk in iter(lambda: handle.read(8 * 1024 * 1024), b""):
|
| 268 |
+
digest.update(chunk)
|
| 269 |
+
return digest.hexdigest()
|
| 270 |
+
|
| 271 |
+
|
| 272 |
+
def tensor_sha256(tensor: torch.Tensor) -> str:
|
| 273 |
+
contiguous = tensor.detach().cpu().contiguous()
|
| 274 |
+
if tensor.dtype == torch.float16:
|
| 275 |
+
array = contiguous.view(torch.int16).numpy()
|
| 276 |
+
payload = array.astype("<i2", copy=False).tobytes(order="C")
|
| 277 |
+
elif tensor.dtype == torch.float32:
|
| 278 |
+
array = contiguous.view(torch.int32).numpy()
|
| 279 |
+
payload = array.astype("<i4", copy=False).tobytes(order="C")
|
| 280 |
+
else:
|
| 281 |
+
raise ValueError(f"unsupported tensor dtype for hashing: {tensor.dtype}")
|
| 282 |
+
return hashlib.sha256(payload).hexdigest()
|
| 283 |
+
|
| 284 |
+
|
| 285 |
+
def read_json(path: Path) -> dict[str, Any]:
|
| 286 |
+
value = json.loads(path.read_text(encoding="utf-8"))
|
| 287 |
+
require(isinstance(value, dict), f"{path.name} must contain a JSON object")
|
| 288 |
+
return value
|
| 289 |
+
|
| 290 |
+
|
| 291 |
+
def normalized_words(value: str) -> str:
|
| 292 |
+
return " ".join(value.split())
|
| 293 |
+
|
| 294 |
+
|
| 295 |
+
def extract_card_front_matter(readme: str) -> str:
|
| 296 |
+
require(readme.startswith("---\n"), "README model-card front matter is missing")
|
| 297 |
+
closing = readme.find("\n---\n", 4)
|
| 298 |
+
require(closing != -1, "README model-card front matter is not closed")
|
| 299 |
+
return readme[4:closing]
|
| 300 |
+
|
| 301 |
+
|
| 302 |
+
def validate_release_file_set(root: Path) -> None:
|
| 303 |
+
actual_files: set[str] = set()
|
| 304 |
+
for path in root.rglob("*"):
|
| 305 |
+
relative = path.relative_to(root)
|
| 306 |
+
if relative.parts and relative.parts[0] == ".git":
|
| 307 |
+
continue
|
| 308 |
+
require(not path.is_symlink(), f"release tree contains a symlink: {relative}")
|
| 309 |
+
if path.is_file():
|
| 310 |
+
actual_files.add(relative.as_posix())
|
| 311 |
+
require(
|
| 312 |
+
actual_files == EXPECTED_RELEASE_FILES,
|
| 313 |
+
"release file set mismatch: "
|
| 314 |
+
f"missing={sorted(EXPECTED_RELEASE_FILES - actual_files)}, "
|
| 315 |
+
f"extra={sorted(actual_files - EXPECTED_RELEASE_FILES)}",
|
| 316 |
+
)
|
| 317 |
+
|
| 318 |
+
|
| 319 |
+
def validate_pinned_public_files(root: Path) -> None:
|
| 320 |
+
for relative_path, expected_sha256 in PINNED_PUBLIC_FILE_SHA256.items():
|
| 321 |
+
require(
|
| 322 |
+
sha256_file(root / relative_path) == expected_sha256,
|
| 323 |
+
f"pinned public file hash mismatch: {relative_path}",
|
| 324 |
+
)
|
| 325 |
+
|
| 326 |
+
|
| 327 |
+
def validate_public_text(root: Path) -> None:
|
| 328 |
+
binary_files = {
|
| 329 |
+
"assets/jlens-model-banner.png",
|
| 330 |
+
"assets/two-lens-files.png",
|
| 331 |
+
"evaluation.safetensors",
|
| 332 |
+
"model.safetensors",
|
| 333 |
+
}
|
| 334 |
+
for relative_path in sorted(EXPECTED_RELEASE_FILES - binary_files):
|
| 335 |
+
text = (root / relative_path).read_text(encoding="utf-8")
|
| 336 |
+
for label, pattern in FORBIDDEN_PUBLIC_PATTERNS.items():
|
| 337 |
+
require(
|
| 338 |
+
pattern.search(text) is None,
|
| 339 |
+
f"{relative_path} contains a forbidden {label}",
|
| 340 |
+
)
|
| 341 |
+
for label, literal in FORBIDDEN_PUBLIC_LITERALS.items():
|
| 342 |
+
require(
|
| 343 |
+
literal not in text,
|
| 344 |
+
f"{relative_path} contains a forbidden {label}",
|
| 345 |
+
)
|
| 346 |
+
|
| 347 |
+
|
| 348 |
+
def parse_publication_rows(readme: str) -> tuple[dict[str, str], list[str]]:
|
| 349 |
+
values: dict[str, str] = {}
|
| 350 |
+
for key, label in PUBLICATION_ROWS.items():
|
| 351 |
+
pattern = re.compile(
|
| 352 |
+
rf"^\| {re.escape(label)} \| (?:`([^`\r\n]+)`|\[([^\]\r\n]+)\]\([^\)\r\n]+\)) \|$",
|
| 353 |
+
re.MULTILINE,
|
| 354 |
+
)
|
| 355 |
+
matches = pattern.findall(readme)
|
| 356 |
+
require(len(matches) == 1, f"README publication row mismatch: {key}")
|
| 357 |
+
values[key] = matches[0][0] or matches[0][1]
|
| 358 |
+
|
| 359 |
+
placeholders_in_readme = set(re.findall(r"\{\{([A-Z0-9_]+)\}\}", readme))
|
| 360 |
+
unresolved = sorted(
|
| 361 |
+
key for key, value in values.items() if value == PUBLICATION_PLACEHOLDERS[key]
|
| 362 |
+
)
|
| 363 |
+
require(
|
| 364 |
+
placeholders_in_readme == set(unresolved),
|
| 365 |
+
"README publication placeholders do not match the publication table",
|
| 366 |
+
)
|
| 367 |
+
require(
|
| 368 |
+
len(unresolved) in {0, len(PUBLICATION_ROWS)},
|
| 369 |
+
"publication fields must be fully unresolved or fully resolved",
|
| 370 |
+
)
|
| 371 |
+
require(
|
| 372 |
+
f'export JLENS_CODE_REPO_URL="{values["CODE_REPOSITORY_URL"]}"' in readme,
|
| 373 |
+
"source URL example differs from the publication table",
|
| 374 |
+
)
|
| 375 |
+
require(
|
| 376 |
+
f'export JLENS_MODEL_REPO_ID="{values["MODEL_REPOSITORY_ID"]}"' in readme,
|
| 377 |
+
"model repository example differs from the publication table",
|
| 378 |
+
)
|
| 379 |
+
return values, unresolved
|
| 380 |
+
|
| 381 |
+
|
| 382 |
+
def require_public_component(value: str, label: str) -> None:
|
| 383 |
+
require(value == value.strip(), f"{label} has surrounding whitespace")
|
| 384 |
+
require(value.casefold() not in PUBLICATION_SENTINELS, f"{label} is a placeholder")
|
| 385 |
+
require("{{" not in value and "}}" not in value, f"{label} is unresolved")
|
| 386 |
+
require(
|
| 387 |
+
not any(character.isspace() for character in value),
|
| 388 |
+
f"{label} contains whitespace",
|
| 389 |
+
)
|
| 390 |
+
require("\\" not in value, f"{label} contains a local path separator")
|
| 391 |
+
|
| 392 |
+
|
| 393 |
+
def parse_https_url(value: str, label: str) -> Any:
|
| 394 |
+
require_public_component(value, label)
|
| 395 |
+
parsed = urlsplit(value)
|
| 396 |
+
require(parsed.scheme == "https", f"{label} must use HTTPS")
|
| 397 |
+
require(parsed.hostname is not None, f"{label} has no hostname")
|
| 398 |
+
require(
|
| 399 |
+
parsed.username is None and parsed.password is None,
|
| 400 |
+
f"{label} contains credentials",
|
| 401 |
+
)
|
| 402 |
+
try:
|
| 403 |
+
port = parsed.port
|
| 404 |
+
except ValueError as error:
|
| 405 |
+
raise ValueError(f"{label} has an invalid port") from error
|
| 406 |
+
require(port is None, f"{label} must not specify a port")
|
| 407 |
+
require(not parsed.query, f"{label} must not contain a query")
|
| 408 |
+
require(not parsed.fragment, f"{label} must not contain a fragment")
|
| 409 |
+
require("%" not in parsed.path, f"{label} must not contain encoded path fragments")
|
| 410 |
+
require("//" not in parsed.path, f"{label} contains an empty path fragment")
|
| 411 |
+
require(";" not in parsed.path, f"{label} contains a parameter fragment")
|
| 412 |
+
return parsed
|
| 413 |
+
|
| 414 |
+
|
| 415 |
+
def validate_publication_values(values: dict[str, str]) -> None:
|
| 416 |
+
code = parse_https_url(values["CODE_REPOSITORY_URL"], "source repository URL")
|
| 417 |
+
trace = parse_https_url(values["TRACE_REPOSITORY_URL"], "trace repository URL")
|
| 418 |
+
site = parse_https_url(values["PUBLIC_SITE_URL"], "Pages site URL")
|
| 419 |
+
model_id = values["MODEL_REPOSITORY_ID"]
|
| 420 |
+
require_public_component(model_id, "model repository ID")
|
| 421 |
+
|
| 422 |
+
code_parts = [part for part in code.path.split("/") if part]
|
| 423 |
+
require(
|
| 424 |
+
code.hostname.casefold() == "github.com",
|
| 425 |
+
"source repository must use github.com",
|
| 426 |
+
)
|
| 427 |
+
require(
|
| 428 |
+
len(code_parts) == 2, "source repository URL must contain namespace/repository"
|
| 429 |
+
)
|
| 430 |
+
require(
|
| 431 |
+
code.path == "/" + "/".join(code_parts),
|
| 432 |
+
"source repository URL is not canonical",
|
| 433 |
+
)
|
| 434 |
+
require(
|
| 435 |
+
all(REPOSITORY_COMPONENT.fullmatch(part) for part in code_parts),
|
| 436 |
+
"source repository URL contains an invalid path fragment",
|
| 437 |
+
)
|
| 438 |
+
require(
|
| 439 |
+
code_parts[0].casefold() not in PUBLICATION_SENTINELS,
|
| 440 |
+
"source namespace is a placeholder",
|
| 441 |
+
)
|
| 442 |
+
require(code_parts[1] == CODE_REPOSITORY_NAME, "source repository name mismatch")
|
| 443 |
+
|
| 444 |
+
trace_parts = [part for part in trace.path.split("/") if part]
|
| 445 |
+
require(
|
| 446 |
+
trace.hostname.casefold() == "huggingface.co",
|
| 447 |
+
"trace repository must use huggingface.co",
|
| 448 |
+
)
|
| 449 |
+
require(
|
| 450 |
+
len(trace_parts) == 3 and trace_parts[0] == "datasets",
|
| 451 |
+
"trace repository URL must use /datasets/namespace/repository",
|
| 452 |
+
)
|
| 453 |
+
require(
|
| 454 |
+
trace.path == "/" + "/".join(trace_parts),
|
| 455 |
+
"trace repository URL is not canonical",
|
| 456 |
+
)
|
| 457 |
+
require(
|
| 458 |
+
all(REPOSITORY_COMPONENT.fullmatch(part) for part in trace_parts[1:]),
|
| 459 |
+
"trace repository URL contains an invalid path fragment",
|
| 460 |
+
)
|
| 461 |
+
require(
|
| 462 |
+
trace_parts[1].casefold() not in PUBLICATION_SENTINELS,
|
| 463 |
+
"trace namespace is a placeholder",
|
| 464 |
+
)
|
| 465 |
+
require(trace_parts[2] == TRACE_REPOSITORY_NAME, "trace repository name mismatch")
|
| 466 |
+
|
| 467 |
+
model_parts = model_id.split("/")
|
| 468 |
+
require(len(model_parts) == 2, "model repository ID must use namespace/repository")
|
| 469 |
+
require(
|
| 470 |
+
all(REPOSITORY_COMPONENT.fullmatch(part) for part in model_parts),
|
| 471 |
+
"model repository ID contains an invalid fragment",
|
| 472 |
+
)
|
| 473 |
+
require(
|
| 474 |
+
model_parts[0].casefold() not in PUBLICATION_SENTINELS,
|
| 475 |
+
"model namespace is a placeholder",
|
| 476 |
+
)
|
| 477 |
+
require(model_parts[1] == MODEL_REPOSITORY_NAME, "model repository name mismatch")
|
| 478 |
+
require(
|
| 479 |
+
model_parts[0].casefold() == trace_parts[1].casefold(),
|
| 480 |
+
"model and trace repositories must use the same Hugging Face namespace",
|
| 481 |
+
)
|
| 482 |
+
|
| 483 |
+
require(
|
| 484 |
+
site.hostname.casefold() == f"{code_parts[0].casefold()}.github.io",
|
| 485 |
+
"Pages hostname must match the source repository namespace",
|
| 486 |
+
)
|
| 487 |
+
require(
|
| 488 |
+
site.path.rstrip("/") == f"/{CODE_REPOSITORY_NAME}",
|
| 489 |
+
"Pages path must match the source repository name",
|
| 490 |
+
)
|
| 491 |
+
require(
|
| 492 |
+
site.path in {f"/{CODE_REPOSITORY_NAME}", f"/{CODE_REPOSITORY_NAME}/"},
|
| 493 |
+
"Pages URL is not canonical",
|
| 494 |
+
)
|
| 495 |
+
|
| 496 |
+
|
| 497 |
+
def validate_model_card(readme: str) -> None:
|
| 498 |
+
require(
|
| 499 |
+
extract_card_front_matter(readme) == EXPECTED_CARD_FRONT_MATTER,
|
| 500 |
+
"README model-card front matter mismatch",
|
| 501 |
+
)
|
| 502 |
+
readme_words = normalized_words(readme)
|
| 503 |
+
for required_text in (
|
| 504 |
+
MODEL_ID,
|
| 505 |
+
MODEL_REVISION,
|
| 506 |
+
ARTIFACT_SHA256,
|
| 507 |
+
EVALUATION_LENS_SHA256,
|
| 508 |
+
EVALUATION_ARTIFACT_SHA256,
|
| 509 |
+
"LICENSES/VIBETHINKER-LICENSE-NOTE.txt",
|
| 510 |
+
"The static site cannot analyze a new prompt.",
|
| 511 |
+
"50,050 readout rows",
|
| 512 |
+
):
|
| 513 |
+
require(
|
| 514 |
+
required_text in readme_words,
|
| 515 |
+
f"README missing required binding: {required_text}",
|
| 516 |
+
)
|
| 517 |
+
|
| 518 |
+
|
| 519 |
+
def validate_licenses_and_notices(root: Path) -> None:
|
| 520 |
+
notice = (root / "NOTICE").read_text(encoding="utf-8")
|
| 521 |
+
third_party = (root / "THIRD_PARTY_NOTICES.md").read_text(encoding="utf-8")
|
| 522 |
+
note = (root / "LICENSES/VIBETHINKER-LICENSE-NOTE.txt").read_text(encoding="utf-8")
|
| 523 |
+
qwen_license = (root / "LICENSES/QWEN-RESEARCH.txt").read_text(encoding="utf-8")
|
| 524 |
+
required_qwen_notice = (
|
| 525 |
+
"Qwen is licensed under the Qwen RESEARCH LICENSE AGREEMENT, "
|
| 526 |
+
"Copyright (c) Alibaba Cloud. All Rights Reserved."
|
| 527 |
+
)
|
| 528 |
+
require(
|
| 529 |
+
required_qwen_notice in normalized_words(notice),
|
| 530 |
+
"NOTICE is missing the required Qwen attribution",
|
| 531 |
+
)
|
| 532 |
+
require(
|
| 533 |
+
required_qwen_notice in normalized_words(qwen_license),
|
| 534 |
+
"Qwen license is missing its attribution clause",
|
| 535 |
+
)
|
| 536 |
+
require(
|
| 537 |
+
"Built with Qwen." in notice, "NOTICE is missing the Qwen product attribution"
|
| 538 |
+
)
|
| 539 |
+
require(
|
| 540 |
+
"LICENSES/VIBETHINKER-LICENSE-NOTE.txt" in notice
|
| 541 |
+
and "LICENSES/VIBETHINKER-LICENSE-NOTE.txt" in third_party,
|
| 542 |
+
"VibeThinker metadata note is not linked from the notices",
|
| 543 |
+
)
|
| 544 |
+
require(
|
| 545 |
+
"license: mit" in normalized_words(note),
|
| 546 |
+
"VibeThinker metadata declaration missing",
|
| 547 |
+
)
|
| 548 |
+
require(MODEL_REVISION in note, "VibeThinker metadata note revision mismatch")
|
| 549 |
+
require(
|
| 550 |
+
"It is not an upstream license text" in normalized_words(note),
|
| 551 |
+
"VibeThinker metadata note scope missing",
|
| 552 |
+
)
|
| 553 |
+
require(
|
| 554 |
+
not (root / "LICENSES" / ("VIBETHINKER-" + "MIT.txt")).exists(),
|
| 555 |
+
"removed VibeThinker license file is present",
|
| 556 |
+
)
|
| 557 |
+
|
| 558 |
+
|
| 559 |
+
def validate_metadata_records(
|
| 560 |
+
config: dict[str, Any],
|
| 561 |
+
provenance: dict[str, Any],
|
| 562 |
+
frozen_validation: dict[str, Any],
|
| 563 |
+
evaluation: dict[str, Any],
|
| 564 |
+
evaluation_path: Path,
|
| 565 |
+
) -> None:
|
| 566 |
+
require(config.get("schema_version") == 1, "config schema mismatch")
|
| 567 |
+
require(config.get("artifact_kind") == "jacobian_lens", "config kind mismatch")
|
| 568 |
+
require(config.get("d_model") == D_MODEL, "config width mismatch")
|
| 569 |
+
require(config.get("n_prompts") == 1000, "config prompt count mismatch")
|
| 570 |
+
require(
|
| 571 |
+
config.get("source_layers") == list(SOURCE_LAYERS),
|
| 572 |
+
"config source layers mismatch",
|
| 573 |
+
)
|
| 574 |
+
require(config.get("target_layer") == 35, "config target layer mismatch")
|
| 575 |
+
require(config.get("tensor_dtype") == "float16", "config tensor dtype mismatch")
|
| 576 |
+
require(
|
| 577 |
+
config.get("tensor_key_pattern") == "J.{source_layer}",
|
| 578 |
+
"config key pattern mismatch",
|
| 579 |
+
)
|
| 580 |
+
require(
|
| 581 |
+
config.get("model")
|
| 582 |
+
== {
|
| 583 |
+
"architecture": "Qwen2ForCausalLM",
|
| 584 |
+
"id": MODEL_ID,
|
| 585 |
+
"n_layers": 36,
|
| 586 |
+
"revision": MODEL_REVISION,
|
| 587 |
+
},
|
| 588 |
+
"config model binding mismatch",
|
| 589 |
+
)
|
| 590 |
+
require(
|
| 591 |
+
config.get("source_checkpoint", {}).get("sha256") == SOURCE_CHECKPOINT_SHA256,
|
| 592 |
+
"config source checkpoint mismatch",
|
| 593 |
+
)
|
| 594 |
+
require(
|
| 595 |
+
config.get("artifact")
|
| 596 |
+
== {
|
| 597 |
+
"filename": "model.safetensors",
|
| 598 |
+
"format": "safetensors",
|
| 599 |
+
"sha256": ARTIFACT_SHA256,
|
| 600 |
+
"size_bytes": ARTIFACT_SIZE_BYTES,
|
| 601 |
+
},
|
| 602 |
+
"config artifact binding mismatch",
|
| 603 |
+
)
|
| 604 |
+
require(
|
| 605 |
+
config.get("evaluation_artifact")
|
| 606 |
+
== {
|
| 607 |
+
"filename": "evaluation.safetensors",
|
| 608 |
+
"format": "safetensors",
|
| 609 |
+
"sha256": EVALUATION_ARTIFACT_SHA256,
|
| 610 |
+
"size_bytes": EVALUATION_ARTIFACT_SIZE_BYTES,
|
| 611 |
+
"source_checkpoint_sha256": EVALUATION_LENS_SHA256,
|
| 612 |
+
"tensor_conversion": "lossless_fp32_reserialization",
|
| 613 |
+
"tensor_dtype": "float32",
|
| 614 |
+
},
|
| 615 |
+
"config evaluation artifact binding mismatch",
|
| 616 |
+
)
|
| 617 |
+
|
| 618 |
+
require(provenance.get("schema_version") == 1, "provenance schema mismatch")
|
| 619 |
+
require(
|
| 620 |
+
provenance.get("model", {}).get("id") == MODEL_ID,
|
| 621 |
+
"provenance model ID mismatch",
|
| 622 |
+
)
|
| 623 |
+
require(
|
| 624 |
+
provenance.get("model", {}).get("revision") == MODEL_REVISION,
|
| 625 |
+
"provenance model revision mismatch",
|
| 626 |
+
)
|
| 627 |
+
require(
|
| 628 |
+
provenance.get("artifact", {}).get("sha256") == ARTIFACT_SHA256,
|
| 629 |
+
"provenance artifact hash mismatch",
|
| 630 |
+
)
|
| 631 |
+
require(
|
| 632 |
+
provenance.get("artifact", {}).get("size_bytes") == ARTIFACT_SIZE_BYTES,
|
| 633 |
+
"provenance artifact size mismatch",
|
| 634 |
+
)
|
| 635 |
+
require(
|
| 636 |
+
provenance.get("artifact", {}).get("tensor_conversion")
|
| 637 |
+
== "lossless_fp16_reserialization",
|
| 638 |
+
"provenance conversion boundary mismatch",
|
| 639 |
+
)
|
| 640 |
+
require(
|
| 641 |
+
provenance.get("source_checkpoint", {}).get("sha256")
|
| 642 |
+
== SOURCE_CHECKPOINT_SHA256,
|
| 643 |
+
"provenance source checkpoint mismatch",
|
| 644 |
+
)
|
| 645 |
+
|
| 646 |
+
require(frozen_validation.get("schema_version") == 1, "validation schema mismatch")
|
| 647 |
+
require(
|
| 648 |
+
frozen_validation.get("artifact") == "model.safetensors",
|
| 649 |
+
"validation filename mismatch",
|
| 650 |
+
)
|
| 651 |
+
require(
|
| 652 |
+
frozen_validation.get("artifact_sha256") == ARTIFACT_SHA256,
|
| 653 |
+
"validation artifact hash mismatch",
|
| 654 |
+
)
|
| 655 |
+
require(
|
| 656 |
+
frozen_validation.get("artifact_size_bytes") == ARTIFACT_SIZE_BYTES,
|
| 657 |
+
"validation artifact size mismatch",
|
| 658 |
+
)
|
| 659 |
+
require(frozen_validation.get("ok") is True, "frozen validation is not successful")
|
| 660 |
+
require(
|
| 661 |
+
frozen_validation.get("tensor_values_changed") == 0,
|
| 662 |
+
"frozen validation records changed values",
|
| 663 |
+
)
|
| 664 |
+
require(
|
| 665 |
+
frozen_validation.get("exact_tensor_matches") == len(SOURCE_LAYERS),
|
| 666 |
+
"frozen validation tensor count mismatch",
|
| 667 |
+
)
|
| 668 |
+
require(
|
| 669 |
+
frozen_validation.get("expected_tensor_matches") == len(SOURCE_LAYERS),
|
| 670 |
+
"frozen validation expected tensor count mismatch",
|
| 671 |
+
)
|
| 672 |
+
require(
|
| 673 |
+
frozen_validation.get("source_checkpoint_sha256") == SOURCE_CHECKPOINT_SHA256,
|
| 674 |
+
"frozen validation source checkpoint mismatch",
|
| 675 |
+
)
|
| 676 |
+
checks = frozen_validation.get("checks")
|
| 677 |
+
require(
|
| 678 |
+
isinstance(checks, dict)
|
| 679 |
+
and checks
|
| 680 |
+
and all(value is True for value in checks.values()),
|
| 681 |
+
"frozen validation contains a failed or malformed check",
|
| 682 |
+
)
|
| 683 |
+
|
| 684 |
+
require(
|
| 685 |
+
sha256_file(evaluation_path) == EVALUATION_FILE_SHA256,
|
| 686 |
+
"evaluation file hash mismatch",
|
| 687 |
+
)
|
| 688 |
+
require(
|
| 689 |
+
evaluation.get("artifact_kind") == "frozen_vibethinker_v1_readout_reference",
|
| 690 |
+
"evaluation artifact kind mismatch",
|
| 691 |
+
)
|
| 692 |
+
require(
|
| 693 |
+
evaluation.get("classification") == "validated_readout_only",
|
| 694 |
+
"evaluation classification mismatch",
|
| 695 |
+
)
|
| 696 |
+
require(evaluation.get("model") == MODEL_ID, "evaluation model ID mismatch")
|
| 697 |
+
require(
|
| 698 |
+
evaluation.get("model_revision") == MODEL_REVISION,
|
| 699 |
+
"evaluation model revision mismatch",
|
| 700 |
+
)
|
| 701 |
+
require(
|
| 702 |
+
evaluation.get("lens_sha256") == EVALUATION_LENS_SHA256,
|
| 703 |
+
"evaluation lens hash mismatch",
|
| 704 |
+
)
|
| 705 |
+
require(
|
| 706 |
+
evaluation.get("lens_variant") == "fp32_evaluation_safetensors_included",
|
| 707 |
+
"evaluation lens variant mismatch",
|
| 708 |
+
)
|
| 709 |
+
require(
|
| 710 |
+
evaluation.get("released_lens_artifact")
|
| 711 |
+
== {
|
| 712 |
+
"filename": "evaluation.safetensors",
|
| 713 |
+
"format": "safetensors",
|
| 714 |
+
"sha256": EVALUATION_ARTIFACT_SHA256,
|
| 715 |
+
"size_bytes": EVALUATION_ARTIFACT_SIZE_BYTES,
|
| 716 |
+
"source_checkpoint_sha256": EVALUATION_LENS_SHA256,
|
| 717 |
+
"tensor_conversion": "lossless_fp32_reserialization",
|
| 718 |
+
},
|
| 719 |
+
"evaluation released artifact binding mismatch",
|
| 720 |
+
)
|
| 721 |
+
require(
|
| 722 |
+
evaluation.get("claim_boundary") == EXPECTED_EVALUATION_CLAIM_BOUNDARY,
|
| 723 |
+
"evaluation claim boundary mismatch",
|
| 724 |
+
)
|
| 725 |
+
require(
|
| 726 |
+
evaluation.get("note") == EXPECTED_EVALUATION_NOTE,
|
| 727 |
+
"evaluation FP16/FP32 note mismatch",
|
| 728 |
+
)
|
| 729 |
+
require(
|
| 730 |
+
evaluation.get("selected_band") == list(SELECTED_BAND),
|
| 731 |
+
"evaluation selected band mismatch",
|
| 732 |
+
)
|
| 733 |
+
require(
|
| 734 |
+
evaluation.get("coverage_scope")
|
| 735 |
+
== "frozen_readout_source_run_not_the_100_prompt_ui_test_pack",
|
| 736 |
+
"evaluation coverage scope mismatch",
|
| 737 |
+
)
|
| 738 |
+
require(
|
| 739 |
+
evaluation.get("task_definition") == EXPECTED_EVALUATION_TASK,
|
| 740 |
+
"evaluation task definition mismatch",
|
| 741 |
+
)
|
| 742 |
+
evidence = evaluation.get("evidence", {})
|
| 743 |
+
require(
|
| 744 |
+
evidence.get("validated_readout_signal") is True,
|
| 745 |
+
"evaluation readout signal boundary mismatch",
|
| 746 |
+
)
|
| 747 |
+
require(
|
| 748 |
+
evidence.get("token_specific_vs_shuffled_target") is True,
|
| 749 |
+
"evaluation token control boundary mismatch",
|
| 750 |
+
)
|
| 751 |
+
require(
|
| 752 |
+
evidence.get("layer_mapping_specific_vs_shuffled_jacobian") is True,
|
| 753 |
+
"evaluation layer control boundary mismatch",
|
| 754 |
+
)
|
| 755 |
+
require(
|
| 756 |
+
evidence.get("incremental_over_ordinary_logit_lens") is False,
|
| 757 |
+
"evaluation logit-lens boundary mismatch",
|
| 758 |
+
)
|
| 759 |
+
require(
|
| 760 |
+
evidence.get("paired_bootstrap")
|
| 761 |
+
== {
|
| 762 |
+
"confidence": 0.95,
|
| 763 |
+
"input": "paired_item_metric_differences",
|
| 764 |
+
"interval": "percentile",
|
| 765 |
+
"metrics": ["pass@10", "mean_reciprocal_rank"],
|
| 766 |
+
"samples": 2000,
|
| 767 |
+
"scope": {
|
| 768 |
+
"kind": "selected_band",
|
| 769 |
+
"source_layers": list(SELECTED_BAND),
|
| 770 |
+
},
|
| 771 |
+
"split": "test",
|
| 772 |
+
},
|
| 773 |
+
"evaluation paired-bootstrap method mismatch",
|
| 774 |
+
)
|
| 775 |
+
require(
|
| 776 |
+
evidence.get("paired_bootstrap_decisions")
|
| 777 |
+
== {
|
| 778 |
+
"incremental_over_ordinary_logit_lens": {
|
| 779 |
+
"comparison": "jlens_band_minus_logit_lens_band",
|
| 780 |
+
"criterion": "at_least_one_lower_bound_greater_than_zero",
|
| 781 |
+
"met": False,
|
| 782 |
+
},
|
| 783 |
+
"layer_mapping_specific_vs_shuffled_jacobian": {
|
| 784 |
+
"comparison": "jlens_band_minus_shuffled_layer_band",
|
| 785 |
+
"criterion": "both_lower_bounds_greater_than_zero",
|
| 786 |
+
"met": True,
|
| 787 |
+
},
|
| 788 |
+
"token_specific_vs_shuffled_target": {
|
| 789 |
+
"comparison": "jlens_band_minus_shuffled_token_band",
|
| 790 |
+
"criterion": "both_lower_bounds_greater_than_zero",
|
| 791 |
+
"met": True,
|
| 792 |
+
},
|
| 793 |
+
},
|
| 794 |
+
"evaluation paired-bootstrap decisions mismatch",
|
| 795 |
+
)
|
| 796 |
+
require(
|
| 797 |
+
evidence.get("specificity_scope")
|
| 798 |
+
== {"kind": "selected_band", "source_layers": list(SELECTED_BAND)},
|
| 799 |
+
"evaluation specificity scope mismatch",
|
| 800 |
+
)
|
| 801 |
+
availability = evaluation.get("source_artifact_availability", {})
|
| 802 |
+
require(
|
| 803 |
+
availability
|
| 804 |
+
== {
|
| 805 |
+
"fp32_compatibility_check_output_included": True,
|
| 806 |
+
"fp32_compatibility_check_output_path": ("evaluation_compatibility.json"),
|
| 807 |
+
"fp32_derivation_record_included": True,
|
| 808 |
+
"fp32_derivation_record_path": "evaluation_provenance.json",
|
| 809 |
+
"fp32_evaluation_lens_included": True,
|
| 810 |
+
"fp32_evaluation_lens_path": "evaluation.safetensors",
|
| 811 |
+
"item_level_evaluation_rows_included": False,
|
| 812 |
+
"item_level_evaluation_rows_location": (
|
| 813 |
+
"companion_trace_repository:data/evaluation-results/"
|
| 814 |
+
"readout-trials.jsonl"
|
| 815 |
+
),
|
| 816 |
+
"paired_bootstrap_interval_bounds_included": False,
|
| 817 |
+
"paired_bootstrap_interval_bounds_location": (
|
| 818 |
+
"companion_trace_repository:data/evaluation-results/"
|
| 819 |
+
"readout-bootstrap-intervals.json"
|
| 820 |
+
),
|
| 821 |
+
"source_evaluation_bundle_included": False,
|
| 822 |
+
"source_hashes_included": True,
|
| 823 |
+
"source_hashes_location": (
|
| 824 |
+
"evaluation_provenance.json_and_companion_trace_repository:"
|
| 825 |
+
"data/readout-reference.json"
|
| 826 |
+
),
|
| 827 |
+
"release_content": (
|
| 828 |
+
"fp16_trace_lens_fp32_evaluation_lens_aggregate_reference_"
|
| 829 |
+
"provenance_and_compatibility"
|
| 830 |
+
),
|
| 831 |
+
},
|
| 832 |
+
"evaluation source-artifact boundary mismatch",
|
| 833 |
+
)
|
| 834 |
+
require(
|
| 835 |
+
EVALUATION_LENS_SHA256 != ARTIFACT_SHA256,
|
| 836 |
+
"FP32 and FP16 artifact hashes were conflated",
|
| 837 |
+
)
|
| 838 |
+
|
| 839 |
+
|
| 840 |
+
def validate_tensor_manifest(manifest: dict[str, Any]) -> None:
|
| 841 |
+
require(
|
| 842 |
+
set(manifest)
|
| 843 |
+
== {
|
| 844 |
+
"artifact",
|
| 845 |
+
"artifact_sha256",
|
| 846 |
+
"artifact_size_bytes",
|
| 847 |
+
"schema_version",
|
| 848 |
+
"tensor_count",
|
| 849 |
+
"tensor_storage_bytes",
|
| 850 |
+
"tensors",
|
| 851 |
+
},
|
| 852 |
+
"tensor manifest top-level fields mismatch",
|
| 853 |
+
)
|
| 854 |
+
require(
|
| 855 |
+
manifest.get("artifact") == "model.safetensors",
|
| 856 |
+
"tensor manifest filename mismatch",
|
| 857 |
+
)
|
| 858 |
+
require(
|
| 859 |
+
manifest.get("artifact_sha256") == ARTIFACT_SHA256,
|
| 860 |
+
"tensor manifest artifact hash mismatch",
|
| 861 |
+
)
|
| 862 |
+
require(
|
| 863 |
+
manifest.get("artifact_size_bytes") == ARTIFACT_SIZE_BYTES,
|
| 864 |
+
"tensor manifest artifact size mismatch",
|
| 865 |
+
)
|
| 866 |
+
require(manifest.get("schema_version") == 1, "tensor manifest schema mismatch")
|
| 867 |
+
require(
|
| 868 |
+
manifest.get("tensor_count") == len(SOURCE_LAYERS),
|
| 869 |
+
"tensor manifest count mismatch",
|
| 870 |
+
)
|
| 871 |
+
require(
|
| 872 |
+
manifest.get("tensor_storage_bytes") == TENSOR_STORAGE_BYTES,
|
| 873 |
+
"tensor manifest storage total mismatch",
|
| 874 |
+
)
|
| 875 |
+
tensors = manifest.get("tensors")
|
| 876 |
+
require(isinstance(tensors, dict), "tensor manifest tensors must be an object")
|
| 877 |
+
require(set(tensors) == EXPECTED_KEYS, "tensor manifest key set mismatch")
|
| 878 |
+
|
| 879 |
+
recorded_storage = 0
|
| 880 |
+
for layer in SOURCE_LAYERS:
|
| 881 |
+
key = f"J.{layer}"
|
| 882 |
+
record = tensors[key]
|
| 883 |
+
require(isinstance(record, dict), f"{key} descriptor must be an object")
|
| 884 |
+
require(
|
| 885 |
+
set(record) == EXPECTED_TENSOR_RECORD_KEYS,
|
| 886 |
+
f"{key} descriptor fields mismatch",
|
| 887 |
+
)
|
| 888 |
+
require(record.get("source_layer") == layer, f"{key} source layer mismatch")
|
| 889 |
+
require(record.get("dtype") == "float16", f"{key} descriptor dtype mismatch")
|
| 890 |
+
require(
|
| 891 |
+
record.get("shape") == [D_MODEL, D_MODEL],
|
| 892 |
+
f"{key} descriptor shape mismatch",
|
| 893 |
+
)
|
| 894 |
+
require(
|
| 895 |
+
record.get("numel") == D_MODEL * D_MODEL,
|
| 896 |
+
f"{key} descriptor element count mismatch",
|
| 897 |
+
)
|
| 898 |
+
require(
|
| 899 |
+
record.get("nbytes") == TENSOR_NBYTES,
|
| 900 |
+
f"{key} descriptor byte count mismatch",
|
| 901 |
+
)
|
| 902 |
+
tensor_hash = record.get("sha256_c_contiguous_little_endian_bytes")
|
| 903 |
+
require(
|
| 904 |
+
isinstance(tensor_hash, str) and SHA256_PATTERN.fullmatch(tensor_hash),
|
| 905 |
+
f"{key} descriptor hash format mismatch",
|
| 906 |
+
)
|
| 907 |
+
recorded_storage += record["nbytes"]
|
| 908 |
+
require(
|
| 909 |
+
recorded_storage == TENSOR_STORAGE_BYTES,
|
| 910 |
+
"tensor descriptor byte total mismatch",
|
| 911 |
+
)
|
| 912 |
+
|
| 913 |
+
|
| 914 |
+
def validate_evaluation_artifact(
|
| 915 |
+
root: Path,
|
| 916 |
+
manifest: dict[str, Any],
|
| 917 |
+
provenance: dict[str, Any],
|
| 918 |
+
frozen_validation: dict[str, Any],
|
| 919 |
+
compatibility: dict[str, Any],
|
| 920 |
+
) -> int:
|
| 921 |
+
artifact_path = root / "evaluation.safetensors"
|
| 922 |
+
require(
|
| 923 |
+
set(manifest)
|
| 924 |
+
== {
|
| 925 |
+
"artifact",
|
| 926 |
+
"artifact_sha256",
|
| 927 |
+
"artifact_size_bytes",
|
| 928 |
+
"schema_version",
|
| 929 |
+
"source_checkpoint_sha256",
|
| 930 |
+
"tensor_count",
|
| 931 |
+
"tensor_storage_bytes",
|
| 932 |
+
"tensors",
|
| 933 |
+
},
|
| 934 |
+
"evaluation tensor manifest top-level fields mismatch",
|
| 935 |
+
)
|
| 936 |
+
require(
|
| 937 |
+
manifest.get("artifact") == "evaluation.safetensors",
|
| 938 |
+
"evaluation tensor manifest filename mismatch",
|
| 939 |
+
)
|
| 940 |
+
require(
|
| 941 |
+
manifest.get("artifact_sha256") == EVALUATION_ARTIFACT_SHA256,
|
| 942 |
+
"evaluation tensor manifest artifact hash mismatch",
|
| 943 |
+
)
|
| 944 |
+
require(
|
| 945 |
+
manifest.get("artifact_size_bytes") == EVALUATION_ARTIFACT_SIZE_BYTES,
|
| 946 |
+
"evaluation tensor manifest artifact size mismatch",
|
| 947 |
+
)
|
| 948 |
+
require(
|
| 949 |
+
manifest.get("source_checkpoint_sha256") == EVALUATION_LENS_SHA256,
|
| 950 |
+
"evaluation tensor manifest source hash mismatch",
|
| 951 |
+
)
|
| 952 |
+
require(
|
| 953 |
+
manifest.get("tensor_count") == len(SOURCE_LAYERS),
|
| 954 |
+
"evaluation tensor manifest count mismatch",
|
| 955 |
+
)
|
| 956 |
+
require(
|
| 957 |
+
manifest.get("tensor_storage_bytes") == EVALUATION_TENSOR_STORAGE_BYTES,
|
| 958 |
+
"evaluation tensor manifest storage total mismatch",
|
| 959 |
+
)
|
| 960 |
+
tensors = manifest.get("tensors")
|
| 961 |
+
require(isinstance(tensors, dict), "evaluation manifest tensors must be an object")
|
| 962 |
+
require(set(tensors) == EXPECTED_KEYS, "evaluation manifest key set mismatch")
|
| 963 |
+
for layer in SOURCE_LAYERS:
|
| 964 |
+
key = f"J.{layer}"
|
| 965 |
+
record = tensors[key]
|
| 966 |
+
require(
|
| 967 |
+
set(record) == EXPECTED_TENSOR_RECORD_KEYS,
|
| 968 |
+
f"evaluation {key} descriptor fields mismatch",
|
| 969 |
+
)
|
| 970 |
+
require(record.get("source_layer") == layer, f"evaluation {key} layer mismatch")
|
| 971 |
+
require(record.get("dtype") == "float32", f"evaluation {key} dtype mismatch")
|
| 972 |
+
require(
|
| 973 |
+
record.get("shape") == [D_MODEL, D_MODEL],
|
| 974 |
+
f"evaluation {key} shape mismatch",
|
| 975 |
+
)
|
| 976 |
+
require(
|
| 977 |
+
record.get("numel") == D_MODEL * D_MODEL,
|
| 978 |
+
f"evaluation {key} element count mismatch",
|
| 979 |
+
)
|
| 980 |
+
require(
|
| 981 |
+
record.get("nbytes") == EVALUATION_TENSOR_NBYTES,
|
| 982 |
+
f"evaluation {key} byte count mismatch",
|
| 983 |
+
)
|
| 984 |
+
require(
|
| 985 |
+
isinstance(record.get("sha256_c_contiguous_little_endian_bytes"), str)
|
| 986 |
+
and SHA256_PATTERN.fullmatch(
|
| 987 |
+
record["sha256_c_contiguous_little_endian_bytes"]
|
| 988 |
+
),
|
| 989 |
+
f"evaluation {key} tensor hash format mismatch",
|
| 990 |
+
)
|
| 991 |
+
|
| 992 |
+
require(
|
| 993 |
+
provenance.get("artifact_kind") == "jacobian_lens_evaluation_fp32_provenance",
|
| 994 |
+
"evaluation provenance kind mismatch",
|
| 995 |
+
)
|
| 996 |
+
require(
|
| 997 |
+
provenance.get("artifact")
|
| 998 |
+
== {
|
| 999 |
+
"filename": "evaluation.safetensors",
|
| 1000 |
+
"format": "safetensors",
|
| 1001 |
+
"sha256": EVALUATION_ARTIFACT_SHA256,
|
| 1002 |
+
"size_bytes": EVALUATION_ARTIFACT_SIZE_BYTES,
|
| 1003 |
+
"tensor_conversion": "lossless_fp32_reserialization",
|
| 1004 |
+
},
|
| 1005 |
+
"evaluation provenance artifact binding mismatch",
|
| 1006 |
+
)
|
| 1007 |
+
require(
|
| 1008 |
+
provenance.get("source_checkpoint")
|
| 1009 |
+
== {"format": "pytorch", "sha256": EVALUATION_LENS_SHA256},
|
| 1010 |
+
"evaluation provenance source binding mismatch",
|
| 1011 |
+
)
|
| 1012 |
+
derivation = provenance.get("derivation", {})
|
| 1013 |
+
require(
|
| 1014 |
+
derivation.get("formula") == "jacobian_sum[layer] / n_done",
|
| 1015 |
+
"evaluation derivation formula mismatch",
|
| 1016 |
+
)
|
| 1017 |
+
require(derivation.get("n_done") == 1000, "evaluation derivation count mismatch")
|
| 1018 |
+
require(
|
| 1019 |
+
derivation.get("source_fit_checkpoint_sha256")
|
| 1020 |
+
== EXPECTED_EVALUATION_METADATA["source_fit_checkpoint_sha256"],
|
| 1021 |
+
"evaluation fit checkpoint binding mismatch",
|
| 1022 |
+
)
|
| 1023 |
+
require(
|
| 1024 |
+
provenance.get("model", {}).get("id") == MODEL_ID
|
| 1025 |
+
and provenance.get("model", {}).get("revision") == MODEL_REVISION,
|
| 1026 |
+
"evaluation provenance model binding mismatch",
|
| 1027 |
+
)
|
| 1028 |
+
require(
|
| 1029 |
+
provenance.get("lens", {}).get("dtype") == "float32"
|
| 1030 |
+
and provenance.get("lens", {}).get("source_layers") == list(SOURCE_LAYERS)
|
| 1031 |
+
and provenance.get("lens", {}).get("target_layer") == 35
|
| 1032 |
+
and provenance.get("lens", {}).get("n_prompts") == 1000,
|
| 1033 |
+
"evaluation provenance lens metadata mismatch",
|
| 1034 |
+
)
|
| 1035 |
+
runtime = provenance.get("software", {}).get("conversion_runtime", {})
|
| 1036 |
+
require(
|
| 1037 |
+
runtime.get("safetensors") == "0.8.0" and runtime.get("torch") == "2.13.0",
|
| 1038 |
+
"evaluation conversion runtime mismatch",
|
| 1039 |
+
)
|
| 1040 |
+
|
| 1041 |
+
require(
|
| 1042 |
+
frozen_validation.get("artifact") == "evaluation.safetensors",
|
| 1043 |
+
"evaluation validation filename mismatch",
|
| 1044 |
+
)
|
| 1045 |
+
require(
|
| 1046 |
+
frozen_validation.get("artifact_sha256") == EVALUATION_ARTIFACT_SHA256,
|
| 1047 |
+
"evaluation validation artifact hash mismatch",
|
| 1048 |
+
)
|
| 1049 |
+
require(
|
| 1050 |
+
frozen_validation.get("artifact_size_bytes") == EVALUATION_ARTIFACT_SIZE_BYTES,
|
| 1051 |
+
"evaluation validation artifact size mismatch",
|
| 1052 |
+
)
|
| 1053 |
+
require(
|
| 1054 |
+
frozen_validation.get("source_checkpoint_sha256") == EVALUATION_LENS_SHA256,
|
| 1055 |
+
"evaluation validation source hash mismatch",
|
| 1056 |
+
)
|
| 1057 |
+
require(frozen_validation.get("ok") is True, "evaluation validation failed")
|
| 1058 |
+
require(
|
| 1059 |
+
frozen_validation.get("tensor_values_changed") == 0,
|
| 1060 |
+
"evaluation validation records changed values",
|
| 1061 |
+
)
|
| 1062 |
+
require(
|
| 1063 |
+
frozen_validation.get("exact_tensor_matches") == len(SOURCE_LAYERS),
|
| 1064 |
+
"evaluation validation tensor count mismatch",
|
| 1065 |
+
)
|
| 1066 |
+
checks = frozen_validation.get("checks")
|
| 1067 |
+
require(
|
| 1068 |
+
isinstance(checks, dict)
|
| 1069 |
+
and checks
|
| 1070 |
+
and all(value is True for value in checks.values()),
|
| 1071 |
+
"evaluation validation contains a failed check",
|
| 1072 |
+
)
|
| 1073 |
+
|
| 1074 |
+
require(
|
| 1075 |
+
compatibility.get("artifact_kind") == "fp32_to_fp16_lens_compatibility",
|
| 1076 |
+
"evaluation compatibility kind mismatch",
|
| 1077 |
+
)
|
| 1078 |
+
require(
|
| 1079 |
+
compatibility.get("fp32_artifact") == "evaluation.safetensors"
|
| 1080 |
+
and compatibility.get("fp32_artifact_sha256") == EVALUATION_ARTIFACT_SHA256
|
| 1081 |
+
and compatibility.get("fp32_source_checkpoint_sha256")
|
| 1082 |
+
== EVALUATION_LENS_SHA256,
|
| 1083 |
+
"evaluation compatibility FP32 binding mismatch",
|
| 1084 |
+
)
|
| 1085 |
+
require(
|
| 1086 |
+
compatibility.get("fp16_artifact") == "model.safetensors"
|
| 1087 |
+
and compatibility.get("fp16_artifact_sha256") == ARTIFACT_SHA256,
|
| 1088 |
+
"evaluation compatibility FP16 binding mismatch",
|
| 1089 |
+
)
|
| 1090 |
+
require(
|
| 1091 |
+
compatibility.get("all_layer_casts_match_exactly") is True,
|
| 1092 |
+
"evaluation compatibility records a cast mismatch",
|
| 1093 |
+
)
|
| 1094 |
+
require(
|
| 1095 |
+
compatibility.get("max_absolute_error") == 0.00048828125,
|
| 1096 |
+
"evaluation compatibility maximum error mismatch",
|
| 1097 |
+
)
|
| 1098 |
+
require(
|
| 1099 |
+
math.isclose(
|
| 1100 |
+
compatibility.get("relative_frobenius_error", math.inf),
|
| 1101 |
+
0.00020901276774552773,
|
| 1102 |
+
rel_tol=1e-12,
|
| 1103 |
+
abs_tol=0.0,
|
| 1104 |
+
),
|
| 1105 |
+
"evaluation compatibility relative error mismatch",
|
| 1106 |
+
)
|
| 1107 |
+
per_layer = compatibility.get("per_layer")
|
| 1108 |
+
require(
|
| 1109 |
+
isinstance(per_layer, dict)
|
| 1110 |
+
and set(per_layer) == {str(layer) for layer in SOURCE_LAYERS},
|
| 1111 |
+
"evaluation compatibility layer set mismatch",
|
| 1112 |
+
)
|
| 1113 |
+
require(
|
| 1114 |
+
all(
|
| 1115 |
+
record.get("cast_matches_model_safetensors_exactly") is True
|
| 1116 |
+
for record in per_layer.values()
|
| 1117 |
+
),
|
| 1118 |
+
"evaluation compatibility layer cast mismatch",
|
| 1119 |
+
)
|
| 1120 |
+
|
| 1121 |
+
require(
|
| 1122 |
+
sha256_file(artifact_path) == EVALUATION_ARTIFACT_SHA256,
|
| 1123 |
+
"evaluation artifact SHA-256 mismatch",
|
| 1124 |
+
)
|
| 1125 |
+
require(
|
| 1126 |
+
artifact_path.stat().st_size == EVALUATION_ARTIFACT_SIZE_BYTES,
|
| 1127 |
+
"evaluation artifact size mismatch",
|
| 1128 |
+
)
|
| 1129 |
+
checked = 0
|
| 1130 |
+
with (
|
| 1131 |
+
safe_open(artifact_path, framework="pt", device="cpu") as artifact,
|
| 1132 |
+
safe_open(
|
| 1133 |
+
root / "model.safetensors", framework="pt", device="cpu"
|
| 1134 |
+
) as fp16_artifact,
|
| 1135 |
+
):
|
| 1136 |
+
require(
|
| 1137 |
+
(artifact.metadata() or {}) == EXPECTED_EVALUATION_METADATA,
|
| 1138 |
+
"evaluation Safetensors metadata mismatch",
|
| 1139 |
+
)
|
| 1140 |
+
require(set(artifact.keys()) == EXPECTED_KEYS, "evaluation tensor key mismatch")
|
| 1141 |
+
for layer in SOURCE_LAYERS:
|
| 1142 |
+
key = f"J.{layer}"
|
| 1143 |
+
tensor = artifact.get_tensor(key)
|
| 1144 |
+
record = tensors[key]
|
| 1145 |
+
require(tensor.dtype == torch.float32, f"evaluation {key} dtype mismatch")
|
| 1146 |
+
require(
|
| 1147 |
+
tuple(tensor.shape) == (D_MODEL, D_MODEL),
|
| 1148 |
+
f"evaluation {key} shape mismatch",
|
| 1149 |
+
)
|
| 1150 |
+
require(tensor.is_contiguous(), f"evaluation {key} is not contiguous")
|
| 1151 |
+
require(
|
| 1152 |
+
bool(torch.isfinite(tensor).all()),
|
| 1153 |
+
f"evaluation {key} contains non-finite values",
|
| 1154 |
+
)
|
| 1155 |
+
require(
|
| 1156 |
+
tensor_sha256(tensor)
|
| 1157 |
+
== record["sha256_c_contiguous_little_endian_bytes"],
|
| 1158 |
+
f"evaluation {key} tensor hash mismatch",
|
| 1159 |
+
)
|
| 1160 |
+
require(
|
| 1161 |
+
torch.equal(
|
| 1162 |
+
tensor.to(torch.float16).view(torch.int16),
|
| 1163 |
+
fp16_artifact.get_tensor(key).view(torch.int16),
|
| 1164 |
+
),
|
| 1165 |
+
f"evaluation {key} FP16 cast mismatch",
|
| 1166 |
+
)
|
| 1167 |
+
checked += 1
|
| 1168 |
+
return checked
|
| 1169 |
+
|
| 1170 |
+
|
| 1171 |
+
def run_git(root: Path, *arguments: str) -> subprocess.CompletedProcess[bytes]:
|
| 1172 |
+
return subprocess.run(
|
| 1173 |
+
["git", "-C", str(root), *arguments],
|
| 1174 |
+
check=False,
|
| 1175 |
+
capture_output=True,
|
| 1176 |
+
)
|
| 1177 |
+
|
| 1178 |
+
|
| 1179 |
+
def validate_git_lfs(
|
| 1180 |
+
root: Path, *, require_evaluation_head: bool = False
|
| 1181 |
+
) -> tuple[bool, bool]:
|
| 1182 |
+
repository_check = run_git(root, "rev-parse", "--is-inside-work-tree")
|
| 1183 |
+
if repository_check.returncode != 0:
|
| 1184 |
+
return False, False
|
| 1185 |
+
|
| 1186 |
+
model_pointer = (
|
| 1187 |
+
"version https://git-lfs.github.com/spec/v1\n"
|
| 1188 |
+
f"oid sha256:{ARTIFACT_SHA256}\n"
|
| 1189 |
+
f"size {ARTIFACT_SIZE_BYTES}\n"
|
| 1190 |
+
).encode("ascii")
|
| 1191 |
+
for object_name in ("HEAD:model.safetensors", ":model.safetensors"):
|
| 1192 |
+
pointer = run_git(root, "show", object_name)
|
| 1193 |
+
require(pointer.returncode == 0, f"cannot read Git LFS pointer: {object_name}")
|
| 1194 |
+
require(
|
| 1195 |
+
pointer.stdout == model_pointer, f"Git LFS pointer mismatch: {object_name}"
|
| 1196 |
+
)
|
| 1197 |
+
|
| 1198 |
+
evaluation_pointer = (
|
| 1199 |
+
"version https://git-lfs.github.com/spec/v1\n"
|
| 1200 |
+
f"oid sha256:{EVALUATION_ARTIFACT_SHA256}\n"
|
| 1201 |
+
f"size {EVALUATION_ARTIFACT_SIZE_BYTES}\n"
|
| 1202 |
+
).encode("ascii")
|
| 1203 |
+
staged_evaluation = run_git(root, "show", ":evaluation.safetensors")
|
| 1204 |
+
require(
|
| 1205 |
+
staged_evaluation.returncode == 0,
|
| 1206 |
+
"cannot read staged Git LFS pointer: evaluation.safetensors",
|
| 1207 |
+
)
|
| 1208 |
+
require(
|
| 1209 |
+
staged_evaluation.stdout == evaluation_pointer,
|
| 1210 |
+
"staged Git LFS pointer mismatch: evaluation.safetensors",
|
| 1211 |
+
)
|
| 1212 |
+
head_evaluation = run_git(root, "show", "HEAD:evaluation.safetensors")
|
| 1213 |
+
if require_evaluation_head:
|
| 1214 |
+
require(
|
| 1215 |
+
head_evaluation.returncode == 0,
|
| 1216 |
+
"publication check requires evaluation.safetensors in HEAD",
|
| 1217 |
+
)
|
| 1218 |
+
if head_evaluation.returncode == 0:
|
| 1219 |
+
require(
|
| 1220 |
+
head_evaluation.stdout == evaluation_pointer,
|
| 1221 |
+
"Git LFS pointer mismatch: HEAD:evaluation.safetensors",
|
| 1222 |
+
)
|
| 1223 |
+
|
| 1224 |
+
fsck = run_git(root, "lfs", "fsck", "HEAD")
|
| 1225 |
+
require(fsck.returncode == 0, "git lfs fsck failed")
|
| 1226 |
+
return True, True
|
| 1227 |
+
|
| 1228 |
+
|
| 1229 |
+
def validate(root: Path, *, require_publication_fields: bool = False) -> dict[str, Any]:
|
| 1230 |
+
artifact_path = root / "model.safetensors"
|
| 1231 |
+
readme = (root / "README.md").read_text(encoding="utf-8")
|
| 1232 |
+
config = read_json(root / "lens_config.json")
|
| 1233 |
+
manifest = read_json(root / "tensor_manifest.json")
|
| 1234 |
+
provenance = read_json(root / "provenance.json")
|
| 1235 |
+
frozen_validation = read_json(root / "validation.json")
|
| 1236 |
+
evaluation_path = root / "evaluation.json"
|
| 1237 |
+
evaluation = read_json(evaluation_path)
|
| 1238 |
+
evaluation_manifest = read_json(root / "evaluation_tensor_manifest.json")
|
| 1239 |
+
evaluation_provenance = read_json(root / "evaluation_provenance.json")
|
| 1240 |
+
evaluation_validation = read_json(root / "evaluation_validation.json")
|
| 1241 |
+
evaluation_compatibility = read_json(root / "evaluation_compatibility.json")
|
| 1242 |
+
|
| 1243 |
+
validate_release_file_set(root)
|
| 1244 |
+
validate_pinned_public_files(root)
|
| 1245 |
+
validate_public_text(root)
|
| 1246 |
+
require(
|
| 1247 |
+
(root / "requirements.txt").read_text(encoding="utf-8")
|
| 1248 |
+
== "safetensors==0.8.0\ntorch==2.13.0\n",
|
| 1249 |
+
"requirements do not match the evaluation conversion runtime",
|
| 1250 |
+
)
|
| 1251 |
+
validate_model_card(readme)
|
| 1252 |
+
validate_licenses_and_notices(root)
|
| 1253 |
+
validate_tensor_manifest(manifest)
|
| 1254 |
+
evaluation_tensors_checked = validate_evaluation_artifact(
|
| 1255 |
+
root,
|
| 1256 |
+
evaluation_manifest,
|
| 1257 |
+
evaluation_provenance,
|
| 1258 |
+
evaluation_validation,
|
| 1259 |
+
evaluation_compatibility,
|
| 1260 |
+
)
|
| 1261 |
+
validate_metadata_records(
|
| 1262 |
+
config,
|
| 1263 |
+
provenance,
|
| 1264 |
+
frozen_validation,
|
| 1265 |
+
evaluation,
|
| 1266 |
+
evaluation_path,
|
| 1267 |
+
)
|
| 1268 |
+
|
| 1269 |
+
publication_values, unresolved_publication_fields = parse_publication_rows(readme)
|
| 1270 |
+
if require_publication_fields:
|
| 1271 |
+
require(
|
| 1272 |
+
not unresolved_publication_fields,
|
| 1273 |
+
"publication fields remain unresolved: "
|
| 1274 |
+
+ ", ".join(unresolved_publication_fields),
|
| 1275 |
+
)
|
| 1276 |
+
if not unresolved_publication_fields:
|
| 1277 |
+
validate_publication_values(publication_values)
|
| 1278 |
+
|
| 1279 |
+
expected_checksum_file = (
|
| 1280 |
+
f"{ARTIFACT_SHA256} model.safetensors\n"
|
| 1281 |
+
f"{EVALUATION_ARTIFACT_SHA256} evaluation.safetensors\n"
|
| 1282 |
+
)
|
| 1283 |
+
require(
|
| 1284 |
+
(root / "SHA256SUMS").read_text(encoding="ascii") == expected_checksum_file,
|
| 1285 |
+
"SHA256SUMS content mismatch",
|
| 1286 |
+
)
|
| 1287 |
+
artifact_checksum = sha256_file(artifact_path)
|
| 1288 |
+
require(artifact_checksum == ARTIFACT_SHA256, "artifact SHA-256 mismatch")
|
| 1289 |
+
require(
|
| 1290 |
+
artifact_path.stat().st_size == ARTIFACT_SIZE_BYTES, "artifact size mismatch"
|
| 1291 |
+
)
|
| 1292 |
+
|
| 1293 |
+
checked = 0
|
| 1294 |
+
with safe_open(artifact_path, framework="pt", device="cpu") as artifact:
|
| 1295 |
+
require(
|
| 1296 |
+
(artifact.metadata() or {}) == EXPECTED_METADATA,
|
| 1297 |
+
"safetensors metadata mismatch",
|
| 1298 |
+
)
|
| 1299 |
+
require(
|
| 1300 |
+
set(artifact.keys()) == EXPECTED_KEYS, "safetensors tensor key mismatch"
|
| 1301 |
+
)
|
| 1302 |
+
for layer in SOURCE_LAYERS:
|
| 1303 |
+
key = f"J.{layer}"
|
| 1304 |
+
tensor = artifact.get_tensor(key)
|
| 1305 |
+
record = manifest["tensors"][key]
|
| 1306 |
+
require(tensor.dtype == torch.float16, f"{key} dtype mismatch")
|
| 1307 |
+
require(tuple(tensor.shape) == (D_MODEL, D_MODEL), f"{key} shape mismatch")
|
| 1308 |
+
require(tensor.is_contiguous(), f"{key} is not contiguous")
|
| 1309 |
+
require(
|
| 1310 |
+
bool(torch.isfinite(tensor).all()), f"{key} contains non-finite values"
|
| 1311 |
+
)
|
| 1312 |
+
require(tensor.numel() == record["numel"], f"{key} element count mismatch")
|
| 1313 |
+
require(
|
| 1314 |
+
tensor.numel() * tensor.element_size() == record["nbytes"],
|
| 1315 |
+
f"{key} byte count mismatch",
|
| 1316 |
+
)
|
| 1317 |
+
require(
|
| 1318 |
+
tensor_sha256(tensor)
|
| 1319 |
+
== record["sha256_c_contiguous_little_endian_bytes"],
|
| 1320 |
+
f"{key} tensor hash mismatch",
|
| 1321 |
+
)
|
| 1322 |
+
checked += 1
|
| 1323 |
+
|
| 1324 |
+
lfs_pointer_exact, lfs_fsck_ok = validate_git_lfs(
|
| 1325 |
+
root,
|
| 1326 |
+
require_evaluation_head=require_publication_fields,
|
| 1327 |
+
)
|
| 1328 |
+
if require_publication_fields:
|
| 1329 |
+
require(
|
| 1330 |
+
lfs_pointer_exact and lfs_fsck_ok,
|
| 1331 |
+
"publication check requires Git LFS validation",
|
| 1332 |
+
)
|
| 1333 |
+
|
| 1334 |
+
return {
|
| 1335 |
+
"artifact": artifact_path.name,
|
| 1336 |
+
"artifact_kind": EXPECTED_METADATA["artifact_kind"],
|
| 1337 |
+
"artifact_sha256": artifact_checksum,
|
| 1338 |
+
"artifact_size_bytes": artifact_path.stat().st_size,
|
| 1339 |
+
"evaluation_artifact_separation": True,
|
| 1340 |
+
"evaluation_artifact_sha256": EVALUATION_ARTIFACT_SHA256,
|
| 1341 |
+
"evaluation_artifact_size_bytes": EVALUATION_ARTIFACT_SIZE_BYTES,
|
| 1342 |
+
"evaluation_lens_sha256": EVALUATION_LENS_SHA256,
|
| 1343 |
+
"evaluation_tensor_count": evaluation_tensors_checked,
|
| 1344 |
+
"fp32_to_fp16_cast_exact": True,
|
| 1345 |
+
"license_notice_hashes_exact": True,
|
| 1346 |
+
"lfs_fsck_ok": lfs_fsck_ok,
|
| 1347 |
+
"lfs_pointer_exact": lfs_pointer_exact,
|
| 1348 |
+
"metadata_exact": True,
|
| 1349 |
+
"model_binding_exact": True,
|
| 1350 |
+
"ok": (
|
| 1351 |
+
checked == len(EXPECTED_KEYS)
|
| 1352 |
+
and evaluation_tensors_checked == len(EXPECTED_KEYS)
|
| 1353 |
+
),
|
| 1354 |
+
"privacy_text_scan": True,
|
| 1355 |
+
"publication_fields_resolved": not unresolved_publication_fields,
|
| 1356 |
+
"release_file_set_exact": True,
|
| 1357 |
+
"tensor_count": checked,
|
| 1358 |
+
"tensor_descriptors_exact": True,
|
| 1359 |
+
"tensor_hashes_exact": True,
|
| 1360 |
+
"unresolved_publication_fields": unresolved_publication_fields,
|
| 1361 |
+
}
|
| 1362 |
+
|
| 1363 |
+
|
| 1364 |
+
def main() -> None:
|
| 1365 |
+
parser = argparse.ArgumentParser(description=__doc__)
|
| 1366 |
+
parser.add_argument(
|
| 1367 |
+
"--root",
|
| 1368 |
+
type=Path,
|
| 1369 |
+
default=Path(__file__).resolve().parents[1],
|
| 1370 |
+
help="Artifact repository root",
|
| 1371 |
+
)
|
| 1372 |
+
parser.add_argument(
|
| 1373 |
+
"--publication",
|
| 1374 |
+
action="store_true",
|
| 1375 |
+
help="Require resolved and valid companion publication fields",
|
| 1376 |
+
)
|
| 1377 |
+
args = parser.parse_args()
|
| 1378 |
+
try:
|
| 1379 |
+
result = validate(
|
| 1380 |
+
args.root.resolve(),
|
| 1381 |
+
require_publication_fields=args.publication,
|
| 1382 |
+
)
|
| 1383 |
+
except ValueError as error:
|
| 1384 |
+
parser.error(str(error))
|
| 1385 |
+
print(
|
| 1386 |
+
json.dumps(
|
| 1387 |
+
result,
|
| 1388 |
+
indent=2,
|
| 1389 |
+
sort_keys=True,
|
| 1390 |
+
)
|
| 1391 |
+
)
|
| 1392 |
+
|
| 1393 |
+
|
| 1394 |
+
if __name__ == "__main__":
|
| 1395 |
+
main()
|
tensor_manifest.json
ADDED
|
@@ -0,0 +1,208 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"artifact": "model.safetensors",
|
| 3 |
+
"artifact_sha256": "089d776979408f23e5377539c15aa8025d171633718ccdce709bcd3372e7942c",
|
| 4 |
+
"artifact_size_bytes": 150996824,
|
| 5 |
+
"schema_version": 1,
|
| 6 |
+
"tensor_count": 18,
|
| 7 |
+
"tensor_storage_bytes": 150994944,
|
| 8 |
+
"tensors": {
|
| 9 |
+
"J.0": {
|
| 10 |
+
"dtype": "float16",
|
| 11 |
+
"nbytes": 8388608,
|
| 12 |
+
"numel": 4194304,
|
| 13 |
+
"sha256_c_contiguous_little_endian_bytes": "70b0dec033aa1707f047f17b9149ed137431e86b17958bd4a7e48d50fdda3d51",
|
| 14 |
+
"shape": [
|
| 15 |
+
2048,
|
| 16 |
+
2048
|
| 17 |
+
],
|
| 18 |
+
"source_layer": 0
|
| 19 |
+
},
|
| 20 |
+
"J.10": {
|
| 21 |
+
"dtype": "float16",
|
| 22 |
+
"nbytes": 8388608,
|
| 23 |
+
"numel": 4194304,
|
| 24 |
+
"sha256_c_contiguous_little_endian_bytes": "30da4128931002f756acb8f26c7a807b0a03aae37ebd5d1fce859ca8872203fe",
|
| 25 |
+
"shape": [
|
| 26 |
+
2048,
|
| 27 |
+
2048
|
| 28 |
+
],
|
| 29 |
+
"source_layer": 10
|
| 30 |
+
},
|
| 31 |
+
"J.12": {
|
| 32 |
+
"dtype": "float16",
|
| 33 |
+
"nbytes": 8388608,
|
| 34 |
+
"numel": 4194304,
|
| 35 |
+
"sha256_c_contiguous_little_endian_bytes": "1585c024e196065636ace0c6693304b5d238589b778d8469975cb735cac4a58e",
|
| 36 |
+
"shape": [
|
| 37 |
+
2048,
|
| 38 |
+
2048
|
| 39 |
+
],
|
| 40 |
+
"source_layer": 12
|
| 41 |
+
},
|
| 42 |
+
"J.14": {
|
| 43 |
+
"dtype": "float16",
|
| 44 |
+
"nbytes": 8388608,
|
| 45 |
+
"numel": 4194304,
|
| 46 |
+
"sha256_c_contiguous_little_endian_bytes": "25474372408e43f00aa30c8787e0953b56674d84a3d4e110619c8122de6939fc",
|
| 47 |
+
"shape": [
|
| 48 |
+
2048,
|
| 49 |
+
2048
|
| 50 |
+
],
|
| 51 |
+
"source_layer": 14
|
| 52 |
+
},
|
| 53 |
+
"J.16": {
|
| 54 |
+
"dtype": "float16",
|
| 55 |
+
"nbytes": 8388608,
|
| 56 |
+
"numel": 4194304,
|
| 57 |
+
"sha256_c_contiguous_little_endian_bytes": "f8451021c23b81883bed655568aaf5fb91fcbf0ef5abe5f92400df8012f759c9",
|
| 58 |
+
"shape": [
|
| 59 |
+
2048,
|
| 60 |
+
2048
|
| 61 |
+
],
|
| 62 |
+
"source_layer": 16
|
| 63 |
+
},
|
| 64 |
+
"J.18": {
|
| 65 |
+
"dtype": "float16",
|
| 66 |
+
"nbytes": 8388608,
|
| 67 |
+
"numel": 4194304,
|
| 68 |
+
"sha256_c_contiguous_little_endian_bytes": "8af4ab11ea1c9443b0e6ba6abf0af5073032b8c4eced8365b6a3c5ea10ec32ba",
|
| 69 |
+
"shape": [
|
| 70 |
+
2048,
|
| 71 |
+
2048
|
| 72 |
+
],
|
| 73 |
+
"source_layer": 18
|
| 74 |
+
},
|
| 75 |
+
"J.2": {
|
| 76 |
+
"dtype": "float16",
|
| 77 |
+
"nbytes": 8388608,
|
| 78 |
+
"numel": 4194304,
|
| 79 |
+
"sha256_c_contiguous_little_endian_bytes": "8a8bc773b72f5bb7e697700ba780ba105fb3efa5c966ff90121f12f97e4eb601",
|
| 80 |
+
"shape": [
|
| 81 |
+
2048,
|
| 82 |
+
2048
|
| 83 |
+
],
|
| 84 |
+
"source_layer": 2
|
| 85 |
+
},
|
| 86 |
+
"J.20": {
|
| 87 |
+
"dtype": "float16",
|
| 88 |
+
"nbytes": 8388608,
|
| 89 |
+
"numel": 4194304,
|
| 90 |
+
"sha256_c_contiguous_little_endian_bytes": "714121ea97b8bb8ba6d72b9656dfbb33e777f2345f0c990494c8960cfa493c77",
|
| 91 |
+
"shape": [
|
| 92 |
+
2048,
|
| 93 |
+
2048
|
| 94 |
+
],
|
| 95 |
+
"source_layer": 20
|
| 96 |
+
},
|
| 97 |
+
"J.22": {
|
| 98 |
+
"dtype": "float16",
|
| 99 |
+
"nbytes": 8388608,
|
| 100 |
+
"numel": 4194304,
|
| 101 |
+
"sha256_c_contiguous_little_endian_bytes": "ea197cfe6095e67364ef2d88a44fb07c80b7731c78d1137c7372771902ddf784",
|
| 102 |
+
"shape": [
|
| 103 |
+
2048,
|
| 104 |
+
2048
|
| 105 |
+
],
|
| 106 |
+
"source_layer": 22
|
| 107 |
+
},
|
| 108 |
+
"J.24": {
|
| 109 |
+
"dtype": "float16",
|
| 110 |
+
"nbytes": 8388608,
|
| 111 |
+
"numel": 4194304,
|
| 112 |
+
"sha256_c_contiguous_little_endian_bytes": "2f0e30882300a6914fd9bc089b1b038beda30e4ed1eca0da0051a532eb367c70",
|
| 113 |
+
"shape": [
|
| 114 |
+
2048,
|
| 115 |
+
2048
|
| 116 |
+
],
|
| 117 |
+
"source_layer": 24
|
| 118 |
+
},
|
| 119 |
+
"J.26": {
|
| 120 |
+
"dtype": "float16",
|
| 121 |
+
"nbytes": 8388608,
|
| 122 |
+
"numel": 4194304,
|
| 123 |
+
"sha256_c_contiguous_little_endian_bytes": "e5b3789b2802efb00d9e2e3e2753059e16db80ebef26c9e363e8a43f1a08f3da",
|
| 124 |
+
"shape": [
|
| 125 |
+
2048,
|
| 126 |
+
2048
|
| 127 |
+
],
|
| 128 |
+
"source_layer": 26
|
| 129 |
+
},
|
| 130 |
+
"J.28": {
|
| 131 |
+
"dtype": "float16",
|
| 132 |
+
"nbytes": 8388608,
|
| 133 |
+
"numel": 4194304,
|
| 134 |
+
"sha256_c_contiguous_little_endian_bytes": "de9fd5a978c691d30c46289fa179332cee671f69aba8c4442d7f61382cd56e40",
|
| 135 |
+
"shape": [
|
| 136 |
+
2048,
|
| 137 |
+
2048
|
| 138 |
+
],
|
| 139 |
+
"source_layer": 28
|
| 140 |
+
},
|
| 141 |
+
"J.30": {
|
| 142 |
+
"dtype": "float16",
|
| 143 |
+
"nbytes": 8388608,
|
| 144 |
+
"numel": 4194304,
|
| 145 |
+
"sha256_c_contiguous_little_endian_bytes": "143d3abe3632641ab2bb6b2c0582ef9d461e0b144b1a29f1f67e716645ea5f4c",
|
| 146 |
+
"shape": [
|
| 147 |
+
2048,
|
| 148 |
+
2048
|
| 149 |
+
],
|
| 150 |
+
"source_layer": 30
|
| 151 |
+
},
|
| 152 |
+
"J.32": {
|
| 153 |
+
"dtype": "float16",
|
| 154 |
+
"nbytes": 8388608,
|
| 155 |
+
"numel": 4194304,
|
| 156 |
+
"sha256_c_contiguous_little_endian_bytes": "bbb1664e19ef450f68a8202c7c6007633e160c96712866c3793cb80474f589dc",
|
| 157 |
+
"shape": [
|
| 158 |
+
2048,
|
| 159 |
+
2048
|
| 160 |
+
],
|
| 161 |
+
"source_layer": 32
|
| 162 |
+
},
|
| 163 |
+
"J.34": {
|
| 164 |
+
"dtype": "float16",
|
| 165 |
+
"nbytes": 8388608,
|
| 166 |
+
"numel": 4194304,
|
| 167 |
+
"sha256_c_contiguous_little_endian_bytes": "a2745b1142ae10a8b7fd449eba62e5530216c4026c748aed78eda69b7aeb35f9",
|
| 168 |
+
"shape": [
|
| 169 |
+
2048,
|
| 170 |
+
2048
|
| 171 |
+
],
|
| 172 |
+
"source_layer": 34
|
| 173 |
+
},
|
| 174 |
+
"J.4": {
|
| 175 |
+
"dtype": "float16",
|
| 176 |
+
"nbytes": 8388608,
|
| 177 |
+
"numel": 4194304,
|
| 178 |
+
"sha256_c_contiguous_little_endian_bytes": "c4970a9604879056b2f4493b1e8b1b0401ae115c236613b37875edf991425cdd",
|
| 179 |
+
"shape": [
|
| 180 |
+
2048,
|
| 181 |
+
2048
|
| 182 |
+
],
|
| 183 |
+
"source_layer": 4
|
| 184 |
+
},
|
| 185 |
+
"J.6": {
|
| 186 |
+
"dtype": "float16",
|
| 187 |
+
"nbytes": 8388608,
|
| 188 |
+
"numel": 4194304,
|
| 189 |
+
"sha256_c_contiguous_little_endian_bytes": "57e34398437bd673ecfe10b8c696306a706287d18307edceb4b69bf1fced1905",
|
| 190 |
+
"shape": [
|
| 191 |
+
2048,
|
| 192 |
+
2048
|
| 193 |
+
],
|
| 194 |
+
"source_layer": 6
|
| 195 |
+
},
|
| 196 |
+
"J.8": {
|
| 197 |
+
"dtype": "float16",
|
| 198 |
+
"nbytes": 8388608,
|
| 199 |
+
"numel": 4194304,
|
| 200 |
+
"sha256_c_contiguous_little_endian_bytes": "dfecda781a5e26a3573a7c7f8ff8dc173bc4b55c6e6f1a80b3c7a918d08eece6",
|
| 201 |
+
"shape": [
|
| 202 |
+
2048,
|
| 203 |
+
2048
|
| 204 |
+
],
|
| 205 |
+
"source_layer": 8
|
| 206 |
+
}
|
| 207 |
+
}
|
| 208 |
+
}
|
validation.json
ADDED
|
@@ -0,0 +1,28 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"artifact": "model.safetensors",
|
| 3 |
+
"artifact_sha256": "089d776979408f23e5377539c15aa8025d171633718ccdce709bcd3372e7942c",
|
| 4 |
+
"artifact_size_bytes": 150996824,
|
| 5 |
+
"checks": {
|
| 6 |
+
"all_source_tensors_contiguous": true,
|
| 7 |
+
"all_source_tensors_finite": true,
|
| 8 |
+
"all_source_tensors_fp16": true,
|
| 9 |
+
"all_source_tensors_shape_2048x2048": true,
|
| 10 |
+
"roundtrip_all_tensor_bit_patterns_equal": true,
|
| 11 |
+
"roundtrip_all_tensor_byte_hashes_equal": true,
|
| 12 |
+
"roundtrip_all_tensor_dtypes_equal": true,
|
| 13 |
+
"roundtrip_all_tensor_shapes_equal": true,
|
| 14 |
+
"roundtrip_all_tensor_values_equal": true,
|
| 15 |
+
"roundtrip_key_set_exact": true,
|
| 16 |
+
"safetensors_header_public_safe": true,
|
| 17 |
+
"safetensors_header_roundtrip_exact": true,
|
| 18 |
+
"source_checkpoint_sha256_exact": true,
|
| 19 |
+
"source_metadata_exact": true,
|
| 20 |
+
"source_top_level_key_set_exact": true
|
| 21 |
+
},
|
| 22 |
+
"exact_tensor_matches": 18,
|
| 23 |
+
"expected_tensor_matches": 18,
|
| 24 |
+
"ok": true,
|
| 25 |
+
"schema_version": 1,
|
| 26 |
+
"source_checkpoint_sha256": "f36a99447623e0d777c70951a9148a7a52e42e0df82942e22ce6326f63d8d664",
|
| 27 |
+
"tensor_values_changed": 0
|
| 28 |
+
}
|