jvogan commited on
Commit
6ca9813
·
0 Parent(s):

Prepare VibeThinker-3B J Lens model release

Browse files
.gitattributes ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ *.safetensors filter=lfs diff=lfs merge=lfs -text
2
+ *.png filter=lfs diff=lfs merge=lfs -text
3
+ LICENSES/QWEN-RESEARCH.txt text eol=lf whitespace=-trailing-space
.gitignore ADDED
@@ -0,0 +1,36 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Public release allowlist. Add each new intended file explicitly.
2
+ *
3
+
4
+ !/.gitignore
5
+ !/.gitattributes
6
+ !/README.md
7
+ !/NOTICE
8
+ !/THIRD_PARTY_NOTICES.md
9
+ !/SHA256SUMS
10
+ !/model.safetensors
11
+ !/evaluation.safetensors
12
+ !/lens_config.json
13
+ !/tensor_manifest.json
14
+ !/evaluation_tensor_manifest.json
15
+ !/provenance.json
16
+ !/evaluation_provenance.json
17
+ !/validation.json
18
+ !/evaluation_validation.json
19
+ !/evaluation.json
20
+ !/evaluation_compatibility.json
21
+ !/requirements.txt
22
+
23
+ !/assets/
24
+ !/assets/jlens-model-banner.png
25
+ !/assets/two-lens-files.png
26
+ !/assets/two-lens-files.svg
27
+
28
+ !/LICENSES/
29
+ !/LICENSES/APACHE-2.0.txt
30
+ !/LICENSES/QWEN-RESEARCH.txt
31
+ !/LICENSES/VIBETHINKER-LICENSE-NOTE.txt
32
+
33
+ !/scripts/
34
+ !/scripts/convert_checkpoint.py
35
+ !/scripts/convert_evaluation_checkpoint.py
36
+ !/scripts/validate_artifact.py
LICENSES/APACHE-2.0.txt ADDED
@@ -0,0 +1,201 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Apache License
2
+ Version 2.0, January 2004
3
+ http://www.apache.org/licenses/
4
+
5
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
6
+
7
+ 1. Definitions.
8
+
9
+ "License" shall mean the terms and conditions for use, reproduction,
10
+ and distribution as defined by Sections 1 through 9 of this document.
11
+
12
+ "Licensor" shall mean the copyright owner or entity authorized by
13
+ the copyright owner that is granting the License.
14
+
15
+ "Legal Entity" shall mean the union of the acting entity and all
16
+ other entities that control, are controlled by, or are under common
17
+ control with that entity. For the purposes of this definition,
18
+ "control" means (i) the power, direct or indirect, to cause the
19
+ direction or management of such entity, whether by contract or
20
+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
21
+ outstanding shares, or (iii) beneficial ownership of such entity.
22
+
23
+ "You" (or "Your") shall mean an individual or Legal Entity
24
+ exercising permissions granted by this License.
25
+
26
+ "Source" form shall mean the preferred form for making modifications,
27
+ including but not limited to software source code, documentation
28
+ source, and configuration files.
29
+
30
+ "Object" form shall mean any form resulting from mechanical
31
+ transformation or translation of a Source form, including but
32
+ not limited to compiled object code, generated documentation,
33
+ and conversions to other media types.
34
+
35
+ "Work" shall mean the work of authorship, whether in Source or
36
+ Object form, made available under the License, as indicated by a
37
+ copyright notice that is included in or attached to the work
38
+ (an example is provided in the Appendix below).
39
+
40
+ "Derivative Works" shall mean any work, whether in Source or Object
41
+ form, that is based on (or derived from) the Work and for which the
42
+ editorial revisions, annotations, elaborations, or other modifications
43
+ represent, as a whole, an original work of authorship. For the purposes
44
+ of this License, Derivative Works shall not include works that remain
45
+ separable from, or merely link (or bind by name) to the interfaces of,
46
+ the Work and Derivative Works thereof.
47
+
48
+ "Contribution" shall mean any work of authorship, including
49
+ the original version of the Work and any modifications or additions
50
+ to that Work or Derivative Works thereof, that is intentionally
51
+ submitted to Licensor for inclusion in the Work by the copyright owner
52
+ or by an individual or Legal Entity authorized to submit on behalf of
53
+ the copyright owner. For the purposes of this definition, "submitted"
54
+ means any form of electronic, verbal, or written communication sent
55
+ to the Licensor or its representatives, including but not limited to
56
+ communication on electronic mailing lists, source code control systems,
57
+ and issue tracking systems that are managed by, or on behalf of, the
58
+ Licensor for the purpose of discussing and improving the Work, but
59
+ excluding communication that is conspicuously marked or otherwise
60
+ designated in writing by the copyright owner as "Not a Contribution."
61
+
62
+ "Contributor" shall mean Licensor and any individual or Legal Entity
63
+ on behalf of whom a Contribution has been received by Licensor and
64
+ subsequently incorporated within the Work.
65
+
66
+ 2. Grant of Copyright License. Subject to the terms and conditions of
67
+ this License, each Contributor hereby grants to You a perpetual,
68
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
69
+ copyright license to reproduce, prepare Derivative Works of,
70
+ publicly display, publicly perform, sublicense, and distribute the
71
+ Work and such Derivative Works in Source or Object form.
72
+
73
+ 3. Grant of Patent License. Subject to the terms and conditions of
74
+ this License, each Contributor hereby grants to You a perpetual,
75
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
76
+ (except as stated in this section) patent license to make, have made,
77
+ use, offer to sell, sell, import, and otherwise transfer the Work,
78
+ where such license applies only to those patent claims licensable
79
+ by such Contributor that are necessarily infringed by their
80
+ Contribution(s) alone or by combination of their Contribution(s)
81
+ with the Work to which such Contribution(s) was submitted. If You
82
+ institute patent litigation against any entity (including a
83
+ cross-claim or counterclaim in a lawsuit) alleging that the Work
84
+ or a Contribution incorporated within the Work constitutes direct
85
+ or contributory patent infringement, then any patent licenses
86
+ granted to You under this License for that Work shall terminate as
87
+ of the date such litigation is filed.
88
+
89
+ 4. Redistribution. You may reproduce and distribute copies of the
90
+ Work or Derivative Works thereof in any medium, with or without
91
+ modifications, and in Source or Object form, provided that You
92
+ meet the following conditions:
93
+
94
+ (a) You must give any other recipients of the Work or
95
+ Derivative Works a copy of this License; and
96
+
97
+ (b) You must cause any modified files to carry prominent notices
98
+ stating that You changed the files; and
99
+
100
+ (c) You must retain, in the Source form of any Derivative Works
101
+ that You distribute, all copyright, patent, trademark, and
102
+ attribution notices from the Source form of the Work,
103
+ excluding those notices that do not pertain to any part of
104
+ the Derivative Works; and
105
+
106
+ (d) If the Work includes a "NOTICE" text file as part of its
107
+ distribution, then any Derivative Works that You distribute must
108
+ include a readable copy of the attribution notices contained
109
+ within such NOTICE file, excluding those notices that do not
110
+ pertain to any part of the Derivative Works, in at least one
111
+ of the following places: within a NOTICE text file distributed
112
+ as part of the Derivative Works; within the Source form or
113
+ documentation, if provided along with the Derivative Works; or,
114
+ within a display generated by the Derivative Works, if and
115
+ wherever such third-party notices normally appear. The contents
116
+ of the NOTICE file are for informational purposes only and
117
+ do not modify the License. You may add Your own attribution
118
+ notices within Derivative Works that You distribute, alongside
119
+ or as an addendum to the NOTICE text from the Work, provided
120
+ that such additional attribution notices cannot be construed
121
+ as modifying the License.
122
+
123
+ You may add Your own copyright statement to Your modifications and
124
+ may provide additional or different license terms and conditions
125
+ for use, reproduction, or distribution of Your modifications, or
126
+ for any such Derivative Works as a whole, provided Your use,
127
+ reproduction, and distribution of the Work otherwise complies with
128
+ the conditions stated in this License.
129
+
130
+ 5. Submission of Contributions. Unless You explicitly state otherwise,
131
+ any Contribution intentionally submitted for inclusion in the Work
132
+ by You to the Licensor shall be under the terms and conditions of
133
+ this License, without any additional terms or conditions.
134
+ Notwithstanding the above, nothing herein shall supersede or modify
135
+ the terms of any separate license agreement you may have executed
136
+ with Licensor regarding such Contributions.
137
+
138
+ 6. Trademarks. This License does not grant permission to use the trade
139
+ names, trademarks, service marks, or product names of the Licensor,
140
+ except as required for reasonable and customary use in describing the
141
+ origin of the Work and reproducing the content of the NOTICE file.
142
+
143
+ 7. Disclaimer of Warranty. Unless required by applicable law or
144
+ agreed to in writing, Licensor provides the Work (and each
145
+ Contributor provides its Contributions) on an "AS IS" BASIS,
146
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
147
+ implied, including, without limitation, any warranties or conditions
148
+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
149
+ PARTICULAR PURPOSE. You are solely responsible for determining the
150
+ appropriateness of using or redistributing the Work and assume any
151
+ risks associated with Your exercise of permissions under this License.
152
+
153
+ 8. Limitation of Liability. In no event and under no legal theory,
154
+ whether in tort (including negligence), contract, or otherwise,
155
+ unless required by applicable law (such as deliberate and grossly
156
+ negligent acts) or agreed to in writing, shall any Contributor be
157
+ liable to You for damages, including any direct, indirect, special,
158
+ incidental, or consequential damages of any character arising as a
159
+ result of this License or out of the use or inability to use the
160
+ Work (including but not limited to damages for loss of goodwill,
161
+ work stoppage, computer failure or malfunction, or any and all other
162
+ commercial damages or losses), even if such Contributor has been
163
+ advised of the possibility of such damages.
164
+
165
+ 9. Accepting Warranty or Additional Liability. While redistributing
166
+ the Work or Derivative Works thereof, You may choose to offer,
167
+ and charge a fee for, acceptance of support, warranty, indemnity,
168
+ or other liability obligations and/or rights consistent with this
169
+ License. However, in accepting such obligations, You may act only
170
+ on Your own behalf and on Your sole responsibility, not on behalf
171
+ of any other Contributor, and only if You agree to indemnify,
172
+ defend, and hold each Contributor harmless for any liability
173
+ incurred by, or claims asserted against, such Contributor by reason
174
+ of your accepting any such warranty or additional liability.
175
+
176
+ END OF TERMS AND CONDITIONS
177
+
178
+ APPENDIX: How to apply the Apache License to your work.
179
+
180
+ To apply the Apache License to your work, attach the following
181
+ boilerplate notice, with the fields enclosed by brackets "[]"
182
+ replaced with your own identifying information. (Don't include
183
+ the brackets!) The text should be enclosed in the appropriate
184
+ comment syntax for the file format. We also recommend that a
185
+ file or class name and description of purpose be included on the
186
+ same "printed page" as the copyright notice for easier
187
+ identification within third-party archives.
188
+
189
+ Copyright [yyyy] [name of copyright owner]
190
+
191
+ Licensed under the Apache License, Version 2.0 (the "License");
192
+ you may not use this file except in compliance with the License.
193
+ You may obtain a copy of the License at
194
+
195
+ http://www.apache.org/licenses/LICENSE-2.0
196
+
197
+ Unless required by applicable law or agreed to in writing, software
198
+ distributed under the License is distributed on an "AS IS" BASIS,
199
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
200
+ See the License for the specific language governing permissions and
201
+ limitations under the License.
LICENSES/QWEN-RESEARCH.txt ADDED
@@ -0,0 +1,54 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Qwen RESEARCH LICENSE AGREEMENT
2
+
3
+ Qwen RESEARCH LICENSE AGREEMENT Release Date: September 19, 2024
4
+
5
+ By clicking to agree or by using or distributing any portion or element of the Qwen Materials, you will be deemed to have recognized and accepted the content of this Agreement, which is effective immediately.
6
+
7
+ 1. Definitions
8
+ a. This Qwen RESEARCH LICENSE AGREEMENT (this "Agreement") shall mean the terms and conditions for use, reproduction, distribution and modification of the Materials as defined by this Agreement.
9
+ b. "We" (or "Us") shall mean Alibaba Cloud.
10
+ c. "You" (or "Your") shall mean a natural person or legal entity exercising the rights granted by this Agreement and/or using the Materials for any purpose and in any field of use.
11
+ d. "Third Parties" shall mean individuals or legal entities that are not under common control with us or you.
12
+ e. "Qwen" shall mean the large language models, and software and algorithms, consisting of trained model weights, parameters (including optimizer states), machine-learning model code, inference-enabling code, training-enabling code, fine-tuning enabling code and other elements of the foregoing distributed by us.
13
+ f. "Materials" shall mean, collectively, Alibaba Cloud's proprietary Qwen and Documentation (and any portion thereof) made available under this Agreement.
14
+ g. "Source" form shall mean the preferred form for making modifications, including but not limited to model source code, documentation source, and configuration files.
15
+ h. "Object" form shall mean any form resulting from mechanical transformation or translation of a Source form, including but not limited to compiled object code, generated documentation, and conversions to other media types.
16
+ i. "Non-Commercial" shall mean for research or evaluation purposes only.
17
+
18
+ 2. Grant of Rights
19
+ a. You are granted a non-exclusive, worldwide, non-transferable and royalty-free limited license under Alibaba Cloud's intellectual property or other rights owned by us embodied in the Materials to use, reproduce, distribute, copy, create derivative works of, and make modifications to the Materials FOR NON-COMMERCIAL PURPOSES ONLY.
20
+ b. If you are commercially using the Materials, you shall request a license from us.
21
+
22
+ 3. Redistribution
23
+ You may distribute copies or make the Materials, or derivative works thereof, available as part of a product or service that contains any of them, with or without modifications, and in Source or Object form, provided that you meet the following conditions:
24
+ a. You shall give any other recipients of the Materials or derivative works a copy of this Agreement;
25
+ b. You shall cause any modified files to carry prominent notices stating that you changed the files;
26
+ c. You shall retain in all copies of the Materials that you distribute the following attribution notices within a "Notice" text file distributed as a part of such copies: "Qwen is licensed under the Qwen RESEARCH LICENSE AGREEMENT, Copyright (c) Alibaba Cloud. All Rights Reserved."; and
27
+ d. You may add your own copyright statement to your modifications and may provide additional or different license terms and conditions for use, reproduction, or distribution of your modifications, or for any such derivative works as a whole, provided your use, reproduction, and distribution of the work otherwise complies with the terms and conditions of this Agreement.
28
+
29
+ 4. Rules of use
30
+ a. The Materials may be subject to export controls or restrictions in China, the United States or other countries or regions. You shall comply with applicable laws and regulations in your use of the Materials.
31
+ b. If you use the Materials or any outputs or results therefrom to create, train, fine-tune, or improve an AI model that is distributed or made available, you shall prominently display “Built with Qwen” or “Improved using Qwen” in the related product documentation.
32
+
33
+ 5. Intellectual Property
34
+ a. We retain ownership of all intellectual property rights in and to the Materials and derivatives made by or for us. Conditioned upon compliance with the terms and conditions of this Agreement, with respect to any derivative works and modifications of the Materials that are made by you, you are and will be the owner of such derivative works and modifications.
35
+ b. No trademark license is granted to use the trade names, trademarks, service marks, or product names of us, except as required to fulfill notice requirements under this Agreement or as required for reasonable and customary use in describing and redistributing the Materials.
36
+ c. If you commence a lawsuit or other proceedings (including a cross-claim or counterclaim in a lawsuit) against us or any entity alleging that the Materials or any output therefrom, or any part of the foregoing, infringe any intellectual property or other right owned or licensable by you, then all licenses granted to you under this Agreement shall terminate as of the date such lawsuit or other proceeding is commenced or brought.
37
+
38
+ 6. Disclaimer of Warranty and Limitation of Liability
39
+ a. We are not obligated to support, update, provide training for, or develop any further version of the Qwen Materials or to grant any license thereto.
40
+ b. THE MATERIALS ARE PROVIDED "AS IS" WITHOUT ANY EXPRESS OR IMPLIED WARRANTY OF ANY KIND INCLUDING WARRANTIES OF MERCHANTABILITY, NONINFRINGEMENT, OR FITNESS FOR A PARTICULAR PURPOSE. WE MAKE NO WARRANTY AND ASSUME NO RESPONSIBILITY FOR THE SAFETY OR STABILITY OF THE MATERIALS AND ANY OUTPUT THEREFROM.
41
+ c. IN NO EVENT SHALL WE BE LIABLE TO YOU FOR ANY DAMAGES, INCLUDING, BUT NOT LIMITED TO ANY DIRECT, OR INDIRECT, SPECIAL OR CONSEQUENTIAL DAMAGES ARISING FROM YOUR USE OR INABILITY TO USE THE MATERIALS OR ANY OUTPUT OF IT, NO MATTER HOW IT’S CAUSED.
42
+ d. You will defend, indemnify and hold harmless us from and against any claim by any third party arising out of or related to your use or distribution of the Materials.
43
+
44
+ 7. Survival and Termination.
45
+ a. The term of this Agreement shall commence upon your acceptance of this Agreement or access to the Materials and will continue in full force and effect until terminated in accordance with the terms and conditions herein.
46
+ b. We may terminate this Agreement if you breach any of the terms or conditions of this Agreement. Upon termination of this Agreement, you must delete and cease use of the Materials. Sections 6 and 8 shall survive the termination of this Agreement.
47
+
48
+ 8. Governing Law and Jurisdiction.
49
+ a. This Agreement and any dispute arising out of or relating to it will be governed by the laws of China, without regard to conflict of law principles, and the UN Convention on Contracts for the International Sale of Goods does not apply to this Agreement.
50
+ b. The People's Courts in Hangzhou City shall have exclusive jurisdiction over any dispute arising out of this Agreement.
51
+
52
+ 9. Other Terms and Conditions.
53
+ a. Any arrangements, understandings, or agreements regarding the Material not stated herein are separate from and independent of the terms and conditions of this Agreement. You shall request a separate license from us, if you use the Materials in ways not expressly agreed to in this Agreement.
54
+ b. We shall not be bound by any additional or different terms or conditions communicated by you unless expressly agreed.
LICENSES/VIBETHINKER-LICENSE-NOTE.txt ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ VibeThinker-3B license metadata note
2
+
3
+ Upstream repository: WeiboAI/VibeThinker-3B
4
+ Pinned revision: 77bd2cced09193c8b9a59a32bd8577bbd1f3e01c
5
+ Pinned revision URL:
6
+ https://huggingface.co/WeiboAI/VibeThinker-3B/tree/77bd2cced09193c8b9a59a32bd8577bbd1f3e01c
7
+
8
+ The README model-card metadata at this revision declares:
9
+
10
+ license: mit
11
+
12
+ The pinned repository tree does not contain a standalone LICENSE file or
13
+ NOTICE file. It does not contain a standalone copyright notice that identifies
14
+ a holder and year.
15
+
16
+ This file records the upstream metadata and file inventory. It is not an
17
+ upstream license text. This release does not assign a VibeThinker copyright
18
+ holder or year.
19
+
20
+ The Qwen Research License Agreement used for this distribution is included
21
+ separately at LICENSES/QWEN-RESEARCH.txt.
NOTICE ADDED
@@ -0,0 +1,25 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ VibeThinker-3B Jacobian Lens
2
+
3
+ This package contains a fitted average-Jacobian readout artifact derived from
4
+ activations of WeiboAI/VibeThinker-3B. The upstream model card identifies
5
+ Qwen/Qwen2.5-Coder-3B as its base model.
6
+
7
+ Qwen is licensed under the Qwen RESEARCH LICENSE AGREEMENT, Copyright (c)
8
+ Alibaba Cloud. All Rights Reserved.
9
+
10
+ Built with Qwen.
11
+
12
+ The pinned VibeThinker model-card metadata declares `license: mit`. The pinned
13
+ repository tree contains no standalone LICENSE file, NOTICE file, or copyright
14
+ notice that identifies a holder and year. The metadata record and pinned
15
+ revision URL are in LICENSES/VIBETHINKER-LICENSE-NOTE.txt.
16
+
17
+ Modification notice: model.safetensors is a derived Jacobian lens tensor
18
+ artifact, not a copy of the upstream language-model weights. It was converted
19
+ losslessly from the fitted PyTorch serialization to safetensors; every FP16
20
+ tensor bit pattern was preserved. No VibeThinker or Qwen base-model weights are
21
+ included in this repository.
22
+
23
+ The lens was fit using Anthropic PBC's jacobian-lens reference implementation,
24
+ licensed under the Apache License 2.0. The fitting corpus was sourced from
25
+ WikiText-103; no source prompt text is redistributed here.
README.md ADDED
@@ -0,0 +1,322 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: other
3
+ license_name: qwen-research-license
4
+ license_link: https://huggingface.co/JacobMolBio/vibethinker-3b-jlens-model/blob/main/LICENSES/QWEN-RESEARCH.txt
5
+ tags:
6
+ - vibethinker-3b
7
+ - jacobian-lens
8
+ - mechanistic-interpretability
9
+ - interpretability
10
+ - qwen2
11
+ - safetensors
12
+ ---
13
+
14
+ ![Stacked VibeThinker-3B and J Lens title beside a folded black-layer sculpture divided by a cyan plane. The subtitle reads: Watch the model’s answer take shape, layer by layer. Weights, traces, code, and a viewer.](assets/jlens-model-banner.png)
15
+
16
+ # VibeThinker-3B J Lens
17
+
18
+ The J Lens is a Jacobian lens fitted to
19
+ [`WeiboAI/VibeThinker-3B`](https://huggingface.co/WeiboAI/VibeThinker-3B/tree/77bd2cced09193c8b9a59a32bd8577bbd1f3e01c).
20
+ Pick one of the 18 fitted source layers and a token position, and the lens
21
+ decodes that residual-stream activation into a ranked list of vocabulary
22
+ tokens. Following one position across layers shows how the decoded ranking
23
+ changes on the way to the model's final output.
24
+
25
+ The lens contains 18 matrices, one for each even-numbered source layer from 0
26
+ through 34. Each matrix maps its source-layer residual-stream activation into
27
+ layer-35 coordinates. VibeThinker-3B's final normalization and vocabulary
28
+ projection produce the ranked token scores.
29
+
30
+ This repository contains the fitted lens in two Safetensors precisions.
31
+ `model.safetensors` is the FP16 lens used to capture the released traces.
32
+ `evaluation.safetensors` is the FP32 lens used for the recorded readout
33
+ evaluation. Casting each FP32 matrix to FP16 reproduces the FP16 file exactly;
34
+ both are included so every published result stays paired with the tensor
35
+ values that produced it.
36
+
37
+ The companion trace repository contains six saved traces. Each trace pairs a
38
+ prompt and its generated response with the top 12 decoded tokens at each
39
+ captured position for the 18 source layers and the final model layer.
40
+
41
+ ## Choose a path
42
+
43
+ | Goal | Use |
44
+ |---|---|
45
+ | Browse saved traces | Open the [static viewer](https://jvogan.github.io/vibethinker-3b-jlens). It reads released trace JSON in the browser; it does not download model weights or run inference. |
46
+ | Run a new prompt locally | Use `scripts/render_slice.py` in the [source repository](https://github.com/jvogan/vibethinker-3b-jlens). It uses the pinned VibeThinker-3B base-model weights and `model.safetensors`, the FP16 J Lens. |
47
+ | Load the released lens in Python | Use `model.safetensors`, the FP16 trace lens. |
48
+ | Recalculate the recorded readout statistics | Use `evaluation.safetensors`, the FP32 evaluation lens, with the released rank rows and `scripts/recalculate_readout.py` in the trace repository. |
49
+
50
+ ## Companion repositories
51
+
52
+ | Release component | Location |
53
+ |---|---|
54
+ | Source code and Pages source | [https://github.com/jvogan/vibethinker-3b-jlens](https://github.com/jvogan/vibethinker-3b-jlens) |
55
+ | Captured trace dataset | [https://huggingface.co/datasets/JacobMolBio/vibethinker-3b-jlens-traces](https://huggingface.co/datasets/JacobMolBio/vibethinker-3b-jlens-traces) |
56
+ | Static Pages site | [https://jvogan.github.io/vibethinker-3b-jlens](https://jvogan.github.io/vibethinker-3b-jlens) |
57
+ | Hugging Face model repository ID | [JacobMolBio/vibethinker-3b-jlens-model](https://huggingface.co/JacobMolBio/vibethinker-3b-jlens-model) |
58
+
59
+ ## Artifact identity
60
+
61
+ | Field | Trace artifact | Evaluation artifact |
62
+ |---|---|---|
63
+ | File | `model.safetensors` | `evaluation.safetensors` |
64
+ | SHA-256 | `089d776979408f23e5377539c15aa8025d171633718ccdce709bcd3372e7942c` | `0cc184eb65d273ac8bfee5450a141c8cdf8dd6ec8caa68d7d47260ef621777d1` |
65
+ | Size | 150,996,824 bytes | 301,991,904 bytes |
66
+ | Dtype | FP16 | FP32 |
67
+ | Role | Captured trace readouts | Recorded readout evaluation |
68
+
69
+ [![Use model.safetensors for FP16 trace work and evaluation.safetensors for the recorded FP32 evaluation.](assets/two-lens-files.png)](assets/two-lens-files.svg)
70
+
71
+ Both artifacts contain 18 matrices with shape 2,048 × 2,048. Their source
72
+ layers are every even layer from 0 through 34, and their target layer is 35.
73
+ The lens was fit on 1,000 prompts for model revision
74
+ `77bd2cced09193c8b9a59a32bd8577bbd1f3e01c`. The revision binding was
75
+ reconstructed from the Hub head after the fit;
76
+ [`provenance.json`](provenance.json) records that binding and its limits.
77
+
78
+ ## Files
79
+
80
+ - `model.safetensors` — the 18 losslessly reserialized FP16 lens matrices.
81
+ - `evaluation.safetensors` — the 18 losslessly reserialized FP32 evaluation matrices.
82
+ - `lens_config.json` — model binding, layer layout, fit parameters, and artifact ID.
83
+ - `tensor_manifest.json` — shape, dtype, size, and raw-byte SHA-256 for every tensor.
84
+ - `evaluation_tensor_manifest.json` — the corresponding FP32 tensor manifest.
85
+ - `evaluation_provenance.json` — derivation and source-hash bindings.
86
+ - `evaluation_compatibility.json` — the FP32-to-FP16 comparison for every layer.
87
+ - `evaluation_validation.json` — the exact FP32 conversion checks.
88
+ - `provenance.json` — fit, corpus, software, and conversion provenance.
89
+ - `validation.json` — exhaustive source-to-safetensors comparison result.
90
+ - `evaluation.json` — recorded readout-only evaluation and its scope.
91
+ - `SHA256SUMS` — checksums for both Safetensors files.
92
+ - `requirements.txt` — pinned packages for loading and validation.
93
+ - `scripts/validate_artifact.py` — validates the artifact without pickle.
94
+ - `scripts/convert_checkpoint.py` — reproducible checkpoint conversion for a user-provided source file.
95
+ - `scripts/convert_evaluation_checkpoint.py` — reproducible FP32 conversion.
96
+ - `LICENSES/QWEN-RESEARCH.txt` — complete distribution terms for the artifacts.
97
+ - `NOTICE` and `THIRD_PARTY_NOTICES.md` — required attribution and license context.
98
+
99
+ ## Tensor layout
100
+
101
+ The keys are `J.<source_layer>`:
102
+
103
+ ```text
104
+ J.0, J.2, J.4, J.6, J.8, J.10, J.12, J.14, J.16,
105
+ J.18, J.20, J.22, J.24, J.26, J.28, J.30, J.32, J.34
106
+ ```
107
+
108
+ Each matrix maps the post-transformer-block residual at its source layer into
109
+ the layer-35 residual basis. With row-major batches of residual vectors, the
110
+ transport used by the reference implementation is:
111
+
112
+ ```python
113
+ transported = residual.float() @ J.float().T
114
+ logits = unembed(transported)
115
+ ```
116
+
117
+ Both Safetensors headers contain the model ID and revision, source and target
118
+ layers, width, prompt count, source-checkpoint hash, and key pattern.
119
+
120
+ `requirements.txt` pins the runtime used to create and validate
121
+ `evaluation.safetensors`. `provenance.json` records the earlier runtime used
122
+ to convert `model.safetensors`.
123
+
124
+ ## Install and load
125
+
126
+ The `jlens` Python package comes from the companion source repository. This
127
+ repository contains the fitted tensors and the dependencies needed to inspect
128
+ and validate them.
129
+
130
+ From this repository checkout, install both parts:
131
+
132
+ ```bash
133
+ export JLENS_CODE_REPO_URL="https://github.com/jvogan/vibethinker-3b-jlens"
134
+ git clone "$JLENS_CODE_REPO_URL" ../vibethinker-3b-jlens
135
+ python3 -m pip install -r requirements.txt
136
+ python3 -m pip install ../vibethinker-3b-jlens
137
+ ```
138
+
139
+ Load the local artifact:
140
+
141
+ ```python
142
+ from jlens import JacobianLens
143
+
144
+ lens = JacobianLens.load("model.safetensors")
145
+ print(lens.source_layers)
146
+ ```
147
+
148
+ Or load an immutable Hugging Face revision, setting `JLENS_MODEL_REVISION`
149
+ to the model-repository commit recorded in the source repository's
150
+ `release-manifest.json`:
151
+
152
+ ```bash
153
+ export JLENS_MODEL_REPO_ID="JacobMolBio/vibethinker-3b-jlens-model"
154
+ export JLENS_MODEL_REVISION="<commit from the source repository's release-manifest.json>"
155
+ ```
156
+
157
+ ```python
158
+ import os
159
+
160
+ from jlens import JacobianLens
161
+
162
+ lens = JacobianLens.from_pretrained(
163
+ os.environ["JLENS_MODEL_REPO_ID"],
164
+ filename="model.safetensors",
165
+ revision=os.environ["JLENS_MODEL_REVISION"],
166
+ )
167
+ ```
168
+
169
+ To inspect tensors without the companion package, install `requirements.txt`
170
+ and use Safetensors directly:
171
+
172
+ ```python
173
+ from safetensors import safe_open
174
+
175
+ with safe_open("model.safetensors", framework="pt", device="cpu") as handle:
176
+ metadata = handle.metadata()
177
+ jacobians = {
178
+ int(key.split(".", 1)[1]): handle.get_tensor(key).clone()
179
+ for key in handle.keys()
180
+ }
181
+
182
+ assert metadata["model_revision"] == (
183
+ "77bd2cced09193c8b9a59a32bd8577bbd1f3e01c"
184
+ )
185
+ assert sorted(jacobians) == list(range(0, 36, 2))
186
+ ```
187
+
188
+ Load VibeThinker itself from the upstream repository at the exact bound
189
+ revision; this repository does not include VibeThinker or Qwen weights:
190
+
191
+ ```python
192
+ from transformers import AutoModelForCausalLM, AutoTokenizer
193
+
194
+ model_id = "WeiboAI/VibeThinker-3B"
195
+ revision = "77bd2cced09193c8b9a59a32bd8577bbd1f3e01c"
196
+ tokenizer = AutoTokenizer.from_pretrained(model_id, revision=revision)
197
+ model = AutoModelForCausalLM.from_pretrained(
198
+ model_id,
199
+ revision=revision,
200
+ dtype="auto",
201
+ )
202
+ ```
203
+
204
+ Review and accept the upstream terms before downloading or using the model.
205
+
206
+ ## Viewer and new-prompt compute
207
+
208
+ The companion Pages site is a static viewer for captured trace JSON. It
209
+ parses, filters, and renders those records in the browser; loading a local
210
+ JSON file uses the browser's file reader and does not upload the file. The
211
+ static site cannot analyze a new prompt. That requires the companion source
212
+ package, this lens artifact, the pinned VibeThinker weights, and local Python
213
+ compute.
214
+
215
+ ## Fit method
216
+
217
+ The lens was fit with the Apache-2.0 Anthropic Jacobian Lens reference
218
+ implementation at commit
219
+ [`581d398613e5602a5af361e1c34d3a92ea82ba8e`](https://github.com/anthropics/jacobian-lens/tree/581d398613e5602a5af361e1c34d3a92ea82ba8e).
220
+ For each fitted source layer, the estimator sums cotangents over valid causal
221
+ targets at or after a source position, averages over valid source positions,
222
+ and gives each prompt equal weight.
223
+
224
+ - Hook point: post-transformer-block output residual
225
+ - Source layers: every even layer from 0 through 34
226
+ - Target layer: 35
227
+ - Sequence cap: 128 tokens
228
+ - Leading positions skipped: 16
229
+ - Final position excluded from fitting
230
+ - Dimension batch: 8
231
+ - Fit runtime dtype: BF16
232
+ - Trace artifact dtype: FP16
233
+ - Evaluation artifact dtype: FP32
234
+ - Fit manifest: 1,000 examples from WikiText-103 raw train, minimum 600
235
+ characters
236
+
237
+ [`provenance.json`](provenance.json) records the fit manifest, a later local
238
+ rematerialization of the prompt set, and the hash checks that bind them.
239
+ Prompt text is not included in this repository.
240
+
241
+ ## Validation
242
+
243
+ Validation checked the following for every tensor in both artifacts:
244
+
245
+ - exact key set;
246
+ - exact shape and each artifact's declared FP16 or FP32 dtype;
247
+ - exact numerical equality with the recorded source checkpoint;
248
+ - exact FP16 or FP32 bit-pattern equality;
249
+ - exact raw tensor-byte SHA-256;
250
+ - contiguity and full finiteness; and
251
+ - a Safetensors header free of private source paths.
252
+
253
+ All 18 tensors in each artifact passed, and zero values changed. The
254
+ FP32-to-FP16 cast also matches `model.safetensors` exactly for all 18 layers.
255
+ Recheck both artifacts with:
256
+
257
+ ```bash
258
+ python3 scripts/validate_artifact.py
259
+ ```
260
+
261
+ The validator verifies both file checksums, cross-file model and artifact
262
+ bindings, header metadata, tensor shapes and hashes, license and notice files,
263
+ and FP32-to-FP16 compatibility.
264
+
265
+ ## Evaluation
266
+
267
+ The recorded evaluation measures ranked-token readouts: for each prompt in a
268
+ 551-item suite, does a known target term rank highly in the decoded tokens?
269
+ The paired token-target and shuffled-layer mapping checks passed for the
270
+ selected band at layers 24, 26, 28, 30, 32, and 34. The comparison with the
271
+ ordinary logit lens did not establish improvement.
272
+ [`evaluation.json`](evaluation.json) records this readout-only scope as
273
+ `validated_readout_only`.
274
+
275
+ The 551 items cover association, multihop, multilingual,
276
+ order-of-operations, poetry, and typo prompts, each pairing a prompt with one
277
+ or more target terms. A deterministic split assigned 377 eligible items to
278
+ the test set; the band was selected on development items. The companion trace
279
+ repository contains all evaluation inputs, the 50,050 readout rows, the
280
+ aggregate metrics, the bootstrap intervals, and the exact method, and its
281
+ `scripts/recalculate_readout.py` recomputes every recorded statistic from the
282
+ released rows. Regenerating the rows themselves requires the pinned base
283
+ model, `evaluation.safetensors`, and a separate evaluator implementation.
284
+
285
+ The recorded metrics bind to `evaluation.safetensors`
286
+ (SHA-256 `0cc184eb65d273ac8bfee5450a141c8cdf8dd6ec8caa68d7d47260ef621777d1`,
287
+ from source FP32 checkpoint SHA-256
288
+ `8f752032a26a5196c1cb447ef63f01e8a29820ff0c57178dd80c6e26d32b12a9`);
289
+ they do not evaluate `model.safetensors`. The evaluation does not measure
290
+ general model accuracy or test causal steering or free-generation behavior.
291
+
292
+ The lens is stride-2, was fit on 1,000 prompts, and is model-revision
293
+ specific. Single-cell readouts can be noisy; compare patterns across nearby
294
+ layers and positions before interpreting an isolated token.
295
+
296
+ ## License and attribution
297
+
298
+ The Qwen Research License applies to this distribution and limits use to
299
+ non-commercial research and evaluation. Commercial use requires a separate
300
+ license from the upstream rights holder. Redistribution must include the
301
+ complete [Qwen Research License](LICENSES/QWEN-RESEARCH.txt) and retain the
302
+ required notice in [`NOTICE`](NOTICE). Built with Qwen. No VibeThinker or
303
+ Qwen base-model weights are redistributed here.
304
+
305
+ The pinned VibeThinker model-card metadata declares `license: mit` and names
306
+ `Qwen/Qwen2.5-Coder-3B` as its base model; the pinned tree contains no
307
+ standalone license file.
308
+ [`LICENSES/VIBETHINKER-LICENSE-NOTE.txt`](LICENSES/VIBETHINKER-LICENSE-NOTE.txt)
309
+ records that metadata and the pinned revision URL.
310
+
311
+ The fitting implementation is Anthropic's Apache-2.0
312
+ [`jacobian-lens`](https://github.com/anthropics/jacobian-lens). WikiText-103
313
+ is available under CC BY-SA 4.0; no source text is redistributed here. See
314
+ [`THIRD_PARTY_NOTICES.md`](THIRD_PARTY_NOTICES.md) for details and links.
315
+
316
+ ## References
317
+
318
+ - Anthropic, [*Verbalizable Representations Form a Global Workspace in Language Models*](https://transformer-circuits.pub/2026/workspace/index.html)
319
+ - Anthropic, [`jacobian-lens`](https://github.com/anthropics/jacobian-lens)
320
+ - [`WeiboAI/VibeThinker-3B`](https://huggingface.co/WeiboAI/VibeThinker-3B/tree/77bd2cced09193c8b9a59a32bd8577bbd1f3e01c)
321
+ - [`Qwen/Qwen2.5-Coder-3B`](https://huggingface.co/Qwen/Qwen2.5-Coder-3B)
322
+ - [`Salesforce/wikitext`](https://huggingface.co/datasets/Salesforce/wikitext)
SHA256SUMS ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ 089d776979408f23e5377539c15aa8025d171633718ccdce709bcd3372e7942c model.safetensors
2
+ 0cc184eb65d273ac8bfee5450a141c8cdf8dd6ec8caa68d7d47260ef621777d1 evaluation.safetensors
THIRD_PARTY_NOTICES.md ADDED
@@ -0,0 +1,53 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Third-party notices
2
+
3
+ This document records upstream materials associated with the artifact. It does
4
+ not replace their license texts or expand the rights they grant.
5
+
6
+ ## VibeThinker-3B and Qwen2.5-Coder-3B
7
+
8
+ The lens was fit from activations of
9
+ [`WeiboAI/VibeThinker-3B`](https://huggingface.co/WeiboAI/VibeThinker-3B/tree/77bd2cced09193c8b9a59a32bd8577bbd1f3e01c), bound
10
+ for reproducibility to revision
11
+ `77bd2cced09193c8b9a59a32bd8577bbd1f3e01c`. The VibeThinker card identifies
12
+ [`Qwen/Qwen2.5-Coder-3B`](https://huggingface.co/Qwen/Qwen2.5-Coder-3B) as its
13
+ base model.
14
+
15
+ The pinned VibeThinker model-card metadata declares `license: mit`. The pinned
16
+ tree contains no standalone LICENSE file, NOTICE file, or copyright notice
17
+ that identifies a holder and year.
18
+ [`LICENSES/VIBETHINKER-LICENSE-NOTE.txt`](LICENSES/VIBETHINKER-LICENSE-NOTE.txt)
19
+ records the declaration, pinned revision, and file-availability boundary. It
20
+ is not an upstream license text.
21
+
22
+ The Qwen base model is subject to the Qwen Research License. This repository
23
+ therefore uses `license: other` and applies the non-commercial
24
+ research/evaluation restriction. The complete Qwen terms are included at
25
+ [`LICENSES/QWEN-RESEARCH.txt`](LICENSES/QWEN-RESEARCH.txt).
26
+
27
+ Required notice:
28
+
29
+ > Qwen is licensed under the Qwen RESEARCH LICENSE AGREEMENT, Copyright (c)
30
+ > Alibaba Cloud. All Rights Reserved.
31
+
32
+ Built with Qwen.
33
+
34
+ No upstream VibeThinker or Qwen model weights are included.
35
+
36
+ ## Anthropic Jacobian Lens
37
+
38
+ The artifact was fit using Anthropic PBC's
39
+ [`jacobian-lens`](https://github.com/anthropics/jacobian-lens) reference
40
+ implementation at commit
41
+ `581d398613e5602a5af361e1c34d3a92ea82ba8e`.
42
+
43
+ Copyright 2026 Anthropic PBC. Licensed under the Apache License, Version 2.0.
44
+ The full license is included at
45
+ [`LICENSES/APACHE-2.0.txt`](LICENSES/APACHE-2.0.txt).
46
+
47
+ ## WikiText-103
48
+
49
+ The fitting corpus was selected from the WikiText-103 raw train split in
50
+ [`Salesforce/wikitext`](https://huggingface.co/datasets/Salesforce/wikitext).
51
+ The dataset card states that WikiText is available under the Creative Commons
52
+ Attribution-ShareAlike 4.0 license. This repository records only corpus
53
+ provenance and cryptographic digests; it does not redistribute the prompt text.
assets/jlens-model-banner.png ADDED

Git LFS Details

  • SHA256: f27c4abb0c84481c2b0e67ed0c390716905bd1fe6fbcb09f2a65bbe15ecdc935
  • Pointer size: 131 Bytes
  • Size of remote file: 805 kB
assets/two-lens-files.png ADDED

Git LFS Details

  • SHA256: b8b0657697c757e952f0cf79a7e5b2fa75709473a851e2fb076365ccc3794f5c
  • Pointer size: 130 Bytes
  • Size of remote file: 74.8 kB
assets/two-lens-files.svg ADDED
evaluation.json ADDED
@@ -0,0 +1,144 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "artifact_kind": "frozen_vibethinker_v1_readout_reference",
3
+ "claim_boundary": "This reference records the V1 readout results. It excludes causal-assay results and does not establish free-generation steering or a global workspace.",
4
+ "classification": "validated_readout_only",
5
+ "coverage": {
6
+ "locked_test_items": 377,
7
+ "readout_items_completed": 539,
8
+ "readout_items_no_eligible_targets": 12,
9
+ "released_prompts": 551
10
+ },
11
+ "coverage_scope": "frozen_readout_source_run_not_the_100_prompt_ui_test_pack",
12
+ "task_definition": {
13
+ "aggregate_mean_reciprocal_rank": "mean_across_eligible_target_terms_of_reciprocal_best_rank",
14
+ "eligible_target": "at_least_one_candidate_form_tokenizes_to_exactly_one_token",
15
+ "final_model": "rank_target_terms_in_the_model_next_token_logits_at_the_score_position",
16
+ "item": "one_prompt_with_one_or_more_target_terms",
17
+ "layer_scope_reduction": "best_target_rank_across_layers_in_the_reported_scope",
18
+ "no_eligible_target_item": "none_of_the_item_target_terms_has_an_eligible_single_token_form",
19
+ "paired_bootstrap_mean_reciprocal_rank": "within_item_mean_of_reciprocal_best_rank_across_eligible_target_terms",
20
+ "pass_at_k": "mean_across_items_of_the_fraction_of_item_target_terms_with_best_rank_at_most_k",
21
+ "score_position": {
22
+ "default": "final_prompt_token",
23
+ "poetry": "last_newline_token"
24
+ },
25
+ "split": {
26
+ "dev_fraction": 0.3,
27
+ "method": "sha256_stable_split",
28
+ "seed": "vibethinker-jlens-v1"
29
+ },
30
+ "suites": ["association", "multihop", "multilingual", "order-ops", "poetry", "typo"],
31
+ "target_candidate_forms": [
32
+ "original_lowercase_and_capitalized_forms_with_and_without_leading_space",
33
+ "order_ops_also_adds_configured_operation_synonyms_and_number_word_digit_forms"
34
+ ],
35
+ "target_rank": "best_one_based_vocabulary_rank_among_eligible_target_token_ids"
36
+ },
37
+ "evidence": {
38
+ "incremental_over_ordinary_logit_lens": false,
39
+ "layer_mapping_specific_vs_shuffled_jacobian": true,
40
+ "paired_bootstrap": {
41
+ "confidence": 0.95,
42
+ "input": "paired_item_metric_differences",
43
+ "interval": "percentile",
44
+ "metrics": ["pass@10", "mean_reciprocal_rank"],
45
+ "samples": 2000,
46
+ "scope": {
47
+ "kind": "selected_band",
48
+ "source_layers": [24, 26, 28, 30, 32, 34]
49
+ },
50
+ "split": "test"
51
+ },
52
+ "paired_bootstrap_decisions": {
53
+ "incremental_over_ordinary_logit_lens": {
54
+ "comparison": "jlens_band_minus_logit_lens_band",
55
+ "criterion": "at_least_one_lower_bound_greater_than_zero",
56
+ "met": false
57
+ },
58
+ "layer_mapping_specific_vs_shuffled_jacobian": {
59
+ "comparison": "jlens_band_minus_shuffled_layer_band",
60
+ "criterion": "both_lower_bounds_greater_than_zero",
61
+ "met": true
62
+ },
63
+ "token_specific_vs_shuffled_target": {
64
+ "comparison": "jlens_band_minus_shuffled_token_band",
65
+ "criterion": "both_lower_bounds_greater_than_zero",
66
+ "met": true
67
+ }
68
+ },
69
+ "specificity_scope": {
70
+ "kind": "selected_band",
71
+ "source_layers": [24, 26, 28, 30, 32, 34]
72
+ },
73
+ "token_specific_vs_shuffled_target": true,
74
+ "validated_readout_signal": true
75
+ },
76
+ "lens_sha256": "8f752032a26a5196c1cb447ef63f01e8a29820ff0c57178dd80c6e26d32b12a9",
77
+ "lens_variant": "fp32_evaluation_safetensors_included",
78
+ "released_lens_artifact": {
79
+ "filename": "evaluation.safetensors",
80
+ "format": "safetensors",
81
+ "sha256": "0cc184eb65d273ac8bfee5450a141c8cdf8dd6ec8caa68d7d47260ef621777d1",
82
+ "size_bytes": 301991904,
83
+ "source_checkpoint_sha256": "8f752032a26a5196c1cb447ef63f01e8a29820ff0c57178dd80c6e26d32b12a9",
84
+ "tensor_conversion": "lossless_fp32_reserialization"
85
+ },
86
+ "source_artifact_availability": {
87
+ "fp32_compatibility_check_output_included": true,
88
+ "fp32_compatibility_check_output_path": "evaluation_compatibility.json",
89
+ "fp32_derivation_record_included": true,
90
+ "fp32_derivation_record_path": "evaluation_provenance.json",
91
+ "fp32_evaluation_lens_included": true,
92
+ "fp32_evaluation_lens_path": "evaluation.safetensors",
93
+ "item_level_evaluation_rows_included": false,
94
+ "item_level_evaluation_rows_location": "companion_trace_repository:data/evaluation-results/readout-trials.jsonl",
95
+ "paired_bootstrap_interval_bounds_included": false,
96
+ "paired_bootstrap_interval_bounds_location": "companion_trace_repository:data/evaluation-results/readout-bootstrap-intervals.json",
97
+ "source_evaluation_bundle_included": false,
98
+ "source_hashes_included": true,
99
+ "source_hashes_location": "evaluation_provenance.json_and_companion_trace_repository:data/readout-reference.json",
100
+ "release_content": "fp16_trace_lens_fp32_evaluation_lens_aggregate_reference_provenance_and_compatibility"
101
+ },
102
+ "metrics": {
103
+ "jlens_all_layers": {
104
+ "mean_reciprocal_rank": 0.027840488103301416,
105
+ "n_items": 377,
106
+ "pass_at_1": 0.009283819628647215,
107
+ "pass_at_10": 0.08819628647214854,
108
+ "pass_at_50": 0.1655614500442087
109
+ },
110
+ "jlens_selected_band": {
111
+ "mean_reciprocal_rank": 0.01888596779741093,
112
+ "n_items": 377,
113
+ "pass_at_1": 0.003978779840848806,
114
+ "pass_at_10": 0.0629973474801061,
115
+ "pass_at_50": 0.11914235190097258
116
+ },
117
+ "logit_lens_all_layers": {
118
+ "mean_reciprocal_rank": 0.027249274540932285,
119
+ "n_items": 377,
120
+ "pass_at_1": 0.007294429708222812,
121
+ "pass_at_10": 0.08377541998231654,
122
+ "pass_at_50": 0.16114058355437666
123
+ },
124
+ "shuffled_layer_all_layers": {
125
+ "mean_reciprocal_rank": 0.042073366910739256,
126
+ "n_items": 377,
127
+ "pass_at_1": 0.04509283819628647,
128
+ "pass_at_10": 0.08819628647214854,
129
+ "pass_at_50": 0.1823607427055703
130
+ },
131
+ "final_model": {
132
+ "mean_reciprocal_rank": 0.019635164837169684,
133
+ "n_items": 377,
134
+ "pass_at_1": 0.003978779840848806,
135
+ "pass_at_10": 0.0665340406719717,
136
+ "pass_at_50": 0.10587975243147656
137
+ }
138
+ },
139
+ "model": "WeiboAI/VibeThinker-3B",
140
+ "model_revision": "77bd2cced09193c8b9a59a32bd8577bbd1f3e01c",
141
+ "note": "The recorded readout metrics bind to evaluation.safetensors, the FP32 evaluation lens in this repository. They do not evaluate model.safetensors, the FP16 lens used for the captured traces.",
142
+ "selected_band": [24, 26, 28, 30, 32, 34],
143
+ "schema_version": 1
144
+ }
evaluation.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0cc184eb65d273ac8bfee5450a141c8cdf8dd6ec8caa68d7d47260ef621777d1
3
+ size 301991904
evaluation_compatibility.json ADDED
@@ -0,0 +1,105 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "all_layer_casts_match_exactly": true,
3
+ "artifact_kind": "fp32_to_fp16_lens_compatibility",
4
+ "conversion": "IEEE_FP32_to_FP16_round_to_nearest_even",
5
+ "fp16_artifact": "model.safetensors",
6
+ "fp16_artifact_sha256": "089d776979408f23e5377539c15aa8025d171633718ccdce709bcd3372e7942c",
7
+ "fp32_artifact": "evaluation.safetensors",
8
+ "fp32_artifact_sha256": "0cc184eb65d273ac8bfee5450a141c8cdf8dd6ec8caa68d7d47260ef621777d1",
9
+ "fp32_source_checkpoint_sha256": "8f752032a26a5196c1cb447ef63f01e8a29820ff0c57178dd80c6e26d32b12a9",
10
+ "max_absolute_error": 0.00048828125,
11
+ "per_layer": {
12
+ "0": {
13
+ "cast_matches_model_safetensors_exactly": true,
14
+ "max_absolute_error": 0.00037920475006103516,
15
+ "relative_frobenius_error": 0.000206997521647552
16
+ },
17
+ "10": {
18
+ "cast_matches_model_safetensors_exactly": true,
19
+ "max_absolute_error": 0.0002696514129638672,
20
+ "relative_frobenius_error": 0.00020761765345058065
21
+ },
22
+ "12": {
23
+ "cast_matches_model_safetensors_exactly": true,
24
+ "max_absolute_error": 0.00046837329864501953,
25
+ "relative_frobenius_error": 0.00020767621517046658
26
+ },
27
+ "14": {
28
+ "cast_matches_model_safetensors_exactly": true,
29
+ "max_absolute_error": 0.00048673152923583984,
30
+ "relative_frobenius_error": 0.00020743575797500317
31
+ },
32
+ "16": {
33
+ "cast_matches_model_safetensors_exactly": true,
34
+ "max_absolute_error": 0.0004863739013671875,
35
+ "relative_frobenius_error": 0.00020778469258608042
36
+ },
37
+ "18": {
38
+ "cast_matches_model_safetensors_exactly": true,
39
+ "max_absolute_error": 0.0004839897155761719,
40
+ "relative_frobenius_error": 0.00020779679841448286
41
+ },
42
+ "2": {
43
+ "cast_matches_model_safetensors_exactly": true,
44
+ "max_absolute_error": 0.0004121065139770508,
45
+ "relative_frobenius_error": 0.00020722711230071437
46
+ },
47
+ "20": {
48
+ "cast_matches_model_safetensors_exactly": true,
49
+ "max_absolute_error": 0.0004881620407104492,
50
+ "relative_frobenius_error": 0.00020744978906595194
51
+ },
52
+ "22": {
53
+ "cast_matches_model_safetensors_exactly": true,
54
+ "max_absolute_error": 0.0004878044128417969,
55
+ "relative_frobenius_error": 0.00020796356930850946
56
+ },
57
+ "24": {
58
+ "cast_matches_model_safetensors_exactly": true,
59
+ "max_absolute_error": 0.0004870891571044922,
60
+ "relative_frobenius_error": 0.00020680286914392338
61
+ },
62
+ "26": {
63
+ "cast_matches_model_safetensors_exactly": true,
64
+ "max_absolute_error": 0.0004875659942626953,
65
+ "relative_frobenius_error": 0.00020638797987813289
66
+ },
67
+ "28": {
68
+ "cast_matches_model_safetensors_exactly": true,
69
+ "max_absolute_error": 0.0004818439483642578,
70
+ "relative_frobenius_error": 0.00020315668925901246
71
+ },
72
+ "30": {
73
+ "cast_matches_model_safetensors_exactly": true,
74
+ "max_absolute_error": 0.00048804283142089844,
75
+ "relative_frobenius_error": 0.000219304047701117
76
+ },
77
+ "32": {
78
+ "cast_matches_model_safetensors_exactly": true,
79
+ "max_absolute_error": 0.00048828125,
80
+ "relative_frobenius_error": 0.0002275111331836927
81
+ },
82
+ "34": {
83
+ "cast_matches_model_safetensors_exactly": true,
84
+ "max_absolute_error": 0.00048804283142089844,
85
+ "relative_frobenius_error": 0.00025735070181156144
86
+ },
87
+ "4": {
88
+ "cast_matches_model_safetensors_exactly": true,
89
+ "max_absolute_error": 0.000270843505859375,
90
+ "relative_frobenius_error": 0.00020699111471641674
91
+ },
92
+ "6": {
93
+ "cast_matches_model_safetensors_exactly": true,
94
+ "max_absolute_error": 0.00024312734603881836,
95
+ "relative_frobenius_error": 0.00020716959804749024
96
+ },
97
+ "8": {
98
+ "cast_matches_model_safetensors_exactly": true,
99
+ "max_absolute_error": 0.00040781497955322266,
100
+ "relative_frobenius_error": 0.00020697079418794512
101
+ }
102
+ },
103
+ "relative_frobenius_error": 0.00020901276774552775,
104
+ "schema_version": 1
105
+ }
evaluation_provenance.json ADDED
@@ -0,0 +1,90 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "artifact": {
3
+ "filename": "evaluation.safetensors",
4
+ "format": "safetensors",
5
+ "sha256": "0cc184eb65d273ac8bfee5450a141c8cdf8dd6ec8caa68d7d47260ef621777d1",
6
+ "size_bytes": 301991904,
7
+ "tensor_conversion": "lossless_fp32_reserialization"
8
+ },
9
+ "artifact_kind": "jacobian_lens_evaluation_fp32_provenance",
10
+ "compatibility": {
11
+ "all_layer_casts_match_exactly": true,
12
+ "file": "evaluation_compatibility.json",
13
+ "fp16_artifact": "model.safetensors",
14
+ "fp16_artifact_sha256": "089d776979408f23e5377539c15aa8025d171633718ccdce709bcd3372e7942c"
15
+ },
16
+ "derivation": {
17
+ "formula": "jacobian_sum[layer] / n_done",
18
+ "n_done": 1000,
19
+ "source_export_script_sha256": "d1d9e0b7afd1d69771907a839f62b30936995fdd63b17ea2ac80f724b3cdce17",
20
+ "source_fit_checkpoint_sha256": "a1236cfe5d04601575b3de150ffe50e3a67e755ede1197ecf74c205a31bdc258",
21
+ "source_matrix_stats_sha256": "a21fe7c1f661fa9b5455d89c0a415beae794012eab38f782b667b27fac942401",
22
+ "source_provenance_sha256": "b6e5764fa1580a142403a425ecd03cafe55d4cc36ca1e6b90da7fc19a35aad36",
23
+ "source_validation_sha256": "4aea71008a10ef2d043129f7767e5d387372fe1fd5022550b59017a44e0b965d"
24
+ },
25
+ "lens": {
26
+ "d_model": 2048,
27
+ "dtype": "float32",
28
+ "estimator": {
29
+ "dim_batch": 8,
30
+ "exclude_final_position": true,
31
+ "fit_dtype": "bfloat16",
32
+ "max_seq_len": 128,
33
+ "name": "causal_all_current_and_future_targets_mean_jacobian",
34
+ "prompt_aggregation": "equal_weight_mean_over_prompts",
35
+ "skip_first": 16,
36
+ "source_position_aggregation": "mean_over_valid_source_positions",
37
+ "target_position_aggregation": "sum_over_valid_targets_at_or_after_source"
38
+ },
39
+ "hook_convention": "post_transformer_block_output_residual",
40
+ "n_prompts": 1000,
41
+ "source_layers": [
42
+ 0,
43
+ 2,
44
+ 4,
45
+ 6,
46
+ 8,
47
+ 10,
48
+ 12,
49
+ 14,
50
+ 16,
51
+ 18,
52
+ 20,
53
+ 22,
54
+ 24,
55
+ 26,
56
+ 28,
57
+ 30,
58
+ 32,
59
+ 34
60
+ ],
61
+ "target_layer": 35
62
+ },
63
+ "limitations": [
64
+ "The original fit did not store the resolved Hugging Face commit; the revision binding was reconstructed from the Hub head and its last-modified timestamp.",
65
+ "The recorded evaluation validates token readout and does not establish causal steering or a global workspace."
66
+ ],
67
+ "model": {
68
+ "architecture": "Qwen2ForCausalLM",
69
+ "d_model": 2048,
70
+ "id": "WeiboAI/VibeThinker-3B",
71
+ "n_layers": 36,
72
+ "revision": "77bd2cced09193c8b9a59a32bd8577bbd1f3e01c",
73
+ "revision_binding": "inferred_hub_head_unchanged_since_before_fit",
74
+ "revision_last_modified": "2026-06-30T11:35:41+00:00",
75
+ "tied_embeddings": true
76
+ },
77
+ "schema_version": 1,
78
+ "software": {
79
+ "anthropic_jacobian_lens_commit": "581d398613e5602a5af361e1c34d3a92ea82ba8e",
80
+ "conversion_runtime": {
81
+ "python_implementation": "CPython",
82
+ "safetensors": "0.8.0",
83
+ "torch": "2.13.0"
84
+ }
85
+ },
86
+ "source_checkpoint": {
87
+ "format": "pytorch",
88
+ "sha256": "8f752032a26a5196c1cb447ef63f01e8a29820ff0c57178dd80c6e26d32b12a9"
89
+ }
90
+ }
evaluation_tensor_manifest.json ADDED
@@ -0,0 +1,209 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "artifact": "evaluation.safetensors",
3
+ "artifact_sha256": "0cc184eb65d273ac8bfee5450a141c8cdf8dd6ec8caa68d7d47260ef621777d1",
4
+ "artifact_size_bytes": 301991904,
5
+ "schema_version": 1,
6
+ "source_checkpoint_sha256": "8f752032a26a5196c1cb447ef63f01e8a29820ff0c57178dd80c6e26d32b12a9",
7
+ "tensor_count": 18,
8
+ "tensor_storage_bytes": 301989888,
9
+ "tensors": {
10
+ "J.0": {
11
+ "dtype": "float32",
12
+ "nbytes": 16777216,
13
+ "numel": 4194304,
14
+ "sha256_c_contiguous_little_endian_bytes": "4a77bed8711fd26317619e7265da5c9ce73aadd2d4851800715c8c5ec6012c6e",
15
+ "shape": [
16
+ 2048,
17
+ 2048
18
+ ],
19
+ "source_layer": 0
20
+ },
21
+ "J.10": {
22
+ "dtype": "float32",
23
+ "nbytes": 16777216,
24
+ "numel": 4194304,
25
+ "sha256_c_contiguous_little_endian_bytes": "e2856885c149fa8244e327369c3bb081cfe651e1a6958bda317f8691e2db3c3b",
26
+ "shape": [
27
+ 2048,
28
+ 2048
29
+ ],
30
+ "source_layer": 10
31
+ },
32
+ "J.12": {
33
+ "dtype": "float32",
34
+ "nbytes": 16777216,
35
+ "numel": 4194304,
36
+ "sha256_c_contiguous_little_endian_bytes": "e6a50d9c6495acebeac769533bd6900ce94443fc977506f836881e5dcb0335a0",
37
+ "shape": [
38
+ 2048,
39
+ 2048
40
+ ],
41
+ "source_layer": 12
42
+ },
43
+ "J.14": {
44
+ "dtype": "float32",
45
+ "nbytes": 16777216,
46
+ "numel": 4194304,
47
+ "sha256_c_contiguous_little_endian_bytes": "b27ade6bee39486cad9b0cdf091f0690528509b0f4936d67e30ce3dc6eeeb6b8",
48
+ "shape": [
49
+ 2048,
50
+ 2048
51
+ ],
52
+ "source_layer": 14
53
+ },
54
+ "J.16": {
55
+ "dtype": "float32",
56
+ "nbytes": 16777216,
57
+ "numel": 4194304,
58
+ "sha256_c_contiguous_little_endian_bytes": "e80280578bd50abdb0c1610d453eb11495be80a8580fe8f490ff63d6ab1e2e1b",
59
+ "shape": [
60
+ 2048,
61
+ 2048
62
+ ],
63
+ "source_layer": 16
64
+ },
65
+ "J.18": {
66
+ "dtype": "float32",
67
+ "nbytes": 16777216,
68
+ "numel": 4194304,
69
+ "sha256_c_contiguous_little_endian_bytes": "23a2a10e7961625232b257e0943ad10893a9e2e051bf65b9f3e57778e293397f",
70
+ "shape": [
71
+ 2048,
72
+ 2048
73
+ ],
74
+ "source_layer": 18
75
+ },
76
+ "J.2": {
77
+ "dtype": "float32",
78
+ "nbytes": 16777216,
79
+ "numel": 4194304,
80
+ "sha256_c_contiguous_little_endian_bytes": "f90ad9dae2c7b8701cc5b6623ddb0a77cb072c5b95a8a45fff1eca8680e796e3",
81
+ "shape": [
82
+ 2048,
83
+ 2048
84
+ ],
85
+ "source_layer": 2
86
+ },
87
+ "J.20": {
88
+ "dtype": "float32",
89
+ "nbytes": 16777216,
90
+ "numel": 4194304,
91
+ "sha256_c_contiguous_little_endian_bytes": "c927effa01101f88b8c06023db5b9f727416ffde7f40be1515d48fa8215389cf",
92
+ "shape": [
93
+ 2048,
94
+ 2048
95
+ ],
96
+ "source_layer": 20
97
+ },
98
+ "J.22": {
99
+ "dtype": "float32",
100
+ "nbytes": 16777216,
101
+ "numel": 4194304,
102
+ "sha256_c_contiguous_little_endian_bytes": "bb319fcc2484b93bc7a7754e508e807cfcd5387474ad7fb97c786da57b22aeea",
103
+ "shape": [
104
+ 2048,
105
+ 2048
106
+ ],
107
+ "source_layer": 22
108
+ },
109
+ "J.24": {
110
+ "dtype": "float32",
111
+ "nbytes": 16777216,
112
+ "numel": 4194304,
113
+ "sha256_c_contiguous_little_endian_bytes": "a34fb656e93ee78c27bb88b42895aa11ecfc525d790ef54b1a18f72b7cf583e7",
114
+ "shape": [
115
+ 2048,
116
+ 2048
117
+ ],
118
+ "source_layer": 24
119
+ },
120
+ "J.26": {
121
+ "dtype": "float32",
122
+ "nbytes": 16777216,
123
+ "numel": 4194304,
124
+ "sha256_c_contiguous_little_endian_bytes": "1865fccec5a832ca6b0766324bd078d5ea12e2715c7d25664dfd56da8cd82b21",
125
+ "shape": [
126
+ 2048,
127
+ 2048
128
+ ],
129
+ "source_layer": 26
130
+ },
131
+ "J.28": {
132
+ "dtype": "float32",
133
+ "nbytes": 16777216,
134
+ "numel": 4194304,
135
+ "sha256_c_contiguous_little_endian_bytes": "9022027a77af3525b4c602723fedbd3871ffc0b4cf812081cdf6739ffb0f778b",
136
+ "shape": [
137
+ 2048,
138
+ 2048
139
+ ],
140
+ "source_layer": 28
141
+ },
142
+ "J.30": {
143
+ "dtype": "float32",
144
+ "nbytes": 16777216,
145
+ "numel": 4194304,
146
+ "sha256_c_contiguous_little_endian_bytes": "ed63bd90333f5feb4441f1887bef102e3c5b95776a6f9c3c88a657ba1a92845f",
147
+ "shape": [
148
+ 2048,
149
+ 2048
150
+ ],
151
+ "source_layer": 30
152
+ },
153
+ "J.32": {
154
+ "dtype": "float32",
155
+ "nbytes": 16777216,
156
+ "numel": 4194304,
157
+ "sha256_c_contiguous_little_endian_bytes": "c16a4dc0bd1af68302094d81cc898fa7fa30dca7f6ca0397f765c2b470716852",
158
+ "shape": [
159
+ 2048,
160
+ 2048
161
+ ],
162
+ "source_layer": 32
163
+ },
164
+ "J.34": {
165
+ "dtype": "float32",
166
+ "nbytes": 16777216,
167
+ "numel": 4194304,
168
+ "sha256_c_contiguous_little_endian_bytes": "92c6ab150ddf0ece4eeb3a9998d9a87a9d8fabbfdad35f15b18f27a80153ecc5",
169
+ "shape": [
170
+ 2048,
171
+ 2048
172
+ ],
173
+ "source_layer": 34
174
+ },
175
+ "J.4": {
176
+ "dtype": "float32",
177
+ "nbytes": 16777216,
178
+ "numel": 4194304,
179
+ "sha256_c_contiguous_little_endian_bytes": "59beda69c9baf4392cc55731fda3e3527c79948651fe9f38a9ce21669fe2aefa",
180
+ "shape": [
181
+ 2048,
182
+ 2048
183
+ ],
184
+ "source_layer": 4
185
+ },
186
+ "J.6": {
187
+ "dtype": "float32",
188
+ "nbytes": 16777216,
189
+ "numel": 4194304,
190
+ "sha256_c_contiguous_little_endian_bytes": "faca3b5a874af143d133a705ca92d8bcbb71fe862f17559464d11d9bf3a5015c",
191
+ "shape": [
192
+ 2048,
193
+ 2048
194
+ ],
195
+ "source_layer": 6
196
+ },
197
+ "J.8": {
198
+ "dtype": "float32",
199
+ "nbytes": 16777216,
200
+ "numel": 4194304,
201
+ "sha256_c_contiguous_little_endian_bytes": "2e9e5f1d2773f627ed231a0c5a6697077761354574350a63ad82035540a1944b",
202
+ "shape": [
203
+ 2048,
204
+ 2048
205
+ ],
206
+ "source_layer": 8
207
+ }
208
+ }
209
+ }
evaluation_validation.json ADDED
@@ -0,0 +1,29 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "artifact": "evaluation.safetensors",
3
+ "artifact_sha256": "0cc184eb65d273ac8bfee5450a141c8cdf8dd6ec8caa68d7d47260ef621777d1",
4
+ "artifact_size_bytes": 301991904,
5
+ "checks": {
6
+ "all_source_tensors_contiguous": true,
7
+ "all_source_tensors_finite": true,
8
+ "all_source_tensors_fp32": true,
9
+ "all_source_tensors_shape_2048x2048": true,
10
+ "fp16_cast_matches_companion_artifact": true,
11
+ "roundtrip_all_tensor_bit_patterns_equal": true,
12
+ "roundtrip_all_tensor_byte_hashes_equal": true,
13
+ "roundtrip_all_tensor_dtypes_equal": true,
14
+ "roundtrip_all_tensor_shapes_equal": true,
15
+ "roundtrip_all_tensor_values_equal": true,
16
+ "roundtrip_key_set_exact": true,
17
+ "safetensors_header_public_safe": true,
18
+ "safetensors_header_roundtrip_exact": true,
19
+ "source_checkpoint_sha256_exact": true,
20
+ "source_metadata_exact": true,
21
+ "source_top_level_key_set_exact": true
22
+ },
23
+ "exact_tensor_matches": 18,
24
+ "expected_tensor_matches": 18,
25
+ "ok": true,
26
+ "schema_version": 1,
27
+ "source_checkpoint_sha256": "8f752032a26a5196c1cb447ef63f01e8a29820ff0c57178dd80c6e26d32b12a9",
28
+ "tensor_values_changed": 0
29
+ }
lens_config.json ADDED
@@ -0,0 +1,62 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "artifact": {
3
+ "filename": "model.safetensors",
4
+ "format": "safetensors",
5
+ "sha256": "089d776979408f23e5377539c15aa8025d171633718ccdce709bcd3372e7942c",
6
+ "size_bytes": 150996824
7
+ },
8
+ "artifact_kind": "jacobian_lens",
9
+ "evaluation_artifact": {
10
+ "filename": "evaluation.safetensors",
11
+ "format": "safetensors",
12
+ "sha256": "0cc184eb65d273ac8bfee5450a141c8cdf8dd6ec8caa68d7d47260ef621777d1",
13
+ "size_bytes": 301991904,
14
+ "source_checkpoint_sha256": "8f752032a26a5196c1cb447ef63f01e8a29820ff0c57178dd80c6e26d32b12a9",
15
+ "tensor_conversion": "lossless_fp32_reserialization",
16
+ "tensor_dtype": "float32"
17
+ },
18
+ "d_model": 2048,
19
+ "fit": {
20
+ "dim_batch": 8,
21
+ "exclude_final_position": true,
22
+ "fit_dtype": "bfloat16",
23
+ "max_seq_len": 128,
24
+ "n_prompts": 1000,
25
+ "skip_first": 16
26
+ },
27
+ "model": {
28
+ "architecture": "Qwen2ForCausalLM",
29
+ "id": "WeiboAI/VibeThinker-3B",
30
+ "n_layers": 36,
31
+ "revision": "77bd2cced09193c8b9a59a32bd8577bbd1f3e01c"
32
+ },
33
+ "n_prompts": 1000,
34
+ "schema_version": 1,
35
+ "source_checkpoint": {
36
+ "format": "pytorch",
37
+ "sha256": "f36a99447623e0d777c70951a9148a7a52e42e0df82942e22ce6326f63d8d664"
38
+ },
39
+ "source_layers": [
40
+ 0,
41
+ 2,
42
+ 4,
43
+ 6,
44
+ 8,
45
+ 10,
46
+ 12,
47
+ 14,
48
+ 16,
49
+ 18,
50
+ 20,
51
+ 22,
52
+ 24,
53
+ 26,
54
+ 28,
55
+ 30,
56
+ 32,
57
+ 34
58
+ ],
59
+ "target_layer": 35,
60
+ "tensor_dtype": "float16",
61
+ "tensor_key_pattern": "J.{source_layer}"
62
+ }
model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:089d776979408f23e5377539c15aa8025d171633718ccdce709bcd3372e7942c
3
+ size 150996824
provenance.json ADDED
@@ -0,0 +1,128 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "artifact": {
3
+ "filename": "model.safetensors",
4
+ "format": "safetensors",
5
+ "sha256": "089d776979408f23e5377539c15aa8025d171633718ccdce709bcd3372e7942c",
6
+ "size_bytes": 150996824,
7
+ "tensor_conversion": "lossless_fp16_reserialization"
8
+ },
9
+ "artifact_kind": "jacobian_lens_provenance",
10
+ "conversion": {
11
+ "safetensors": "0.7.0",
12
+ "source_and_output_tensor_bit_patterns_equal": true,
13
+ "source_and_output_tensor_dtypes_equal": true,
14
+ "source_and_output_tensor_keys_equal": true,
15
+ "source_and_output_tensor_shapes_equal": true,
16
+ "torch": "2.12.0"
17
+ },
18
+ "fit": {
19
+ "d_model": 2048,
20
+ "dim_batch": 8,
21
+ "fit_date": null,
22
+ "fit_date_available": false,
23
+ "estimator": {
24
+ "exclude_final_position": true,
25
+ "name": "causal_all_current_and_future_targets_mean_jacobian",
26
+ "prompt_aggregation": "equal_weight_mean_over_prompts",
27
+ "source_position_aggregation": "mean_over_valid_source_positions",
28
+ "target_position_aggregation": "sum_over_valid_targets_at_or_after_source"
29
+ },
30
+ "fit_dtype": "bfloat16",
31
+ "hook_convention": "post_transformer_block_output_residual",
32
+ "max_seq_len": 128,
33
+ "n_prompts": 1000,
34
+ "skip_first": 16,
35
+ "source_layers": [
36
+ 0,
37
+ 2,
38
+ 4,
39
+ 6,
40
+ 8,
41
+ 10,
42
+ 12,
43
+ 14,
44
+ 16,
45
+ 18,
46
+ 20,
47
+ 22,
48
+ 24,
49
+ 26,
50
+ 28,
51
+ 30,
52
+ 32,
53
+ 34
54
+ ],
55
+ "stored_dtype": "float16",
56
+ "target_layer": 35
57
+ },
58
+ "limitations": [
59
+ "The original fit did not persist the resolved Hugging Face commit. The revision binding was reconstructed from the Hub head and its last-modified timestamp after the fit.",
60
+ "Artifact integrity and provenance do not establish causal intervention validity.",
61
+ "The fitting prompt text is not redistributed in this repository.",
62
+ "The original remote prompt JSONL is absent from the fetched fit artifacts. A later local materialization matches the remote manifest's aggregate row-hash chain and summary fields, but exact remote JSONL byte identity was not verified."
63
+ ],
64
+ "model": {
65
+ "architecture": "Qwen2ForCausalLM",
66
+ "d_model": 2048,
67
+ "id": "WeiboAI/VibeThinker-3B",
68
+ "n_layers": 36,
69
+ "revision": "77bd2cced09193c8b9a59a32bd8577bbd1f3e01c",
70
+ "revision_binding": "inferred_hub_head_unchanged_since_before_fit",
71
+ "revision_last_modified": "2026-06-30T11:35:41+00:00",
72
+ "tied_embeddings": true
73
+ },
74
+ "prompt_corpus": {
75
+ "dataset": "Salesforce/wikitext",
76
+ "dataset_config": "wikitext-103-raw-v1",
77
+ "dataset_revision": null,
78
+ "fit_manifest": {
79
+ "first_id": "wikitext-0000",
80
+ "last_id": "wikitext-0999",
81
+ "manifest_sha256": "96593961cce7f84f72698cddfe67f2c0d2ea334083d93ca56aaa150ad42923c2",
82
+ "min_chars": 600,
83
+ "n_prompts": 1000,
84
+ "prompt_sha256_chain": "d49717ca76e73c570e18d0229c735e7049d636b7d5fdf3cc371e4877351cf368",
85
+ "prompt_sha256_chain_algorithm": "sha256(concatenated_per_row_sha256_hex)",
86
+ "total_chars": 982826
87
+ },
88
+ "later_local_materialization": {
89
+ "concatenated_per_row_sha256_hex": "d49717ca76e73c570e18d0229c735e7049d636b7d5fdf3cc371e4877351cf368",
90
+ "file_sha256": "1bf8cac8f1fae950201eb4161eb76a75bc5f7b4b9665c1a9b16e17aee3d37334",
91
+ "first_id": "wikitext-0000",
92
+ "last_id": "wikitext-0999",
93
+ "local_manifest_sha256": "a3aac6dd98f456bb2a95843f10ca71241d8f40e794bf1f1c83c71bcafd7c06ed",
94
+ "n_prompts": 1000,
95
+ "newline_joined_per_row_sha256_hex": "993212d9409cddaf3820330095b656b50231cb15578007ec2ea97229ca9792ad",
96
+ "row_hash_validation": "all_local_rows_match_embedded_sha256_utf8_text",
97
+ "total_chars": 982826
98
+ },
99
+ "reconciliation": {
100
+ "aggregate_row_hash_chain_matches": true,
101
+ "exact_remote_jsonl_byte_identity_verified": false,
102
+ "original_remote_jsonl_available": false,
103
+ "row_by_row_remote_comparison_performed": false,
104
+ "status": "later_local_materialization_matches_remote_aggregate_manifest",
105
+ "summary_fields_match": [
106
+ "n_prompts",
107
+ "total_chars",
108
+ "first_id",
109
+ "last_id"
110
+ ]
111
+ },
112
+ "selection_rule_reported": "first 1000 rows with at least 600 characters",
113
+ "split": "train"
114
+ },
115
+ "schema_version": 1,
116
+ "software": {
117
+ "anthropic_jacobian_lens_commit": "581d398613e5602a5af361e1c34d3a92ea82ba8e",
118
+ "fit_runtime": {
119
+ "python": "3.10.12",
120
+ "torch": "2.1.0+cu118",
121
+ "transformers": "4.44.2"
122
+ }
123
+ },
124
+ "source_checkpoint": {
125
+ "format": "pytorch",
126
+ "sha256": "f36a99447623e0d777c70951a9148a7a52e42e0df82942e22ce6326f63d8d664"
127
+ }
128
+ }
requirements.txt ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ safetensors==0.8.0
2
+ torch==2.13.0
scripts/convert_checkpoint.py ADDED
@@ -0,0 +1,275 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ """Convert the frozen VibeThinker J Lens checkpoint to Safetensors.
3
+
4
+ The input path is intentionally required at runtime and is never copied into
5
+ the safetensors header or any generated public metadata. The conversion is
6
+ lossless: every FP16 bit pattern is compared after a safetensors round trip.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import argparse
12
+ import hashlib
13
+ import json
14
+ import os
15
+ from collections.abc import Mapping
16
+ from pathlib import Path
17
+ from typing import Any
18
+
19
+ import torch
20
+ from safetensors import safe_open
21
+
22
+ SOURCE_SHA256 = "f36a99447623e0d777c70951a9148a7a52e42e0df82942e22ce6326f63d8d664"
23
+ MODEL_ID = "WeiboAI/VibeThinker-3B"
24
+ MODEL_REVISION = "77bd2cced09193c8b9a59a32bd8577bbd1f3e01c"
25
+ SOURCE_LAYERS = tuple(range(0, 36, 2))
26
+ TARGET_LAYER = 35
27
+ D_MODEL = 2048
28
+ N_PROMPTS = 1000
29
+ EXPECTED_TOP_LEVEL_KEYS = {"J", "n_prompts", "source_layers", "d_model"}
30
+ FORBIDDEN_PUBLIC_FRAGMENTS = (
31
+ os.sep.join(("", "Users", "")),
32
+ os.sep.join(("", "Volumes", "")),
33
+ os.sep.join(("", "workspace")),
34
+ )
35
+
36
+
37
+ def sha256_file(path: Path) -> str:
38
+ digest = hashlib.sha256()
39
+ with path.open("rb") as handle:
40
+ for chunk in iter(lambda: handle.read(8 * 1024 * 1024), b""):
41
+ digest.update(chunk)
42
+ return digest.hexdigest()
43
+
44
+
45
+ def tensor_storage_bytes(tensor: torch.Tensor) -> bytes:
46
+ """Return C-contiguous little-endian bytes, including FP16 bit patterns."""
47
+
48
+ array = tensor.detach().cpu().contiguous().view(torch.int16).numpy()
49
+ return array.astype("<i2", copy=False).tobytes(order="C")
50
+
51
+
52
+ def tensor_sha256(tensor: torch.Tensor) -> str:
53
+ return hashlib.sha256(tensor_storage_bytes(tensor)).hexdigest()
54
+
55
+
56
+ def write_json(path: Path, value: Any) -> None:
57
+ path.write_text(
58
+ json.dumps(value, indent=2, sort_keys=True, ensure_ascii=True) + "\n",
59
+ encoding="utf-8",
60
+ )
61
+
62
+
63
+ def require(condition: bool, message: str) -> None:
64
+ if not condition:
65
+ raise ValueError(message)
66
+
67
+
68
+ def validate_source(checkpoint: Any) -> dict[int, torch.Tensor]:
69
+ require(isinstance(checkpoint, Mapping), "checkpoint must be a mapping")
70
+ require(
71
+ set(checkpoint) == EXPECTED_TOP_LEVEL_KEYS,
72
+ f"unexpected checkpoint keys: {sorted(checkpoint)}",
73
+ )
74
+ require(checkpoint["n_prompts"] == N_PROMPTS, "unexpected n_prompts")
75
+ require(checkpoint["d_model"] == D_MODEL, "unexpected d_model")
76
+ require(
77
+ tuple(checkpoint["source_layers"]) == SOURCE_LAYERS,
78
+ "unexpected source_layers",
79
+ )
80
+
81
+ matrices = checkpoint["J"]
82
+ require(isinstance(matrices, Mapping), "J must be a layer-to-tensor mapping")
83
+ require(set(matrices) == set(SOURCE_LAYERS), "unexpected J layer keys")
84
+
85
+ validated: dict[int, torch.Tensor] = {}
86
+ for layer in SOURCE_LAYERS:
87
+ tensor = matrices[layer]
88
+ require(isinstance(tensor, torch.Tensor), f"J[{layer}] is not a tensor")
89
+ require(tensor.device.type == "cpu", f"J[{layer}] is not on CPU")
90
+ require(tensor.dtype == torch.float16, f"J[{layer}] is not FP16")
91
+ require(tuple(tensor.shape) == (D_MODEL, D_MODEL), f"J[{layer}] shape mismatch")
92
+ require(tensor.is_contiguous(), f"J[{layer}] is not contiguous")
93
+ require(bool(torch.isfinite(tensor).all()), f"J[{layer}] contains non-finite values")
94
+ validated[layer] = tensor
95
+ return validated
96
+
97
+
98
+ def public_header() -> dict[str, str]:
99
+ return {
100
+ "artifact_kind": "jacobian_lens",
101
+ "d_model": str(D_MODEL),
102
+ "format": "pt",
103
+ "model_id": MODEL_ID,
104
+ "model_revision": MODEL_REVISION,
105
+ "n_prompts": str(N_PROMPTS),
106
+ "schema_version": "1",
107
+ "source_checkpoint_sha256": SOURCE_SHA256,
108
+ "source_layers": json.dumps(SOURCE_LAYERS, separators=(",", ":")),
109
+ "target_layer": str(TARGET_LAYER),
110
+ "tensor_dtype": "float16",
111
+ "tensor_key_pattern": "J.{source_layer}",
112
+ }
113
+
114
+
115
+ def assert_public_header(metadata: Mapping[str, str]) -> None:
116
+ encoded = json.dumps(dict(metadata), sort_keys=True)
117
+ for fragment in FORBIDDEN_PUBLIC_FRAGMENTS:
118
+ require(fragment not in encoded, f"private fragment found in safetensors header: {fragment}")
119
+
120
+
121
+ def save_deterministic_safetensors(
122
+ tensors: Mapping[str, torch.Tensor],
123
+ path: Path,
124
+ metadata: Mapping[str, str],
125
+ ) -> None:
126
+ """Write the documented safetensors format with canonical key ordering.
127
+
128
+ The upstream writer preserves tensor data exactly, but its Rust metadata
129
+ map can serialize keys in a process-random order. Canonical JSON ordering
130
+ makes the complete artifact reproducible byte for byte across runs.
131
+ """
132
+
133
+ offset = 0
134
+ header: dict[str, Any] = {
135
+ "__metadata__": {key: metadata[key] for key in sorted(metadata)}
136
+ }
137
+ for key in sorted(tensors):
138
+ tensor = tensors[key]
139
+ require(tensor.dtype == torch.float16, f"{key} is not FP16")
140
+ nbytes = tensor.numel() * tensor.element_size()
141
+ header[key] = {
142
+ "dtype": "F16",
143
+ "shape": list(tensor.shape),
144
+ "data_offsets": [offset, offset + nbytes],
145
+ }
146
+ offset += nbytes
147
+
148
+ encoded_header = json.dumps(
149
+ header,
150
+ ensure_ascii=False,
151
+ separators=(",", ":"),
152
+ ).encode("utf-8")
153
+ padding = (-len(encoded_header)) % 8
154
+ encoded_header += b" " * padding
155
+
156
+ with path.open("wb") as handle:
157
+ handle.write(len(encoded_header).to_bytes(8, byteorder="little", signed=False))
158
+ handle.write(encoded_header)
159
+ for key in sorted(tensors):
160
+ handle.write(tensor_storage_bytes(tensors[key]))
161
+
162
+
163
+ def convert(source: Path, output_dir: Path, overwrite: bool) -> dict[str, Any]:
164
+ source_digest = sha256_file(source)
165
+ require(source_digest == SOURCE_SHA256, "source checkpoint SHA-256 mismatch")
166
+
167
+ checkpoint = torch.load(source, map_location="cpu", weights_only=True)
168
+ matrices = validate_source(checkpoint)
169
+ tensors = {f"J.{layer}": matrices[layer] for layer in SOURCE_LAYERS}
170
+
171
+ output_dir.mkdir(parents=True, exist_ok=True)
172
+ output_path = output_dir / "model.safetensors"
173
+ if output_path.exists() and not overwrite:
174
+ raise FileExistsError(f"refusing to overwrite {output_path.name}; pass --overwrite")
175
+
176
+ metadata = public_header()
177
+ assert_public_header(metadata)
178
+ temporary_path = output_dir / ".model.safetensors.tmp"
179
+ save_deterministic_safetensors(tensors, temporary_path, metadata)
180
+ os.replace(temporary_path, output_path)
181
+
182
+ manifest_tensors: dict[str, Any] = {}
183
+ exact_matches = 0
184
+ with safe_open(output_path, framework="pt", device="cpu") as artifact:
185
+ stored_metadata = artifact.metadata() or {}
186
+ require(stored_metadata == metadata, "safetensors metadata changed during serialization")
187
+ assert_public_header(stored_metadata)
188
+ require(set(artifact.keys()) == set(tensors), "safetensors key set mismatch")
189
+
190
+ for layer in SOURCE_LAYERS:
191
+ key = f"J.{layer}"
192
+ source_tensor = matrices[layer]
193
+ output_tensor = artifact.get_tensor(key)
194
+ require(output_tensor.dtype == source_tensor.dtype, f"{key} dtype mismatch")
195
+ require(tuple(output_tensor.shape) == tuple(source_tensor.shape), f"{key} shape mismatch")
196
+ require(output_tensor.is_contiguous(), f"{key} is not contiguous")
197
+ require(torch.equal(output_tensor, source_tensor), f"{key} value mismatch")
198
+ require(
199
+ torch.equal(output_tensor.view(torch.int16), source_tensor.view(torch.int16)),
200
+ f"{key} FP16 bit-pattern mismatch",
201
+ )
202
+ source_tensor_digest = tensor_sha256(source_tensor)
203
+ output_tensor_digest = tensor_sha256(output_tensor)
204
+ require(source_tensor_digest == output_tensor_digest, f"{key} byte hash mismatch")
205
+ exact_matches += 1
206
+ manifest_tensors[key] = {
207
+ "dtype": "float16",
208
+ "nbytes": source_tensor.numel() * source_tensor.element_size(),
209
+ "numel": source_tensor.numel(),
210
+ "sha256_c_contiguous_little_endian_bytes": source_tensor_digest,
211
+ "shape": list(source_tensor.shape),
212
+ "source_layer": layer,
213
+ }
214
+
215
+ output_digest = sha256_file(output_path)
216
+ output_size = output_path.stat().st_size
217
+ tensor_manifest = {
218
+ "artifact": "model.safetensors",
219
+ "artifact_sha256": output_digest,
220
+ "artifact_size_bytes": output_size,
221
+ "schema_version": 1,
222
+ "tensor_count": len(manifest_tensors),
223
+ "tensor_storage_bytes": sum(item["nbytes"] for item in manifest_tensors.values()),
224
+ "tensors": manifest_tensors,
225
+ }
226
+ validation = {
227
+ "artifact": "model.safetensors",
228
+ "artifact_sha256": output_digest,
229
+ "artifact_size_bytes": output_size,
230
+ "checks": {
231
+ "all_source_tensors_contiguous": True,
232
+ "all_source_tensors_finite": True,
233
+ "all_source_tensors_fp16": True,
234
+ "all_source_tensors_shape_2048x2048": True,
235
+ "roundtrip_all_tensor_byte_hashes_equal": True,
236
+ "roundtrip_all_tensor_dtypes_equal": True,
237
+ "roundtrip_all_tensor_shapes_equal": True,
238
+ "roundtrip_all_tensor_values_equal": True,
239
+ "roundtrip_all_tensor_bit_patterns_equal": True,
240
+ "roundtrip_key_set_exact": True,
241
+ "safetensors_header_public_safe": True,
242
+ "safetensors_header_roundtrip_exact": True,
243
+ "source_checkpoint_sha256_exact": True,
244
+ "source_metadata_exact": True,
245
+ "source_top_level_key_set_exact": True,
246
+ },
247
+ "exact_tensor_matches": exact_matches,
248
+ "expected_tensor_matches": len(SOURCE_LAYERS),
249
+ "ok": exact_matches == len(SOURCE_LAYERS),
250
+ "schema_version": 1,
251
+ "source_checkpoint_sha256": source_digest,
252
+ "tensor_values_changed": 0,
253
+ }
254
+ write_json(output_dir / "tensor_manifest.json", tensor_manifest)
255
+ write_json(output_dir / "validation.json", validation)
256
+ (output_dir / "SHA256SUMS").write_text(
257
+ f"{output_digest} model.safetensors\n",
258
+ encoding="ascii",
259
+ )
260
+ return validation
261
+
262
+
263
+ def main() -> None:
264
+ parser = argparse.ArgumentParser(description=__doc__)
265
+ parser.add_argument("--source", required=True, type=Path, help="Private source .pt checkpoint")
266
+ parser.add_argument("--output-dir", required=True, type=Path, help="Public artifact directory")
267
+ parser.add_argument("--overwrite", action="store_true")
268
+ args = parser.parse_args()
269
+
270
+ result = convert(args.source, args.output_dir, args.overwrite)
271
+ print(json.dumps(result, indent=2, sort_keys=True))
272
+
273
+
274
+ if __name__ == "__main__":
275
+ main()
scripts/convert_evaluation_checkpoint.py ADDED
@@ -0,0 +1,439 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ """Convert the frozen FP32 evaluation lens to deterministic Safetensors."""
3
+
4
+ from __future__ import annotations
5
+
6
+ import argparse
7
+ import hashlib
8
+ import json
9
+ import os
10
+ from collections.abc import Mapping
11
+ from pathlib import Path
12
+ from typing import Any
13
+
14
+ import safetensors
15
+ import torch
16
+ from safetensors import safe_open
17
+
18
+ SOURCE_SHA256 = "8f752032a26a5196c1cb447ef63f01e8a29820ff0c57178dd80c6e26d32b12a9"
19
+ SOURCE_PROVENANCE_SHA256 = (
20
+ "b6e5764fa1580a142403a425ecd03cafe55d4cc36ca1e6b90da7fc19a35aad36"
21
+ )
22
+ SOURCE_VALIDATION_SHA256 = (
23
+ "4aea71008a10ef2d043129f7767e5d387372fe1fd5022550b59017a44e0b965d"
24
+ )
25
+ SOURCE_MATRIX_STATS_SHA256 = (
26
+ "a21fe7c1f661fa9b5455d89c0a415beae794012eab38f782b667b27fac942401"
27
+ )
28
+ SOURCE_FIT_CHECKPOINT_SHA256 = (
29
+ "a1236cfe5d04601575b3de150ffe50e3a67e755ede1197ecf74c205a31bdc258"
30
+ )
31
+ SOURCE_EXPORT_SCRIPT_SHA256 = (
32
+ "d1d9e0b7afd1d69771907a839f62b30936995fdd63b17ea2ac80f724b3cdce17"
33
+ )
34
+ FP16_ARTIFACT_SHA256 = (
35
+ "089d776979408f23e5377539c15aa8025d171633718ccdce709bcd3372e7942c"
36
+ )
37
+ MODEL_ID = "WeiboAI/VibeThinker-3B"
38
+ MODEL_REVISION = "77bd2cced09193c8b9a59a32bd8577bbd1f3e01c"
39
+ SOURCE_LAYERS = tuple(range(0, 36, 2))
40
+ TARGET_LAYER = 35
41
+ D_MODEL = 2048
42
+ N_PROMPTS = 1000
43
+ EXPECTED_TOP_LEVEL_KEYS = {"J", "n_prompts", "source_layers", "d_model"}
44
+ FORBIDDEN_PUBLIC_FRAGMENTS = (
45
+ os.sep.join(("", "Users", "")),
46
+ os.sep.join(("", "Volumes", "")),
47
+ os.sep.join(("", "workspace")),
48
+ )
49
+
50
+
51
+ def sha256_file(path: Path) -> str:
52
+ digest = hashlib.sha256()
53
+ with path.open("rb") as handle:
54
+ for chunk in iter(lambda: handle.read(8 * 1024 * 1024), b""):
55
+ digest.update(chunk)
56
+ return digest.hexdigest()
57
+
58
+
59
+ def tensor_storage_bytes(tensor: torch.Tensor) -> bytes:
60
+ """Return C-contiguous little-endian FP32 bytes without value conversion."""
61
+
62
+ array = tensor.detach().cpu().contiguous().view(torch.int32).numpy()
63
+ return array.astype("<i4", copy=False).tobytes(order="C")
64
+
65
+
66
+ def tensor_sha256(tensor: torch.Tensor) -> str:
67
+ return hashlib.sha256(tensor_storage_bytes(tensor)).hexdigest()
68
+
69
+
70
+ def write_json(path: Path, value: Any) -> None:
71
+ path.write_text(
72
+ json.dumps(value, indent=2, sort_keys=True, ensure_ascii=True) + "\n",
73
+ encoding="utf-8",
74
+ )
75
+
76
+
77
+ def require(condition: bool, message: str) -> None:
78
+ if not condition:
79
+ raise ValueError(message)
80
+
81
+
82
+ def validate_source(checkpoint: Any) -> dict[int, torch.Tensor]:
83
+ require(isinstance(checkpoint, Mapping), "checkpoint must be a mapping")
84
+ require(
85
+ set(checkpoint) == EXPECTED_TOP_LEVEL_KEYS,
86
+ f"unexpected checkpoint keys: {sorted(checkpoint)}",
87
+ )
88
+ require(checkpoint["n_prompts"] == N_PROMPTS, "unexpected n_prompts")
89
+ require(checkpoint["d_model"] == D_MODEL, "unexpected d_model")
90
+ require(
91
+ tuple(checkpoint["source_layers"]) == SOURCE_LAYERS,
92
+ "unexpected source_layers",
93
+ )
94
+ matrices = checkpoint["J"]
95
+ require(isinstance(matrices, Mapping), "J must be a layer-to-tensor mapping")
96
+ require(set(matrices) == set(SOURCE_LAYERS), "unexpected J layer keys")
97
+
98
+ validated: dict[int, torch.Tensor] = {}
99
+ for layer in SOURCE_LAYERS:
100
+ tensor = matrices[layer]
101
+ require(isinstance(tensor, torch.Tensor), f"J[{layer}] is not a tensor")
102
+ require(tensor.device.type == "cpu", f"J[{layer}] is not on CPU")
103
+ require(tensor.dtype == torch.float32, f"J[{layer}] is not FP32")
104
+ require(
105
+ tuple(tensor.shape) == (D_MODEL, D_MODEL),
106
+ f"J[{layer}] shape mismatch",
107
+ )
108
+ require(tensor.is_contiguous(), f"J[{layer}] is not contiguous")
109
+ require(
110
+ bool(torch.isfinite(tensor).all()),
111
+ f"J[{layer}] contains non-finite values",
112
+ )
113
+ validated[layer] = tensor
114
+ return validated
115
+
116
+
117
+ def public_header() -> dict[str, str]:
118
+ return {
119
+ "artifact_kind": "jacobian_lens_evaluation_fp32",
120
+ "d_model": str(D_MODEL),
121
+ "format": "pt",
122
+ "model_id": MODEL_ID,
123
+ "model_revision": MODEL_REVISION,
124
+ "n_prompts": str(N_PROMPTS),
125
+ "schema_version": "1",
126
+ "source_fit_checkpoint_sha256": SOURCE_FIT_CHECKPOINT_SHA256,
127
+ "source_fp32_checkpoint_sha256": SOURCE_SHA256,
128
+ "source_layers": json.dumps(SOURCE_LAYERS, separators=(",", ":")),
129
+ "target_layer": str(TARGET_LAYER),
130
+ "tensor_dtype": "float32",
131
+ "tensor_key_pattern": "J.{source_layer}",
132
+ }
133
+
134
+
135
+ def assert_public_header(metadata: Mapping[str, str]) -> None:
136
+ encoded = json.dumps(dict(metadata), sort_keys=True)
137
+ for fragment in FORBIDDEN_PUBLIC_FRAGMENTS:
138
+ require(
139
+ fragment not in encoded,
140
+ f"private fragment found in Safetensors header: {fragment}",
141
+ )
142
+
143
+
144
+ def save_deterministic_safetensors(
145
+ tensors: Mapping[str, torch.Tensor],
146
+ path: Path,
147
+ metadata: Mapping[str, str],
148
+ ) -> None:
149
+ offset = 0
150
+ header: dict[str, Any] = {
151
+ "__metadata__": {key: metadata[key] for key in sorted(metadata)}
152
+ }
153
+ for key in sorted(tensors):
154
+ tensor = tensors[key]
155
+ require(tensor.dtype == torch.float32, f"{key} is not FP32")
156
+ nbytes = tensor.numel() * tensor.element_size()
157
+ header[key] = {
158
+ "dtype": "F32",
159
+ "shape": list(tensor.shape),
160
+ "data_offsets": [offset, offset + nbytes],
161
+ }
162
+ offset += nbytes
163
+
164
+ encoded_header = json.dumps(
165
+ header,
166
+ ensure_ascii=False,
167
+ separators=(",", ":"),
168
+ ).encode("utf-8")
169
+ encoded_header += b" " * ((-len(encoded_header)) % 8)
170
+ with path.open("wb") as handle:
171
+ handle.write(len(encoded_header).to_bytes(8, "little", signed=False))
172
+ handle.write(encoded_header)
173
+ for key in sorted(tensors):
174
+ handle.write(tensor_storage_bytes(tensors[key]))
175
+
176
+
177
+ def compare_fp16(
178
+ matrices: Mapping[int, torch.Tensor],
179
+ fp16_path: Path,
180
+ ) -> dict[str, Any]:
181
+ require(
182
+ sha256_file(fp16_path) == FP16_ARTIFACT_SHA256,
183
+ "FP16 companion artifact SHA-256 mismatch",
184
+ )
185
+ per_layer: dict[str, Any] = {}
186
+ error_squared = 0.0
187
+ reference_squared = 0.0
188
+ with safe_open(fp16_path, framework="pt", device="cpu") as fp16_artifact:
189
+ require(
190
+ set(fp16_artifact.keys()) == {f"J.{layer}" for layer in SOURCE_LAYERS},
191
+ "FP16 companion key set mismatch",
192
+ )
193
+ for layer in SOURCE_LAYERS:
194
+ fp32 = matrices[layer]
195
+ stored_fp16 = fp16_artifact.get_tensor(f"J.{layer}")
196
+ cast_fp16 = fp32.to(torch.float16)
197
+ require(
198
+ torch.equal(cast_fp16.view(torch.int16), stored_fp16.view(torch.int16)),
199
+ f"FP32-to-FP16 cast differs at J.{layer}",
200
+ )
201
+ error = cast_fp16.float() - fp32
202
+ error_norm = float(torch.linalg.vector_norm(error))
203
+ reference_norm = float(torch.linalg.vector_norm(fp32))
204
+ layer_error_squared = error_norm**2
205
+ layer_reference_squared = reference_norm**2
206
+ error_squared += layer_error_squared
207
+ reference_squared += layer_reference_squared
208
+ per_layer[str(layer)] = {
209
+ "cast_matches_model_safetensors_exactly": True,
210
+ "max_absolute_error": float(error.abs().max().item()),
211
+ "relative_frobenius_error": error_norm / reference_norm,
212
+ }
213
+ return {
214
+ "schema_version": 1,
215
+ "artifact_kind": "fp32_to_fp16_lens_compatibility",
216
+ "fp32_source_checkpoint_sha256": SOURCE_SHA256,
217
+ "fp16_artifact": "model.safetensors",
218
+ "fp16_artifact_sha256": FP16_ARTIFACT_SHA256,
219
+ "conversion": "IEEE_FP32_to_FP16_round_to_nearest_even",
220
+ "all_layer_casts_match_exactly": True,
221
+ "relative_frobenius_error": (error_squared / reference_squared) ** 0.5,
222
+ "max_absolute_error": max(
223
+ record["max_absolute_error"] for record in per_layer.values()
224
+ ),
225
+ "per_layer": per_layer,
226
+ }
227
+
228
+
229
+ def convert(source: Path, output_dir: Path, overwrite: bool) -> dict[str, Any]:
230
+ source_digest = sha256_file(source)
231
+ require(source_digest == SOURCE_SHA256, "source checkpoint SHA-256 mismatch")
232
+ checkpoint = torch.load(source, map_location="cpu", weights_only=True)
233
+ matrices = validate_source(checkpoint)
234
+ tensors = {f"J.{layer}": matrices[layer] for layer in SOURCE_LAYERS}
235
+
236
+ output_dir.mkdir(parents=True, exist_ok=True)
237
+ output_path = output_dir / "evaluation.safetensors"
238
+ if output_path.exists() and not overwrite:
239
+ raise FileExistsError(
240
+ f"refusing to overwrite {output_path.name}; pass --overwrite"
241
+ )
242
+ metadata = public_header()
243
+ assert_public_header(metadata)
244
+ temporary_path = output_dir / ".evaluation.safetensors.tmp"
245
+ save_deterministic_safetensors(tensors, temporary_path, metadata)
246
+ os.replace(temporary_path, output_path)
247
+
248
+ manifest_tensors: dict[str, Any] = {}
249
+ exact_matches = 0
250
+ with safe_open(output_path, framework="pt", device="cpu") as artifact:
251
+ stored_metadata = artifact.metadata() or {}
252
+ require(stored_metadata == metadata, "Safetensors metadata changed")
253
+ assert_public_header(stored_metadata)
254
+ require(set(artifact.keys()) == set(tensors), "Safetensors key set mismatch")
255
+ for layer in SOURCE_LAYERS:
256
+ key = f"J.{layer}"
257
+ source_tensor = matrices[layer]
258
+ output_tensor = artifact.get_tensor(key)
259
+ require(output_tensor.dtype == torch.float32, f"{key} dtype mismatch")
260
+ require(
261
+ tuple(output_tensor.shape) == (D_MODEL, D_MODEL),
262
+ f"{key} shape mismatch",
263
+ )
264
+ require(output_tensor.is_contiguous(), f"{key} is not contiguous")
265
+ require(torch.equal(output_tensor, source_tensor), f"{key} value mismatch")
266
+ require(
267
+ torch.equal(
268
+ output_tensor.view(torch.int32),
269
+ source_tensor.view(torch.int32),
270
+ ),
271
+ f"{key} FP32 bit-pattern mismatch",
272
+ )
273
+ source_tensor_digest = tensor_sha256(source_tensor)
274
+ require(
275
+ tensor_sha256(output_tensor) == source_tensor_digest,
276
+ f"{key} raw byte hash mismatch",
277
+ )
278
+ exact_matches += 1
279
+ manifest_tensors[key] = {
280
+ "dtype": "float32",
281
+ "nbytes": source_tensor.numel() * source_tensor.element_size(),
282
+ "numel": source_tensor.numel(),
283
+ "sha256_c_contiguous_little_endian_bytes": source_tensor_digest,
284
+ "shape": list(source_tensor.shape),
285
+ "source_layer": layer,
286
+ }
287
+
288
+ output_digest = sha256_file(output_path)
289
+ output_size = output_path.stat().st_size
290
+ manifest = {
291
+ "schema_version": 1,
292
+ "artifact": "evaluation.safetensors",
293
+ "artifact_sha256": output_digest,
294
+ "artifact_size_bytes": output_size,
295
+ "source_checkpoint_sha256": source_digest,
296
+ "tensor_count": len(manifest_tensors),
297
+ "tensor_storage_bytes": sum(
298
+ record["nbytes"] for record in manifest_tensors.values()
299
+ ),
300
+ "tensors": manifest_tensors,
301
+ }
302
+ compatibility = compare_fp16(matrices, output_dir / "model.safetensors")
303
+ compatibility["fp32_artifact"] = "evaluation.safetensors"
304
+ compatibility["fp32_artifact_sha256"] = output_digest
305
+ validation = {
306
+ "schema_version": 1,
307
+ "artifact": "evaluation.safetensors",
308
+ "artifact_sha256": output_digest,
309
+ "artifact_size_bytes": output_size,
310
+ "source_checkpoint_sha256": source_digest,
311
+ "exact_tensor_matches": exact_matches,
312
+ "expected_tensor_matches": len(SOURCE_LAYERS),
313
+ "tensor_values_changed": 0,
314
+ "checks": {
315
+ "all_source_tensors_contiguous": True,
316
+ "all_source_tensors_finite": True,
317
+ "all_source_tensors_fp32": True,
318
+ "all_source_tensors_shape_2048x2048": True,
319
+ "fp16_cast_matches_companion_artifact": True,
320
+ "roundtrip_all_tensor_byte_hashes_equal": True,
321
+ "roundtrip_all_tensor_dtypes_equal": True,
322
+ "roundtrip_all_tensor_shapes_equal": True,
323
+ "roundtrip_all_tensor_values_equal": True,
324
+ "roundtrip_all_tensor_bit_patterns_equal": True,
325
+ "roundtrip_key_set_exact": True,
326
+ "safetensors_header_public_safe": True,
327
+ "safetensors_header_roundtrip_exact": True,
328
+ "source_checkpoint_sha256_exact": True,
329
+ "source_metadata_exact": True,
330
+ "source_top_level_key_set_exact": True,
331
+ },
332
+ "ok": exact_matches == len(SOURCE_LAYERS),
333
+ }
334
+ provenance = {
335
+ "schema_version": 1,
336
+ "artifact_kind": "jacobian_lens_evaluation_fp32_provenance",
337
+ "artifact": {
338
+ "filename": "evaluation.safetensors",
339
+ "format": "safetensors",
340
+ "sha256": output_digest,
341
+ "size_bytes": output_size,
342
+ "tensor_conversion": "lossless_fp32_reserialization",
343
+ },
344
+ "source_checkpoint": {
345
+ "format": "pytorch",
346
+ "sha256": source_digest,
347
+ },
348
+ "derivation": {
349
+ "formula": "jacobian_sum[layer] / n_done",
350
+ "n_done": N_PROMPTS,
351
+ "source_fit_checkpoint_sha256": SOURCE_FIT_CHECKPOINT_SHA256,
352
+ "source_export_script_sha256": SOURCE_EXPORT_SCRIPT_SHA256,
353
+ "source_matrix_stats_sha256": SOURCE_MATRIX_STATS_SHA256,
354
+ "source_provenance_sha256": SOURCE_PROVENANCE_SHA256,
355
+ "source_validation_sha256": SOURCE_VALIDATION_SHA256,
356
+ },
357
+ "model": {
358
+ "architecture": "Qwen2ForCausalLM",
359
+ "d_model": D_MODEL,
360
+ "id": MODEL_ID,
361
+ "n_layers": 36,
362
+ "revision": MODEL_REVISION,
363
+ "revision_binding": "inferred_hub_head_unchanged_since_before_fit",
364
+ "revision_last_modified": "2026-06-30T11:35:41+00:00",
365
+ "tied_embeddings": True,
366
+ },
367
+ "lens": {
368
+ "d_model": D_MODEL,
369
+ "dtype": "float32",
370
+ "hook_convention": "post_transformer_block_output_residual",
371
+ "n_prompts": N_PROMPTS,
372
+ "source_layers": list(SOURCE_LAYERS),
373
+ "target_layer": TARGET_LAYER,
374
+ "estimator": {
375
+ "dim_batch": 8,
376
+ "exclude_final_position": True,
377
+ "fit_dtype": "bfloat16",
378
+ "max_seq_len": 128,
379
+ "name": "causal_all_current_and_future_targets_mean_jacobian",
380
+ "prompt_aggregation": "equal_weight_mean_over_prompts",
381
+ "skip_first": 16,
382
+ "source_position_aggregation": "mean_over_valid_source_positions",
383
+ "target_position_aggregation": (
384
+ "sum_over_valid_targets_at_or_after_source"
385
+ ),
386
+ },
387
+ },
388
+ "compatibility": {
389
+ "file": "evaluation_compatibility.json",
390
+ "fp16_artifact": "model.safetensors",
391
+ "fp16_artifact_sha256": FP16_ARTIFACT_SHA256,
392
+ "all_layer_casts_match_exactly": True,
393
+ },
394
+ "software": {
395
+ "conversion_runtime": {
396
+ "python_implementation": "CPython",
397
+ "safetensors": safetensors.__version__,
398
+ "torch": torch.__version__,
399
+ },
400
+ "anthropic_jacobian_lens_commit": (
401
+ "581d398613e5602a5af361e1c34d3a92ea82ba8e"
402
+ ),
403
+ },
404
+ "limitations": [
405
+ "The original fit did not store the resolved Hugging Face commit; the revision binding was reconstructed from the Hub head and its last-modified timestamp.",
406
+ "The recorded evaluation validates token readout and does not establish causal steering or a global workspace.",
407
+ ],
408
+ }
409
+
410
+ write_json(output_dir / "evaluation_tensor_manifest.json", manifest)
411
+ write_json(output_dir / "evaluation_compatibility.json", compatibility)
412
+ write_json(output_dir / "evaluation_validation.json", validation)
413
+ write_json(output_dir / "evaluation_provenance.json", provenance)
414
+ return {
415
+ "artifact": output_path.name,
416
+ "artifact_sha256": output_digest,
417
+ "artifact_size_bytes": output_size,
418
+ "exact_tensor_matches": exact_matches,
419
+ "fp16_cast_matches": compatibility["all_layer_casts_match_exactly"],
420
+ "source_checkpoint_sha256": source_digest,
421
+ }
422
+
423
+
424
+ def main() -> None:
425
+ parser = argparse.ArgumentParser(description=__doc__)
426
+ parser.add_argument("--source", required=True, type=Path)
427
+ parser.add_argument(
428
+ "--output-dir",
429
+ type=Path,
430
+ default=Path(__file__).resolve().parents[1],
431
+ )
432
+ parser.add_argument("--overwrite", action="store_true")
433
+ args = parser.parse_args()
434
+ result = convert(args.source, args.output_dir.resolve(), args.overwrite)
435
+ print(json.dumps(result, indent=2, sort_keys=True))
436
+
437
+
438
+ if __name__ == "__main__":
439
+ main()
scripts/validate_artifact.py ADDED
@@ -0,0 +1,1395 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ """Validate the public model-release candidate without the private source file."""
3
+
4
+ from __future__ import annotations
5
+
6
+ import argparse
7
+ import hashlib
8
+ import json
9
+ import math
10
+ import re
11
+ import subprocess
12
+ from pathlib import Path
13
+ from typing import Any
14
+ from urllib.parse import urlsplit
15
+
16
+ import torch
17
+ from safetensors import safe_open
18
+
19
+ MODEL_ID = "WeiboAI/VibeThinker-3B"
20
+ MODEL_REVISION = "77bd2cced09193c8b9a59a32bd8577bbd1f3e01c"
21
+ SOURCE_CHECKPOINT_SHA256 = (
22
+ "f36a99447623e0d777c70951a9148a7a52e42e0df82942e22ce6326f63d8d664"
23
+ )
24
+ ARTIFACT_SHA256 = "089d776979408f23e5377539c15aa8025d171633718ccdce709bcd3372e7942c"
25
+ ARTIFACT_SIZE_BYTES = 150_996_824
26
+ EVALUATION_LENS_SHA256 = (
27
+ "8f752032a26a5196c1cb447ef63f01e8a29820ff0c57178dd80c6e26d32b12a9"
28
+ )
29
+ EVALUATION_ARTIFACT_SHA256 = (
30
+ "0cc184eb65d273ac8bfee5450a141c8cdf8dd6ec8caa68d7d47260ef621777d1"
31
+ )
32
+ EVALUATION_ARTIFACT_SIZE_BYTES = 301_991_904
33
+ EVALUATION_FILE_SHA256 = (
34
+ "073f2886e370acec7dd1564f2d7e834b4253493e84b026ec0cc017a6069a5f78"
35
+ )
36
+ SOURCE_LAYERS = tuple(range(0, 36, 2))
37
+ SELECTED_BAND = (24, 26, 28, 30, 32, 34)
38
+ D_MODEL = 2048
39
+ TENSOR_NBYTES = D_MODEL * D_MODEL * 2
40
+ TENSOR_STORAGE_BYTES = len(SOURCE_LAYERS) * TENSOR_NBYTES
41
+ EVALUATION_TENSOR_NBYTES = D_MODEL * D_MODEL * 4
42
+ EVALUATION_TENSOR_STORAGE_BYTES = len(SOURCE_LAYERS) * EVALUATION_TENSOR_NBYTES
43
+ EXPECTED_KEYS = {f"J.{layer}" for layer in SOURCE_LAYERS}
44
+ EXPECTED_TENSOR_RECORD_KEYS = {
45
+ "dtype",
46
+ "nbytes",
47
+ "numel",
48
+ "sha256_c_contiguous_little_endian_bytes",
49
+ "shape",
50
+ "source_layer",
51
+ }
52
+ EXPECTED_METADATA = {
53
+ "artifact_kind": "jacobian_lens",
54
+ "d_model": str(D_MODEL),
55
+ "format": "pt",
56
+ "model_id": MODEL_ID,
57
+ "model_revision": MODEL_REVISION,
58
+ "n_prompts": "1000",
59
+ "schema_version": "1",
60
+ "source_checkpoint_sha256": SOURCE_CHECKPOINT_SHA256,
61
+ "source_layers": "[0,2,4,6,8,10,12,14,16,18,20,22,24,26,28,30,32,34]",
62
+ "target_layer": "35",
63
+ "tensor_dtype": "float16",
64
+ "tensor_key_pattern": "J.{source_layer}",
65
+ }
66
+ EXPECTED_EVALUATION_METADATA = {
67
+ "artifact_kind": "jacobian_lens_evaluation_fp32",
68
+ "d_model": str(D_MODEL),
69
+ "format": "pt",
70
+ "model_id": MODEL_ID,
71
+ "model_revision": MODEL_REVISION,
72
+ "n_prompts": "1000",
73
+ "schema_version": "1",
74
+ "source_fit_checkpoint_sha256": (
75
+ "a1236cfe5d04601575b3de150ffe50e3a67e755ede1197ecf74c205a31bdc258"
76
+ ),
77
+ "source_fp32_checkpoint_sha256": EVALUATION_LENS_SHA256,
78
+ "source_layers": "[0,2,4,6,8,10,12,14,16,18,20,22,24,26,28,30,32,34]",
79
+ "target_layer": "35",
80
+ "tensor_dtype": "float32",
81
+ "tensor_key_pattern": "J.{source_layer}",
82
+ }
83
+ EXPECTED_CARD_FRONT_MATTER = """license: other
84
+ license_name: qwen-research-license
85
+ license_link: https://huggingface.co/JacobMolBio/vibethinker-3b-jlens-model/blob/main/LICENSES/QWEN-RESEARCH.txt
86
+ tags:
87
+ - vibethinker-3b
88
+ - jacobian-lens
89
+ - mechanistic-interpretability
90
+ - interpretability
91
+ - qwen2
92
+ - safetensors"""
93
+ EXPECTED_RELEASE_FILES = {
94
+ ".gitattributes",
95
+ ".gitignore",
96
+ "assets/jlens-model-banner.png",
97
+ "assets/two-lens-files.png",
98
+ "assets/two-lens-files.svg",
99
+ "LICENSES/APACHE-2.0.txt",
100
+ "LICENSES/QWEN-RESEARCH.txt",
101
+ "LICENSES/VIBETHINKER-LICENSE-NOTE.txt",
102
+ "NOTICE",
103
+ "README.md",
104
+ "SHA256SUMS",
105
+ "THIRD_PARTY_NOTICES.md",
106
+ "evaluation.json",
107
+ "evaluation.safetensors",
108
+ "evaluation_compatibility.json",
109
+ "evaluation_provenance.json",
110
+ "evaluation_tensor_manifest.json",
111
+ "evaluation_validation.json",
112
+ "lens_config.json",
113
+ "model.safetensors",
114
+ "provenance.json",
115
+ "requirements.txt",
116
+ "scripts/convert_checkpoint.py",
117
+ "scripts/convert_evaluation_checkpoint.py",
118
+ "scripts/validate_artifact.py",
119
+ "tensor_manifest.json",
120
+ "validation.json",
121
+ }
122
+ PINNED_PUBLIC_FILE_SHA256 = {
123
+ ".gitattributes": "cd0273298656ca90cb8d08fd25b7601483f7e13417e3115ab809856c42e794f7",
124
+ ".gitignore": "4c7486a4b7c5ad04e0e62225b9a55c9ae00b039168d34614d402ac1f73acc459",
125
+ "assets/jlens-model-banner.png": "f27c4abb0c84481c2b0e67ed0c390716905bd1fe6fbcb09f2a65bbe15ecdc935",
126
+ "LICENSES/APACHE-2.0.txt": "ec01a6d25ea6a6b50430eda7af9e23e1c510048502d0a3adb5b54559e687d4fa",
127
+ "LICENSES/QWEN-RESEARCH.txt": "ef52482bb785733093dc9a2e8edd8e764c77d12d8e9d8f10a80c9b547d32d0f9",
128
+ "LICENSES/VIBETHINKER-LICENSE-NOTE.txt": "a74ea19436fbab1210ff16aa9037337d95ca9f93675406da4383bd8de4f4d25b",
129
+ "NOTICE": "35d9d57e97593a99aff4434875b43943d7627258495466bd5f8f81cdd8d40289",
130
+ "THIRD_PARTY_NOTICES.md": "d68e3b08646b56bf8d8c80d1d1acd98edfb40e72cb287762a3150d9b9d75a18f",
131
+ }
132
+ PUBLICATION_ROWS = {
133
+ "CODE_REPOSITORY_URL": "Source code and Pages source",
134
+ "TRACE_REPOSITORY_URL": "Captured trace dataset",
135
+ "PUBLIC_SITE_URL": "Static Pages site",
136
+ "MODEL_REPOSITORY_ID": "Hugging Face model repository ID",
137
+ }
138
+ PUBLICATION_PLACEHOLDERS = {key: "{{" + key + "}}" for key in PUBLICATION_ROWS}
139
+ CODE_REPOSITORY_NAME = "vibethinker-3b-jlens"
140
+ TRACE_REPOSITORY_NAME = "vibethinker-3b-jlens-traces"
141
+ MODEL_REPOSITORY_NAME = "vibethinker-3b-jlens-model"
142
+ REPOSITORY_COMPONENT = re.compile(r"[A-Za-z0-9][A-Za-z0-9_.-]*")
143
+ SHA256_PATTERN = re.compile(r"[0-9a-f]{64}")
144
+ PUBLICATION_SENTINELS = {
145
+ "",
146
+ "example",
147
+ "local",
148
+ "none",
149
+ "null",
150
+ "org",
151
+ "organization",
152
+ "placeholder",
153
+ "repo",
154
+ "repository",
155
+ "tbd",
156
+ "todo",
157
+ "user",
158
+ "username",
159
+ }
160
+ FORBIDDEN_PUBLIC_PATTERNS = {
161
+ "email address": re.compile(
162
+ r"\b[A-Za-z0-9.!#$%&'*+/=?^_`{|}~-]+"
163
+ + chr(64)
164
+ + r"[A-Za-z0-9](?:[A-Za-z0-9.-]{0,61}[A-Za-z0-9])?"
165
+ + r"\.[A-Za-z]{2,}\b"
166
+ ),
167
+ "AWS access key": re.compile(r"\b" + "AK" + r"IA[0-9A-Z]{16}\b"),
168
+ "AWS temporary access key": re.compile(r"\b" + "AS" + r"IA[0-9A-Z]{16}\b"),
169
+ "GitHub access token": re.compile(r"\b" + "gh" + r"[pousr]_[A-Za-z0-9]{20,}\b"),
170
+ "GitHub fine-grained token": re.compile(
171
+ r"\b" + "github" + r"_pat_[A-Za-z0-9_]{20,}\b"
172
+ ),
173
+ "GitLab access token": re.compile(r"\b" + "gl" + r"pat-[A-Za-z0-9_-]{20,}\b"),
174
+ "Hugging Face access token": re.compile(r"\b" + "hf" + r"_[A-Za-z0-9]{20,}\b"),
175
+ "OpenAI-style access token": re.compile(r"\b" + "sk" + r"-[A-Za-z0-9_-]{20,}\b"),
176
+ "Google API key": re.compile(r"\b" + "AI" + r"za[0-9A-Za-z_-]{30,}\b"),
177
+ "Slack access token": re.compile(r"\b" + "xo" + r"[abprs]-[A-Za-z0-9-]{20,}\b"),
178
+ "private key": re.compile(
179
+ "-----BEGIN " + r"(?:DSA |EC |OPENSSH |RSA )?PRIVATE KEY-----"
180
+ ),
181
+ "bearer credential": re.compile(
182
+ r"\b" + "Bearer" + r"\s+[A-Za-z0-9._~+/=-]{20,}", re.IGNORECASE
183
+ ),
184
+ "local file URI": re.compile(r"\b" + "file:" + r"//", re.IGNORECASE),
185
+ "macOS user path": re.compile(r"(?<![A-Za-z0-9:])/" + r"Users/[^/\s]+/"),
186
+ "mounted volume path": re.compile(r"(?<![A-Za-z0-9:])/" + r"Volumes/[^/\s]+/"),
187
+ "Unix home path": re.compile(r"(?<![A-Za-z0-9:])/" + r"home/[^/\s]+/"),
188
+ "root home path": re.compile(r"(?<![A-Za-z0-9:])/" + r"root(?:/|\b)"),
189
+ "temporary path": re.compile(
190
+ r"(?<![A-Za-z0-9:])/" + r"(?:tmp|private/tmp|var/folders)/"
191
+ ),
192
+ "workspace path": re.compile(r"(?<![A-Za-z0-9:])/" + r"workspaces?/[^\s]+"),
193
+ "mounted data path": re.compile(r"(?<![A-Za-z0-9:])/" + r"mnt/[^\s]+"),
194
+ "Windows user path": re.compile(r"[A-Za-z]:\\" + r"Users\\[^\\\s]+\\"),
195
+ "loopback hostname": re.compile(r"\b" + "local" + r"host\b", re.IGNORECASE),
196
+ "loopback IPv4 address": re.compile(r"\b127(?:\.[0-9]{1,3}){3}\b"),
197
+ "unspecified IPv4 address": re.compile(r"\b0\.0\.0\.0\b"),
198
+ }
199
+ FORBIDDEN_PUBLIC_LITERALS = {
200
+ "local account name": "jacob" + "vogan",
201
+ "local account alias": "jaco" + "vogan",
202
+ "invented VibeThinker copyright": "Copyright (c) 2025 " + "WeiboAI",
203
+ "removed VibeThinker license filename": "VIBETHINKER-" + "MIT.txt",
204
+ }
205
+ EXPECTED_EVALUATION_CLAIM_BOUNDARY = (
206
+ "This reference records the V1 readout results. It excludes causal-assay "
207
+ "results and does not establish free-generation steering or a global "
208
+ "workspace."
209
+ )
210
+ EXPECTED_EVALUATION_NOTE = (
211
+ "The recorded readout metrics bind to evaluation.safetensors, the FP32 "
212
+ "evaluation lens in this repository. They do not evaluate "
213
+ "model.safetensors, the FP16 lens used for the captured traces."
214
+ )
215
+ EXPECTED_EVALUATION_TASK = {
216
+ "aggregate_mean_reciprocal_rank": (
217
+ "mean_across_eligible_target_terms_of_reciprocal_best_rank"
218
+ ),
219
+ "eligible_target": ("at_least_one_candidate_form_tokenizes_to_exactly_one_token"),
220
+ "final_model": (
221
+ "rank_target_terms_in_the_model_next_token_logits_at_the_score_position"
222
+ ),
223
+ "item": "one_prompt_with_one_or_more_target_terms",
224
+ "layer_scope_reduction": ("best_target_rank_across_layers_in_the_reported_scope"),
225
+ "no_eligible_target_item": (
226
+ "none_of_the_item_target_terms_has_an_eligible_single_token_form"
227
+ ),
228
+ "paired_bootstrap_mean_reciprocal_rank": (
229
+ "within_item_mean_of_reciprocal_best_rank_across_eligible_target_terms"
230
+ ),
231
+ "pass_at_k": (
232
+ "mean_across_items_of_the_fraction_of_item_target_terms_with_best_rank_at_most_k"
233
+ ),
234
+ "score_position": {
235
+ "default": "final_prompt_token",
236
+ "poetry": "last_newline_token",
237
+ },
238
+ "split": {
239
+ "dev_fraction": 0.3,
240
+ "method": "sha256_stable_split",
241
+ "seed": "vibethinker-jlens-v1",
242
+ },
243
+ "suites": [
244
+ "association",
245
+ "multihop",
246
+ "multilingual",
247
+ "order-ops",
248
+ "poetry",
249
+ "typo",
250
+ ],
251
+ "target_candidate_forms": [
252
+ "original_lowercase_and_capitalized_forms_with_and_without_leading_space",
253
+ "order_ops_also_adds_configured_operation_synonyms_and_number_word_digit_forms",
254
+ ],
255
+ "target_rank": ("best_one_based_vocabulary_rank_among_eligible_target_token_ids"),
256
+ }
257
+
258
+
259
+ def require(condition: bool, message: str) -> None:
260
+ if not condition:
261
+ raise ValueError(message)
262
+
263
+
264
+ def sha256_file(path: Path) -> str:
265
+ digest = hashlib.sha256()
266
+ with path.open("rb") as handle:
267
+ for chunk in iter(lambda: handle.read(8 * 1024 * 1024), b""):
268
+ digest.update(chunk)
269
+ return digest.hexdigest()
270
+
271
+
272
+ def tensor_sha256(tensor: torch.Tensor) -> str:
273
+ contiguous = tensor.detach().cpu().contiguous()
274
+ if tensor.dtype == torch.float16:
275
+ array = contiguous.view(torch.int16).numpy()
276
+ payload = array.astype("<i2", copy=False).tobytes(order="C")
277
+ elif tensor.dtype == torch.float32:
278
+ array = contiguous.view(torch.int32).numpy()
279
+ payload = array.astype("<i4", copy=False).tobytes(order="C")
280
+ else:
281
+ raise ValueError(f"unsupported tensor dtype for hashing: {tensor.dtype}")
282
+ return hashlib.sha256(payload).hexdigest()
283
+
284
+
285
+ def read_json(path: Path) -> dict[str, Any]:
286
+ value = json.loads(path.read_text(encoding="utf-8"))
287
+ require(isinstance(value, dict), f"{path.name} must contain a JSON object")
288
+ return value
289
+
290
+
291
+ def normalized_words(value: str) -> str:
292
+ return " ".join(value.split())
293
+
294
+
295
+ def extract_card_front_matter(readme: str) -> str:
296
+ require(readme.startswith("---\n"), "README model-card front matter is missing")
297
+ closing = readme.find("\n---\n", 4)
298
+ require(closing != -1, "README model-card front matter is not closed")
299
+ return readme[4:closing]
300
+
301
+
302
+ def validate_release_file_set(root: Path) -> None:
303
+ actual_files: set[str] = set()
304
+ for path in root.rglob("*"):
305
+ relative = path.relative_to(root)
306
+ if relative.parts and relative.parts[0] == ".git":
307
+ continue
308
+ require(not path.is_symlink(), f"release tree contains a symlink: {relative}")
309
+ if path.is_file():
310
+ actual_files.add(relative.as_posix())
311
+ require(
312
+ actual_files == EXPECTED_RELEASE_FILES,
313
+ "release file set mismatch: "
314
+ f"missing={sorted(EXPECTED_RELEASE_FILES - actual_files)}, "
315
+ f"extra={sorted(actual_files - EXPECTED_RELEASE_FILES)}",
316
+ )
317
+
318
+
319
+ def validate_pinned_public_files(root: Path) -> None:
320
+ for relative_path, expected_sha256 in PINNED_PUBLIC_FILE_SHA256.items():
321
+ require(
322
+ sha256_file(root / relative_path) == expected_sha256,
323
+ f"pinned public file hash mismatch: {relative_path}",
324
+ )
325
+
326
+
327
+ def validate_public_text(root: Path) -> None:
328
+ binary_files = {
329
+ "assets/jlens-model-banner.png",
330
+ "assets/two-lens-files.png",
331
+ "evaluation.safetensors",
332
+ "model.safetensors",
333
+ }
334
+ for relative_path in sorted(EXPECTED_RELEASE_FILES - binary_files):
335
+ text = (root / relative_path).read_text(encoding="utf-8")
336
+ for label, pattern in FORBIDDEN_PUBLIC_PATTERNS.items():
337
+ require(
338
+ pattern.search(text) is None,
339
+ f"{relative_path} contains a forbidden {label}",
340
+ )
341
+ for label, literal in FORBIDDEN_PUBLIC_LITERALS.items():
342
+ require(
343
+ literal not in text,
344
+ f"{relative_path} contains a forbidden {label}",
345
+ )
346
+
347
+
348
+ def parse_publication_rows(readme: str) -> tuple[dict[str, str], list[str]]:
349
+ values: dict[str, str] = {}
350
+ for key, label in PUBLICATION_ROWS.items():
351
+ pattern = re.compile(
352
+ rf"^\| {re.escape(label)} \| (?:`([^`\r\n]+)`|\[([^\]\r\n]+)\]\([^\)\r\n]+\)) \|$",
353
+ re.MULTILINE,
354
+ )
355
+ matches = pattern.findall(readme)
356
+ require(len(matches) == 1, f"README publication row mismatch: {key}")
357
+ values[key] = matches[0][0] or matches[0][1]
358
+
359
+ placeholders_in_readme = set(re.findall(r"\{\{([A-Z0-9_]+)\}\}", readme))
360
+ unresolved = sorted(
361
+ key for key, value in values.items() if value == PUBLICATION_PLACEHOLDERS[key]
362
+ )
363
+ require(
364
+ placeholders_in_readme == set(unresolved),
365
+ "README publication placeholders do not match the publication table",
366
+ )
367
+ require(
368
+ len(unresolved) in {0, len(PUBLICATION_ROWS)},
369
+ "publication fields must be fully unresolved or fully resolved",
370
+ )
371
+ require(
372
+ f'export JLENS_CODE_REPO_URL="{values["CODE_REPOSITORY_URL"]}"' in readme,
373
+ "source URL example differs from the publication table",
374
+ )
375
+ require(
376
+ f'export JLENS_MODEL_REPO_ID="{values["MODEL_REPOSITORY_ID"]}"' in readme,
377
+ "model repository example differs from the publication table",
378
+ )
379
+ return values, unresolved
380
+
381
+
382
+ def require_public_component(value: str, label: str) -> None:
383
+ require(value == value.strip(), f"{label} has surrounding whitespace")
384
+ require(value.casefold() not in PUBLICATION_SENTINELS, f"{label} is a placeholder")
385
+ require("{{" not in value and "}}" not in value, f"{label} is unresolved")
386
+ require(
387
+ not any(character.isspace() for character in value),
388
+ f"{label} contains whitespace",
389
+ )
390
+ require("\\" not in value, f"{label} contains a local path separator")
391
+
392
+
393
+ def parse_https_url(value: str, label: str) -> Any:
394
+ require_public_component(value, label)
395
+ parsed = urlsplit(value)
396
+ require(parsed.scheme == "https", f"{label} must use HTTPS")
397
+ require(parsed.hostname is not None, f"{label} has no hostname")
398
+ require(
399
+ parsed.username is None and parsed.password is None,
400
+ f"{label} contains credentials",
401
+ )
402
+ try:
403
+ port = parsed.port
404
+ except ValueError as error:
405
+ raise ValueError(f"{label} has an invalid port") from error
406
+ require(port is None, f"{label} must not specify a port")
407
+ require(not parsed.query, f"{label} must not contain a query")
408
+ require(not parsed.fragment, f"{label} must not contain a fragment")
409
+ require("%" not in parsed.path, f"{label} must not contain encoded path fragments")
410
+ require("//" not in parsed.path, f"{label} contains an empty path fragment")
411
+ require(";" not in parsed.path, f"{label} contains a parameter fragment")
412
+ return parsed
413
+
414
+
415
+ def validate_publication_values(values: dict[str, str]) -> None:
416
+ code = parse_https_url(values["CODE_REPOSITORY_URL"], "source repository URL")
417
+ trace = parse_https_url(values["TRACE_REPOSITORY_URL"], "trace repository URL")
418
+ site = parse_https_url(values["PUBLIC_SITE_URL"], "Pages site URL")
419
+ model_id = values["MODEL_REPOSITORY_ID"]
420
+ require_public_component(model_id, "model repository ID")
421
+
422
+ code_parts = [part for part in code.path.split("/") if part]
423
+ require(
424
+ code.hostname.casefold() == "github.com",
425
+ "source repository must use github.com",
426
+ )
427
+ require(
428
+ len(code_parts) == 2, "source repository URL must contain namespace/repository"
429
+ )
430
+ require(
431
+ code.path == "/" + "/".join(code_parts),
432
+ "source repository URL is not canonical",
433
+ )
434
+ require(
435
+ all(REPOSITORY_COMPONENT.fullmatch(part) for part in code_parts),
436
+ "source repository URL contains an invalid path fragment",
437
+ )
438
+ require(
439
+ code_parts[0].casefold() not in PUBLICATION_SENTINELS,
440
+ "source namespace is a placeholder",
441
+ )
442
+ require(code_parts[1] == CODE_REPOSITORY_NAME, "source repository name mismatch")
443
+
444
+ trace_parts = [part for part in trace.path.split("/") if part]
445
+ require(
446
+ trace.hostname.casefold() == "huggingface.co",
447
+ "trace repository must use huggingface.co",
448
+ )
449
+ require(
450
+ len(trace_parts) == 3 and trace_parts[0] == "datasets",
451
+ "trace repository URL must use /datasets/namespace/repository",
452
+ )
453
+ require(
454
+ trace.path == "/" + "/".join(trace_parts),
455
+ "trace repository URL is not canonical",
456
+ )
457
+ require(
458
+ all(REPOSITORY_COMPONENT.fullmatch(part) for part in trace_parts[1:]),
459
+ "trace repository URL contains an invalid path fragment",
460
+ )
461
+ require(
462
+ trace_parts[1].casefold() not in PUBLICATION_SENTINELS,
463
+ "trace namespace is a placeholder",
464
+ )
465
+ require(trace_parts[2] == TRACE_REPOSITORY_NAME, "trace repository name mismatch")
466
+
467
+ model_parts = model_id.split("/")
468
+ require(len(model_parts) == 2, "model repository ID must use namespace/repository")
469
+ require(
470
+ all(REPOSITORY_COMPONENT.fullmatch(part) for part in model_parts),
471
+ "model repository ID contains an invalid fragment",
472
+ )
473
+ require(
474
+ model_parts[0].casefold() not in PUBLICATION_SENTINELS,
475
+ "model namespace is a placeholder",
476
+ )
477
+ require(model_parts[1] == MODEL_REPOSITORY_NAME, "model repository name mismatch")
478
+ require(
479
+ model_parts[0].casefold() == trace_parts[1].casefold(),
480
+ "model and trace repositories must use the same Hugging Face namespace",
481
+ )
482
+
483
+ require(
484
+ site.hostname.casefold() == f"{code_parts[0].casefold()}.github.io",
485
+ "Pages hostname must match the source repository namespace",
486
+ )
487
+ require(
488
+ site.path.rstrip("/") == f"/{CODE_REPOSITORY_NAME}",
489
+ "Pages path must match the source repository name",
490
+ )
491
+ require(
492
+ site.path in {f"/{CODE_REPOSITORY_NAME}", f"/{CODE_REPOSITORY_NAME}/"},
493
+ "Pages URL is not canonical",
494
+ )
495
+
496
+
497
+ def validate_model_card(readme: str) -> None:
498
+ require(
499
+ extract_card_front_matter(readme) == EXPECTED_CARD_FRONT_MATTER,
500
+ "README model-card front matter mismatch",
501
+ )
502
+ readme_words = normalized_words(readme)
503
+ for required_text in (
504
+ MODEL_ID,
505
+ MODEL_REVISION,
506
+ ARTIFACT_SHA256,
507
+ EVALUATION_LENS_SHA256,
508
+ EVALUATION_ARTIFACT_SHA256,
509
+ "LICENSES/VIBETHINKER-LICENSE-NOTE.txt",
510
+ "The static site cannot analyze a new prompt.",
511
+ "50,050 readout rows",
512
+ ):
513
+ require(
514
+ required_text in readme_words,
515
+ f"README missing required binding: {required_text}",
516
+ )
517
+
518
+
519
+ def validate_licenses_and_notices(root: Path) -> None:
520
+ notice = (root / "NOTICE").read_text(encoding="utf-8")
521
+ third_party = (root / "THIRD_PARTY_NOTICES.md").read_text(encoding="utf-8")
522
+ note = (root / "LICENSES/VIBETHINKER-LICENSE-NOTE.txt").read_text(encoding="utf-8")
523
+ qwen_license = (root / "LICENSES/QWEN-RESEARCH.txt").read_text(encoding="utf-8")
524
+ required_qwen_notice = (
525
+ "Qwen is licensed under the Qwen RESEARCH LICENSE AGREEMENT, "
526
+ "Copyright (c) Alibaba Cloud. All Rights Reserved."
527
+ )
528
+ require(
529
+ required_qwen_notice in normalized_words(notice),
530
+ "NOTICE is missing the required Qwen attribution",
531
+ )
532
+ require(
533
+ required_qwen_notice in normalized_words(qwen_license),
534
+ "Qwen license is missing its attribution clause",
535
+ )
536
+ require(
537
+ "Built with Qwen." in notice, "NOTICE is missing the Qwen product attribution"
538
+ )
539
+ require(
540
+ "LICENSES/VIBETHINKER-LICENSE-NOTE.txt" in notice
541
+ and "LICENSES/VIBETHINKER-LICENSE-NOTE.txt" in third_party,
542
+ "VibeThinker metadata note is not linked from the notices",
543
+ )
544
+ require(
545
+ "license: mit" in normalized_words(note),
546
+ "VibeThinker metadata declaration missing",
547
+ )
548
+ require(MODEL_REVISION in note, "VibeThinker metadata note revision mismatch")
549
+ require(
550
+ "It is not an upstream license text" in normalized_words(note),
551
+ "VibeThinker metadata note scope missing",
552
+ )
553
+ require(
554
+ not (root / "LICENSES" / ("VIBETHINKER-" + "MIT.txt")).exists(),
555
+ "removed VibeThinker license file is present",
556
+ )
557
+
558
+
559
+ def validate_metadata_records(
560
+ config: dict[str, Any],
561
+ provenance: dict[str, Any],
562
+ frozen_validation: dict[str, Any],
563
+ evaluation: dict[str, Any],
564
+ evaluation_path: Path,
565
+ ) -> None:
566
+ require(config.get("schema_version") == 1, "config schema mismatch")
567
+ require(config.get("artifact_kind") == "jacobian_lens", "config kind mismatch")
568
+ require(config.get("d_model") == D_MODEL, "config width mismatch")
569
+ require(config.get("n_prompts") == 1000, "config prompt count mismatch")
570
+ require(
571
+ config.get("source_layers") == list(SOURCE_LAYERS),
572
+ "config source layers mismatch",
573
+ )
574
+ require(config.get("target_layer") == 35, "config target layer mismatch")
575
+ require(config.get("tensor_dtype") == "float16", "config tensor dtype mismatch")
576
+ require(
577
+ config.get("tensor_key_pattern") == "J.{source_layer}",
578
+ "config key pattern mismatch",
579
+ )
580
+ require(
581
+ config.get("model")
582
+ == {
583
+ "architecture": "Qwen2ForCausalLM",
584
+ "id": MODEL_ID,
585
+ "n_layers": 36,
586
+ "revision": MODEL_REVISION,
587
+ },
588
+ "config model binding mismatch",
589
+ )
590
+ require(
591
+ config.get("source_checkpoint", {}).get("sha256") == SOURCE_CHECKPOINT_SHA256,
592
+ "config source checkpoint mismatch",
593
+ )
594
+ require(
595
+ config.get("artifact")
596
+ == {
597
+ "filename": "model.safetensors",
598
+ "format": "safetensors",
599
+ "sha256": ARTIFACT_SHA256,
600
+ "size_bytes": ARTIFACT_SIZE_BYTES,
601
+ },
602
+ "config artifact binding mismatch",
603
+ )
604
+ require(
605
+ config.get("evaluation_artifact")
606
+ == {
607
+ "filename": "evaluation.safetensors",
608
+ "format": "safetensors",
609
+ "sha256": EVALUATION_ARTIFACT_SHA256,
610
+ "size_bytes": EVALUATION_ARTIFACT_SIZE_BYTES,
611
+ "source_checkpoint_sha256": EVALUATION_LENS_SHA256,
612
+ "tensor_conversion": "lossless_fp32_reserialization",
613
+ "tensor_dtype": "float32",
614
+ },
615
+ "config evaluation artifact binding mismatch",
616
+ )
617
+
618
+ require(provenance.get("schema_version") == 1, "provenance schema mismatch")
619
+ require(
620
+ provenance.get("model", {}).get("id") == MODEL_ID,
621
+ "provenance model ID mismatch",
622
+ )
623
+ require(
624
+ provenance.get("model", {}).get("revision") == MODEL_REVISION,
625
+ "provenance model revision mismatch",
626
+ )
627
+ require(
628
+ provenance.get("artifact", {}).get("sha256") == ARTIFACT_SHA256,
629
+ "provenance artifact hash mismatch",
630
+ )
631
+ require(
632
+ provenance.get("artifact", {}).get("size_bytes") == ARTIFACT_SIZE_BYTES,
633
+ "provenance artifact size mismatch",
634
+ )
635
+ require(
636
+ provenance.get("artifact", {}).get("tensor_conversion")
637
+ == "lossless_fp16_reserialization",
638
+ "provenance conversion boundary mismatch",
639
+ )
640
+ require(
641
+ provenance.get("source_checkpoint", {}).get("sha256")
642
+ == SOURCE_CHECKPOINT_SHA256,
643
+ "provenance source checkpoint mismatch",
644
+ )
645
+
646
+ require(frozen_validation.get("schema_version") == 1, "validation schema mismatch")
647
+ require(
648
+ frozen_validation.get("artifact") == "model.safetensors",
649
+ "validation filename mismatch",
650
+ )
651
+ require(
652
+ frozen_validation.get("artifact_sha256") == ARTIFACT_SHA256,
653
+ "validation artifact hash mismatch",
654
+ )
655
+ require(
656
+ frozen_validation.get("artifact_size_bytes") == ARTIFACT_SIZE_BYTES,
657
+ "validation artifact size mismatch",
658
+ )
659
+ require(frozen_validation.get("ok") is True, "frozen validation is not successful")
660
+ require(
661
+ frozen_validation.get("tensor_values_changed") == 0,
662
+ "frozen validation records changed values",
663
+ )
664
+ require(
665
+ frozen_validation.get("exact_tensor_matches") == len(SOURCE_LAYERS),
666
+ "frozen validation tensor count mismatch",
667
+ )
668
+ require(
669
+ frozen_validation.get("expected_tensor_matches") == len(SOURCE_LAYERS),
670
+ "frozen validation expected tensor count mismatch",
671
+ )
672
+ require(
673
+ frozen_validation.get("source_checkpoint_sha256") == SOURCE_CHECKPOINT_SHA256,
674
+ "frozen validation source checkpoint mismatch",
675
+ )
676
+ checks = frozen_validation.get("checks")
677
+ require(
678
+ isinstance(checks, dict)
679
+ and checks
680
+ and all(value is True for value in checks.values()),
681
+ "frozen validation contains a failed or malformed check",
682
+ )
683
+
684
+ require(
685
+ sha256_file(evaluation_path) == EVALUATION_FILE_SHA256,
686
+ "evaluation file hash mismatch",
687
+ )
688
+ require(
689
+ evaluation.get("artifact_kind") == "frozen_vibethinker_v1_readout_reference",
690
+ "evaluation artifact kind mismatch",
691
+ )
692
+ require(
693
+ evaluation.get("classification") == "validated_readout_only",
694
+ "evaluation classification mismatch",
695
+ )
696
+ require(evaluation.get("model") == MODEL_ID, "evaluation model ID mismatch")
697
+ require(
698
+ evaluation.get("model_revision") == MODEL_REVISION,
699
+ "evaluation model revision mismatch",
700
+ )
701
+ require(
702
+ evaluation.get("lens_sha256") == EVALUATION_LENS_SHA256,
703
+ "evaluation lens hash mismatch",
704
+ )
705
+ require(
706
+ evaluation.get("lens_variant") == "fp32_evaluation_safetensors_included",
707
+ "evaluation lens variant mismatch",
708
+ )
709
+ require(
710
+ evaluation.get("released_lens_artifact")
711
+ == {
712
+ "filename": "evaluation.safetensors",
713
+ "format": "safetensors",
714
+ "sha256": EVALUATION_ARTIFACT_SHA256,
715
+ "size_bytes": EVALUATION_ARTIFACT_SIZE_BYTES,
716
+ "source_checkpoint_sha256": EVALUATION_LENS_SHA256,
717
+ "tensor_conversion": "lossless_fp32_reserialization",
718
+ },
719
+ "evaluation released artifact binding mismatch",
720
+ )
721
+ require(
722
+ evaluation.get("claim_boundary") == EXPECTED_EVALUATION_CLAIM_BOUNDARY,
723
+ "evaluation claim boundary mismatch",
724
+ )
725
+ require(
726
+ evaluation.get("note") == EXPECTED_EVALUATION_NOTE,
727
+ "evaluation FP16/FP32 note mismatch",
728
+ )
729
+ require(
730
+ evaluation.get("selected_band") == list(SELECTED_BAND),
731
+ "evaluation selected band mismatch",
732
+ )
733
+ require(
734
+ evaluation.get("coverage_scope")
735
+ == "frozen_readout_source_run_not_the_100_prompt_ui_test_pack",
736
+ "evaluation coverage scope mismatch",
737
+ )
738
+ require(
739
+ evaluation.get("task_definition") == EXPECTED_EVALUATION_TASK,
740
+ "evaluation task definition mismatch",
741
+ )
742
+ evidence = evaluation.get("evidence", {})
743
+ require(
744
+ evidence.get("validated_readout_signal") is True,
745
+ "evaluation readout signal boundary mismatch",
746
+ )
747
+ require(
748
+ evidence.get("token_specific_vs_shuffled_target") is True,
749
+ "evaluation token control boundary mismatch",
750
+ )
751
+ require(
752
+ evidence.get("layer_mapping_specific_vs_shuffled_jacobian") is True,
753
+ "evaluation layer control boundary mismatch",
754
+ )
755
+ require(
756
+ evidence.get("incremental_over_ordinary_logit_lens") is False,
757
+ "evaluation logit-lens boundary mismatch",
758
+ )
759
+ require(
760
+ evidence.get("paired_bootstrap")
761
+ == {
762
+ "confidence": 0.95,
763
+ "input": "paired_item_metric_differences",
764
+ "interval": "percentile",
765
+ "metrics": ["pass@10", "mean_reciprocal_rank"],
766
+ "samples": 2000,
767
+ "scope": {
768
+ "kind": "selected_band",
769
+ "source_layers": list(SELECTED_BAND),
770
+ },
771
+ "split": "test",
772
+ },
773
+ "evaluation paired-bootstrap method mismatch",
774
+ )
775
+ require(
776
+ evidence.get("paired_bootstrap_decisions")
777
+ == {
778
+ "incremental_over_ordinary_logit_lens": {
779
+ "comparison": "jlens_band_minus_logit_lens_band",
780
+ "criterion": "at_least_one_lower_bound_greater_than_zero",
781
+ "met": False,
782
+ },
783
+ "layer_mapping_specific_vs_shuffled_jacobian": {
784
+ "comparison": "jlens_band_minus_shuffled_layer_band",
785
+ "criterion": "both_lower_bounds_greater_than_zero",
786
+ "met": True,
787
+ },
788
+ "token_specific_vs_shuffled_target": {
789
+ "comparison": "jlens_band_minus_shuffled_token_band",
790
+ "criterion": "both_lower_bounds_greater_than_zero",
791
+ "met": True,
792
+ },
793
+ },
794
+ "evaluation paired-bootstrap decisions mismatch",
795
+ )
796
+ require(
797
+ evidence.get("specificity_scope")
798
+ == {"kind": "selected_band", "source_layers": list(SELECTED_BAND)},
799
+ "evaluation specificity scope mismatch",
800
+ )
801
+ availability = evaluation.get("source_artifact_availability", {})
802
+ require(
803
+ availability
804
+ == {
805
+ "fp32_compatibility_check_output_included": True,
806
+ "fp32_compatibility_check_output_path": ("evaluation_compatibility.json"),
807
+ "fp32_derivation_record_included": True,
808
+ "fp32_derivation_record_path": "evaluation_provenance.json",
809
+ "fp32_evaluation_lens_included": True,
810
+ "fp32_evaluation_lens_path": "evaluation.safetensors",
811
+ "item_level_evaluation_rows_included": False,
812
+ "item_level_evaluation_rows_location": (
813
+ "companion_trace_repository:data/evaluation-results/"
814
+ "readout-trials.jsonl"
815
+ ),
816
+ "paired_bootstrap_interval_bounds_included": False,
817
+ "paired_bootstrap_interval_bounds_location": (
818
+ "companion_trace_repository:data/evaluation-results/"
819
+ "readout-bootstrap-intervals.json"
820
+ ),
821
+ "source_evaluation_bundle_included": False,
822
+ "source_hashes_included": True,
823
+ "source_hashes_location": (
824
+ "evaluation_provenance.json_and_companion_trace_repository:"
825
+ "data/readout-reference.json"
826
+ ),
827
+ "release_content": (
828
+ "fp16_trace_lens_fp32_evaluation_lens_aggregate_reference_"
829
+ "provenance_and_compatibility"
830
+ ),
831
+ },
832
+ "evaluation source-artifact boundary mismatch",
833
+ )
834
+ require(
835
+ EVALUATION_LENS_SHA256 != ARTIFACT_SHA256,
836
+ "FP32 and FP16 artifact hashes were conflated",
837
+ )
838
+
839
+
840
+ def validate_tensor_manifest(manifest: dict[str, Any]) -> None:
841
+ require(
842
+ set(manifest)
843
+ == {
844
+ "artifact",
845
+ "artifact_sha256",
846
+ "artifact_size_bytes",
847
+ "schema_version",
848
+ "tensor_count",
849
+ "tensor_storage_bytes",
850
+ "tensors",
851
+ },
852
+ "tensor manifest top-level fields mismatch",
853
+ )
854
+ require(
855
+ manifest.get("artifact") == "model.safetensors",
856
+ "tensor manifest filename mismatch",
857
+ )
858
+ require(
859
+ manifest.get("artifact_sha256") == ARTIFACT_SHA256,
860
+ "tensor manifest artifact hash mismatch",
861
+ )
862
+ require(
863
+ manifest.get("artifact_size_bytes") == ARTIFACT_SIZE_BYTES,
864
+ "tensor manifest artifact size mismatch",
865
+ )
866
+ require(manifest.get("schema_version") == 1, "tensor manifest schema mismatch")
867
+ require(
868
+ manifest.get("tensor_count") == len(SOURCE_LAYERS),
869
+ "tensor manifest count mismatch",
870
+ )
871
+ require(
872
+ manifest.get("tensor_storage_bytes") == TENSOR_STORAGE_BYTES,
873
+ "tensor manifest storage total mismatch",
874
+ )
875
+ tensors = manifest.get("tensors")
876
+ require(isinstance(tensors, dict), "tensor manifest tensors must be an object")
877
+ require(set(tensors) == EXPECTED_KEYS, "tensor manifest key set mismatch")
878
+
879
+ recorded_storage = 0
880
+ for layer in SOURCE_LAYERS:
881
+ key = f"J.{layer}"
882
+ record = tensors[key]
883
+ require(isinstance(record, dict), f"{key} descriptor must be an object")
884
+ require(
885
+ set(record) == EXPECTED_TENSOR_RECORD_KEYS,
886
+ f"{key} descriptor fields mismatch",
887
+ )
888
+ require(record.get("source_layer") == layer, f"{key} source layer mismatch")
889
+ require(record.get("dtype") == "float16", f"{key} descriptor dtype mismatch")
890
+ require(
891
+ record.get("shape") == [D_MODEL, D_MODEL],
892
+ f"{key} descriptor shape mismatch",
893
+ )
894
+ require(
895
+ record.get("numel") == D_MODEL * D_MODEL,
896
+ f"{key} descriptor element count mismatch",
897
+ )
898
+ require(
899
+ record.get("nbytes") == TENSOR_NBYTES,
900
+ f"{key} descriptor byte count mismatch",
901
+ )
902
+ tensor_hash = record.get("sha256_c_contiguous_little_endian_bytes")
903
+ require(
904
+ isinstance(tensor_hash, str) and SHA256_PATTERN.fullmatch(tensor_hash),
905
+ f"{key} descriptor hash format mismatch",
906
+ )
907
+ recorded_storage += record["nbytes"]
908
+ require(
909
+ recorded_storage == TENSOR_STORAGE_BYTES,
910
+ "tensor descriptor byte total mismatch",
911
+ )
912
+
913
+
914
+ def validate_evaluation_artifact(
915
+ root: Path,
916
+ manifest: dict[str, Any],
917
+ provenance: dict[str, Any],
918
+ frozen_validation: dict[str, Any],
919
+ compatibility: dict[str, Any],
920
+ ) -> int:
921
+ artifact_path = root / "evaluation.safetensors"
922
+ require(
923
+ set(manifest)
924
+ == {
925
+ "artifact",
926
+ "artifact_sha256",
927
+ "artifact_size_bytes",
928
+ "schema_version",
929
+ "source_checkpoint_sha256",
930
+ "tensor_count",
931
+ "tensor_storage_bytes",
932
+ "tensors",
933
+ },
934
+ "evaluation tensor manifest top-level fields mismatch",
935
+ )
936
+ require(
937
+ manifest.get("artifact") == "evaluation.safetensors",
938
+ "evaluation tensor manifest filename mismatch",
939
+ )
940
+ require(
941
+ manifest.get("artifact_sha256") == EVALUATION_ARTIFACT_SHA256,
942
+ "evaluation tensor manifest artifact hash mismatch",
943
+ )
944
+ require(
945
+ manifest.get("artifact_size_bytes") == EVALUATION_ARTIFACT_SIZE_BYTES,
946
+ "evaluation tensor manifest artifact size mismatch",
947
+ )
948
+ require(
949
+ manifest.get("source_checkpoint_sha256") == EVALUATION_LENS_SHA256,
950
+ "evaluation tensor manifest source hash mismatch",
951
+ )
952
+ require(
953
+ manifest.get("tensor_count") == len(SOURCE_LAYERS),
954
+ "evaluation tensor manifest count mismatch",
955
+ )
956
+ require(
957
+ manifest.get("tensor_storage_bytes") == EVALUATION_TENSOR_STORAGE_BYTES,
958
+ "evaluation tensor manifest storage total mismatch",
959
+ )
960
+ tensors = manifest.get("tensors")
961
+ require(isinstance(tensors, dict), "evaluation manifest tensors must be an object")
962
+ require(set(tensors) == EXPECTED_KEYS, "evaluation manifest key set mismatch")
963
+ for layer in SOURCE_LAYERS:
964
+ key = f"J.{layer}"
965
+ record = tensors[key]
966
+ require(
967
+ set(record) == EXPECTED_TENSOR_RECORD_KEYS,
968
+ f"evaluation {key} descriptor fields mismatch",
969
+ )
970
+ require(record.get("source_layer") == layer, f"evaluation {key} layer mismatch")
971
+ require(record.get("dtype") == "float32", f"evaluation {key} dtype mismatch")
972
+ require(
973
+ record.get("shape") == [D_MODEL, D_MODEL],
974
+ f"evaluation {key} shape mismatch",
975
+ )
976
+ require(
977
+ record.get("numel") == D_MODEL * D_MODEL,
978
+ f"evaluation {key} element count mismatch",
979
+ )
980
+ require(
981
+ record.get("nbytes") == EVALUATION_TENSOR_NBYTES,
982
+ f"evaluation {key} byte count mismatch",
983
+ )
984
+ require(
985
+ isinstance(record.get("sha256_c_contiguous_little_endian_bytes"), str)
986
+ and SHA256_PATTERN.fullmatch(
987
+ record["sha256_c_contiguous_little_endian_bytes"]
988
+ ),
989
+ f"evaluation {key} tensor hash format mismatch",
990
+ )
991
+
992
+ require(
993
+ provenance.get("artifact_kind") == "jacobian_lens_evaluation_fp32_provenance",
994
+ "evaluation provenance kind mismatch",
995
+ )
996
+ require(
997
+ provenance.get("artifact")
998
+ == {
999
+ "filename": "evaluation.safetensors",
1000
+ "format": "safetensors",
1001
+ "sha256": EVALUATION_ARTIFACT_SHA256,
1002
+ "size_bytes": EVALUATION_ARTIFACT_SIZE_BYTES,
1003
+ "tensor_conversion": "lossless_fp32_reserialization",
1004
+ },
1005
+ "evaluation provenance artifact binding mismatch",
1006
+ )
1007
+ require(
1008
+ provenance.get("source_checkpoint")
1009
+ == {"format": "pytorch", "sha256": EVALUATION_LENS_SHA256},
1010
+ "evaluation provenance source binding mismatch",
1011
+ )
1012
+ derivation = provenance.get("derivation", {})
1013
+ require(
1014
+ derivation.get("formula") == "jacobian_sum[layer] / n_done",
1015
+ "evaluation derivation formula mismatch",
1016
+ )
1017
+ require(derivation.get("n_done") == 1000, "evaluation derivation count mismatch")
1018
+ require(
1019
+ derivation.get("source_fit_checkpoint_sha256")
1020
+ == EXPECTED_EVALUATION_METADATA["source_fit_checkpoint_sha256"],
1021
+ "evaluation fit checkpoint binding mismatch",
1022
+ )
1023
+ require(
1024
+ provenance.get("model", {}).get("id") == MODEL_ID
1025
+ and provenance.get("model", {}).get("revision") == MODEL_REVISION,
1026
+ "evaluation provenance model binding mismatch",
1027
+ )
1028
+ require(
1029
+ provenance.get("lens", {}).get("dtype") == "float32"
1030
+ and provenance.get("lens", {}).get("source_layers") == list(SOURCE_LAYERS)
1031
+ and provenance.get("lens", {}).get("target_layer") == 35
1032
+ and provenance.get("lens", {}).get("n_prompts") == 1000,
1033
+ "evaluation provenance lens metadata mismatch",
1034
+ )
1035
+ runtime = provenance.get("software", {}).get("conversion_runtime", {})
1036
+ require(
1037
+ runtime.get("safetensors") == "0.8.0" and runtime.get("torch") == "2.13.0",
1038
+ "evaluation conversion runtime mismatch",
1039
+ )
1040
+
1041
+ require(
1042
+ frozen_validation.get("artifact") == "evaluation.safetensors",
1043
+ "evaluation validation filename mismatch",
1044
+ )
1045
+ require(
1046
+ frozen_validation.get("artifact_sha256") == EVALUATION_ARTIFACT_SHA256,
1047
+ "evaluation validation artifact hash mismatch",
1048
+ )
1049
+ require(
1050
+ frozen_validation.get("artifact_size_bytes") == EVALUATION_ARTIFACT_SIZE_BYTES,
1051
+ "evaluation validation artifact size mismatch",
1052
+ )
1053
+ require(
1054
+ frozen_validation.get("source_checkpoint_sha256") == EVALUATION_LENS_SHA256,
1055
+ "evaluation validation source hash mismatch",
1056
+ )
1057
+ require(frozen_validation.get("ok") is True, "evaluation validation failed")
1058
+ require(
1059
+ frozen_validation.get("tensor_values_changed") == 0,
1060
+ "evaluation validation records changed values",
1061
+ )
1062
+ require(
1063
+ frozen_validation.get("exact_tensor_matches") == len(SOURCE_LAYERS),
1064
+ "evaluation validation tensor count mismatch",
1065
+ )
1066
+ checks = frozen_validation.get("checks")
1067
+ require(
1068
+ isinstance(checks, dict)
1069
+ and checks
1070
+ and all(value is True for value in checks.values()),
1071
+ "evaluation validation contains a failed check",
1072
+ )
1073
+
1074
+ require(
1075
+ compatibility.get("artifact_kind") == "fp32_to_fp16_lens_compatibility",
1076
+ "evaluation compatibility kind mismatch",
1077
+ )
1078
+ require(
1079
+ compatibility.get("fp32_artifact") == "evaluation.safetensors"
1080
+ and compatibility.get("fp32_artifact_sha256") == EVALUATION_ARTIFACT_SHA256
1081
+ and compatibility.get("fp32_source_checkpoint_sha256")
1082
+ == EVALUATION_LENS_SHA256,
1083
+ "evaluation compatibility FP32 binding mismatch",
1084
+ )
1085
+ require(
1086
+ compatibility.get("fp16_artifact") == "model.safetensors"
1087
+ and compatibility.get("fp16_artifact_sha256") == ARTIFACT_SHA256,
1088
+ "evaluation compatibility FP16 binding mismatch",
1089
+ )
1090
+ require(
1091
+ compatibility.get("all_layer_casts_match_exactly") is True,
1092
+ "evaluation compatibility records a cast mismatch",
1093
+ )
1094
+ require(
1095
+ compatibility.get("max_absolute_error") == 0.00048828125,
1096
+ "evaluation compatibility maximum error mismatch",
1097
+ )
1098
+ require(
1099
+ math.isclose(
1100
+ compatibility.get("relative_frobenius_error", math.inf),
1101
+ 0.00020901276774552773,
1102
+ rel_tol=1e-12,
1103
+ abs_tol=0.0,
1104
+ ),
1105
+ "evaluation compatibility relative error mismatch",
1106
+ )
1107
+ per_layer = compatibility.get("per_layer")
1108
+ require(
1109
+ isinstance(per_layer, dict)
1110
+ and set(per_layer) == {str(layer) for layer in SOURCE_LAYERS},
1111
+ "evaluation compatibility layer set mismatch",
1112
+ )
1113
+ require(
1114
+ all(
1115
+ record.get("cast_matches_model_safetensors_exactly") is True
1116
+ for record in per_layer.values()
1117
+ ),
1118
+ "evaluation compatibility layer cast mismatch",
1119
+ )
1120
+
1121
+ require(
1122
+ sha256_file(artifact_path) == EVALUATION_ARTIFACT_SHA256,
1123
+ "evaluation artifact SHA-256 mismatch",
1124
+ )
1125
+ require(
1126
+ artifact_path.stat().st_size == EVALUATION_ARTIFACT_SIZE_BYTES,
1127
+ "evaluation artifact size mismatch",
1128
+ )
1129
+ checked = 0
1130
+ with (
1131
+ safe_open(artifact_path, framework="pt", device="cpu") as artifact,
1132
+ safe_open(
1133
+ root / "model.safetensors", framework="pt", device="cpu"
1134
+ ) as fp16_artifact,
1135
+ ):
1136
+ require(
1137
+ (artifact.metadata() or {}) == EXPECTED_EVALUATION_METADATA,
1138
+ "evaluation Safetensors metadata mismatch",
1139
+ )
1140
+ require(set(artifact.keys()) == EXPECTED_KEYS, "evaluation tensor key mismatch")
1141
+ for layer in SOURCE_LAYERS:
1142
+ key = f"J.{layer}"
1143
+ tensor = artifact.get_tensor(key)
1144
+ record = tensors[key]
1145
+ require(tensor.dtype == torch.float32, f"evaluation {key} dtype mismatch")
1146
+ require(
1147
+ tuple(tensor.shape) == (D_MODEL, D_MODEL),
1148
+ f"evaluation {key} shape mismatch",
1149
+ )
1150
+ require(tensor.is_contiguous(), f"evaluation {key} is not contiguous")
1151
+ require(
1152
+ bool(torch.isfinite(tensor).all()),
1153
+ f"evaluation {key} contains non-finite values",
1154
+ )
1155
+ require(
1156
+ tensor_sha256(tensor)
1157
+ == record["sha256_c_contiguous_little_endian_bytes"],
1158
+ f"evaluation {key} tensor hash mismatch",
1159
+ )
1160
+ require(
1161
+ torch.equal(
1162
+ tensor.to(torch.float16).view(torch.int16),
1163
+ fp16_artifact.get_tensor(key).view(torch.int16),
1164
+ ),
1165
+ f"evaluation {key} FP16 cast mismatch",
1166
+ )
1167
+ checked += 1
1168
+ return checked
1169
+
1170
+
1171
+ def run_git(root: Path, *arguments: str) -> subprocess.CompletedProcess[bytes]:
1172
+ return subprocess.run(
1173
+ ["git", "-C", str(root), *arguments],
1174
+ check=False,
1175
+ capture_output=True,
1176
+ )
1177
+
1178
+
1179
+ def validate_git_lfs(
1180
+ root: Path, *, require_evaluation_head: bool = False
1181
+ ) -> tuple[bool, bool]:
1182
+ repository_check = run_git(root, "rev-parse", "--is-inside-work-tree")
1183
+ if repository_check.returncode != 0:
1184
+ return False, False
1185
+
1186
+ model_pointer = (
1187
+ "version https://git-lfs.github.com/spec/v1\n"
1188
+ f"oid sha256:{ARTIFACT_SHA256}\n"
1189
+ f"size {ARTIFACT_SIZE_BYTES}\n"
1190
+ ).encode("ascii")
1191
+ for object_name in ("HEAD:model.safetensors", ":model.safetensors"):
1192
+ pointer = run_git(root, "show", object_name)
1193
+ require(pointer.returncode == 0, f"cannot read Git LFS pointer: {object_name}")
1194
+ require(
1195
+ pointer.stdout == model_pointer, f"Git LFS pointer mismatch: {object_name}"
1196
+ )
1197
+
1198
+ evaluation_pointer = (
1199
+ "version https://git-lfs.github.com/spec/v1\n"
1200
+ f"oid sha256:{EVALUATION_ARTIFACT_SHA256}\n"
1201
+ f"size {EVALUATION_ARTIFACT_SIZE_BYTES}\n"
1202
+ ).encode("ascii")
1203
+ staged_evaluation = run_git(root, "show", ":evaluation.safetensors")
1204
+ require(
1205
+ staged_evaluation.returncode == 0,
1206
+ "cannot read staged Git LFS pointer: evaluation.safetensors",
1207
+ )
1208
+ require(
1209
+ staged_evaluation.stdout == evaluation_pointer,
1210
+ "staged Git LFS pointer mismatch: evaluation.safetensors",
1211
+ )
1212
+ head_evaluation = run_git(root, "show", "HEAD:evaluation.safetensors")
1213
+ if require_evaluation_head:
1214
+ require(
1215
+ head_evaluation.returncode == 0,
1216
+ "publication check requires evaluation.safetensors in HEAD",
1217
+ )
1218
+ if head_evaluation.returncode == 0:
1219
+ require(
1220
+ head_evaluation.stdout == evaluation_pointer,
1221
+ "Git LFS pointer mismatch: HEAD:evaluation.safetensors",
1222
+ )
1223
+
1224
+ fsck = run_git(root, "lfs", "fsck", "HEAD")
1225
+ require(fsck.returncode == 0, "git lfs fsck failed")
1226
+ return True, True
1227
+
1228
+
1229
+ def validate(root: Path, *, require_publication_fields: bool = False) -> dict[str, Any]:
1230
+ artifact_path = root / "model.safetensors"
1231
+ readme = (root / "README.md").read_text(encoding="utf-8")
1232
+ config = read_json(root / "lens_config.json")
1233
+ manifest = read_json(root / "tensor_manifest.json")
1234
+ provenance = read_json(root / "provenance.json")
1235
+ frozen_validation = read_json(root / "validation.json")
1236
+ evaluation_path = root / "evaluation.json"
1237
+ evaluation = read_json(evaluation_path)
1238
+ evaluation_manifest = read_json(root / "evaluation_tensor_manifest.json")
1239
+ evaluation_provenance = read_json(root / "evaluation_provenance.json")
1240
+ evaluation_validation = read_json(root / "evaluation_validation.json")
1241
+ evaluation_compatibility = read_json(root / "evaluation_compatibility.json")
1242
+
1243
+ validate_release_file_set(root)
1244
+ validate_pinned_public_files(root)
1245
+ validate_public_text(root)
1246
+ require(
1247
+ (root / "requirements.txt").read_text(encoding="utf-8")
1248
+ == "safetensors==0.8.0\ntorch==2.13.0\n",
1249
+ "requirements do not match the evaluation conversion runtime",
1250
+ )
1251
+ validate_model_card(readme)
1252
+ validate_licenses_and_notices(root)
1253
+ validate_tensor_manifest(manifest)
1254
+ evaluation_tensors_checked = validate_evaluation_artifact(
1255
+ root,
1256
+ evaluation_manifest,
1257
+ evaluation_provenance,
1258
+ evaluation_validation,
1259
+ evaluation_compatibility,
1260
+ )
1261
+ validate_metadata_records(
1262
+ config,
1263
+ provenance,
1264
+ frozen_validation,
1265
+ evaluation,
1266
+ evaluation_path,
1267
+ )
1268
+
1269
+ publication_values, unresolved_publication_fields = parse_publication_rows(readme)
1270
+ if require_publication_fields:
1271
+ require(
1272
+ not unresolved_publication_fields,
1273
+ "publication fields remain unresolved: "
1274
+ + ", ".join(unresolved_publication_fields),
1275
+ )
1276
+ if not unresolved_publication_fields:
1277
+ validate_publication_values(publication_values)
1278
+
1279
+ expected_checksum_file = (
1280
+ f"{ARTIFACT_SHA256} model.safetensors\n"
1281
+ f"{EVALUATION_ARTIFACT_SHA256} evaluation.safetensors\n"
1282
+ )
1283
+ require(
1284
+ (root / "SHA256SUMS").read_text(encoding="ascii") == expected_checksum_file,
1285
+ "SHA256SUMS content mismatch",
1286
+ )
1287
+ artifact_checksum = sha256_file(artifact_path)
1288
+ require(artifact_checksum == ARTIFACT_SHA256, "artifact SHA-256 mismatch")
1289
+ require(
1290
+ artifact_path.stat().st_size == ARTIFACT_SIZE_BYTES, "artifact size mismatch"
1291
+ )
1292
+
1293
+ checked = 0
1294
+ with safe_open(artifact_path, framework="pt", device="cpu") as artifact:
1295
+ require(
1296
+ (artifact.metadata() or {}) == EXPECTED_METADATA,
1297
+ "safetensors metadata mismatch",
1298
+ )
1299
+ require(
1300
+ set(artifact.keys()) == EXPECTED_KEYS, "safetensors tensor key mismatch"
1301
+ )
1302
+ for layer in SOURCE_LAYERS:
1303
+ key = f"J.{layer}"
1304
+ tensor = artifact.get_tensor(key)
1305
+ record = manifest["tensors"][key]
1306
+ require(tensor.dtype == torch.float16, f"{key} dtype mismatch")
1307
+ require(tuple(tensor.shape) == (D_MODEL, D_MODEL), f"{key} shape mismatch")
1308
+ require(tensor.is_contiguous(), f"{key} is not contiguous")
1309
+ require(
1310
+ bool(torch.isfinite(tensor).all()), f"{key} contains non-finite values"
1311
+ )
1312
+ require(tensor.numel() == record["numel"], f"{key} element count mismatch")
1313
+ require(
1314
+ tensor.numel() * tensor.element_size() == record["nbytes"],
1315
+ f"{key} byte count mismatch",
1316
+ )
1317
+ require(
1318
+ tensor_sha256(tensor)
1319
+ == record["sha256_c_contiguous_little_endian_bytes"],
1320
+ f"{key} tensor hash mismatch",
1321
+ )
1322
+ checked += 1
1323
+
1324
+ lfs_pointer_exact, lfs_fsck_ok = validate_git_lfs(
1325
+ root,
1326
+ require_evaluation_head=require_publication_fields,
1327
+ )
1328
+ if require_publication_fields:
1329
+ require(
1330
+ lfs_pointer_exact and lfs_fsck_ok,
1331
+ "publication check requires Git LFS validation",
1332
+ )
1333
+
1334
+ return {
1335
+ "artifact": artifact_path.name,
1336
+ "artifact_kind": EXPECTED_METADATA["artifact_kind"],
1337
+ "artifact_sha256": artifact_checksum,
1338
+ "artifact_size_bytes": artifact_path.stat().st_size,
1339
+ "evaluation_artifact_separation": True,
1340
+ "evaluation_artifact_sha256": EVALUATION_ARTIFACT_SHA256,
1341
+ "evaluation_artifact_size_bytes": EVALUATION_ARTIFACT_SIZE_BYTES,
1342
+ "evaluation_lens_sha256": EVALUATION_LENS_SHA256,
1343
+ "evaluation_tensor_count": evaluation_tensors_checked,
1344
+ "fp32_to_fp16_cast_exact": True,
1345
+ "license_notice_hashes_exact": True,
1346
+ "lfs_fsck_ok": lfs_fsck_ok,
1347
+ "lfs_pointer_exact": lfs_pointer_exact,
1348
+ "metadata_exact": True,
1349
+ "model_binding_exact": True,
1350
+ "ok": (
1351
+ checked == len(EXPECTED_KEYS)
1352
+ and evaluation_tensors_checked == len(EXPECTED_KEYS)
1353
+ ),
1354
+ "privacy_text_scan": True,
1355
+ "publication_fields_resolved": not unresolved_publication_fields,
1356
+ "release_file_set_exact": True,
1357
+ "tensor_count": checked,
1358
+ "tensor_descriptors_exact": True,
1359
+ "tensor_hashes_exact": True,
1360
+ "unresolved_publication_fields": unresolved_publication_fields,
1361
+ }
1362
+
1363
+
1364
+ def main() -> None:
1365
+ parser = argparse.ArgumentParser(description=__doc__)
1366
+ parser.add_argument(
1367
+ "--root",
1368
+ type=Path,
1369
+ default=Path(__file__).resolve().parents[1],
1370
+ help="Artifact repository root",
1371
+ )
1372
+ parser.add_argument(
1373
+ "--publication",
1374
+ action="store_true",
1375
+ help="Require resolved and valid companion publication fields",
1376
+ )
1377
+ args = parser.parse_args()
1378
+ try:
1379
+ result = validate(
1380
+ args.root.resolve(),
1381
+ require_publication_fields=args.publication,
1382
+ )
1383
+ except ValueError as error:
1384
+ parser.error(str(error))
1385
+ print(
1386
+ json.dumps(
1387
+ result,
1388
+ indent=2,
1389
+ sort_keys=True,
1390
+ )
1391
+ )
1392
+
1393
+
1394
+ if __name__ == "__main__":
1395
+ main()
tensor_manifest.json ADDED
@@ -0,0 +1,208 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "artifact": "model.safetensors",
3
+ "artifact_sha256": "089d776979408f23e5377539c15aa8025d171633718ccdce709bcd3372e7942c",
4
+ "artifact_size_bytes": 150996824,
5
+ "schema_version": 1,
6
+ "tensor_count": 18,
7
+ "tensor_storage_bytes": 150994944,
8
+ "tensors": {
9
+ "J.0": {
10
+ "dtype": "float16",
11
+ "nbytes": 8388608,
12
+ "numel": 4194304,
13
+ "sha256_c_contiguous_little_endian_bytes": "70b0dec033aa1707f047f17b9149ed137431e86b17958bd4a7e48d50fdda3d51",
14
+ "shape": [
15
+ 2048,
16
+ 2048
17
+ ],
18
+ "source_layer": 0
19
+ },
20
+ "J.10": {
21
+ "dtype": "float16",
22
+ "nbytes": 8388608,
23
+ "numel": 4194304,
24
+ "sha256_c_contiguous_little_endian_bytes": "30da4128931002f756acb8f26c7a807b0a03aae37ebd5d1fce859ca8872203fe",
25
+ "shape": [
26
+ 2048,
27
+ 2048
28
+ ],
29
+ "source_layer": 10
30
+ },
31
+ "J.12": {
32
+ "dtype": "float16",
33
+ "nbytes": 8388608,
34
+ "numel": 4194304,
35
+ "sha256_c_contiguous_little_endian_bytes": "1585c024e196065636ace0c6693304b5d238589b778d8469975cb735cac4a58e",
36
+ "shape": [
37
+ 2048,
38
+ 2048
39
+ ],
40
+ "source_layer": 12
41
+ },
42
+ "J.14": {
43
+ "dtype": "float16",
44
+ "nbytes": 8388608,
45
+ "numel": 4194304,
46
+ "sha256_c_contiguous_little_endian_bytes": "25474372408e43f00aa30c8787e0953b56674d84a3d4e110619c8122de6939fc",
47
+ "shape": [
48
+ 2048,
49
+ 2048
50
+ ],
51
+ "source_layer": 14
52
+ },
53
+ "J.16": {
54
+ "dtype": "float16",
55
+ "nbytes": 8388608,
56
+ "numel": 4194304,
57
+ "sha256_c_contiguous_little_endian_bytes": "f8451021c23b81883bed655568aaf5fb91fcbf0ef5abe5f92400df8012f759c9",
58
+ "shape": [
59
+ 2048,
60
+ 2048
61
+ ],
62
+ "source_layer": 16
63
+ },
64
+ "J.18": {
65
+ "dtype": "float16",
66
+ "nbytes": 8388608,
67
+ "numel": 4194304,
68
+ "sha256_c_contiguous_little_endian_bytes": "8af4ab11ea1c9443b0e6ba6abf0af5073032b8c4eced8365b6a3c5ea10ec32ba",
69
+ "shape": [
70
+ 2048,
71
+ 2048
72
+ ],
73
+ "source_layer": 18
74
+ },
75
+ "J.2": {
76
+ "dtype": "float16",
77
+ "nbytes": 8388608,
78
+ "numel": 4194304,
79
+ "sha256_c_contiguous_little_endian_bytes": "8a8bc773b72f5bb7e697700ba780ba105fb3efa5c966ff90121f12f97e4eb601",
80
+ "shape": [
81
+ 2048,
82
+ 2048
83
+ ],
84
+ "source_layer": 2
85
+ },
86
+ "J.20": {
87
+ "dtype": "float16",
88
+ "nbytes": 8388608,
89
+ "numel": 4194304,
90
+ "sha256_c_contiguous_little_endian_bytes": "714121ea97b8bb8ba6d72b9656dfbb33e777f2345f0c990494c8960cfa493c77",
91
+ "shape": [
92
+ 2048,
93
+ 2048
94
+ ],
95
+ "source_layer": 20
96
+ },
97
+ "J.22": {
98
+ "dtype": "float16",
99
+ "nbytes": 8388608,
100
+ "numel": 4194304,
101
+ "sha256_c_contiguous_little_endian_bytes": "ea197cfe6095e67364ef2d88a44fb07c80b7731c78d1137c7372771902ddf784",
102
+ "shape": [
103
+ 2048,
104
+ 2048
105
+ ],
106
+ "source_layer": 22
107
+ },
108
+ "J.24": {
109
+ "dtype": "float16",
110
+ "nbytes": 8388608,
111
+ "numel": 4194304,
112
+ "sha256_c_contiguous_little_endian_bytes": "2f0e30882300a6914fd9bc089b1b038beda30e4ed1eca0da0051a532eb367c70",
113
+ "shape": [
114
+ 2048,
115
+ 2048
116
+ ],
117
+ "source_layer": 24
118
+ },
119
+ "J.26": {
120
+ "dtype": "float16",
121
+ "nbytes": 8388608,
122
+ "numel": 4194304,
123
+ "sha256_c_contiguous_little_endian_bytes": "e5b3789b2802efb00d9e2e3e2753059e16db80ebef26c9e363e8a43f1a08f3da",
124
+ "shape": [
125
+ 2048,
126
+ 2048
127
+ ],
128
+ "source_layer": 26
129
+ },
130
+ "J.28": {
131
+ "dtype": "float16",
132
+ "nbytes": 8388608,
133
+ "numel": 4194304,
134
+ "sha256_c_contiguous_little_endian_bytes": "de9fd5a978c691d30c46289fa179332cee671f69aba8c4442d7f61382cd56e40",
135
+ "shape": [
136
+ 2048,
137
+ 2048
138
+ ],
139
+ "source_layer": 28
140
+ },
141
+ "J.30": {
142
+ "dtype": "float16",
143
+ "nbytes": 8388608,
144
+ "numel": 4194304,
145
+ "sha256_c_contiguous_little_endian_bytes": "143d3abe3632641ab2bb6b2c0582ef9d461e0b144b1a29f1f67e716645ea5f4c",
146
+ "shape": [
147
+ 2048,
148
+ 2048
149
+ ],
150
+ "source_layer": 30
151
+ },
152
+ "J.32": {
153
+ "dtype": "float16",
154
+ "nbytes": 8388608,
155
+ "numel": 4194304,
156
+ "sha256_c_contiguous_little_endian_bytes": "bbb1664e19ef450f68a8202c7c6007633e160c96712866c3793cb80474f589dc",
157
+ "shape": [
158
+ 2048,
159
+ 2048
160
+ ],
161
+ "source_layer": 32
162
+ },
163
+ "J.34": {
164
+ "dtype": "float16",
165
+ "nbytes": 8388608,
166
+ "numel": 4194304,
167
+ "sha256_c_contiguous_little_endian_bytes": "a2745b1142ae10a8b7fd449eba62e5530216c4026c748aed78eda69b7aeb35f9",
168
+ "shape": [
169
+ 2048,
170
+ 2048
171
+ ],
172
+ "source_layer": 34
173
+ },
174
+ "J.4": {
175
+ "dtype": "float16",
176
+ "nbytes": 8388608,
177
+ "numel": 4194304,
178
+ "sha256_c_contiguous_little_endian_bytes": "c4970a9604879056b2f4493b1e8b1b0401ae115c236613b37875edf991425cdd",
179
+ "shape": [
180
+ 2048,
181
+ 2048
182
+ ],
183
+ "source_layer": 4
184
+ },
185
+ "J.6": {
186
+ "dtype": "float16",
187
+ "nbytes": 8388608,
188
+ "numel": 4194304,
189
+ "sha256_c_contiguous_little_endian_bytes": "57e34398437bd673ecfe10b8c696306a706287d18307edceb4b69bf1fced1905",
190
+ "shape": [
191
+ 2048,
192
+ 2048
193
+ ],
194
+ "source_layer": 6
195
+ },
196
+ "J.8": {
197
+ "dtype": "float16",
198
+ "nbytes": 8388608,
199
+ "numel": 4194304,
200
+ "sha256_c_contiguous_little_endian_bytes": "dfecda781a5e26a3573a7c7f8ff8dc173bc4b55c6e6f1a80b3c7a918d08eece6",
201
+ "shape": [
202
+ 2048,
203
+ 2048
204
+ ],
205
+ "source_layer": 8
206
+ }
207
+ }
208
+ }
validation.json ADDED
@@ -0,0 +1,28 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "artifact": "model.safetensors",
3
+ "artifact_sha256": "089d776979408f23e5377539c15aa8025d171633718ccdce709bcd3372e7942c",
4
+ "artifact_size_bytes": 150996824,
5
+ "checks": {
6
+ "all_source_tensors_contiguous": true,
7
+ "all_source_tensors_finite": true,
8
+ "all_source_tensors_fp16": true,
9
+ "all_source_tensors_shape_2048x2048": true,
10
+ "roundtrip_all_tensor_bit_patterns_equal": true,
11
+ "roundtrip_all_tensor_byte_hashes_equal": true,
12
+ "roundtrip_all_tensor_dtypes_equal": true,
13
+ "roundtrip_all_tensor_shapes_equal": true,
14
+ "roundtrip_all_tensor_values_equal": true,
15
+ "roundtrip_key_set_exact": true,
16
+ "safetensors_header_public_safe": true,
17
+ "safetensors_header_roundtrip_exact": true,
18
+ "source_checkpoint_sha256_exact": true,
19
+ "source_metadata_exact": true,
20
+ "source_top_level_key_set_exact": true
21
+ },
22
+ "exact_tensor_matches": 18,
23
+ "expected_tensor_matches": 18,
24
+ "ok": true,
25
+ "schema_version": 1,
26
+ "source_checkpoint_sha256": "f36a99447623e0d777c70951a9148a7a52e42e0df82942e22ce6326f63d8d664",
27
+ "tensor_values_changed": 0
28
+ }