Kev-4B Core AI: fp16 hidden-state decoder, Metal-kernel graph, 128 tokens per call (K128), with the pointer head and tokenizer
Browse files- .gitattributes +2 -0
- LICENSE +202 -0
- README.md +373 -0
- SHA256SUMS +12 -0
- adapter_config.json +56 -0
- config.json +104 -0
- gpu-pipelined/kev_4b_decode_fp16_metal_pf128/head/head.safetensors +3 -0
- gpu-pipelined/kev_4b_decode_fp16_metal_pf128/head/kev_head.json +67 -0
- gpu-pipelined/kev_4b_decode_fp16_metal_pf128/kev_4b_decode_fp16_metal_pf128.aimodel/main.hash +1 -0
- gpu-pipelined/kev_4b_decode_fp16_metal_pf128/kev_4b_decode_fp16_metal_pf128.aimodel/main.mlirb +3 -0
- gpu-pipelined/kev_4b_decode_fp16_metal_pf128/kev_4b_decode_fp16_metal_pf128.aimodel/metadata.json +5 -0
- gpu-pipelined/kev_4b_decode_fp16_metal_pf128/metadata.json +276 -0
- gpu-pipelined/kev_4b_decode_fp16_metal_pf128/tokenizer/tokenizer.json +3 -0
- gpu-pipelined/kev_4b_decode_fp16_metal_pf128/tokenizer/tokenizer_config.json +305 -0
.gitattributes
CHANGED
|
@@ -33,3 +33,5 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
|
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
+
gpu-pipelined/kev_4b_decode_fp16_metal_pf128/kev_4b_decode_fp16_metal_pf128.aimodel/main.mlirb filter=lfs diff=lfs merge=lfs -text
|
| 37 |
+
gpu-pipelined/kev_4b_decode_fp16_metal_pf128/tokenizer/tokenizer.json filter=lfs diff=lfs merge=lfs -text
|
LICENSE
ADDED
|
@@ -0,0 +1,202 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
|
| 2 |
+
Apache License
|
| 3 |
+
Version 2.0, January 2004
|
| 4 |
+
http://www.apache.org/licenses/
|
| 5 |
+
|
| 6 |
+
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
| 7 |
+
|
| 8 |
+
1. Definitions.
|
| 9 |
+
|
| 10 |
+
"License" shall mean the terms and conditions for use, reproduction,
|
| 11 |
+
and distribution as defined by Sections 1 through 9 of this document.
|
| 12 |
+
|
| 13 |
+
"Licensor" shall mean the copyright owner or entity authorized by
|
| 14 |
+
the copyright owner that is granting the License.
|
| 15 |
+
|
| 16 |
+
"Legal Entity" shall mean the union of the acting entity and all
|
| 17 |
+
other entities that control, are controlled by, or are under common
|
| 18 |
+
control with that entity. For the purposes of this definition,
|
| 19 |
+
"control" means (i) the power, direct or indirect, to cause the
|
| 20 |
+
direction or management of such entity, whether by contract or
|
| 21 |
+
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
| 22 |
+
outstanding shares, or (iii) beneficial ownership of such entity.
|
| 23 |
+
|
| 24 |
+
"You" (or "Your") shall mean an individual or Legal Entity
|
| 25 |
+
exercising permissions granted by this License.
|
| 26 |
+
|
| 27 |
+
"Source" form shall mean the preferred form for making modifications,
|
| 28 |
+
including but not limited to software source code, documentation
|
| 29 |
+
source, and configuration files.
|
| 30 |
+
|
| 31 |
+
"Object" form shall mean any form resulting from mechanical
|
| 32 |
+
transformation or translation of a Source form, including but
|
| 33 |
+
not limited to compiled object code, generated documentation,
|
| 34 |
+
and conversions to other media types.
|
| 35 |
+
|
| 36 |
+
"Work" shall mean the work of authorship, whether in Source or
|
| 37 |
+
Object form, made available under the License, as indicated by a
|
| 38 |
+
copyright notice that is included in or attached to the work
|
| 39 |
+
(an example is provided in the Appendix below).
|
| 40 |
+
|
| 41 |
+
"Derivative Works" shall mean any work, whether in Source or Object
|
| 42 |
+
form, that is based on (or derived from) the Work and for which the
|
| 43 |
+
editorial revisions, annotations, elaborations, or other modifications
|
| 44 |
+
represent, as a whole, an original work of authorship. For the purposes
|
| 45 |
+
of this License, Derivative Works shall not include works that remain
|
| 46 |
+
separable from, or merely link (or bind by name) to the interfaces of,
|
| 47 |
+
the Work and Derivative Works thereof.
|
| 48 |
+
|
| 49 |
+
"Contribution" shall mean any work of authorship, including
|
| 50 |
+
the original version of the Work and any modifications or additions
|
| 51 |
+
to that Work or Derivative Works thereof, that is intentionally
|
| 52 |
+
submitted to Licensor for inclusion in the Work by the copyright owner
|
| 53 |
+
or by an individual or Legal Entity authorized to submit on behalf of
|
| 54 |
+
the copyright owner. For the purposes of this definition, "submitted"
|
| 55 |
+
means any form of electronic, verbal, or written communication sent
|
| 56 |
+
to the Licensor or its representatives, including but not limited to
|
| 57 |
+
communication on electronic mailing lists, source code control systems,
|
| 58 |
+
and issue tracking systems that are managed by, or on behalf of, the
|
| 59 |
+
Licensor for the purpose of discussing and improving the Work, but
|
| 60 |
+
excluding communication that is conspicuously marked or otherwise
|
| 61 |
+
designated in writing by the copyright owner as "Not a Contribution."
|
| 62 |
+
|
| 63 |
+
"Contributor" shall mean Licensor and any individual or Legal Entity
|
| 64 |
+
on behalf of whom a Contribution has been received by Licensor and
|
| 65 |
+
subsequently incorporated within the Work.
|
| 66 |
+
|
| 67 |
+
2. Grant of Copyright License. Subject to the terms and conditions of
|
| 68 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 69 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 70 |
+
copyright license to reproduce, prepare Derivative Works of,
|
| 71 |
+
publicly display, publicly perform, sublicense, and distribute the
|
| 72 |
+
Work and such Derivative Works in Source or Object form.
|
| 73 |
+
|
| 74 |
+
3. Grant of Patent License. Subject to the terms and conditions of
|
| 75 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 76 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 77 |
+
(except as stated in this section) patent license to make, have made,
|
| 78 |
+
use, offer to sell, sell, import, and otherwise transfer the Work,
|
| 79 |
+
where such license applies only to those patent claims licensable
|
| 80 |
+
by such Contributor that are necessarily infringed by their
|
| 81 |
+
Contribution(s) alone or by combination of their Contribution(s)
|
| 82 |
+
with the Work to which such Contribution(s) was submitted. If You
|
| 83 |
+
institute patent litigation against any entity (including a
|
| 84 |
+
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
| 85 |
+
or a Contribution incorporated within the Work constitutes direct
|
| 86 |
+
or contributory patent infringement, then any patent licenses
|
| 87 |
+
granted to You under this License for that Work shall terminate
|
| 88 |
+
as of the date such litigation is filed.
|
| 89 |
+
|
| 90 |
+
4. Redistribution. You may reproduce and distribute copies of the
|
| 91 |
+
Work or Derivative Works thereof in any medium, with or without
|
| 92 |
+
modifications, and in Source or Object form, provided that You
|
| 93 |
+
meet the following conditions:
|
| 94 |
+
|
| 95 |
+
(a) You must give any other recipients of the Work or
|
| 96 |
+
Derivative Works a copy of this License; and
|
| 97 |
+
|
| 98 |
+
(b) You must cause any modified files to carry prominent notices
|
| 99 |
+
stating that You changed the files; and
|
| 100 |
+
|
| 101 |
+
(c) You must retain, in the Source form of any Derivative Works
|
| 102 |
+
that You distribute, all copyright, patent, trademark, and
|
| 103 |
+
attribution notices from the Source form of the Work,
|
| 104 |
+
excluding those notices that do not pertain to any part of
|
| 105 |
+
the Derivative Works; and
|
| 106 |
+
|
| 107 |
+
(d) If the Work includes a "NOTICE" text file as part of its
|
| 108 |
+
distribution, then any Derivative Works that You distribute must
|
| 109 |
+
include a readable copy of the attribution notices contained
|
| 110 |
+
within such NOTICE file, excluding those notices that do not
|
| 111 |
+
pertain to any part of the Derivative Works, in at least one
|
| 112 |
+
of the following places: within a NOTICE text file distributed
|
| 113 |
+
as part of the Derivative Works; within the Source form or
|
| 114 |
+
documentation, if provided along with the Derivative Works; or,
|
| 115 |
+
within a display generated by the Derivative Works, if and
|
| 116 |
+
wherever such third-party notices normally appear. The contents
|
| 117 |
+
of the NOTICE file are for informational purposes only and
|
| 118 |
+
do not modify the License. You may add Your own attribution
|
| 119 |
+
notices within Derivative Works that You distribute, alongside
|
| 120 |
+
or as an addendum to the NOTICE text from the Work, provided
|
| 121 |
+
that such additional attribution notices cannot be construed
|
| 122 |
+
as modifying the License.
|
| 123 |
+
|
| 124 |
+
You may add Your own copyright statement to Your modifications and
|
| 125 |
+
may provide additional or different license terms and conditions
|
| 126 |
+
for use, reproduction, or distribution of Your modifications, or
|
| 127 |
+
for any such Derivative Works as a whole, provided Your use,
|
| 128 |
+
reproduction, and distribution of the Work otherwise complies with
|
| 129 |
+
the conditions stated in this License.
|
| 130 |
+
|
| 131 |
+
5. Submission of Contributions. Unless You explicitly state otherwise,
|
| 132 |
+
any Contribution intentionally submitted for inclusion in the Work
|
| 133 |
+
by You to the Licensor shall be under the terms and conditions of
|
| 134 |
+
this License, without any additional terms or conditions.
|
| 135 |
+
Notwithstanding the above, nothing herein shall supersede or modify
|
| 136 |
+
the terms of any separate license agreement you may have executed
|
| 137 |
+
with Licensor regarding such Contributions.
|
| 138 |
+
|
| 139 |
+
6. Trademarks. This License does not grant permission to use the trade
|
| 140 |
+
names, trademarks, service marks, or product names of the Licensor,
|
| 141 |
+
except as required for reasonable and customary use in describing the
|
| 142 |
+
origin of the Work and reproducing the content of the NOTICE file.
|
| 143 |
+
|
| 144 |
+
7. Disclaimer of Warranty. Unless required by applicable law or
|
| 145 |
+
agreed to in writing, Licensor provides the Work (and each
|
| 146 |
+
Contributor provides its Contributions) on an "AS IS" BASIS,
|
| 147 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
| 148 |
+
implied, including, without limitation, any warranties or conditions
|
| 149 |
+
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
| 150 |
+
PARTICULAR PURPOSE. You are solely responsible for determining the
|
| 151 |
+
appropriateness of using or redistributing the Work and assume any
|
| 152 |
+
risks associated with Your exercise of permissions under this License.
|
| 153 |
+
|
| 154 |
+
8. Limitation of Liability. In no event and under no legal theory,
|
| 155 |
+
whether in tort (including negligence), contract, or otherwise,
|
| 156 |
+
unless required by applicable law (such as deliberate and grossly
|
| 157 |
+
negligent acts) or agreed to in writing, shall any Contributor be
|
| 158 |
+
liable to You for damages, including any direct, indirect, special,
|
| 159 |
+
incidental, or consequential damages of any character arising as a
|
| 160 |
+
result of this License or out of the use or inability to use the
|
| 161 |
+
Work (including but not limited to damages for loss of goodwill,
|
| 162 |
+
work stoppage, computer failure or malfunction, or any and all
|
| 163 |
+
other commercial damages or losses), even if such Contributor
|
| 164 |
+
has been advised of the possibility of such damages.
|
| 165 |
+
|
| 166 |
+
9. Accepting Warranty or Additional Liability. While redistributing
|
| 167 |
+
the Work or Derivative Works thereof, You may choose to offer,
|
| 168 |
+
and charge a fee for, acceptance of support, warranty, indemnity,
|
| 169 |
+
or other liability obligations and/or rights consistent with this
|
| 170 |
+
License. However, in accepting such obligations, You may act only
|
| 171 |
+
on Your own behalf and on Your sole responsibility, not on behalf
|
| 172 |
+
of any other Contributor, and only if You agree to indemnify,
|
| 173 |
+
defend, and hold each Contributor harmless for any liability
|
| 174 |
+
incurred by, or claims asserted against, such Contributor by reason
|
| 175 |
+
of your accepting any such warranty or additional liability.
|
| 176 |
+
|
| 177 |
+
END OF TERMS AND CONDITIONS
|
| 178 |
+
|
| 179 |
+
APPENDIX: How to apply the Apache License to your work.
|
| 180 |
+
|
| 181 |
+
To apply the Apache License to your work, attach the following
|
| 182 |
+
boilerplate notice, with the fields enclosed by brackets "[]"
|
| 183 |
+
replaced with your own identifying information. (Don't include
|
| 184 |
+
the brackets!) The text should be enclosed in the appropriate
|
| 185 |
+
comment syntax for the file format. We also recommend that a
|
| 186 |
+
file or class name and description of purpose be included on the
|
| 187 |
+
same "printed page" as the copyright notice for easier
|
| 188 |
+
identification within third-party archives.
|
| 189 |
+
|
| 190 |
+
Copyright 2026 Alibaba Cloud
|
| 191 |
+
|
| 192 |
+
Licensed under the Apache License, Version 2.0 (the "License");
|
| 193 |
+
you may not use this file except in compliance with the License.
|
| 194 |
+
You may obtain a copy of the License at
|
| 195 |
+
|
| 196 |
+
http://www.apache.org/licenses/LICENSE-2.0
|
| 197 |
+
|
| 198 |
+
Unless required by applicable law or agreed to in writing, software
|
| 199 |
+
distributed under the License is distributed on an "AS IS" BASIS,
|
| 200 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
| 201 |
+
See the License for the specific language governing permissions and
|
| 202 |
+
limitations under the License.
|
README.md
ADDED
|
@@ -0,0 +1,373 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
license: apache-2.0
|
| 3 |
+
base_model: jaredpalmer/kev-4b
|
| 4 |
+
base_model_relation: quantized
|
| 5 |
+
pipeline_tag: text-classification
|
| 6 |
+
library_name: coreai
|
| 7 |
+
language:
|
| 8 |
+
- en
|
| 9 |
+
tags:
|
| 10 |
+
- coreai
|
| 11 |
+
- core-ai
|
| 12 |
+
- apple
|
| 13 |
+
- on-device
|
| 14 |
+
- decision-model
|
| 15 |
+
- systemone
|
| 16 |
+
- typed-decisions
|
| 17 |
+
- structured-output
|
| 18 |
+
---
|
| 19 |
+
|
| 20 |
+
# Kev-4B — Core AI
|
| 21 |
+
|
| 22 |
+
Apple Core AI (`.aimodel`) conversion of [jaredpalmer/kev-4b](https://huggingface.co/jaredpalmer/kev-4b/tree/6cfce5c2fa4b4bd64026336ab649c5ca78857d52) (tag `v1.0`, Apache-2.0): a state and typed questions in, the probability of every option of every question out.
|
| 23 |
+
|
| 24 |
+
| Variant | Path | Size | Requires | Tested on |
|
| 25 |
+
|---|---|---:|---|---|
|
| 26 |
+
| Decoder, fp16, with the host's pointer head | `gpu-pipelined/kev_4b_decode_fp16_metal_pf128/` | 8,431 MB | macOS 27 | M4 Max, macOS 27.0 (26A428), 2026-10-04 |
|
| 27 |
+
|
| 28 |
+
One question of 94 tokens takes 100.2 ms on the M4 Max, in Swift ([measured below](#time-per-decision-on-the-mac)).
|
| 29 |
+
|
| 30 |
+
```swift
|
| 31 |
+
import Kev
|
| 32 |
+
|
| 33 |
+
let bundle = URL(filePath: "Kev-4B-CoreAI/gpu-pipelined/kev_4b_decode_fp16_metal_pf128")
|
| 34 |
+
let kev = try await KevDecider(bundle: bundle) // asset: nil = the .aimodel, specialized here (GPU, frequent reshapes)
|
| 35 |
+
let response = try await kev.decide(requestJSON: requestData, shared: true) // shared: the state's whole calls run once
|
| 36 |
+
print(PythonFormat.dumps(response, asciiOnly: false))
|
| 37 |
+
```
|
| 38 |
+
|
| 39 |
+
The snippet uses the [`Kev`](https://github.com/john-rocky/coreai-model-zoo/tree/main/apps/Kev) Swift package from the model zoo. The zoo card and the gate scripts live in [coreai-model-zoo](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-4b/README.md); the rest of this page is the same text with repository links.
|
| 40 |
+
|
| 41 |
+
Source [jaredpalmer/kev-4b](https://huggingface.co/jaredpalmer/kev-4b/tree/6cfce5c2fa4b4bd64026336ab649c5ca78857d52) (tag `v1.0`, commit `6cfce5c`) · base [Qwen/Qwen3.5-4B-Base](https://huggingface.co/Qwen/Qwen3.5-4B-Base/tree/1001bb4d826a52d1f399e183466143f4da7b741b) (revision `1001bb4`)
|
| 42 |
+
|
| 43 |
+
A **decision model**. Give it a state (text, or a JSON object or array) and typed questions: `noul`
|
| 44 |
+
(yes / no), `choice` (named options) or `score` (ordered levels). It returns a probability for every
|
| 45 |
+
option of every question. It never generates text. Requests and responses use the SystemOne-compatible
|
| 46 |
+
request shape: `{model, state, questions}` in, `{model, answers, usage}` out, one typed decision per
|
| 47 |
+
question.
|
| 48 |
+
|
| 49 |
+
Jared Palmer trained it as a rank-16 LoRA adapter and a pointer head on Qwen3.5-4B-Base (32 layers:
|
| 50 |
+
24 Gated DeltaNet and 8 full attention, hidden size 2,560). The author's card: "It is intended for
|
| 51 |
+
developers who classify, route, triage or check documents and who need probabilities that can be
|
| 52 |
+
thresholded, for example to send uncertain cases to human review." Each question is its own row: the
|
| 53 |
+
state, then the question and its options. The head scores each option's closing token against the
|
| 54 |
+
question's last token. The author's card reports benchmark results; none of them are re-measured here.
|
| 55 |
+
|
| 56 |
+
This port merges the adapter into the base with the author's own merge script and exports the text
|
| 57 |
+
backbone as one Core AI graph. The graph takes 128 token ids per call and returns the final-norm hidden
|
| 58 |
+
state at every position; it has no vocabulary head. Each Gated DeltaNet layer runs its recurrence in an
|
| 59 |
+
fp32 Metal kernel, one GPU dispatch per layer per call. Weights are fp16 (8.41 GB). The host runs the
|
| 60 |
+
pointer head from the author's head weights. The gate is probability parity with the author's own fp32
|
| 61 |
+
code, on every option of every question, on a fixture of 434 questions and on a held-out set of 130.
|
| 62 |
+
|
| 63 |
+
It runs on the Mac GPU. This release is for the Mac: the one load of an iPhone AOT asset of an earlier
|
| 64 |
+
graph on an iPhone 18 Pro crashed (below). [Kev-0.8B](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-0.8b/README.md), the same port at 0.8B, is
|
| 65 |
+
measured on the iPhone 18 Pro.
|
| 66 |
+
|
| 67 |
+
## Readout contract
|
| 68 |
+
|
| 69 |
+
The contract is Kev-0.8B's, with this checkpoint's head and temperature; the full description is in the
|
| 70 |
+
[Kev-0.8B card](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-0.8b/README.md#readout-contract). In short:
|
| 71 |
+
|
| 72 |
+
```
|
| 73 |
+
[<|fim_prefix|>] + render(state) + [<|fim_middle|>] + render(instructions)
|
| 74 |
+
+ for each option: [<|box_start|>] + option_text + [<|box_end|>]
|
| 75 |
+
+ [<|fim_suffix|>]
|
| 76 |
+
```
|
| 77 |
+
|
| 78 |
+
- One row per question, zeroed states, positions 0..L − 1. Delimiters from the base tokenizer: 248060,
|
| 79 |
+
248061, 248049 / 248050, 248062; pad = eos = 248044; no bos. The 4B base's `tokenizer.json` is
|
| 80 |
+
byte-identical to the 0.8B base's; its `tokenizer_config.json` differs in the chat template's thinking
|
| 81 |
+
default, which this readout does not use. The `kev-4b` repository still carries Qwen3-era
|
| 82 |
+
`vocab.json`, `merges.txt` and `added_tokens.json` (other delimiter ids); the author's loader reads the
|
| 83 |
+
tokenizer from the base, and so does this port.
|
| 84 |
+
- The head: `z_k = ((W_k h_opt_k + b_k) · (W_q h_decide + b_q)) / 16`, `p = softmax(z / T)` within the
|
| 85 |
+
question, T = 2.406050072164233 (`head.pt`); `W_q`, `W_k` are `[256, 2560]` with biases, fp32.
|
| 86 |
+
- A row holds at most **3,968 tokens**. The last call is padded to 128 tokens and must end at or below
|
| 87 |
+
the graph's position bound of 4,095.
|
| 88 |
+
- Shared prefix (optional, exact): the state's whole 128-token calls run once and the four states are
|
| 89 |
+
copied per question. Every row's hidden state equals its direct run bit for bit (below).
|
| 90 |
+
|
| 91 |
+
The fixture and the held-out set are Kev-0.8B's (434 and 130 questions), with this checkpoint's own fp32
|
| 92 |
+
oracle run (the author's package, one CPU thread): [`fixtures-kev-4b.json`](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-4b/fixtures-kev-4b.json). It has
|
| 93 |
+
twelve near-ties on the fixture and four on the held-out set.
|
| 94 |
+
|
| 95 |
+
## Core AI shape
|
| 96 |
+
|
| 97 |
+
The same module as Kev-0.8B, `conversion/kev/qwen3_5_kev_decoder.py`, with no code change: the overlay's
|
| 98 |
+
stateful Qwen3.5 text decoder on the merged weights, no vocabulary head. One function, `main`, at a
|
| 99 |
+
static 128 tokens. Each Gated DeltaNet layer runs its recurrence in the overlay's fp32 Metal chunk kernel
|
| 100 |
+
(`qwen3_5_gdn_metal`): one GPU dispatch per layer per call, the recurrent state kept in fp32 through the
|
| 101 |
+
call.
|
| 102 |
+
|
| 103 |
+
| | name | shape, type |
|
| 104 |
+
|---|---|---|
|
| 105 |
+
| inputs | `input_ids` | [1, 128] int32 |
|
| 106 |
+
| | `position_ids` | [1, seq] int32, the ramp 0 .. seq − 1 |
|
| 107 |
+
| states | `keyCache`, `valueCache` | [8, 1, 4, ctx, 256] fp16, ctx up to 4,096 |
|
| 108 |
+
| | `convState`, `recState` | [24, 1, 8192, 3], [24, 1, 32, 128, 128] fp16 |
|
| 109 |
+
| output | `hidden` | [1, 128, 2560] fp16, every position |
|
| 110 |
+
|
| 111 |
+
Per row of T ids, from zeroed states: ⌈T / 128⌉ calls, the last padded with `<|endoftext|>` (248044) and
|
| 112 |
+
its padded rows dropped. `metadata.json` says `kind: decision-backbone` and `language.prefill_chunk: 128`,
|
| 113 |
+
and carries the readout contract. The host (`conversion/kev/decide.py`, [`apps/Kev`](https://github.com/john-rocky/coreai-model-zoo/tree/main/apps/Kev)) is
|
| 114 |
+
Kev-0.8B's: the head in float64 from `head/head.safetensors`, p rounded to fp32 once, the
|
| 115 |
+
SystemOne-compatible response, the shared prefix and the prepared state. No engine, no runtime patch.
|
| 116 |
+
|
| 117 |
+
## Measured (Apple M4 Max, macOS 27.0 26A428, 2026-10-03 – 10-04)
|
| 118 |
+
|
| 119 |
+
The bar is Kev-0.8B's, fixed before any graph ran: the argmax equal to the oracle's on every question
|
| 120 |
+
whose oracle top-2 margin is above 0.02 (near-ties listed apart), max |Δp| ≤ 0.02 over every option of
|
| 121 |
+
every question, the mean over rows of each row's mean |Δp| ≤ 0.002, every process's re-run bit-equal.
|
| 122 |
+
The Python gates load the AOT `.aimodelc` (h16c, `--expect-frequent-reshapes`) with
|
| 123 |
+
`SpecializationOptions.default()`; their times are on a GPU shared with other work.
|
| 124 |
+
|
| 125 |
+
### Before any graph: the merge and fp32 torch
|
| 126 |
+
|
| 127 |
+
- The merged checkpoint (`scripts/merge_lora_checkpoint.py`, tag `kev-1.0`; fp32, four shards of
|
| 128 |
+
16.8 GB, 426 tensors, 248 of them adapted) answers through the author's loader as the adapter
|
| 129 |
+
checkpoint does: 434/434 questions, logits bit-equal. `W + (B @ A) × 2` reproduces three merged
|
| 130 |
+
tensors bit for bit (the adapter moves them by 1.4–2.0 %)
|
| 131 |
+
([`gate-kev-4b-merge.json`](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-4b/gate-kev-4b-merge.json)).
|
| 132 |
+
- The decoder module in fp32 on the CPU, driven like the graph with the recurrence unrolled, read through
|
| 133 |
+
the author's head: 434/434 argmax (the 12 near-ties included), max |Δp| 6.6e-6, lowest per-position
|
| 134 |
+
hidden cosine 0.9999999965 on 21 rows; the module's loader reads all 426 keys and leaves none; it equals
|
| 135 |
+
the overlay's plain text decoder bit for bit (3 rows, one of 1,518 tokens)
|
| 136 |
+
([`gate-kev-4b-torch-parity.json`](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-4b/gate-kev-4b-torch-parity.json)).
|
| 137 |
+
- The Metal kernel has no torch values: its `torch_defn` returns zeros of the right shapes. The GPU gate
|
| 138 |
+
below is its parity check.
|
| 139 |
+
|
| 140 |
+
### The graph alone on the Mac GPU
|
| 141 |
+
|
| 142 |
+
| set | questions | argmax (margin > 0.02) | near-ties agreeing | max \|Δp\| | mean of row means | bar |
|
| 143 |
+
|---|---:|---:|---:|---:|---:|---|
|
| 144 |
+
| **fixture** | 434 | **422/422** | **10/12** | **0.0153** | **0.00077** | **PASS** |
|
| 145 |
+
| held out | 130 | 126/126 | 3/4 | 0.0096 | 0.00069 | PASS |
|
| 146 |
+
|
| 147 |
+
The near-ties that flip have oracle margins of 0.0005–0.0024 and move by at most 0.0029. The worst
|
| 148 |
+
fixture row is a SemIf item (0.0153), the worst held-out row a `score` item (0.0096). The lowest
|
| 149 |
+
per-position hidden cosine on the 21 recorded rows is 0.9921. Every value is finite and every process's
|
| 150 |
+
re-run is bit-equal. The red arms move p: a swapped state moves 5 of 5 argmaxes (max |Δp| 0.988), a
|
| 151 |
+
grammatical "not" 3 of 5 (0.957). Transcript: [`gate-kev-4b-readout.json`](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-4b/gate-kev-4b-readout.json).
|
| 152 |
+
|
| 153 |
+
### From the request: Python and Swift
|
| 154 |
+
|
| 155 |
+
- `conversion/kev/host.py`, without the author's package, rebuilds every oracle row from the raw request
|
| 156 |
+
(ids, `<|fim_suffix|>` / `<|box_end|>` indices, keys): 434/434 and 130/130 rows with two tokenizer
|
| 157 |
+
implementations. `decide.py` on the AOT graph: hidden rows bit-equal to the gate's on 434/434 and
|
| 158 |
+
130/130 rows, and the shared prefix equal to the direct run on every row
|
| 159 |
+
([`gate-kev-4b-host.json`](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-4b/gate-kev-4b-host.json)).
|
| 160 |
+
- The Swift host ([`apps/Kev`](https://github.com/john-rocky/coreai-model-zoo/tree/main/apps/Kev), Release): the ids from the raw request equal the
|
| 161 |
+
oracle's on 434/434 and 130/130 rows. On the same AOT asset every row's hidden state equals the Python
|
| 162 |
+
gate's bit for bit (434/434 and 130/130), and its float64 head gives p within 4.5e-7 of the gate's fp32
|
| 163 |
+
head. The bar is the gate's (fixture 0.0153 / 0.00077, held out 0.0096 / 0.00069). With the shared
|
| 164 |
+
prefix every row's hidden state and p equal the direct run's (434/434 and 130/130).
|
| 165 |
+
- **JIT = AOT.** The `.aimodel` specialized by the Swift runtime (GPU preferred,
|
| 166 |
+
`expectFrequentReshapes`) gives the AOT asset's hidden state and p bit for bit on 25 rows of 10 records.
|
| 167 |
+
The specialization took 16.6 s in `AIModel(contentsOf:)` (18.4 s with the tokenizer and the head)
|
| 168 |
+
and left a 15.55 GB entry in the runtime's cache; a later load took 1.74 s
|
| 169 |
+
([`gate-kev-4b-swift.json`](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-4b/gate-kev-4b-swift.json)).
|
| 170 |
+
|
| 171 |
+
### Time per decision on the Mac
|
| 172 |
+
|
| 173 |
+
Swift, Release CLI, the AOT asset, in one machine-wide GPU lock window (2026-10-04 14:05–14:30 JST). Two
|
| 174 |
+
processes per graph counted only when no other GPU job ran and the one-minute load average was at most 12;
|
| 175 |
+
each made 6 decisions per item after one warm-up, and the table gives the median (p10–p90) over the 12.
|
| 176 |
+
The earlier unrolled graph (16 tokens per call) ran in the same window:
|
| 177 |
+
|
| 178 |
+
| request | tokens | calls | ms | p10–p90 | unrolled S = 16, ms |
|
| 179 |
+
|---|---:|---:|---:|---|---:|
|
| 180 |
+
| one question | 94 | 1 | 100.2 | 99.3–101.2 | 269.5 |
|
| 181 |
+
| one question | 380 | 3 | 298.0 | 296.1–304.5 | 1,075.4 |
|
| 182 |
+
| one question | 1,518 | 12 | 1,189.0 | 1,182.9–1,194.9 | 4,341.9 |
|
| 183 |
+
| one question | 1,802 | 15 | 1,488.1 | 1,484.3–1,498.5 | 5,200.2 |
|
| 184 |
+
| five questions on one 137-token state, each row from zero | 240 | 10 | 986.7 | 983.0–990.2 | 2,367.9 |
|
| 185 |
+
| the same five questions, the state run once (shared) | 240 | 6 | 615.6 | 613.3–624.5 | 894.7 |
|
| 186 |
+
| eight questions on that state, each row from zero | 320 | 16 | 1,578.7 | 1,574.2–1,591.6 | 3,776.0 |
|
| 187 |
+
| the same eight questions, shared | 320 | 9 | 920.0 | 916.9–930.6 | 1,262.4 |
|
| 188 |
+
| four questions on one 1,477-token state, each row from zero | 1,622 | 49 | 4,851.7 | 4,839.2–4,870.6 | 17,087.7 |
|
| 189 |
+
| the same four questions, shared | 1,622 | 16 | 1,614.5 | 1,607.5–1,647.8 | 4,713.7 |
|
| 190 |
+
|
| 191 |
+
Loading the AOT asset took 9.90 / 9.91 s at the start of each process and 1.30 / 1.22 s when loaded
|
| 192 |
+
again. The two processes ended at 0.89 and 0.95 GB of footprint. Transcript:
|
| 193 |
+
[`gate-kev-4b-timing-mac.json`](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-4b/gate-kev-4b-timing-mac.json).
|
| 194 |
+
|
| 195 |
+
### iPhone 18 Pro: not in this release
|
| 196 |
+
|
| 197 |
+
An earlier graph of this port (the recurrence unrolled, 16 tokens per call) compiled for the iPhone 18
|
| 198 |
+
Pro's GPU (h19p, `--expect-frequent-reshapes`) is 15,557,759,365 bytes. Loaded once in the headless gate
|
| 199 |
+
app (iOS 27.0 24A437, `.default` options, no increased-memory-limit entitlement), the app died inside
|
| 200 |
+
`AIModel(contentsOf:)` with `EXC_BAD_ACCESS (SIGSEGV)` in the on-device compile for delegates
|
| 201 |
+
(`-[MPSGraphAICodeCompilerDelegate getInitializedAICodeBytecodeWithPayloadPrefix:delegateId:]`). Its last
|
| 202 |
+
memory sample, 0.32 s in, read a footprint of 175.9 MB with 3,364.1 MB available. The cause is not
|
| 203 |
+
isolated and the load was not retried; the graph of this release was not loaded on the phone
|
| 204 |
+
([`../kev-0.8b/gate-kev-0.8b-iphone.json`](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-0.8b/gate-kev-0.8b-iphone.json)
|
| 205 |
+
`round8.device.summary.load_4b`, `round8.load_4b_memory_samples`).
|
| 206 |
+
|
| 207 |
+
## Forms measured
|
| 208 |
+
|
| 209 |
+
Every form below was exported and compiled the same way and read against the same oracle: the full gate,
|
| 210 |
+
or the 130-row subset where the table says so. The times come from different windows, so compare forms only
|
| 211 |
+
within one window. Full records: [`gate-kev-4b-forms.json`](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-4b/gate-kev-4b-forms.json).
|
| 212 |
+
|
| 213 |
+
Mac (Swift Release CLI, the 94-token question, median of the counted processes):
|
| 214 |
+
|
| 215 |
+
| form | 94-token decision, ms | window | fixture max \|Δp\| | note |
|
| 216 |
+
|---|---:|---|---:|---|
|
| 217 |
+
| Gated DeltaNet unrolled, S = 16 | 269.5 | round 12 | 0.0154 | |
|
| 218 |
+
| unrolled, S = 32 / 64 | — | — | 0.0078 / 0.0070 (130 rows) | not timed in a lock window (104.8 / 174.1 ms per call on a shared GPU) |
|
| 219 |
+
| Metal kernel, S = 32 | — | — | 0.0074 (130 rows) | not timed in a lock window (48.4 ms per call on a shared GPU) |
|
| 220 |
+
| Metal kernel, S = 64 | 118.0 | round 12 | 0.0170 | five questions shared: 435.9 ms (S = 128: 615.6) |
|
| 221 |
+
| **Metal kernel, S = 128 (this release)** | **100.2** | round 12 | **0.0153** | |
|
| 222 |
+
| Metal kernel, query length 2..512, host calls of at most 512 ids | 83.3 | round 12 | 0.0132 | not shipped: memory grows (below) |
|
| 223 |
+
| in-graph chunk scan | — | — | — | not run at 4B; on Kev-0.8B it fails from S = 32 |
|
| 224 |
+
| int8, every linear (unrolled S = 16) | — | — | 0.0601 | fails the bar (Precision) |
|
| 225 |
+
| any form on the Neural Engine | — | — | — | not run: the recurrence's fp32 state is not an ANE element type ([`qwen3.5-static-ane.md`](https://github.com/john-rocky/coreai-model-zoo/blob/main/knowledge/qwen3.5-static-ane.md)) |
|
| 226 |
+
|
| 227 |
+
**S = 64 for several short questions on one state.** In the same window, five questions on one
|
| 228 |
+
137-token state, shared, took 435.9 ms with S = 64 and 615.6 ms with S = 128; one 94-token question took
|
| 229 |
+
118.0 and 100.2 ms. This release ships S = 128.
|
| 230 |
+
|
| 231 |
+
**Why the dynamic-length graph does not ship.** It passed the gate and decided the 94-token question in
|
| 232 |
+
83.3 ms in the same window. Its two timing processes ended at 6.31 and 6.32 GB of footprint, against 0.89
|
| 233 |
+
and 0.95 GB for S = 128. On Kev-0.8B the same form keeps growing while the call length keeps changing,
|
| 234 |
+
until the `AIModel` is created again ([Kev-0.8B card](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-0.8b/README.md#forms-measured)).
|
| 235 |
+
|
| 236 |
+
## Precision
|
| 237 |
+
|
| 238 |
+
**int8 was measured and is not shipped.** It was measured on the unrolled S = 16 graph; the Metal-kernel
|
| 239 |
+
graphs were not measured in int8. int8 per block of 32 over every decoder linear fails the bar:
|
| 240 |
+
|
| 241 |
+
| decoder (unrolled S = 16) | fp16 kept | `main.mlirb` bytes | fixture max \|Δp\| / mean | held out max \|Δp\| / mean | bar |
|
| 242 |
+
|---|---:|---:|---|---|---|
|
| 243 |
+
| fp16 | 100 % | 8,414,007,636 | 0.0154 / 0.00078 | 0.0109 / 0.00072 | PASS |
|
| 244 |
+
| int8lin: every linear int8 (`symmetric_with_clipping`) | 0 % | 5,068,103,201 | 0.0601 / 0.00192 | 0.0285 / 0.00169 | FAIL |
|
| 245 |
+
| asymmetric int8, `in_proj_qkv` + `out_proj` + `v_proj` fp16 | 21.7 % | 5,882,832,315 | 0.0175 / 0.00135 | 0.0277 / 0.00123 | FAIL |
|
| 246 |
+
| asymmetric int8, `in_proj_qkv` + `out_proj` fp16 | 21.2 % | 5,863,831,482 | 0.0174 / 0.00138 | not run | — |
|
| 247 |
+
|
| 248 |
+
The two asymmetric sets are the ones an fp32 torch bisect chose under a rule written before any result
|
| 249 |
+
(at most 25 % of the linear weights fp16, the target worst row ≤ 0.006 met by none; the two best by mean
|
| 250 |
+
taken). The one with `v_proj` fp16 passes the fixture and fails the held-out set on one `score` row, the row that also
|
| 251 |
+
breaks int8lin and is fp16's held-out worst. A set built before the rule's last stage (layers 0–3 and
|
| 252 |
+
`out_proj` fp16) fails both (0.0272, 0.0293). The quantizer's `qscheme` accepts `symmetric`,
|
| 253 |
+
`asymmetric` and `symmetric_with_clipping`; block 16 and an int8 embedding table also compile
|
| 254 |
+
([`gate-kev-4b-int8.json`](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-4b/gate-kev-4b-int8.json)).
|
| 255 |
+
|
| 256 |
+
**What the compiled asset holds.** `--expect-frequent-reshapes` adds an fp16 copy of every linear weight:
|
| 257 |
+
for the unrolled graph the h19p asset of the fp16 bundle is 15,557,759,365 bytes with it and 8,413,701,860
|
| 258 |
+
without, and the int8 sets above compile to 13,026,670,589 and 13,007,670,689 bytes with it. This
|
| 259 |
+
release's Mac h16c asset is 15,552,786,962 bytes against an 8,412,500,461-byte `main.mlirb`. Without the
|
| 260 |
+
flag the runtime specializes the unrolled graph again for every new position length: 9.6–14.6 s per new
|
| 261 |
+
length on the Mac GPU (calls at a length already seen: 46.4 ms median, 33 rows); an int8 asset without the
|
| 262 |
+
flag had not finished specializing its opening length after 12 minutes. A Mac gate worker on the efr fp16
|
| 263 |
+
asset of the unrolled graph showed up to 16.6 GB rss, up to 16.4 GB of it clean mapped file, and a
|
| 264 |
+
phys_footprint of at most 1.15 GB.
|
| 265 |
+
|
| 266 |
+
## ⬇️ Bundle
|
| 267 |
+
|
| 268 |
+
[mlboydaisuke/Kev-4B-CoreAI](https://huggingface.co/mlboydaisuke/Kev-4B-CoreAI), one folder under
|
| 269 |
+
`gpu-pipelined/`:
|
| 270 |
+
|
| 271 |
+
| file | what | bytes | sha256 |
|
| 272 |
+
|---|---|---:|---|
|
| 273 |
+
| `kev_4b_decode_fp16_metal_pf128.aimodel/main.mlirb` | the decoder, fp16 | 8,412,500,461 | `da529dd4…413eb17f` |
|
| 274 |
+
| `metadata.json` | `kind: decision-backbone`, the readout contract, the gate | 12,643 | `e641e44f…c9039b08` |
|
| 275 |
+
| `tokenizer/tokenizer.json` | the base model's, verbatim | 12,807,196 | `fe000e3e…d50d2927` |
|
| 276 |
+
| `tokenizer/tokenizer_config.json` | the base model's, verbatim | 16,713 | `3891e840…b6b9f89c` |
|
| 277 |
+
| `head/head.safetensors` | the pointer head: q / k weight `[256, 2560]` and bias, fp32 | 5,245,304 | `36c392f7…54157106` |
|
| 278 |
+
| `head/kev_head.json` | head size, scale, temperature, delimiter ids, provenance | 2,038 | `5ff071ff…cfba7416` |
|
| 279 |
+
|
| 280 |
+
The repository root carries the base model's `config.json`, the adapter's `adapter_config.json` and the
|
| 281 |
+
Apache-2.0 `LICENSE`, verbatim. `SHA256SUMS` lists every file. The Hub revision is recorded here after the
|
| 282 |
+
upload.
|
| 283 |
+
|
| 284 |
+
No AOT asset ships: the Swift runtime's specialization of the `.aimodel` equals the AOT asset bit for bit
|
| 285 |
+
(above). To compile one for the Mac anyway (15.55 GB, the flag is required):
|
| 286 |
+
|
| 287 |
+
```bash
|
| 288 |
+
xcrun coreai-build compile kev_4b_decode_fp16_metal_pf128.aimodel --output aot --preferred-compute gpu \
|
| 289 |
+
--platform macOS --architecture h16c --expect-frequent-reshapes
|
| 290 |
+
```
|
| 291 |
+
|
| 292 |
+
## Use it
|
| 293 |
+
|
| 294 |
+
Swift, with the [`Kev`](https://github.com/john-rocky/coreai-model-zoo/tree/main/apps/Kev) package (macOS 27; the system CoreAI framework, Accelerate and
|
| 295 |
+
swift-transformers' tokenizer), on a download of the repository:
|
| 296 |
+
|
| 297 |
+
```swift
|
| 298 |
+
import Kev
|
| 299 |
+
|
| 300 |
+
let bundle = URL(filePath: "Kev-4B-CoreAI/gpu-pipelined/kev_4b_decode_fp16_metal_pf128")
|
| 301 |
+
let kev = try await KevDecider(bundle: bundle) // the .aimodel, specialized here (GPU, frequent reshapes)
|
| 302 |
+
let response = try await kev.decide(requestJSON: requestData, shared: true) // shared: the state's whole calls run once
|
| 303 |
+
|
| 304 |
+
// questions that arrive later, on the same state
|
| 305 |
+
let prepared = try await kev.prepare(state: try JSONParser.parse(stateData)) // the state's whole calls, once
|
| 306 |
+
let later = try await kev.decide(prepared: prepared, questionsJSON: questionsData) // only the questions' rows run
|
| 307 |
+
```
|
| 308 |
+
|
| 309 |
+
Loading the bundle specializes the graph once (16.6 s on the M4 Max) and caches it; later loads read the cache.
|
| 310 |
+
The Mac CLI: `kev run --bundle Kev-4B-CoreAI/gpu-pipelined/kev_4b_decode_fp16_metal_pf128 --asset jit
|
| 311 |
+
--request req.json --shared --out resp.json` (`swift build -c release --package-path apps/Kev`). The
|
| 312 |
+
request shape and the Python read-out (`conversion/kev/decide.py run --model kev-4b …`, AOT assets only)
|
| 313 |
+
are as in the [Kev-0.8B card](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-0.8b/README.md#use-it).
|
| 314 |
+
|
| 315 |
+
## Reproduce
|
| 316 |
+
|
| 317 |
+
Environment: the zoo overlay venv (coreai-core 1.0.0b2, coreai-torch 0.4.1, torch 2.9.0, transformers
|
| 318 |
+
4.57.6), Xcode 27.0 RC; the author's code (the oracle and the merge) in its own venv from the author's
|
| 319 |
+
`uv.lock` at tag `kev-1.0`. Order and flags: [`conversion/kev/README.md`](https://github.com/john-rocky/coreai-model-zoo/blob/main/conversion/kev/README.md).
|
| 320 |
+
|
| 321 |
+
```bash
|
| 322 |
+
(cd $ZOO_WORK_ROOT/_kev/kev-src && python scripts/merge_lora_checkpoint.py \
|
| 323 |
+
--lora jaredpalmer/kev-4b@591dcb5bd6d05eb0b5131ea6608f93f10243335c --out $ZOO_WORK_ROOT/_kev/merged/kev-4b-v1.0)
|
| 324 |
+
# the decoder (--aot adds the h16c .aimodelc the Python gates load)
|
| 325 |
+
python conversion/kev/export_decoder.py fp16 --model kev-4b --gdn-scan metal --prefill-chunk 128 --aot
|
| 326 |
+
# the published bundle: the gated .aimodel bytes and their metadata, without exporting again
|
| 327 |
+
python conversion/kev/export_decoder.py fp16 --model kev-4b --gdn-scan metal --prefill-chunk 128 --metadata-only \
|
| 328 |
+
--from-aimodel <bundles>/kev_4b_decode_fp16_metal_pf128/kev_4b_decode_fp16_metal_pf128.aimodel \
|
| 329 |
+
--bundle-dir <ship>/kev_4b_decode_fp16_metal_pf128
|
| 330 |
+
```
|
| 331 |
+
|
| 332 |
+
Recipe: [`recipe.toml`](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-4b/recipe.toml). Port notes: [`knowledge/kev-port.md`](https://github.com/john-rocky/coreai-model-zoo/blob/main/knowledge/kev-port.md).
|
| 333 |
+
|
| 334 |
+
## Other formats
|
| 335 |
+
|
| 336 |
+
On the Hub (2026-10-04), not run here:
|
| 337 |
+
|
| 338 |
+
- ONNX: [`midudev/kev-4b-ONNX`](https://huggingface.co/midudev/kev-4b-ONNX) (ONNX Runtime Web, int4
|
| 339 |
+
weights, the pointer head as `head.bin`);
|
| 340 |
+
[`onnx-community/kev-4b-ONNX`](https://huggingface.co/onnx-community/kev-4b-ONNX) (Transformers.js,
|
| 341 |
+
`q4` / `q4f16`, the pointer head in the graph; its card names Qwen/Qwen3-4B-Base as the base, the
|
| 342 |
+
Qwen3 generation of Kev-4B before Kev 1.0).
|
| 343 |
+
- GGUF: [`ggml-org/Kev-4B-GGUF`](https://huggingface.co/ggml-org/Kev-4B-GGUF) ("a decision model, to be
|
| 344 |
+
used via `/v1/systemone` API", per its card).
|
| 345 |
+
- MLX: [`RoderickQiu/kev-4b-mlx-8bit`](https://huggingface.co/RoderickQiu/kev-4b-mlx-8bit) (merged, 8-bit,
|
| 346 |
+
`head.pt` fp32) and [`aselea/Kev-4B-MLX-Serve-8bit`](https://huggingface.co/aselea/Kev-4B-MLX-Serve-8bit)
|
| 347 |
+
(for mlx-serve, the head as `kev_head.safetensors`); the author's repository also serves Kev through
|
| 348 |
+
MLX on Apple silicon.
|
| 349 |
+
|
| 350 |
+
No other Core AI conversion of Kev was listed on the Hub on 2026-10-04.
|
| 351 |
+
|
| 352 |
+
## License
|
| 353 |
+
|
| 354 |
+
Apache-2.0: the adapter and the head (jaredpalmer/kev-4b) and the base (Qwen/Qwen3.5-4B-Base); the bundle
|
| 355 |
+
inherits it, and the Hugging Face repository carries the base's `LICENSE`. The author's package runs only
|
| 356 |
+
in the oracle and the merge, at gate time. The fixture file's terms are Kev-0.8B's: SemIf authored144
|
| 357 |
+
under MIT with its notice, transfer-v4 records by reference, 11 of the 20 records written for the port
|
| 358 |
+
published and 9 withheld (an invented name in each was found in use on the web).
|
| 359 |
+
|
| 360 |
+
## Limits
|
| 361 |
+
|
| 362 |
+
From the [author's card](https://huggingface.co/jaredpalmer/kev-4b): English only; text generation, chat
|
| 363 |
+
and fully automated consequential decisions about people are out of scope; questions that need facts
|
| 364 |
+
that are neither in the state nor general knowledge are out of scope; the temperature was fitted on the
|
| 365 |
+
author's development rows (the card explains how to measure and refit it). This port adds:
|
| 366 |
+
|
| 367 |
+
- A row holds at most 3,968 tokens, while the author's server accepts states up to 65,536.
|
| 368 |
+
- The release is for the Mac.
|
| 369 |
+
- A process's opening call is slower. On the Mac the opening 128-token call of each Python gate process
|
| 370 |
+
took 1,121–9,881 ms, against a median of 99.0 ms for its later calls.
|
| 371 |
+
- If you export this decoder with a dynamic query length yourself, keep the call lengths to one or two
|
| 372 |
+
values. A process that cycles more lengths grows its memory until the `AIModel` is created again
|
| 373 |
+
(measured on Kev-0.8B on the Mac and the iPhone; which runtime layer keeps the memory is not isolated).
|
SHA256SUMS
ADDED
|
@@ -0,0 +1,12 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
50cbab8a892c5f2993b8c7351a99182507472def3b1374558308605d99b86b32 LICENSE
|
| 2 |
+
49f9ef81dab003ae1275678400962136811b0de05d2132fb9c8527a22d89c1fb README.md
|
| 3 |
+
8a05dfd6c5e7f61a62e6db5d8de8093a8dbbd105aa7ee7b5e7b73f6249f45617 adapter_config.json
|
| 4 |
+
ddc63e1c717afa86c865bb5e01313d89d72bb53b97ad4a8a03ba8510c0621670 config.json
|
| 5 |
+
36c392f75fa5eff2243b07664e1edb1dc4e771dfc9d145469ba175db54157106 gpu-pipelined/kev_4b_decode_fp16_metal_pf128/head/head.safetensors
|
| 6 |
+
5ff071ffaa5b620999f44770a014b4b5b1da2a8225acbde10b412652cfba7416 gpu-pipelined/kev_4b_decode_fp16_metal_pf128/head/kev_head.json
|
| 7 |
+
862ef6355b0169ce54a2b9a9159b9fcdbcdae2b227aa18075baf86fbcc9902e4 gpu-pipelined/kev_4b_decode_fp16_metal_pf128/kev_4b_decode_fp16_metal_pf128.aimodel/main.hash
|
| 8 |
+
da529dd48f08a7535ca6b8dfb5640723641c2058daf7757a52a9a527413eb17f gpu-pipelined/kev_4b_decode_fp16_metal_pf128/kev_4b_decode_fp16_metal_pf128.aimodel/main.mlirb
|
| 9 |
+
2ab4242dec3222dd5338649a17fb1dc0cfc60f969f9d9bef27ce38fb84ec675b gpu-pipelined/kev_4b_decode_fp16_metal_pf128/kev_4b_decode_fp16_metal_pf128.aimodel/metadata.json
|
| 10 |
+
e641e44fc9d540614e7bc192e49df78c9d8a7d8e23309fe1bef499a5c9039b08 gpu-pipelined/kev_4b_decode_fp16_metal_pf128/metadata.json
|
| 11 |
+
fe000e3ed39ed12b8d2481d527d44f93c65d37e87645d2dcc80d1bf9d50d2927 gpu-pipelined/kev_4b_decode_fp16_metal_pf128/tokenizer/tokenizer.json
|
| 12 |
+
3891e840d7dc5fca0af33d3a25083a735e36fe06214e3f707024820cb6b9f89c gpu-pipelined/kev_4b_decode_fp16_metal_pf128/tokenizer/tokenizer_config.json
|
adapter_config.json
ADDED
|
@@ -0,0 +1,56 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"alora_invocation_tokens": null,
|
| 3 |
+
"alpha_pattern": {},
|
| 4 |
+
"arrow_config": null,
|
| 5 |
+
"auto_mapping": null,
|
| 6 |
+
"base_model_name_or_path": "Qwen/Qwen3.5-4B-Base",
|
| 7 |
+
"bias": "none",
|
| 8 |
+
"corda_config": null,
|
| 9 |
+
"ensure_weight_tying": false,
|
| 10 |
+
"eva_config": null,
|
| 11 |
+
"exclude_modules": null,
|
| 12 |
+
"fan_in_fan_out": false,
|
| 13 |
+
"inference_mode": true,
|
| 14 |
+
"init_lora_weights": true,
|
| 15 |
+
"kasa_config": null,
|
| 16 |
+
"layer_replication": null,
|
| 17 |
+
"layers_pattern": null,
|
| 18 |
+
"layers_to_transform": null,
|
| 19 |
+
"loftq_config": {},
|
| 20 |
+
"lora_alpha": 32,
|
| 21 |
+
"lora_bias": false,
|
| 22 |
+
"lora_dropout": 0.05,
|
| 23 |
+
"lora_ga_config": null,
|
| 24 |
+
"megatron_config": null,
|
| 25 |
+
"megatron_core": "megatron.core",
|
| 26 |
+
"modules_to_save": null,
|
| 27 |
+
"monteclora_config": null,
|
| 28 |
+
"peft_type": "LORA",
|
| 29 |
+
"peft_version": "0.21.0",
|
| 30 |
+
"qalora_group_size": 16,
|
| 31 |
+
"r": 16,
|
| 32 |
+
"rank_pattern": {},
|
| 33 |
+
"revision": null,
|
| 34 |
+
"target_modules": [
|
| 35 |
+
"in_proj_b",
|
| 36 |
+
"in_proj_qkv",
|
| 37 |
+
"gate_proj",
|
| 38 |
+
"v_proj",
|
| 39 |
+
"o_proj",
|
| 40 |
+
"k_proj",
|
| 41 |
+
"up_proj",
|
| 42 |
+
"in_proj_z",
|
| 43 |
+
"in_proj_a",
|
| 44 |
+
"out_proj",
|
| 45 |
+
"down_proj",
|
| 46 |
+
"q_proj"
|
| 47 |
+
],
|
| 48 |
+
"target_parameters": null,
|
| 49 |
+
"task_type": "FEATURE_EXTRACTION",
|
| 50 |
+
"trainable_token_indices": null,
|
| 51 |
+
"use_bdlora": null,
|
| 52 |
+
"use_dora": false,
|
| 53 |
+
"use_qalora": false,
|
| 54 |
+
"use_rslora": false,
|
| 55 |
+
"velora_config": null
|
| 56 |
+
}
|
config.json
ADDED
|
@@ -0,0 +1,104 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"architectures": [
|
| 3 |
+
"Qwen3_5ForConditionalGeneration"
|
| 4 |
+
],
|
| 5 |
+
"image_token_id": 248056,
|
| 6 |
+
"model_type": "qwen3_5",
|
| 7 |
+
"text_config": {
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"attn_output_gate": true,
|
| 11 |
+
"dtype": "bfloat16",
|
| 12 |
+
"eos_token_id": 248044,
|
| 13 |
+
"full_attention_interval": 4,
|
| 14 |
+
"head_dim": 256,
|
| 15 |
+
"hidden_act": "silu",
|
| 16 |
+
"hidden_size": 2560,
|
| 17 |
+
"initializer_range": 0.02,
|
| 18 |
+
"intermediate_size": 9216,
|
| 19 |
+
"layer_types": [
|
| 20 |
+
"linear_attention",
|
| 21 |
+
"linear_attention",
|
| 22 |
+
"linear_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"linear_attention",
|
| 25 |
+
"linear_attention",
|
| 26 |
+
"linear_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"linear_attention",
|
| 29 |
+
"linear_attention",
|
| 30 |
+
"linear_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"linear_attention",
|
| 33 |
+
"linear_attention",
|
| 34 |
+
"linear_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"linear_attention",
|
| 37 |
+
"linear_attention",
|
| 38 |
+
"linear_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"linear_attention",
|
| 41 |
+
"linear_attention",
|
| 42 |
+
"linear_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"linear_attention",
|
| 45 |
+
"linear_attention",
|
| 46 |
+
"linear_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"linear_attention",
|
| 49 |
+
"linear_attention",
|
| 50 |
+
"linear_attention",
|
| 51 |
+
"full_attention"
|
| 52 |
+
],
|
| 53 |
+
"linear_conv_kernel_dim": 4,
|
| 54 |
+
"linear_key_head_dim": 128,
|
| 55 |
+
"linear_num_key_heads": 16,
|
| 56 |
+
"linear_num_value_heads": 32,
|
| 57 |
+
"linear_value_head_dim": 128,
|
| 58 |
+
"max_position_embeddings": 262144,
|
| 59 |
+
"mlp_only_layers": [],
|
| 60 |
+
"model_type": "qwen3_5_text",
|
| 61 |
+
"mtp_num_hidden_layers": 1,
|
| 62 |
+
"mtp_use_dedicated_embeddings": false,
|
| 63 |
+
"num_attention_heads": 16,
|
| 64 |
+
"num_hidden_layers": 32,
|
| 65 |
+
"num_key_value_heads": 4,
|
| 66 |
+
"rms_norm_eps": 1e-06,
|
| 67 |
+
"tie_word_embeddings": true,
|
| 68 |
+
"use_cache": true,
|
| 69 |
+
"vocab_size": 248320,
|
| 70 |
+
"mamba_ssm_dtype": "float32",
|
| 71 |
+
"rope_parameters": {
|
| 72 |
+
"mrope_interleaved": true,
|
| 73 |
+
"mrope_section": [
|
| 74 |
+
11,
|
| 75 |
+
11,
|
| 76 |
+
10
|
| 77 |
+
],
|
| 78 |
+
"rope_type": "default",
|
| 79 |
+
"rope_theta": 10000000,
|
| 80 |
+
"partial_rotary_factor": 0.25
|
| 81 |
+
}
|
| 82 |
+
},
|
| 83 |
+
"tie_word_embeddings": true,
|
| 84 |
+
"transformers_version": "4.57.0.dev0",
|
| 85 |
+
"video_token_id": 248057,
|
| 86 |
+
"vision_config": {
|
| 87 |
+
"deepstack_visual_indexes": [],
|
| 88 |
+
"depth": 24,
|
| 89 |
+
"hidden_act": "gelu_pytorch_tanh",
|
| 90 |
+
"hidden_size": 1024,
|
| 91 |
+
"in_channels": 3,
|
| 92 |
+
"initializer_range": 0.02,
|
| 93 |
+
"intermediate_size": 4096,
|
| 94 |
+
"model_type": "qwen3_5",
|
| 95 |
+
"num_heads": 16,
|
| 96 |
+
"num_position_embeddings": 2304,
|
| 97 |
+
"out_hidden_size": 2560,
|
| 98 |
+
"patch_size": 16,
|
| 99 |
+
"spatial_merge_size": 2,
|
| 100 |
+
"temporal_patch_size": 2
|
| 101 |
+
},
|
| 102 |
+
"vision_end_token_id": 248054,
|
| 103 |
+
"vision_start_token_id": 248053
|
| 104 |
+
}
|
gpu-pipelined/kev_4b_decode_fp16_metal_pf128/head/head.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:36c392f75fa5eff2243b07664e1edb1dc4e771dfc9d145469ba175db54157106
|
| 3 |
+
size 5245304
|
gpu-pipelined/kev_4b_decode_fp16_metal_pf128/head/kev_head.json
ADDED
|
@@ -0,0 +1,67 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"model": "kev-4b",
|
| 3 |
+
"hidden_size": 2560,
|
| 4 |
+
"head_dim": 256,
|
| 5 |
+
"scale": 0.0625,
|
| 6 |
+
"temperature": 2.406050072164233,
|
| 7 |
+
"formula": "z_k = ((k.weight @ h_opt_k + k.bias) . (q.weight @ h_decide + q.bias)) * scale; p = softmax_k(z / temperature)",
|
| 8 |
+
"readout": {
|
| 9 |
+
"decide": "hidden state at the row's last token (<decide>)",
|
| 10 |
+
"option_k": "hidden state at option k's </opt> token",
|
| 11 |
+
"hidden": "the decoder's last_hidden_state (after the final RMSNorm), fp32"
|
| 12 |
+
},
|
| 13 |
+
"row": "one causal row per question: [state] + user_tokens(render(state)) + [q] + user_tokens(render(instructions)) + for each option ([opt] + user_tokens(option) + [opt_end]) + [decide]; positions 0..L-1; fresh state per row",
|
| 14 |
+
"user_tokens": "re.sub(r'<\\|([A-Za-z0-9_]+)\\|>', r'<¦\\1¦>', text), tokenized without special tokens",
|
| 15 |
+
"delimiters": {
|
| 16 |
+
"state": {
|
| 17 |
+
"token": "<|fim_prefix|>",
|
| 18 |
+
"id": 248060
|
| 19 |
+
},
|
| 20 |
+
"q": {
|
| 21 |
+
"token": "<|fim_middle|>",
|
| 22 |
+
"id": 248061
|
| 23 |
+
},
|
| 24 |
+
"opt": {
|
| 25 |
+
"token": "<|box_start|>",
|
| 26 |
+
"id": 248049
|
| 27 |
+
},
|
| 28 |
+
"opt_end": {
|
| 29 |
+
"token": "<|box_end|>",
|
| 30 |
+
"id": 248050
|
| 31 |
+
},
|
| 32 |
+
"decide": {
|
| 33 |
+
"token": "<|fim_suffix|>",
|
| 34 |
+
"id": 248062
|
| 35 |
+
}
|
| 36 |
+
},
|
| 37 |
+
"pad": {
|
| 38 |
+
"token": "<|endoftext|>",
|
| 39 |
+
"id": 248044
|
| 40 |
+
},
|
| 41 |
+
"base": {
|
| 42 |
+
"repo": "Qwen/Qwen3.5-4B-Base",
|
| 43 |
+
"revision": "1001bb4d826a52d1f399e183466143f4da7b741b"
|
| 44 |
+
},
|
| 45 |
+
"adapter": {
|
| 46 |
+
"repo": "jaredpalmer/kev-4b",
|
| 47 |
+
"tag": "v1.0",
|
| 48 |
+
"revision": "591dcb5bd6d05eb0b5131ea6608f93f10243335c",
|
| 49 |
+
"resolved_commit": "6cfce5c2fa4b4bd64026336ab649c5ca78857d52",
|
| 50 |
+
"adapter_sha256": "90e817356246e7f18bfa7ca3d31794cd4fbeb3332a66a84cb51d9ceae925f2b2",
|
| 51 |
+
"head_pt_sha256": "dd633435998ecc751ac538717a3742e32149500fabf7d7276287dbf0693f347c"
|
| 52 |
+
},
|
| 53 |
+
"head_safetensors_sha256": "36c392f75fa5eff2243b07664e1edb1dc4e771dfc9d145469ba175db54157106",
|
| 54 |
+
"temperature_fit": {
|
| 55 |
+
"rows": "runs/release/kev-4b-r10/development/rows.json",
|
| 56 |
+
"n": 1264,
|
| 57 |
+
"method": "min micro mean NLL over a 121-point log grid 0.25..4"
|
| 58 |
+
},
|
| 59 |
+
"record0_check": [
|
| 60 |
+
{
|
| 61 |
+
"record": "tv4_000",
|
| 62 |
+
"qid": "answer",
|
| 63 |
+
"bit_equal": true,
|
| 64 |
+
"max_abs_diff": 0.0
|
| 65 |
+
}
|
| 66 |
+
]
|
| 67 |
+
}
|
gpu-pipelined/kev_4b_decode_fp16_metal_pf128/kev_4b_decode_fp16_metal_pf128.aimodel/main.hash
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
�R�ԏ�S\��ߵd#d X��uzR��'A>�
|
gpu-pipelined/kev_4b_decode_fp16_metal_pf128/kev_4b_decode_fp16_metal_pf128.aimodel/main.mlirb
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:da529dd48f08a7535ca6b8dfb5640723641c2058daf7757a52a9a527413eb17f
|
| 3 |
+
size 8412500461
|
gpu-pipelined/kev_4b_decode_fp16_metal_pf128/kev_4b_decode_fp16_metal_pf128.aimodel/metadata.json
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"creationDate" : "20261004T000304Z",
|
| 3 |
+
"producer" : "coreai-core 1.0.0b2",
|
| 4 |
+
"assetVersion" : "2.0"
|
| 5 |
+
}
|
gpu-pipelined/kev_4b_decode_fp16_metal_pf128/metadata.json
ADDED
|
@@ -0,0 +1,276 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"metadata_version": "0.2",
|
| 3 |
+
"kind": "decision-backbone",
|
| 4 |
+
"name": "kev_4b_decode_fp16_metal_pf128",
|
| 5 |
+
"assets": {
|
| 6 |
+
"main": "kev_4b_decode_fp16_metal_pf128.aimodel"
|
| 7 |
+
},
|
| 8 |
+
"language": {
|
| 9 |
+
"tokenizer": "Qwen/Qwen3.5-4B-Base",
|
| 10 |
+
"vocab_size": 248320,
|
| 11 |
+
"max_context_length": 4096,
|
| 12 |
+
"embedded_tokenizer": true,
|
| 13 |
+
"prefill_chunk": 128,
|
| 14 |
+
"static_inputs": [],
|
| 15 |
+
"output": "hidden [1, 128, 2560] fp16, every position",
|
| 16 |
+
"function_map": {
|
| 17 |
+
"main": [
|
| 18 |
+
"main"
|
| 19 |
+
]
|
| 20 |
+
},
|
| 21 |
+
"tokenizer_revision": "1001bb4d826a52d1f399e183466143f4da7b741b"
|
| 22 |
+
},
|
| 23 |
+
"source": {
|
| 24 |
+
"model_definition": "torch (coreai_models.models.macos.qwen3_5, conversion/kev/qwen3_5_kev_decoder.py)",
|
| 25 |
+
"adapter": {
|
| 26 |
+
"repo": "jaredpalmer/kev-4b",
|
| 27 |
+
"tag": "v1.0",
|
| 28 |
+
"tag_target": "591dcb5bd6d05eb0b5131ea6608f93f10243335c",
|
| 29 |
+
"resolved_commit": "6cfce5c2fa4b4bd64026336ab649c5ca78857d52",
|
| 30 |
+
"weights_identical_to": "139fdd94f1b6a6ad80cc15e08fcb99cac885a101 (the card's weight revision; same LFS oids)",
|
| 31 |
+
"adapter_model_safetensors_sha256": "90e817356246e7f18bfa7ca3d31794cd4fbeb3332a66a84cb51d9ceae925f2b2",
|
| 32 |
+
"head_pt_sha256": "dd633435998ecc751ac538717a3742e32149500fabf7d7276287dbf0693f347c",
|
| 33 |
+
"license": "apache-2.0"
|
| 34 |
+
},
|
| 35 |
+
"base": {
|
| 36 |
+
"repo": "Qwen/Qwen3.5-4B-Base",
|
| 37 |
+
"revision": "1001bb4d826a52d1f399e183466143f4da7b741b",
|
| 38 |
+
"safetensors_sha256": {
|
| 39 |
+
"model.safetensors-00001-of-00002.safetensors": "df547074dce70532a0493e5433152bd17a65efb89088cfabc2e7e2371a93d712",
|
| 40 |
+
"model.safetensors-00002-of-00002.safetensors": "590fbaac095dd31db886c322d9d2f7df47777966391acf306ddddc3e4e3a15ef"
|
| 41 |
+
},
|
| 42 |
+
"license": "apache-2.0"
|
| 43 |
+
},
|
| 44 |
+
"merge": {
|
| 45 |
+
"tool": "scripts/merge_lora_checkpoint.py (https://github.com/jaredpalmer/kev tag kev-1.0, commit 6b719c3c3f367295f6ef336f4f751cf5ff970abc)",
|
| 46 |
+
"formula": "W + (B @ A) * alpha / r (alpha / r = 2.0) on every adapted weight, in fp32, written fp32 (12 target kinds, 248 adapted tensors); every other weight unchanged",
|
| 47 |
+
"weights_sha256": "904380cbf0be134e122f2abd7c7bab52f1e2a0f373d3a2e7947a3025e882f3f3",
|
| 48 |
+
"model_safetensors_sha256": {
|
| 49 |
+
"model-00001-of-00004.safetensors": "45a919c6e0c4a907324e8735cd3be10ab8bdad2ba67fa5844367e0fe3a4bd18f",
|
| 50 |
+
"model-00002-of-00004.safetensors": "9e35c50caf92e662b03a092fc6b57db69683f9ae4076661a2aef962ea175aaab",
|
| 51 |
+
"model-00003-of-00004.safetensors": "c9036c7fed9c87d419f81435955b4a84e6803c1d8b7029e37cd9979aaf6815f3",
|
| 52 |
+
"model-00004-of-00004.safetensors": "47cde07c0ef1909177ea7c6f6bc75033443ca186f8567d72a9ef5bfd000d2b89"
|
| 53 |
+
},
|
| 54 |
+
"tensors": 426
|
| 55 |
+
},
|
| 56 |
+
"hf_model_id": "jaredpalmer/kev-4b",
|
| 57 |
+
"hf_revision": "6cfce5c2fa4b4bd64026336ab649c5ca78857d52",
|
| 58 |
+
"tokenizer": {
|
| 59 |
+
"repo": "Qwen/Qwen3.5-4B-Base",
|
| 60 |
+
"revision": "1001bb4d826a52d1f399e183466143f4da7b741b",
|
| 61 |
+
"files_sha256": {
|
| 62 |
+
"tokenizer.json": "fe000e3ed39ed12b8d2481d527d44f93c65d37e87645d2dcc80d1bf9d50d2927",
|
| 63 |
+
"tokenizer_config.json": "3891e840d7dc5fca0af33d3a25083a735e36fe06214e3f707024820cb6b9f89c"
|
| 64 |
+
},
|
| 65 |
+
"shipped_in": "tokenizer/ (verbatim copies of the two files; embedded_tokenizer)"
|
| 66 |
+
},
|
| 67 |
+
"hf_revision_note": "the commit the Hub resolves the tag v1.0 to; the adapter and head.pt are the LFS objects named in source.adapter. The exporter read the merged checkpoint (source.merge) through a local cache id, which is not recorded here."
|
| 68 |
+
},
|
| 69 |
+
"compression": null,
|
| 70 |
+
"decision": {
|
| 71 |
+
"output": "hidden [1, S, 2560] per call: the final-norm hidden state at every position (no vocabulary head in the graph)",
|
| 72 |
+
"readout": "one row per question, fresh zero states per row; a row of T ids runs as ceil(T / 128) calls of 'main' (static S = 128): call k gets ids[128k : 128k + 128] with position_ids 0..128k+127; the last call is padded with <|endoftext|> (248044) and the hidden rows of the padded positions are discarded (causal: they cannot reach a real position). The hidden rows [1, 128, 2560] of every call, concatenated and cut to T, are the backbone's final-norm last_hidden_state [T, 2560] the head reads. A row fits when ceil(T / 128) * 128 <= 4095 (the position dim's upper bound; the KV sequence dim is allocated at 4096).",
|
| 73 |
+
"row": {
|
| 74 |
+
"source": "the author's kev.model.encode + rows_of (row form, https://github.com/jaredpalmer/kev tag kev-1.0); conversion/kev/oracle_kev.py records the oracle's rows",
|
| 75 |
+
"layout": "[<state>] + user_tokens(render(state)) + [<q>] + user_tokens(render(instructions)) + for each option ([<opt>] + user_tokens(option_text) + [</opt>]) + [<decide>]",
|
| 76 |
+
"delimiters": {
|
| 77 |
+
"state": {
|
| 78 |
+
"token": "<|fim_prefix|>",
|
| 79 |
+
"id": 248060
|
| 80 |
+
},
|
| 81 |
+
"q": {
|
| 82 |
+
"token": "<|fim_middle|>",
|
| 83 |
+
"id": 248061
|
| 84 |
+
},
|
| 85 |
+
"opt": {
|
| 86 |
+
"token": "<|box_start|>",
|
| 87 |
+
"id": 248049
|
| 88 |
+
},
|
| 89 |
+
"opt_end": {
|
| 90 |
+
"token": "<|box_end|>",
|
| 91 |
+
"id": 248050
|
| 92 |
+
},
|
| 93 |
+
"decide": {
|
| 94 |
+
"token": "<|fim_suffix|>",
|
| 95 |
+
"id": 248062
|
| 96 |
+
}
|
| 97 |
+
},
|
| 98 |
+
"pad": {
|
| 99 |
+
"token": "<|endoftext|>",
|
| 100 |
+
"id": 248044
|
| 101 |
+
},
|
| 102 |
+
"bos": null,
|
| 103 |
+
"positions": "0..L-1 (one causal row = the state ids then the question's branch ids)",
|
| 104 |
+
"user_tokens": "re.sub(r'<\\|([A-Za-z0-9_]+)\\|>', r'<¦\\1¦>', text), then tokenized without special tokens (user text can never produce a delimiter)",
|
| 105 |
+
"tokenizer": "tokenizer/ (the base model's pinned tokenizer.json + tokenizer_config.json)",
|
| 106 |
+
"readout_positions": {
|
| 107 |
+
"decide": "the row's last token (<decide>)",
|
| 108 |
+
"options": "each option's </opt> token, in option order"
|
| 109 |
+
},
|
| 110 |
+
"limits_author_serving": "state <= 65,536 tokens (the <state> token included), a row <= 73,728; this graph: one row <= the readout bound above"
|
| 111 |
+
},
|
| 112 |
+
"request": {
|
| 113 |
+
"source": "kev.api.to_record (SystemOne request -> internal record)",
|
| 114 |
+
"render": "None -> ''; str / int / float / bool -> str(v); array -> one line per item '- ' + render(item) (nested items indented by two spaces per level); object -> one line per key 'key: value', or 'key:' then the nested render indented by two spaces when the value is an object or an array",
|
| 115 |
+
"state": "render(state)",
|
| 116 |
+
"instructions": "render(instructions) (omitted -> '': the <q> token is followed by the options)",
|
| 117 |
+
"option_text": "name if description is None or '' else 'name: ' + render(description)",
|
| 118 |
+
"options": {
|
| 119 |
+
"noul": [
|
| 120 |
+
"option_text('no', criteria.false)",
|
| 121 |
+
"option_text('yes', criteria.true)"
|
| 122 |
+
],
|
| 123 |
+
"choice": "option_text(name, description) for each criteria entry, in criteria order",
|
| 124 |
+
"score": "render(level) for each level, in order"
|
| 125 |
+
},
|
| 126 |
+
"question_keys": {
|
| 127 |
+
"noul": [
|
| 128 |
+
"false",
|
| 129 |
+
"true"
|
| 130 |
+
],
|
| 131 |
+
"choice": "the criteria names, in order",
|
| 132 |
+
"score": "'0'..'n-1'"
|
| 133 |
+
},
|
| 134 |
+
"max_options": 255
|
| 135 |
+
},
|
| 136 |
+
"head": {
|
| 137 |
+
"files": [
|
| 138 |
+
"head/head.safetensors",
|
| 139 |
+
"head/kev_head.json"
|
| 140 |
+
],
|
| 141 |
+
"formula": "z_k = ((k.weight @ h_opt_k + k.bias) . (q.weight @ h_decide + q.bias)) * 0.0625; p = softmax_k(z / temperature)",
|
| 142 |
+
"scale": 0.0625,
|
| 143 |
+
"temperature": "kev_head.json temperature (2.406050072164233, the author's calibration in head.pt)",
|
| 144 |
+
"code": "kev.model.PointerHead.forward (q, k = nn.Linear(2560, 256)); weights from head.pt, fp32",
|
| 145 |
+
"dtype_of_record": "fp32 (the gate reads the graph's fp16 hidden through the fp32 head)"
|
| 146 |
+
},
|
| 147 |
+
"softmax": "per question, fp32, over that question's options, after the temperature",
|
| 148 |
+
"response": {
|
| 149 |
+
"shape": "SystemOne-compatible: {model, answers: {question_id: answer}, usage: {input_tokens, output_tokens}, latency_ms} (kev.serve.Server._body)",
|
| 150 |
+
"noul": "{type: 'noul', noul: p(true)}",
|
| 151 |
+
"choice": "{type: 'choice', choice: the key of the first max p, confidence: (p_max - 1/K) / (1 - 1/K) (1 when K = 1), probabilities: {key: p}}",
|
| 152 |
+
"score": "{type: 'score', score: sum(i * p_i), legend: {key: render(level)}, probabilities: {key: p}, confidence: max(0, 1 - E|level - mode| / D), D = mean |i - (L-1)/2| over the L levels, mode = the first most likely level (1 when L = 1)}",
|
| 153 |
+
"normalize": "the confidences normalize p to sum 1 first (all zeros -> uniform)",
|
| 154 |
+
"rounding": "4 decimals (kev.api.round_prob)",
|
| 155 |
+
"usage": "input_tokens = the encoded request's tokens (the author's packed form: state once + every branch); output_tokens = tokens of json.dumps(answers) (kev.api.output_tokens)"
|
| 156 |
+
}
|
| 157 |
+
},
|
| 158 |
+
"gdn_scan": {
|
| 159 |
+
"form": "metal",
|
| 160 |
+
"kernel": "qwen3_5_gdn_chunk_s128",
|
| 161 |
+
"chunk_max": 128,
|
| 162 |
+
"kernel_source": "coreai_models.models.macos.qwen3_5_gdn_metal.build_gdn_chunk_kernel (overlay, unchanged)",
|
| 163 |
+
"threads_per_grid": [
|
| 164 |
+
128,
|
| 165 |
+
32,
|
| 166 |
+
1
|
| 167 |
+
],
|
| 168 |
+
"threads_per_thread_group": [
|
| 169 |
+
128,
|
| 170 |
+
1,
|
| 171 |
+
1
|
| 172 |
+
]
|
| 173 |
+
},
|
| 174 |
+
"compilation": {
|
| 175 |
+
"date": "2026-10-04T05:42:03.402948+00:00",
|
| 176 |
+
"targets": []
|
| 177 |
+
},
|
| 178 |
+
"gate": {
|
| 179 |
+
"bar": {
|
| 180 |
+
"argmax": "every question whose oracle top-2 margin is above 0.02 (near-ties listed apart)",
|
| 181 |
+
"max_abs_dp": 0.02,
|
| 182 |
+
"mean_of_run_mean_abs_dp": 0.002,
|
| 183 |
+
"reset": "bit-equal in every process",
|
| 184 |
+
"finite": true
|
| 185 |
+
},
|
| 186 |
+
"oracle": "the author's fp32 checkpoint on the CPU (kev.model.admit + the row-form forward, threads 1)",
|
| 187 |
+
"graph": "the AOT h16c .aimodelc of this bundle's .aimodel, Python runtime, SpecializationOptions.default(), 128 ids per call",
|
| 188 |
+
"fixture_434": {
|
| 189 |
+
"result": "PASS",
|
| 190 |
+
"rows": 434,
|
| 191 |
+
"argmax_non_near_tie": "422/422",
|
| 192 |
+
"argmax_near_tie": "10/12",
|
| 193 |
+
"max_abs_dp": 0.015306,
|
| 194 |
+
"mean_of_run_mean_abs_dp": 0.0007686,
|
| 195 |
+
"worst_row": {
|
| 196 |
+
"id": "semif_1105577da4c8dad4609d",
|
| 197 |
+
"q": 0,
|
| 198 |
+
"max_abs_dp": 0.015306353569030762
|
| 199 |
+
},
|
| 200 |
+
"reset_bit_equal_all_processes": true,
|
| 201 |
+
"finite": true,
|
| 202 |
+
"oracle_sha256": "eb3031df83a75b746beec3480a4bbb47787e13f8a12412ff135feb9f609c70f9",
|
| 203 |
+
"when": "2026-10-04T02:09:21+00:00",
|
| 204 |
+
"lane_file_sha256": "41dca9cbed8d58bf331480137dee62f212449769c2e723b2f7d79ca7f43ab6b4",
|
| 205 |
+
"transcript": "coreai-model-zoo models/kev-4b/gate-kev-4b-readout.json"
|
| 206 |
+
},
|
| 207 |
+
"heldout_130": {
|
| 208 |
+
"result": "PASS",
|
| 209 |
+
"rows": 130,
|
| 210 |
+
"argmax_non_near_tie": "126/126",
|
| 211 |
+
"argmax_near_tie": "3/4",
|
| 212 |
+
"max_abs_dp": 0.00961,
|
| 213 |
+
"mean_of_run_mean_abs_dp": 0.0006898,
|
| 214 |
+
"worst_row": {
|
| 215 |
+
"id": "tv4sh_11",
|
| 216 |
+
"q": 0,
|
| 217 |
+
"max_abs_dp": 0.009610295295715332
|
| 218 |
+
},
|
| 219 |
+
"reset_bit_equal_all_processes": true,
|
| 220 |
+
"finite": true,
|
| 221 |
+
"oracle_sha256": "dfa101a559450a8b75c140bafd120728561273908a527ad89ce44d5558d31c59",
|
| 222 |
+
"when": "2026-10-04T02:10:22+00:00",
|
| 223 |
+
"lane_file_sha256": "96f097809ee73d5898cdd64614c05033c253a4012175e15a1d818046585d1738",
|
| 224 |
+
"transcript": "coreai-model-zoo models/kev-4b/gate-kev-4b-readout.json"
|
| 225 |
+
},
|
| 226 |
+
"red_arms_v2": {
|
| 227 |
+
"result": "RED (every arm)",
|
| 228 |
+
"transcript": "coreai-model-zoo models/kev-4b/gate-kev-4b-readout.json"
|
| 229 |
+
},
|
| 230 |
+
"swift": {
|
| 231 |
+
"result": "PASS",
|
| 232 |
+
"hidden_sha256_equal_python_gate": {
|
| 233 |
+
"fixture": 434,
|
| 234 |
+
"heldout": 130
|
| 235 |
+
},
|
| 236 |
+
"shared_hidden_bit_equal_direct": {
|
| 237 |
+
"fixture": 434,
|
| 238 |
+
"heldout": 130
|
| 239 |
+
},
|
| 240 |
+
"jit_vs_aot": {
|
| 241 |
+
"records": 10,
|
| 242 |
+
"rows": 25,
|
| 243 |
+
"hidden_sha256_equal": 25,
|
| 244 |
+
"p_bit_equal": 25
|
| 245 |
+
},
|
| 246 |
+
"transcript": "coreai-model-zoo models/kev-4b/gate-kev-4b-swift.json"
|
| 247 |
+
},
|
| 248 |
+
"host": {
|
| 249 |
+
"transcript": "coreai-model-zoo models/kev-4b/gate-kev-4b-host.json"
|
| 250 |
+
},
|
| 251 |
+
"fixture": "coreai-model-zoo models/kev-4b/fixtures-kev-4b.json",
|
| 252 |
+
"devices": "Mac only in this release"
|
| 253 |
+
},
|
| 254 |
+
"metadata_of_record": {
|
| 255 |
+
"exported_metadata_json_sha256": "dcdd0ea1c4ea4369282176db2f937779aab00b944b4f26034e9314b192d88e87",
|
| 256 |
+
"edits": [
|
| 257 |
+
"language.tokenizer = the base repo id (was the local id of the merged checkpoint) + tokenizer_revision",
|
| 258 |
+
"source.hf_model_id / hf_revision = the Kev checkpoint (was the local HF-cache id of the merged weights)",
|
| 259 |
+
"source.tokenizer.shipped_in added",
|
| 260 |
+
"gate: the published form's gate numbers and the zoo transcripts that hold them"
|
| 261 |
+
],
|
| 262 |
+
"host_read_keys_unchanged": [
|
| 263 |
+
"name",
|
| 264 |
+
"kind",
|
| 265 |
+
"assets",
|
| 266 |
+
"language.vocab_size",
|
| 267 |
+
"language.max_context_length",
|
| 268 |
+
"language.prefill_chunk",
|
| 269 |
+
"language.query_len_range",
|
| 270 |
+
"language.query_len_call_max",
|
| 271 |
+
"language.query_len_multiple",
|
| 272 |
+
"decision"
|
| 273 |
+
],
|
| 274 |
+
"written": "2026-10-04 (round 17)"
|
| 275 |
+
}
|
| 276 |
+
}
|
gpu-pipelined/kev_4b_decode_fp16_metal_pf128/tokenizer/tokenizer.json
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:fe000e3ed39ed12b8d2481d527d44f93c65d37e87645d2dcc80d1bf9d50d2927
|
| 3 |
+
size 12807196
|
gpu-pipelined/kev_4b_decode_fp16_metal_pf128/tokenizer/tokenizer_config.json
ADDED
|
@@ -0,0 +1,305 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"add_prefix_space": false,
|
| 3 |
+
"added_tokens_decoder": {
|
| 4 |
+
"248044": {
|
| 5 |
+
"content": "<|endoftext|>",
|
| 6 |
+
"lstrip": false,
|
| 7 |
+
"normalized": false,
|
| 8 |
+
"rstrip": false,
|
| 9 |
+
"single_word": false,
|
| 10 |
+
"special": true
|
| 11 |
+
},
|
| 12 |
+
"248045": {
|
| 13 |
+
"content": "<|im_start|>",
|
| 14 |
+
"lstrip": false,
|
| 15 |
+
"normalized": false,
|
| 16 |
+
"rstrip": false,
|
| 17 |
+
"single_word": false,
|
| 18 |
+
"special": true
|
| 19 |
+
},
|
| 20 |
+
"248046": {
|
| 21 |
+
"content": "<|im_end|>",
|
| 22 |
+
"lstrip": false,
|
| 23 |
+
"normalized": false,
|
| 24 |
+
"rstrip": false,
|
| 25 |
+
"single_word": false,
|
| 26 |
+
"special": true
|
| 27 |
+
},
|
| 28 |
+
"248047": {
|
| 29 |
+
"content": "<|object_ref_start|>",
|
| 30 |
+
"lstrip": false,
|
| 31 |
+
"normalized": false,
|
| 32 |
+
"rstrip": false,
|
| 33 |
+
"single_word": false,
|
| 34 |
+
"special": true
|
| 35 |
+
},
|
| 36 |
+
"248048": {
|
| 37 |
+
"content": "<|object_ref_end|>",
|
| 38 |
+
"lstrip": false,
|
| 39 |
+
"normalized": false,
|
| 40 |
+
"rstrip": false,
|
| 41 |
+
"single_word": false,
|
| 42 |
+
"special": true
|
| 43 |
+
},
|
| 44 |
+
"248049": {
|
| 45 |
+
"content": "<|box_start|>",
|
| 46 |
+
"lstrip": false,
|
| 47 |
+
"normalized": false,
|
| 48 |
+
"rstrip": false,
|
| 49 |
+
"single_word": false,
|
| 50 |
+
"special": true
|
| 51 |
+
},
|
| 52 |
+
"248050": {
|
| 53 |
+
"content": "<|box_end|>",
|
| 54 |
+
"lstrip": false,
|
| 55 |
+
"normalized": false,
|
| 56 |
+
"rstrip": false,
|
| 57 |
+
"single_word": false,
|
| 58 |
+
"special": true
|
| 59 |
+
},
|
| 60 |
+
"248051": {
|
| 61 |
+
"content": "<|quad_start|>",
|
| 62 |
+
"lstrip": false,
|
| 63 |
+
"normalized": false,
|
| 64 |
+
"rstrip": false,
|
| 65 |
+
"single_word": false,
|
| 66 |
+
"special": true
|
| 67 |
+
},
|
| 68 |
+
"248052": {
|
| 69 |
+
"content": "<|quad_end|>",
|
| 70 |
+
"lstrip": false,
|
| 71 |
+
"normalized": false,
|
| 72 |
+
"rstrip": false,
|
| 73 |
+
"single_word": false,
|
| 74 |
+
"special": true
|
| 75 |
+
},
|
| 76 |
+
"248053": {
|
| 77 |
+
"content": "<|vision_start|>",
|
| 78 |
+
"lstrip": false,
|
| 79 |
+
"normalized": false,
|
| 80 |
+
"rstrip": false,
|
| 81 |
+
"single_word": false,
|
| 82 |
+
"special": true
|
| 83 |
+
},
|
| 84 |
+
"248054": {
|
| 85 |
+
"content": "<|vision_end|>",
|
| 86 |
+
"lstrip": false,
|
| 87 |
+
"normalized": false,
|
| 88 |
+
"rstrip": false,
|
| 89 |
+
"single_word": false,
|
| 90 |
+
"special": true
|
| 91 |
+
},
|
| 92 |
+
"248055": {
|
| 93 |
+
"content": "<|vision_pad|>",
|
| 94 |
+
"lstrip": false,
|
| 95 |
+
"normalized": false,
|
| 96 |
+
"rstrip": false,
|
| 97 |
+
"single_word": false,
|
| 98 |
+
"special": true
|
| 99 |
+
},
|
| 100 |
+
"248056": {
|
| 101 |
+
"content": "<|image_pad|>",
|
| 102 |
+
"lstrip": false,
|
| 103 |
+
"normalized": false,
|
| 104 |
+
"rstrip": false,
|
| 105 |
+
"single_word": false,
|
| 106 |
+
"special": true
|
| 107 |
+
},
|
| 108 |
+
"248057": {
|
| 109 |
+
"content": "<|video_pad|>",
|
| 110 |
+
"lstrip": false,
|
| 111 |
+
"normalized": false,
|
| 112 |
+
"rstrip": false,
|
| 113 |
+
"single_word": false,
|
| 114 |
+
"special": true
|
| 115 |
+
},
|
| 116 |
+
"248058": {
|
| 117 |
+
"content": "<tool_call>",
|
| 118 |
+
"lstrip": false,
|
| 119 |
+
"normalized": false,
|
| 120 |
+
"rstrip": false,
|
| 121 |
+
"single_word": false,
|
| 122 |
+
"special": false
|
| 123 |
+
},
|
| 124 |
+
"248059": {
|
| 125 |
+
"content": "</tool_call>",
|
| 126 |
+
"lstrip": false,
|
| 127 |
+
"normalized": false,
|
| 128 |
+
"rstrip": false,
|
| 129 |
+
"single_word": false,
|
| 130 |
+
"special": false
|
| 131 |
+
},
|
| 132 |
+
"248060": {
|
| 133 |
+
"content": "<|fim_prefix|>",
|
| 134 |
+
"lstrip": false,
|
| 135 |
+
"normalized": false,
|
| 136 |
+
"rstrip": false,
|
| 137 |
+
"single_word": false,
|
| 138 |
+
"special": false
|
| 139 |
+
},
|
| 140 |
+
"248061": {
|
| 141 |
+
"content": "<|fim_middle|>",
|
| 142 |
+
"lstrip": false,
|
| 143 |
+
"normalized": false,
|
| 144 |
+
"rstrip": false,
|
| 145 |
+
"single_word": false,
|
| 146 |
+
"special": false
|
| 147 |
+
},
|
| 148 |
+
"248062": {
|
| 149 |
+
"content": "<|fim_suffix|>",
|
| 150 |
+
"lstrip": false,
|
| 151 |
+
"normalized": false,
|
| 152 |
+
"rstrip": false,
|
| 153 |
+
"single_word": false,
|
| 154 |
+
"special": false
|
| 155 |
+
},
|
| 156 |
+
"248063": {
|
| 157 |
+
"content": "<|fim_pad|>",
|
| 158 |
+
"lstrip": false,
|
| 159 |
+
"normalized": false,
|
| 160 |
+
"rstrip": false,
|
| 161 |
+
"single_word": false,
|
| 162 |
+
"special": false
|
| 163 |
+
},
|
| 164 |
+
"248064": {
|
| 165 |
+
"content": "<|repo_name|>",
|
| 166 |
+
"lstrip": false,
|
| 167 |
+
"normalized": false,
|
| 168 |
+
"rstrip": false,
|
| 169 |
+
"single_word": false,
|
| 170 |
+
"special": false
|
| 171 |
+
},
|
| 172 |
+
"248065": {
|
| 173 |
+
"content": "<|file_sep|>",
|
| 174 |
+
"lstrip": false,
|
| 175 |
+
"normalized": false,
|
| 176 |
+
"rstrip": false,
|
| 177 |
+
"single_word": false,
|
| 178 |
+
"special": false
|
| 179 |
+
},
|
| 180 |
+
"248066": {
|
| 181 |
+
"content": "<tool_response>",
|
| 182 |
+
"lstrip": false,
|
| 183 |
+
"normalized": false,
|
| 184 |
+
"rstrip": false,
|
| 185 |
+
"single_word": false,
|
| 186 |
+
"special": false
|
| 187 |
+
},
|
| 188 |
+
"248067": {
|
| 189 |
+
"content": "</tool_response>",
|
| 190 |
+
"lstrip": false,
|
| 191 |
+
"normalized": false,
|
| 192 |
+
"rstrip": false,
|
| 193 |
+
"single_word": false,
|
| 194 |
+
"special": false
|
| 195 |
+
},
|
| 196 |
+
"248068": {
|
| 197 |
+
"content": "<think>",
|
| 198 |
+
"lstrip": false,
|
| 199 |
+
"normalized": false,
|
| 200 |
+
"rstrip": false,
|
| 201 |
+
"single_word": false,
|
| 202 |
+
"special": false
|
| 203 |
+
},
|
| 204 |
+
"248069": {
|
| 205 |
+
"content": "</think>",
|
| 206 |
+
"lstrip": false,
|
| 207 |
+
"normalized": false,
|
| 208 |
+
"rstrip": false,
|
| 209 |
+
"single_word": false,
|
| 210 |
+
"special": false
|
| 211 |
+
},
|
| 212 |
+
"248070": {
|
| 213 |
+
"content": "<|audio_start|>",
|
| 214 |
+
"lstrip": false,
|
| 215 |
+
"normalized": false,
|
| 216 |
+
"rstrip": false,
|
| 217 |
+
"single_word": false,
|
| 218 |
+
"special": true
|
| 219 |
+
},
|
| 220 |
+
"248071": {
|
| 221 |
+
"content": "<|audio_end|>",
|
| 222 |
+
"lstrip": false,
|
| 223 |
+
"normalized": false,
|
| 224 |
+
"rstrip": false,
|
| 225 |
+
"single_word": false,
|
| 226 |
+
"special": true
|
| 227 |
+
},
|
| 228 |
+
"248072": {
|
| 229 |
+
"content": "<tts_pad>",
|
| 230 |
+
"lstrip": false,
|
| 231 |
+
"normalized": false,
|
| 232 |
+
"rstrip": false,
|
| 233 |
+
"single_word": false,
|
| 234 |
+
"special": true
|
| 235 |
+
},
|
| 236 |
+
"248073": {
|
| 237 |
+
"content": "<tts_text_bos>",
|
| 238 |
+
"lstrip": false,
|
| 239 |
+
"normalized": false,
|
| 240 |
+
"rstrip": false,
|
| 241 |
+
"single_word": false,
|
| 242 |
+
"special": true
|
| 243 |
+
},
|
| 244 |
+
"248074": {
|
| 245 |
+
"content": "<tts_text_eod>",
|
| 246 |
+
"lstrip": false,
|
| 247 |
+
"normalized": false,
|
| 248 |
+
"rstrip": false,
|
| 249 |
+
"single_word": false,
|
| 250 |
+
"special": true
|
| 251 |
+
},
|
| 252 |
+
"248075": {
|
| 253 |
+
"content": "<tts_text_bos_single>",
|
| 254 |
+
"lstrip": false,
|
| 255 |
+
"normalized": false,
|
| 256 |
+
"rstrip": false,
|
| 257 |
+
"single_word": false,
|
| 258 |
+
"special": true
|
| 259 |
+
},
|
| 260 |
+
"248076": {
|
| 261 |
+
"content": "<|audio_pad|>",
|
| 262 |
+
"lstrip": false,
|
| 263 |
+
"normalized": false,
|
| 264 |
+
"rstrip": false,
|
| 265 |
+
"single_word": false,
|
| 266 |
+
"special": true
|
| 267 |
+
}
|
| 268 |
+
},
|
| 269 |
+
"additional_special_tokens": [
|
| 270 |
+
"<|im_start|>",
|
| 271 |
+
"<|im_end|>",
|
| 272 |
+
"<|object_ref_start|>",
|
| 273 |
+
"<|object_ref_end|>",
|
| 274 |
+
"<|box_start|>",
|
| 275 |
+
"<|box_end|>",
|
| 276 |
+
"<|quad_start|>",
|
| 277 |
+
"<|quad_end|>",
|
| 278 |
+
"<|vision_start|>",
|
| 279 |
+
"<|vision_end|>",
|
| 280 |
+
"<|vision_pad|>",
|
| 281 |
+
"<|image_pad|>",
|
| 282 |
+
"<|video_pad|>"
|
| 283 |
+
],
|
| 284 |
+
"bos_token": null,
|
| 285 |
+
"chat_template": "{%- set image_count = namespace(value=0) %}\n{%- set video_count = namespace(value=0) %}\n{%- macro render_content(content, do_vision_count, is_system_content=false) %}\n {%- if content is string %}\n {{- content }}\n {%- elif content is iterable and content is not mapping %}\n {%- for item in content %}\n {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}\n {%- if is_system_content %}\n {{- raise_exception('System message cannot contain images.') }}\n {%- endif %}\n {%- if do_vision_count %}\n {%- set image_count.value = image_count.value + 1 %}\n {%- endif %}\n {%- if add_vision_id %}\n {{- 'Picture ' ~ image_count.value ~ ': ' }}\n {%- endif %}\n {{- '<|vision_start|><|image_pad|><|vision_end|>' }}\n {%- elif 'video' in item or item.type == 'video' %}\n {%- if is_system_content %}\n {{- raise_exception('System message cannot contain videos.') }}\n {%- endif %}\n {%- if do_vision_count %}\n {%- set video_count.value = video_count.value + 1 %}\n {%- endif %}\n {%- if add_vision_id %}\n {{- 'Video ' ~ video_count.value ~ ': ' }}\n {%- endif %}\n {{- '<|vision_start|><|video_pad|><|vision_end|>' }}\n {%- elif 'text' in item %}\n {{- item.text }}\n {%- else %}\n {{- raise_exception('Unexpected item type in content.') }}\n {%- endif %}\n {%- endfor %}\n {%- elif content is none or content is undefined %}\n {{- '' }}\n {%- else %}\n {{- raise_exception('Unexpected content type.') }}\n {%- endif %}\n{%- endmacro %}\n{%- if not messages %}\n {{- raise_exception('No messages provided.') }}\n{%- endif %}\n{%- if tools and tools is iterable and tools is not mapping %}\n {{- '<|im_start|>system\\n' }}\n {{- \"# Tools\\n\\nYou have access to the following functions:\\n\\n<tools>\" }}\n {%- for tool in tools %}\n {{- \"\\n\" }}\n {{- tool | tojson }}\n {%- endfor %}\n {{- \"\\n</tools>\" }}\n {{- '\\n\\nIf you choose to call a function ONLY reply in the following format with NO suffix:\\n\\n<tool_call>\\n<function=example_function_name>\\n<parameter=example_parameter_1>\\nvalue_1\\n</parameter>\\n<parameter=example_parameter_2>\\nThis is the value for the second parameter\\nthat can span\\nmultiple lines\\n</parameter>\\n</function>\\n</tool_call>\\n\\n<IMPORTANT>\\nReminder:\\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\\n- Required parameters MUST be specified\\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\\n</IMPORTANT>' }}\n {%- if messages[0].role == 'system' %}\n {%- set content = render_content(messages[0].content, false, true)|trim %}\n {%- if content %}\n {{- '\\n\\n' + content }}\n {%- endif %}\n {%- endif %}\n {{- '<|im_end|>\\n' }}\n{%- else %}\n {%- if messages[0].role == 'system' %}\n {%- set content = render_content(messages[0].content, false, true)|trim %}\n {{- '<|im_start|>system\\n' + content + '<|im_end|>\\n' }}\n {%- endif %}\n{%- endif %}\n{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}\n{%- for message in messages[::-1] %}\n {%- set index = (messages|length - 1) - loop.index0 %}\n {%- if ns.multi_step_tool and message.role == \"user\" %}\n {%- set content = render_content(message.content, false)|trim %}\n {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}\n {%- set ns.multi_step_tool = false %}\n {%- set ns.last_query_index = index %}\n {%- endif %}\n {%- endif %}\n{%- endfor %}\n{%- if ns.multi_step_tool %}\n {{- raise_exception('No user query found in messages.') }}\n{%- endif %}\n{%- for message in messages %}\n {%- set content = render_content(message.content, true)|trim %}\n {%- if message.role == \"system\" %}\n {%- if not loop.first %}\n {{- raise_exception('System message must be at the beginning.') }}\n {%- endif %}\n {%- elif message.role == \"user\" %}\n {{- '<|im_start|>' + message.role + '\\n' + content + '<|im_end|>' + '\\n' }}\n {%- elif message.role == \"assistant\" %}\n {%- set reasoning_content = '' %}\n {%- if message.reasoning_content is string %}\n {%- set reasoning_content = message.reasoning_content %}\n {%- else %}\n {%- if '</think>' in content %}\n {%- set reasoning_content = content.split('</think>')[0].rstrip('\\n').split('<think>')[-1].lstrip('\\n') %}\n {%- set content = content.split('</think>')[-1].lstrip('\\n') %}\n {%- endif %}\n {%- endif %}\n {%- set reasoning_content = reasoning_content|trim %}\n {%- if loop.index0 > ns.last_query_index %}\n {{- '<|im_start|>' + message.role + '\\n<think>\\n' + reasoning_content + '\\n</think>\\n\\n' + content }}\n {%- else %}\n {{- '<|im_start|>' + message.role + '\\n' + content }}\n {%- endif %}\n {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}\n {%- for tool_call in message.tool_calls %}\n {%- if tool_call.function is defined %}\n {%- set tool_call = tool_call.function %}\n {%- endif %}\n {%- if loop.first %}\n {%- if content|trim %}\n {{- '\\n\\n<tool_call>\\n<function=' + tool_call.name + '>\\n' }}\n {%- else %}\n {{- '<tool_call>\\n<function=' + tool_call.name + '>\\n' }}\n {%- endif %}\n {%- else %}\n {{- '\\n<tool_call>\\n<function=' + tool_call.name + '>\\n' }}\n {%- endif %}\n {%- if tool_call.arguments is defined %}\n {%- for args_name, args_value in tool_call.arguments|items %}\n {{- '<parameter=' + args_name + '>\\n' }}\n {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}\n {{- args_value }}\n {{- '\\n</parameter>\\n' }}\n {%- endfor %}\n {%- endif %}\n {{- '</function>\\n</tool_call>' }}\n {%- endfor %}\n {%- endif %}\n {{- '<|im_end|>\\n' }}\n {%- elif message.role == \"tool\" %}\n {%- if loop.previtem and loop.previtem.role != \"tool\" %}\n {{- '<|im_start|>user' }}\n {%- endif %}\n {{- '\\n<tool_response>\\n' }}\n {{- content }}\n {{- '\\n</tool_response>' }}\n {%- if not loop.last and loop.nextitem.role != \"tool\" %}\n {{- '<|im_end|>\\n' }}\n {%- elif loop.last %}\n {{- '<|im_end|>\\n' }}\n {%- endif %}\n {%- else %}\n {{- raise_exception('Unexpected message role.') }}\n {%- endif %}\n{%- endfor %}\n{%- if add_generation_prompt %}\n {{- '<|im_start|>assistant\\n' }}\n {%- if enable_thinking is defined and enable_thinking is false %}\n {{- '<think>\\n\\n</think>\\n\\n' }}\n {%- else %}\n {{- '<think>\\n' }}\n {%- endif %}\n{%- endif %}",
|
| 286 |
+
"clean_up_tokenization_spaces": false,
|
| 287 |
+
"eos_token": "<|endoftext|>",
|
| 288 |
+
"errors": "replace",
|
| 289 |
+
"model_max_length": 262144,
|
| 290 |
+
"pad_token": "<|endoftext|>",
|
| 291 |
+
"split_special_tokens": false,
|
| 292 |
+
"tokenizer_class": "Qwen2Tokenizer",
|
| 293 |
+
"unk_token": null,
|
| 294 |
+
"add_bos_token": false,
|
| 295 |
+
"pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
|
| 296 |
+
"extra_special_tokens": {
|
| 297 |
+
"audio_bos_token": "<|audio_start|>",
|
| 298 |
+
"audio_eos_token": "<|audio_end|>",
|
| 299 |
+
"audio_token": "<|audio_pad|>",
|
| 300 |
+
"image_token": "<|image_pad|>",
|
| 301 |
+
"video_token": "<|video_pad|>",
|
| 302 |
+
"vision_bos_token": "<|vision_start|>",
|
| 303 |
+
"vision_eos_token": "<|vision_end|>"
|
| 304 |
+
}
|
| 305 |
+
}
|