Ollaya package for Cloudflare/clef-flash
Browse files- README.md +44 -0
- flash/calibration.json +9 -0
- flash/decision.json +100 -0
- flash/model-fp32.onnx +3 -0
README.md
ADDED
|
@@ -0,0 +1,44 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
license: apache-2.0
|
| 3 |
+
base_model:
|
| 4 |
+
- Cloudflare/clef-flash
|
| 5 |
+
library_name: onnx
|
| 6 |
+
tags:
|
| 7 |
+
- ollaya
|
| 8 |
+
- onnx
|
| 9 |
+
- decision-model
|
| 10 |
+
- system-one
|
| 11 |
+
pipeline_tag: text-classification
|
| 12 |
+
---
|
| 13 |
+
|
| 14 |
+
# clef for Ollaya
|
| 15 |
+
|
| 16 |
+
[Ollaya](https://github.com/ollaya-dev/ollaya) package of **[Cloudflare/clef-flash](https://huggingface.co/Cloudflare/clef-flash)** by Cloudflare (post-trained model and joint schema head) and the Qwen team (base model).
|
| 17 |
+
Ollaya runs open decision models locally, the way Ollama runs LLMs: typed questions in,
|
| 18 |
+
calibrated answers out, behind a TypeSafe-compatible API.
|
| 19 |
+
|
| 20 |
+
```sh
|
| 21 |
+
ollaya run clef
|
| 22 |
+
```
|
| 23 |
+
|
| 24 |
+
## What is in this repository
|
| 25 |
+
|
| 26 |
+
This repository holds only the files Ollaya derives, with no weights. Each graph is an ONNX export of the
|
| 27 |
+
original model whose weights **reference the authors' own weight files by byte offset**,
|
| 28 |
+
so `ollaya pull` downloads the weights from the upstream repositories, unmodified and pinned to a
|
| 29 |
+
commit, and verifies their sha256.
|
| 30 |
+
|
| 31 |
+
| Tag | Upstream | Files |
|
| 32 |
+
|---|---|---|
|
| 33 |
+
| `clef:flash` | [Cloudflare/clef-flash@17f0b0a](https://huggingface.co/Cloudflare/clef-flash/tree/17f0b0ad64efb65d273590632833508766b2aae6) | `flash/model-fp32.onnx`, `flash/decision.json`, `flash/calibration.json` |
|
| 34 |
+
|
| 35 |
+
Each tag has an fp32 graph, used on CPU and GPU. Each tag also has `decision.json` (sequence layout, special tokens) and
|
| 36 |
+
`calibration.json` (temperatures).
|
| 37 |
+
|
| 38 |
+
## Parity
|
| 39 |
+
|
| 40 |
+
Ollaya's Rust runtime matches the authors' own code (joint_schema_model.py: their encoder, Qwen3.5 model and joint schema head, fp32) on 571 questions from 131 requests, on CUDA: identical token ids and spans, the same 13 rejected requests, the same decision on every question, logits within 4.3e-5 and probabilities within 6.3e-6.
|
| 41 |
+
|
| 42 |
+
## License
|
| 43 |
+
|
| 44 |
+
Same as the upstream model (Apache-2.0). Ollaya itself is Apache-2.0.
|
flash/calibration.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"temperature": [
|
| 3 |
+
1.0,
|
| 4 |
+
1.0,
|
| 5 |
+
1.0
|
| 6 |
+
],
|
| 7 |
+
"temperature_by_options": {},
|
| 8 |
+
"source": "none: upstream answers with a plain softmax of the head's logits"
|
| 9 |
+
}
|
flash/decision.json
ADDED
|
@@ -0,0 +1,100 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"engine": "onnx",
|
| 3 |
+
"family": "clef",
|
| 4 |
+
"layout": "clef-joint-v1",
|
| 5 |
+
"upstream": {
|
| 6 |
+
"repo": "Cloudflare/clef-flash",
|
| 7 |
+
"revision": "17f0b0ad64efb65d273590632833508766b2aae6",
|
| 8 |
+
"code": "joint_schema_model.py in the model repository (encode_record, JointSchemaHead)"
|
| 9 |
+
},
|
| 10 |
+
"contract": {
|
| 11 |
+
"inputs": {
|
| 12 |
+
"input_ids": {
|
| 13 |
+
"dtype": "int64",
|
| 14 |
+
"shape": [
|
| 15 |
+
1,
|
| 16 |
+
"seq"
|
| 17 |
+
],
|
| 18 |
+
"note": "the request's ids; seq a multiple of 64; right-pad with the pad id"
|
| 19 |
+
},
|
| 20 |
+
"token_positions": {
|
| 21 |
+
"dtype": "int64",
|
| 22 |
+
"shape": [
|
| 23 |
+
"tokens"
|
| 24 |
+
],
|
| 25 |
+
"note": "0..tokens-1"
|
| 26 |
+
},
|
| 27 |
+
"question_spans": {
|
| 28 |
+
"dtype": "int64",
|
| 29 |
+
"shape": [
|
| 30 |
+
"questions",
|
| 31 |
+
2
|
| 32 |
+
],
|
| 33 |
+
"note": "instruction tokens [start, end) per question"
|
| 34 |
+
},
|
| 35 |
+
"question_types": {
|
| 36 |
+
"dtype": "int64",
|
| 37 |
+
"shape": [
|
| 38 |
+
"questions"
|
| 39 |
+
],
|
| 40 |
+
"note": "type_index"
|
| 41 |
+
},
|
| 42 |
+
"option_spans": {
|
| 43 |
+
"dtype": "int64",
|
| 44 |
+
"shape": [
|
| 45 |
+
"options",
|
| 46 |
+
2
|
| 47 |
+
],
|
| 48 |
+
"note": "option tokens [start, end), every question's options in order"
|
| 49 |
+
},
|
| 50 |
+
"option_question": {
|
| 51 |
+
"dtype": "int64",
|
| 52 |
+
"shape": [
|
| 53 |
+
"options"
|
| 54 |
+
],
|
| 55 |
+
"note": "the question index of each option"
|
| 56 |
+
}
|
| 57 |
+
},
|
| 58 |
+
"outputs": {
|
| 59 |
+
"logits": {
|
| 60 |
+
"dtype": "float32",
|
| 61 |
+
"shape": [
|
| 62 |
+
"options"
|
| 63 |
+
],
|
| 64 |
+
"note": "raw option logits"
|
| 65 |
+
}
|
| 66 |
+
},
|
| 67 |
+
"seq_multiple": 64,
|
| 68 |
+
"positions": "0..seq-1, implicit",
|
| 69 |
+
"attention": "causal; no mask input"
|
| 70 |
+
},
|
| 71 |
+
"prompt": {
|
| 72 |
+
"prefix": "<|im_start|>system\nRead the complete state and schema. Decide every field jointly. Each answer must be exactly one of that field's allowed options.<|im_end|>\n<|im_start|>user\nSTATE:\n",
|
| 73 |
+
"schema": "\n\nSCHEMA FIELDS:\n",
|
| 74 |
+
"field": "\nFIELD {n}\nID: {id}\nTYPE: {type}\nINSTRUCTION: ",
|
| 75 |
+
"options": "\nALLOWED OPTIONS:\n",
|
| 76 |
+
"option": "OPTION {n}: ",
|
| 77 |
+
"option_end": "\n",
|
| 78 |
+
"field_end": "END FIELD\n",
|
| 79 |
+
"suffix": "\n<|im_end|>\n<|im_start|>assistant\n<think>\n\n</think>\n\nJOINT SCHEMA DECISIONS:"
|
| 80 |
+
},
|
| 81 |
+
"noul_criteria": {
|
| 82 |
+
"true": "The proposition is true or the answer is yes.",
|
| 83 |
+
"false": "The proposition is false or the answer is no."
|
| 84 |
+
},
|
| 85 |
+
"max_tokens": 4096,
|
| 86 |
+
"pad": 248044,
|
| 87 |
+
"type_index": {
|
| 88 |
+
"noul": 0,
|
| 89 |
+
"choice": 1,
|
| 90 |
+
"score": 2
|
| 91 |
+
},
|
| 92 |
+
"option_logits": {
|
| 93 |
+
"noul": "[true, false]",
|
| 94 |
+
"choice": "labels sorted by code point",
|
| 95 |
+
"score": "levels in order"
|
| 96 |
+
},
|
| 97 |
+
"opset": 20,
|
| 98 |
+
"precision": "fp32 compute; weights BF16 (widened by Cast)",
|
| 99 |
+
"weights_in_memory": "bf16"
|
| 100 |
+
}
|
flash/model-fp32.onnx
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:5cc3f10d3ad4158c34411e2e29247d914fe6b70a2dc92b53458ee7bff6b62a57
|
| 3 |
+
size 11991785
|