mertcobanov commited on
Commit
85b1c5c
·
verified ·
1 Parent(s): 9a31984

Ollaya package for Cloudflare/clef-flash

Browse files
README.md ADDED
@@ -0,0 +1,44 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: apache-2.0
3
+ base_model:
4
+ - Cloudflare/clef-flash
5
+ library_name: onnx
6
+ tags:
7
+ - ollaya
8
+ - onnx
9
+ - decision-model
10
+ - system-one
11
+ pipeline_tag: text-classification
12
+ ---
13
+
14
+ # clef for Ollaya
15
+
16
+ [Ollaya](https://github.com/ollaya-dev/ollaya) package of **[Cloudflare/clef-flash](https://huggingface.co/Cloudflare/clef-flash)** by Cloudflare (post-trained model and joint schema head) and the Qwen team (base model).
17
+ Ollaya runs open decision models locally, the way Ollama runs LLMs: typed questions in,
18
+ calibrated answers out, behind a TypeSafe-compatible API.
19
+
20
+ ```sh
21
+ ollaya run clef
22
+ ```
23
+
24
+ ## What is in this repository
25
+
26
+ This repository holds only the files Ollaya derives, with no weights. Each graph is an ONNX export of the
27
+ original model whose weights **reference the authors' own weight files by byte offset**,
28
+ so `ollaya pull` downloads the weights from the upstream repositories, unmodified and pinned to a
29
+ commit, and verifies their sha256.
30
+
31
+ | Tag | Upstream | Files |
32
+ |---|---|---|
33
+ | `clef:flash` | [Cloudflare/clef-flash@17f0b0a](https://huggingface.co/Cloudflare/clef-flash/tree/17f0b0ad64efb65d273590632833508766b2aae6) | `flash/model-fp32.onnx`, `flash/decision.json`, `flash/calibration.json` |
34
+
35
+ Each tag has an fp32 graph, used on CPU and GPU. Each tag also has `decision.json` (sequence layout, special tokens) and
36
+ `calibration.json` (temperatures).
37
+
38
+ ## Parity
39
+
40
+ Ollaya's Rust runtime matches the authors' own code (joint_schema_model.py: their encoder, Qwen3.5 model and joint schema head, fp32) on 571 questions from 131 requests, on CUDA: identical token ids and spans, the same 13 rejected requests, the same decision on every question, logits within 4.3e-5 and probabilities within 6.3e-6.
41
+
42
+ ## License
43
+
44
+ Same as the upstream model (Apache-2.0). Ollaya itself is Apache-2.0.
flash/calibration.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "temperature": [
3
+ 1.0,
4
+ 1.0,
5
+ 1.0
6
+ ],
7
+ "temperature_by_options": {},
8
+ "source": "none: upstream answers with a plain softmax of the head's logits"
9
+ }
flash/decision.json ADDED
@@ -0,0 +1,100 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "engine": "onnx",
3
+ "family": "clef",
4
+ "layout": "clef-joint-v1",
5
+ "upstream": {
6
+ "repo": "Cloudflare/clef-flash",
7
+ "revision": "17f0b0ad64efb65d273590632833508766b2aae6",
8
+ "code": "joint_schema_model.py in the model repository (encode_record, JointSchemaHead)"
9
+ },
10
+ "contract": {
11
+ "inputs": {
12
+ "input_ids": {
13
+ "dtype": "int64",
14
+ "shape": [
15
+ 1,
16
+ "seq"
17
+ ],
18
+ "note": "the request's ids; seq a multiple of 64; right-pad with the pad id"
19
+ },
20
+ "token_positions": {
21
+ "dtype": "int64",
22
+ "shape": [
23
+ "tokens"
24
+ ],
25
+ "note": "0..tokens-1"
26
+ },
27
+ "question_spans": {
28
+ "dtype": "int64",
29
+ "shape": [
30
+ "questions",
31
+ 2
32
+ ],
33
+ "note": "instruction tokens [start, end) per question"
34
+ },
35
+ "question_types": {
36
+ "dtype": "int64",
37
+ "shape": [
38
+ "questions"
39
+ ],
40
+ "note": "type_index"
41
+ },
42
+ "option_spans": {
43
+ "dtype": "int64",
44
+ "shape": [
45
+ "options",
46
+ 2
47
+ ],
48
+ "note": "option tokens [start, end), every question's options in order"
49
+ },
50
+ "option_question": {
51
+ "dtype": "int64",
52
+ "shape": [
53
+ "options"
54
+ ],
55
+ "note": "the question index of each option"
56
+ }
57
+ },
58
+ "outputs": {
59
+ "logits": {
60
+ "dtype": "float32",
61
+ "shape": [
62
+ "options"
63
+ ],
64
+ "note": "raw option logits"
65
+ }
66
+ },
67
+ "seq_multiple": 64,
68
+ "positions": "0..seq-1, implicit",
69
+ "attention": "causal; no mask input"
70
+ },
71
+ "prompt": {
72
+ "prefix": "<|im_start|>system\nRead the complete state and schema. Decide every field jointly. Each answer must be exactly one of that field's allowed options.<|im_end|>\n<|im_start|>user\nSTATE:\n",
73
+ "schema": "\n\nSCHEMA FIELDS:\n",
74
+ "field": "\nFIELD {n}\nID: {id}\nTYPE: {type}\nINSTRUCTION: ",
75
+ "options": "\nALLOWED OPTIONS:\n",
76
+ "option": "OPTION {n}: ",
77
+ "option_end": "\n",
78
+ "field_end": "END FIELD\n",
79
+ "suffix": "\n<|im_end|>\n<|im_start|>assistant\n<think>\n\n</think>\n\nJOINT SCHEMA DECISIONS:"
80
+ },
81
+ "noul_criteria": {
82
+ "true": "The proposition is true or the answer is yes.",
83
+ "false": "The proposition is false or the answer is no."
84
+ },
85
+ "max_tokens": 4096,
86
+ "pad": 248044,
87
+ "type_index": {
88
+ "noul": 0,
89
+ "choice": 1,
90
+ "score": 2
91
+ },
92
+ "option_logits": {
93
+ "noul": "[true, false]",
94
+ "choice": "labels sorted by code point",
95
+ "score": "levels in order"
96
+ },
97
+ "opset": 20,
98
+ "precision": "fp32 compute; weights BF16 (widened by Cast)",
99
+ "weights_in_memory": "bf16"
100
+ }
flash/model-fp32.onnx ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5cc3f10d3ad4158c34411e2e29247d914fe6b70a2dc92b53458ee7bff6b62a57
3
+ size 11991785