oplatek commited on
Commit
7afdccf
·
verified ·
1 Parent(s): 962e65c

ThinkingCap Qwen3.8-27B ZeroGPU chat demo

Browse files
README.md CHANGED
@@ -1,13 +1,18 @@
1
  ---
2
  title: ThinkingCap Qwen3.8 27B
3
- emoji: 🐨
4
- colorFrom: pink
5
- colorTo: pink
6
  sdk: gradio
7
  sdk_version: 6.28.0
8
- python_version: '3.12'
9
  app_file: app.py
10
- pinned: false
 
 
 
 
11
  ---
12
 
13
- Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference
 
 
 
1
  ---
2
  title: ThinkingCap Qwen3.8 27B
3
+ emoji: 🧢
4
+ colorFrom: blue
5
+ colorTo: indigo
6
  sdk: gradio
7
  sdk_version: 6.28.0
 
8
  app_file: app.py
9
+ python_version: "3.12"
10
+ startup_duration_timeout: 1h
11
+ short_description: Qwen3.8-27B reasoning with 37% fewer thinking tokens
12
+ models:
13
+ - bottlecapai/ThinkingCap-Qwen3.8-27B
14
  ---
15
 
16
+ Chat demo for [bottlecapai/ThinkingCap-Qwen3.8-27B](https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B),
17
+ a token-efficient reasoning fine-tune of Qwen3.8-27B. Runs bf16 transformers on ZeroGPU (xlarge).
18
+ Each reply shows how many tokens the model spent reasoning.
__pycache__/app.cpython-312.pyc ADDED
Binary file (12.8 kB). View file
 
__pycache__/chat_utils.cpython-312.pyc ADDED
Binary file (7.4 kB). View file
 
app.py ADDED
@@ -0,0 +1,221 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import spaces # must precede torch: it patches torch.cuda for ZeroGPU
2
+
3
+ import queue
4
+ import time
5
+ from threading import Thread
6
+
7
+ import gradio as gr
8
+ import torch
9
+ from transformers import AutoModelForImageTextToText, AutoProcessor
10
+ from transformers.generation.streamers import BaseStreamer
11
+
12
+ from chat_utils import notice, render_reply, split_ids, to_messages
13
+
14
+ MODEL_ID = "bottlecapai/ThinkingCap-Qwen3.8-27B"
15
+
16
+ # One ZeroGPU reservation. A long answer is generated as a chain of reservations, each
17
+ # re-prefilling prompt + tokens so far, so a small value keeps queue priority high and
18
+ # lets even signed-out visitors (2 min/day, charged 2x on xlarge) run one step.
19
+ GPU_DURATION = 60
20
+ # Headroom inside a reservation for worker start-up and the final yield; generation is
21
+ # cut by max_time before ZeroGPU would abort the worker mid-stream.
22
+ GPU_MARGIN = 15
23
+
24
+ EFFORT_CHOICES = [("xhigh (recommended)", "xhigh"), ("medium", "medium"), ("low", "low"),
25
+ ("off: no thinking", "off")]
26
+
27
+ processor = AutoProcessor.from_pretrained(MODEL_ID)
28
+ tokenizer = processor.tokenizer
29
+ # bf16 is ~54 GB, so it needs the full 96 GB card (size="xlarge" below).
30
+ model = AutoModelForImageTextToText.from_pretrained(
31
+ MODEL_ID, dtype=torch.bfloat16, device_map="cuda").eval()
32
+
33
+ EOS_IDS = set(model.generation_config.eos_token_id)
34
+ THINK_END_ID = tokenizer.convert_tokens_to_ids("</think>")
35
+
36
+
37
+ class IdStreamer(BaseStreamer):
38
+ """Hands generated token ids to the consumer thread; skips the prompt."""
39
+
40
+ def __init__(self):
41
+ self.queue = queue.Queue()
42
+ self.prompt_seen = False
43
+
44
+ def put(self, value):
45
+ if not self.prompt_seen:
46
+ self.prompt_seen = True
47
+ return
48
+ self.queue.put(value.flatten().tolist())
49
+
50
+ def end(self):
51
+ self.queue.put(None)
52
+
53
+ def __iter__(self):
54
+ while (ids := self.queue.get()) is not None:
55
+ yield ids
56
+
57
+
58
+ def extend_inputs(prompt, generated):
59
+ """Append already generated tokens to the processor output so generation resumes."""
60
+ prompt_ids = prompt["input_ids"]
61
+ extra = torch.tensor([generated], dtype=prompt_ids.dtype)
62
+ inputs = {}
63
+ for key, value in prompt.items():
64
+ if key == "input_ids":
65
+ value = torch.cat([value, extra], dim=1)
66
+ elif torch.is_tensor(value) and value.shape == prompt_ids.shape:
67
+ # Per-token side inputs: attention mask continues with 1s, type ids with 0s
68
+ # (generated tokens are text).
69
+ fill = 1 if key == "attention_mask" else 0
70
+ value = torch.cat([value, torch.full_like(extra, fill, dtype=value.dtype)], dim=1)
71
+ inputs[key] = value
72
+ return inputs
73
+
74
+
75
+ @spaces.GPU(duration=GPU_DURATION, size="xlarge")
76
+ def generate_step(prompt: dict, generated: list, max_new_tokens: int, sampling: dict):
77
+ """Generate up to one reservation's worth of tokens, yielding id lists as they arrive."""
78
+ started = time.monotonic()
79
+ inputs = {k: v.to("cuda") if torch.is_tensor(v) else v
80
+ for k, v in extend_inputs(prompt, generated).items()}
81
+ streamer = IdStreamer()
82
+ failure = []
83
+
84
+ def run():
85
+ try:
86
+ model.generate(**inputs, **sampling, streamer=streamer,
87
+ max_new_tokens=max_new_tokens,
88
+ max_time=GPU_DURATION - GPU_MARGIN - (time.monotonic() - started))
89
+ except Exception as exc: # surfaced below instead of hanging the stream
90
+ failure.append(exc)
91
+ streamer.end()
92
+
93
+ thread = Thread(target=run)
94
+ thread.start()
95
+ n = 0
96
+ for ids in streamer:
97
+ n += len(ids)
98
+ yield ids
99
+ thread.join()
100
+ elapsed = time.monotonic() - started
101
+ print(f"[step] context={inputs['input_ids'].shape[1]} new={n} "
102
+ f"{elapsed:.1f}s {n / max(elapsed, 1e-6):.1f} tok/s", flush=True)
103
+ if failure:
104
+ raise failure[0]
105
+
106
+
107
+ def respond(message: dict, history: list, reasoning_effort: str = "xhigh",
108
+ max_new_tokens: int = 8192, temperature: float = 1.0, top_p: float = 0.95,
109
+ top_k: int = 20):
110
+ """Chat with ThinkingCap-Qwen3.8-27B, a token-efficient reasoning model. Streams the reply.
111
+
112
+ Args:
113
+ message: The user turn: {"text": str, "files": [image or text file paths]}.
114
+ history: Prior conversation turns (Gradio messages format).
115
+ reasoning_effort: Thinking budget: "xhigh" (recommended), "medium", "low" or "off".
116
+ max_new_tokens: Cap on reasoning + answer tokens for this reply.
117
+ temperature: Sampling temperature; 0 means greedy decoding.
118
+ top_p: Nucleus sampling cutoff.
119
+ top_k: Top-k sampling cutoff.
120
+ """
121
+ try:
122
+ messages = to_messages(message, history)
123
+ except (ValueError, OSError) as exc:
124
+ raise gr.Error(str(exc))
125
+
126
+ thinking = reasoning_effort != "off"
127
+ template_kwargs = ({"reasoning_effort": reasoning_effort} if thinking
128
+ else {"enable_thinking": False})
129
+ prompt = dict(processor.apply_chat_template(
130
+ messages, add_generation_prompt=True, tokenize=True, return_dict=True,
131
+ return_tensors="pt", preserve_thinking=False, **template_kwargs))
132
+ sampling = (dict(do_sample=True, temperature=temperature, top_p=top_p, top_k=int(top_k))
133
+ if temperature > 0 else dict(do_sample=False))
134
+
135
+ generated, started, n_reasoning = [], time.monotonic(), 0
136
+
137
+ def render(done):
138
+ nonlocal n_reasoning
139
+ if not thinking:
140
+ return render_reply(None, tokenizer.decode(generated, skip_special_tokens=True),
141
+ 0, 0, done)
142
+ reasoning_ids, answer_ids = split_ids(generated, THINK_END_ID)
143
+ n_reasoning = len(reasoning_ids)
144
+ answer = (None if answer_ids is None
145
+ else tokenizer.decode(answer_ids, skip_special_tokens=True))
146
+ return render_reply(tokenizer.decode(reasoning_ids, skip_special_tokens=True),
147
+ answer, n_reasoning, time.monotonic() - started, done)
148
+
149
+ stopped_by = None
150
+ while len(generated) < max_new_tokens:
151
+ before = len(generated)
152
+ try:
153
+ for ids in generate_step(prompt, list(generated), max_new_tokens - before, sampling):
154
+ generated.extend(ids)
155
+ yield render(done=False)
156
+ except Exception as exc:
157
+ if not generated:
158
+ raise
159
+ stopped_by = str(exc) or type(exc).__name__ # e.g. visitor's GPU quota ran out
160
+ break
161
+ if len(generated) == before or generated[-1] in EOS_IDS:
162
+ break
163
+
164
+ out = render(done=True)
165
+ if stopped_by:
166
+ out.append(notice("⚠️ Stopped early",
167
+ f"{stopped_by}\n\nSign in to Hugging Face for a larger daily GPU "
168
+ "quota, or lower **Reasoning effort**."))
169
+ elif generated and generated[-1] not in EOS_IDS:
170
+ where = " while still reasoning" if thinking and len(out) == 1 else ""
171
+ out.append(notice("⚠️ Reply was cut off",
172
+ f"Hit the {max_new_tokens:,}-token limit{where}. Raise **Max new "
173
+ "tokens** or lower **Reasoning effort** under Settings."))
174
+ yield out
175
+
176
+
177
+ EXAMPLES = [[{"text": t, "files": []}] for t in (
178
+ "Find all real x such that √(x + 3) = x − 3.",
179
+ "A bat and a ball cost $1.10 in total. The bat costs $1.00 more than the ball. "
180
+ "How much does the ball cost?",
181
+ "Write a Python function that returns the longest palindromic substring of a string "
182
+ "in O(n²) time, with a short explanation.",
183
+ "Explain the difference between TCP and UDP to a new backend engineer, with one "
184
+ "example where each is the right choice.",
185
+ )]
186
+
187
+ DESCRIPTION = f"""\
188
+ # 🧢 ThinkingCap · Qwen3.8-27B
189
+ [**{MODEL_ID}**](https://huggingface.co/{MODEL_ID}) keeps Qwen3.8-27B's accuracy on hard and
190
+ agentic tasks while thinking **37% fewer reasoning tokens** on average. Each reply shows how many
191
+ tokens the model spent reasoning. Text and image input.
192
+ [Blog post](https://bottlecapai.com/post/thinkingcap-qwen3-8-27b/) ·
193
+ [FP8](https://huggingface.co/{MODEL_ID}-FP8) · [NVFP4](https://huggingface.co/{MODEL_ID}-NVFP4) ·
194
+ [GGUF](https://huggingface.co/{MODEL_ID}-GGUF) · by [BottleCap AI](https://www.bottlecapai.com/)
195
+
196
+ <sub>Runs bf16 on ZeroGPU and uses your daily Hugging Face GPU quota. Sign in for more.</sub>
197
+ """
198
+
199
+ with gr.Blocks(title="ThinkingCap Qwen3.8-27B", fill_height=True) as demo:
200
+ gr.Markdown(DESCRIPTION)
201
+ with gr.Accordion("Settings", open=False):
202
+ effort = gr.Dropdown(EFFORT_CHOICES, value="xhigh", label="Reasoning effort")
203
+ max_tokens = gr.Slider(256, 16384, value=8192, step=256, label="Max new tokens")
204
+ temperature = gr.Slider(0.0, 1.5, value=1.0, step=0.05, label="Temperature")
205
+ top_p = gr.Slider(0.05, 1.0, value=0.95, step=0.01, label="Top-p")
206
+ top_k = gr.Slider(1, 100, value=20, step=1, label="Top-k")
207
+ gr.ChatInterface(
208
+ fn=respond,
209
+ multimodal=True,
210
+ chatbot=gr.Chatbot(height="65vh", label="ThinkingCap", buttons=["copy"]),
211
+ textbox=gr.MultimodalTextbox(placeholder="Ask something hard, or attach an image…",
212
+ file_types=["image", "text"], show_label=False,
213
+ max_plain_text_length=100_000),
214
+ additional_inputs=[effort, max_tokens, temperature, top_p, top_k],
215
+ examples=EXAMPLES,
216
+ cache_examples=False,
217
+ concurrency_limit=8,
218
+ )
219
+
220
+ if __name__ == "__main__":
221
+ demo.launch(mcp_server=True, theme=gr.themes.Soft())
chat_utils.py ADDED
@@ -0,0 +1,135 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Gradio chat history <-> Qwen chat-template messages, and reply rendering.
2
+
3
+ Pure Python (no torch / gradio imports) so it is unit-testable without the Space runtime.
4
+ Rendered replies are plain dicts, which Gradio's Chatbot accepts as messages.
5
+ """
6
+ import os
7
+
8
+ IMAGE_SUFFIXES = {".jpg", ".jpeg", ".png", ".webp", ".gif", ".bmp", ".tif", ".tiff"}
9
+ TEXT_SUFFIXES = {".txt", ".md", ".rst", ".log", ".csv", ".tsv", ".json", ".jsonl", ".yaml",
10
+ ".yml", ".toml", ".xml", ".html", ".css", ".js", ".ts", ".py", ".c", ".h",
11
+ ".cpp", ".java", ".go", ".rs", ".rb", ".sh", ".sql", ".tex"}
12
+ MAX_FILE_CHARS = 30_000
13
+
14
+
15
+ def _file_path(item):
16
+ """Path of an attachment in any of the shapes Gradio uses, else None."""
17
+ if isinstance(item, (list, tuple)):
18
+ return item[0] if item else None
19
+ if isinstance(item, dict):
20
+ nested = item.get("file")
21
+ if isinstance(nested, dict):
22
+ return nested.get("path") or nested.get("url")
23
+ return item.get("path") or item.get("url")
24
+ return None
25
+
26
+
27
+ def file_part(path):
28
+ """Turn an attachment into an image or inlined-text part; raise ValueError otherwise."""
29
+ name = os.path.basename(path)
30
+ suffix = os.path.splitext(path)[1].lower()
31
+ if suffix in IMAGE_SUFFIXES:
32
+ return {"type": "image", "image": path}
33
+ if suffix in TEXT_SUFFIXES:
34
+ with open(path, encoding="utf-8", errors="replace") as fh:
35
+ body = fh.read(MAX_FILE_CHARS + 1)
36
+ if len(body) > MAX_FILE_CHARS:
37
+ body = body[:MAX_FILE_CHARS] + "\n[... truncated]"
38
+ # Gradio turns a long paste into pasted_text.txt; that is the message itself.
39
+ if name == "pasted_text.txt":
40
+ return {"type": "text", "text": body}
41
+ return {"type": "text", "text": f"--- {name} ---\n{body}\n--- end of {name} ---"}
42
+ raise ValueError(f"Unsupported attachment '{name}'. Attach an image or a text file.")
43
+
44
+
45
+ def content_parts(content):
46
+ """Normalise one turn's content (str, tuple, list of str/dict) into chat-template parts."""
47
+ # A tuple is Gradio's file-only shape: ("/tmp/a.png",).
48
+ if isinstance(content, tuple):
49
+ return [file_part(p) for p in content if p]
50
+ items = content if isinstance(content, list) else [content]
51
+ parts = []
52
+ for item in items:
53
+ if isinstance(item, str):
54
+ if item.strip():
55
+ parts.append({"type": "text", "text": item})
56
+ elif isinstance(item, dict) and item.get("type") == "text":
57
+ if (item.get("text") or "").strip():
58
+ parts.append({"type": "text", "text": item["text"]})
59
+ elif isinstance(item, dict) and item.get("type") == "image" and "image" in item:
60
+ parts.append(item)
61
+ elif (path := _file_path(item)):
62
+ parts.append(file_part(path))
63
+ return parts
64
+
65
+
66
+ def to_messages(message, history):
67
+ """Build chat-template messages from a MultimodalTextbox value and Chatbot history.
68
+
69
+ Assistant turns carrying a metadata title (reasoning bubbles, notices) are dropped:
70
+ Qwen's template expects earlier turns without their thinking.
71
+ """
72
+ messages = []
73
+ for turn in history or []:
74
+ role = turn.get("role") if isinstance(turn, dict) else None
75
+ if role not in ("user", "assistant"):
76
+ continue
77
+ if role == "assistant" and (turn.get("metadata") or {}).get("title"):
78
+ continue
79
+ parts = content_parts(turn.get("content"))
80
+ if not parts:
81
+ continue
82
+ # Consecutive same-role entries (Gradio stores a file and its caption separately)
83
+ # are one turn for the model.
84
+ if messages and messages[-1]["role"] == role:
85
+ messages[-1]["content"].extend(parts)
86
+ else:
87
+ messages.append({"role": role, "content": parts})
88
+
89
+ if isinstance(message, str):
90
+ message = {"text": message, "files": []}
91
+ parts = [file_part(p) for p in (_file_path(f) or f for f in message.get("files") or []) if p]
92
+ if (message.get("text") or "").strip():
93
+ parts.append({"type": "text", "text": message["text"]})
94
+ if not parts:
95
+ raise ValueError("Nothing to send: type a message or attach a file.")
96
+ if messages and messages[-1]["role"] == "user":
97
+ messages[-1]["content"].extend(parts)
98
+ else:
99
+ messages.append({"role": "user", "content": parts})
100
+ return messages
101
+
102
+
103
+ def split_ids(ids, think_end_id):
104
+ """Split generated ids at the first </think>: (reasoning_ids, answer_ids or None).
105
+
106
+ The prompt already opens <think>, so the model only ever emits the closing tag;
107
+ answer_ids is None while the model is still reasoning.
108
+ """
109
+ if think_end_id in ids:
110
+ i = ids.index(think_end_id)
111
+ return ids[:i], ids[i + 1:]
112
+ return ids, None
113
+
114
+
115
+ def render_reply(reasoning, answer, n_reasoning, seconds, done):
116
+ """Chatbot messages for one reply: a collapsible reasoning bubble, then the answer.
117
+
118
+ Pass reasoning=None when thinking is off.
119
+ """
120
+ out = []
121
+ if reasoning is not None:
122
+ still_thinking = answer is None and not done
123
+ title = f"Reasoning · {n_reasoning:,} tokens"
124
+ out.append({"role": "assistant", "content": reasoning.strip() or "…",
125
+ "metadata": {"title": title,
126
+ "status": "pending" if still_thinking else "done",
127
+ "duration": round(seconds, 1)}})
128
+ if answer is not None and answer.strip():
129
+ out.append({"role": "assistant", "content": answer.strip()})
130
+ return out
131
+
132
+
133
+ def notice(title, text):
134
+ """A titled assistant message; to_messages drops it from later turns' context."""
135
+ return {"role": "assistant", "content": text, "metadata": {"title": title}}
requirements.txt ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ # torchvision must match the torch ZeroGPU preinstalls; unpinned, pip pulls a newer
2
+ # torchvision that drags torch outside ZeroGPU's supported set.
3
+ torch==2.11.0
4
+ torchvision==0.26.0
5
+ transformers==5.17.0
6
+ accelerate==1.15.0
7
+ # Triton kernels for the Gated-DeltaNet layers (48 of 64); without them transformers
8
+ # falls back to a reference PyTorch path that is an order of magnitude slower.
9
+ flash-linear-attention==0.5.2