akhaliq HF Staff GLM-5.2 commited on
Commit
69efd3e
·
1 Parent(s): 6b5cd6a

Remove image upload; switch to text-only coding assistant

Browse files

The model is agentic-coding focused and does not accept images. Drop the
image upload path from both the frontend and backend and use the text-only
serving recipe from the model card: AutoTokenizer + AutoModelForCausalLM
with tokenizer.apply_chat_template(..., tokenize=False).

Also keeps the terminal-style UI and Ornith logo header.

Co-Authored-By: GLM-5.2 <noreply@z.ai>

Files changed (2) hide show
  1. app.py +16 -50
  2. index.html +6 -53
app.py CHANGED
@@ -1,7 +1,7 @@
1
  """
2
  Ornith-1.0-9B — a gradio.Server chat app.
3
 
4
- This Space demonstrates `deepreinforce-ai/Ornith-1.0-9B`, a multimodal
5
  (qwen3.5-based) reasoning model, served through `gradio.Server`.
6
 
7
  `gradio.Server` extends FastAPI and adds Gradio's API engine on top
@@ -21,16 +21,14 @@ from threading import Thread
21
  from typing import Optional
22
 
23
  import torch
24
- from PIL import Image
25
  from transformers import (
26
- AutoModelForMultimodalLM,
27
- AutoProcessor,
28
  TextIteratorStreamer,
29
  )
30
 
31
  from fastapi.responses import HTMLResponse
32
  from gradio import Server
33
- from gradio.data_classes import FileData
34
 
35
  # `spaces` is only available on Hugging Face Spaces. Guard the import so the
36
  # file still parses/imports on a plain machine (the GPU decorator becomes a
@@ -74,8 +72,8 @@ THINK_CLOSE = None # resolved lazily from tokenizer config in _split_think
74
  # and call .to("cuda") once - the CUDA emulation handles that safely. Real
75
  # CUDA then becomes available inside the @spaces.GPU-decorated function.
76
  print(f"[ornith] loading {MODEL_ID} ...", flush=True)
77
- _processor = AutoProcessor.from_pretrained(MODEL_ID)
78
- _model = AutoModelForMultimodalLM.from_pretrained(
79
  MODEL_ID,
80
  torch_dtype=torch.bfloat16,
81
  low_cpu_mem_usage=True,
@@ -84,16 +82,6 @@ _model.to("cuda")
84
  _model.eval()
85
  print("[ornith] model ready on cuda", flush=True)
86
 
87
- _tokenizer = _processor.tokenizer if hasattr(_processor, "tokenizer") else _processor
88
-
89
-
90
- def _file_path(image):
91
- if image is None:
92
- return None
93
- if isinstance(image, dict):
94
- return image.get("path")
95
- return getattr(image, "path", None)
96
-
97
 
98
  def _resolve_markers():
99
  open_tok = None
@@ -127,27 +115,6 @@ def _split_think(text, open_tok, close_tok):
127
  return text.replace(open_tok or "", "").strip(), ""
128
 
129
 
130
- def _build_messages(message, image_path, history):
131
- messages = []
132
- for turn in history or []:
133
- role = turn.get("role")
134
- content = turn.get("content")
135
- if role in ("system", "user", "assistant") and content:
136
- messages.append({"role": role, "content": content})
137
-
138
- parts = []
139
- if image_path:
140
- try:
141
- img = Image.open(image_path).convert("RGB")
142
- parts.append({"type": "image", "image": img})
143
- except Exception:
144
- parts.append({"type": "image", "image": image_path})
145
- parts.append({"type": "text", "text": message or ""})
146
-
147
- messages.append({"role": "user", "content": parts})
148
- return messages
149
-
150
-
151
  # --------------------------------------------------------------------------- #
152
  # App + streaming endpoint
153
  # --------------------------------------------------------------------------- #
@@ -158,7 +125,6 @@ app = Server()
158
  @GPU(duration=300)
159
  def generate(
160
  message: str,
161
- image: Optional[FileData] = None,
162
  history: list = None,
163
  max_new_tokens: int = DEFAULT_MAX_NEW_TOKENS,
164
  temperature: float = DEFAULT_TEMPERATURE,
@@ -167,23 +133,23 @@ def generate(
167
  # {reasoning, answer, status, error}.
168
  if history is None:
169
  history = []
170
- image_path = _file_path(image)
171
  open_tok, close_tok = _resolve_markers()
172
 
173
  try:
174
- messages = _build_messages(message, image_path, history)
175
-
176
- inputs = _processor.apply_chat_template(
 
 
 
 
 
 
177
  messages,
 
178
  add_generation_prompt=True,
179
- tokenize=True,
180
- return_dict=True,
181
- return_tensors="pt",
182
  )
183
- inputs = {
184
- k: (v.to("cuda") if hasattr(v, "to") else v)
185
- for k, v in inputs.items()
186
- }
187
 
188
  streamer = TextIteratorStreamer(
189
  _tokenizer,
 
1
  """
2
  Ornith-1.0-9B — a gradio.Server chat app.
3
 
4
+ This Space demonstrates `deepreinforce-ai/Ornith-1.0-9B`, an agentic coding
5
  (qwen3.5-based) reasoning model, served through `gradio.Server`.
6
 
7
  `gradio.Server` extends FastAPI and adds Gradio's API engine on top
 
21
  from typing import Optional
22
 
23
  import torch
 
24
  from transformers import (
25
+ AutoModelForCausalLM,
26
+ AutoTokenizer,
27
  TextIteratorStreamer,
28
  )
29
 
30
  from fastapi.responses import HTMLResponse
31
  from gradio import Server
 
32
 
33
  # `spaces` is only available on Hugging Face Spaces. Guard the import so the
34
  # file still parses/imports on a plain machine (the GPU decorator becomes a
 
72
  # and call .to("cuda") once - the CUDA emulation handles that safely. Real
73
  # CUDA then becomes available inside the @spaces.GPU-decorated function.
74
  print(f"[ornith] loading {MODEL_ID} ...", flush=True)
75
+ _tokenizer = AutoTokenizer.from_pretrained(MODEL_ID)
76
+ _model = AutoModelForCausalLM.from_pretrained(
77
  MODEL_ID,
78
  torch_dtype=torch.bfloat16,
79
  low_cpu_mem_usage=True,
 
82
  _model.eval()
83
  print("[ornith] model ready on cuda", flush=True)
84
 
 
 
 
 
 
 
 
 
 
 
85
 
86
  def _resolve_markers():
87
  open_tok = None
 
115
  return text.replace(open_tok or "", "").strip(), ""
116
 
117
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
118
  # --------------------------------------------------------------------------- #
119
  # App + streaming endpoint
120
  # --------------------------------------------------------------------------- #
 
125
  @GPU(duration=300)
126
  def generate(
127
  message: str,
 
128
  history: list = None,
129
  max_new_tokens: int = DEFAULT_MAX_NEW_TOKENS,
130
  temperature: float = DEFAULT_TEMPERATURE,
 
133
  # {reasoning, answer, status, error}.
134
  if history is None:
135
  history = []
 
136
  open_tok, close_tok = _resolve_markers()
137
 
138
  try:
139
+ messages = []
140
+ for turn in history:
141
+ role = turn.get("role")
142
+ content = turn.get("content")
143
+ if role in ("system", "user", "assistant") and content:
144
+ messages.append({"role": role, "content": content})
145
+ messages.append({"role": "user", "content": message or ""})
146
+
147
+ prompt = _tokenizer.apply_chat_template(
148
  messages,
149
+ tokenize=False,
150
  add_generation_prompt=True,
 
 
 
151
  )
152
+ inputs = _tokenizer(prompt, return_tensors="pt").to("cuda")
 
 
 
153
 
154
  streamer = TextIteratorStreamer(
155
  _tokenizer,
index.html CHANGED
@@ -3,7 +3,7 @@
3
  <head>
4
  <meta charset="UTF-8" />
5
  <meta name="viewport" content="width=device-width, initial-scale=1.0" />
6
- <title>Ornith-1.0-9B · Claude Code</title>
7
  <style>
8
  :root {
9
  --bg: #0a0a0a;
@@ -65,7 +65,6 @@
65
  .content { flex: 1; min-width: 0; }
66
 
67
  .user-input { color: var(--text); white-space: pre-wrap; }
68
- .user-image { margin-top: 8px; max-width: 240px; border: 1px solid var(--border); border-radius: 4px; display: block; }
69
 
70
  .thinking { margin: 6px 0 10px; }
71
  .thinking summary {
@@ -113,15 +112,6 @@
113
  #settings label { display: flex; flex-direction: column; gap: 4px; }
114
  #settings input[type=range] { width: 150px; accent-color: var(--green); }
115
 
116
- .previews { display: flex; gap: 8px; flex-wrap: wrap; margin-bottom: 8px; }
117
- .previews .thumb { position: relative; }
118
- .previews .thumb img { height: 48px; border-radius: 4px; border: 1px solid var(--border); }
119
- .previews .thumb .x {
120
- position: absolute; top: -6px; right: -6px; cursor: pointer;
121
- background: var(--panel-2); border: 1px solid var(--border); border-radius: 50%;
122
- width: 16px; height: 16px; text-align: center; line-height: 14px; font-size: 11px;
123
- }
124
-
125
  .input-row { display: flex; align-items: flex-end; gap: 8px; }
126
  #text {
127
  flex: 1; resize: none; min-height: 42px; max-height: 160px;
@@ -157,7 +147,7 @@
157
  <div id="chat">
158
  <div class="welcome">
159
  <h2>Welcome to Ornith-1.0-9B</h2>
160
- <p>Agentic coding assistant. Ask me to write, debug, or explain code. You can also attach an image for multimodal questions.</p>
161
  </div>
162
  </div>
163
  <div class="status" id="status"></div>
@@ -168,32 +158,25 @@
168
  <label>Temperature <input type="range" id="temp" min="0.1" max="1.5" step="0.1" value="0.6" /><span id="temp-v">0.6</span></label>
169
  <label>Max tokens <input type="range" id="maxtok" min="256" max="8192" step="256" value="2048" /><span id="maxtok-v">2048</span></label>
170
  </div>
171
- <div class="previews" id="previews"></div>
172
  <div class="input-row">
173
- <button class="icon-btn" id="attach" title="Attach image">📎</button>
174
  <textarea id="text" placeholder="Ask Ornith to write code, explain, or debug…"></textarea>
175
  <button id="send">Send</button>
176
  <button id="stop">Stop</button>
177
  </div>
178
- <div class="hint">Shift+Enter for newline · Enter to send · Image-text-to-text supported</div>
179
  </div>
180
  </div>
181
 
182
- <input type="file" id="file" accept="image/*" hidden />
183
-
184
  <script src="https://cdn.jsdelivr.net/npm/marked/marked.min.js"></script>
185
  <script src="https://cdn.jsdelivr.net/npm/dompurify@3/dist/purify.min.js"></script>
186
  <script type="module">
187
- import { Client, handle_file } from "https://cdn.jsdelivr.net/npm/@gradio/client/dist/index.min.js";
188
 
189
  const chatEl = document.getElementById("chat");
190
  const statusEl = document.getElementById("status");
191
  const textEl = document.getElementById("text");
192
  const sendBtn = document.getElementById("send");
193
  const stopBtn = document.getElementById("stop");
194
- const attachBtn= document.getElementById("attach");
195
- const fileEl = document.getElementById("file");
196
- const prevEl = document.getElementById("previews");
197
  const setToggle= document.getElementById("settings-toggle");
198
  const settings = document.getElementById("settings");
199
  const tempEl = document.getElementById("temp");
@@ -202,7 +185,6 @@ const maxEl = document.getElementById("maxtok");
202
  const maxV = document.getElementById("maxtok-v");
203
 
204
  let conversation = [];
205
- let pendingImage = null;
206
  let client = null;
207
  let job = null;
208
 
@@ -215,26 +197,6 @@ setToggle.addEventListener("click", () => settings.classList.toggle("open"));
215
  tempEl.addEventListener("input", () => tempV.textContent = tempEl.value);
216
  maxEl.addEventListener("input", () => maxV.textContent = maxEl.value);
217
 
218
- attachBtn.addEventListener("click", () => fileEl.click());
219
- fileEl.addEventListener("change", () => {
220
- if (fileEl.files && fileEl.files[0]) {
221
- pendingImage = fileEl.files[0];
222
- renderPreviews();
223
- }
224
- fileEl.value = "";
225
- });
226
-
227
- function renderPreviews() {
228
- prevEl.innerHTML = "";
229
- if (!pendingImage) return;
230
- const url = URL.createObjectURL(pendingImage);
231
- const wrap = document.createElement("div");
232
- wrap.className = "thumb";
233
- wrap.innerHTML = `<img src="${url}" alt="attachment"/><div class="x" title="remove">×</div>`;
234
- wrap.querySelector(".x").addEventListener("click", () => { pendingImage = null; renderPreviews(); });
235
- prevEl.appendChild(wrap);
236
- }
237
-
238
  textEl.addEventListener("input", () => {
239
  textEl.style.height = "auto";
240
  textEl.style.height = Math.min(textEl.scrollHeight, 160) + "px";
@@ -273,19 +235,11 @@ function addTurn(role) {
273
 
274
  async function send() {
275
  const message = textEl.value.trim();
276
- if (!message && !pendingImage) return;
277
  if (job) return;
278
 
279
  const userContent = addTurn("user");
280
- let html = `<div class="user-input">${escapeHtml(message)}</div>`;
281
- if (pendingImage) {
282
- const url = URL.createObjectURL(pendingImage);
283
- html += `<img class="user-image" src="${url}" alt="attachment"/>`;
284
- }
285
- userContent.innerHTML = html;
286
-
287
- const sentImage = pendingImage;
288
- pendingImage = null; renderPreviews();
289
  textEl.value = ""; textEl.style.height = "auto";
290
 
291
  const history = conversation.slice();
@@ -301,7 +255,6 @@ async function send() {
301
  const c = await connect();
302
  const payload = {
303
  message,
304
- image: sentImage ? handle_file(sentImage) : null,
305
  history,
306
  max_new_tokens: parseInt(maxEl.value, 10),
307
  temperature: parseFloat(tempEl.value),
 
3
  <head>
4
  <meta charset="UTF-8" />
5
  <meta name="viewport" content="width=device-width, initial-scale=1.0" />
6
+ <title>Ornith-1.0-9B · Terminal</title>
7
  <style>
8
  :root {
9
  --bg: #0a0a0a;
 
65
  .content { flex: 1; min-width: 0; }
66
 
67
  .user-input { color: var(--text); white-space: pre-wrap; }
 
68
 
69
  .thinking { margin: 6px 0 10px; }
70
  .thinking summary {
 
112
  #settings label { display: flex; flex-direction: column; gap: 4px; }
113
  #settings input[type=range] { width: 150px; accent-color: var(--green); }
114
 
 
 
 
 
 
 
 
 
 
115
  .input-row { display: flex; align-items: flex-end; gap: 8px; }
116
  #text {
117
  flex: 1; resize: none; min-height: 42px; max-height: 160px;
 
147
  <div id="chat">
148
  <div class="welcome">
149
  <h2>Welcome to Ornith-1.0-9B</h2>
150
+ <p>Agentic coding assistant. Ask me to write, debug, or explain code.</p>
151
  </div>
152
  </div>
153
  <div class="status" id="status"></div>
 
158
  <label>Temperature <input type="range" id="temp" min="0.1" max="1.5" step="0.1" value="0.6" /><span id="temp-v">0.6</span></label>
159
  <label>Max tokens <input type="range" id="maxtok" min="256" max="8192" step="256" value="2048" /><span id="maxtok-v">2048</span></label>
160
  </div>
 
161
  <div class="input-row">
 
162
  <textarea id="text" placeholder="Ask Ornith to write code, explain, or debug…"></textarea>
163
  <button id="send">Send</button>
164
  <button id="stop">Stop</button>
165
  </div>
166
+ <div class="hint">Shift+Enter for newline · Enter to send</div>
167
  </div>
168
  </div>
169
 
 
 
170
  <script src="https://cdn.jsdelivr.net/npm/marked/marked.min.js"></script>
171
  <script src="https://cdn.jsdelivr.net/npm/dompurify@3/dist/purify.min.js"></script>
172
  <script type="module">
173
+ import { Client } from "https://cdn.jsdelivr.net/npm/@gradio/client/dist/index.min.js";
174
 
175
  const chatEl = document.getElementById("chat");
176
  const statusEl = document.getElementById("status");
177
  const textEl = document.getElementById("text");
178
  const sendBtn = document.getElementById("send");
179
  const stopBtn = document.getElementById("stop");
 
 
 
180
  const setToggle= document.getElementById("settings-toggle");
181
  const settings = document.getElementById("settings");
182
  const tempEl = document.getElementById("temp");
 
185
  const maxV = document.getElementById("maxtok-v");
186
 
187
  let conversation = [];
 
188
  let client = null;
189
  let job = null;
190
 
 
197
  tempEl.addEventListener("input", () => tempV.textContent = tempEl.value);
198
  maxEl.addEventListener("input", () => maxV.textContent = maxEl.value);
199
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
200
  textEl.addEventListener("input", () => {
201
  textEl.style.height = "auto";
202
  textEl.style.height = Math.min(textEl.scrollHeight, 160) + "px";
 
235
 
236
  async function send() {
237
  const message = textEl.value.trim();
238
+ if (!message) return;
239
  if (job) return;
240
 
241
  const userContent = addTurn("user");
242
+ userContent.innerHTML = `<div class="user-input">${escapeHtml(message)}</div>`;
 
 
 
 
 
 
 
 
243
  textEl.value = ""; textEl.style.height = "auto";
244
 
245
  const history = conversation.slice();
 
255
  const c = await connect();
256
  const payload = {
257
  message,
 
258
  history,
259
  max_new_tokens: parseInt(maxEl.value, 10),
260
  temperature: parseFloat(tempEl.value),