akhaliq HF Staff Claude commited on
Commit
73278e4
·
1 Parent(s): efc2be9

Fix streaming contract: annotate @app .api return type + match JS client submit() API

Browse files

The UI showed nothing while logs printed each yield because gradio.Server
derives an endpoint's output components from the function's *return type
annotation*. race_endpoint had no annotation, so it registered zero
outputs — the queue ran and logged every yielded dict, but the client
received nothing to render. Annotate -> dict (the yielded-value type,
not Generator[dict,...]) to declare one dict output; the queue still
iterates the generator and streams each dict as a data event.

Wire the frontend to the documented @gradio/client contract: submit()
returns an async iterator (not a promise); each msg is {type:'data',
data:[record]} for this single-output endpoint or {type:'status',...}
for queue stage. Subscribe with events:['data','status'] and surface
queue position while waiting on the GPU slice. normalizeRecord unwraps
the [dict] array shape that postprocess_data produces for one output.

ZeroGPU engagement is unchanged from the gr.Blocks version: @spaces.GPU
on the inner classify/generate functions is a no-op unless the Space
hardware is set to ZeroGPU (SPACES_ZERO_GPU=true), independent of
whether the caller is a .click() handler or an @app .api generator.

Co-Authored-By: Claude <noreply@anthropic.com>

Files changed (2) hide show
  1. app.py +9 -2
  2. index.html +25 -22
app.py CHANGED
@@ -262,12 +262,19 @@ app = Server()
262
 
263
 
264
  @app.api(name="race")
265
- def race_endpoint(example_name: str, custom_html: str):
266
  """Queued, GPU-managed streaming endpoint. Each yielded record is delivered
267
  to the Gradio client (JS or Python) as a separate SSE event, so the frontend
268
  sees pulpie finish, then dripper's tokens arrive live, then the verdict.
269
  The final record carries every field, so a non-streaming predict() call
270
- still converges to the complete result."""
 
 
 
 
 
 
 
271
  yield from race(example_name or "", custom_html or "")
272
 
273
 
 
262
 
263
 
264
  @app.api(name="race")
265
+ def race_endpoint(example_name: str, custom_html: str) -> dict:
266
  """Queued, GPU-managed streaming endpoint. Each yielded record is delivered
267
  to the Gradio client (JS or Python) as a separate SSE event, so the frontend
268
  sees pulpie finish, then dripper's tokens arrive live, then the verdict.
269
  The final record carries every field, so a non-streaming predict() call
270
+ still converges to the complete result.
271
+
272
+ The `-> dict` return annotation is load-bearing: gradio.Server derives the
273
+ endpoint's output components from the return type, and a generator without
274
+ one registers zero outputs — the queue runs (the logs show each yield) but
275
+ the client receives nothing to render. Annotating with the *yielded* type
276
+ (dict, not Generator[dict, ...]) declares one dict output while the queue
277
+ still iterates the generator and streams each dict as a data event."""
278
  yield from race(example_name or "", custom_html or "")
279
 
280
 
index.html CHANGED
@@ -401,7 +401,10 @@
401
  // ── Connect client ──
402
  let client = null;
403
  try {
404
- client = await Client.connect(window.location.origin);
 
 
 
405
  connPill.classList.remove("connecting");
406
  connText.textContent = "READY";
407
  } catch (e) {
@@ -422,29 +425,29 @@
422
  const params = { example_name: exampleName, custom_html: custom };
423
 
424
  try {
425
- // submit() returns a streaming Job; iterating delivers records as the
426
- // generator yields them (live tokens + running timers). If streaming is
427
- // unavailable on this client version, the job still resolves with the
428
- // final record — applyRecord is idempotent and converges either way.
429
- const job = client.submit("/race", params);
430
- const iter = (typeof job[Symbol.asyncIterator] === "function")
431
- ? job
432
- : (job && typeof job.then === "function" ? await job : null);
433
-
434
- if (iter && typeof iter[Symbol.asyncIterator] === "function") {
435
- for await (const msg of iter) {
436
- // msg may be { type, data } (gradio job) or the record itself
437
- const payload = (msg && msg.type === "data" && msg.data !== undefined) ? msg.data
438
- : (msg && msg.data !== undefined ? msg.data : msg);
439
- applyRecord(normalizeRecord(payload));
 
 
 
 
 
440
  }
441
- } else if (iter) {
442
- applyRecord(normalizeRecord(iter));
443
- } else if (job && job.data !== undefined) {
444
- applyRecord(normalizeRecord(job.data));
445
- } else {
446
- applyRecord(normalizeRecord(job));
447
  }
 
448
  if (!verdictEl.innerHTML) setProgress(100);
449
  } catch (e) {
450
  showError("Race failed: " + (e?.message || e));
 
401
  // ── Connect client ──
402
  let client = null;
403
  try {
404
+ // events: ["data","status"] — status fires queue-position / stage updates
405
+ // (pending → generating → complete), which we surface while waiting on the
406
+ // GPU slice. Data fires once per generator yield.
407
+ client = await Client.connect(window.location.origin, { events: ["data", "status"] });
408
  connPill.classList.remove("connecting");
409
  connText.textContent = "READY";
410
  } catch (e) {
 
425
  const params = { example_name: exampleName, custom_html: custom };
426
 
427
  try {
428
+ // Per the @gradio/client docs: submit() returns an async iterator (not a
429
+ // promise — do not await it). Each yielded msg has type "data" (a computed
430
+ // value — fires once per generator yield) or "status" (queue stage). For
431
+ // a single-output @app.api endpoint, msg.data is a one-element array
432
+ // [record]; normalizeRecord unwraps it.
433
+ const submission = client.submit("/race", params);
434
+ let sawData = false;
435
+ for await (const msg of submission) {
436
+ if (msg.type === "status") {
437
+ // surface queue position while waiting for the GPU slice
438
+ if (msg.stage === "pending" && msg.position != null) {
439
+ verdictEl.innerHTML = '<div class="verdict-bar" style="color:var(--soft)">QUEUED · POSITION ' + msg.position + '</div>';
440
+ } else if (msg.stage === "generating") {
441
+ verdictEl.innerHTML = "";
442
+ }
443
+ continue;
444
+ }
445
+ if (msg.type === "data") {
446
+ sawData = true;
447
+ applyRecord(normalizeRecord(msg.data));
448
  }
 
 
 
 
 
 
449
  }
450
+ if (!sawData) showError("Backend returned no data. Check the /race endpoint and Space logs.");
451
  if (!verdictEl.innerHTML) setProgress(100);
452
  } catch (e) {
453
  showError("Race failed: " + (e?.message || e));