Fix streaming contract: annotate @app .api return type + match JS client submit() API
Browse filesThe UI showed nothing while logs printed each yield because gradio.Server
derives an endpoint's output components from the function's *return type
annotation*. race_endpoint had no annotation, so it registered zero
outputs — the queue ran and logged every yielded dict, but the client
received nothing to render. Annotate -> dict (the yielded-value type,
not Generator[dict,...]) to declare one dict output; the queue still
iterates the generator and streams each dict as a data event.
Wire the frontend to the documented @gradio/client contract: submit()
returns an async iterator (not a promise); each msg is {type:'data',
data:[record]} for this single-output endpoint or {type:'status',...}
for queue stage. Subscribe with events:['data','status'] and surface
queue position while waiting on the GPU slice. normalizeRecord unwraps
the [dict] array shape that postprocess_data produces for one output.
ZeroGPU engagement is unchanged from the gr.Blocks version: @spaces.GPU
on the inner classify/generate functions is a no-op unless the Space
hardware is set to ZeroGPU (SPACES_ZERO_GPU=true), independent of
whether the caller is a .click() handler or an @app .api generator.
Co-Authored-By: Claude <noreply@anthropic.com>
- app.py +9 -2
- index.html +25 -22
|
@@ -262,12 +262,19 @@ app = Server()
|
|
| 262 |
|
| 263 |
|
| 264 |
@app.api(name="race")
|
| 265 |
-
def race_endpoint(example_name: str, custom_html: str):
|
| 266 |
"""Queued, GPU-managed streaming endpoint. Each yielded record is delivered
|
| 267 |
to the Gradio client (JS or Python) as a separate SSE event, so the frontend
|
| 268 |
sees pulpie finish, then dripper's tokens arrive live, then the verdict.
|
| 269 |
The final record carries every field, so a non-streaming predict() call
|
| 270 |
-
still converges to the complete result.
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 271 |
yield from race(example_name or "", custom_html or "")
|
| 272 |
|
| 273 |
|
|
|
|
| 262 |
|
| 263 |
|
| 264 |
@app.api(name="race")
|
| 265 |
+
def race_endpoint(example_name: str, custom_html: str) -> dict:
|
| 266 |
"""Queued, GPU-managed streaming endpoint. Each yielded record is delivered
|
| 267 |
to the Gradio client (JS or Python) as a separate SSE event, so the frontend
|
| 268 |
sees pulpie finish, then dripper's tokens arrive live, then the verdict.
|
| 269 |
The final record carries every field, so a non-streaming predict() call
|
| 270 |
+
still converges to the complete result.
|
| 271 |
+
|
| 272 |
+
The `-> dict` return annotation is load-bearing: gradio.Server derives the
|
| 273 |
+
endpoint's output components from the return type, and a generator without
|
| 274 |
+
one registers zero outputs — the queue runs (the logs show each yield) but
|
| 275 |
+
the client receives nothing to render. Annotating with the *yielded* type
|
| 276 |
+
(dict, not Generator[dict, ...]) declares one dict output while the queue
|
| 277 |
+
still iterates the generator and streams each dict as a data event."""
|
| 278 |
yield from race(example_name or "", custom_html or "")
|
| 279 |
|
| 280 |
|
|
@@ -401,7 +401,10 @@
|
|
| 401 |
// ── Connect client ──
|
| 402 |
let client = null;
|
| 403 |
try {
|
| 404 |
-
|
|
|
|
|
|
|
|
|
|
| 405 |
connPill.classList.remove("connecting");
|
| 406 |
connText.textContent = "READY";
|
| 407 |
} catch (e) {
|
|
@@ -422,29 +425,29 @@
|
|
| 422 |
const params = { example_name: exampleName, custom_html: custom };
|
| 423 |
|
| 424 |
try {
|
| 425 |
-
// submit() returns
|
| 426 |
-
//
|
| 427 |
-
//
|
| 428 |
-
//
|
| 429 |
-
|
| 430 |
-
const
|
| 431 |
-
|
| 432 |
-
|
| 433 |
-
|
| 434 |
-
|
| 435 |
-
|
| 436 |
-
|
| 437 |
-
|
| 438 |
-
|
| 439 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 440 |
}
|
| 441 |
-
} else if (iter) {
|
| 442 |
-
applyRecord(normalizeRecord(iter));
|
| 443 |
-
} else if (job && job.data !== undefined) {
|
| 444 |
-
applyRecord(normalizeRecord(job.data));
|
| 445 |
-
} else {
|
| 446 |
-
applyRecord(normalizeRecord(job));
|
| 447 |
}
|
|
|
|
| 448 |
if (!verdictEl.innerHTML) setProgress(100);
|
| 449 |
} catch (e) {
|
| 450 |
showError("Race failed: " + (e?.message || e));
|
|
|
|
| 401 |
// ── Connect client ──
|
| 402 |
let client = null;
|
| 403 |
try {
|
| 404 |
+
// events: ["data","status"] — status fires queue-position / stage updates
|
| 405 |
+
// (pending → generating → complete), which we surface while waiting on the
|
| 406 |
+
// GPU slice. Data fires once per generator yield.
|
| 407 |
+
client = await Client.connect(window.location.origin, { events: ["data", "status"] });
|
| 408 |
connPill.classList.remove("connecting");
|
| 409 |
connText.textContent = "READY";
|
| 410 |
} catch (e) {
|
|
|
|
| 425 |
const params = { example_name: exampleName, custom_html: custom };
|
| 426 |
|
| 427 |
try {
|
| 428 |
+
// Per the @gradio/client docs: submit() returns an async iterator (not a
|
| 429 |
+
// promise — do not await it). Each yielded msg has type "data" (a computed
|
| 430 |
+
// value — fires once per generator yield) or "status" (queue stage). For
|
| 431 |
+
// a single-output @app.api endpoint, msg.data is a one-element array
|
| 432 |
+
// [record]; normalizeRecord unwraps it.
|
| 433 |
+
const submission = client.submit("/race", params);
|
| 434 |
+
let sawData = false;
|
| 435 |
+
for await (const msg of submission) {
|
| 436 |
+
if (msg.type === "status") {
|
| 437 |
+
// surface queue position while waiting for the GPU slice
|
| 438 |
+
if (msg.stage === "pending" && msg.position != null) {
|
| 439 |
+
verdictEl.innerHTML = '<div class="verdict-bar" style="color:var(--soft)">QUEUED · POSITION ' + msg.position + '</div>';
|
| 440 |
+
} else if (msg.stage === "generating") {
|
| 441 |
+
verdictEl.innerHTML = "";
|
| 442 |
+
}
|
| 443 |
+
continue;
|
| 444 |
+
}
|
| 445 |
+
if (msg.type === "data") {
|
| 446 |
+
sawData = true;
|
| 447 |
+
applyRecord(normalizeRecord(msg.data));
|
| 448 |
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 449 |
}
|
| 450 |
+
if (!sawData) showError("Backend returned no data. Check the /race endpoint and Space logs.");
|
| 451 |
if (!verdictEl.innerHTML) setProgress(100);
|
| 452 |
} catch (e) {
|
| 453 |
showError("Race failed: " + (e?.message || e));
|