Spaces:
Running on Zero
Running on Zero
Download app.py from tencent/HY-Embodied-0.5: direct link, hf CLI and curl.
- Browser
- Download file 28 kB
-
https://huggingface.co/spaces/tencent/HY-Embodied-0.5/resolve/main/app.py
- Command line
-
hf download hf://spaces/tencent/HY-Embodied-0.5/app.py
-
curl -L -o app.py https://huggingface.co/spaces/tencent/HY-Embodied-0.5/resolve/main/app.py
28 kB
| """ | |
| HY-Embodied-0.5 Chatbot — Hugging Face Space | |
| Based on the official repository: https://github.com/Tencent-Hunyuan/HY-Embodied | |
| Model weights: https://huggingface.co/tencent/HY-Embodied-0.5 | |
| UI layout follows the tencent/HunyuanOCR Space pattern (gr.Blocks + gr.Chatbot | |
| + gr.MultimodalTextbox + gr.State), which avoids the | |
| `TypeError: argument of type 'bool' is not iterable` in `gradio_client.utils` | |
| that is triggered by `ChatInterface(multimodal=True, additional_inputs=[...])` | |
| on current gradio releases. | |
| """ | |
| import copy | |
| import html | |
| import os | |
| import re | |
| import threading | |
| from typing import Any, Dict, List | |
| # ---------- Monkey-patch gradio_client schema parser ---------- | |
| # gradio_client.utils.get_type / _json_schema_to_python_type crash with | |
| # `TypeError: argument of type 'bool' is not iterable` | |
| # when a JSON Schema node is a bare `True` / `False` (valid per JSON Schema, | |
| # e.g. `additionalProperties: false`). This is a known upstream bug that | |
| # affects Gradio 4.x and 5.x. We patch the two entry points to treat bool / | |
| # non-dict nodes as empty schemas so `/info` can render. | |
| def _patch_gradio_client_schema_bug() -> None: | |
| try: | |
| from gradio_client import utils as _gc_utils | |
| except Exception: | |
| return | |
| _orig_get_type = _gc_utils.get_type | |
| _orig_json_schema = _gc_utils._json_schema_to_python_type | |
| def _safe_get_type(schema): | |
| if not isinstance(schema, dict): | |
| return "Any" | |
| return _orig_get_type(schema) | |
| def _safe_json_schema_to_python_type(schema, defs=None): | |
| if isinstance(schema, bool): | |
| return "Any" if schema else "None" | |
| if not isinstance(schema, dict): | |
| return "Any" | |
| try: | |
| return _orig_json_schema(schema, defs) | |
| except (TypeError, KeyError, RecursionError): | |
| return "Any" | |
| _gc_utils.get_type = _safe_get_type | |
| _gc_utils._json_schema_to_python_type = _safe_json_schema_to_python_type | |
| _patch_gradio_client_schema_bug() | |
| import gradio as gr | |
| # `spaces` only exists on Hugging Face ZeroGPU hardware. Provide a no-op | |
| # shim when running locally so the `@spaces.GPU(...)` decorator still works. | |
| try: | |
| import spaces # ZeroGPU | |
| except ImportError: | |
| class _SpacesShim: | |
| def GPU(*dargs, **dkwargs): | |
| def _decorator(fn): | |
| return fn | |
| # Allow both `@spaces.GPU` and `@spaces.GPU(duration=...)` | |
| if len(dargs) == 1 and callable(dargs[0]) and not dkwargs: | |
| return dargs[0] | |
| return _decorator | |
| spaces = _SpacesShim() # type: ignore | |
| import torch | |
| from PIL import Image | |
| from transformers import ( | |
| AutoModelForImageTextToText, | |
| AutoProcessor, | |
| TextIteratorStreamer, | |
| ) | |
| # ---------- Configuration ---------- | |
| MODEL_PATH = "tencent/HY-Embodied-0.5" | |
| DEVICE = "cuda" if torch.cuda.is_available() else "cpu" | |
| DTYPE = torch.bfloat16 | |
| DEFAULT_MAX_NEW_TOKENS = 8192 | |
| DEFAULT_TEMPERATURE = 0.6 | |
| DEFAULT_TOP_P = 0.9 | |
| MIN_MAX_NEW_TOKENS = 1024 | |
| MAX_MAX_NEW_TOKENS = 32768 | |
| # ---------- Model Loading ---------- | |
| print(f"Loading HY-Embodied processor from {MODEL_PATH} ...") | |
| processor = AutoProcessor.from_pretrained(MODEL_PATH) | |
| # Load chat template if bundled with the model repo | |
| try: | |
| from huggingface_hub import hf_hub_download | |
| chat_template_path = hf_hub_download( | |
| repo_id=MODEL_PATH, filename="chat_template.jinja" | |
| ) | |
| with open(chat_template_path, "r", encoding="utf-8") as f: | |
| processor.chat_template = f.read() | |
| print("Loaded custom chat_template.jinja") | |
| except Exception as e: | |
| print(f"No custom chat_template.jinja found, using default: {e}") | |
| print("Loading HY-Embodied model weights ...") | |
| model = AutoModelForImageTextToText.from_pretrained( | |
| MODEL_PATH, | |
| torch_dtype=DTYPE, | |
| ) | |
| model.to(DEVICE).eval() | |
| print(f"Model ready on {DEVICE}.") | |
| # ---------- Helpers ---------- | |
| IMAGE_EXTS = {".jpg", ".jpeg", ".png", ".bmp", ".webp", ".gif", ".tiff"} | |
| def _is_image(path: str) -> bool: | |
| if not isinstance(path, str): | |
| return False | |
| return os.path.splitext(path)[1].lower() in IMAGE_EXTS | |
| def _history_to_messages(task_history: List) -> List[Dict[str, Any]]: | |
| """ | |
| Convert internal `task_history` — a list of (user_turn, assistant_reply) | |
| tuples, where `user_turn` is either a str (text) or a (path,) tuple (image) — | |
| into HY-Embodied's structured chat-template format. | |
| Consecutive image-turns + a trailing text turn get merged into a single | |
| user message with multi-part content, matching how the model is typically | |
| prompted. | |
| """ | |
| messages: List[Dict[str, Any]] = [] | |
| pending_content: List[Dict[str, Any]] = [] | |
| for user_turn, assistant_reply in task_history: | |
| if isinstance(user_turn, (tuple, list)): | |
| img_path = user_turn[0] | |
| if _is_image(img_path): | |
| pending_content.append( | |
| {"type": "image", "image": Image.open(img_path).convert("RGB")} | |
| ) | |
| continue | |
| # Text turn — flush pending images + this text as one user message | |
| if isinstance(user_turn, str) and user_turn: | |
| pending_content.append({"type": "text", "text": user_turn}) | |
| if pending_content: | |
| messages.append({"role": "user", "content": pending_content}) | |
| pending_content = [] | |
| if assistant_reply: | |
| messages.append( | |
| { | |
| "role": "assistant", | |
| "content": [{"type": "text", "text": assistant_reply}], | |
| } | |
| ) | |
| # Trailing images without a text prompt — still send as a user turn | |
| if pending_content: | |
| messages.append({"role": "user", "content": pending_content}) | |
| return messages | |
| # ---------- Inference ---------- | |
| def _generate_stream( | |
| task_history: List, | |
| max_new_tokens: int, | |
| temperature: float, | |
| top_p: float, | |
| enable_thinking: bool, | |
| ): | |
| """Yield the incrementally-decoded model output for the current history.""" | |
| messages = _history_to_messages(task_history) | |
| if not messages: | |
| yield "Please enter a message or upload an image." | |
| return | |
| # ---- Log what the user actually sent to the model ---- | |
| # Walk the final structured messages so we count only what's being fed to | |
| # the model this turn (images merged from prior image-only turns included). | |
| num_images = 0 | |
| text_prompts: List[str] = [] | |
| for msg in messages: | |
| if msg.get("role") != "user": | |
| continue | |
| for part in msg.get("content", []): | |
| if part.get("type") == "image": | |
| num_images += 1 | |
| elif part.get("type") == "text": | |
| t = part.get("text", "") | |
| if t: | |
| text_prompts.append(t) | |
| latest_prompt = text_prompts[-1] if text_prompts else "(no text prompt)" | |
| print("=" * 60) | |
| print(f"[inference] images: {num_images} | thinking: {enable_thinking} | " | |
| f"max_new_tokens: {max_new_tokens} | temperature: {temperature} | top_p: {top_p}") | |
| print(f"[inference] user text prompt: {latest_prompt!r}") | |
| print("=" * 60, flush=True) | |
| inputs = processor.apply_chat_template( | |
| messages, | |
| tokenize=True, | |
| add_generation_prompt=True, | |
| return_dict=True, | |
| return_tensors="pt", | |
| enable_thinking=enable_thinking, | |
| ).to(model.device) | |
| streamer = TextIteratorStreamer( | |
| processor.tokenizer if hasattr(processor, "tokenizer") else processor, | |
| skip_prompt=True, | |
| skip_special_tokens=True, | |
| ) | |
| gen_kwargs = dict( | |
| **inputs, | |
| max_new_tokens=int(max_new_tokens), | |
| use_cache=True, | |
| temperature=float(temperature), | |
| top_p=float(top_p), | |
| do_sample=float(temperature) > 0, | |
| streamer=streamer, | |
| ) | |
| thread = threading.Thread(target=model.generate, kwargs=gen_kwargs) | |
| thread.start() | |
| partial = "" | |
| for new_text in streamer: | |
| partial += new_text | |
| yield partial | |
| thread.join() | |
| # ---------- Output formatting ---------- | |
| _ANSWER_OPEN_RE = re.compile(r"<answer>\s*", re.IGNORECASE) | |
| _ANSWER_CLOSE_RE = re.compile(r"\s*</answer>\s*", re.IGNORECASE) | |
| _THINK_OPEN_RE = re.compile(r"<think>", re.IGNORECASE) | |
| _THINK_CLOSE_RE = re.compile(r"</think>", re.IGNORECASE) | |
| _THINK_LEADING_OPEN_RE = re.compile(r"^\s*<think>\s*", re.IGNORECASE) | |
| # Matches a markdown fenced code block (```...```) or an inline code span (`...`). | |
| # We skip HTML-escaping inside these so user/model code examples still render. | |
| _CODE_SPAN_RE = re.compile( | |
| r"```.*?```|`[^`\n]+`", | |
| re.DOTALL, | |
| ) | |
| def _escape_html_outside_code(text: str) -> str: | |
| """ | |
| HTML-escape `<`, `>`, `&` everywhere EXCEPT inside markdown code fences / | |
| inline code spans. This is what makes tags like `<point>(x,y)</point>`, | |
| `<box>...</box>`, `<tool_call>...</tool_call>`, `<image_pad>`, and any | |
| other special marker the model emits render as literal text in the chat | |
| bubble — Gradio's markdown renderer would otherwise treat them as unknown | |
| HTML tags and silently drop them (or their contents). | |
| """ | |
| out: List[str] = [] | |
| cursor = 0 | |
| for m in _CODE_SPAN_RE.finditer(text): | |
| out.append(html.escape(text[cursor : m.start()], quote=False)) | |
| # Code regions pass through verbatim — markdown's own code renderer | |
| # already escapes their contents safely. | |
| out.append(m.group(0)) | |
| cursor = m.end() | |
| out.append(html.escape(text[cursor:], quote=False)) | |
| return "".join(out) | |
| def _format_for_chatbot(raw: str, enable_thinking: bool = False) -> str: | |
| """ | |
| Render raw model output safely in a Gradio chat bubble. | |
| Gradio's markdown renderer hands unknown HTML tags to the browser, which | |
| either drops them (most inline tags) or eats their contents (block tags | |
| like `<answer>`). The model can emit any of: `<think>...</think>`, | |
| `<answer>...</answer>`, `<point>(x,y)</point>`, `<box>...</box>`, | |
| `<tool_call>...</tool_call>`, `<image_pad>`, etc. Stripping each one by | |
| name is fragile — a new tag breaks the UI silently. | |
| Thinking-mode quirk: the Qwen-style chat template pre-fills `<think>\\n` | |
| into the assistant prompt before generation. Because the `TextIteratorStreamer` | |
| is configured with `skip_prompt=True`, the opening `<think>` token is never | |
| streamed back — we only see the thinking body and the closing `</think>`. | |
| So we cannot rely on a matched pair; instead: | |
| * If `</think>` is present in the raw text, treat everything before the | |
| first `</think>` as the think block (stripping a leading `<think>` if | |
| one happens to be there). | |
| * Otherwise, if a literal `<think>` appears, or we know thinking mode | |
| was enabled, treat the whole stream-in-flight as open-thinking. | |
| * Otherwise, render the text as a plain answer. | |
| The rest of the output (after unwrapping `<answer>...</answer>`) is | |
| HTML-escaped outside of markdown code spans so every other special tag | |
| renders as literal text. | |
| """ | |
| if not raw: | |
| return raw | |
| think_text: str = "" | |
| open_thinking_fragment: str = "" | |
| text: str | |
| close_match = _THINK_CLOSE_RE.search(raw) | |
| open_match = _THINK_OPEN_RE.search(raw) | |
| if close_match: | |
| # Closed thinking segment. Everything before `</think>` is reasoning, | |
| # everything after is the real reply. Any `<think>` inside the reasoning | |
| # body gets stripped — we don't want to re-surface it. | |
| before = raw[: close_match.start()] | |
| after = raw[close_match.end() :] | |
| before = _THINK_LEADING_OPEN_RE.sub("", before) | |
| # Also drop any stray `<think>` that might appear mid-reasoning. | |
| before = _THINK_OPEN_RE.sub("", before) | |
| think_text = before.strip() | |
| text = after | |
| elif open_match or enable_thinking: | |
| # Streaming in think-mode and we haven't hit `</think>` yet. Show the | |
| # partial thought as an open `<details>` so the user sees progress. | |
| if open_match: | |
| open_thinking_fragment = raw[open_match.end() :].strip() | |
| text = raw[: open_match.start()] | |
| else: | |
| # No literal `<think>` because the chat template pre-filled it — | |
| # the whole stream is still inside the thinking region. | |
| open_thinking_fragment = raw.strip() | |
| text = "" | |
| else: | |
| text = raw | |
| # Unwrap <answer>...</answer> — strip only the wrapper tags, keep contents. | |
| text = _ANSWER_OPEN_RE.sub("", text) | |
| text = _ANSWER_CLOSE_RE.sub("", text) | |
| text = text.strip() | |
| # Escape remaining angle brackets so tags like <point>, <box>, <tool_call>, | |
| # <image_pad>, etc. render verbatim instead of being eaten by the HTML | |
| # renderer. | |
| text = _escape_html_outside_code(text) | |
| # Re-inflate captured <think> blocks as collapsible sections. Their inner | |
| # text is also escaped (so any nested special tokens stay visible), but | |
| # the <details>/<summary> wrapper itself is our own HTML and stays raw. | |
| # NOTE: keeping <details> / </details> on their own lines with a blank line | |
| # around them is required by markdown-it (Gradio's renderer) — otherwise | |
| # the whole block is treated as a single raw-HTML region and the trailing | |
| # answer text fails to render. | |
| parts: List[str] = [] | |
| if think_text: | |
| parts.append( | |
| "<details>\n" | |
| "<summary>💭 Thinking</summary>\n\n" | |
| f"{_escape_html_outside_code(think_text)}\n\n" | |
| "</details>" | |
| ) | |
| if open_thinking_fragment: | |
| parts.append( | |
| "<details open>\n" | |
| "<summary>💭 Thinking…</summary>\n\n" | |
| f"{_escape_html_outside_code(open_thinking_fragment)}\n\n" | |
| "</details>" | |
| ) | |
| if text: | |
| parts.append(text) | |
| return "\n\n".join(parts).strip() | |
| # ---------- Gradio event handlers ---------- | |
| def _empty_mm_value(): | |
| """Value a `MultimodalTextbox` resets to — no text, no attached files.""" | |
| return gr.update(value={"text": "", "files": []}) | |
| def add_multimodal(chatbot, task_history, mm_input): | |
| """Flush any pending text + attached image files into the chat. | |
| `MultimodalTextbox` delivers its payload as a dict: | |
| {"text": "...", "files": ["/tmp/gradio/a.jpg", "/tmp/gradio/b.jpg"]} | |
| We turn each attached image into its own user image-turn (the chat template | |
| handler expects images as separate entries to match the model's | |
| multi-image prompt format) and then append the text turn. | |
| Text is HTML-escaped for display so tags like `<point>(x,y)</point>` the | |
| user typed render literally; the raw string is kept in `task_history` for | |
| the model. | |
| """ | |
| if mm_input is None: | |
| return chatbot, task_history, _empty_mm_value() | |
| # MultimodalTextbox can yield either a dict or, historically, a bare string. | |
| if isinstance(mm_input, dict): | |
| text = (mm_input.get("text") or "").strip() | |
| files = list(mm_input.get("files") or []) | |
| else: | |
| text = (mm_input or "").strip() | |
| files = [] | |
| if not text and not files: | |
| return chatbot, task_history, _empty_mm_value() | |
| for f in files: | |
| path = f.name if hasattr(f, "name") else f | |
| if not isinstance(path, str) or not _is_image(path): | |
| continue | |
| chatbot = chatbot + [((path,), None)] | |
| task_history = task_history + [((path,), None)] | |
| if text: | |
| chatbot = chatbot + [(_escape_html_outside_code(text), None)] | |
| task_history = task_history + [(text, None)] | |
| return chatbot, task_history, _empty_mm_value() | |
| def reset_state(chatbot, task_history): | |
| task_history.clear() | |
| chatbot.clear() | |
| if torch.cuda.is_available(): | |
| torch.cuda.empty_cache() | |
| return [] | |
| def predict(chatbot, task_history, max_new_tokens, temperature, top_p, enable_thinking): | |
| """Stream the assistant reply for the latest user turn.""" | |
| if not task_history: | |
| yield chatbot | |
| return | |
| # The last turn must currently be pending — a user text with assistant=None. | |
| last_user, last_reply = task_history[-1] | |
| if last_reply is not None or isinstance(last_user, (tuple, list)): | |
| # Nothing to generate (either already answered, or user only uploaded an | |
| # image without a question yet). | |
| yield chatbot | |
| return | |
| history_for_model = copy.deepcopy(task_history) | |
| # `last_user` is the raw text we feed the model with; the chatbot bubble | |
| # always holds an HTML-escaped version so tags like <point>(x,y)</point> | |
| # the user typed render literally instead of being eaten by the markdown | |
| # renderer. | |
| display_user = _escape_html_outside_code(last_user) if isinstance(last_user, str) else last_user | |
| partial = "" | |
| for partial in _generate_stream( | |
| history_for_model, | |
| max_new_tokens=max_new_tokens, | |
| temperature=temperature, | |
| top_p=top_p, | |
| enable_thinking=enable_thinking, | |
| ): | |
| chatbot[-1] = (display_user, _format_for_chatbot(partial, enable_thinking)) | |
| yield chatbot | |
| # Persist the raw model output in task_history so regenerate / future turns | |
| # see the unmodified assistant reply; the chatbot only ever shows the | |
| # formatted version. | |
| task_history[-1] = (last_user, partial) | |
| chatbot[-1] = (display_user, _format_for_chatbot(partial, enable_thinking)) | |
| yield chatbot | |
| def regenerate(chatbot, task_history, max_new_tokens, temperature, top_p, enable_thinking): | |
| """Re-run generation for the most recent answered turn.""" | |
| if not task_history: | |
| yield chatbot | |
| return | |
| last_user, last_reply = task_history[-1] | |
| if last_reply is None or isinstance(last_user, (tuple, list)): | |
| yield chatbot | |
| return | |
| task_history[-1] = (last_user, None) | |
| chatbot[-1] = ( | |
| _escape_html_outside_code(last_user) if isinstance(last_user, str) else last_user, | |
| None, | |
| ) | |
| yield from predict( | |
| chatbot, task_history, max_new_tokens, temperature, top_p, enable_thinking | |
| ) | |
| # ---------- Examples ---------- | |
| # Each entry: local image directory + relative image filenames + question + | |
| # the generation settings we want to apply when this example is loaded. Paths | |
| # that don't exist on disk are filtered out silently so the app still boots on | |
| # a fresh deployment where these files aren't bundled. | |
| # | |
| # We use a path relative to this file (rather than an absolute one) so the | |
| # examples keep working when the project is cloned into a Hugging Face Space | |
| # container or any other host where the original source tree isn't mounted. | |
| _EXAMPLE_ROOT = os.path.join(os.path.dirname(os.path.abspath(__file__)), "examples") | |
| _EXAMPLES_RAW: List[Dict[str, Any]] = [ | |
| { | |
| "dir": os.path.join(_EXAMPLE_ROOT, "0"), | |
| "images": ["1.jpg", "2.jpg", "3.jpg", "4.jpg"], | |
| "question": ( | |
| "Based on these four images (image 1, 2, 3, and 4) showing the red bottle from different viewpoints (front, left, back, and right)" | |
| ", with each camera aligned with room walls and partially capturing the surroundings: From the viewpoint presented in image 4," | |
| "what is to the left of the red bottle? " | |
| "A. Wall and door B. Gray-green tufted backrest C. Curtain D. TV and electric fan." | |
| ), | |
| "enable_thinking": True, | |
| "temperature": DEFAULT_TEMPERATURE, | |
| }, | |
| { | |
| "dir": os.path.join(_EXAMPLE_ROOT, "1"), | |
| "images": ["1.jpg"], | |
| "question": ( | |
| "Which of the arrows points at a motorcycle? " | |
| "Choices: A. Yellow. B. green. C. Purple. D. blue. " | |
| "Answer the question briefly." | |
| ), | |
| "enable_thinking": True, | |
| "temperature": DEFAULT_TEMPERATURE, | |
| }, | |
| { | |
| "dir": os.path.join(_EXAMPLE_ROOT, "2"), | |
| "images": ["0.jpg", "1.jpg", "2.jpg", "3.jpg"], | |
| "question": "以上图片展示了制作花篮的过程,请按照正确的先后顺序为图片排序。", | |
| "enable_thinking": True, | |
| "temperature": DEFAULT_TEMPERATURE, | |
| }, | |
| { | |
| "dir": os.path.join(_EXAMPLE_ROOT, "3"), | |
| "images": [ | |
| "0.jpg", "1.jpg", "2.jpg", "3.jpg", "4.jpg", "5.jpg", "6.jpg", "7.jpg", | |
| "8.jpg", "9.jpg", "10.jpg", "11.jpg", "12.jpg", "13.jpg", | |
| "14.jpg", "15.jpg", | |
| ], | |
| "question": ( | |
| "These are frames of a video. How many trash bin(s) are in this room? " | |
| "Please answer the question using a single word or phrase." | |
| ), | |
| "enable_thinking": True, | |
| "temperature": DEFAULT_TEMPERATURE, | |
| }, | |
| ] | |
| def _resolve_example(ex: Dict[str, Any]) -> Dict[str, Any]: | |
| """Expand image filenames to absolute paths.""" | |
| resolved = dict(ex) | |
| resolved["image_paths"] = [os.path.join(ex["dir"], f) for f in ex["images"]] | |
| return resolved | |
| EXAMPLES = [ | |
| _resolve_example(ex) | |
| for ex in _EXAMPLES_RAW | |
| if all(os.path.exists(os.path.join(ex["dir"], f)) for f in ex["images"]) | |
| ] | |
| if len(EXAMPLES) != len(_EXAMPLES_RAW): | |
| print( | |
| f"[examples] {len(_EXAMPLES_RAW) - len(EXAMPLES)} example(s) skipped — " | |
| "source images not found on this host." | |
| ) | |
| def _make_example_loader(ex: Dict[str, Any]): | |
| """Factory so each button captures its own `ex` (avoids the classic | |
| closure-in-loop bug). | |
| """ | |
| image_paths = ex["image_paths"] | |
| question = ex["question"] | |
| thinking = bool(ex.get("enable_thinking", False)) | |
| temp = float(ex.get("temperature", DEFAULT_TEMPERATURE)) | |
| def _load(): | |
| chatbot: List[Any] = [] | |
| task_history: List[Any] = [] | |
| for p in image_paths: | |
| chatbot.append(((p,), None)) | |
| task_history.append(((p,), None)) | |
| chatbot.append((_escape_html_outside_code(question), None)) | |
| task_history.append((question, None)) | |
| return ( | |
| chatbot, | |
| task_history, | |
| gr.update(value=thinking), | |
| gr.update(value=temp), | |
| _empty_mm_value(), | |
| ) | |
| return _load | |
| # ---------- UI ---------- | |
| TITLE = "# HY-Embodied-0.5 Chatbot" | |
| DESCRIPTION = """ | |
| Official demo for **[HY-Embodied-0.5](https://github.com/Tencent-Hunyuan/HY-Embodied)** — | |
| a vision-language foundation model for embodied AI and real-world robotic agents by | |
| **Tencent Robotics X / HY Vision Team**. | |
| - Upload an image with the button below, then type your question and hit Send | |
| - Supports the model's `<think>` / `<answer>` thinking mode | |
| - Coordinates normalized to (0, 1000) | |
| """ | |
| with gr.Blocks(title="HY-Embodied-0.5 Chatbot", fill_height=True) as demo: | |
| gr.Markdown(TITLE) | |
| gr.Markdown(DESCRIPTION) | |
| with gr.Accordion("Generation settings", open=False): | |
| max_new_tokens = gr.Slider( | |
| minimum=MIN_MAX_NEW_TOKENS, maximum=MAX_MAX_NEW_TOKENS, | |
| value=DEFAULT_MAX_NEW_TOKENS, step=16, | |
| label="Max new tokens", | |
| ) | |
| temperature = gr.Slider( | |
| minimum=0.0, maximum=1.5, value=DEFAULT_TEMPERATURE, step=0.05, | |
| label="Temperature (0 = greedy)", | |
| ) | |
| top_p = gr.Slider( | |
| minimum=0.1, maximum=1.0, value=DEFAULT_TOP_P, step=0.05, | |
| label="Top-p", | |
| ) | |
| enable_thinking = gr.Checkbox( | |
| value=False, label="Enable thinking mode (<think>...</think>)" | |
| ) | |
| # NOTE: `type="tuples"` is required here — Gradio 5+ defaults Chatbot to | |
| # `type="messages"` (OpenAI-style dicts), which silently drops our | |
| # (user, assistant) tuples and makes every reply render as empty. | |
| chatbot = gr.Chatbot( | |
| label="Dialogue", | |
| height=520, | |
| show_copy_button=True, | |
| type="tuples", | |
| ) | |
| task_history = gr.State([]) | |
| with gr.Row(): | |
| # MultimodalTextbox natively supports drag-and-drop and paste of images | |
| # alongside typed text, and emits them together on submit as | |
| # {"text": str, "files": [paths]}. Image attachments stay cached inside | |
| # the box (with thumbnails) until the user hits Send — they do not go | |
| # to the chatbot on upload. | |
| query = gr.MultimodalTextbox( | |
| show_label=False, | |
| placeholder=( | |
| "Type a message, drag / paste images here, then hit Send. " | |
| "Attachments stay in the box until you submit." | |
| ), | |
| file_types=["image"], | |
| file_count="multiple", | |
| lines=2, | |
| scale=6, | |
| ) | |
| submit_btn = gr.Button("Send", variant="primary", scale=1) | |
| regen_btn = gr.Button("Regenerate", scale=1) | |
| empty_btn = gr.Button("Clear", scale=1) | |
| gen_inputs = [max_new_tokens, temperature, top_p, enable_thinking] | |
| # ---- Examples ---- | |
| # One card per case: a gallery showing its images + a button whose label is | |
| # a short preview of the question. Clicking the button populates the chat, | |
| # sets thinking / temperature, then chains into `predict` so the model | |
| # immediately answers — users get a one-click demo. | |
| if EXAMPLES: | |
| gr.Markdown("### Examples — click to load & run") | |
| with gr.Row(): | |
| for ex in EXAMPLES: | |
| q_preview = ex["question"].strip().replace("\n", " ") | |
| if len(q_preview) > 80: | |
| q_preview = q_preview[:77] + "…" | |
| n_imgs = len(ex["image_paths"]) | |
| with gr.Column(scale=1, min_width=220): | |
| gr.Gallery( | |
| value=ex["image_paths"], | |
| label=f"{n_imgs} image{'s' if n_imgs > 1 else ''}", | |
| columns=min(n_imgs, 3), | |
| height=160, | |
| show_label=True, | |
| show_download_button=False, | |
| show_share_button=False, | |
| preview=False, | |
| allow_preview=True, | |
| interactive=False, | |
| ) | |
| ex_btn = gr.Button(q_preview, size="sm") | |
| ex_btn.click( | |
| _make_example_loader(ex), | |
| inputs=None, | |
| outputs=[chatbot, task_history, enable_thinking, temperature, query], | |
| ).then( | |
| predict, | |
| [chatbot, task_history] + gen_inputs, | |
| [chatbot], | |
| show_progress=True, | |
| ) | |
| submit_btn.click( | |
| add_multimodal, | |
| [chatbot, task_history, query], | |
| [chatbot, task_history, query], | |
| ).then( | |
| predict, | |
| [chatbot, task_history] + gen_inputs, | |
| [chatbot], | |
| show_progress=True, | |
| ) | |
| query.submit( | |
| add_multimodal, | |
| [chatbot, task_history, query], | |
| [chatbot, task_history, query], | |
| ).then( | |
| predict, | |
| [chatbot, task_history] + gen_inputs, | |
| [chatbot], | |
| show_progress=True, | |
| ) | |
| regen_btn.click( | |
| regenerate, | |
| [chatbot, task_history] + gen_inputs, | |
| [chatbot], | |
| show_progress=True, | |
| ) | |
| empty_btn.click(reset_state, [chatbot, task_history], [chatbot]) | |
| gr.Markdown( | |
| "---\n" | |
| "Model: [`tencent/HY-Embodied-0.5`](https://huggingface.co/tencent/HY-Embodied-0.5) · " | |
| "Code: [GitHub](https://github.com/Tencent-Hunyuan/HY-Embodied)" | |
| ) | |
| if __name__ == "__main__": | |
| # `server_name="0.0.0.0"` makes it reachable from the host when running in | |
| # a container. Locally you can also set `share=True` via env var to get a | |
| # public gradio.live URL. | |
| demo.queue(max_size=16).launch( | |
| server_name=os.environ.get("GRADIO_SERVER_NAME", "0.0.0.0"), | |
| server_port=int(os.environ.get("GRADIO_SERVER_PORT", "7860")), | |
| share=os.environ.get("GRADIO_SHARE", "").lower() in {"1", "true", "yes"}, | |
| show_api=False, | |
| ) | |