import gradio as gr from huggingface_hub import InferenceClient from huggingface_hub.errors import HfHubHTTPError import base64 from PIL import Image import io from typing import Optional # Inference always runs on the visitor's own Hugging Face account (Sign in with HF). There is # deliberately no Space-level token fallback: visitors must not spend the owner's quota. # Function to encode image to base64 def encode_image(image_path): if not image_path: print("No image path provided") return None try: print(f"Encoding image from path: {image_path}") # If it's already a PIL Image if isinstance(image_path, Image.Image): image = image_path else: # Try to open the image file image = Image.open(image_path) # Convert to RGB if image has an alpha channel (RGBA) if image.mode == 'RGBA': image = image.convert('RGB') # Encode to base64 buffered = io.BytesIO() image.save(buffered, format="JPEG") img_str = base64.b64encode(buffered.getvalue()).decode("utf-8") print("Image encoded successfully") return img_str except Exception as e: print(f"Error encoding image: {e}") return None def render_profile(profile: Optional[gr.OAuthProfile]) -> str: """Display the current authentication status in the UI.""" if profile is None: return "Not signed in. Use the button above to sign in with Hugging Face and run inference with your own account." display_name = getattr(profile, "name", None) or getattr(profile, "username", "Hugging Face user") return f"Signed in as **{display_name}**." def refresh_auth_info( profile: Optional[gr.OAuthProfile], oauth_token: Optional[gr.OAuthToken], ): """Capture OAuth credentials for downstream callbacks.""" return oauth_token, render_profile(profile) def respond( message, image_files, # Changed parameter name and structure history: list, # prior turns, already in OpenAI chat format system_message, max_tokens, temperature, top_p, frequency_penalty, seed, custom_model, model_search_term, selected_model, oauth_token: Optional[gr.OAuthToken] = None ): print(f"Received message: {message}") print(f"Received {len(image_files) if image_files else 0} images") print(f"History: {history}") print(f"System message: {system_message}") print(f"Max tokens: {max_tokens}, Temperature: {temperature}, Top-P: {top_p}") print(f"Frequency Penalty: {frequency_penalty}, Seed: {seed}") print(f"Selected model (custom_model): {custom_model}") print(f"Model search term: {model_search_term}") if oauth_token is None or not getattr(oauth_token, "token", None): raise gr.Error( "Sign in with your Hugging Face account (button at the top) to chat. " "Requests run on your own account and Inference Providers quota." ) print("Using OAuth token from signed-in user for inference.") client = InferenceClient(token=oauth_token.token) print("Hugging Face Inference Client initialized.") # Convert seed to None if -1 (meaning random) if seed == -1: seed = None # Create multimodal content if images are present if image_files and len(image_files) > 0: # Process the user message to include images user_content = [] # Add text part if there is any if message and message.strip(): user_content.append({ "type": "text", "text": message }) # Add image parts for img in image_files: if img is not None: # Get raw image data from path try: encoded_image = encode_image(img) if encoded_image: user_content.append({ "type": "image_url", "image_url": { "url": f"data:image/jpeg;base64,{encoded_image}" } }) except Exception as e: print(f"Error encoding image: {e}") else: # Text-only message user_content = message # Prepare messages in the format expected by the API messages = [{"role": "system", "content": system_message}] print("Initial messages array constructed.") # Add conversation history to the context messages.extend(history) # Append the latest user message messages.append({"role": "user", "content": user_content}) print(f"Latest user message appended (content type: {type(user_content)})") # Determine which model to use, prioritizing custom_model if provided model_to_use = custom_model.strip() if custom_model.strip() != "" else selected_model print(f"Model selected for inference: {model_to_use}") # Start with an empty string to build the response as tokens stream in response = "" print(f"Sending request to Hugging Face inference.") # Prepare parameters for the chat completion request parameters = { "max_tokens": max_tokens, "temperature": temperature, "top_p": top_p, "frequency_penalty": frequency_penalty, } if seed is not None: parameters["seed"] = seed # Use the InferenceClient for making the request try: # Create a generator for the streaming response stream = client.chat_completion( model=model_to_use, messages=messages, stream=True, **parameters ) print("Received tokens: ", end="", flush=True) # Process the streaming response for chunk in stream: if hasattr(chunk, 'choices') and len(chunk.choices) > 0: # Extract the content from the response if hasattr(chunk.choices[0], 'delta') and hasattr(chunk.choices[0].delta, 'content'): token_text = chunk.choices[0].delta.content if token_text: print(token_text, end="", flush=True) response += token_text yield response print() except HfHubHTTPError as e: status = getattr(e.response, "status_code", None) if status in (401, 403): raise gr.Error( ( "Failed to generate response: {}\n\n" "Your Hugging Face session must grant the **inference-api** (Make calls to Inference Providers) " "permission. Sign out, then sign back in and approve the requested scopes." ).format(e) ) from e print(f"Error during inference: {e}") response += f"\nError: {str(e)}" yield response except Exception as e: print(f"Error during inference: {e}") response += f"\nError: {str(e)}" yield response print("Completed response generation.") # GRADIO UI with gr.Blocks() as demo: with gr.Row(elem_id="oauth-row"): login_button = gr.LoginButton() auth_status = gr.Markdown(render_profile(None), elem_id="oauth-status") # Create the chatbot component chatbot = gr.Chatbot( height=600, buttons=["copy"], placeholder="Select a model and begin chatting. Now supports multimodal inputs.", layout="panel" ) print("Chatbot interface created.") # Multimodal textbox for messages (combines text and file uploads) msg = gr.MultimodalTextbox( placeholder="Type a message or upload images...", show_label=False, container=False, scale=12, file_types=["image"], file_count="multiple", sources=["upload"] ) # Create accordion for settings with gr.Accordion("Settings", open=False): # System message system_message_box = gr.Textbox( value="You are a helpful AI assistant that can understand images and text.", placeholder="You are a helpful assistant.", label="System Prompt" ) # Generation parameters with gr.Row(): with gr.Column(): max_tokens_slider = gr.Slider( minimum=1, maximum=4096, value=512, step=1, label="Max tokens" ) temperature_slider = gr.Slider( minimum=0.1, maximum=4.0, value=0.7, step=0.1, label="Temperature" ) top_p_slider = gr.Slider( minimum=0.1, maximum=1.0, value=0.95, step=0.05, label="Top-P" ) with gr.Column(): frequency_penalty_slider = gr.Slider( minimum=-2.0, maximum=2.0, value=0.0, step=0.1, label="Frequency Penalty" ) seed_slider = gr.Slider( minimum=-1, maximum=65535, value=-1, step=1, label="Seed (-1 for random)" ) # Custom model box custom_model_box = gr.Textbox( value="", label="Custom Model", info="(Optional) Provide a custom Hugging Face model path. Overrides any selected featured model.", placeholder="meta-llama/Llama-3.3-70B-Instruct" ) # Model search model_search_box = gr.Textbox( label="Filter Models", placeholder="Search for a featured model...", lines=1 ) # Featured models list # Updated to include multimodal models models_list = [ "Qwen/Qwen3-VL-30B-A3B-Instruct", "google/gemma-3-27b-it", "meta-llama/Llama-4-Scout-17B-16E-Instruct", "Qwen/Qwen2.5-VL-72B-Instruct", "meta-llama/Llama-3.3-70B-Instruct", "meta-llama/Llama-3.1-70B-Instruct", "meta-llama/Llama-3.2-3B-Instruct", "meta-llama/Llama-3.2-1B-Instruct", "meta-llama/Llama-3.1-8B-Instruct", "NousResearch/Hermes-3-Llama-3.1-8B", "Qwen/Qwen3-235B-A22B", "Qwen/Qwen3-32B", "Qwen/Qwen2.5-72B-Instruct", "Qwen/Qwen2.5-3B-Instruct", "Qwen/Qwen2.5-0.5B-Instruct", "Qwen/QwQ-32B", "Qwen/Qwen2.5-Coder-32B-Instruct", "microsoft/Phi-3-mini-4k-instruct", ] featured_model_radio = gr.Radio( label="Select a model below", choices=models_list, value="Qwen/Qwen3-VL-30B-A3B-Instruct", # Default to a multimodal model interactive=True ) gr.Markdown("[View all Text-to-Text models](https://huggingface.co/models?inference_provider=all&pipeline_tag=text-generation&sort=trending) | [View all multimodal models](https://huggingface.co/models?inference_provider=all&pipeline_tag=image-text-to-text&sort=trending)") # Chat history state chat_history = gr.State([]) oauth_token_state = gr.State(None) # Function to filter models def filter_models(search_term): print(f"Filtering models with search term: {search_term}") filtered = [m for m in models_list if search_term.lower() in m.lower()] print(f"Filtered models: {filtered}") return gr.update(choices=filtered) # Function to set custom model from radio def set_custom_model_from_radio(selected): print(f"Featured model selected: {selected}") return selected # Function for the chat interface def user(user_message, history): """Append the submitted text and images to the chat as user messages.""" history = list(history or []) if not user_message: return history text_content = (user_message.get("text") or "").strip() files = user_message.get("files") or [] if text_content: history.append({"role": "user", "content": text_content}) for file_path in files: path = file_path.get("path") if isinstance(file_path, dict) else file_path if path: history.append({"role": "user", "content": {"path": path}}) return history def _parts(content): """Gradio 6 message content -> (texts, image paths).""" items = content if isinstance(content, list) else [content] texts, images = [], [] for item in items: if isinstance(item, str): texts.append(item) elif isinstance(item, dict): if item.get("type") == "text": texts.append(item.get("text", "")) elif item.get("type") == "file" or "file" in item or "path" in item: f = item.get("file", item) path = f.get("path") if isinstance(f, dict) else None if path: images.append(path) return texts, images def _to_api(role, texts, images): text = "\n".join(t for t in texts if t) if role == "user" and images: content = [{"type": "text", "text": text}] if text else [] for img in images: encoded = encode_image(img) if encoded: content.append({"type": "image_url", "image_url": {"url": f"data:image/jpeg;base64,{encoded}"}}) return {"role": role, "content": content} return {"role": role, "content": text} # Define bot response function def bot( history, system_msg, max_tokens, temperature, top_p, freq_penalty, seed, custom_model, search_term, selected_model, oauth_token_obj: Optional[gr.OAuthToken] = None, ): history = list(history or []) if not history or history[-1].get("role") != "user": yield history return # The new turn = trailing user messages (text and/or images) after the last reply start = len(history) while start > 0 and history[start - 1].get("role") == "user": start -= 1 cur_texts, cur_images = [], [] for m in history[start:]: t, i = _parts(m["content"]) cur_texts += t cur_images += i # Earlier turns, merged per role, in OpenAI format prior = [] role, texts, images = None, [], [] for m in history[:start]: r = m.get("role") if r not in ("user", "assistant"): continue t, i = _parts(m["content"]) if r != role and role is not None: prior.append(_to_api(role, texts, images)) texts, images = [], [] role = r texts += t images += i if role is not None: prior.append(_to_api(role, texts, images)) history.append({"role": "assistant", "content": ""}) for response in respond( "\n".join(t for t in cur_texts if t), cur_images or None, prior, system_msg, max_tokens, temperature, top_p, freq_penalty, seed, custom_model, search_term, selected_model, oauth_token=oauth_token_obj, ): history[-1]["content"] = response yield history # Event handlers - only using the MultimodalTextbox's built-in submit functionality msg.submit( user, [msg, chatbot], [chatbot], queue=False ).then( bot, [chatbot, system_message_box, max_tokens_slider, temperature_slider, top_p_slider, frequency_penalty_slider, seed_slider, custom_model_box, model_search_box, featured_model_radio], [chatbot] ).then( lambda: {"text": "", "files": []}, # Clear inputs after submission None, [msg] ) # Connect the model filter to update the radio choices model_search_box.change( fn=filter_models, inputs=model_search_box, outputs=featured_model_radio ) print("Model search box change event linked.") # Connect the featured model radio to update the custom model box featured_model_radio.change( fn=set_custom_model_from_radio, inputs=featured_model_radio, outputs=custom_model_box ) print("Featured model radio button change event linked.") login_button.click( refresh_auth_info, inputs=None, outputs=[oauth_token_state, auth_status], ) demo.load(refresh_auth_info, inputs=None, outputs=[oauth_token_state, auth_status]) print("Gradio interface initialized.") if __name__ == "__main__": print("Launching the demo application.") demo.launch(mcp_server=True, theme="Nymbo/Nymbo_Theme")