# -*- coding: utf-8 -*- """ app.py ====== Gradio interface for the Blind Navigation Assistant. This file ONLY handles inference / the user interface. All model TRAINING happens separately in `train_model.py` — run that first (or copy an already-trained `best.pt` into place) before launching this app. What it does: 1. User uploads a photo (e.g. taken from a phone camera). 2. The trained YOLO classification model predicts the most likely scene/object label(s) for the image. 3. A Depth-Anything-V2 model estimates rough distance-to-obstacle. 4. A simple rule-based safety check flags common vehicle hazards. 5. Qwen2.5-VL turns all of that into a short, spoken-style navigation instruction. 6. gTTS converts the guidance text to speech, played back in-browser. Run locally: python app.py Deploy on Hugging Face Spaces: Just push this file + requirements.txt + your trained weights folder to a Space using the "Gradio" SDK. See README instructions below. """ import os import torch import spaces import gradio as gr from PIL import Image from gtts import gTTS from ultralytics import YOLO from transformers import pipeline, AutoProcessor, Qwen2_5_VLForConditionalGeneration from huggingface_hub import login # ── CONFIG ─────────────────────────────────────────────────────────────── YOLO_MODEL_PATH = os.environ.get( "YOLO_MODEL_PATH", "runs/classify/train/weights/best.pt" ) DEPTH_MODEL_NAME = "depth-anything/Depth-Anything-V2-Large-hf" QWEN_MODEL_NAME = "Qwen/Qwen2.5-VL-3B-Instruct" DANGEROUS_OBJECTS = {"car", "bus", "truck", "motorcycle", "bicycle"} DEVICE = "cuda" if torch.cuda.is_available() else "cpu" # ── HUGGING FACE AUTHENTICATION (NO HARDCODED TOKENS) ─────────────────── def authenticate_huggingface(): """ Reads a Hugging Face token from environment variables so no token is ever hardcoded in source. - Locally / Hugging Face Spaces: set a Repository Secret or env var named HF_TOKEN (Spaces -> Settings -> Repository secrets). - Google Colab: set it as a Colab secret named HF_TOKEN, or export it as an environment variable before running this file. """ token = os.environ.get("HF_TOKEN") if token: login(token=token) print("Hugging Face login successful.") else: print("No HF_TOKEN found in environment — continuing without login " "(fine for public models).") authenticate_huggingface() # ── MODEL LOADING (with error handling) ────────────────────────────────── def load_yolo_model(path: str) -> YOLO: if not os.path.exists(path): raise FileNotFoundError( f"YOLO model weights not found at '{path}'.\n" "Run train_model.py first, or set the YOLO_MODEL_PATH " "environment variable to point at your trained best.pt file." ) try: return YOLO(path) except Exception as e: raise RuntimeError(f"Failed to load YOLO model from '{path}': {e}") def load_depth_pipeline(): try: return pipeline( "depth-estimation", model=DEPTH_MODEL_NAME, device=0 if DEVICE == "cuda" else -1, ) except Exception as e: raise RuntimeError(f"Failed to load depth estimation model: {e}") def load_qwen_model(): try: processor = AutoProcessor.from_pretrained(QWEN_MODEL_NAME) model = Qwen2_5_VLForConditionalGeneration.from_pretrained( QWEN_MODEL_NAME, torch_dtype=torch.float16 if DEVICE == "cuda" else torch.float32, device_map="auto", ) model.eval() return processor, model except Exception as e: raise RuntimeError(f"Failed to load Qwen2.5-VL model: {e}") print("Loading models — this can take a minute the first time...") yolo_model = load_yolo_model(YOLO_MODEL_PATH) depth_pipe = load_depth_pipeline() qwen_processor, qwen_model = load_qwen_model() print("All models loaded. Ready to serve requests.") # ── CORE LOGIC (ported from the original notebook) ────────────────────── def predict_objects(image: Image.Image, top_k: int = 3): """Runs the trained YOLO classifier and returns its top-k predicted labels with confidence, since the trained model is a classifier (not a detector) — it labels the overall scene rather than boxing individual objects.""" result = yolo_model(image, verbose=False)[0] probs = result.probs top_indices = probs.top5[:top_k] labels = [(yolo_model.names[i], float(probs.data[i])) for i in top_indices] return labels def analyze_depth(image: Image.Image) -> str: out = depth_pipe(image) depth_map = out["predicted_depth"] mean_d, min_d, max_d = depth_map.mean().item(), depth_map.min().item(), depth_map.max().item() rng = max_d - min_d if max_d != min_d else 1.0 score = round(10 * (mean_d - min_d) / rng, 1) if score < 3: return f"Very close obstacle ahead (depth score {score}/10)" elif score < 6: return f"Obstacle at medium distance (depth score {score}/10)" else: return f"Path appears clear, objects far away (depth score {score}/10)" def check_safety(objs: list) -> str: found = [o for o in objs if o.lower() in DANGEROUS_OBJECTS] if found: return f"Caution: possible vehicle hazard detected ({', '.join(found)})" return "No immediate vehicle hazard detected" def generate_guidance(image: Image.Image, objs: list, depth_info: str, safety_note: str) -> str: from qwen_vl_utils import process_vision_info prompt = ( "You are a blind navigation assistant. " f"Detected objects: {', '.join(objs) if objs else 'none confidently detected'}. " f"Depth info: {depth_info}. " f"Safety: {safety_note}. " "Give short, clear navigation guidance in 2-3 sentences." ) messages = [{ "role": "user", "content": [ {"type": "image", "image": image}, {"type": "text", "text": prompt}, ], }] text = qwen_processor.apply_chat_template(messages, tokenize=False, add_generation_prompt=True) img_in, vid_in = process_vision_info(messages) inputs = qwen_processor(text=[text], images=img_in, videos=vid_in, return_tensors="pt").to(DEVICE) with torch.no_grad(): output_ids = qwen_model.generate(**inputs, max_new_tokens=150) input_len = inputs["input_ids"].shape[1] new_tokens = output_ids[:, input_len:] return qwen_processor.batch_decode(new_tokens, skip_special_tokens=True)[0].strip() def text_to_speech(text: str, out_path: str = "voice.mp3") -> str | None: try: tts = gTTS(text=text[:900], lang="en") tts.save(out_path) return out_path except Exception as e: print("Voice generation error:", e) return None # ── MAIN PIPELINE CALLED BY THE UI ─────────────────────────────────────── @spaces.GPU def run_pipeline(image: Image.Image): if image is None: return "Please upload an image.", "", "", None image = image.convert("RGB") try: preds = predict_objects(image) objs = [label for label, _ in preds] objects_text = "\n".join(f"- {label} ({conf * 100:.1f}% confidence)" for label, conf in preds) except Exception as e: return f"Object prediction failed: {e}", "", "", None try: depth_info = analyze_depth(image) except Exception as e: depth_info = f"Depth analysis failed: {e}" safety_note = check_safety(objs) depth_safety_text = f"{depth_info}\n{safety_note}" try: guidance = generate_guidance(image, objs, depth_info, safety_note) except Exception as e: guidance = f"Guidance generation failed: {e}" audio_path = text_to_speech(guidance) if not guidance.startswith("Guidance generation failed") else None return objects_text, depth_safety_text, guidance, audio_path # ── GRADIO INTERFACE ────────────────────────────────────────────────────── with gr.Blocks(title="Blind Navigation Assistant") as demo: gr.Markdown( """ # 🦯 Blind Navigation Assistant Upload a photo of the scene ahead. The app will identify likely objects/scene type, estimate obstacle distance, flag vehicle hazards, and speak short navigation guidance. """ ) with gr.Row(): with gr.Column(): image_input = gr.Image(type="pil", label="Upload Image") submit_btn = gr.Button("Analyze", variant="primary") with gr.Column(): objects_output = gr.Textbox(label="Detected / Predicted Objects", lines=4) depth_safety_output = gr.Textbox(label="Depth & Safety Analysis", lines=3) guidance_output = gr.Textbox(label="Navigation Guidance", lines=4) audio_output = gr.Audio(label="Voice Guidance", autoplay=True) submit_btn.click( fn=run_pipeline, inputs=image_input, outputs=[objects_output, depth_safety_output, guidance_output, audio_output], ) gr.Examples( examples=[], # add local sample image paths here if you have any, e.g. ["sample1.jpg"] inputs=image_input, label="Sample Images (optional)", ) if __name__ == "__main__": demo.launch()