Spaces:
Running on Zero
Running on Zero
Download app.py from BAI-23F-116/blind-navigation-assistant: direct link, hf CLI and curl.
- Browser
- Download file 9.8 kB
-
https://huggingface.co/spaces/BAI-23F-116/blind-navigation-assistant/resolve/main/app.py
- Command line
-
hf download hf://spaces/BAI-23F-116/blind-navigation-assistant/app.py
-
curl -L -o app.py https://huggingface.co/spaces/BAI-23F-116/blind-navigation-assistant/resolve/main/app.py
9.8 kB
| # -*- coding: utf-8 -*- | |
| """ | |
| app.py | |
| ====== | |
| Gradio interface for the Blind Navigation Assistant. | |
| This file ONLY handles inference / the user interface. All model | |
| TRAINING happens separately in `train_model.py` β run that first (or | |
| copy an already-trained `best.pt` into place) before launching this app. | |
| What it does: | |
| 1. User uploads a photo (e.g. taken from a phone camera). | |
| 2. The trained YOLO classification model predicts the most likely | |
| scene/object label(s) for the image. | |
| 3. A Depth-Anything-V2 model estimates rough distance-to-obstacle. | |
| 4. A simple rule-based safety check flags common vehicle hazards. | |
| 5. Qwen2.5-VL turns all of that into a short, spoken-style navigation | |
| instruction. | |
| 6. gTTS converts the guidance text to speech, played back in-browser. | |
| Run locally: | |
| python app.py | |
| Deploy on Hugging Face Spaces: | |
| Just push this file + requirements.txt + your trained weights folder | |
| to a Space using the "Gradio" SDK. See README instructions below. | |
| """ | |
| import os | |
| import torch | |
| import spaces | |
| import gradio as gr | |
| from PIL import Image | |
| from gtts import gTTS | |
| from ultralytics import YOLO | |
| from transformers import pipeline, AutoProcessor, Qwen2_5_VLForConditionalGeneration | |
| from huggingface_hub import login | |
| # ββ CONFIG βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| YOLO_MODEL_PATH = os.environ.get( | |
| "YOLO_MODEL_PATH", "runs/classify/train/weights/best.pt" | |
| ) | |
| DEPTH_MODEL_NAME = "depth-anything/Depth-Anything-V2-Large-hf" | |
| QWEN_MODEL_NAME = "Qwen/Qwen2.5-VL-3B-Instruct" | |
| DANGEROUS_OBJECTS = {"car", "bus", "truck", "motorcycle", "bicycle"} | |
| DEVICE = "cuda" if torch.cuda.is_available() else "cpu" | |
| # ββ HUGGING FACE AUTHENTICATION (NO HARDCODED TOKENS) βββββββββββββββββββ | |
| def authenticate_huggingface(): | |
| """ | |
| Reads a Hugging Face token from environment variables so no token is | |
| ever hardcoded in source. | |
| - Locally / Hugging Face Spaces: set a Repository Secret or env var | |
| named HF_TOKEN (Spaces -> Settings -> Repository secrets). | |
| - Google Colab: set it as a Colab secret named HF_TOKEN, or export | |
| it as an environment variable before running this file. | |
| """ | |
| token = os.environ.get("HF_TOKEN") | |
| if token: | |
| login(token=token) | |
| print("Hugging Face login successful.") | |
| else: | |
| print("No HF_TOKEN found in environment β continuing without login " | |
| "(fine for public models).") | |
| authenticate_huggingface() | |
| # ββ MODEL LOADING (with error handling) ββββββββββββββββββββββββββββββββββ | |
| def load_yolo_model(path: str) -> YOLO: | |
| if not os.path.exists(path): | |
| raise FileNotFoundError( | |
| f"YOLO model weights not found at '{path}'.\n" | |
| "Run train_model.py first, or set the YOLO_MODEL_PATH " | |
| "environment variable to point at your trained best.pt file." | |
| ) | |
| try: | |
| return YOLO(path) | |
| except Exception as e: | |
| raise RuntimeError(f"Failed to load YOLO model from '{path}': {e}") | |
| def load_depth_pipeline(): | |
| try: | |
| return pipeline( | |
| "depth-estimation", | |
| model=DEPTH_MODEL_NAME, | |
| device=0 if DEVICE == "cuda" else -1, | |
| ) | |
| except Exception as e: | |
| raise RuntimeError(f"Failed to load depth estimation model: {e}") | |
| def load_qwen_model(): | |
| try: | |
| processor = AutoProcessor.from_pretrained(QWEN_MODEL_NAME) | |
| model = Qwen2_5_VLForConditionalGeneration.from_pretrained( | |
| QWEN_MODEL_NAME, | |
| torch_dtype=torch.float16 if DEVICE == "cuda" else torch.float32, | |
| device_map="auto", | |
| ) | |
| model.eval() | |
| return processor, model | |
| except Exception as e: | |
| raise RuntimeError(f"Failed to load Qwen2.5-VL model: {e}") | |
| print("Loading models β this can take a minute the first time...") | |
| yolo_model = load_yolo_model(YOLO_MODEL_PATH) | |
| depth_pipe = load_depth_pipeline() | |
| qwen_processor, qwen_model = load_qwen_model() | |
| print("All models loaded. Ready to serve requests.") | |
| # ββ CORE LOGIC (ported from the original notebook) ββββββββββββββββββββββ | |
| def predict_objects(image: Image.Image, top_k: int = 3): | |
| """Runs the trained YOLO classifier and returns its top-k predicted | |
| labels with confidence, since the trained model is a classifier | |
| (not a detector) β it labels the overall scene rather than boxing | |
| individual objects.""" | |
| result = yolo_model(image, verbose=False)[0] | |
| probs = result.probs | |
| top_indices = probs.top5[:top_k] | |
| labels = [(yolo_model.names[i], float(probs.data[i])) for i in top_indices] | |
| return labels | |
| def analyze_depth(image: Image.Image) -> str: | |
| out = depth_pipe(image) | |
| depth_map = out["predicted_depth"] | |
| mean_d, min_d, max_d = depth_map.mean().item(), depth_map.min().item(), depth_map.max().item() | |
| rng = max_d - min_d if max_d != min_d else 1.0 | |
| score = round(10 * (mean_d - min_d) / rng, 1) | |
| if score < 3: | |
| return f"Very close obstacle ahead (depth score {score}/10)" | |
| elif score < 6: | |
| return f"Obstacle at medium distance (depth score {score}/10)" | |
| else: | |
| return f"Path appears clear, objects far away (depth score {score}/10)" | |
| def check_safety(objs: list) -> str: | |
| found = [o for o in objs if o.lower() in DANGEROUS_OBJECTS] | |
| if found: | |
| return f"Caution: possible vehicle hazard detected ({', '.join(found)})" | |
| return "No immediate vehicle hazard detected" | |
| def generate_guidance(image: Image.Image, objs: list, depth_info: str, safety_note: str) -> str: | |
| from qwen_vl_utils import process_vision_info | |
| prompt = ( | |
| "You are a blind navigation assistant. " | |
| f"Detected objects: {', '.join(objs) if objs else 'none confidently detected'}. " | |
| f"Depth info: {depth_info}. " | |
| f"Safety: {safety_note}. " | |
| "Give short, clear navigation guidance in 2-3 sentences." | |
| ) | |
| messages = [{ | |
| "role": "user", | |
| "content": [ | |
| {"type": "image", "image": image}, | |
| {"type": "text", "text": prompt}, | |
| ], | |
| }] | |
| text = qwen_processor.apply_chat_template(messages, tokenize=False, add_generation_prompt=True) | |
| img_in, vid_in = process_vision_info(messages) | |
| inputs = qwen_processor(text=[text], images=img_in, videos=vid_in, return_tensors="pt").to(DEVICE) | |
| with torch.no_grad(): | |
| output_ids = qwen_model.generate(**inputs, max_new_tokens=150) | |
| input_len = inputs["input_ids"].shape[1] | |
| new_tokens = output_ids[:, input_len:] | |
| return qwen_processor.batch_decode(new_tokens, skip_special_tokens=True)[0].strip() | |
| def text_to_speech(text: str, out_path: str = "voice.mp3") -> str | None: | |
| try: | |
| tts = gTTS(text=text[:900], lang="en") | |
| tts.save(out_path) | |
| return out_path | |
| except Exception as e: | |
| print("Voice generation error:", e) | |
| return None | |
| # ββ MAIN PIPELINE CALLED BY THE UI βββββββββββββββββββββββββββββββββββββββ | |
| def run_pipeline(image: Image.Image): | |
| if image is None: | |
| return "Please upload an image.", "", "", None | |
| image = image.convert("RGB") | |
| try: | |
| preds = predict_objects(image) | |
| objs = [label for label, _ in preds] | |
| objects_text = "\n".join(f"- {label} ({conf * 100:.1f}% confidence)" for label, conf in preds) | |
| except Exception as e: | |
| return f"Object prediction failed: {e}", "", "", None | |
| try: | |
| depth_info = analyze_depth(image) | |
| except Exception as e: | |
| depth_info = f"Depth analysis failed: {e}" | |
| safety_note = check_safety(objs) | |
| depth_safety_text = f"{depth_info}\n{safety_note}" | |
| try: | |
| guidance = generate_guidance(image, objs, depth_info, safety_note) | |
| except Exception as e: | |
| guidance = f"Guidance generation failed: {e}" | |
| audio_path = text_to_speech(guidance) if not guidance.startswith("Guidance generation failed") else None | |
| return objects_text, depth_safety_text, guidance, audio_path | |
| # ββ GRADIO INTERFACE ββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| with gr.Blocks(title="Blind Navigation Assistant") as demo: | |
| gr.Markdown( | |
| """ | |
| # π¦― Blind Navigation Assistant | |
| Upload a photo of the scene ahead. The app will identify likely | |
| objects/scene type, estimate obstacle distance, flag vehicle | |
| hazards, and speak short navigation guidance. | |
| """ | |
| ) | |
| with gr.Row(): | |
| with gr.Column(): | |
| image_input = gr.Image(type="pil", label="Upload Image") | |
| submit_btn = gr.Button("Analyze", variant="primary") | |
| with gr.Column(): | |
| objects_output = gr.Textbox(label="Detected / Predicted Objects", lines=4) | |
| depth_safety_output = gr.Textbox(label="Depth & Safety Analysis", lines=3) | |
| guidance_output = gr.Textbox(label="Navigation Guidance", lines=4) | |
| audio_output = gr.Audio(label="Voice Guidance", autoplay=True) | |
| submit_btn.click( | |
| fn=run_pipeline, | |
| inputs=image_input, | |
| outputs=[objects_output, depth_safety_output, guidance_output, audio_output], | |
| ) | |
| gr.Examples( | |
| examples=[], # add local sample image paths here if you have any, e.g. ["sample1.jpg"] | |
| inputs=image_input, | |
| label="Sample Images (optional)", | |
| ) | |
| if __name__ == "__main__": | |
| demo.launch() | |