"""Professional Hugging Face Space for multi-level satellite image analysis.""" from __future__ import annotations import os import time import uuid from functools import lru_cache from pathlib import Path import gradio as gr import numpy as np import torch from huggingface_hub import hf_hub_download from PIL import Image from transformers import ( AutoImageProcessor, AutoModelForImageClassification, Mask2FormerForUniversalSegmentation, ) try: import spaces except ImportError: class _SpacesFallback: @staticmethod def GPU(*_args, **_kwargs): def decorator(function): return function return decorator spaces = _SpacesFallback() from satellite_utils import ( build_analysis_summary, build_class_table, build_detection_summary, build_detection_table, build_lulc_table, normalized_entropy, render_detections, render_lulc_assessment, render_segmentation, resize_for_inference, write_class_csv, write_detection_csv, write_json, write_lulc_csv, write_pixel_geojson, ) CLASSIFICATION_MODEL_ID = "mrm8488/convnext-tiny-finetuned-eurosat" SEGMENTATION_MODEL_ID = "mfaytin/mask2former-satellite" DETECTION_MODEL_ID = "bluelabel/satellite-equipment-detection-yolov8n-vhr10" DETECTION_FILENAME = "best.pt" OUTPUT_ROOT = Path("/tmp/satellite-vision-toolkit") OPEN_EARTH_MAP_LABELS = { 0: "background", 1: "bareland", 2: "grass", 3: "pavement", 4: "road", 5: "tree", 6: "water", 7: "cropland", 8: "building", } CASE_STUDIES = { "residential": { "title": "Dense residential block", "image": "assets/cases/dense_residential.jpg", "source": "https://huggingface.co/datasets/blanchon/UC_Merced", "sensor": "USGS Urban Area Imagery · RGB · 0.3 m · 256×256 px", "focus": "Inspect buildings, impervious surfaces, street texture, and residential scene confidence.", }, "intersection": { "title": "Urban intersection", "image": "assets/cases/urban_intersection.jpg", "source": "https://huggingface.co/datasets/blanchon/UC_Merced", "sensor": "USGS Urban Area Imagery · RGB · 0.3 m · 256×256 px", "focus": "Test road/pavement segmentation and small-vehicle sensitivity at a city junction.", }, "harbor": { "title": "Urban marina / harbor", "image": "assets/cases/harbor_marina.jpg", "source": "https://huggingface.co/datasets/blanchon/UC_Merced", "sensor": "USGS Urban Area Imagery · RGB · 0.3 m · 256×256 px", "focus": "Compare water segmentation with supported ship/harbor object predictions.", }, "parking": { "title": "Urban parking lot", "image": "assets/cases/parking_lot.jpg", "source": "https://huggingface.co/datasets/blanchon/UC_Merced", "sensor": "USGS Urban Area Imagery · RGB · 0.3 m · 256×256 px", "focus": "Probe pavement coverage and the detector's limits for tightly packed small vehicles.", }, } def _device() -> torch.device: torch.set_num_threads(max(1, min(4, os.cpu_count() or 1))) return torch.device("cuda" if torch.cuda.is_available() else "cpu") @lru_cache(maxsize=1) def load_classifier(): device = _device() processor = AutoImageProcessor.from_pretrained(CLASSIFICATION_MODEL_ID) model = AutoModelForImageClassification.from_pretrained(CLASSIFICATION_MODEL_ID).to(device).eval() id2label = {int(key): str(value) for key, value in model.config.id2label.items()} return processor, model, id2label, device @lru_cache(maxsize=1) def load_segmenter(): device = _device() processor = AutoImageProcessor.from_pretrained(SEGMENTATION_MODEL_ID) model = Mask2FormerForUniversalSegmentation.from_pretrained(SEGMENTATION_MODEL_ID).to(device).eval() return processor, model, OPEN_EARTH_MAP_LABELS, device @lru_cache(maxsize=1) def load_detector(): from ultralytics import YOLO weights = hf_hub_download(repo_id=DETECTION_MODEL_ID, filename=DETECTION_FILENAME) return YOLO(weights) def _new_output_dir() -> Path: output_dir = OUTPUT_ROOT / uuid.uuid4().hex output_dir.mkdir(parents=True, exist_ok=True) return output_dir def _require_image(image: Image.Image | None) -> Image.Image: if image is None: raise gr.Error("Please upload a satellite or aerial image first.") return resize_for_inference(image) def load_case_study(case_key: str): """Load a documented NASA case into the shared image input.""" case = CASE_STUDIES[case_key] note = ( f"### {case['title']}\n" f"**Acquisition:** {case['sensor']} \n" f"**Suggested analysis:** {case['focus']} \n" f"[Open UC Merced / USGS source]({case['source']}) · " "High-resolution aerial imagery is intentionally outside EuroSAT's Sentinel-2 scale; review domain shift and do not treat the dataset label as model ground truth." ) return case["image"], note def _classify_impl(prepared: Image.Image, top_k: int, output_dir: Path) -> dict[str, object]: processor, model, id2label, device = load_classifier() inputs = {name: tensor.to(device) for name, tensor in processor(images=prepared, return_tensors="pt").items()} with torch.inference_mode(): logits = model(**inputs).logits[0] probabilities = torch.softmax(logits, dim=-1).detach().cpu().tolist() rows = build_lulc_table(probabilities, id2label, top_k) entropy = normalized_entropy(probabilities) csv_path = output_dir / "lulc_classification.csv" json_path = output_dir / "lulc_classification.json" write_lulc_csv(csv_path, rows) write_json( json_path, { "task": "scene_level_lulc_classification", "model": CLASSIFICATION_MODEL_ID, "processed_image_size": {"width": prepared.width, "height": prepared.height}, "normalized_entropy": round(entropy, 6), "predictions": [ {"rank": row[0], "class": row[1], "probability_percent": row[2], "confidence_tier": row[3]} for row in rows ], "scope_note": "Whole-scene EuroSAT class; not a cadastral or planning land-use designation.", }, ) return { "rows": rows, "entropy": entropy, "assessment": render_lulc_assessment(rows, entropy), "files": [str(csv_path), str(json_path)], "device": device.type, } def _segment_impl( prepared: Image.Image, opacity: float, min_share_percent: float, output_dir: Path, ) -> dict[str, object]: processor, model, id2label, device = load_segmenter() inputs = {name: tensor.to(device) for name, tensor in processor(images=prepared, return_tensors="pt").items()} with torch.inference_mode(): outputs = model(**inputs) class_map = processor.post_process_semantic_segmentation( outputs, target_sizes=[(prepared.height, prepared.width)], )[0].cpu().numpy().astype(np.uint8) overlay, color_mask = render_segmentation(prepared, class_map, id2label, float(opacity)) rows = build_class_table(class_map, id2label, float(min_share_percent)) overlay_path = output_dir / "land_cover_overlay.png" mask_path = output_dir / "land_cover_color_mask.png" ids_path = output_dir / "land_cover_class_ids.png" csv_path = output_dir / "land_cover_summary.csv" overlay.save(overlay_path) color_mask.save(mask_path) Image.fromarray(class_map).save(ids_path) write_class_csv(csv_path, rows) return { "overlay": overlay, "mask": color_mask, "rows": rows, "files": [str(overlay_path), str(mask_path), str(ids_path), str(csv_path)], "device": device.type, } def _detect_impl( prepared: Image.Image, confidence_threshold: float, iou_threshold: float, output_dir: Path, ) -> dict[str, object]: device = "cuda" if torch.cuda.is_available() else "cpu" detector = load_detector() prediction = detector.predict( source=np.asarray(prepared), conf=float(confidence_threshold), iou=float(iou_threshold), imgsz=1024, device=device, max_det=500, verbose=False, )[0] detections: list[dict[str, object]] = [] if prediction.boxes is not None: for coordinates, confidence, class_id_value in zip( prediction.boxes.xyxy.detach().cpu().tolist(), prediction.boxes.conf.detach().cpu().tolist(), prediction.boxes.cls.detach().cpu().tolist(), ): class_id = int(class_id_value) detections.append( { "class_id": class_id, "class_name": str(prediction.names[class_id]), "confidence": float(confidence), "x1": float(coordinates[0]), "y1": float(coordinates[1]), "x2": float(coordinates[2]), "y2": float(coordinates[3]), } ) overlay = render_detections(prepared, detections) summary_rows = build_detection_summary(detections) detail_rows = build_detection_table(detections, prepared.size) overlay_path = output_dir / "satellite_detection_overlay.png" csv_path = output_dir / "satellite_detections.csv" geojson_path = output_dir / "satellite_detections_pixel_coordinates.geojson" overlay.save(overlay_path) write_detection_csv(csv_path, detail_rows) write_pixel_geojson(geojson_path, detections, prepared.size) return { "overlay": overlay, "summary": summary_rows, "details": detail_rows, "files": [str(overlay_path), str(csv_path), str(geojson_path)], "device": device, } @spaces.GPU(duration=120) def classify_lulc(image: Image.Image | None, top_k: int): started_at = time.perf_counter() prepared = _require_image(image) try: result = _classify_impl(prepared, int(top_k), _new_output_dir()) except Exception as exc: raise gr.Error(f"LULC classification failed: {type(exc).__name__}: {exc}") from exc status = ( f"Complete · {prepared.width}×{prepared.height} · top class {result['rows'][0][1]} " f"({result['rows'][0][2]:.1f}%) · {time.perf_counter() - started_at:.1f}s · device={result['device']}" ) return result["assessment"], result["rows"], result["files"], status @spaces.GPU(duration=120) def segment_satellite_image(image: Image.Image | None, opacity: float, min_share_percent: float): started_at = time.perf_counter() prepared = _require_image(image) try: result = _segment_impl(prepared, opacity, min_share_percent, _new_output_dir()) except Exception as exc: raise gr.Error(f"Land-cover segmentation failed: {type(exc).__name__}: {exc}") from exc status = ( f"Complete · {prepared.width}×{prepared.height} · {len(result['rows'])} reported cover classes · " f"{time.perf_counter() - started_at:.1f}s · device={result['device']}" ) return result["overlay"], result["mask"], result["rows"], result["files"], status @spaces.GPU(duration=120) def detect_satellite_objects(image: Image.Image | None, confidence_threshold: float, iou_threshold: float): started_at = time.perf_counter() prepared = _require_image(image) try: result = _detect_impl(prepared, confidence_threshold, iou_threshold, _new_output_dir()) except Exception as exc: raise gr.Error(f"Satellite object detection failed: {type(exc).__name__}: {exc}") from exc status = ( f"Complete · {prepared.width}×{prepared.height} · {len(result['details'])} objects · " f"{len(result['summary'])} classes · {time.perf_counter() - started_at:.1f}s · device={result['device']}" ) return result["overlay"], result["summary"], result["details"], result["files"], status @spaces.GPU(duration=180) def analyze_satellite_image( image: Image.Image | None, top_k: int, opacity: float, min_share_percent: float, confidence_threshold: float, iou_threshold: float, ): started_at = time.perf_counter() prepared = _require_image(image) output_dir = _new_output_dir() try: classification = _classify_impl(prepared, int(top_k), output_dir) segmentation = _segment_impl(prepared, opacity, min_share_percent, output_dir) detection = _detect_impl(prepared, confidence_threshold, iou_threshold, output_dir) except Exception as exc: raise gr.Error(f"Complete analysis failed: {type(exc).__name__}: {exc}") from exc elapsed = time.perf_counter() - started_at summary = build_analysis_summary( classification["rows"], float(classification["entropy"]), segmentation["rows"], detection["summary"], elapsed, ) report_path = output_dir / "analysis_report.json" write_json( report_path, { "processed_image_size": {"width": prepared.width, "height": prepared.height}, "models": { "classification": CLASSIFICATION_MODEL_ID, "segmentation": SEGMENTATION_MODEL_ID, "detection": DETECTION_MODEL_ID, }, "lulc_classification": classification["rows"], "lulc_normalized_entropy": round(float(classification["entropy"]), 6), "land_cover_pixel_shares": segmentation["rows"], "detection_summary": detection["summary"], "detection_details": detection["details"], "elapsed_seconds": round(elapsed, 3), "coordinate_note": "Detection GeoJSON is in top-left-origin image pixels and has no geographic CRS.", }, ) files = classification["files"] + segmentation["files"] + detection["files"] + [str(report_path)] status = f"Complete multi-model assessment · {prepared.width}×{prepared.height} · {elapsed:.1f}s" return ( summary, classification["assessment"], classification["rows"], segmentation["overlay"], segmentation["mask"], segmentation["rows"], detection["overlay"], detection["summary"], detection["details"], files, status, ) CSS = """ .gradio-container {max-width: 1380px !important; background: #f4f7f9; color:#172b35;} .hero {padding: 1.55rem 1.8rem; border-radius: 20px; color:#fff !important; background: linear-gradient(118deg,#071d31 0%,#0c4656 58%,#13806b 100%); box-shadow:0 14px 36px rgba(7,29,49,.22); margin: .35rem 0 .8rem;} .hero-grid {display:grid;grid-template-columns:minmax(0,1fr) auto;gap:24px;align-items:center;} .hero h1,#hero-title {font-size:2.25rem;line-height:1.08;margin:.3rem 0 .5rem;letter-spacing:-.035em;color:#fff !important;text-shadow:0 1px 1px rgba(0,0,0,.12);} .hero p {max-width:800px;margin:.35rem 0;color:#e5f7f5 !important;font-size:.98rem;line-height:1.5;} .hero .eyebrow {color:#8ff0dc !important;}.hero-links{display:flex;flex-wrap:wrap;gap:8px;margin-top:12px}.hero-links a{color:#fff!important;text-decoration:none;border:1px solid rgba(255,255,255,.35);background:rgba(255,255,255,.09);padding:5px 10px;border-radius:999px;font-size:.8rem;font-weight:700}.hero-links a:hover{background:rgba(255,255,255,.18)} .hero-stats{display:grid;grid-template-columns:repeat(2,88px);gap:8px}.hero-stats div{border:1px solid rgba(255,255,255,.2);background:rgba(255,255,255,.08);border-radius:13px;padding:11px;text-align:center}.hero-stats strong{display:block;color:#fff;font-size:1.3rem}.hero-stats span{color:#cce9e5;font-size:.7rem} .method-strip {display:grid;grid-template-columns:repeat(3,1fr);gap:10px;margin:0 0 1rem;} .method-strip div,.assessment-card,.metric-card {background:white;border:1px solid #d9e4e9;border-radius:13px;padding:12px 14px;box-shadow:0 4px 14px rgba(20,50,70,.045);} .method-strip b {display:block;color:#0b5363;margin-bottom:3px;font-size:.86rem}.method-strip span,.micro-note {color:#617681;font-size:.78rem;} .summary-grid {display:grid;grid-template-columns:repeat(4,1fr);gap:12px;margin:12px 0;} .metric-card span,.eyebrow {display:block;color:#66808c;font-size:.72rem;font-weight:750;letter-spacing:.1em;text-transform:uppercase;} .metric-card strong {display:block;font-size:1.45rem;margin:6px 0;color:#113544;}.metric-card small {color:#60747e;} .assessment-card h2 {margin:.25rem 0;color:#123c49}.assessment-card p {color:#526b76;} .prob-row {display:grid;grid-template-columns:155px 1fr 62px;gap:10px;align-items:center;margin:8px 0;font-size:.86rem;} .prob-row b {text-align:right}.prob-track {height:9px;background:#e5edf1;border-radius:20px;overflow:hidden}.prob-track i {display:block;height:100%;background:linear-gradient(90deg,#169c7d,#36b7c5);border-radius:20px;} .section-note {padding:12px 14px;border-left:4px solid #15947a;background:#eef9f6;border-radius:8px;color:#315c62;} .case-heading{display:flex;justify-content:space-between;align-items:end;margin:.35rem 2px .5rem}.case-heading h2{margin:0;color:#123746;font-size:1.2rem}.case-heading p{margin:0;color:#647984;font-size:.82rem} .case-grid{display:grid!important;grid-template-columns:repeat(4,minmax(0,1fr))!important;gap:12px!important}.case-grid>div{min-width:0!important} .case-card {background:#fff;border:1px solid #d7e2e8!important;border-radius:15px!important;padding:8px!important;box-shadow:0 6px 16px rgba(20,50,70,.06);min-width:0!important}.case-card:hover{border-color:#62a99a!important;box-shadow:0 9px 22px rgba(20,80,70,.1)} .case-thumb {border-radius:10px!important;overflow:hidden;background:#e7eef1}.case-thumb img{width:100%!important;height:100%!important;object-fit:cover!important;image-rendering:auto!important}.case-card h3{margin:2px 2px 0!important;color:#143643;font-size:.92rem!important}.case-card p{margin:0 2px 3px!important;color:#667b85;font-size:.73rem!important;line-height:1.35}.case-card button{min-height:34px!important;font-size:.78rem!important} .case-note{background:#eaf7f3;border:1px solid #b9ded3;border-radius:11px;padding:1px 12px;margin:.4rem 0 .75rem}.case-note h3{font-size:.95rem;margin:.6rem 0 .2rem}.case-note p{font-size:.8rem} .workspace-title h3{margin-bottom:.25rem!important}.controls-card{background:#fff;border:1px solid #dae5ea;border-radius:15px;padding:14px!important} @media(max-width:950px){.case-grid{grid-template-columns:repeat(2,minmax(0,1fr))!important}} @media(max-width:850px){.hero-grid{grid-template-columns:1fr}.hero-stats{display:none}.method-strip,.summary-grid{grid-template-columns:1fr}.prob-row{grid-template-columns:115px 1fr 56px}.case-heading{display:block}} @media(max-width:560px){.case-grid{grid-template-columns:1fr!important}} """ with gr.Blocks(title="Satellite Vision Toolkit Pro", css=CSS, theme=gr.themes.Soft()) as demo: gr.HTML("""
Clear, multi-level interpretation of local urban overhead imagery—from whole-scene LULC context to pixel cover and individual objects.
High-resolution 256×256 USGS aerial chips · click any card to load