"""Bootstrap canopy height rasters for an AOI.""" from __future__ import annotations import argparse import json from collections.abc import Iterable from pathlib import Path from typing import Any from urllib.error import HTTPError, URLError from urllib.request import Request, urlopen META_CHM_V2_INDEX_BASE_URL = ( "https://dataforgood-fb-data.s3.amazonaws.com/" "forests/v2/global/dinov3_global_chm_v2_ml3" ) META_CHM_V2_BASE_URL = "https://data.source.coop/tge-labs/meta-chm-v2/chm" META_CHM_V2_INDEX_URL = f"{META_CHM_V2_INDEX_BASE_URL}/tiles.geojson" USER_AGENT = "RadioApi canopy bootstrap/0.1" def parse_bbox(raw: str) -> tuple[float, float, float, float]: parts = [float(part.strip()) for part in raw.split(",")] if len(parts) != 4: raise argparse.ArgumentTypeError("bbox must be minlon,minlat,maxlon,maxlat") minlon, minlat, maxlon, maxlat = parts if minlon >= maxlon or minlat >= maxlat: raise argparse.ArgumentTypeError("bbox min values must be lower than max values") return minlon, minlat, maxlon, maxlat def intersects( a: tuple[float, float, float, float], b: tuple[float, float, float, float], ) -> bool: return not (a[2] < b[0] or a[0] > b[2] or a[3] < b[1] or a[1] > b[3]) def _walk_coordinates(value: Any) -> Iterable[tuple[float, float]]: if ( isinstance(value, list) and len(value) >= 2 and isinstance(value[0], int | float) and isinstance(value[1], int | float) ): yield float(value[0]), float(value[1]) return if isinstance(value, list): for item in value: yield from _walk_coordinates(item) def feature_bbox(feature: dict[str, Any]) -> tuple[float, float, float, float] | None: bbox = feature.get("bbox") if isinstance(bbox, list) and len(bbox) >= 4: return float(bbox[0]), float(bbox[1]), float(bbox[2]), float(bbox[3]) geometry = feature.get("geometry") or {} coordinates = list(_walk_coordinates(geometry.get("coordinates"))) if not coordinates: return None lons = [coord[0] for coord in coordinates] lats = [coord[1] for coord in coordinates] return min(lons), min(lats), max(lons), max(lats) def _asset_url(feature: dict[str, Any], base_url: str = META_CHM_V2_BASE_URL) -> str | None: properties = feature.get("properties") or {} for key in ("url", "href", "asset_href", "s3_url", "cog_url"): value = properties.get(key) or feature.get(key) if isinstance(value, str) and value: if value.startswith("s3://dataforgood-fb-data/"): return value.replace( "s3://dataforgood-fb-data", "https://dataforgood-fb-data.s3.amazonaws.com", 1, ) return value for key in ("filename", "file", "name", "tile", "quadkey"): value = properties.get(key) or feature.get(key) if isinstance(value, str) and value: filename = value if value.lower().endswith((".tif", ".tiff")) else f"{value}.tif" return f"{base_url.rstrip('/')}/{filename}" return None def validate_urls(urls: list[str], sample_size: int = 2) -> None: for url in urls[:sample_size]: request = Request(url, headers={"Range": "bytes=0-0", "User-Agent": USER_AGENT}) try: with urlopen(request) as response: if response.status >= 400: raise SystemExit(f"Canopy URL is not reachable: {url} ({response.status})") except HTTPError as exc: raise SystemExit( f"Canopy URL is not reachable: {url} ({exc.code}). " "Check --index-url/--urls-file or the dataset URL template." ) from exc except URLError as exc: raise SystemExit(f"Canopy URL validation failed: {url}: {exc}") from exc def load_index(index_url: str) -> dict[str, Any]: path = Path(index_url) if path.exists(): return json.loads(path.read_text(encoding="utf-8")) request = Request(index_url, headers={"User-Agent": USER_AGENT}) with urlopen(request) as response: return json.loads(response.read().decode("utf-8")) def select_tile_urls( index: dict[str, Any], bbox: tuple[float, float, float, float], *, base_url: str = META_CHM_V2_BASE_URL, ) -> list[str]: urls: list[str] = [] for feature in index.get("features", []): tile_bbox = feature_bbox(feature) if tile_bbox is None or not intersects(tile_bbox, bbox): continue url = _asset_url(feature, base_url=base_url) if url is not None: urls.append(url) return sorted(set(urls)) def urls_from_file(path: Path) -> list[str]: return [ line.strip() for line in path.read_text(encoding="utf-8").splitlines() if line.strip() and not line.lstrip().startswith("#") ] def destination_for(url: str, output_dir: Path) -> Path: name = url.rstrip("/").rsplit("/", 1)[-1] if not name.lower().endswith((".tif", ".tiff")): name = f"{name}.tif" return output_dir / name def download_urls(urls: list[str], output_dir: Path, overwrite: bool) -> tuple[int, int, list[str]]: downloaded = 0 skipped = 0 failed: list[str] = [] output_dir.mkdir(parents=True, exist_ok=True) for url in urls: destination = destination_for(url, output_dir) if destination.exists() and not overwrite: skipped += 1 print(f"skip existing {destination}") continue print(f"download {url}") try: request = Request(url, headers={"User-Agent": USER_AGENT}) with urlopen(request) as response, destination.open("wb") as output: output.write(response.read()) except (HTTPError, URLError) as exc: failed.append(f"{url}: {exc}") if destination.exists(): destination.unlink() continue downloaded += 1 return downloaded, skipped, failed def parse_args() -> argparse.Namespace: parser = argparse.ArgumentParser(description="Bootstrap Meta/WRI canopy height COGs") parser.add_argument("--bbox", type=parse_bbox, help="AOI bbox: minlon,minlat,maxlon,maxlat") parser.add_argument("--output-dir", default=Path("data/canopy"), type=Path) parser.add_argument("--source", default="meta-chm-v2", choices=["meta-chm-v2"]) parser.add_argument("--index-url", default=META_CHM_V2_INDEX_URL) parser.add_argument("--urls-file", type=Path) parser.add_argument("--skip-url-check", action="store_true") parser.add_argument("--overwrite", action="store_true") return parser.parse_args() def main() -> None: args = parse_args() if args.urls_file is None and args.bbox is None: raise SystemExit("--bbox is required unless --urls-file is provided") if args.urls_file is not None: urls = urls_from_file(args.urls_file) else: index = load_index(args.index_url) urls = select_tile_urls(index, args.bbox) if not urls: raise SystemExit("No canopy tiles intersect requested AOI") if not args.skip_url_check: validate_urls(urls) downloaded, skipped, failed = download_urls(urls, args.output_dir, args.overwrite) print(f"downloaded={downloaded} skipped={skipped} failed={len(failed)}") if failed: for item in failed: print(f"failed {item}") raise SystemExit(1) if __name__ == "__main__": main()