added conopy script
This commit is contained in:
@@ -4,6 +4,7 @@ from __future__ import annotations
|
|||||||
|
|
||||||
import argparse
|
import argparse
|
||||||
import json
|
import json
|
||||||
|
import time
|
||||||
from collections.abc import Iterable
|
from collections.abc import Iterable
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any
|
from typing import Any
|
||||||
@@ -142,7 +143,43 @@ def destination_for(url: str, output_dir: Path) -> Path:
|
|||||||
return output_dir / name
|
return output_dir / name
|
||||||
|
|
||||||
|
|
||||||
def download_urls(urls: list[str], output_dir: Path, overwrite: bool) -> tuple[int, int, list[str]]:
|
def _format_mb(value: int) -> str:
|
||||||
|
return f"{value / (1024 * 1024):.1f} MB"
|
||||||
|
|
||||||
|
|
||||||
|
def _download_one(url: str, destination: Path, timeout_s: int, chunk_size: int) -> None:
|
||||||
|
partial = destination.with_suffix(f"{destination.suffix}.part")
|
||||||
|
request = Request(url, headers={"User-Agent": USER_AGENT})
|
||||||
|
with urlopen(request, timeout=timeout_s) as response, partial.open("wb") as output:
|
||||||
|
total_header = response.headers.get("Content-Length")
|
||||||
|
total = int(total_header) if total_header and total_header.isdigit() else None
|
||||||
|
read = 0
|
||||||
|
last_report = time.monotonic()
|
||||||
|
while True:
|
||||||
|
chunk = response.read(chunk_size)
|
||||||
|
if not chunk:
|
||||||
|
break
|
||||||
|
output.write(chunk)
|
||||||
|
read += len(chunk)
|
||||||
|
now = time.monotonic()
|
||||||
|
if now - last_report >= 5:
|
||||||
|
if total:
|
||||||
|
pct = (read / total) * 100
|
||||||
|
print(f" {_format_mb(read)} / {_format_mb(total)} ({pct:.1f}%)")
|
||||||
|
else:
|
||||||
|
print(f" {_format_mb(read)}")
|
||||||
|
last_report = now
|
||||||
|
partial.replace(destination)
|
||||||
|
|
||||||
|
|
||||||
|
def download_urls(
|
||||||
|
urls: list[str],
|
||||||
|
output_dir: Path,
|
||||||
|
overwrite: bool,
|
||||||
|
*,
|
||||||
|
timeout_s: int = 60,
|
||||||
|
chunk_size: int = 1024 * 1024,
|
||||||
|
) -> tuple[int, int, list[str]]:
|
||||||
downloaded = 0
|
downloaded = 0
|
||||||
skipped = 0
|
skipped = 0
|
||||||
failed: list[str] = []
|
failed: list[str] = []
|
||||||
@@ -155,15 +192,17 @@ def download_urls(urls: list[str], output_dir: Path, overwrite: bool) -> tuple[i
|
|||||||
print(f"skip existing {destination}")
|
print(f"skip existing {destination}")
|
||||||
continue
|
continue
|
||||||
|
|
||||||
|
partial = destination.with_suffix(f"{destination.suffix}.part")
|
||||||
|
if partial.exists() and not overwrite:
|
||||||
|
partial.unlink()
|
||||||
|
|
||||||
print(f"download {url}")
|
print(f"download {url}")
|
||||||
try:
|
try:
|
||||||
request = Request(url, headers={"User-Agent": USER_AGENT})
|
_download_one(url, destination, timeout_s, chunk_size)
|
||||||
with urlopen(request) as response, destination.open("wb") as output:
|
|
||||||
output.write(response.read())
|
|
||||||
except (HTTPError, URLError) as exc:
|
except (HTTPError, URLError) as exc:
|
||||||
failed.append(f"{url}: {exc}")
|
failed.append(f"{url}: {exc}")
|
||||||
if destination.exists():
|
if partial.exists():
|
||||||
destination.unlink()
|
partial.unlink()
|
||||||
continue
|
continue
|
||||||
downloaded += 1
|
downloaded += 1
|
||||||
|
|
||||||
@@ -178,6 +217,7 @@ def parse_args() -> argparse.Namespace:
|
|||||||
parser.add_argument("--index-url", default=META_CHM_V2_INDEX_URL)
|
parser.add_argument("--index-url", default=META_CHM_V2_INDEX_URL)
|
||||||
parser.add_argument("--urls-file", type=Path)
|
parser.add_argument("--urls-file", type=Path)
|
||||||
parser.add_argument("--skip-url-check", action="store_true")
|
parser.add_argument("--skip-url-check", action="store_true")
|
||||||
|
parser.add_argument("--timeout-s", default=60, type=int)
|
||||||
parser.add_argument("--overwrite", action="store_true")
|
parser.add_argument("--overwrite", action="store_true")
|
||||||
return parser.parse_args()
|
return parser.parse_args()
|
||||||
|
|
||||||
@@ -199,7 +239,13 @@ def main() -> None:
|
|||||||
if not args.skip_url_check:
|
if not args.skip_url_check:
|
||||||
validate_urls(urls)
|
validate_urls(urls)
|
||||||
|
|
||||||
downloaded, skipped, failed = download_urls(urls, args.output_dir, args.overwrite)
|
print(f"selected_tiles={len(urls)}")
|
||||||
|
downloaded, skipped, failed = download_urls(
|
||||||
|
urls,
|
||||||
|
args.output_dir,
|
||||||
|
args.overwrite,
|
||||||
|
timeout_s=args.timeout_s,
|
||||||
|
)
|
||||||
print(f"downloaded={downloaded} skipped={skipped} failed={len(failed)}")
|
print(f"downloaded={downloaded} skipped={skipped} failed={len(failed)}")
|
||||||
if failed:
|
if failed:
|
||||||
for item in failed:
|
for item in failed:
|
||||||
|
|||||||
Reference in New Issue
Block a user