Commit db18d70b authored by Eric Duminil's avatar Eric Duminil
Browse files

Broken code, with NOTE & TODO

parent cc16a44a
# https://www.lgln.niedersachsen.de/startseite/geodaten_karten/3d_geobasisdaten/3d_gebaudemodelle/3d-gebaudemodelle-lod1-und-lod2-142891.html # https://www.lgln.niedersachsen.de/startseite/geodaten_karten/3d_geobasisdaten/3d_gebaudemodelle/3d-gebaudemodelle-lod1-und-lod2-142891.html
# Für alle offenen Geodaten des LGLN gilt die Lizenz CC BY 4.0. # Für alle offenen Geodaten des LGLN gilt die Lizenz CC BY 4.0.
import json
import os
import sys
from concurrent.futures import ThreadPoolExecutor, as_completed
from pathlib import Path
from download_metalink import SCRIPT_DIR, download_file
NIEDERSACHSEN_GEOJSON_URL = (
"https://arcgis-geojson.s3.eu-de.cloud-object-storage.appdomain.cloud/lod2/lgln-opengeodata-lod2.geojson"
)
NIEDERSACHSEN_JSON = SCRIPT_DIR / "tmp" / "niedersachsen_lod2.geojson"
tmp_path = SCRIPT_DIR / "tmp"
tmp_path.mkdir(parents=True, exist_ok=True)
if not NIEDERSACHSEN_JSON.exists():
download_file(NIEDERSACHSEN_GEOJSON_URL, NIEDERSACHSEN_JSON)
with open(NIEDERSACHSEN_JSON) as json_file:
data = json.load(json_file)
features = data["features"]
# {
# 'type': 'Feature',
# 'properties': {
# 'xml': 'https://lod2.s3.eu-de.cloud-object-storage.appdomain.cloud/LoD2_32_598_5897_1_ni.gml',
# 'shp': 'https://lod2.s3.eu-de.cloud-object-storage.appdomain.cloud/SHP/LoD2_32_598_5897_1_ni.zip',
# 'Aktualitaet': '2024-06-17 00:00:00',
# 'tile_id': '25985897'
# },
# 'geometry': {
# 'type': 'Polygon',
# 'crs': {'type': 'name', 'properties': {'name': 'EPSG:4326'}},
# 'coordinates': [
# [
# [10.467457605796172, 53.21322495460642],
# [10.48289772131542, 53.21322495460642],
# [10.48289772131542, 53.222063537210374],
# [10.467457605796172, 53.222063537210374],
# [10.467457605796172, 53.21322495460642]
# ]
# ]
# }
# },
# TODO: Dry with 02_bayern.py and 10_nordrhein_westfalen.py, and move some functions to utils.py?
# NOTE: Could define a CityGML class: URL, Path, Size, SHA, Date, UTM, Source, Bundesland. And "Check" method, as well as "Download". "Select", "Compress" too?
def download_and_verify(file_info: dict, tmp_dir: Path, download_dir: Path, max_retries=3) -> bool:
"""Download a file, verify its size, and move to final location."""
filename = file_info["name"]
expected_size = int(file_info["size"])
urls = [NRW_SERVER + filename]
final_path = download_dir / filename
# Check if file already exists and is valid
if final_path.exists():
print(f"Checking existing file: {filename}")
if os.path.getsize(final_path) == expected_size:
print(f"✓ {filename} already downloaded and verified")
return True
else:
print(f" Existing file has incorrect hash, will re-download")
final_path.unlink()
tmp_path = tmp_dir / filename
for attempt in range(max_retries):
print(f"Downloading {filename} (attempt {attempt + 1}/{max_retries})")
# Try each URL until one succeeds
download_success = False
for url in urls:
print(f" Trying {url}")
if download_file(url, tmp_path):
download_success = True
break
if not download_success:
print(f" Failed to download from all URLs")
if tmp_path.exists():
tmp_path.unlink()
continue
print(f" Verifying size for {filename}")
actual_size = os.path.getsize(tmp_path)
if actual_size == expected_size:
# Move to final location
final_path.parent.mkdir(parents=True, exist_ok=True)
tmp_path.rename(final_path)
print(f"✓ {filename} downloaded and verified successfully")
return True
else:
print(f" Hash mismatch for {filename}")
print(f" Expected: {expected_size}")
print(f" Got: {actual_size}")
tmp_path.unlink()
print(f"✗ Failed to download {filename} after {max_retries} attempts")
return False
def download_all_files(
json_filename: str, download_path: Path, tmp_path: Path = SCRIPT_DIR / "tmp", jobs: int = 4, retries: int = 3
):
# Create directories
tmp_path.mkdir(parents=True, exist_ok=True)
download_file(
json_filename,
NRW_JSON,
)
with open(NRW_JSON) as json_file:
data = json.load(json_file)
# Parse metalink file
files = [file for sets in data["datasets"] for file in sets["files"]]
print(f"Found {len(files)} files to download\n")
download_path.mkdir(parents=True, exist_ok=True)
# Download files in parallel
successful = 0
failed = 0
with ThreadPoolExecutor(max_workers=jobs) as executor:
futures = {
executor.submit(download_and_verify, file_info, tmp_path, download_path, retries): file_info["name"]
for file_info in files
}
for future in as_completed(futures):
filename = futures[future]
try:
if future.result():
successful += 1
else:
failed += 1
except Exception as e:
print(f"✗ Exception while processing {filename}: {e}")
failed += 1
# Summary
print(f"\n{'='*60}")
print(f"Download complete: {successful} successful, {failed} failed")
print(f"{'='*60}")
if failed > 0:
sys.exit(1)
if __name__ == "__main__":
download_all_files(NRW_SERVER + "index.json", SCRIPT_DIR / "citygml" / "nordrhein_westfalen")
Supports Markdown
0% or .
You are about to add 0 people to the discussion. Proceed with caution.
Finish editing this message first!
Please register or to comment