API, GeoJSON, CBOR Support

This commit is contained in:
randogoth 2025-02-27 10:33:19 +00:00
parent 08eca3e042
commit a9f98c80df
8 changed files with 113 additions and 40 deletions

View file

@ -5,6 +5,8 @@ import geopandas as gpd
import pandas as pd
import pyarrow.parquet as pq
import pyarrow as pa
import json
import cbor2
from pathlib import Path
from shapely import wkb
from typing import List, Dict
@ -29,36 +31,41 @@ def append_to_parquet(file_path: Path, new_data: gpd.GeoDataFrame):
"""Appends new data to an existing Parquet file while ensuring correct merging."""
if file_path.exists():
try:
# ✅ Load existing data
existing_data = gpd.read_parquet(file_path)
# ✅ Merge with new data
combined_data = pd.concat([existing_data, new_data], ignore_index=True)
# ✅ Drop duplicates *only if feature_id exists*
if "feature_id" in combined_data.columns:
combined_data = combined_data.drop_duplicates(subset="feature_id", keep="last")
# ✅ Reset index before saving
combined_data = combined_data.reset_index(drop=True)
# ✅ Save back to the same file
combined_data.to_parquet(file_path, index=False)
print(f"✅ Appended new data to {file_path}")
except Exception as e:
print(f"❌ Error while merging {file_path}: {e}")
else:
# If the file doesn't exist, create it
new_data.to_parquet(file_path, index=False)
print(f"✅ Created new file: {file_path}")
def slice_and_split_geoparquet(input_file: str, output_dir: str):
"""
Splits the GeoParquet file into multiple files based on 4-character Geohash boxes.
Appends new data to existing Geohash Parquet files while preventing duplicate geometries.
"""
def save_as_geojson(geohash_code: str, gdf: gpd.GeoDataFrame, output_path: Path):
"""Saves a GeoDataFrame as a GeoJSON file."""
geojson_file = output_path / f"{geohash_code}.geojson"
geojson_dict = json.loads(gdf.to_json())
with open(geojson_file, "w", encoding="utf-8") as f:
json.dump(geojson_dict, f)
print(f"✅ Saved GeoJSON: {geojson_file}")
def save_as_cbor(geohash_code: str, gdf: gpd.GeoDataFrame, output_path: Path):
"""Saves a GeoDataFrame as a CBOR file (compact binary format)."""
cbor_file = output_path / f"{geohash_code}.cbor"
geojson_dict = json.loads(gdf.to_json())
with open(cbor_file, "wb") as f:
cbor2.dump(geojson_dict, f)
print(f"✅ Saved CBOR: {cbor_file}")
def slice_and_split_geoparquet(input_file: str, output_dir: str, export_parquet: bool, export_geojson: bool, export_cbor: bool):
"""Splits the GeoParquet file into multiple files based on 4-character Geohash tiles."""
input_path, output_path = Path(input_file), Path(output_dir)
if not input_path.exists():
raise FileNotFoundError(f"File '{input_file}' not found.")
@ -69,17 +76,14 @@ def slice_and_split_geoparquet(input_file: str, output_dir: str):
con = duckdb.connect()
con.execute("INSTALL spatial; LOAD spatial;")
# Create a temporary table of all non POINT geometries
con.execute(f"CREATE TEMP TABLE geoparquet AS SELECT * FROM read_parquet('{input_path}') WHERE ST_GeometryType(geometry) IS NOT NULL AND ST_GeometryType(geometry) != 'POINT'")
# Compute bounding box for each geometry
bbox_results = con.execute("""
SELECT feature_id, rowid, ST_XMin(ST_Envelope(geometry)), ST_YMin(ST_Envelope(geometry)),
ST_XMax(ST_Envelope(geometry)), ST_YMax(ST_Envelope(geometry))
FROM geoparquet;
""").fetchall()
# Compute all intersecting geohashes
geohash_mapping: Dict[str, List[int]] = {}
for feature_id, rowid, min_x, min_y, max_x, max_y in bbox_results:
@ -89,13 +93,11 @@ def slice_and_split_geoparquet(input_file: str, output_dir: str):
geohash_mapping[geohash_code] = []
geohash_mapping[geohash_code].append(rowid)
# Dictionary to store in-memory results before writing
geohash_data: Dict[str, gpd.GeoDataFrame] = {}
for geohash_code, rowids in geohash_mapping.items():
geohash_bbox = geohash.bbox(geohash_code)
# Clip geometries inside this geohash box
filtered_data = con.execute(f"""
SELECT feature_id, tags,
ST_AsWKB(ST_Intersection(geometry,
@ -110,46 +112,51 @@ def slice_and_split_geoparquet(input_file: str, output_dir: str):
""").fetchdf()
if not filtered_data.empty:
# ✅ Remove non-WKB values
filtered_data = filtered_data[filtered_data["clipped_geom"].apply(lambda x: isinstance(x, (bytes, bytearray)))]
if not filtered_data.empty:
# ✅ Convert WKB to Shapely geometries safely
filtered_data["geometry"] = filtered_data["clipped_geom"].apply(
lambda x: wkb.loads(bytes(x)) if isinstance(x, (bytes, bytearray)) else None
)
filtered_data.drop(columns=["clipped_geom"], inplace=True)
# Remove rows where geometry conversion failed
filtered_data = filtered_data.dropna(subset=["geometry"])
if not filtered_data.empty:
# Convert to GeoPandas GeoDataFrame
gdf = gpd.GeoDataFrame(filtered_data, geometry="geometry", crs="EPSG:4326")
# Store in memory
geohash_data[geohash_code] = gdf
# Step 4: Write each geohash's data **only once** and prevent duplicates
for geohash_code, gdf in geohash_data.items():
geohash_file = output_path / f"{geohash_code}.parquet"
append_to_parquet(geohash_file, gdf)
print(f"✅ Updated: {geohash_file}")
if export_parquet:
geohash_file = output_path / f"{geohash_code}.parquet"
append_to_parquet(geohash_file, gdf)
if export_geojson:
save_as_geojson(geohash_code, gdf, output_path)
if export_cbor:
save_as_cbor(geohash_code, gdf, output_path)
con.close()
print("🎉 Processing complete!")
def main():
parser = argparse.ArgumentParser(description="Split a GeoParquet file into Geohash tiles.")
parser = argparse.ArgumentParser(description="Split a GeoParquet file into Geohash tiles with export format options.")
parser.add_argument("-i", "--input", required=True, help="Path to the input GeoParquet file.")
parser.add_argument("-o", "--output", required=True, help="Directory to save output Geohash tiles.")
parser.add_argument("-o", "--output", required=True, help="Directory to save output files.")
parser.add_argument("--parquet", action="store_true", help="Export as GeoParquet")
parser.add_argument("--geojson", action="store_true", help="Export as GeoJSON")
parser.add_argument("--cbor", action="store_true", help="Export as CBOR")
args = parser.parse_args()
if not (args.parquet or args.geojson or args.cbor):
args.parquet = True # Default to GeoParquet if no options are given
try:
slice_and_split_geoparquet(args.input, args.output)
slice_and_split_geoparquet(args.input, args.output, args.parquet, args.geojson, args.cbor)
except Exception as e:
print(f"❌ Error: {e}")
if __name__ == "__main__":
main()
main()