lookup endpoint, better tag filtering
This commit is contained in:
parent
be5938ffb0
commit
9168a7cbb8
3 changed files with 102 additions and 10 deletions
|
|
@ -22,9 +22,9 @@ process file:
|
||||||
#!/usr/bin/env bash
|
#!/usr/bin/env bash
|
||||||
stem=$(basename {{file}} .pbf)
|
stem=$(basename {{file}} .pbf)
|
||||||
echo "Extracting Data from OpenStreetMaps PBF"
|
echo "Extracting Data from OpenStreetMaps PBF"
|
||||||
quackosm {{file}} --osm-tags-filter-file tags.json --compact --output ${stem}.parquet
|
quackosm {{file}} --osm-tags-filter-file tags.json --all-tags --compact --output ${stem}.parquet
|
||||||
echo "Generating geohash parquet files"
|
echo "Generating geohash parquet files"
|
||||||
python3 slice.py --input ${stem}.parquet --water water.parquet --output geohash --parquet
|
python3 slice.py --input ${stem}.parquet --water water.parquet --tags tags.json --output geohash --parquet
|
||||||
echo "Cleaning up"
|
echo "Cleaning up"
|
||||||
rm ${stem}.parquet
|
rm ${stem}.parquet
|
||||||
|
|
||||||
|
|
|
||||||
65
api.py
65
api.py
|
|
@ -1,9 +1,13 @@
|
||||||
from fastapi import FastAPI, HTTPException
|
from fastapi import FastAPI, HTTPException, Query
|
||||||
from fastapi.responses import Response, FileResponse
|
from fastapi.responses import Response, FileResponse, JSONResponse
|
||||||
from fastapi.middleware.gzip import GZipMiddleware
|
from fastapi.middleware.gzip import GZipMiddleware
|
||||||
import geopandas as gpd
|
import geopandas as gpd
|
||||||
import cbor2
|
import cbor2
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
import pygeohash as pgh
|
||||||
|
from shapely.geometry import Point, LineString, MultiLineString
|
||||||
|
from pyproj import Geod
|
||||||
|
import json
|
||||||
|
|
||||||
app = FastAPI()
|
app = FastAPI()
|
||||||
|
|
||||||
|
|
@ -44,4 +48,59 @@ async def get_geodata(geohash: str, format: str = "cbor"):
|
||||||
return Response(content=cbor_data, media_type="application/cbor",
|
return Response(content=cbor_data, media_type="application/cbor",
|
||||||
headers={"Content-Disposition": f"attachment; filename={base_geohash}.cbor"})
|
headers={"Content-Disposition": f"attachment; filename={base_geohash}.cbor"})
|
||||||
|
|
||||||
raise HTTPException(status_code=400, detail="Invalid format. Use 'parquet', 'geojson', or 'cbor'.")
|
raise HTTPException(status_code=400, detail="Invalid format. Use 'parquet', 'geojson', or 'cbor'.")
|
||||||
|
|
||||||
|
@app.get("/lookup/")
|
||||||
|
async def lookup(
|
||||||
|
latitude: float = Query(..., description="Latitude of the point"),
|
||||||
|
longitude: float = Query(..., description="Longitude of the point"),
|
||||||
|
distance: float = Query(10, description="Distance threshold in meters for nearby lines")
|
||||||
|
):
|
||||||
|
"""
|
||||||
|
Finds polygons containing the given point and lines within a threshold (in meters).
|
||||||
|
"""
|
||||||
|
# Compute geohash (precision can be adjusted)
|
||||||
|
geohash = pgh.encode(latitude, longitude, precision=4) # Adjust precision as needed
|
||||||
|
parquet_file = GEOHASH_FOLDER / f"{geohash}.parquet"
|
||||||
|
|
||||||
|
if not parquet_file.exists():
|
||||||
|
raise HTTPException(status_code=404, detail=f"No data found for geohash '{geohash}'.")
|
||||||
|
|
||||||
|
try:
|
||||||
|
# Load the Parquet file into a GeoDataFrame
|
||||||
|
gdf = gpd.read_parquet(parquet_file)
|
||||||
|
except Exception as e:
|
||||||
|
raise HTTPException(status_code=500, detail=f"Failed to read Parquet file: {str(e)}")
|
||||||
|
|
||||||
|
point = Point(longitude, latitude)
|
||||||
|
geod = Geod(ellps="WGS84") # Accurate geodetic distance calculations
|
||||||
|
|
||||||
|
found_geometries = []
|
||||||
|
|
||||||
|
# Process each row
|
||||||
|
for _, row in gdf.iterrows():
|
||||||
|
geom = row.geometry # Directly use the Shapely geometry object
|
||||||
|
|
||||||
|
metadata = {}
|
||||||
|
if isinstance(row.get("tags"), dict):
|
||||||
|
metadata = {k: v for k, v in row["tags"].items() if v is not None} # Remove empty values
|
||||||
|
|
||||||
|
if geom.contains(point):
|
||||||
|
found_geometries.append({"type": "polygon", **metadata})
|
||||||
|
elif isinstance(geom, (LineString, MultiLineString)):
|
||||||
|
try:
|
||||||
|
if isinstance(geom, LineString):
|
||||||
|
min_distance = min(geod.inv(point.x, point.y, p[0], p[1])[2] for p in geom.coords)
|
||||||
|
elif isinstance(geom, MultiLineString):
|
||||||
|
min_distance = min(
|
||||||
|
min(geod.inv(point.x, point.y, p[0], p[1])[2] for p in line.coords)
|
||||||
|
for line in geom.geoms
|
||||||
|
)
|
||||||
|
|
||||||
|
if min_distance <= distance:
|
||||||
|
found_geometries.append({"type": "line" if isinstance(geom, LineString) else "multi_line", **metadata})
|
||||||
|
except Exception as e:
|
||||||
|
raise HTTPException(status_code=500, detail=f"Failed distance calculation: {str(e)}")
|
||||||
|
|
||||||
|
# Return JSON result
|
||||||
|
return JSONResponse(content=found_geometries)
|
||||||
43
slice.py
43
slice.py
|
|
@ -1,5 +1,4 @@
|
||||||
import argparse
|
import argparse
|
||||||
import json
|
|
||||||
import duckdb
|
import duckdb
|
||||||
import geohash
|
import geohash
|
||||||
import geopandas as gpd
|
import geopandas as gpd
|
||||||
|
|
@ -9,12 +8,31 @@ from shapely import wkb
|
||||||
from shapely.geometry import box
|
from shapely.geometry import box
|
||||||
from tqdm import tqdm # For progress bar
|
from tqdm import tqdm # For progress bar
|
||||||
from typing import List, Dict, Optional
|
from typing import List, Dict, Optional
|
||||||
|
import json
|
||||||
|
|
||||||
# Configure logging
|
# Configure logging
|
||||||
import logging
|
import logging
|
||||||
logging.basicConfig(format="%(levelname)s: %(message)s", level=logging.INFO)
|
logging.basicConfig(format="%(levelname)s: %(message)s", level=logging.INFO)
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
def load_tags_config(tags_file: Path) -> Dict:
|
||||||
|
"""Loads the tags configuration from a JSON file."""
|
||||||
|
with open(tags_file, "r") as f:
|
||||||
|
return json.load(f)
|
||||||
|
|
||||||
|
def filter_tags(tags: Dict, tags_config: Dict) -> Dict:
|
||||||
|
"""Filters tags based on the rules defined in the tags configuration."""
|
||||||
|
filtered_tags = {}
|
||||||
|
for key, value in tags.items():
|
||||||
|
if key in tags_config:
|
||||||
|
if tags_config[key] is True:
|
||||||
|
# Keep all values for this key
|
||||||
|
filtered_tags[key] = value
|
||||||
|
elif isinstance(tags_config[key], list) and value in tags_config[key]:
|
||||||
|
# Keep only specific values for this key
|
||||||
|
filtered_tags[key] = value
|
||||||
|
return filtered_tags
|
||||||
|
|
||||||
def get_geohashes_from_bbox(min_x, min_y, max_x, max_y) -> List[str]:
|
def get_geohashes_from_bbox(min_x, min_y, max_x, max_y) -> List[str]:
|
||||||
"""Generates all 4-character Geohashes that intersect a bounding box."""
|
"""Generates all 4-character Geohashes that intersect a bounding box."""
|
||||||
geohashes = set()
|
geohashes = set()
|
||||||
|
|
@ -120,13 +138,21 @@ def add_water_to_geohash(geohash_code: str, geohash_file: Path, water_gdf: gpd.G
|
||||||
updated_gdf.to_parquet(geohash_file, index=False)
|
updated_gdf.to_parquet(geohash_file, index=False)
|
||||||
return True # Water polygons added
|
return True # Water polygons added
|
||||||
|
|
||||||
def slice_and_split_geoparquet(input_file: str, output_dir: str, water_file: Optional[str], export_parquet: bool, export_geojson: bool, export_cbor: bool):
|
def slice_and_split_geoparquet(input_file: str, output_dir: str, water_file: Optional[str], tags_file: Optional[str], export_parquet: bool, export_geojson: bool, export_cbor: bool):
|
||||||
"""Splits the GeoParquet file into multiple files based on 4-character Geohash tiles and optionally adds water polygons."""
|
"""Splits the GeoParquet file into multiple files based on 4-character Geohash tiles, optionally adds water polygons, and filters tags."""
|
||||||
input_path, output_path = Path(input_file), Path(output_dir)
|
input_path, output_path = Path(input_file), Path(output_dir)
|
||||||
if not input_path.exists():
|
if not input_path.exists():
|
||||||
raise FileNotFoundError(f"File '{input_file}' not found.")
|
raise FileNotFoundError(f"File '{input_file}' not found.")
|
||||||
output_path.mkdir(parents=True, exist_ok=True)
|
output_path.mkdir(parents=True, exist_ok=True)
|
||||||
|
|
||||||
|
# Load tags configuration if provided
|
||||||
|
tags_config = {}
|
||||||
|
if tags_file:
|
||||||
|
tags_path = Path(tags_file)
|
||||||
|
if not tags_path.exists():
|
||||||
|
raise FileNotFoundError(f"File '{tags_file}' not found.")
|
||||||
|
tags_config = load_tags_config(tags_path)
|
||||||
|
|
||||||
# Load water dataset if provided
|
# Load water dataset if provided
|
||||||
water_gdf = None
|
water_gdf = None
|
||||||
if water_file:
|
if water_file:
|
||||||
|
|
@ -191,6 +217,12 @@ def slice_and_split_geoparquet(input_file: str, output_dir: str, water_file: Opt
|
||||||
filtered_data = filtered_data.dropna(subset=["geometry"])
|
filtered_data = filtered_data.dropna(subset=["geometry"])
|
||||||
|
|
||||||
if not filtered_data.empty:
|
if not filtered_data.empty:
|
||||||
|
# Filter tags if tags configuration is provided
|
||||||
|
if tags_config:
|
||||||
|
filtered_data["tags"] = filtered_data["tags"].apply(
|
||||||
|
lambda tags: filter_tags(tags, tags_config) if isinstance(tags, dict) else {}
|
||||||
|
)
|
||||||
|
|
||||||
gdf = gpd.GeoDataFrame(filtered_data, geometry="geometry", crs="EPSG:4326")
|
gdf = gpd.GeoDataFrame(filtered_data, geometry="geometry", crs="EPSG:4326")
|
||||||
|
|
||||||
# Export to GeoParquet
|
# Export to GeoParquet
|
||||||
|
|
@ -214,10 +246,11 @@ def slice_and_split_geoparquet(input_file: str, output_dir: str, water_file: Opt
|
||||||
print(f"🎉 Processing complete! Generated {total_tiles} tiles, {tiles_with_water} of which had water polygons added.")
|
print(f"🎉 Processing complete! Generated {total_tiles} tiles, {tiles_with_water} of which had water polygons added.")
|
||||||
|
|
||||||
def main():
|
def main():
|
||||||
parser = argparse.ArgumentParser(description="Split a GeoParquet file into Geohash tiles, optionally add water polygons, and export in desired formats.")
|
parser = argparse.ArgumentParser(description="Split a GeoParquet file into Geohash tiles, optionally add water polygons, filter tags, and export in desired formats.")
|
||||||
parser.add_argument("-i", "--input", required=True, help="Path to the input GeoParquet file.")
|
parser.add_argument("-i", "--input", required=True, help="Path to the input GeoParquet file.")
|
||||||
parser.add_argument("-o", "--output", required=True, help="Directory to save output files.")
|
parser.add_argument("-o", "--output", required=True, help="Directory to save output files.")
|
||||||
parser.add_argument("-w", "--water", required=False, help="Path to the water.parquet file.")
|
parser.add_argument("-w", "--water", required=False, help="Path to the water.parquet file.")
|
||||||
|
parser.add_argument("-t", "--tags", required=False, help="Path to the tags.json file.")
|
||||||
parser.add_argument("--parquet", action="store_true", help="Export as GeoParquet")
|
parser.add_argument("--parquet", action="store_true", help="Export as GeoParquet")
|
||||||
parser.add_argument("--geojson", action="store_true", help="Export as GeoJSON")
|
parser.add_argument("--geojson", action="store_true", help="Export as GeoJSON")
|
||||||
parser.add_argument("--cbor", action="store_true", help="Export as CBOR")
|
parser.add_argument("--cbor", action="store_true", help="Export as CBOR")
|
||||||
|
|
@ -228,7 +261,7 @@ def main():
|
||||||
args.parquet = True # Default to GeoParquet if no options are given
|
args.parquet = True # Default to GeoParquet if no options are given
|
||||||
|
|
||||||
try:
|
try:
|
||||||
slice_and_split_geoparquet(args.input, args.output, args.water, args.parquet, args.geojson, args.cbor)
|
slice_and_split_geoparquet(args.input, args.output, args.water, args.tags, args.parquet, args.geojson, args.cbor)
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.error(f"❌ Error: {e}")
|
logger.error(f"❌ Error: {e}")
|
||||||
|
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue