diff --git a/import.py b/import.py deleted file mode 100644 index d2b6c81..0000000 --- a/import.py +++ /dev/null @@ -1,94 +0,0 @@ -import duckdb -from pathlib import Path -import geohash -import osmium -import subprocess -from pprint import pprint -from typing import List, Tuple, Dict - -def remove_points(input_file: str, output_file: str): - try: - input_path, output_path = Path(input_file), Path(output_file) - if not input_path.exists(): - raise FileNotFoundError(f"File '{input_file}' not found.") - - con = duckdb.connect() - con.execute("INSTALL spatial; LOAD spatial;") - - con.execute(f""" - CREATE VIEW geoparquet_filtered AS - SELECT * FROM read_parquet('{input_path}') - WHERE ST_GeometryType(geometry) IS NOT NULL - AND ST_GeometryType(geometry) != 'POINT' - """) - - if con.execute("SELECT COUNT(*) FROM geoparquet_filtered").fetchone()[0] == 0: - raise ValueError("No non-POINT geometries found.") - - output_path.parent.mkdir(parents=True, exist_ok=True) - con.execute(f"COPY geoparquet_filtered TO '{output_path}' (FORMAT PARQUET)") - print(f"✅ Saved to {output_path}") - - except Exception as e: - print(f"❌ Error: {e}") - - finally: - if 'con' in locals(): - con.close() - -def bbox(file_path: str) -> Dict[str, float]: - try: - result = subprocess.run( - ["osmium", "fileinfo", "-e", file_path], - capture_output=True, text=True, check=True - ) - for line in result.stdout.split("\n"): - if "Bounding box:" in line: - bbox_str = line.strip().replace("Bounding box:", "").strip() - # Convert to a tuple of floats - bbox_tuple = tuple(map(float, bbox_str.strip("()").split(","))) - return bbox_tuple - except subprocess.CalledProcessError as e: - print(f"❌ Error running osmium: {e}") - -def check_data(file: str): - con = duckdb.connect() - con.execute("INSTALL spatial; LOAD spatial;") - con.execute(f"CREATE VIEW geoparquet AS SELECT * FROM read_parquet('{file}')") - columns = con.execute("DESCRIBE geoparquet").fetchdf() - print("Columns in the file:\n", columns) - dc = con.execute("SELECT count(*) FROM geoparquet LIMIT 10").fetchdf() - print("\nTotal rows:", dc) - df = con.execute("SELECT * FROM geoparquet LIMIT 10").fetchdf() - print("\nFirst 10 rows:\n", df) - try: - geo_types = con.execute("SELECT DISTINCT ST_GeometryType(geometry) AS geom_type FROM geoparquet").fetchdf() - print("\nDetected Geometry Types:\n", geo_types) - except Exception as e: - print("\nCould not determine geometry type. Possible reason: No valid geometry column found.\n", e) - con.close() - -def bbox_geohashes(bbox: Tuple[float, float, float, float]) -> List[str]: - min_x, min_y, max_x, max_y = bbox - geohashes = set() - - # Define step size based on 4-character Geohash precision (~39km × 19.5km) - step_size = 0.25 # Approximate step size in degrees (varies slightly by latitude) - - lat = min_y - while lat <= max_y: - lon = min_x - while lon <= max_x: - gh = geohash.encode(lat, lon, precision=4) - geohashes.add(gh) - lon += step_size # Move eastward - lat += step_size # Move northward - - return sorted(geohashes) - -# remove_points('compact.parquet', 'processed.parquet') -# bbox = bbox('israel.pbf') -# hashes = bbox_geohashes(bbox) -# print(hashes) - -check_data("geohash/sv2c.parquet") \ No newline at end of file diff --git a/readme.md b/readme.md new file mode 100644 index 0000000..f75358f --- /dev/null +++ b/readme.md @@ -0,0 +1,136 @@ + +

 

+ +# dopecarpet + +Generate GeoHash parquet files from OpenStreetMap PBF files with custom tag filter to look up metadata for coordinates. + +## Dependencies + +Installation (Fedora) + +```bash +sudo dnf install osmium-tool jq python3-devel pip +pip install quackosm[cli] psycopg2-binary pyosmium geoalchemy2 geojson shapely pygeohash pyarrow pandas click +``` + +## Basic Use + +### 1. Customize Filter + +Edit the `tags.json` to specify which OSM tags you want to include in the metadata. This affects what geometries are extracted from the OSM data and has a direct influence on the file sizes. + +```json +{ + "name:en": true, + "name": true, + "aeroway": true, + "boundary": ["aboriginal_lands", "border_zone", "forest", "hazard", "national_park", "protected_area", "disputed"], + "geological": true, + "highway": ["motorway", "trunk", "primary", "raceway"], + "historic": true, + "landuse": true, + "leisure": ["dog_park", "garden", "miniature_golf", "nature_reserve", "park", "playground", "stadium", "swimming_pool", "water_park"], + "military": ["airfield", "base", "bunker", "barracks", "danger_area", "nuclear_explosion_site", "range", "training_area"], + "natural": true, + "place": ["plot", "farm", "island", "islet", "sea", "ocean"], + "power": ["plant"], + "railway": ["abandoned", "construction", "disused", "rail"], + "tourism": ["attraction", "yes"], + "water": true, + "waterway": true +} +``` + +### 2. Generate GeoHash Files + +This extracts the desired data from the PBF file and generates four digit geohash parquet files in the `geohash/` directory. + +```bash +just process +``` + +### 3. Lookup Coordinate Metadata + +Prints metadata found for a coordinate as JSON object list. + +```bash +just lookup , +``` + +Sample Output: `just lookup 52.496846793890256 13.435128880554208` + +```json +[ + { + "type": "polygon", + "leisure": "park", + "name": "Görlitzer Park" + }, + { + "type": "polygon", + "landuse": "forest" + }, + { + "type": "polygon", + "name": "Luisenstadt" + }, + { + "type": "polygon", + "name": "Jugendverkehrsschule Wiener Straße" + }, + { + "type": "polygon", + "name": "Friedrichshain-Kreuzberg" + }, + { + "type": "polygon", + "name": "Kreuzberg" + } +] +``` + +## Advanced Use + +### Extract and Filter Data + +To just extract the filtered data from an OSM PBF file and save it as parquet file + +```bash +just extract +``` + +### Download OSM Data File for Region + +Automatically download a file from Geofabrik based on the name of a region, filter it, and save it as parquet file. + +```bash +just download '' +``` + +The original PBF file remains saved in the `files/` folder. + +### Split Parquet File to GeoHash Parquet Files + +```bash +just convert +``` + +### Plot Parquet File as PNG + +Plot the geometry of a parquet file as a PNG with the same filename. + +```bash +just plot +``` + +Sample Output: +![snippet](u33d.png) + +## TO DO + +- [X] GeoHash parquet file generation +- [X] Append data to existing parquet files +- [ ] Create web API server +- [ ] Dockerize it +- [ ] Scheduled update routine \ No newline at end of file diff --git a/u33d.png b/u33d.png new file mode 100644 index 0000000..52a4a0a Binary files /dev/null and b/u33d.png differ