Split address pipeline: extract convert-addresses.py, simplify compare-addresses.py, delete dead scripts
This commit is contained in:
+278
-787
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,192 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""
|
||||||
|
Convert county address shapefile (ZIP) to OSM-formatted GeoJSON.
|
||||||
|
|
||||||
|
Applies LIFECYCLE filtering, CRS conversion, field mapping via qgis-functions,
|
||||||
|
and street name exceptions from /data/exceptions.yml.
|
||||||
|
|
||||||
|
Usage:
|
||||||
|
python convert-addresses.py /data/latest/sumter/addresses.shp.zip /data/latest/sumter/county-addresses.geojson
|
||||||
|
python convert-addresses.py /data/latest/lake/addresses.shp.zip /data/latest/lake/county-addresses.geojson
|
||||||
|
"""
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import shutil
|
||||||
|
import sys
|
||||||
|
import zipfile
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import geopandas as gpd
|
||||||
|
import pandas as pd
|
||||||
|
import importlib
|
||||||
|
import warnings
|
||||||
|
|
||||||
|
warnings.filterwarnings('ignore')
|
||||||
|
|
||||||
|
qgis_functions = importlib.import_module("qgis-functions")
|
||||||
|
|
||||||
|
|
||||||
|
def load_exceptions(exceptions_path='/data/exceptions.yml'):
|
||||||
|
path = Path(exceptions_path)
|
||||||
|
if not path.exists():
|
||||||
|
return {}
|
||||||
|
import yaml
|
||||||
|
with open(path) as f:
|
||||||
|
data = yaml.safe_load(f) or {}
|
||||||
|
return {
|
||||||
|
item['from']: item['to']
|
||||||
|
for item in data.get('corrections', [])
|
||||||
|
if 'from' in item and 'to' in item
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def process_address_fields(gdf, exceptions):
|
||||||
|
"""Map county shapefile fields to OSM address schema."""
|
||||||
|
processed = gdf.copy()
|
||||||
|
mapping = {}
|
||||||
|
|
||||||
|
# House number
|
||||||
|
for field in ['ADD_NUM', 'AddressNum', 'ADDRESS_NUM', 'HOUSE_NUM']:
|
||||||
|
if field in processed.columns:
|
||||||
|
series = pd.to_numeric(processed[field], errors='coerce')
|
||||||
|
mapping['addr:housenumber'] = series.round().astype('Int64')
|
||||||
|
break
|
||||||
|
|
||||||
|
# Unit
|
||||||
|
for field in ['UNIT', 'UnitNumber', 'UNIT_NUM', 'APT']:
|
||||||
|
if field in processed.columns:
|
||||||
|
series = processed[field].copy().replace(['nan', 'None', '', None], None)
|
||||||
|
mapping['addr:unit'] = series.where(series.notna(), None)
|
||||||
|
break
|
||||||
|
|
||||||
|
# Street name
|
||||||
|
if 'SADD' in processed.columns:
|
||||||
|
# Sumter: full address string in SADD
|
||||||
|
mapping['addr:street'] = [
|
||||||
|
qgis_functions.title(qgis_functions.getstreetfromaddress(str(v), None, None))
|
||||||
|
if pd.notna(v) else None
|
||||||
|
for v in processed['SADD']
|
||||||
|
]
|
||||||
|
elif 'FullAddres' in processed.columns:
|
||||||
|
# Lake: full address string in FullAddres
|
||||||
|
mapping['addr:street'] = [
|
||||||
|
qgis_functions.title(qgis_functions.getstreetfromaddress(str(v), None, None))
|
||||||
|
if pd.notna(v) else None
|
||||||
|
for v in processed['FullAddres']
|
||||||
|
]
|
||||||
|
elif 'BaseStreet' in processed.columns:
|
||||||
|
# Lake alternative: assemble from components
|
||||||
|
street_names = []
|
||||||
|
for _, row in processed.iterrows():
|
||||||
|
parts = []
|
||||||
|
for col in ['PrefixDire', 'PrefixType', 'BaseStreet', 'SuffixType']:
|
||||||
|
if col in row and pd.notna(row[col]):
|
||||||
|
parts.append(str(row[col]).strip())
|
||||||
|
street_names.append(qgis_functions.title(' '.join(parts)) if parts else None)
|
||||||
|
mapping['addr:street'] = street_names
|
||||||
|
|
||||||
|
# Apply street name exceptions
|
||||||
|
if 'addr:street' in mapping and exceptions:
|
||||||
|
mapping['addr:street'] = [
|
||||||
|
exceptions.get(name, name) if name is not None else None
|
||||||
|
for name in mapping['addr:street']
|
||||||
|
]
|
||||||
|
|
||||||
|
# City
|
||||||
|
for field in ['POST_COMM', 'PostalCity', 'CITY', 'Jurisdicti']:
|
||||||
|
if field in processed.columns:
|
||||||
|
mapping['addr:city'] = [
|
||||||
|
qgis_functions.title(str(v)) if pd.notna(v) else None
|
||||||
|
for v in processed[field]
|
||||||
|
]
|
||||||
|
break
|
||||||
|
|
||||||
|
# Postcode
|
||||||
|
for field in ['POST_CODE', 'ZipCode', 'ZIP', 'POSTAL_CODE']:
|
||||||
|
if field in processed.columns:
|
||||||
|
series = pd.to_numeric(processed[field], errors='coerce')
|
||||||
|
mapping['addr:postcode'] = series.round().astype('Int64')
|
||||||
|
break
|
||||||
|
|
||||||
|
mapping['addr:state'] = 'FL'
|
||||||
|
|
||||||
|
for key, value in mapping.items():
|
||||||
|
processed[key] = value
|
||||||
|
|
||||||
|
return processed
|
||||||
|
|
||||||
|
|
||||||
|
def convert(zip_path, output_path):
|
||||||
|
zip_path = Path(zip_path)
|
||||||
|
output_path = Path(output_path)
|
||||||
|
|
||||||
|
# Skip if output is newer than input
|
||||||
|
if (output_path.exists() and zip_path.exists() and
|
||||||
|
output_path.stat().st_mtime > zip_path.stat().st_mtime):
|
||||||
|
print(f"Output is up to date: {output_path}")
|
||||||
|
return
|
||||||
|
|
||||||
|
print(f"Converting {zip_path} ...")
|
||||||
|
|
||||||
|
exceptions = load_exceptions()
|
||||||
|
|
||||||
|
temp_dir = zip_path.parent / "temp_extract"
|
||||||
|
temp_dir.mkdir(exist_ok=True)
|
||||||
|
try:
|
||||||
|
with zipfile.ZipFile(zip_path, 'r') as zf:
|
||||||
|
zf.extractall(temp_dir)
|
||||||
|
|
||||||
|
shp_files = list(temp_dir.glob("*.shp"))
|
||||||
|
if not shp_files:
|
||||||
|
print("Error: no .shp file found in ZIP", file=sys.stderr)
|
||||||
|
sys.exit(1)
|
||||||
|
|
||||||
|
gdf = gpd.read_file(shp_files[0])
|
||||||
|
|
||||||
|
# LIFECYCLE filter (Sumter addresses use 'Current'; Lake has no such field)
|
||||||
|
LIFECYCLE_FIELD = 'LIFECYCLE'
|
||||||
|
ACTIVE_VALUE = 'Current'
|
||||||
|
if LIFECYCLE_FIELD in gdf.columns:
|
||||||
|
before = len(gdf)
|
||||||
|
gdf = gdf[gdf[LIFECYCLE_FIELD] == ACTIVE_VALUE].copy().reset_index(drop=True)
|
||||||
|
filtered = before - len(gdf)
|
||||||
|
if filtered:
|
||||||
|
print(f"Filtered out {filtered} non-active addresses (LIFECYCLE != '{ACTIVE_VALUE}')")
|
||||||
|
else:
|
||||||
|
print(f"All {len(gdf)} addresses are active")
|
||||||
|
|
||||||
|
# CRS conversion
|
||||||
|
if gdf.crs and gdf.crs != 'EPSG:4326':
|
||||||
|
print(f"Converting CRS from {gdf.crs} to EPSG:4326")
|
||||||
|
gdf = gdf.to_crs('EPSG:4326')
|
||||||
|
|
||||||
|
gdf = process_address_fields(gdf, exceptions)
|
||||||
|
|
||||||
|
# Points only, must have a house number
|
||||||
|
gdf = gdf[gdf.geometry.type == 'Point'].copy()
|
||||||
|
gdf = gdf[gdf['addr:housenumber'].notna()].copy()
|
||||||
|
|
||||||
|
osm_fields = ['addr:housenumber', 'addr:unit', 'addr:street',
|
||||||
|
'addr:city', 'addr:postcode', 'addr:state']
|
||||||
|
keep = [f for f in osm_fields if f in gdf.columns]
|
||||||
|
gdf = gdf[keep + ['geometry']]
|
||||||
|
|
||||||
|
output_path.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
gdf.to_file(output_path, driver='GeoJSON')
|
||||||
|
print(f"Saved {len(gdf)} addresses to {output_path}")
|
||||||
|
|
||||||
|
finally:
|
||||||
|
if temp_dir.exists():
|
||||||
|
shutil.rmtree(temp_dir)
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
parser = argparse.ArgumentParser(description="Convert county address shapefile to OSM-formatted GeoJSON")
|
||||||
|
parser.add_argument('input_zip', help='Path to county address ZIP (containing .shp)')
|
||||||
|
parser.add_argument('output_geojson', help='Output GeoJSON path')
|
||||||
|
args = parser.parse_args()
|
||||||
|
convert(args.input_zip, args.output_geojson)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == '__main__':
|
||||||
|
main()
|
||||||
@@ -1,41 +0,0 @@
|
|||||||
#!/usr/bin/env python3
|
|
||||||
"""
|
|
||||||
Simple wrapper script for comparing Lake County addresses
|
|
||||||
"""
|
|
||||||
|
|
||||||
import subprocess
|
|
||||||
import sys
|
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
def main():
|
|
||||||
# Change to script directory
|
|
||||||
script_dir = Path(__file__).parent
|
|
||||||
|
|
||||||
# Define the command
|
|
||||||
cmd = [
|
|
||||||
sys.executable,
|
|
||||||
"compare-addresses.py",
|
|
||||||
"Lake",
|
|
||||||
"Florida",
|
|
||||||
"--local-zip", "original data/Lake/Addresspoints 2025-06.zip",
|
|
||||||
"--tolerance", "50",
|
|
||||||
"--output-dir", "processed data/Lake"
|
|
||||||
]
|
|
||||||
|
|
||||||
print("Running Lake County address comparison...")
|
|
||||||
print("Command:", " ".join(cmd))
|
|
||||||
print()
|
|
||||||
|
|
||||||
# Run the command
|
|
||||||
result = subprocess.run(cmd, cwd=script_dir)
|
|
||||||
|
|
||||||
if result.returncode == 0:
|
|
||||||
print("\nAddress comparison completed successfully!")
|
|
||||||
print("Results saved in: processed data/Lake/")
|
|
||||||
else:
|
|
||||||
print(f"\nError: Script failed with return code {result.returncode}")
|
|
||||||
|
|
||||||
return result.returncode
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
sys.exit(main())
|
|
||||||
@@ -1,258 +0,0 @@
|
|||||||
#!/usr/bin/env python3
|
|
||||||
"""
|
|
||||||
Shapefile to GeoJSON Converter for Address Data
|
|
||||||
Converts ESRI:102659 CRS shapefile to EPSG:4326 GeoJSON with OSM-style address tags
|
|
||||||
"""
|
|
||||||
|
|
||||||
import geopandas as gpd
|
|
||||||
import json
|
|
||||||
import sys
|
|
||||||
import os
|
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
import importlib
|
|
||||||
qgis_functions = importlib.import_module("qgis-functions")
|
|
||||||
title = qgis_functions.title
|
|
||||||
getstreetfromaddress = qgis_functions.getstreetfromaddress
|
|
||||||
|
|
||||||
def convert_crs(gdf, source_crs='ESRI:102659', target_crs='EPSG:4326'):
|
|
||||||
"""
|
|
||||||
Convert coordinate reference system from source to target CRS
|
|
||||||
|
|
||||||
Args:
|
|
||||||
gdf: GeoDataFrame to convert
|
|
||||||
source_crs: Source coordinate reference system (default: ESRI:102659)
|
|
||||||
target_crs: Target coordinate reference system (default: EPSG:4326)
|
|
||||||
|
|
||||||
Returns:
|
|
||||||
GeoDataFrame with converted CRS
|
|
||||||
"""
|
|
||||||
if gdf.crs is None:
|
|
||||||
print(f"Warning: No CRS detected, assuming {source_crs}")
|
|
||||||
gdf.crs = source_crs
|
|
||||||
|
|
||||||
if gdf.crs != target_crs:
|
|
||||||
print(f"Converting from {gdf.crs} to {target_crs}")
|
|
||||||
gdf = gdf.to_crs(target_crs)
|
|
||||||
|
|
||||||
return gdf
|
|
||||||
|
|
||||||
def process_address_fields(gdf):
|
|
||||||
"""
|
|
||||||
Process and map address fields according to OSM address schema
|
|
||||||
|
|
||||||
Args:
|
|
||||||
gdf: GeoDataFrame with address data
|
|
||||||
|
|
||||||
Returns:
|
|
||||||
GeoDataFrame with processed address fields
|
|
||||||
"""
|
|
||||||
processed_gdf = gdf.copy()
|
|
||||||
|
|
||||||
# Create new columns for OSM address tags
|
|
||||||
address_mapping = {}
|
|
||||||
|
|
||||||
# ADD_NUM -> addr:housenumber (as integer)
|
|
||||||
if 'ADD_NUM' in processed_gdf.columns:
|
|
||||||
# Handle NaN values and convert to nullable integer
|
|
||||||
add_num_series = processed_gdf['ADD_NUM'].copy()
|
|
||||||
# Convert to numeric, coercing errors to NaN
|
|
||||||
add_num_series = pd.to_numeric(add_num_series, errors='coerce')
|
|
||||||
# Round to remove decimal places, then convert to nullable integer
|
|
||||||
address_mapping['addr:housenumber'] = add_num_series.round().astype('Int64')
|
|
||||||
|
|
||||||
# UNIT -> addr:unit (as string)
|
|
||||||
if 'UNIT' in processed_gdf.columns:
|
|
||||||
unit_series = processed_gdf['UNIT'].copy()
|
|
||||||
# Replace NaN, empty strings, and 'None' string with actual None
|
|
||||||
unit_series = unit_series.replace(['nan', 'None', '', None], None)
|
|
||||||
# Only keep non-null values as strings
|
|
||||||
unit_series = unit_series.where(unit_series.notna(), None)
|
|
||||||
address_mapping['addr:unit'] = unit_series
|
|
||||||
|
|
||||||
# SADD -> addr:street via title(getstreetfromaddress("SADD"))
|
|
||||||
if 'SADD' in processed_gdf.columns:
|
|
||||||
street_names = []
|
|
||||||
for sadd_value in processed_gdf['SADD']:
|
|
||||||
if pd.notna(sadd_value):
|
|
||||||
street_from_addr = getstreetfromaddress(str(sadd_value), None, None)
|
|
||||||
street_titled = title(street_from_addr)
|
|
||||||
street_names.append(street_titled)
|
|
||||||
else:
|
|
||||||
street_names.append(None)
|
|
||||||
address_mapping['addr:street'] = street_names
|
|
||||||
|
|
||||||
# POST_COMM -> addr:city via title("POST_COMM")
|
|
||||||
if 'POST_COMM' in processed_gdf.columns:
|
|
||||||
city_names = []
|
|
||||||
for post_comm in processed_gdf['POST_COMM']:
|
|
||||||
if pd.notna(post_comm):
|
|
||||||
city_titled = title(str(post_comm))
|
|
||||||
city_names.append(city_titled)
|
|
||||||
else:
|
|
||||||
city_names.append(None)
|
|
||||||
address_mapping['addr:city'] = city_names
|
|
||||||
|
|
||||||
# POST_CODE -> addr:postcode (as integer)
|
|
||||||
if 'POST_CODE' in processed_gdf.columns:
|
|
||||||
# Handle NaN values and convert to nullable integer
|
|
||||||
post_code_series = processed_gdf['POST_CODE'].copy()
|
|
||||||
# Convert to numeric, coercing errors to NaN
|
|
||||||
post_code_series = pd.to_numeric(post_code_series, errors='coerce')
|
|
||||||
# Round to remove decimal places, then convert to nullable integer
|
|
||||||
address_mapping['addr:postcode'] = post_code_series.round().astype('Int64')
|
|
||||||
|
|
||||||
# Manually add addr:state = 'FL'
|
|
||||||
address_mapping['addr:state'] = 'FL'
|
|
||||||
|
|
||||||
# Add the new address columns to the GeoDataFrame
|
|
||||||
for key, value in address_mapping.items():
|
|
||||||
processed_gdf[key] = value
|
|
||||||
|
|
||||||
return processed_gdf
|
|
||||||
|
|
||||||
def clean_output_data(gdf, keep_original_fields=False):
|
|
||||||
"""
|
|
||||||
Clean the output data, optionally keeping original fields
|
|
||||||
|
|
||||||
Args:
|
|
||||||
gdf: GeoDataFrame to clean
|
|
||||||
keep_original_fields: Whether to keep original shapefile fields
|
|
||||||
|
|
||||||
Returns:
|
|
||||||
Cleaned GeoDataFrame
|
|
||||||
"""
|
|
||||||
# Define the OSM address fields we want to keep
|
|
||||||
osm_fields = [
|
|
||||||
'addr:housenumber', 'addr:unit', 'addr:street',
|
|
||||||
'addr:city', 'addr:postcode', 'addr:state'
|
|
||||||
]
|
|
||||||
|
|
||||||
if keep_original_fields:
|
|
||||||
# Keep both original and OSM fields
|
|
||||||
original_fields = ['ADD_NUM', 'UNIT', 'SADD', 'POST_COMM', 'POST_CODE']
|
|
||||||
fields_to_keep = list(set(osm_fields + original_fields + ['geometry']))
|
|
||||||
else:
|
|
||||||
# Keep only OSM fields and geometry
|
|
||||||
fields_to_keep = osm_fields + ['geometry']
|
|
||||||
|
|
||||||
# Filter to only existing columns
|
|
||||||
existing_fields = [field for field in fields_to_keep if field in gdf.columns]
|
|
||||||
|
|
||||||
return gdf[existing_fields]
|
|
||||||
|
|
||||||
def convert_shapefile_to_geojson(
|
|
||||||
input_shapefile,
|
|
||||||
output_geojson,
|
|
||||||
keep_original_fields=False,
|
|
||||||
source_crs='ESRI:102659',
|
|
||||||
target_crs='EPSG:4326'
|
|
||||||
):
|
|
||||||
"""
|
|
||||||
Main conversion function
|
|
||||||
|
|
||||||
Args:
|
|
||||||
input_shapefile: Path to input shapefile
|
|
||||||
output_geojson: Path to output GeoJSON file
|
|
||||||
keep_original_fields: Whether to keep original shapefile fields
|
|
||||||
source_crs: Source coordinate reference system
|
|
||||||
target_crs: Target coordinate reference system
|
|
||||||
"""
|
|
||||||
try:
|
|
||||||
# Read shapefile
|
|
||||||
print(f"Reading shapefile: {input_shapefile}")
|
|
||||||
gdf = gpd.read_file(input_shapefile)
|
|
||||||
print(f"Loaded {len(gdf)} features")
|
|
||||||
|
|
||||||
# Display original columns
|
|
||||||
print(f"Original columns: {list(gdf.columns)}")
|
|
||||||
|
|
||||||
# Convert CRS if needed
|
|
||||||
gdf = convert_crs(gdf, source_crs, target_crs)
|
|
||||||
|
|
||||||
# Process address fields
|
|
||||||
print("Processing address fields...")
|
|
||||||
gdf = process_address_fields(gdf)
|
|
||||||
|
|
||||||
# Clean output data
|
|
||||||
gdf = clean_output_data(gdf, keep_original_fields)
|
|
||||||
|
|
||||||
# Remove rows with no valid geometry
|
|
||||||
gdf = gdf[gdf.geometry.notna()]
|
|
||||||
|
|
||||||
print(f"Final columns: {list(gdf.columns)}")
|
|
||||||
print(f"Final feature count: {len(gdf)}")
|
|
||||||
|
|
||||||
# Write to GeoJSON
|
|
||||||
print(f"Writing GeoJSON: {output_geojson}")
|
|
||||||
gdf.to_file(output_geojson, driver='GeoJSON')
|
|
||||||
|
|
||||||
print(f"Conversion completed successfully!")
|
|
||||||
|
|
||||||
# Display sample of processed data
|
|
||||||
if len(gdf) > 0:
|
|
||||||
print("\nSample of processed data:")
|
|
||||||
sample_cols = [col for col in gdf.columns if col.startswith('addr:')]
|
|
||||||
if sample_cols:
|
|
||||||
print(gdf[sample_cols].head())
|
|
||||||
|
|
||||||
except Exception as e:
|
|
||||||
print(f"Error during conversion: {str(e)}")
|
|
||||||
sys.exit(1)
|
|
||||||
|
|
||||||
def main():
|
|
||||||
"""
|
|
||||||
Main function to handle command line arguments
|
|
||||||
"""
|
|
||||||
import argparse
|
|
||||||
|
|
||||||
parser = argparse.ArgumentParser(
|
|
||||||
description='Convert shapefile to GeoJSON with OSM address tags'
|
|
||||||
)
|
|
||||||
parser.add_argument(
|
|
||||||
'input_shapefile',
|
|
||||||
help='Path to input shapefile'
|
|
||||||
)
|
|
||||||
parser.add_argument(
|
|
||||||
'output_geojson',
|
|
||||||
help='Path to output GeoJSON file'
|
|
||||||
)
|
|
||||||
parser.add_argument(
|
|
||||||
'--keep-original',
|
|
||||||
action='store_true',
|
|
||||||
help='Keep original shapefile fields in addition to OSM fields'
|
|
||||||
)
|
|
||||||
parser.add_argument(
|
|
||||||
'--source-crs',
|
|
||||||
default='ESRI:102659',
|
|
||||||
help='Source coordinate reference system (default: ESRI:102659)'
|
|
||||||
)
|
|
||||||
parser.add_argument(
|
|
||||||
'--target-crs',
|
|
||||||
default='EPSG:4326',
|
|
||||||
help='Target coordinate reference system (default: EPSG:4326)'
|
|
||||||
)
|
|
||||||
|
|
||||||
args = parser.parse_args()
|
|
||||||
|
|
||||||
# Validate input file
|
|
||||||
if not os.path.exists(args.input_shapefile):
|
|
||||||
print(f"Error: Input shapefile '{args.input_shapefile}' not found")
|
|
||||||
sys.exit(1)
|
|
||||||
|
|
||||||
# Create output directory if it doesn't exist
|
|
||||||
output_dir = Path(args.output_geojson).parent
|
|
||||||
output_dir.mkdir(parents=True, exist_ok=True)
|
|
||||||
|
|
||||||
# Run conversion
|
|
||||||
convert_shapefile_to_geojson(
|
|
||||||
args.input_shapefile,
|
|
||||||
args.output_geojson,
|
|
||||||
args.keep_original,
|
|
||||||
args.source_crs,
|
|
||||||
args.target_crs
|
|
||||||
)
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
import pandas as pd
|
|
||||||
main()
|
|
||||||
@@ -1,540 +0,0 @@
|
|||||||
#!/usr/bin/env python3
|
|
||||||
"""
|
|
||||||
GeoJSON Multi Modal Golf Cart Path Comparison Script
|
|
||||||
|
|
||||||
Compares two GeoJSON files containing road data and identifies:
|
|
||||||
1. Roads in file1 that don't have corresponding coverage in file2 (removed roads)
|
|
||||||
2. Roads in file2 that don't have corresponding coverage in file1 (added roads)
|
|
||||||
|
|
||||||
Only reports differences that are significant (above minimum length threshold).
|
|
||||||
Optimized for performance with parallel processing and spatial indexing.
|
|
||||||
|
|
||||||
TODO:
|
|
||||||
- put properties properly on removed roads, so they're visible in JOSM
|
|
||||||
- handle polygons properly (on previous geojson step?) for circular roads
|
|
||||||
"""
|
|
||||||
|
|
||||||
import json
|
|
||||||
import argparse
|
|
||||||
from pathlib import Path
|
|
||||||
from typing import List, Dict, Any, Tuple
|
|
||||||
import geopandas as gpd
|
|
||||||
from shapely.geometry import LineString, MultiLineString, Point, Polygon
|
|
||||||
from shapely.ops import unary_union
|
|
||||||
from shapely.strtree import STRtree
|
|
||||||
import pandas as pd
|
|
||||||
import warnings
|
|
||||||
import multiprocessing as mp
|
|
||||||
from functools import partial
|
|
||||||
import numpy as np
|
|
||||||
from concurrent.futures import ProcessPoolExecutor, as_completed
|
|
||||||
import gc
|
|
||||||
|
|
||||||
# Suppress warnings for cleaner output
|
|
||||||
warnings.filterwarnings('ignore')
|
|
||||||
|
|
||||||
class RoadComparator:
|
|
||||||
def __init__(self, tolerance_feet: float = 50.0, min_gap_length_feet: float = 100.0,
|
|
||||||
n_jobs: int = None, chunk_size: int = 1000):
|
|
||||||
"""
|
|
||||||
Initialize the road comparator.
|
|
||||||
|
|
||||||
Args:
|
|
||||||
tolerance_feet: Distance tolerance for considering roads as overlapping (default: 50 feet)
|
|
||||||
min_gap_length_feet: Minimum length of gap/extra to be considered significant (default: 100 feet)
|
|
||||||
n_jobs: Number of parallel processes to use (default: CPU count - 1)
|
|
||||||
chunk_size: Number of geometries to process per chunk (default: 1000)
|
|
||||||
"""
|
|
||||||
self.tolerance_feet = tolerance_feet
|
|
||||||
self.min_gap_length_feet = min_gap_length_feet
|
|
||||||
self.n_jobs = n_jobs or max(1, mp.cpu_count() - 1)
|
|
||||||
self.chunk_size = chunk_size
|
|
||||||
|
|
||||||
# Convert feet to degrees (approximate conversion for continental US)
|
|
||||||
# 1 degree latitude ≈ 364,000 feet
|
|
||||||
# 1 degree longitude ≈ 288,000 feet (at 40° latitude)
|
|
||||||
self.tolerance_deg = tolerance_feet / 364000.0
|
|
||||||
self.min_gap_length_deg = min_gap_length_feet / 364000.0
|
|
||||||
|
|
||||||
print(f"Using {self.n_jobs} parallel processes with chunk size {self.chunk_size}")
|
|
||||||
|
|
||||||
def load_geojson(self, filepath: str) -> gpd.GeoDataFrame:
|
|
||||||
"""Load and validate GeoJSON file with optimizations."""
|
|
||||||
try:
|
|
||||||
# Use pyogr engine for faster loading of large files
|
|
||||||
gdf = gpd.read_file(filepath, engine='pyogrio')
|
|
||||||
|
|
||||||
# Filter only LineString, MultiLineString, and Polygon geometries
|
|
||||||
line_types = ['LineString', 'MultiLineString', 'Polygon']
|
|
||||||
gdf = gdf[gdf.geometry.type.isin(line_types)].copy()
|
|
||||||
|
|
||||||
if len(gdf) == 0:
|
|
||||||
raise ValueError(f"No line geometries found in {filepath}")
|
|
||||||
|
|
||||||
# Reset index for efficient processing
|
|
||||||
gdf = gdf.reset_index(drop=True)
|
|
||||||
|
|
||||||
# Ensure geometry is valid and fix simple issues
|
|
||||||
invalid_mask = ~gdf.geometry.is_valid
|
|
||||||
if invalid_mask.any():
|
|
||||||
print(f"Fixing {invalid_mask.sum()} invalid geometries...")
|
|
||||||
gdf.loc[invalid_mask, 'geometry'] = gdf.loc[invalid_mask, 'geometry'].buffer(0)
|
|
||||||
|
|
||||||
print(f"Loaded {len(gdf)} road features from {filepath}")
|
|
||||||
return gdf
|
|
||||||
|
|
||||||
except Exception as e:
|
|
||||||
raise Exception(f"Error loading {filepath}: {str(e)}")
|
|
||||||
|
|
||||||
def create_buffered_union_optimized(self, gdf: gpd.GeoDataFrame) -> Any:
|
|
||||||
"""Create a buffered union using chunked processing for memory efficiency."""
|
|
||||||
print("Creating optimized buffered union...")
|
|
||||||
|
|
||||||
# Process in chunks to manage memory
|
|
||||||
chunks = [gdf.iloc[i:i+self.chunk_size] for i in range(0, len(gdf), self.chunk_size)]
|
|
||||||
chunk_unions = []
|
|
||||||
|
|
||||||
# Use partial function for multiprocessing
|
|
||||||
buffer_func = partial(self._buffer_chunk, tolerance=self.tolerance_deg)
|
|
||||||
|
|
||||||
with ProcessPoolExecutor(max_workers=self.n_jobs) as executor:
|
|
||||||
# Submit all chunk processing jobs
|
|
||||||
future_to_chunk = {executor.submit(buffer_func, chunk): i
|
|
||||||
for i, chunk in enumerate(chunks)}
|
|
||||||
|
|
||||||
# Collect results as they complete
|
|
||||||
for future in as_completed(future_to_chunk):
|
|
||||||
chunk_idx = future_to_chunk[future]
|
|
||||||
try:
|
|
||||||
chunk_union = future.result()
|
|
||||||
if chunk_union and not chunk_union.is_empty:
|
|
||||||
chunk_unions.append(chunk_union)
|
|
||||||
print(f"Processed chunk {chunk_idx + 1}/{len(chunks)}")
|
|
||||||
except Exception as e:
|
|
||||||
print(f"Error processing chunk {chunk_idx}: {str(e)}")
|
|
||||||
|
|
||||||
# Union all chunk results
|
|
||||||
print("Combining chunk unions...")
|
|
||||||
if chunk_unions:
|
|
||||||
final_union = unary_union(chunk_unions)
|
|
||||||
# Force garbage collection
|
|
||||||
del chunk_unions
|
|
||||||
gc.collect()
|
|
||||||
return final_union
|
|
||||||
else:
|
|
||||||
raise Exception("No valid geometries to create union")
|
|
||||||
|
|
||||||
@staticmethod
|
|
||||||
def _buffer_chunk(chunk_gdf: gpd.GeoDataFrame, tolerance: float) -> Any:
|
|
||||||
"""Buffer geometries in a chunk and return their union."""
|
|
||||||
try:
|
|
||||||
# Buffer all geometries in the chunk
|
|
||||||
buffered = chunk_gdf.geometry.buffer(tolerance)
|
|
||||||
|
|
||||||
# Create union of buffered geometries
|
|
||||||
if len(buffered) == 1:
|
|
||||||
return buffered.iloc[0]
|
|
||||||
else:
|
|
||||||
return unary_union(buffered.tolist())
|
|
||||||
except Exception as e:
|
|
||||||
print(f"Error in chunk processing: {str(e)}")
|
|
||||||
return None
|
|
||||||
|
|
||||||
def create_spatial_index(self, gdf: gpd.GeoDataFrame) -> STRtree:
|
|
||||||
"""Create spatial index for fast intersection queries."""
|
|
||||||
print("Creating spatial index...")
|
|
||||||
# Create STRtree for fast spatial queries
|
|
||||||
geometries = gdf.geometry.tolist()
|
|
||||||
return STRtree(geometries)
|
|
||||||
|
|
||||||
def find_removed_segments_optimized(self, source_gdf: gpd.GeoDataFrame,
|
|
||||||
target_union: Any) -> List[Dict[str, Any]]:
|
|
||||||
"""
|
|
||||||
Find segments in source_gdf that are not covered by target_union (removed roads).
|
|
||||||
Optimized with parallel processing.
|
|
||||||
"""
|
|
||||||
print("Finding removed segments...")
|
|
||||||
|
|
||||||
# Split into chunks for parallel processing
|
|
||||||
chunks = [source_gdf.iloc[i:i+self.chunk_size]
|
|
||||||
for i in range(0, len(source_gdf), self.chunk_size)]
|
|
||||||
|
|
||||||
all_removed = []
|
|
||||||
|
|
||||||
# Use partial function for multiprocessing
|
|
||||||
process_func = partial(self._process_removed_chunk,
|
|
||||||
target_union=target_union,
|
|
||||||
min_length_deg=self.min_gap_length_deg)
|
|
||||||
|
|
||||||
with ProcessPoolExecutor(max_workers=self.n_jobs) as executor:
|
|
||||||
# Submit all chunk processing jobs
|
|
||||||
future_to_chunk = {executor.submit(process_func, chunk): i
|
|
||||||
for i, chunk in enumerate(chunks)}
|
|
||||||
|
|
||||||
# Collect results as they complete
|
|
||||||
for future in as_completed(future_to_chunk):
|
|
||||||
chunk_idx = future_to_chunk[future]
|
|
||||||
try:
|
|
||||||
chunk_removed = future.result()
|
|
||||||
all_removed.extend(chunk_removed)
|
|
||||||
print(f"Processed removed chunk {chunk_idx + 1}/{len(chunks)}")
|
|
||||||
except Exception as e:
|
|
||||||
print(f"Error processing removed chunk {chunk_idx}: {str(e)}")
|
|
||||||
|
|
||||||
return all_removed
|
|
||||||
|
|
||||||
@staticmethod
|
|
||||||
def _process_removed_chunk(chunk_gdf: gpd.GeoDataFrame, target_union: Any,
|
|
||||||
min_length_deg: float) -> List[Dict[str, Any]]:
|
|
||||||
"""Process a chunk of geometries to find removed segments."""
|
|
||||||
removed_segments = []
|
|
||||||
|
|
||||||
for idx, row in chunk_gdf.iterrows():
|
|
||||||
geom = row.geometry
|
|
||||||
|
|
||||||
# Handle MultiLineString by processing each component
|
|
||||||
if isinstance(geom, MultiLineString):
|
|
||||||
lines = list(geom.geoms)
|
|
||||||
else:
|
|
||||||
lines = [geom] # Polygon and Line can be accessed directly
|
|
||||||
|
|
||||||
for line in lines:
|
|
||||||
try:
|
|
||||||
# Find parts of the line that don't intersect with target_union
|
|
||||||
uncovered = line.difference(target_union)
|
|
||||||
|
|
||||||
if uncovered.is_empty:
|
|
||||||
continue
|
|
||||||
|
|
||||||
# Handle different geometry types returned by difference
|
|
||||||
uncovered_lines = []
|
|
||||||
if hasattr(uncovered, 'geoms'):
|
|
||||||
for geom_part in uncovered.geoms:
|
|
||||||
if isinstance(geom_part, LineString):
|
|
||||||
uncovered_lines.append(geom_part)
|
|
||||||
elif isinstance(uncovered, LineString):
|
|
||||||
uncovered_lines.append(uncovered)
|
|
||||||
|
|
||||||
# Check each uncovered line segment
|
|
||||||
for uncovered_line in uncovered_lines:
|
|
||||||
if uncovered_line.length >= min_length_deg:
|
|
||||||
# Create properties dict with original metadata plus 'removed: true'
|
|
||||||
properties = dict(row.drop('geometry'))
|
|
||||||
properties['removed'] = True
|
|
||||||
|
|
||||||
removed_segments.append({
|
|
||||||
'geometry': uncovered_line,
|
|
||||||
**properties
|
|
||||||
})
|
|
||||||
|
|
||||||
except Exception as e:
|
|
||||||
continue # Skip problematic geometries
|
|
||||||
|
|
||||||
return removed_segments
|
|
||||||
|
|
||||||
def find_added_roads_optimized(self, source_gdf: gpd.GeoDataFrame,
|
|
||||||
target_union: Any) -> List[Dict[str, Any]]:
|
|
||||||
"""
|
|
||||||
Find entire roads in source_gdf that don't significantly overlap with target_union.
|
|
||||||
Optimized with parallel processing.
|
|
||||||
"""
|
|
||||||
print("Finding added roads...")
|
|
||||||
|
|
||||||
# Split into chunks for parallel processing
|
|
||||||
chunks = [source_gdf.iloc[i:i+self.chunk_size]
|
|
||||||
for i in range(0, len(source_gdf), self.chunk_size)]
|
|
||||||
|
|
||||||
all_added = []
|
|
||||||
|
|
||||||
# Use partial function for multiprocessing
|
|
||||||
process_func = partial(self._process_added_chunk,
|
|
||||||
target_union=target_union,
|
|
||||||
min_length_deg=self.min_gap_length_deg)
|
|
||||||
|
|
||||||
with ProcessPoolExecutor(max_workers=self.n_jobs) as executor:
|
|
||||||
# Submit all chunk processing jobs
|
|
||||||
future_to_chunk = {executor.submit(process_func, chunk): i
|
|
||||||
for i, chunk in enumerate(chunks)}
|
|
||||||
|
|
||||||
# Collect results as they complete
|
|
||||||
for future in as_completed(future_to_chunk):
|
|
||||||
chunk_idx = future_to_chunk[future]
|
|
||||||
try:
|
|
||||||
chunk_added = future.result()
|
|
||||||
all_added.extend(chunk_added)
|
|
||||||
print(f"Processed added chunk {chunk_idx + 1}/{len(chunks)}")
|
|
||||||
except Exception as e:
|
|
||||||
print(f"Error processing added chunk {chunk_idx}: {str(e)}")
|
|
||||||
|
|
||||||
return all_added
|
|
||||||
|
|
||||||
@staticmethod
|
|
||||||
def _process_added_chunk(chunk_gdf: gpd.GeoDataFrame, target_union: Any,
|
|
||||||
min_length_deg: float) -> List[Dict[str, Any]]:
|
|
||||||
"""Process a chunk of geometries to find added roads."""
|
|
||||||
added_roads = []
|
|
||||||
|
|
||||||
for idx, row in chunk_gdf.iterrows():
|
|
||||||
geom = row.geometry
|
|
||||||
|
|
||||||
try:
|
|
||||||
# Check what portion of the road is not covered
|
|
||||||
uncovered = geom.difference(target_union)
|
|
||||||
|
|
||||||
if not uncovered.is_empty:
|
|
||||||
# Calculate what percentage of the original road is uncovered
|
|
||||||
uncovered_length = 0
|
|
||||||
if hasattr(uncovered, 'geoms'):
|
|
||||||
for geom_part in uncovered.geoms:
|
|
||||||
if isinstance(geom_part, LineString):
|
|
||||||
uncovered_length += geom_part.length
|
|
||||||
elif isinstance(uncovered, LineString):
|
|
||||||
uncovered_length = uncovered.length
|
|
||||||
|
|
||||||
original_length = geom.length
|
|
||||||
uncovered_ratio = uncovered_length / original_length if original_length > 0 else 0
|
|
||||||
|
|
||||||
# Include the entire road if:
|
|
||||||
# 1. The uncovered portion is above minimum threshold, AND
|
|
||||||
# 2. More than 10% of the road is uncovered
|
|
||||||
if uncovered_ratio > 0.1:
|
|
||||||
#uncovered_length >= min_length_deg and
|
|
||||||
# Include entire original road with all original metadata
|
|
||||||
original_properties = dict(row.drop('geometry'))
|
|
||||||
|
|
||||||
#
|
|
||||||
# For Sumter County Roads
|
|
||||||
#
|
|
||||||
properties = {
|
|
||||||
'surface': 'asphalt'
|
|
||||||
}
|
|
||||||
|
|
||||||
output = True
|
|
||||||
|
|
||||||
for key, value in original_properties.items():
|
|
||||||
if key == 'Part_of_Ro' and value == "Yes":
|
|
||||||
output = False
|
|
||||||
continue # Skip cart paths that are parts of roads
|
|
||||||
else:
|
|
||||||
properties['highway'] = 'residential'
|
|
||||||
properties['bicycle'] = 'yes'
|
|
||||||
properties['foot'] = 'yes'
|
|
||||||
properties['golf'] = 'cartpath'
|
|
||||||
properties['golf_cart'] = 'yes'
|
|
||||||
properties['highway'] = 'path'
|
|
||||||
properties['motor_vehicle'] = 'no'
|
|
||||||
properties['segregated'] = 'no'
|
|
||||||
properties['surface'] = 'asphalt'
|
|
||||||
|
|
||||||
if output:
|
|
||||||
added_roads.append({
|
|
||||||
'geometry': geom,
|
|
||||||
**properties
|
|
||||||
})
|
|
||||||
|
|
||||||
except Exception as e:
|
|
||||||
print(e)
|
|
||||||
continue # Skip problematic geometries
|
|
||||||
|
|
||||||
return added_roads
|
|
||||||
|
|
||||||
def compare_roads(self, file1_path: str, file2_path: str) -> Tuple[List[Dict], List[Dict]]:
|
|
||||||
"""
|
|
||||||
Compare two GeoJSON files and find significant differences.
|
|
||||||
Optimized version with parallel processing.
|
|
||||||
|
|
||||||
Returns:
|
|
||||||
Tuple of (removed_roads, added_roads)
|
|
||||||
"""
|
|
||||||
print(f"Comparing {file1_path} and {file2_path}")
|
|
||||||
print(f"Tolerance: {self.tolerance_feet} feet")
|
|
||||||
print(f"Minimum significant length: {self.min_gap_length_feet} feet")
|
|
||||||
print(f"Parallel processing: {self.n_jobs} workers")
|
|
||||||
print("-" * 50)
|
|
||||||
|
|
||||||
# Load both files
|
|
||||||
gdf1 = self.load_geojson(file1_path)
|
|
||||||
gdf2 = self.load_geojson(file2_path)
|
|
||||||
|
|
||||||
# Ensure both are in the same CRS
|
|
||||||
if gdf1.crs != gdf2.crs:
|
|
||||||
print(f"Warning: CRS mismatch. Converting {file2_path} to match {file1_path}")
|
|
||||||
gdf2 = gdf2.to_crs(gdf1.crs)
|
|
||||||
|
|
||||||
print("Creating optimized spatial unions...")
|
|
||||||
|
|
||||||
# Create buffered unions using optimized method
|
|
||||||
union1 = self.create_buffered_union_optimized(gdf1)
|
|
||||||
union2 = self.create_buffered_union_optimized(gdf2)
|
|
||||||
|
|
||||||
print("Finding removed and added roads with parallel processing...")
|
|
||||||
|
|
||||||
# Find roads using optimized parallel methods
|
|
||||||
removed_roads = self.find_removed_segments_optimized(gdf1, union2)
|
|
||||||
added_roads = self.find_added_roads_optimized(gdf2, union1)
|
|
||||||
|
|
||||||
# Clean up memory
|
|
||||||
del gdf1, gdf2, union1, union2
|
|
||||||
gc.collect()
|
|
||||||
|
|
||||||
return removed_roads, added_roads
|
|
||||||
|
|
||||||
def save_results(self, removed: List[Dict], added: List[Dict], output_path: str):
|
|
||||||
"""Save results to GeoJSON file."""
|
|
||||||
all_results = removed + added
|
|
||||||
|
|
||||||
if not all_results:
|
|
||||||
print("No significant differences found!")
|
|
||||||
return
|
|
||||||
|
|
||||||
# Create GeoDataFrame efficiently
|
|
||||||
print("Saving results...")
|
|
||||||
results_gdf = gpd.GeoDataFrame(all_results)
|
|
||||||
|
|
||||||
# Save to file with optimization
|
|
||||||
results_gdf.to_file(output_path, driver='GeoJSON', engine='pyogrio')
|
|
||||||
print(f"Results saved to: {output_path}")
|
|
||||||
|
|
||||||
def print_summary(self, removed: List[Dict], added: List[Dict], file1_name: str, file2_name: str):
|
|
||||||
"""Print a summary of the comparison results."""
|
|
||||||
print("\n" + "="*60)
|
|
||||||
print("COMPARISON SUMMARY")
|
|
||||||
print("="*60)
|
|
||||||
|
|
||||||
print(f"\nFile 1: {file1_name}")
|
|
||||||
print(f"File 2: {file2_name}")
|
|
||||||
print(f"Tolerance: {self.tolerance_feet} feet")
|
|
||||||
print(f"Minimum significant length: {self.min_gap_length_feet} feet")
|
|
||||||
|
|
||||||
if removed:
|
|
||||||
print(f"\n🔴 REMOVED ROADS ({len(removed)} segments):")
|
|
||||||
print("These road segments exist in File 1 but are missing or incomplete in File 2:")
|
|
||||||
|
|
||||||
# Calculate total length of removed segments
|
|
||||||
total_removed_length = 0
|
|
||||||
removed_by_road = {}
|
|
||||||
|
|
||||||
for segment in removed:
|
|
||||||
geom = segment['geometry']
|
|
||||||
length_feet = geom.length * 364000.0 # Convert to feet
|
|
||||||
total_removed_length += length_feet
|
|
||||||
|
|
||||||
# Get road name
|
|
||||||
road_name = "Unknown"
|
|
||||||
name_fields = ['name', 'NAME', 'road_name', 'street_name', 'FULLNAME']
|
|
||||||
for field in name_fields:
|
|
||||||
if field in segment and pd.notna(segment[field]):
|
|
||||||
road_name = str(segment[field])
|
|
||||||
break
|
|
||||||
if road_name not in removed_by_road:
|
|
||||||
removed_by_road[road_name] = []
|
|
||||||
removed_by_road[road_name].append(length_feet)
|
|
||||||
|
|
||||||
print(f"Total removed length: {total_removed_length:,.1f} feet ({total_removed_length/5280:.2f} miles)")
|
|
||||||
|
|
||||||
for road, lengths in sorted(removed_by_road.items()):
|
|
||||||
road_total = sum(lengths)
|
|
||||||
print(f" • {road}: {len(lengths)} segment(s), {road_total:,.1f} feet")
|
|
||||||
|
|
||||||
if added:
|
|
||||||
print(f"\n🔵 ADDED ROADS ({len(added)} roads):")
|
|
||||||
print("These roads exist in File 2 but are missing or incomplete in File 1:")
|
|
||||||
|
|
||||||
# Calculate total length of added roads
|
|
||||||
total_added_length = 0
|
|
||||||
added_by_road = {}
|
|
||||||
|
|
||||||
for road in added:
|
|
||||||
geom = road['geometry']
|
|
||||||
length_feet = geom.length * 364000.0 # Convert to feet
|
|
||||||
total_added_length += length_feet
|
|
||||||
|
|
||||||
# Get road name
|
|
||||||
road_name = "Unknown"
|
|
||||||
name_fields = ['name', 'NAME', 'road_name', 'street_name', 'FULLNAME']
|
|
||||||
for field in name_fields:
|
|
||||||
if field in road and pd.notna(road[field]):
|
|
||||||
road_name = str(road[field])
|
|
||||||
break
|
|
||||||
|
|
||||||
if road_name not in added_by_road:
|
|
||||||
added_by_road[road_name] = 0
|
|
||||||
added_by_road[road_name] += length_feet
|
|
||||||
|
|
||||||
print(f"Total added length: {total_added_length:,.1f} feet ({total_added_length/5280:.2f} miles)")
|
|
||||||
|
|
||||||
for road, length in sorted(added_by_road.items()):
|
|
||||||
print(f" • {road}: {length:,.1f} feet")
|
|
||||||
|
|
||||||
if not removed and not added:
|
|
||||||
print("\n✅ No significant differences found!")
|
|
||||||
print("The road networks have good coverage overlap within the specified tolerance.")
|
|
||||||
|
|
||||||
|
|
||||||
def main():
|
|
||||||
parser = argparse.ArgumentParser(
|
|
||||||
description="Compare two GeoJSON files containing roads and find significant gaps or extras (Optimized)",
|
|
||||||
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
||||||
epilog="""
|
|
||||||
Examples:
|
|
||||||
python sumter-multi-modal-convert.py osm-multi-modal.geojson county-multi-modal.geojson
|
|
||||||
python sumter-multi-modal-convert.py osm-multi-modal.geojson county-multi-modal.geojson --tolerance 100 --min-length 200
|
|
||||||
python sumter-multi-modal-convert.py osm-multi-modal.geojson county-multi-modal.geojson --output differences.geojson
|
|
||||||
python sumter-multi-modal-convert.py osm-multi-modal.geojson county-multi-modal.geojson --jobs 8 --chunk-size 2000
|
|
||||||
"""
|
|
||||||
)
|
|
||||||
|
|
||||||
parser.add_argument('file1', help='First GeoJSON file')
|
|
||||||
parser.add_argument('file2', help='Second GeoJSON file')
|
|
||||||
parser.add_argument('--tolerance', '-t', type=float, default=50.0,
|
|
||||||
help='Distance tolerance in feet for considering roads as overlapping (default: 50)')
|
|
||||||
parser.add_argument('--min-length', '-m', type=float, default=100.0,
|
|
||||||
help='Minimum length in feet for gaps/extras to be considered significant (default: 100)')
|
|
||||||
parser.add_argument('--output', '-o', help='Output GeoJSON file for results (optional)')
|
|
||||||
parser.add_argument('--jobs', '-j', type=int, default=None,
|
|
||||||
help='Number of parallel processes (default: CPU count - 1)')
|
|
||||||
parser.add_argument('--chunk-size', '-c', type=int, default=1000,
|
|
||||||
help='Number of geometries to process per chunk (default: 1000)')
|
|
||||||
|
|
||||||
args = parser.parse_args()
|
|
||||||
|
|
||||||
# Validate input files
|
|
||||||
if not Path(args.file1).exists():
|
|
||||||
print(f"Error: File {args.file1} does not exist")
|
|
||||||
return 1
|
|
||||||
|
|
||||||
if not Path(args.file2).exists():
|
|
||||||
print(f"Error: File {args.file2} does not exist")
|
|
||||||
return 1
|
|
||||||
|
|
||||||
try:
|
|
||||||
# Create comparator and run comparison
|
|
||||||
comparator = RoadComparator(
|
|
||||||
tolerance_feet=args.tolerance,
|
|
||||||
min_gap_length_feet=args.min_length,
|
|
||||||
n_jobs=args.jobs,
|
|
||||||
chunk_size=args.chunk_size
|
|
||||||
)
|
|
||||||
|
|
||||||
removed, added = comparator.compare_roads(args.file1, args.file2)
|
|
||||||
|
|
||||||
# Print summary
|
|
||||||
comparator.print_summary(removed, added, args.file1, args.file2)
|
|
||||||
|
|
||||||
# Save results if output file specified
|
|
||||||
if args.output:
|
|
||||||
comparator.save_results(removed, added, args.output)
|
|
||||||
elif removed or added:
|
|
||||||
# Auto-generate output filename if differences found
|
|
||||||
output_file = f"multi_modal_differences_{Path(args.file1).stem}_vs_{Path(args.file2).stem}.geojson"
|
|
||||||
comparator.save_results(removed, added, output_file)
|
|
||||||
|
|
||||||
return 0
|
|
||||||
|
|
||||||
except Exception as e:
|
|
||||||
print(f"Error: {str(e)}")
|
|
||||||
return 1
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
exit(main())
|
|
||||||
+6
-14
@@ -93,27 +93,22 @@ def get_script_map():
|
|||||||
'lake': ['python', 'shp-to-geojson.py', '/data/latest/lake/roads.shp.zip', '/data/latest/lake/county-roads.geojson']
|
'lake': ['python', 'shp-to-geojson.py', '/data/latest/lake/roads.shp.zip', '/data/latest/lake/county-roads.geojson']
|
||||||
},
|
},
|
||||||
'convert-addresses': {
|
'convert-addresses': {
|
||||||
'sumter': ['python', 'shp-to-geojson.py', '/data/latest/sumter/addresses.shp.zip', '/data/latest/sumter/county-addresses.geojson'],
|
'sumter': ['python', 'convert-addresses.py', '/data/latest/sumter/addresses.shp.zip', '/data/latest/sumter/county-addresses.geojson'],
|
||||||
'lake': ['python', 'shp-to-geojson.py', '/data/latest/lake/addresses.shp.zip', '/data/latest/lake/county-addresses.geojson']
|
'lake': ['python', 'convert-addresses.py', '/data/latest/lake/addresses.shp.zip', '/data/latest/lake/county-addresses.geojson'],
|
||||||
},
|
},
|
||||||
'convert-paths': {
|
'convert-paths': {
|
||||||
#todo: delete sumter-multi-modal-convert.py ?
|
|
||||||
'sumter': ['python', 'shp-to-geojson.py', '/data/latest/sumter/paths.shp.zip', '/data/latest/sumter/county-paths.geojson'],
|
'sumter': ['python', 'shp-to-geojson.py', '/data/latest/sumter/paths.shp.zip', '/data/latest/sumter/county-paths.geojson'],
|
||||||
},
|
},
|
||||||
'diff-roads': {
|
'diff-roads': {
|
||||||
'lake': ['python', 'diff-highways.py', '/data/latest/lake/osm-roads.geojson', '/data/latest/lake/county-roads.geojson', '--output', '/data/latest/lake/diff-roads.geojson'],
|
'lake': ['python', 'diff-highways.py', '/data/latest/lake/osm-roads.geojson', '/data/latest/lake/county-roads.geojson', '--output', '/data/latest/lake/diff-roads.geojson'],
|
||||||
'sumter': ['python', 'diff-highways.py', '/data/latest/sumter/osm-roads.geojson', '/data/latest/sumter/county-roads.geojson', '--output', '/data/latest/sumter/diff-roads.geojson']
|
'sumter': ['python', 'diff-highways.py', '/data/latest/sumter/osm-roads.geojson', '/data/latest/sumter/county-roads.geojson', '--output', '/data/latest/sumter/diff-roads.geojson'],
|
||||||
},
|
},
|
||||||
'diff-paths': {
|
'diff-paths': {
|
||||||
#todo: no lake county data for paths
|
|
||||||
#'lake': ['python', 'diff-highways.py', '/data/latest/lake/osm-paths.geojson', '/data/latest/lake/county-paths.geojson', '--output', '/data/latest/lake/diff-paths.geojson'],
|
|
||||||
'sumter': ['python', 'diff-highways.py', '/data/latest/sumter/osm-paths.geojson', '/data/latest/sumter/county-paths.geojson', '--output', '/data/latest/sumter/diff-paths.geojson'],
|
'sumter': ['python', 'diff-highways.py', '/data/latest/sumter/osm-paths.geojson', '/data/latest/sumter/county-paths.geojson', '--output', '/data/latest/sumter/diff-paths.geojson'],
|
||||||
},
|
},
|
||||||
# addresses need no osm download or shapefile convert, just county download
|
|
||||||
'diff-addresses': {
|
'diff-addresses': {
|
||||||
#todo: delete sumter-address-convert.py ?
|
'lake': ['python', 'compare-addresses.py', '--local-file', '/data/latest/lake/county-addresses.geojson', '--osm-file', '/data/latest/lake/osm-addresses.geojson', '--output-dir', '/data/latest/lake'],
|
||||||
'lake': ['python', 'compare-addresses.py', 'Lake', 'Florida', '--local-zip', '/data/latest/lake/addresses.shp.zip', '--output-dir', '/data/latest/lake', '--cache-dir', '/data/osm_cache'],
|
'sumter': ['python', 'compare-addresses.py', '--local-file', '/data/latest/sumter/county-addresses.geojson', '--osm-file', '/data/latest/sumter/osm-addresses.geojson', '--output-dir', '/data/latest/sumter'],
|
||||||
'sumter': ['python', 'compare-addresses.py', 'Sumter', 'Florida', '--local-zip', '/data/latest/sumter/addresses.shp.zip', '--output-dir', '/data/latest/sumter', '--cache-dir', '/data/osm_cache']
|
|
||||||
},
|
},
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -165,9 +160,6 @@ def run_script():
|
|||||||
else:
|
else:
|
||||||
cmd = list(cmd_config) # Make a copy to avoid modifying the original
|
cmd = list(cmd_config) # Make a copy to avoid modifying the original
|
||||||
|
|
||||||
# Add --force-download flag for diff-addresses if requested
|
|
||||||
if script_name == 'diff-addresses' and force_download and isinstance(cmd, list):
|
|
||||||
cmd.append('--force-download')
|
|
||||||
else:
|
else:
|
||||||
return jsonify({'error': 'Invalid script configuration'}), 400
|
return jsonify({'error': 'Invalid script configuration'}), 400
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user