Split address pipeline: extract convert-addresses.py, simplify compare-addresses.py, delete dead scripts

This commit is contained in:
zyphlar
2026-04-09 21:55:13 -07:00
parent e8267b1eee
commit f3a1ca6020
6 changed files with 476 additions and 1640 deletions
+278 -787
View File
File diff suppressed because it is too large Load Diff
+192
View File
@@ -0,0 +1,192 @@
#!/usr/bin/env python3
"""
Convert county address shapefile (ZIP) to OSM-formatted GeoJSON.
Applies LIFECYCLE filtering, CRS conversion, field mapping via qgis-functions,
and street name exceptions from /data/exceptions.yml.
Usage:
python convert-addresses.py /data/latest/sumter/addresses.shp.zip /data/latest/sumter/county-addresses.geojson
python convert-addresses.py /data/latest/lake/addresses.shp.zip /data/latest/lake/county-addresses.geojson
"""
import argparse
import shutil
import sys
import zipfile
from pathlib import Path
import geopandas as gpd
import pandas as pd
import importlib
import warnings
warnings.filterwarnings('ignore')
qgis_functions = importlib.import_module("qgis-functions")
def load_exceptions(exceptions_path='/data/exceptions.yml'):
path = Path(exceptions_path)
if not path.exists():
return {}
import yaml
with open(path) as f:
data = yaml.safe_load(f) or {}
return {
item['from']: item['to']
for item in data.get('corrections', [])
if 'from' in item and 'to' in item
}
def process_address_fields(gdf, exceptions):
"""Map county shapefile fields to OSM address schema."""
processed = gdf.copy()
mapping = {}
# House number
for field in ['ADD_NUM', 'AddressNum', 'ADDRESS_NUM', 'HOUSE_NUM']:
if field in processed.columns:
series = pd.to_numeric(processed[field], errors='coerce')
mapping['addr:housenumber'] = series.round().astype('Int64')
break
# Unit
for field in ['UNIT', 'UnitNumber', 'UNIT_NUM', 'APT']:
if field in processed.columns:
series = processed[field].copy().replace(['nan', 'None', '', None], None)
mapping['addr:unit'] = series.where(series.notna(), None)
break
# Street name
if 'SADD' in processed.columns:
# Sumter: full address string in SADD
mapping['addr:street'] = [
qgis_functions.title(qgis_functions.getstreetfromaddress(str(v), None, None))
if pd.notna(v) else None
for v in processed['SADD']
]
elif 'FullAddres' in processed.columns:
# Lake: full address string in FullAddres
mapping['addr:street'] = [
qgis_functions.title(qgis_functions.getstreetfromaddress(str(v), None, None))
if pd.notna(v) else None
for v in processed['FullAddres']
]
elif 'BaseStreet' in processed.columns:
# Lake alternative: assemble from components
street_names = []
for _, row in processed.iterrows():
parts = []
for col in ['PrefixDire', 'PrefixType', 'BaseStreet', 'SuffixType']:
if col in row and pd.notna(row[col]):
parts.append(str(row[col]).strip())
street_names.append(qgis_functions.title(' '.join(parts)) if parts else None)
mapping['addr:street'] = street_names
# Apply street name exceptions
if 'addr:street' in mapping and exceptions:
mapping['addr:street'] = [
exceptions.get(name, name) if name is not None else None
for name in mapping['addr:street']
]
# City
for field in ['POST_COMM', 'PostalCity', 'CITY', 'Jurisdicti']:
if field in processed.columns:
mapping['addr:city'] = [
qgis_functions.title(str(v)) if pd.notna(v) else None
for v in processed[field]
]
break
# Postcode
for field in ['POST_CODE', 'ZipCode', 'ZIP', 'POSTAL_CODE']:
if field in processed.columns:
series = pd.to_numeric(processed[field], errors='coerce')
mapping['addr:postcode'] = series.round().astype('Int64')
break
mapping['addr:state'] = 'FL'
for key, value in mapping.items():
processed[key] = value
return processed
def convert(zip_path, output_path):
zip_path = Path(zip_path)
output_path = Path(output_path)
# Skip if output is newer than input
if (output_path.exists() and zip_path.exists() and
output_path.stat().st_mtime > zip_path.stat().st_mtime):
print(f"Output is up to date: {output_path}")
return
print(f"Converting {zip_path} ...")
exceptions = load_exceptions()
temp_dir = zip_path.parent / "temp_extract"
temp_dir.mkdir(exist_ok=True)
try:
with zipfile.ZipFile(zip_path, 'r') as zf:
zf.extractall(temp_dir)
shp_files = list(temp_dir.glob("*.shp"))
if not shp_files:
print("Error: no .shp file found in ZIP", file=sys.stderr)
sys.exit(1)
gdf = gpd.read_file(shp_files[0])
# LIFECYCLE filter (Sumter addresses use 'Current'; Lake has no such field)
LIFECYCLE_FIELD = 'LIFECYCLE'
ACTIVE_VALUE = 'Current'
if LIFECYCLE_FIELD in gdf.columns:
before = len(gdf)
gdf = gdf[gdf[LIFECYCLE_FIELD] == ACTIVE_VALUE].copy().reset_index(drop=True)
filtered = before - len(gdf)
if filtered:
print(f"Filtered out {filtered} non-active addresses (LIFECYCLE != '{ACTIVE_VALUE}')")
else:
print(f"All {len(gdf)} addresses are active")
# CRS conversion
if gdf.crs and gdf.crs != 'EPSG:4326':
print(f"Converting CRS from {gdf.crs} to EPSG:4326")
gdf = gdf.to_crs('EPSG:4326')
gdf = process_address_fields(gdf, exceptions)
# Points only, must have a house number
gdf = gdf[gdf.geometry.type == 'Point'].copy()
gdf = gdf[gdf['addr:housenumber'].notna()].copy()
osm_fields = ['addr:housenumber', 'addr:unit', 'addr:street',
'addr:city', 'addr:postcode', 'addr:state']
keep = [f for f in osm_fields if f in gdf.columns]
gdf = gdf[keep + ['geometry']]
output_path.parent.mkdir(parents=True, exist_ok=True)
gdf.to_file(output_path, driver='GeoJSON')
print(f"Saved {len(gdf)} addresses to {output_path}")
finally:
if temp_dir.exists():
shutil.rmtree(temp_dir)
def main():
parser = argparse.ArgumentParser(description="Convert county address shapefile to OSM-formatted GeoJSON")
parser.add_argument('input_zip', help='Path to county address ZIP (containing .shp)')
parser.add_argument('output_geojson', help='Output GeoJSON path')
args = parser.parse_args()
convert(args.input_zip, args.output_geojson)
if __name__ == '__main__':
main()
-41
View File
@@ -1,41 +0,0 @@
#!/usr/bin/env python3
"""
Simple wrapper script for comparing Lake County addresses
"""
import subprocess
import sys
from pathlib import Path
def main():
# Change to script directory
script_dir = Path(__file__).parent
# Define the command
cmd = [
sys.executable,
"compare-addresses.py",
"Lake",
"Florida",
"--local-zip", "original data/Lake/Addresspoints 2025-06.zip",
"--tolerance", "50",
"--output-dir", "processed data/Lake"
]
print("Running Lake County address comparison...")
print("Command:", " ".join(cmd))
print()
# Run the command
result = subprocess.run(cmd, cwd=script_dir)
if result.returncode == 0:
print("\nAddress comparison completed successfully!")
print("Results saved in: processed data/Lake/")
else:
print(f"\nError: Script failed with return code {result.returncode}")
return result.returncode
if __name__ == "__main__":
sys.exit(main())
-258
View File
@@ -1,258 +0,0 @@
#!/usr/bin/env python3
"""
Shapefile to GeoJSON Converter for Address Data
Converts ESRI:102659 CRS shapefile to EPSG:4326 GeoJSON with OSM-style address tags
"""
import geopandas as gpd
import json
import sys
import os
from pathlib import Path
import importlib
qgis_functions = importlib.import_module("qgis-functions")
title = qgis_functions.title
getstreetfromaddress = qgis_functions.getstreetfromaddress
def convert_crs(gdf, source_crs='ESRI:102659', target_crs='EPSG:4326'):
"""
Convert coordinate reference system from source to target CRS
Args:
gdf: GeoDataFrame to convert
source_crs: Source coordinate reference system (default: ESRI:102659)
target_crs: Target coordinate reference system (default: EPSG:4326)
Returns:
GeoDataFrame with converted CRS
"""
if gdf.crs is None:
print(f"Warning: No CRS detected, assuming {source_crs}")
gdf.crs = source_crs
if gdf.crs != target_crs:
print(f"Converting from {gdf.crs} to {target_crs}")
gdf = gdf.to_crs(target_crs)
return gdf
def process_address_fields(gdf):
"""
Process and map address fields according to OSM address schema
Args:
gdf: GeoDataFrame with address data
Returns:
GeoDataFrame with processed address fields
"""
processed_gdf = gdf.copy()
# Create new columns for OSM address tags
address_mapping = {}
# ADD_NUM -> addr:housenumber (as integer)
if 'ADD_NUM' in processed_gdf.columns:
# Handle NaN values and convert to nullable integer
add_num_series = processed_gdf['ADD_NUM'].copy()
# Convert to numeric, coercing errors to NaN
add_num_series = pd.to_numeric(add_num_series, errors='coerce')
# Round to remove decimal places, then convert to nullable integer
address_mapping['addr:housenumber'] = add_num_series.round().astype('Int64')
# UNIT -> addr:unit (as string)
if 'UNIT' in processed_gdf.columns:
unit_series = processed_gdf['UNIT'].copy()
# Replace NaN, empty strings, and 'None' string with actual None
unit_series = unit_series.replace(['nan', 'None', '', None], None)
# Only keep non-null values as strings
unit_series = unit_series.where(unit_series.notna(), None)
address_mapping['addr:unit'] = unit_series
# SADD -> addr:street via title(getstreetfromaddress("SADD"))
if 'SADD' in processed_gdf.columns:
street_names = []
for sadd_value in processed_gdf['SADD']:
if pd.notna(sadd_value):
street_from_addr = getstreetfromaddress(str(sadd_value), None, None)
street_titled = title(street_from_addr)
street_names.append(street_titled)
else:
street_names.append(None)
address_mapping['addr:street'] = street_names
# POST_COMM -> addr:city via title("POST_COMM")
if 'POST_COMM' in processed_gdf.columns:
city_names = []
for post_comm in processed_gdf['POST_COMM']:
if pd.notna(post_comm):
city_titled = title(str(post_comm))
city_names.append(city_titled)
else:
city_names.append(None)
address_mapping['addr:city'] = city_names
# POST_CODE -> addr:postcode (as integer)
if 'POST_CODE' in processed_gdf.columns:
# Handle NaN values and convert to nullable integer
post_code_series = processed_gdf['POST_CODE'].copy()
# Convert to numeric, coercing errors to NaN
post_code_series = pd.to_numeric(post_code_series, errors='coerce')
# Round to remove decimal places, then convert to nullable integer
address_mapping['addr:postcode'] = post_code_series.round().astype('Int64')
# Manually add addr:state = 'FL'
address_mapping['addr:state'] = 'FL'
# Add the new address columns to the GeoDataFrame
for key, value in address_mapping.items():
processed_gdf[key] = value
return processed_gdf
def clean_output_data(gdf, keep_original_fields=False):
"""
Clean the output data, optionally keeping original fields
Args:
gdf: GeoDataFrame to clean
keep_original_fields: Whether to keep original shapefile fields
Returns:
Cleaned GeoDataFrame
"""
# Define the OSM address fields we want to keep
osm_fields = [
'addr:housenumber', 'addr:unit', 'addr:street',
'addr:city', 'addr:postcode', 'addr:state'
]
if keep_original_fields:
# Keep both original and OSM fields
original_fields = ['ADD_NUM', 'UNIT', 'SADD', 'POST_COMM', 'POST_CODE']
fields_to_keep = list(set(osm_fields + original_fields + ['geometry']))
else:
# Keep only OSM fields and geometry
fields_to_keep = osm_fields + ['geometry']
# Filter to only existing columns
existing_fields = [field for field in fields_to_keep if field in gdf.columns]
return gdf[existing_fields]
def convert_shapefile_to_geojson(
input_shapefile,
output_geojson,
keep_original_fields=False,
source_crs='ESRI:102659',
target_crs='EPSG:4326'
):
"""
Main conversion function
Args:
input_shapefile: Path to input shapefile
output_geojson: Path to output GeoJSON file
keep_original_fields: Whether to keep original shapefile fields
source_crs: Source coordinate reference system
target_crs: Target coordinate reference system
"""
try:
# Read shapefile
print(f"Reading shapefile: {input_shapefile}")
gdf = gpd.read_file(input_shapefile)
print(f"Loaded {len(gdf)} features")
# Display original columns
print(f"Original columns: {list(gdf.columns)}")
# Convert CRS if needed
gdf = convert_crs(gdf, source_crs, target_crs)
# Process address fields
print("Processing address fields...")
gdf = process_address_fields(gdf)
# Clean output data
gdf = clean_output_data(gdf, keep_original_fields)
# Remove rows with no valid geometry
gdf = gdf[gdf.geometry.notna()]
print(f"Final columns: {list(gdf.columns)}")
print(f"Final feature count: {len(gdf)}")
# Write to GeoJSON
print(f"Writing GeoJSON: {output_geojson}")
gdf.to_file(output_geojson, driver='GeoJSON')
print(f"Conversion completed successfully!")
# Display sample of processed data
if len(gdf) > 0:
print("\nSample of processed data:")
sample_cols = [col for col in gdf.columns if col.startswith('addr:')]
if sample_cols:
print(gdf[sample_cols].head())
except Exception as e:
print(f"Error during conversion: {str(e)}")
sys.exit(1)
def main():
"""
Main function to handle command line arguments
"""
import argparse
parser = argparse.ArgumentParser(
description='Convert shapefile to GeoJSON with OSM address tags'
)
parser.add_argument(
'input_shapefile',
help='Path to input shapefile'
)
parser.add_argument(
'output_geojson',
help='Path to output GeoJSON file'
)
parser.add_argument(
'--keep-original',
action='store_true',
help='Keep original shapefile fields in addition to OSM fields'
)
parser.add_argument(
'--source-crs',
default='ESRI:102659',
help='Source coordinate reference system (default: ESRI:102659)'
)
parser.add_argument(
'--target-crs',
default='EPSG:4326',
help='Target coordinate reference system (default: EPSG:4326)'
)
args = parser.parse_args()
# Validate input file
if not os.path.exists(args.input_shapefile):
print(f"Error: Input shapefile '{args.input_shapefile}' not found")
sys.exit(1)
# Create output directory if it doesn't exist
output_dir = Path(args.output_geojson).parent
output_dir.mkdir(parents=True, exist_ok=True)
# Run conversion
convert_shapefile_to_geojson(
args.input_shapefile,
args.output_geojson,
args.keep_original,
args.source_crs,
args.target_crs
)
if __name__ == "__main__":
import pandas as pd
main()
-540
View File
@@ -1,540 +0,0 @@
#!/usr/bin/env python3
"""
GeoJSON Multi Modal Golf Cart Path Comparison Script
Compares two GeoJSON files containing road data and identifies:
1. Roads in file1 that don't have corresponding coverage in file2 (removed roads)
2. Roads in file2 that don't have corresponding coverage in file1 (added roads)
Only reports differences that are significant (above minimum length threshold).
Optimized for performance with parallel processing and spatial indexing.
TODO:
- put properties properly on removed roads, so they're visible in JOSM
- handle polygons properly (on previous geojson step?) for circular roads
"""
import json
import argparse
from pathlib import Path
from typing import List, Dict, Any, Tuple
import geopandas as gpd
from shapely.geometry import LineString, MultiLineString, Point, Polygon
from shapely.ops import unary_union
from shapely.strtree import STRtree
import pandas as pd
import warnings
import multiprocessing as mp
from functools import partial
import numpy as np
from concurrent.futures import ProcessPoolExecutor, as_completed
import gc
# Suppress warnings for cleaner output
warnings.filterwarnings('ignore')
class RoadComparator:
def __init__(self, tolerance_feet: float = 50.0, min_gap_length_feet: float = 100.0,
n_jobs: int = None, chunk_size: int = 1000):
"""
Initialize the road comparator.
Args:
tolerance_feet: Distance tolerance for considering roads as overlapping (default: 50 feet)
min_gap_length_feet: Minimum length of gap/extra to be considered significant (default: 100 feet)
n_jobs: Number of parallel processes to use (default: CPU count - 1)
chunk_size: Number of geometries to process per chunk (default: 1000)
"""
self.tolerance_feet = tolerance_feet
self.min_gap_length_feet = min_gap_length_feet
self.n_jobs = n_jobs or max(1, mp.cpu_count() - 1)
self.chunk_size = chunk_size
# Convert feet to degrees (approximate conversion for continental US)
# 1 degree latitude ≈ 364,000 feet
# 1 degree longitude ≈ 288,000 feet (at 40° latitude)
self.tolerance_deg = tolerance_feet / 364000.0
self.min_gap_length_deg = min_gap_length_feet / 364000.0
print(f"Using {self.n_jobs} parallel processes with chunk size {self.chunk_size}")
def load_geojson(self, filepath: str) -> gpd.GeoDataFrame:
"""Load and validate GeoJSON file with optimizations."""
try:
# Use pyogr engine for faster loading of large files
gdf = gpd.read_file(filepath, engine='pyogrio')
# Filter only LineString, MultiLineString, and Polygon geometries
line_types = ['LineString', 'MultiLineString', 'Polygon']
gdf = gdf[gdf.geometry.type.isin(line_types)].copy()
if len(gdf) == 0:
raise ValueError(f"No line geometries found in {filepath}")
# Reset index for efficient processing
gdf = gdf.reset_index(drop=True)
# Ensure geometry is valid and fix simple issues
invalid_mask = ~gdf.geometry.is_valid
if invalid_mask.any():
print(f"Fixing {invalid_mask.sum()} invalid geometries...")
gdf.loc[invalid_mask, 'geometry'] = gdf.loc[invalid_mask, 'geometry'].buffer(0)
print(f"Loaded {len(gdf)} road features from {filepath}")
return gdf
except Exception as e:
raise Exception(f"Error loading {filepath}: {str(e)}")
def create_buffered_union_optimized(self, gdf: gpd.GeoDataFrame) -> Any:
"""Create a buffered union using chunked processing for memory efficiency."""
print("Creating optimized buffered union...")
# Process in chunks to manage memory
chunks = [gdf.iloc[i:i+self.chunk_size] for i in range(0, len(gdf), self.chunk_size)]
chunk_unions = []
# Use partial function for multiprocessing
buffer_func = partial(self._buffer_chunk, tolerance=self.tolerance_deg)
with ProcessPoolExecutor(max_workers=self.n_jobs) as executor:
# Submit all chunk processing jobs
future_to_chunk = {executor.submit(buffer_func, chunk): i
for i, chunk in enumerate(chunks)}
# Collect results as they complete
for future in as_completed(future_to_chunk):
chunk_idx = future_to_chunk[future]
try:
chunk_union = future.result()
if chunk_union and not chunk_union.is_empty:
chunk_unions.append(chunk_union)
print(f"Processed chunk {chunk_idx + 1}/{len(chunks)}")
except Exception as e:
print(f"Error processing chunk {chunk_idx}: {str(e)}")
# Union all chunk results
print("Combining chunk unions...")
if chunk_unions:
final_union = unary_union(chunk_unions)
# Force garbage collection
del chunk_unions
gc.collect()
return final_union
else:
raise Exception("No valid geometries to create union")
@staticmethod
def _buffer_chunk(chunk_gdf: gpd.GeoDataFrame, tolerance: float) -> Any:
"""Buffer geometries in a chunk and return their union."""
try:
# Buffer all geometries in the chunk
buffered = chunk_gdf.geometry.buffer(tolerance)
# Create union of buffered geometries
if len(buffered) == 1:
return buffered.iloc[0]
else:
return unary_union(buffered.tolist())
except Exception as e:
print(f"Error in chunk processing: {str(e)}")
return None
def create_spatial_index(self, gdf: gpd.GeoDataFrame) -> STRtree:
"""Create spatial index for fast intersection queries."""
print("Creating spatial index...")
# Create STRtree for fast spatial queries
geometries = gdf.geometry.tolist()
return STRtree(geometries)
def find_removed_segments_optimized(self, source_gdf: gpd.GeoDataFrame,
target_union: Any) -> List[Dict[str, Any]]:
"""
Find segments in source_gdf that are not covered by target_union (removed roads).
Optimized with parallel processing.
"""
print("Finding removed segments...")
# Split into chunks for parallel processing
chunks = [source_gdf.iloc[i:i+self.chunk_size]
for i in range(0, len(source_gdf), self.chunk_size)]
all_removed = []
# Use partial function for multiprocessing
process_func = partial(self._process_removed_chunk,
target_union=target_union,
min_length_deg=self.min_gap_length_deg)
with ProcessPoolExecutor(max_workers=self.n_jobs) as executor:
# Submit all chunk processing jobs
future_to_chunk = {executor.submit(process_func, chunk): i
for i, chunk in enumerate(chunks)}
# Collect results as they complete
for future in as_completed(future_to_chunk):
chunk_idx = future_to_chunk[future]
try:
chunk_removed = future.result()
all_removed.extend(chunk_removed)
print(f"Processed removed chunk {chunk_idx + 1}/{len(chunks)}")
except Exception as e:
print(f"Error processing removed chunk {chunk_idx}: {str(e)}")
return all_removed
@staticmethod
def _process_removed_chunk(chunk_gdf: gpd.GeoDataFrame, target_union: Any,
min_length_deg: float) -> List[Dict[str, Any]]:
"""Process a chunk of geometries to find removed segments."""
removed_segments = []
for idx, row in chunk_gdf.iterrows():
geom = row.geometry
# Handle MultiLineString by processing each component
if isinstance(geom, MultiLineString):
lines = list(geom.geoms)
else:
lines = [geom] # Polygon and Line can be accessed directly
for line in lines:
try:
# Find parts of the line that don't intersect with target_union
uncovered = line.difference(target_union)
if uncovered.is_empty:
continue
# Handle different geometry types returned by difference
uncovered_lines = []
if hasattr(uncovered, 'geoms'):
for geom_part in uncovered.geoms:
if isinstance(geom_part, LineString):
uncovered_lines.append(geom_part)
elif isinstance(uncovered, LineString):
uncovered_lines.append(uncovered)
# Check each uncovered line segment
for uncovered_line in uncovered_lines:
if uncovered_line.length >= min_length_deg:
# Create properties dict with original metadata plus 'removed: true'
properties = dict(row.drop('geometry'))
properties['removed'] = True
removed_segments.append({
'geometry': uncovered_line,
**properties
})
except Exception as e:
continue # Skip problematic geometries
return removed_segments
def find_added_roads_optimized(self, source_gdf: gpd.GeoDataFrame,
target_union: Any) -> List[Dict[str, Any]]:
"""
Find entire roads in source_gdf that don't significantly overlap with target_union.
Optimized with parallel processing.
"""
print("Finding added roads...")
# Split into chunks for parallel processing
chunks = [source_gdf.iloc[i:i+self.chunk_size]
for i in range(0, len(source_gdf), self.chunk_size)]
all_added = []
# Use partial function for multiprocessing
process_func = partial(self._process_added_chunk,
target_union=target_union,
min_length_deg=self.min_gap_length_deg)
with ProcessPoolExecutor(max_workers=self.n_jobs) as executor:
# Submit all chunk processing jobs
future_to_chunk = {executor.submit(process_func, chunk): i
for i, chunk in enumerate(chunks)}
# Collect results as they complete
for future in as_completed(future_to_chunk):
chunk_idx = future_to_chunk[future]
try:
chunk_added = future.result()
all_added.extend(chunk_added)
print(f"Processed added chunk {chunk_idx + 1}/{len(chunks)}")
except Exception as e:
print(f"Error processing added chunk {chunk_idx}: {str(e)}")
return all_added
@staticmethod
def _process_added_chunk(chunk_gdf: gpd.GeoDataFrame, target_union: Any,
min_length_deg: float) -> List[Dict[str, Any]]:
"""Process a chunk of geometries to find added roads."""
added_roads = []
for idx, row in chunk_gdf.iterrows():
geom = row.geometry
try:
# Check what portion of the road is not covered
uncovered = geom.difference(target_union)
if not uncovered.is_empty:
# Calculate what percentage of the original road is uncovered
uncovered_length = 0
if hasattr(uncovered, 'geoms'):
for geom_part in uncovered.geoms:
if isinstance(geom_part, LineString):
uncovered_length += geom_part.length
elif isinstance(uncovered, LineString):
uncovered_length = uncovered.length
original_length = geom.length
uncovered_ratio = uncovered_length / original_length if original_length > 0 else 0
# Include the entire road if:
# 1. The uncovered portion is above minimum threshold, AND
# 2. More than 10% of the road is uncovered
if uncovered_ratio > 0.1:
#uncovered_length >= min_length_deg and
# Include entire original road with all original metadata
original_properties = dict(row.drop('geometry'))
#
# For Sumter County Roads
#
properties = {
'surface': 'asphalt'
}
output = True
for key, value in original_properties.items():
if key == 'Part_of_Ro' and value == "Yes":
output = False
continue # Skip cart paths that are parts of roads
else:
properties['highway'] = 'residential'
properties['bicycle'] = 'yes'
properties['foot'] = 'yes'
properties['golf'] = 'cartpath'
properties['golf_cart'] = 'yes'
properties['highway'] = 'path'
properties['motor_vehicle'] = 'no'
properties['segregated'] = 'no'
properties['surface'] = 'asphalt'
if output:
added_roads.append({
'geometry': geom,
**properties
})
except Exception as e:
print(e)
continue # Skip problematic geometries
return added_roads
def compare_roads(self, file1_path: str, file2_path: str) -> Tuple[List[Dict], List[Dict]]:
"""
Compare two GeoJSON files and find significant differences.
Optimized version with parallel processing.
Returns:
Tuple of (removed_roads, added_roads)
"""
print(f"Comparing {file1_path} and {file2_path}")
print(f"Tolerance: {self.tolerance_feet} feet")
print(f"Minimum significant length: {self.min_gap_length_feet} feet")
print(f"Parallel processing: {self.n_jobs} workers")
print("-" * 50)
# Load both files
gdf1 = self.load_geojson(file1_path)
gdf2 = self.load_geojson(file2_path)
# Ensure both are in the same CRS
if gdf1.crs != gdf2.crs:
print(f"Warning: CRS mismatch. Converting {file2_path} to match {file1_path}")
gdf2 = gdf2.to_crs(gdf1.crs)
print("Creating optimized spatial unions...")
# Create buffered unions using optimized method
union1 = self.create_buffered_union_optimized(gdf1)
union2 = self.create_buffered_union_optimized(gdf2)
print("Finding removed and added roads with parallel processing...")
# Find roads using optimized parallel methods
removed_roads = self.find_removed_segments_optimized(gdf1, union2)
added_roads = self.find_added_roads_optimized(gdf2, union1)
# Clean up memory
del gdf1, gdf2, union1, union2
gc.collect()
return removed_roads, added_roads
def save_results(self, removed: List[Dict], added: List[Dict], output_path: str):
"""Save results to GeoJSON file."""
all_results = removed + added
if not all_results:
print("No significant differences found!")
return
# Create GeoDataFrame efficiently
print("Saving results...")
results_gdf = gpd.GeoDataFrame(all_results)
# Save to file with optimization
results_gdf.to_file(output_path, driver='GeoJSON', engine='pyogrio')
print(f"Results saved to: {output_path}")
def print_summary(self, removed: List[Dict], added: List[Dict], file1_name: str, file2_name: str):
"""Print a summary of the comparison results."""
print("\n" + "="*60)
print("COMPARISON SUMMARY")
print("="*60)
print(f"\nFile 1: {file1_name}")
print(f"File 2: {file2_name}")
print(f"Tolerance: {self.tolerance_feet} feet")
print(f"Minimum significant length: {self.min_gap_length_feet} feet")
if removed:
print(f"\n🔴 REMOVED ROADS ({len(removed)} segments):")
print("These road segments exist in File 1 but are missing or incomplete in File 2:")
# Calculate total length of removed segments
total_removed_length = 0
removed_by_road = {}
for segment in removed:
geom = segment['geometry']
length_feet = geom.length * 364000.0 # Convert to feet
total_removed_length += length_feet
# Get road name
road_name = "Unknown"
name_fields = ['name', 'NAME', 'road_name', 'street_name', 'FULLNAME']
for field in name_fields:
if field in segment and pd.notna(segment[field]):
road_name = str(segment[field])
break
if road_name not in removed_by_road:
removed_by_road[road_name] = []
removed_by_road[road_name].append(length_feet)
print(f"Total removed length: {total_removed_length:,.1f} feet ({total_removed_length/5280:.2f} miles)")
for road, lengths in sorted(removed_by_road.items()):
road_total = sum(lengths)
print(f"{road}: {len(lengths)} segment(s), {road_total:,.1f} feet")
if added:
print(f"\n🔵 ADDED ROADS ({len(added)} roads):")
print("These roads exist in File 2 but are missing or incomplete in File 1:")
# Calculate total length of added roads
total_added_length = 0
added_by_road = {}
for road in added:
geom = road['geometry']
length_feet = geom.length * 364000.0 # Convert to feet
total_added_length += length_feet
# Get road name
road_name = "Unknown"
name_fields = ['name', 'NAME', 'road_name', 'street_name', 'FULLNAME']
for field in name_fields:
if field in road and pd.notna(road[field]):
road_name = str(road[field])
break
if road_name not in added_by_road:
added_by_road[road_name] = 0
added_by_road[road_name] += length_feet
print(f"Total added length: {total_added_length:,.1f} feet ({total_added_length/5280:.2f} miles)")
for road, length in sorted(added_by_road.items()):
print(f"{road}: {length:,.1f} feet")
if not removed and not added:
print("\n✅ No significant differences found!")
print("The road networks have good coverage overlap within the specified tolerance.")
def main():
parser = argparse.ArgumentParser(
description="Compare two GeoJSON files containing roads and find significant gaps or extras (Optimized)",
formatter_class=argparse.RawDescriptionHelpFormatter,
epilog="""
Examples:
python sumter-multi-modal-convert.py osm-multi-modal.geojson county-multi-modal.geojson
python sumter-multi-modal-convert.py osm-multi-modal.geojson county-multi-modal.geojson --tolerance 100 --min-length 200
python sumter-multi-modal-convert.py osm-multi-modal.geojson county-multi-modal.geojson --output differences.geojson
python sumter-multi-modal-convert.py osm-multi-modal.geojson county-multi-modal.geojson --jobs 8 --chunk-size 2000
"""
)
parser.add_argument('file1', help='First GeoJSON file')
parser.add_argument('file2', help='Second GeoJSON file')
parser.add_argument('--tolerance', '-t', type=float, default=50.0,
help='Distance tolerance in feet for considering roads as overlapping (default: 50)')
parser.add_argument('--min-length', '-m', type=float, default=100.0,
help='Minimum length in feet for gaps/extras to be considered significant (default: 100)')
parser.add_argument('--output', '-o', help='Output GeoJSON file for results (optional)')
parser.add_argument('--jobs', '-j', type=int, default=None,
help='Number of parallel processes (default: CPU count - 1)')
parser.add_argument('--chunk-size', '-c', type=int, default=1000,
help='Number of geometries to process per chunk (default: 1000)')
args = parser.parse_args()
# Validate input files
if not Path(args.file1).exists():
print(f"Error: File {args.file1} does not exist")
return 1
if not Path(args.file2).exists():
print(f"Error: File {args.file2} does not exist")
return 1
try:
# Create comparator and run comparison
comparator = RoadComparator(
tolerance_feet=args.tolerance,
min_gap_length_feet=args.min_length,
n_jobs=args.jobs,
chunk_size=args.chunk_size
)
removed, added = comparator.compare_roads(args.file1, args.file2)
# Print summary
comparator.print_summary(removed, added, args.file1, args.file2)
# Save results if output file specified
if args.output:
comparator.save_results(removed, added, args.output)
elif removed or added:
# Auto-generate output filename if differences found
output_file = f"multi_modal_differences_{Path(args.file1).stem}_vs_{Path(args.file2).stem}.geojson"
comparator.save_results(removed, added, output_file)
return 0
except Exception as e:
print(f"Error: {str(e)}")
return 1
if __name__ == "__main__":
exit(main())
+6 -14
View File
@@ -93,27 +93,22 @@ def get_script_map():
'lake': ['python', 'shp-to-geojson.py', '/data/latest/lake/roads.shp.zip', '/data/latest/lake/county-roads.geojson'] 'lake': ['python', 'shp-to-geojson.py', '/data/latest/lake/roads.shp.zip', '/data/latest/lake/county-roads.geojson']
}, },
'convert-addresses': { 'convert-addresses': {
'sumter': ['python', 'shp-to-geojson.py', '/data/latest/sumter/addresses.shp.zip', '/data/latest/sumter/county-addresses.geojson'], 'sumter': ['python', 'convert-addresses.py', '/data/latest/sumter/addresses.shp.zip', '/data/latest/sumter/county-addresses.geojson'],
'lake': ['python', 'shp-to-geojson.py', '/data/latest/lake/addresses.shp.zip', '/data/latest/lake/county-addresses.geojson'] 'lake': ['python', 'convert-addresses.py', '/data/latest/lake/addresses.shp.zip', '/data/latest/lake/county-addresses.geojson'],
}, },
'convert-paths': { 'convert-paths': {
#todo: delete sumter-multi-modal-convert.py ?
'sumter': ['python', 'shp-to-geojson.py', '/data/latest/sumter/paths.shp.zip', '/data/latest/sumter/county-paths.geojson'], 'sumter': ['python', 'shp-to-geojson.py', '/data/latest/sumter/paths.shp.zip', '/data/latest/sumter/county-paths.geojson'],
}, },
'diff-roads': { 'diff-roads': {
'lake': ['python', 'diff-highways.py', '/data/latest/lake/osm-roads.geojson', '/data/latest/lake/county-roads.geojson', '--output', '/data/latest/lake/diff-roads.geojson'], 'lake': ['python', 'diff-highways.py', '/data/latest/lake/osm-roads.geojson', '/data/latest/lake/county-roads.geojson', '--output', '/data/latest/lake/diff-roads.geojson'],
'sumter': ['python', 'diff-highways.py', '/data/latest/sumter/osm-roads.geojson', '/data/latest/sumter/county-roads.geojson', '--output', '/data/latest/sumter/diff-roads.geojson'] 'sumter': ['python', 'diff-highways.py', '/data/latest/sumter/osm-roads.geojson', '/data/latest/sumter/county-roads.geojson', '--output', '/data/latest/sumter/diff-roads.geojson'],
}, },
'diff-paths': { 'diff-paths': {
#todo: no lake county data for paths
#'lake': ['python', 'diff-highways.py', '/data/latest/lake/osm-paths.geojson', '/data/latest/lake/county-paths.geojson', '--output', '/data/latest/lake/diff-paths.geojson'],
'sumter': ['python', 'diff-highways.py', '/data/latest/sumter/osm-paths.geojson', '/data/latest/sumter/county-paths.geojson', '--output', '/data/latest/sumter/diff-paths.geojson'], 'sumter': ['python', 'diff-highways.py', '/data/latest/sumter/osm-paths.geojson', '/data/latest/sumter/county-paths.geojson', '--output', '/data/latest/sumter/diff-paths.geojson'],
}, },
# addresses need no osm download or shapefile convert, just county download
'diff-addresses': { 'diff-addresses': {
#todo: delete sumter-address-convert.py ? 'lake': ['python', 'compare-addresses.py', '--local-file', '/data/latest/lake/county-addresses.geojson', '--osm-file', '/data/latest/lake/osm-addresses.geojson', '--output-dir', '/data/latest/lake'],
'lake': ['python', 'compare-addresses.py', 'Lake', 'Florida', '--local-zip', '/data/latest/lake/addresses.shp.zip', '--output-dir', '/data/latest/lake', '--cache-dir', '/data/osm_cache'], 'sumter': ['python', 'compare-addresses.py', '--local-file', '/data/latest/sumter/county-addresses.geojson', '--osm-file', '/data/latest/sumter/osm-addresses.geojson', '--output-dir', '/data/latest/sumter'],
'sumter': ['python', 'compare-addresses.py', 'Sumter', 'Florida', '--local-zip', '/data/latest/sumter/addresses.shp.zip', '--output-dir', '/data/latest/sumter', '--cache-dir', '/data/osm_cache']
}, },
} }
@@ -165,9 +160,6 @@ def run_script():
else: else:
cmd = list(cmd_config) # Make a copy to avoid modifying the original cmd = list(cmd_config) # Make a copy to avoid modifying the original
# Add --force-download flag for diff-addresses if requested
if script_name == 'diff-addresses' and force_download and isinstance(cmd, list):
cmd.append('--force-download')
else: else:
return jsonify({'error': 'Invalid script configuration'}), 400 return jsonify({'error': 'Invalid script configuration'}), 400