Skip to content

openavmkit.utilities.openstreetmap

OpenStreetMap data fetching service.

Wraps osmnx to download tagged OSM features (parks, water bodies, schools, streets, transportation networks, etc.) for a given bounding box, with caching to avoid re-downloading on subsequent runs. Used by the distance enrichment (data.process.enrich.distances) and the streets enrichment (data.process.enrich.streets) when osm: true is configured for a feature.

OpenStreetMapService

OpenStreetMapService(settings=None)

Service for retrieving and processing data from OpenStreetMap.

Attributes:

Name Type Description
settings dict

Settings dictionary

features dict

Dictionary containing internal features that have been loaded

Initialize the OpenStreetMap service.

Parameters:

Name Type Description Default
settings dict

Configuration settings for the service

None
Source code in openavmkit/utilities/openstreetmap.py
34
35
36
37
38
39
40
41
42
43
def __init__(self, settings: dict = None):
    """Initialize the OpenStreetMap service.

    Parameters
    ----------
    settings : dict
        Configuration settings for the service
    """
    self.settings = settings or {}
    self.features = {}

calculate_distances

calculate_distances(gdf, features, feature_type)

Calculate distances to features, both aggregate and specific top N features.

Parameters:

Name Type Description Default
gdf GeoDataFrame

Parcel GeoDataFrame

required
features GeoDataFrame

Features GeoDataFrame

required
feature_type str

Type of feature (e.g., 'water', 'park')

required

Returns:

Type Description
DataFrame

DataFrame with distances

Source code in openavmkit/utilities/openstreetmap.py
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
def calculate_distances(
    self, gdf: gpd.GeoDataFrame, features: gpd.GeoDataFrame, feature_type: str
) -> pd.DataFrame:
    """Calculate distances to features, both aggregate and specific top N features.

    Parameters
    ----------
    gdf : gpd.GeoDataFrame
        Parcel GeoDataFrame
    features : gpd.GeoDataFrame
        Features GeoDataFrame
    feature_type : str
        Type of feature (e.g., 'water', 'park')

    Returns
    -------
    pd.DataFrame
        DataFrame with distances
    """

    # check if we have already cached this data, AND the settings are the same
    # construct a unique signature:
    signature = {"feature_type": feature_type, "features": hash(features.to_json())}
    if check_cache(
        f"osm/{feature_type}_distances", signature=signature, filetype="df"
    ):
        print("----> using cached distances")
        # if so return the cached version
        return read_cache(f"osm/{feature_type}_distances", "df")

    # Project to UTM for accurate distance calculation
    utm_crs = self._get_utm_crs(gdf.total_bounds)
    gdf_proj = gdf.to_crs(utm_crs)
    features_proj = features.to_crs(utm_crs)

    # Initialize dictionary to store all distance calculations
    distance_data = {}

    # Calculate aggregate distance (distance to nearest feature of any type)
    distance_data[f"dist_to_{feature_type}_any"] = gdf_proj.geometry.apply(
        lambda g: features_proj.geometry.distance(g).min()
    )

    # Calculate distances to top N features if available
    if f"{feature_type}_top" in self.features:
        top_features = self.features[f"{feature_type}_top"]
        for _, feature in top_features.iterrows():
            feature_name = feature["name"]
            feature_geom = feature.geometry
            feature_proj = gpd.GeoSeries([feature_geom]).to_crs(utm_crs)[0]

            distance_data[f"dist_to_{feature_type}_{feature_name}"] = (
                gdf_proj.geometry.apply(lambda g: feature_proj.distance(g))
            )

    # write to cache so we can skip on next run
    write_cache(f"osm/{feature_type}_distances", signature, distance_data, "df")

    # Create DataFrame from all collected distances at once
    return pd.DataFrame(distance_data, index=gdf.index)

enrich_parcels

enrich_parcels(gdf, settings)

Get OpenStreetMap features and prepare them for spatial joins. Returns a dictionary of feature dataframes for use by data.py's spatial join logic.

Parameters:

Name Type Description Default
gdf GeoDataFrame

Parcel GeoDataFrame (used for bbox)

required
settings dict

Settings for enrichment

required

Returns:

Type Description
dict[str, GeoDataFrame]

Dictionary of feature dataframes

Source code in openavmkit/utilities/openstreetmap.py
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
def enrich_parcels(
    self,
    gdf: gpd.GeoDataFrame,
    settings: Dict
) -> Dict[str, gpd.GeoDataFrame]:
    """Get OpenStreetMap features and prepare them for spatial joins. Returns a
    dictionary of feature dataframes for use by data.py's spatial join logic.

    Parameters
    ----------
    gdf : gpd.GeoDataFrame
        Parcel GeoDataFrame (used for bbox)
    settings : dict
        Settings for enrichment

    Returns
    -------
    dict[str, gpd.GeoDataFrame]
        Dictionary of feature dataframes
    """
    # Get the bounding box of the GeoDataFrame
    bbox = gdf.total_bounds

    # Dictionary to store all dataframes
    dataframes = {}

    # Process each feature type based on settings
    if settings.get("water_bodies", {}).get("enabled", False):
        water_bodies = self.get_features(bbox, "water_bodies", settings["water_bodies"])
        if not water_bodies.empty:
            # Store both the main and top features in dataframes
            dataframes["water_bodies"] = self.features["water_bodies"]
            dataframes["water_bodies_top"] = self.features["water_bodies_top"]

    if settings.get("transportation", {}).get("enabled", False):
        transportation = self.get_features(bbox, "transportation", settings["transportation"])
        if not transportation.empty:
            dataframes["transportation"] = self.features["transportation"]
            dataframes["transportation_top"] = self.features["transportation_top"]

    if settings.get("educational", {}).get("enabled", False):
        institutions = self.get_features(bbox, "educational", settings["educational"])
        if not institutions.empty:
            dataframes["educational"] = self.features["educational"]
            dataframes["educational_top"] = self.features["educational_top"]

    if settings.get("parks", {}).get("enabled", False):
        parks = self.get_features(bbox, "parks", settings["parks"])
        if not parks.empty:
            dataframes["parks"] = self.features["parks"]
            dataframes["parks_top"] = self.features["parks_top"]

    if settings.get("golf_courses", {}).get("enabled", False):
        golf_courses = self.get_features(bbox, "golf_courses", settings["golf_courses"])
        if not golf_courses.empty:
            dataframes["golf_courses"] = self.features["golf_courses"]
            dataframes["golf_courses_top"] = self.features["golf_courses_top"]

    return dataframes

init_service_openstreetmap

init_service_openstreetmap(settings=None)

Initialize an OpenStreetMap service with the provided settings.

Parameters:

Name Type Description Default
settings dict

Configuration settings for the service

None

Returns:

Type Description
OpenStreetMapService

Initialized OpenStreetMap service

Source code in openavmkit/utilities/openstreetmap.py
349
350
351
352
353
354
355
356
357
358
359
360
361
362
def init_service_openstreetmap(settings: Dict = None) -> OpenStreetMapService:
    """Initialize an OpenStreetMap service with the provided settings.

    Parameters
    ----------
    settings : dict
        Configuration settings for the service

    Returns
    -------
    OpenStreetMapService
        Initialized OpenStreetMap service
    """
    return OpenStreetMapService(settings)