diff --git a/backend/data_cleaning/clean_zoning.py b/backend/data_cleaning/clean_zoning.py index feeffbf3..470d717b 100644 --- a/backend/data_cleaning/clean_zoning.py +++ b/backend/data_cleaning/clean_zoning.py @@ -65,6 +65,64 @@ } +# Zoning data Municipal_Name -> (corrected Municipal_Name or None (unchanged) +MISSING_GEOID_FIXES = { + "Bennington Landgrove": ("Landgrove", "Landgrove town"), + "Enosburgh Enosburg Falls": ("Enosburg Falls", "Enosburgh town"), + "Huntington": (None, "Huntington town"), + "Hyde Park Town Hyde Park Village": ("Hyde Park Village", "Hyde Park town"), + "Morristown": (None, "Morristown town"), + "Newport Town": ("Newport town", "Newport town"), + "North Bennington": (None, "Bennington town"), + "Old Bennington": (None, "Bennington town"), + "Peru": (None, "Peru town"), + "Saint George": (None, "St. George town"), + "Sandgate F2": ("Sandgate", "Sandgate town"), + "South Burlington": (None, "South Burlington city"), + "Stowe Town": (None, "Stowe town"), + "Stowe Town Stowe Village": ("Stowe Village", "Stowe town"), + "Warren Gore": ("Warren's Gore", "Warren's gore"), + "Westmore": (None, "Westmore town"), +} + + +def fix_missing_geoids( + con: duckdb.DuckDBPyConnection, raw_df: pd.DataFrame +) -> pd.DataFrame: + """ + Fixes for zoning rows whose GEO_ID/Municipal_Name + are missing or wrong (~132 rows). + """ + # Westmore, Newport Town and Peru have a blank Municipal_Name. + # The town is the District_Name. + blank = raw_df["Municipal_Name"].fillna("") == "" + raw_df.loc[blank, "Municipal_Name"] = raw_df.loc[blank, "District_Name"] + + # Get the town names for joining + geoids = con.execute( + "SELECT CAST(GEOID AS VARCHAR), split_part(NAME, ',', 1) " + "FROM lake.RAW.vt_town_lines" + ).fetchall() + geoid_by_name = {name: geoid for geoid, name in geoids} + + for name, (new_name, census_name) in MISSING_GEOID_FIXES.items(): + rows = raw_df["Municipal_Name"] == name + if census_name in geoid_by_name: + raw_df.loc[rows, "GEO_ID"] = geoid_by_name[census_name] + # If there is a mismatch between the names, log it! + else: + print( + "No census match for zoning Municipal_Name" + + f"\033[1m{name}\033[0m to Census name \033[1m{census_name}\033[0m" + ) + if new_name: + raw_df.loc[rows, "Municipal_Name"] = new_name + + # Small fix: Landgrove district (OBJECT_ID=894) had the wrong RPC. + raw_df.loc[raw_df["OBJECT_ID"] == 894, "RPC"] = "BCRC" + return raw_df + + def read_raw_data(con: duckdb.DuckDBPyConnection) -> pd.DataFrame: """ Reads the lake.RAW.zoning table into python memory @@ -94,7 +152,8 @@ def read_raw_data(con: duckdb.DuckDBPyConnection) -> pd.DataFrame: # Hardcoded fix for Burlington's "Residential - High Density" zoning district # Changes overlay status to "No" raw_df.loc[raw_df["OBJECT_ID"] == 1045, "Overlay_District"] = "No" - + # Fill in missing GEOIDs for future joins + raw_df = fix_missing_geoids(con, raw_df) con.register("zoning_raw", raw_df) return raw_df diff --git a/docs/datasets/zoning.md b/docs/datasets/zoning.md index 9064c68e..18a625e8 100644 --- a/docs/datasets/zoning.md +++ b/docs/datasets/zoning.md @@ -46,7 +46,7 @@ tables that downstream queries read; see the | `County` | `VARCHAR` | Vermont county containing the district. | 14 Vermont counties | 1,739 (100.0%) | | `RPC` | `VARCHAR` | Abbreviation of the Regional Planning Commission serving the municipality. | `CCRPC`, `NVDA`, `TRORC`, `RRPC`, `ACRPC`, `CVRPC`, `NWRPC`, `BCRC`, `WRC`, `MARC`, `LCPC` | 1,739 (100.0%) | | `Municipal_Name` | `VARCHAR` | Name of the municipality (town, city, or village) whose bylaw defines the district. | 208 distinct; trailing space on almost every value | 1,739 (100.0%) | -| `GEO_ID` | `VARCHAR` | Census geographic identifier (GEOID) for the municipality, used to join to Census data. Missing for some rows. | 10-digit string, e.g. `5000929125` | 1,607 (92.4%) | +| `GEO_ID` | `VARCHAR` | Census geographic identifier (GEOID) for the municipality, used to join to Census data. Missing for 132 rows (Fixed in `clean_zoning.py`) | 10-digit string, e.g. `5000929125` | 1,607 (92.4%) | | `District_Name` | `VARCHAR` | Full name of the district as written in the municipal bylaw. | Free text (1,026 distinct) | 1,739 (100.0%) | | `Abbreviated_District_Name` | `VARCHAR` | Short abbreviation of the district name used on maps and tables. | Free text (873 distinct) | 1,739 (100.0%) | | `District_Type` | `VARCHAR` | Classification of the district by primary land-use character. | `Mixed with Residential`, `Primarily Residential`, `Nonresidential`, `Overlay not Affecting Use`, `Overlay` | 1,739 (100.0%) | @@ -210,7 +210,7 @@ tables that downstream queries read; see the EPSG:4326 (degrees), but these values run up to about 5.4e5 and 1.4e8, so they were computed in a different projection upstream and are not degrees. Compute area from the geometry instead (the cleaner derives `Acres` this way). -- **Missing identifiers and dates.** `GEO_ID` is null on 132 rows and +- **Missing identifiers and dates.** `GEO_ID` is null on 132 rows (Addressed in the `data_cleaning/clean_zoning.py` script) and `Bylaw_Date` on 504. `Bylaw_Date` spans 2006 to 2024, so districts in different towns reflect different bylaw vintages. - **Sparse rule columns.** Most numeric rule columns are populated on well under @@ -232,18 +232,25 @@ The Atlas documentation also lists `FIPS6`, `GIS_ID`, `Last_Update`, ```sql -- Districts where 4+ unit housing is permitted by right, by RPC -SELECT RPC, COUNT(*) AS districts +SELECT + RPC, + COUNT(*) AS districts FROM lake.RAW.zoning WHERE F4F_Allowance = 'Permitted' GROUP BY RPC ORDER BY districts DESC; -- Trim names before joining to other tables -SELECT TRIM(Municipal_Name) AS town, District_Name, F1F_Min_Lot_Size +SELECT + TRIM(Municipal_Name) AS town, + District_Name, + F1F_Min_Lot_Size FROM lake.RAW.zoning WHERE TRIM(Municipal_Name) = 'Middlebury'; -- Geometry back to a spatial type -SELECT OBJECT_ID, ST_GeomFromWKB(geometry) AS geom +SELECT + OBJECT_ID, + ST_GeomFromWKB(geometry) AS geom FROM lake.RAW.zoning; ```