From e04d3c589ca15418532bb924c539b2ad066b060f Mon Sep 17 00:00:00 2001 From: Isarge05 Date: Wed, 30 Sep 2026 11:12:20 -0400 Subject: [PATCH 1/4] Cleaned up missing GEOIDs and Municipality_Name in raw zoning data --- backend/data_cleaning/clean_zoning.py | 52 ++++++++++++++++++++++++++- 1 file changed, 51 insertions(+), 1 deletion(-) diff --git a/backend/data_cleaning/clean_zoning.py b/backend/data_cleaning/clean_zoning.py index feeffbf3..9328723d 100644 --- a/backend/data_cleaning/clean_zoning.py +++ b/backend/data_cleaning/clean_zoning.py @@ -65,6 +65,55 @@ } +# Zoning data Municipal_Name -> (corrected Municipal_Name or None (unchanged) +MISSING_GEOID_FIXES = { + "Bennington Landgrove": ("Landgrove", "Landgrove town"), + "Enosburgh Enosburg Falls": ("Enosburg Falls", "Enosburgh town"), + "Huntington": (None, "Huntington town"), + "Hyde Park Town Hyde Park Village": ("Hyde Park Village", "Hyde Park town"), + "Morristown": (None, "Morristown town"), + "North Bennington": (None, "Bennington town"), + "Old Bennington": (None, "Bennington town"), + "Saint George": (None, "St. George town"), + "Sandgate F2": ("Sandgate", "Sandgate town"), + "South Burlington": (None, "South Burlington city"), + "Stowe Town": (None, "Stowe town"), + "Stowe Town Stowe Village": ("Stowe Village", "Stowe town"), + "Warren Gore": ("Warren's Gore", "Warren's gore"), +} + + +def fix_missing_geoids( + con: duckdb.DuckDBPyConnection, raw_df: pd.DataFrame +) -> pd.DataFrame: + """ + Fixes for zoning rows whose GEO_ID/Municipal_Name + are missing or wrong (~132 rows). + """ + # Westmore, Newport Town and Peru have a blank Municipal_Name. + # The town is the District_Name. + blank = raw_df["Municipal_Name"].fillna("") == "" + raw_df.loc[blank, "Municipal_Name"] = raw_df.loc[blank, "District_Name"] + + # Get the town names for joining + geoids = con.execute( + "SELECT CAST(GEOID AS VARCHAR), split_part(NAME, ',', 1) " + "FROM lake.RAW.vt_town_lines" + ).fetchall() + geoid_by_name = {name: geoid for geoid, name in geoids} + + for name, (new_name, census_name) in MISSING_GEOID_FIXES.items(): + rows = raw_df["Municipal_Name"] == name + if census_name in geoid_by_name: + raw_df.loc[rows, "GEO_ID"] = geoid_by_name[census_name] + if new_name: + raw_df.loc[rows, "Municipal_Name"] = new_name + + # Small fix: Landgrove district (OBJECT_ID=894) had the wrong RPC. + raw_df.loc[raw_df["OBJECT_ID"] == 894, "RPC"] = "BCRC" + return raw_df + + def read_raw_data(con: duckdb.DuckDBPyConnection) -> pd.DataFrame: """ Reads the lake.RAW.zoning table into python memory @@ -94,7 +143,8 @@ def read_raw_data(con: duckdb.DuckDBPyConnection) -> pd.DataFrame: # Hardcoded fix for Burlington's "Residential - High Density" zoning district # Changes overlay status to "No" raw_df.loc[raw_df["OBJECT_ID"] == 1045, "Overlay_District"] = "No" - + # Fill in missing GEOIDs for future joins + raw_df = fix_missing_geoids(con, raw_df) con.register("zoning_raw", raw_df) return raw_df From 212e00e509e7e53005a39397e0f3cdf0dd68cdd7 Mon Sep 17 00:00:00 2001 From: Isarge05 Date: Wed, 30 Sep 2026 11:24:51 -0400 Subject: [PATCH 2/4] Update the 3 districts with no municipality name into the cleaning mapper --- backend/data_cleaning/clean_zoning.py | 3 +++ 1 file changed, 3 insertions(+) diff --git a/backend/data_cleaning/clean_zoning.py b/backend/data_cleaning/clean_zoning.py index 9328723d..46f8e218 100644 --- a/backend/data_cleaning/clean_zoning.py +++ b/backend/data_cleaning/clean_zoning.py @@ -72,14 +72,17 @@ "Huntington": (None, "Huntington town"), "Hyde Park Town Hyde Park Village": ("Hyde Park Village", "Hyde Park town"), "Morristown": (None, "Morristown town"), + "Newport Town": ("Newport town", "Newport town"), "North Bennington": (None, "Bennington town"), "Old Bennington": (None, "Bennington town"), + "Peru": (None, "Peru town"), "Saint George": (None, "St. George town"), "Sandgate F2": ("Sandgate", "Sandgate town"), "South Burlington": (None, "South Burlington city"), "Stowe Town": (None, "Stowe town"), "Stowe Town Stowe Village": ("Stowe Village", "Stowe town"), "Warren Gore": ("Warren's Gore", "Warren's gore"), + "Westmore": (None, "Westmore town"), } From 774d97b3304f086740a9fb33579ebfb62efd8b48 Mon Sep 17 00:00:00 2001 From: Isarge05 Date: Wed, 30 Sep 2026 11:53:54 -0400 Subject: [PATCH 3/4] Adjusted zoning dataset docs to reflect cleaning step --- docs/datasets/zoning.md | 17 ++++++++++++----- 1 file changed, 12 insertions(+), 5 deletions(-) diff --git a/docs/datasets/zoning.md b/docs/datasets/zoning.md index 9064c68e..18a625e8 100644 --- a/docs/datasets/zoning.md +++ b/docs/datasets/zoning.md @@ -46,7 +46,7 @@ tables that downstream queries read; see the | `County` | `VARCHAR` | Vermont county containing the district. | 14 Vermont counties | 1,739 (100.0%) | | `RPC` | `VARCHAR` | Abbreviation of the Regional Planning Commission serving the municipality. | `CCRPC`, `NVDA`, `TRORC`, `RRPC`, `ACRPC`, `CVRPC`, `NWRPC`, `BCRC`, `WRC`, `MARC`, `LCPC` | 1,739 (100.0%) | | `Municipal_Name` | `VARCHAR` | Name of the municipality (town, city, or village) whose bylaw defines the district. | 208 distinct; trailing space on almost every value | 1,739 (100.0%) | -| `GEO_ID` | `VARCHAR` | Census geographic identifier (GEOID) for the municipality, used to join to Census data. Missing for some rows. | 10-digit string, e.g. `5000929125` | 1,607 (92.4%) | +| `GEO_ID` | `VARCHAR` | Census geographic identifier (GEOID) for the municipality, used to join to Census data. Missing for 132 rows (Fixed in `clean_zoning.py`) | 10-digit string, e.g. `5000929125` | 1,607 (92.4%) | | `District_Name` | `VARCHAR` | Full name of the district as written in the municipal bylaw. | Free text (1,026 distinct) | 1,739 (100.0%) | | `Abbreviated_District_Name` | `VARCHAR` | Short abbreviation of the district name used on maps and tables. | Free text (873 distinct) | 1,739 (100.0%) | | `District_Type` | `VARCHAR` | Classification of the district by primary land-use character. | `Mixed with Residential`, `Primarily Residential`, `Nonresidential`, `Overlay not Affecting Use`, `Overlay` | 1,739 (100.0%) | @@ -210,7 +210,7 @@ tables that downstream queries read; see the EPSG:4326 (degrees), but these values run up to about 5.4e5 and 1.4e8, so they were computed in a different projection upstream and are not degrees. Compute area from the geometry instead (the cleaner derives `Acres` this way). -- **Missing identifiers and dates.** `GEO_ID` is null on 132 rows and +- **Missing identifiers and dates.** `GEO_ID` is null on 132 rows (Addressed in the `data_cleaning/clean_zoning.py` script) and `Bylaw_Date` on 504. `Bylaw_Date` spans 2006 to 2024, so districts in different towns reflect different bylaw vintages. - **Sparse rule columns.** Most numeric rule columns are populated on well under @@ -232,18 +232,25 @@ The Atlas documentation also lists `FIPS6`, `GIS_ID`, `Last_Update`, ```sql -- Districts where 4+ unit housing is permitted by right, by RPC -SELECT RPC, COUNT(*) AS districts +SELECT + RPC, + COUNT(*) AS districts FROM lake.RAW.zoning WHERE F4F_Allowance = 'Permitted' GROUP BY RPC ORDER BY districts DESC; -- Trim names before joining to other tables -SELECT TRIM(Municipal_Name) AS town, District_Name, F1F_Min_Lot_Size +SELECT + TRIM(Municipal_Name) AS town, + District_Name, + F1F_Min_Lot_Size FROM lake.RAW.zoning WHERE TRIM(Municipal_Name) = 'Middlebury'; -- Geometry back to a spatial type -SELECT OBJECT_ID, ST_GeomFromWKB(geometry) AS geom +SELECT + OBJECT_ID, + ST_GeomFromWKB(geometry) AS geom FROM lake.RAW.zoning; ``` From c90f1a7b3ecc74a01f4a8f159678e4e7a195b6f1 Mon Sep 17 00:00:00 2001 From: Isarge05 Date: Thu, 1 Oct 2026 12:22:14 -0400 Subject: [PATCH 4/4] Added logging to mismatched geoid locations --- backend/data_cleaning/clean_zoning.py | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/backend/data_cleaning/clean_zoning.py b/backend/data_cleaning/clean_zoning.py index 46f8e218..470d717b 100644 --- a/backend/data_cleaning/clean_zoning.py +++ b/backend/data_cleaning/clean_zoning.py @@ -109,6 +109,12 @@ def fix_missing_geoids( rows = raw_df["Municipal_Name"] == name if census_name in geoid_by_name: raw_df.loc[rows, "GEO_ID"] = geoid_by_name[census_name] + # If there is a mismatch between the names, log it! + else: + print( + "No census match for zoning Municipal_Name" + + f"\033[1m{name}\033[0m to Census name \033[1m{census_name}\033[0m" + ) if new_name: raw_df.loc[rows, "Municipal_Name"] = new_name