Source Copenhagen transit data from Rejseplanen GTFS

- Replace OSM Overpass transit data with the Rejseplanen GTFS feed
  (routes keyed by (agency, short name); styles from modes.yaml)
- Draw one shape per (style, ref, direction), choosing the shape that
  serves the most in-area stops so lines pass the stops we show
- Collapse stops by names (verified unambiguous); prune stops whose
  serving refs have no drawn line within 300 m
- Patch Københavns Havn + Nordhavn into the area polygon so ferry
  routes and sub-harbour metro tunnels survive clipping; 100 m buffer
  closes relation boundary slivers
- Drop OSM download pipeline; keep tiled basemap
This commit is contained in:
marvin
2026-09-17 21:07:40 +02:00
parent 2dcaed7600
commit a6715eaff5
12 changed files with 522 additions and 751 deletions
+129 -20
View File
@@ -1,41 +1,150 @@
#!/usr/bin/env python3
"""Convert raw GTFS txt files into portable GeoPackage layers + a colour table.
STATUS: BLOCKED on download_gtfs.py (no feed URL yet).
Intended outputs (data/processed):
gtfs_shapes.gpkg (shapes.txt -> LineString per shape_id, EPSG:4326)
Inputs (data/raw/gtfs): routes.txt, trips.txt, stops.txt, shapes.txt
Outputs (data/processed):
gtfs_shapes.gpkg (shapes.txt -> LineString per shape_id, EPSG:4326,
each shape tagged with its route_id)
gtfs_stops.gpkg (stops.txt -> Point per stop, EPSG:4326)
route_colors.csv (route_id, route_short_name, route_type, route_color)
route_colors.csv (route_id, agency_id, route_short_name, route_long_name,
route_type, route_color, route_text_color)
Uses partridge for fast GTFS parsing. prepare.py reads these when present and
falls back to OSM-only data when absent.
The feed is nationwide, so everything is pre-filtered to the Copenhagen area:
stops by location, shapes to those intersecting the area polygon (or a
fallback bbox when data/processed/area.gpkg does not exist yet), and routes
to those having at least one surviving shape. prepare.py clips precisely to
the area later.
Note: this feed leaves route_color empty for the Copenhagen operators, so
prepare.py will mostly fall through to OSM colour tags / the styling palette.
stop_times.txt is not needed here and is intentionally not parsed (220 MB).
"""
import sys
from pathlib import Path
import geopandas as gpd
import pandas as pd
from shapely.geometry import LineString, box
from shapely.geometry.base import BaseGeometry
from _common import GTFS_RAW, PROCESSED
# Generous Copenhagen bbox (covers København, Frederiksberg, Amager, and
# immediate surroundings: airport, Hellerup, Lyngby, Brøndby, ...).
CPH_BBOX = box(12.30, 55.55, 12.75, 55.85)
def load_area() -> BaseGeometry:
"""City Pass area polygon if build_area has run, else the fallback bbox."""
area_path = PROCESSED / "area.gpkg"
if area_path.exists():
return gpd.read_file(area_path).geometry.union_all()
print(f"no {area_path}; using fallback Copenhagen bbox", flush=True)
return CPH_BBOX
def build_stops(area: BaseGeometry) -> gpd.GeoDataFrame:
stops = pd.read_csv(
GTFS_RAW / "stops.txt",
usecols=["stop_id", "stop_name", "stop_lat", "stop_lon",
"location_type", "parent_station"],
dtype={"stop_id": str, "parent_station": str},
)
minx, miny, maxx, maxy = area.bounds
in_bbox = (
stops["stop_lat"].between(miny, maxy)
& stops["stop_lon"].between(minx, maxx)
)
stops = stops[in_bbox].copy()
g = gpd.GeoDataFrame(
stops,
geometry=gpd.points_from_xy(stops["stop_lon"], stops["stop_lat"]),
crs="EPSG:4326",
)
return g[g.intersects(area)].drop(columns=["stop_lat", "stop_lon"])
def build_shapes(area: BaseGeometry) -> gpd.GeoDataFrame:
pts = pd.read_csv(GTFS_RAW / "shapes.txt", dtype={"shape_id": str})
minx, miny, maxx, maxy = area.bounds
hits = pts[
pts["shape_pt_lat"].between(miny, maxy)
& pts["shape_pt_lon"].between(minx, maxx)
]
pts = pts[pts["shape_id"].isin(set(hits["shape_id"].unique()))]
pts = pts.sort_values(["shape_id", "shape_pt_sequence"])
lines = (
pts.groupby("shape_id")
.apply(
lambda g: LineString(zip(g["shape_pt_lon"], g["shape_pt_lat"])),
include_groups=False,
)
.rename("geometry")
.reset_index()
)
g = gpd.GeoDataFrame(lines, geometry="geometry", crs="EPSG:4326")
return g[g.intersects(area)]
def build_routes(
shapes: gpd.GeoDataFrame,
) -> tuple[gpd.GeoDataFrame, pd.DataFrame]:
"""Tag shapes with route_id and keep routes having >= 1 surviving shape."""
trips = pd.read_csv(
GTFS_RAW / "trips.txt",
usecols=["route_id", "shape_id"],
dtype=str,
).dropna(subset=["shape_id"])
shape_to_route = (
trips[trips["shape_id"].isin(set(shapes["shape_id"]))]
.drop_duplicates("shape_id")
.set_index("shape_id")["route_id"]
)
shapes = shapes.copy()
shapes["route_id"] = shapes["shape_id"].map(shape_to_route)
routes = pd.read_csv(GTFS_RAW / "routes.txt", dtype=str).fillna("")
routes = routes[
routes["route_id"].isin(set(shape_to_route.unique()))
].copy()
return shapes, routes[
[
"route_id", "agency_id", "route_short_name", "route_long_name",
"route_type", "route_color", "route_text_color",
]
]
def main():
if not (GTFS_RAW / "routes.txt").exists():
print(
"gtfs_to_geopackage: BLOCKED — no GTFS data found in "
f"{GTFS_RAW}. Run download_gtfs.py first once a feed URL is set.",
"gtfs_to_geopackage: no GTFS data found in "
f"{GTFS_RAW}. Run download_gtfs.py first.",
file=sys.stderr,
)
sys.exit(2)
# TODO (when GTFS available):
# import partridge as pt
# import geopandas as gpd
# from shapely.geometry import LineString, Point
# feed = pt.load_geo_feed(str(GTFS_RAW))
# shapes -> gtfs_shapes.gpkg (LineString per shape_id)
# stops -> gtfs_stops.gpkg
# routes -> route_colors.csv (route_id, route_short_name, route_type, route_color)
print("gtfs_to_geopackage: not yet implemented (GTFS feed unavailable).")
sys.exit(2)
PROCESSED.mkdir(parents=True, exist_ok=True)
area = load_area()
stops = build_stops(area)
print(f"stops in area: {len(stops)}", flush=True)
shapes = build_shapes(area)
print(f"shapes in area: {len(shapes)}", flush=True)
shapes, routes = build_routes(shapes)
print(f"routes in area: {len(routes)}", flush=True)
stops.to_file(PROCESSED / "gtfs_stops.gpkg", driver="GPKG", layer="stops")
shapes.to_file(
PROCESSED / "gtfs_shapes.gpkg", driver="GPKG", layer="shapes"
)
routes.to_csv(PROCESSED / "route_colors.csv", index=False)
print(
"wrote gtfs_stops.gpkg, gtfs_shapes.gpkg, route_colors.csv "
f"into {PROCESSED}",
flush=True,
)
if __name__ == "__main__":