Skip to content

pipeline

predspot.pipeline

Pipeline Module

Convenience functions to run a sensible default prediction pipeline in one call, and to generate small synthetic datasets for quick experiments.

Example

from predspot.pipeline import generate_testdata, run_prediction_pipeline crimes, study_area = generate_testdata(2000, '2019-01-01', '2020-12-31', seed=0) predictions, pipeline = run_prediction_pipeline(crimes, study_area, grid_resolution=1)

generate_testdata

generate_testdata(n_points, start_time, end_time, bounds=DEFAULT_BOUNDS, seed=None)

Generate synthetic crime events inside a rectangular study area.

A thin wrapper around generate_crimes with three hotspots and the default temporal patterns.

Parameters:

Name Type Description Default
n_points int

Number of events.

required
start_time str

First possible timestamp ('YYYY-MM-DD').

required
end_time str

Last possible timestamp ('YYYY-MM-DD').

required
bounds tuple

(west, south, east, north) in WGS84 degrees.

DEFAULT_BOUNDS
seed int

Seed for reproducibility.

None

Returns:

Type Description
tuple

(crimes, study_area) — a DataFrame with tag, t, lon, lat and a one-row GeoDataFrame with the study area.

Source code in src/predspot/pipeline.py
def generate_testdata(n_points, start_time, end_time, bounds=DEFAULT_BOUNDS, seed=None):
    """
    Generate synthetic crime events inside a rectangular study area.

    A thin wrapper around [`generate_crimes`][predspot.synthetic.generate_crimes] with
    three hotspots and the default temporal patterns.

    Args:
        n_points (int): Number of events.
        start_time (str): First possible timestamp (``'YYYY-MM-DD'``).
        end_time (str): Last possible timestamp (``'YYYY-MM-DD'``).
        bounds (tuple): ``(west, south, east, north)`` in WGS84 degrees.
        seed (int, optional): Seed for reproducibility.

    Returns:
        tuple: ``(crimes, study_area)`` — a DataFrame with ``tag``, ``t``,
        ``lon``, ``lat`` and a one-row GeoDataFrame with the study area.
    """
    west, south, east, north = bounds
    study_area = gpd.GeoDataFrame(
        {"name": ["study_area"]}, geometry=[box(west, south, east, north)], crs="EPSG:4326"
    )
    crimes = generate_crimes(
        study_area, n_events=n_points, start=start_time, end=end_time, seed=seed
    )
    return crimes, study_area

build_default_pipeline

build_default_pipeline(study_area, tfreq='M', grid_resolution=1, lags=2, bandwidth='silverman', random_state=None)

Build the default Predspot pipeline: KDE mapping, seasonal/trend/diff features, quantile scaling, RFE feature selection and a random forest.

Parameters:

Name Type Description Default
study_area GeoDataFrame

Study area used to build the point grid.

required
tfreq str

Time frequency ('M', 'W' or 'D').

'M'
grid_resolution float

Grid spacing in kilometers.

1
lags int

Number of lags (and STL period) of the features.

2
bandwidth str or float

KDE bandwidth, see KDE.

'silverman'
random_state int

Seed for the estimator and shuffling.

None

Returns:

Type Description
PredictionPipeline

An unfitted pipeline.

Source code in src/predspot/pipeline.py
def build_default_pipeline(
    study_area, tfreq="M", grid_resolution=1, lags=2, bandwidth="silverman", random_state=None
):
    """
    Build the default Predspot pipeline: KDE mapping, seasonal/trend/diff
    features, quantile scaling, RFE feature selection and a random forest.

    Args:
        study_area (GeoDataFrame): Study area used to build the point grid.
        tfreq (str): Time frequency (``'M'``, ``'W'`` or ``'D'``).
        grid_resolution (float): Grid spacing in kilometers.
        lags (int): Number of lags (and STL period) of the features.
        bandwidth (str or float): KDE bandwidth, see [`KDE`][predspot.crime_mapping.KDE].
        random_state (int, optional): Seed for the estimator and shuffling.

    Returns:
        predspot.ml_modelling.PredictionPipeline: An unfitted pipeline.
    """
    grid = crime_mapping.create_gridpoints(study_area, grid_resolution)
    return ml_modelling.PredictionPipeline(
        mapping=crime_mapping.KDE(tfreq=tfreq, grid=grid, bandwidth=bandwidth),
        fextraction=PandasFeatureUnion(
            [
                ("seasonal", feature_engineering.Seasonality(lags=lags, tfreq=tfreq)),
                ("trend", feature_engineering.Trend(lags=lags, tfreq=tfreq)),
                ("diff", feature_engineering.Diff(lags=lags, tfreq=tfreq)),
            ]
        ),
        estimator=Pipeline(
            [
                (
                    "f_scaling",
                    feature_engineering.FeatureScaling(
                        QuantileTransformer(n_quantiles=10, output_distribution="uniform")
                    ),
                ),
                (
                    "f_selection",
                    ml_modelling.FeatureSelection(
                        RFE(RandomForestRegressor(n_estimators=20, random_state=random_state))
                    ),
                ),
                (
                    "model",
                    ml_modelling.Model(
                        RandomForestRegressor(n_estimators=50, random_state=random_state)
                    ),
                ),
            ]
        ),
        random_state=random_state,
    )

run_prediction_pipeline

run_prediction_pipeline(crime_data, study_area, crime_tags=None, time_range=None, tfreq='M', grid_resolution=1, lags=2, random_state=None)

Fit the default pipeline on crime data and forecast the next period.

Parameters:

Name Type Description Default
crime_data DataFrame

Events with tag, t, lon, lat.

required
study_area GeoDataFrame

Study area boundary.

required
crime_tags list

Keep only these crime types.

None
time_range tuple

('HH:MM', 'HH:MM') to keep only events within this time of day.

None
tfreq str

Time frequency ('M', 'W' or 'D').

'M'
grid_resolution float

Grid spacing in kilometers.

1
lags int

Number of lags (and STL period) of the features.

2
random_state int

Seed for reproducibility.

None

Returns:

Type Description
tuple

(predictions, pipeline) — the forecast for the next period and the fitted PredictionPipeline.

Source code in src/predspot/pipeline.py
def run_prediction_pipeline(
    crime_data,
    study_area,
    crime_tags=None,
    time_range=None,
    tfreq="M",
    grid_resolution=1,
    lags=2,
    random_state=None,
):
    """
    Fit the default pipeline on crime data and forecast the next period.

    Args:
        crime_data (pandas.DataFrame): Events with ``tag``, ``t``, ``lon``, ``lat``.
        study_area (GeoDataFrame): Study area boundary.
        crime_tags (list, optional): Keep only these crime types.
        time_range (tuple, optional): ``('HH:MM', 'HH:MM')`` to keep only
            events within this time of day.
        tfreq (str): Time frequency (``'M'``, ``'W'`` or ``'D'``).
        grid_resolution (float): Grid spacing in kilometers.
        lags (int): Number of lags (and STL period) of the features.
        random_state (int, optional): Seed for reproducibility.

    Returns:
        tuple: ``(predictions, pipeline)`` — the forecast for the next period
        and the fitted [`PredictionPipeline`][predspot.ml_modelling.PredictionPipeline].
    """
    missing = [c for c in ("tag", "t", "lat", "lon") if c not in crime_data.columns]
    if missing:
        raise ValueError(f"Crime data must contain columns tag, t, lat, lon; missing {missing}")
    if crime_tags:
        crime_data = crime_data.loc[crime_data["tag"].isin(crime_tags)]
    if time_range:
        time_ix = pd.DatetimeIndex(pd.to_datetime(crime_data["t"]))
        crime_data = crime_data.iloc[time_ix.indexer_between_time(time_range[0], time_range[1])]

    dataset = dataset_preparation.Dataset(crimes=crime_data, study_area=study_area)
    pipeline = build_default_pipeline(
        study_area,
        tfreq=tfreq,
        grid_resolution=grid_resolution,
        lags=lags,
        random_state=random_state,
    )
    pipeline.fit(dataset)
    predictions = pipeline.predict()
    return predictions, pipeline

evaluate_pipeline

evaluate_pipeline(pipeline, scoring='r2', cv=5)

Cross-validate a fitted pipeline; see PredictionPipeline.evaluate.

Parameters:

Name Type Description Default
pipeline PredictionPipeline

A fitted pipeline.

required
scoring str

'r2' or 'mse'.

'r2'
cv int

Number of folds.

5

Returns:

Type Description
list

One score per fold.

Source code in src/predspot/pipeline.py
def evaluate_pipeline(pipeline, scoring="r2", cv=5):
    """
    Cross-validate a fitted pipeline; see ``PredictionPipeline.evaluate``.

    Args:
        pipeline (PredictionPipeline): A fitted pipeline.
        scoring (str): ``'r2'`` or ``'mse'``.
        cv (int): Number of folds.

    Returns:
        list: One score per fold.
    """
    scores = pipeline.evaluate(scoring=scoring, cv=cv)
    logger.debug("Evaluation complete. Mean score: %.4f", np.mean(scores))
    return scores