diff --git a/activitysim/abm/models/summarize.py b/activitysim/abm/models/summarize.py index fb29fe19d6..a2dc451061 100644 --- a/activitysim/abm/models/summarize.py +++ b/activitysim/abm/models/summarize.py @@ -364,6 +364,14 @@ def summarize( out_file = row["Output"] expr = row["Expression"] + # delete temporary variables listed in Expression when Output == "_del" + if out_file == "_del": + logger.debug(f"Deleting temporary variable(s): {expr}") + with performance_timer.time_expression(expr): + for var in str(expr).split(","): + locals_d.pop(var.strip(), None) + continue + # Save temporary variables starting with underscores in locals_d if out_file.startswith("_"): logger.debug(f"Temp Variable: {expr} -> {out_file}") diff --git a/docs/users-guide/visualization.rst b/docs/users-guide/visualization.rst index 3da5a7fc0b..06b9bb7a92 100644 --- a/docs/users-guide/visualization.rst +++ b/docs/users-guide/visualization.rst @@ -36,6 +36,15 @@ Rows with output values that begin with an alphanumeric character will be saved Rows with output values that begin with underscores (e.g., ``_output_name``) will be stored as temporary variables in the local namespace so they can be used in following expressions. Expressions defining temporary variables can produce any data type. Users are encouraged to follow the ActivitySim convention using capitals to denote constants (e.g., ``_TEMP_CONSTANT``), though this convention is not formally enforced for summarize expressions. +Temporary variables that are no longer needed can be removed from the local +namespace, allowing their memory to be reclaimed when no other references +remain. Set the ``Output`` column to the reserved value ``_del`` and list one or more +comma-separated temporary variable names in ``Expression``. Missing names are +ignored, and the row does not create an output file. For example:: + + Description,Output,Expression + Delete intermediate tables,_del,"_trips_with_income, _work_tours" + Summarize expressions can make use of several convenience functions for binning numeric Pandas Series' into quantiles, equal intervals, or manually-specified ranges. These functions are available in the local namespace used to evaluate summarize expressions (as well as for preprocessing the ``trips_merged`` table; see below), so they can be used directly in summary expressions. These functions include: * ``quantiles``: Construct quantiles from a Series given a number of bins. diff --git a/test/summarize/configs/summarize.csv b/test/summarize/configs/summarize.csv index 4a7492ee10..a66388a6d2 100644 --- a/test/summarize/configs/summarize.csv +++ b/test/summarize/configs/summarize.csv @@ -107,3 +107,8 @@ Description,Output,Expression # TAZ population density quintiles ,_taz_pop_dens,land_use.TOTPOP/land_use.TOTACRE ,taz_population_density_quintiles,"quantiles(_taz_pop_dens, 5, '{rank}').rename('pop_dens_quintile').reset_index()" + +# Delete a temporary dataframe (to save memory) and verify it is unavailable to later expressions +,_temporary_dataframe,trips_merged[['number_of_participants']] +,_del,_temporary_dataframe +,temporary_dataframe_deleted,"pd.Series(['_temporary_dataframe' not in locals()], name='deleted')" diff --git a/test/summarize/test_summarize.py b/test/summarize/test_summarize.py index 1faf737193..6c7ee083f4 100644 --- a/test/summarize/test_summarize.py +++ b/test/summarize/test_summarize.py @@ -8,7 +8,7 @@ import pytest # import models is necessary to initalize the model steps -from activitysim.abm import models +from activitysim.abm import models # noqa: F401 from activitysim.core import los, workflow @@ -18,12 +18,13 @@ def initialize_pipeline( tables: dict[str, str], initialize_network_los: bool, base_dir: Path, + tmp_path_factory: pytest.TempPathFactory, ) -> workflow.State: if base_dir is None: base_dir = Path("test").joinpath(module) configs_dir = base_dir.joinpath("configs") data_dir = base_dir.joinpath("data") - output_dir = base_dir.joinpath("output") + output_dir = tmp_path_factory.mktemp("summarize-output") state = ( workflow.State() @@ -50,7 +51,7 @@ def initialize_pipeline( # Add the dataframes to the pipeline state.checkpoint.restore() - state.checkpoint.add(module) + state.checkpoint.add("init") state.checkpoint.close_store() # By convention, this method needs to yield something @@ -122,8 +123,6 @@ def test_summarize(initialize_pipeline: workflow.State, caplog): output_location = ( model_settings["OUTPUT"] if "OUTPUT" in model_settings else "summaries" ) - output_dir = state.get_output_file_path(output_location) - # Check that households are counted correctly households_count = pd.read_csv( state.get_output_file_path( @@ -143,3 +142,11 @@ def test_summarize(initialize_pipeline: workflow.State, caplog): assert int(trips_by_mode_count.BIKE.iloc[0]) == len( trips[trips.trip_mode == "BIKE"] ) + + # Check that _del removes temporary dataframes from the expression namespace + temporary_dataframe_deleted = pd.read_csv( + state.get_output_file_path( + os.path.join(output_location, "temporary_dataframe_deleted.csv") + ) + ) + assert temporary_dataframe_deleted["deleted"].tolist() == [True]