Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 8 additions & 0 deletions activitysim/abm/models/summarize.py
Original file line number Diff line number Diff line change
Expand Up @@ -364,6 +364,14 @@ def summarize(
out_file = row["Output"]
expr = row["Expression"]

# delete temporary variables listed in Expression when Output == "_del"
if out_file == "_del":
logger.debug(f"Deleting temporary variable(s): {expr}")
with performance_timer.time_expression(expr):
for var in str(expr).split(","):
locals_d.pop(var.strip(), None)
continue

# Save temporary variables starting with underscores in locals_d
if out_file.startswith("_"):
logger.debug(f"Temp Variable: {expr} -> {out_file}")
Expand Down
9 changes: 9 additions & 0 deletions docs/users-guide/visualization.rst
Original file line number Diff line number Diff line change
Expand Up @@ -36,6 +36,15 @@ Rows with output values that begin with an alphanumeric character will be saved

Rows with output values that begin with underscores (e.g., ``_output_name``) will be stored as temporary variables in the local namespace so they can be used in following expressions. Expressions defining temporary variables can produce any data type. Users are encouraged to follow the ActivitySim convention using capitals to denote constants (e.g., ``_TEMP_CONSTANT``), though this convention is not formally enforced for summarize expressions.

Temporary variables that are no longer needed can be removed from the local
namespace, allowing their memory to be reclaimed when no other references
remain. Set the ``Output`` column to the reserved value ``_del`` and list one or more
comma-separated temporary variable names in ``Expression``. Missing names are
ignored, and the row does not create an output file. For example::

Description,Output,Expression
Delete intermediate tables,_del,"_trips_with_income, _work_tours"

Summarize expressions can make use of several convenience functions for binning numeric Pandas Series' into quantiles, equal intervals, or manually-specified ranges. These functions are available in the local namespace used to evaluate summarize expressions (as well as for preprocessing the ``trips_merged`` table; see below), so they can be used directly in summary expressions. These functions include:

* ``quantiles``: Construct quantiles from a Series given a number of bins.
Expand Down
5 changes: 5 additions & 0 deletions test/summarize/configs/summarize.csv
Original file line number Diff line number Diff line change
Expand Up @@ -107,3 +107,8 @@ Description,Output,Expression
# TAZ population density quintiles
,_taz_pop_dens,land_use.TOTPOP/land_use.TOTACRE
,taz_population_density_quintiles,"quantiles(_taz_pop_dens, 5, '{rank}').rename('pop_dens_quintile').reset_index()"

# Delete a temporary dataframe (to save memory) and verify it is unavailable to later expressions
,_temporary_dataframe,trips_merged[['number_of_participants']]
,_del,_temporary_dataframe
,temporary_dataframe_deleted,"pd.Series(['_temporary_dataframe' not in locals()], name='deleted')"
17 changes: 12 additions & 5 deletions test/summarize/test_summarize.py
Original file line number Diff line number Diff line change
Expand Up @@ -8,7 +8,7 @@
import pytest

# import models is necessary to initalize the model steps
from activitysim.abm import models
from activitysim.abm import models # noqa: F401
from activitysim.core import los, workflow


Expand All @@ -18,12 +18,13 @@ def initialize_pipeline(
tables: dict[str, str],
initialize_network_los: bool,
base_dir: Path,
tmp_path_factory: pytest.TempPathFactory,
) -> workflow.State:
if base_dir is None:
base_dir = Path("test").joinpath(module)
configs_dir = base_dir.joinpath("configs")
data_dir = base_dir.joinpath("data")
output_dir = base_dir.joinpath("output")
output_dir = tmp_path_factory.mktemp("summarize-output")

state = (
workflow.State()
Expand All @@ -50,7 +51,7 @@ def initialize_pipeline(

# Add the dataframes to the pipeline
state.checkpoint.restore()
state.checkpoint.add(module)
state.checkpoint.add("init")
state.checkpoint.close_store()

# By convention, this method needs to yield something
Expand Down Expand Up @@ -122,8 +123,6 @@ def test_summarize(initialize_pipeline: workflow.State, caplog):
output_location = (
model_settings["OUTPUT"] if "OUTPUT" in model_settings else "summaries"
)
output_dir = state.get_output_file_path(output_location)

# Check that households are counted correctly
households_count = pd.read_csv(
state.get_output_file_path(
Expand All @@ -143,3 +142,11 @@ def test_summarize(initialize_pipeline: workflow.State, caplog):
assert int(trips_by_mode_count.BIKE.iloc[0]) == len(
trips[trips.trip_mode == "BIKE"]
)

# Check that _del removes temporary dataframes from the expression namespace
temporary_dataframe_deleted = pd.read_csv(
state.get_output_file_path(
os.path.join(output_location, "temporary_dataframe_deleted.csv")
)
)
assert temporary_dataframe_deleted["deleted"].tolist() == [True]
Loading