diff --git a/.github/workflows/httomo_docs.yml b/.github/workflows/httomo_docs.yml index 375284420..d3ecb92fb 100644 --- a/.github/workflows/httomo_docs.yml +++ b/.github/workflows/httomo_docs.yml @@ -8,6 +8,10 @@ on: push: branches: - main + tags: + - 'v*' + schedule: + - cron: '0 6 * * 1' jobs: build-docs-publish: @@ -33,8 +37,7 @@ jobs: init-shell: bash - name: Install httomo-backends - run: | - pip install --no-deps httomo-backends + run: pip install --no-deps -r ./docs/source/doc-pip-requirements.txt - name: Generate full yaml pipelines using pipeline directives run: | @@ -47,10 +50,18 @@ jobs: zip -r pipelines_full_artifact.zip ./docs/source/pipelines_full/ - name: Build docs - run: sphinx-build -a -E -b html ./docs/source/ ./docs/build/ + run: sphinx-build -W --keep-going -a -E -b html ./docs/source/ ./docs/build/ + + - name: Check external links + if: github.event_name == 'schedule' + run: sphinx-build -W --keep-going -a -E -b linkcheck ./docs/source/ ./docs/linkcheck/ - name: Publish docs - if: github.ref_type == 'tag' || github.ref_name == 'main' + if: >- + (github.event_name == 'push' && + (github.ref_type == 'tag' || github.ref_name == 'main')) || + (github.event_name == 'workflow_dispatch' && + github.ref_name == 'main') run: ghp-import -n -p -f ./docs/build env: GITHUB_TOKEN: ${{ github.token }} @@ -59,4 +70,4 @@ jobs: uses: actions/upload-artifact@v4 with: name: full_pipelines-artifact - path: pipelines_full_artifact.zip \ No newline at end of file + path: pipelines_full_artifact.zip diff --git a/README.rst b/README.rst index fd2cec627..d0faea49c 100644 --- a/README.rst +++ b/README.rst @@ -1,26 +1,25 @@ -HTTomo (High Throughput Tomography pipeline) -******************************************************* +High Throughput Tomography software +*********************************** -HTTomo is a user interface (UI) written in Python for fast big data processing using MPI protocols. -It orchestrates I/O data operations and enables processing on a CPU and/or a GPU. HTTomo utilises other libraries, such as `TomoPy `_ and `HTTomolibgpu `_ -as backends for data processing. The methods from the libraries are exposed through YAML templates to enable fast task programming. - -Installation -============ -See detailed instructions for `installation `_ . +HTTomo is a Python framework for high-performance tomographic data processing mainly targeting GPU-compute. +It orchestrates distributed I/O and CPU/GPU workflows using MPI, while providing +YAML-based access to processing methods from libraries such as +`TomoPy `_ and `HTTomolibgpu `_. Documentation -============== -Please check the full `documentation `_. +============= + +The `HTTomo documentation `_ +contains installation instructions, a quickstart, ready-to-use pipelines, +reference material and developer guidance. + +After installation, use :code:`python -m httomo --help` to inspect the command +line interface. A typical workflow is to validate a pipeline and then run it: -Running HTTomo: -================ +.. code-block:: console -* Install the module following any chosen `installation `_ path. -* For help with the command line interface, execute :code:`python -m httomo --help` -* Choose the existing `YAML pipeline `_ or build a new one using ready-to-be-used `templates `_. -* Optional: perform the validity check of the YAML pipeline file with the `YAML checker `_. -* Run HTTomo with :code:`python -m httomo run [OPTIONS] IN_DATA_FILE YAML_CONFIG OUT_DIR`, see more on that `here `_. + python -m httomo check pipeline.yaml input.nxs + python -m httomo run input.nxs pipeline.yaml output_directory Release Tagging Scheme ====================== diff --git a/docs/build_sphinx_docs.bat b/docs/build_sphinx_docs.bat new file mode 100644 index 000000000..6cf6133b8 --- /dev/null +++ b/docs/build_sphinx_docs.bat @@ -0,0 +1,44 @@ +@echo off +setlocal + +echo ******************************************************************************** +echo Starting the Sphinx script +echo ******************************************************************************** +echo Creating plugin API files and HTML pages... + +rem Resolve all paths relative to the directory containing this batch file. +set "SCRIPT_DIR=%~dp0" + +rem Remove old generated API files and build output to avoid obsolete files. +if exist "%SCRIPT_DIR%source\developers\generated" ( + rmdir /s /q "%SCRIPT_DIR%source\developers\generated" + if errorlevel 1 goto :cleanup_failed +) + +if exist "%SCRIPT_DIR%build" ( + rmdir /s /q "%SCRIPT_DIR%build" + if errorlevel 1 goto :cleanup_failed +) + +rem -a writes all output files. +rem -E rebuilds the Sphinx environment without using the saved cache. +rem -b html selects the HTML builder. +rem -W treats warnings as errors; --keep-going reports all warnings in one run. +sphinx-build -W --keep-going -a -E -b html "%SCRIPT_DIR%source" "%SCRIPT_DIR%build" +set "EXIT_CODE=%ERRORLEVEL%" + +if not "%EXIT_CODE%"=="0" ( + echo. + echo ERROR: Sphinx build failed with exit code %EXIT_CODE%. +) else ( + echo. + echo Sphinx documentation built successfully. + echo Output: "%SCRIPT_DIR%build\index.html" +) + +exit /b %EXIT_CODE% + +:cleanup_failed +echo. +echo ERROR: Unable to remove an existing generated or build directory. +exit /b 1 diff --git a/docs/source/_static/3d_setup.png b/docs/source/_static/3d_setup.png deleted file mode 100644 index 81efc91e5..000000000 Binary files a/docs/source/_static/3d_setup.png and /dev/null differ diff --git a/docs/source/_static/add_method_workflow.svg b/docs/source/_static/add_method_workflow.svg new file mode 100644 index 000000000..7b7288c1d --- /dev/null +++ b/docs/source/_static/add_method_workflow.svg @@ -0,0 +1,103 @@ + + + Workflow for adding a processing method to HTTomo + Four steps show a developer implementing and testing a method in a backend library, confirming that its interface fits an HTTomo wrapper, registering its execution metadata and YAML template in httomo-backends, and validating and running a small HTTomo pipeline. If no existing wrapper fits, the developer extends HTTomo and tests the wrapper before registering the method. + + + + + + + + + + From processing function to HTTomo pipeline method + + + + + 1 + BACKEND LIBRARY + Implement the method + Write the processing function + Expose it from its module + Add focused unit tests + + + + + + + 2 + HTTOMO INTERFACE + Check the wrapper + Does an existing wrapper + support the function? + YES + + + + + + + 3 + HTTOMO-BACKENDS + Register the method + Add execution metadata + Add memory/shape helpers + Generate the YAML template + + + + + + + 4 + HTTOMO + Verify the integration + Validate a small pipeline + Run representative data + Submit linked changes + + + + NO + + Extend HTTomo's wrapper layer + Add wrapper tests, then continue + + diff --git a/docs/source/_static/blocks_chunks/blocks.png b/docs/source/_static/blocks_chunks/blocks.png new file mode 100644 index 000000000..a394c7f93 Binary files /dev/null and b/docs/source/_static/blocks_chunks/blocks.png differ diff --git a/docs/source/_static/blocks_chunks/chunks.png b/docs/source/_static/blocks_chunks/chunks.png new file mode 100644 index 000000000..33012112b Binary files /dev/null and b/docs/source/_static/blocks_chunks/chunks.png differ diff --git a/docs/source/_static/blocks_chunks/chunks_blocks.png b/docs/source/_static/blocks_chunks/chunks_blocks.png deleted file mode 100644 index ecad343c8..000000000 Binary files a/docs/source/_static/blocks_chunks/chunks_blocks.png and /dev/null differ diff --git a/docs/source/_static/blocks_chunks/chunks_blocks.svg b/docs/source/_static/blocks_chunks/chunks_blocks.svg deleted file mode 100644 index 4fb7f104f..000000000 --- a/docs/source/_static/blocks_chunks/chunks_blocks.svg +++ /dev/null @@ -1,1665 +0,0 @@ - - - - - - - - - - - - - - - - - - - - - - - - - - - - - chunk1 - 2 - 3 - 4 - - - - - - - - - - - block 1 - input blocks - processed blocks - 2 - 3 - ... - - - - series of CPU/GPU methods - 4 processes - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - diff --git a/docs/source/_static/custom.css b/docs/source/_static/custom.css new file mode 100644 index 000000000..f664011e9 --- /dev/null +++ b/docs/source/_static/custom.css @@ -0,0 +1,32 @@ +/* Highlight selected pages in the primary (left) navigation only. + * + * The direct href selectors cover inactive links. When one of these pages is + * active, Sphinx replaces its href with "#". The :has() selectors identify + * the same navigation group through the other named link and select the + * active entry by its stable first/last position. + */ +.bd-sidebar-primary .bd-sidenav a[href$="yaml.html"], +.bd-sidebar-primary .bd-sidenav:has(a[href$="versioned_downloads.html"]) + > li:first-child + > a, +.bd-sidebar-primary .bd-sidenav a[href$="versioned_downloads.html"], +.bd-sidebar-primary .bd-sidenav:has(a[href$="yaml.html"]) + > li:last-child + > a { + background-color: rgb(33 150 243 / 18%); + box-shadow: inset 4px 0 #2196f3; + border-radius: 0.25rem; + font-weight: 600; +} + +/* The run_httomo link identifies the Getting started navigation group when + * Sphinx replaces the active quickstart link with href="#". */ +.bd-sidebar-primary .bd-sidenav a[href$="quickstart.html"], +.bd-sidebar-primary .bd-sidenav:has(a[href$="run_httomo.html"]) + > li:nth-child(2) + > a { + background-color: rgb(255 193 7 / 22%); + box-shadow: inset 4px 0 #d39e00; + border-radius: 0.25rem; + font-weight: 600; +} diff --git a/docs/source/_static/execution_model.svg b/docs/source/_static/execution_model.svg new file mode 100644 index 000000000..f950186d8 --- /dev/null +++ b/docs/source/_static/execution_model.svg @@ -0,0 +1,154 @@ + + + HTTomo pipeline execution model + Four-stage overview showing a YAML pipeline being validated, the pipeline being divided into sections, data being distributed into MPI rank chunks within each section, chunks being processed block by block through the section methods, and processing moving to the next section or writing results. + + + + + + + + + + + + + 1 + + Create the pipeline + + YAML pipeline + + + Validate + methods and parameters + + + Executable pipeline + + + + + 2 + + Divide the pipeline into sections + Group compatible methods that use the same data pattern + + Executable pipeline + + + Section 1 + methods A–C • projections + + + Section 2 + methods D–E • sinograms + … + + + + + 3 + + Execute one section + + Inside the current section + + 1. Distribute the section data into MPI rank chunks + + Section input + + + MPI rank 0 + chunk 0 + + MPI rank 1 + chunk 1 + … + + MPI rank N−1 + chunk N−1 + + + 2. Split each rank's chunk into memory-sized blocks + + chunk 0 + + + block 1 + + block 2 + + block 3 + … + + 3. Pass one block through every method before starting the next block + + one block + + + method A + + + method B + + + method C + then the + next block + + + + + 4 + + Move to the next section or finish + + Section complete + + + Prepare section boundary + synchronise • re-slice • redistribute + + + + Next section + + Results + diff --git a/docs/source/_static/gpu_reconstruction_httomo.svg b/docs/source/_static/gpu_reconstruction_httomo.svg new file mode 100644 index 000000000..b4cde24be --- /dev/null +++ b/docs/source/_static/gpu_reconstruction_httomo.svg @@ -0,0 +1,2132 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + 2026-09-09T13:59:47.337999 + image/svg+xml + + + Matplotlib v3.10.8, https://matplotlib.org/ + + + + + + + + + + + + + + + GPU Reconstruction ecosystem in HTTomo + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + Direct: FBP / ASTRA + Log-Polar / FFT-CuPy + Iterative methods + Data fidelity + regularisation + Plug-and-play reconstruction + + Open ASTRA Toolbox website + + + + Open ToMoBAR website + + + + Open CuPy website + + + + + + HTTomolibGPU + Higher-level processing library + + + + diff --git a/docs/source/_static/httomo-workflow-dark.png b/docs/source/_static/httomo-workflow-dark.png new file mode 100644 index 000000000..a439dbc8b Binary files /dev/null and b/docs/source/_static/httomo-workflow-dark.png differ diff --git a/docs/source/_static/logo_dark.svg b/docs/source/_static/logo_dark.svg deleted file mode 100644 index 10ef8f667..000000000 --- a/docs/source/_static/logo_dark.svg +++ /dev/null @@ -1,467 +0,0 @@ - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - HTTomo - - - - - - - - - - - - - - - diff --git a/docs/source/_static/logo_light.svg b/docs/source/_static/logo_light.svg deleted file mode 100644 index 6c65eb4cd..000000000 --- a/docs/source/_static/logo_light.svg +++ /dev/null @@ -1,458 +0,0 @@ - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - HTTomo - - - - - - - - diff --git a/docs/source/_static/memory_estimators/gpu_memory_estimation_vector.png b/docs/source/_static/memory_estimators/gpu_memory_estimation_vector.png new file mode 100644 index 000000000..56c872e8b Binary files /dev/null and b/docs/source/_static/memory_estimators/gpu_memory_estimation_vector.png differ diff --git a/docs/source/_static/memory_estimators/memory_estimators.svg b/docs/source/_static/memory_estimators/memory_estimators.svg deleted file mode 100644 index e1abf354d..000000000 --- a/docs/source/_static/memory_estimators/memory_estimators.svg +++ /dev/null @@ -1,1032 +0,0 @@ - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - HTTomo - - - - - - - - - - - - - - - - - GPU memoryavailable - Methods insection - chain of GPU methods - Optimal blocksize for the section - that fits into thechosen GPU device - - - - - - CuPy - - CuPy - - - - - DataSource - - DataSink - - Section (a loop over blocks) - Memory estimatorsfor each exposed method - - - - - diff --git a/docs/source/_static/memory_estimators/memoryest_1.png b/docs/source/_static/memory_estimators/memoryest_1.png deleted file mode 100644 index 387139d0a..000000000 Binary files a/docs/source/_static/memory_estimators/memoryest_1.png and /dev/null differ diff --git a/docs/source/_static/memory_estimators/memoryest_2.png b/docs/source/_static/memory_estimators/memoryest_2.png deleted file mode 100644 index b193e8754..000000000 Binary files a/docs/source/_static/memory_estimators/memoryest_2.png and /dev/null differ diff --git a/docs/source/_static/real_data/recon_sandstone.png b/docs/source/_static/real_data/recon_sandstone.png new file mode 100644 index 000000000..ed774f459 Binary files /dev/null and b/docs/source/_static/real_data/recon_sandstone.png differ diff --git a/docs/source/_static/real_data/stripe_vo_recon.png b/docs/source/_static/real_data/stripe_vo_recon.png new file mode 100644 index 000000000..c3535c42a Binary files /dev/null and b/docs/source/_static/real_data/stripe_vo_recon.png differ diff --git a/docs/source/_static/reslice_gather/gather_chunks.png b/docs/source/_static/reslice_gather/gather_chunks.png deleted file mode 100644 index 4c1afcc48..000000000 Binary files a/docs/source/_static/reslice_gather/gather_chunks.png and /dev/null differ diff --git a/docs/source/_static/reslice_gather/gather_chunks.svg b/docs/source/_static/reslice_gather/gather_chunks.svg deleted file mode 100644 index 8ce32db63..000000000 --- a/docs/source/_static/reslice_gather/gather_chunks.svg +++ /dev/null @@ -1,1886 +0,0 @@ - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - chunk1 - 2 - 3 - 4 - - - - - - - - - block 1 - 2 - 3 - ... - 4 processes - - - - - - - - - - - - - diff --git a/docs/source/_static/reslice_gather/httomo_reslicing_dark.png b/docs/source/_static/reslice_gather/httomo_reslicing_dark.png new file mode 100644 index 000000000..2b042c577 Binary files /dev/null and b/docs/source/_static/reslice_gather/httomo_reslicing_dark.png differ diff --git a/docs/source/_static/reslice_gather/reslice.png b/docs/source/_static/reslice_gather/reslice.png deleted file mode 100644 index ab199a5d8..000000000 Binary files a/docs/source/_static/reslice_gather/reslice.png and /dev/null differ diff --git a/docs/source/_static/sections/sections.svg b/docs/source/_static/sections/sections.svg deleted file mode 100644 index bd96955b4..000000000 --- a/docs/source/_static/sections/sections.svg +++ /dev/null @@ -1,307 +0,0 @@ - - - - - - - - - - - - - - - - - - - - - - - - M1 - M3 - M2 - T1 - T2 - M5 - M4 - T3 - T4 - - L - - pipeline execution direction - - diff --git a/docs/source/_static/sections/sections1.png b/docs/source/_static/sections/sections1.png deleted file mode 100644 index dfec17177..000000000 Binary files a/docs/source/_static/sections/sections1.png and /dev/null differ diff --git a/docs/source/_static/sections/sections2.png b/docs/source/_static/sections/sections2.png deleted file mode 100644 index 874a86e29..000000000 Binary files a/docs/source/_static/sections/sections2.png and /dev/null differ diff --git a/docs/source/_static/sections/sections2.svg b/docs/source/_static/sections/sections2.svg deleted file mode 100644 index cc59ddb7c..000000000 --- a/docs/source/_static/sections/sections2.svg +++ /dev/null @@ -1,356 +0,0 @@ - - - - - - - - - - - - - - - - - - - - - - - - - - M1 - M3 - M2 - T1 - T2 - M5 - M4 - T3 - T4 - - L - - pipeline execution direction - section 1 - section 2 - - diff --git a/docs/source/_static/sections/sections3.png b/docs/source/_static/sections/sections3.png deleted file mode 100644 index c92d00783..000000000 Binary files a/docs/source/_static/sections/sections3.png and /dev/null differ diff --git a/docs/source/_static/sections/sections3.svg b/docs/source/_static/sections/sections3.svg deleted file mode 100644 index 28b168a13..000000000 --- a/docs/source/_static/sections/sections3.svg +++ /dev/null @@ -1,384 +0,0 @@ - - - - - - - - - - - - - - - - - - - - - - - - - - - M1 - M3 - M2 - T1 - T2 - M5 - M4 - T3 - T4 - - L - - pipeline execution direction - section 1 - sec 2 - section 3 - - diff --git a/docs/source/_static/sections/sections4.png b/docs/source/_static/sections/sections4.png deleted file mode 100644 index a9f83b41b..000000000 Binary files a/docs/source/_static/sections/sections4.png and /dev/null differ diff --git a/docs/source/_static/sections/sections4.svg b/docs/source/_static/sections/sections4.svg deleted file mode 100644 index def5a79fd..000000000 --- a/docs/source/_static/sections/sections4.svg +++ /dev/null @@ -1,460 +0,0 @@ - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - M1 - M3 - M2 - T1 - T2 - M5 - M4 - T3 - T4 - - L - - pipeline execution direction - section 2 - sec 1 - sec 4 - sec 3 - - side output - - diff --git a/docs/source/_static/sections/sections_concept.png b/docs/source/_static/sections/sections_concept.png new file mode 100644 index 000000000..ff258ec3c Binary files /dev/null and b/docs/source/_static/sections/sections_concept.png differ diff --git a/docs/source/_static/side_output_reference.svg b/docs/source/_static/side_output_reference.svg new file mode 100644 index 000000000..d54a86602 --- /dev/null +++ b/docs/source/_static/side_output_reference.svg @@ -0,0 +1,203 @@ + + + + Passing a side output between HTTomo methods + Data and angles flow directly from a centre-finding method to a reconstruction method. A separate named centre-of-rotation value is referenced by the reconstruction method's center parameter. + + + + + + + + + + + Find centre of rotation + + Method ID + centering + + Side output + cor → centre_of_rotation + + Reconstruct + + Main input + data, angles + + Parameter + center: ${{ centering.side_outputs. + centre_of_rotation }} + + main data flow + + named supplementary value + diff --git a/docs/source/_static/synth_data_screenshot.png b/docs/source/_static/synth_data_screenshot.png new file mode 100644 index 000000000..b1c1833e8 Binary files /dev/null and b/docs/source/_static/synth_data_screenshot.png differ diff --git a/docs/source/_static/wrappers/httomo_wrappers.png b/docs/source/_static/wrappers/httomo_wrappers.png new file mode 100644 index 000000000..206fa56a5 Binary files /dev/null and b/docs/source/_static/wrappers/httomo_wrappers.png differ diff --git a/docs/source/_static/wrappers/wrappers.png b/docs/source/_static/wrappers/wrappers.png deleted file mode 100644 index 20e29e8af..000000000 Binary files a/docs/source/_static/wrappers/wrappers.png and /dev/null differ diff --git a/docs/source/backends/list.rst b/docs/source/backends/list.rst index 63ce02027..de0109930 100644 --- a/docs/source/backends/list.rst +++ b/docs/source/backends/list.rst @@ -1,43 +1,72 @@ .. _backends_list: -Supported libraries -=================== - -HTTomo currently supports several software packages that are used as -`backends` to perform data processing and reconstruction. The list of -the external packages will be growing in future to meet users' needs. - -If the package has a modular structure with an easy access to every -method, for example as in the `TomoPy `_ -software or in the `scikit `_ library, then the -integration process is straightforward. - -A YAML template can be generated by using `YAML generator `_, -or manually. More complicated in structure packages would need an additional wrapping, see :ref:`info_wrappers`. - -Please see the provided list of :ref:`reference_templates`. - -HTTomolibgpu library (GPU) --------------------------- -`HTTomolibgpu `_ library is developed at `Diamond Light source `_ -by Data Analysis Group to work together with the HTTomo software. - -* HTTomolibgpu is a Python library of GPU accelerated methods written using `CuPy `_ API and CUDA language. -* Most of the original methods have been taken from TomoPy or `Savu `_ software and then re-optimised and GPU-accelerated. -* It is a fully modular library and methods can be used stand-alone, however the GPU memory distribution feature of HTTomo will not be available. This means that the methods can silently fail if the GPU memory is overflowed. - -HTTomolib library (CPU) --------------------------- -`HTTomolib `_ library is similar to HTTomolibgpu, but contains mostly CPU modules. - -TomoPy software (CPU) ---------------------- -`TomoPy `_ is an open-source Python package for -tomographic data processing and image reconstruction developed at -`The Advanced Photon Source `_ in Illinois, USA. -The project is active since 2013 and it gained a `large audience `_ -of users and contributors across tomographic imaging community. - -* TomoPy is an open-source package in Python and C for data processing and reconstruction. TomoPy is mostly a CPU processing library and in HTTomo we expose the CPU modules only. -* It is a CPU-multithreaded package. HTTomo controls parallelisation through MPI on a higher level and also supports local CPU multithreading from TomoPy, for every MPI process. -* TomoPy is a library of stand-alone methods which can be easily integrated into HTTomo. Notably not all TomoPy methods are integrated in HTTomo because of the I/O nature of some modules. Please see the list of available TomoPy :ref:`reference_templates`. \ No newline at end of file +Processing libraries +==================== + +HTTomo coordinates loading, processing and reconstruction methods supplied by +scientific software libraries. A pipeline may combine methods from several +:term:`backends `. + +.. list-table:: + :header-rows: 1 + :widths: 20 12 31 37 + + * - Library + - Processor + - Choose it for + - Notes + * - `HTTomolibGPU`_ + - GPU + - Accelerated preprocessing, artefact correction and reconstruction + - Uses CuPy and CUDA; several reconstruction methods also use TomoBAR or + ASTRA. + * - `HTTomolib`_ + - CPU + - CPU processing and image-output methods maintained for HTTomo + - Useful in CPU pipelines and for output stages following GPU processing. + * - `TomoPy`_ + - CPU + - Established tomography preprocessing and reconstruction methods + - CPU methods exposed through HTTomo can use local multithreading within + each MPI process. + +Only methods listed in :ref:`reference_templates` have the metadata and YAML +templates required for direct use in an HTTomo pipeline. Not every function in +a backend library is exposed. + +HTTomolibGPU +------------ + +HTTomolibGPU is developed by the Data Analysis Group at +`Diamond Light Source`_ for GPU-accelerated tomography. Its methods can be +called independently, but HTTomo adds pipeline orchestration, block sizing, +distributed I/O and GPU memory management. See :ref:`reconstruction_ecosystem` +for the relationship between HTTomolibGPU, TomoBAR, ASTRA and CuPy. + +HTTomolib +--------- + +HTTomolib contains CPU methods maintained alongside HTTomo. It also supplies +utilities such as rescaling and image writing that are commonly used at the +end of both CPU and GPU pipelines. + +TomoPy +------ + +TomoPy is an open-source tomography package from the wider imaging community. +HTTomo exposes a selected set of its CPU preprocessing and reconstruction +functions. HTTomo distributes data between MPI processes; TomoPy may also use +CPU threads inside each process. + +Adding another method or library +-------------------------------- + +Processing-code integration belongs in the developer guide. See +:ref:`developers_add_own_method` for the complete method workflow and +:ref:`developers_httomo_backends` for registering metadata after a method has +been implemented. + +.. _HTTomolibGPU: https://github.com/DiamondLightSource/httomolibgpu +.. _HTTomolib: https://github.com/DiamondLightSource/httomolib +.. _TomoPy: https://tomopy.readthedocs.io +.. _Diamond Light Source: https://www.diamond.ac.uk/ diff --git a/docs/source/backends/templates.rst b/docs/source/backends/templates.rst index da919b863..87d4bb58b 100644 --- a/docs/source/backends/templates.rst +++ b/docs/source/backends/templates.rst @@ -1,41 +1,34 @@ .. _reference_templates: -====================== -Methods YAML Templates -====================== +Available methods +================= -This section refers to YAML templates from the :ref:`backends_list`, which are distributed separately through `HTTomo-backends `_. -Here you can find either fixed (archived) templates that were previously released, or dynamically changing latest templates. -If you installed HTTomo with a specific version associated with the release, please use the :ref:`archived_templates`. If you are a developer, or -you need to access the latest developments, please use the :ref:`latest_templates`. +HTTomo processing methods are provided by the :ref:`backends_list`. Each +available method has a YAML template containing its module path, parameters +and default values. Copy these templates when :ref:`configuring a pipeline +`. -.. _latest_templates: - -Latest Templates -================ - -The templates below are generated for the **current/latest** HTTomo development. These templates can be updated frequently and -if you need to use templates that are linked to a specific released (tagged) HTTomo version, please see the archives above. - -.. note:: At DLS, the templates bellow should work with the :code:`httomo/latest` module. - -`HTTomolibgpu Modules `_ +The generated references below track the latest HTTomo development version. +For a tagged HTTomo release, download the matching templates from +:ref:`versioned_downloads`. -`HTTomolib Modules `_ - -`TomoPy Modules `_ - -.. _archived_templates: - -Archived Templates -=================== +.. _latest_templates: -These are archived YAML templates that can be used with already released and tagged version of HTTomo. +Current method reference +------------------------ -.. only:: builder_html +The method templates are generated and published by `HTTomo-backends +`_. Select the library +used by your pipeline: - :download:`HTTomo version 3.1 templates <../templates_archive/httomo_ver3_1_yaml_templates.zip>` +* `HTTomolibGPU methods (GPU) + `_ +* `HTTomolib methods (CPU) + `_ +* `TomoPy methods (CPU) + `_ - :download:`HTTomo version 3.2 templates <../templates_archive/httomo_ver3_2_yaml_templates.zip>` - - :download:`HTTomo version 3.3 templates <../templates_archive/httomo_ver3_3_yaml_templates.zip>` +.. note:: + At Diamond Light Source, the current references correspond to the + :code:`httomo/latest` module. Use :ref:`versioned_downloads` for production + runs tied to a particular release. diff --git a/docs/source/conf.py b/docs/source/conf.py index 5d35ac904..ca2923df4 100644 --- a/docs/source/conf.py +++ b/docs/source/conf.py @@ -11,7 +11,7 @@ # # Unless required by applicable law or agreed to in writing, software # distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either ecpress or implied. +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. # See the License for the specific language governing permissions and # limitations under the License. # --------------------------------------------------------------------------- @@ -23,6 +23,8 @@ import os import sys from datetime import date +from importlib.metadata import PackageNotFoundError, version as package_version +import subprocess from unittest import mock # If extensions (or modules to document with autodoc) are in another directory, @@ -57,7 +59,8 @@ def __repr__(self): sys.modules["scipy"] = CustomMock() sys.modules["scipy.signal"] = CustomMock() - +sys.modules["loguru"] = CustomMock() +sys.modules["graypy"] = CustomMock() # ------------------------------------------------------------------------------ @@ -68,11 +71,29 @@ def __repr__(self): # Specify a base language to help assistive technology language = "en" -# Save the commit hash, this is displayed in the page title -release = os.popen('git log -1 --format="%H"').read().strip() -# Set version as the latest tag in the current branch -version = os.popen("git describe --tags --abbrev=0").read().strip() +def _git_value(*args: str) -> str: + """Return a Git value without making documentation builds depend on Git.""" + try: + result = subprocess.run( + ["git", *args], + check=True, + capture_output=True, + text=True, + ) + except (FileNotFoundError, subprocess.CalledProcessError): + return "" + return result.stdout.strip() + + +try: + installed_version = package_version("httomo") +except PackageNotFoundError: + installed_version = "development" + +# The commit and nearest tag are displayed in the generated documentation. +release = _git_value("rev-parse", "HEAD") or installed_version +version = _git_value("describe", "--tags", "--abbrev=0") or installed_version # Add any Sphinx extension module names here, as strings. They can be extensions # coming with Sphinx (named 'sphinx.ext.*') or your custom ones. @@ -111,6 +132,7 @@ def __repr__(self): html_favicon = "_static/logo_light.png" html_last_updated_fmt = "" html_static_path = ["_static"] +html_css_files = ["custom.css"] html_use_smartypants = True html_theme_options = { @@ -122,8 +144,17 @@ def __repr__(self): } html_context = { - "github_user": "HTTomo", - "github_repo": "https://github.com/DiamondLightSource/httomo", + "github_user": "DiamondLightSource", + "github_repo": "httomo", "github_version": "main", - "doc_path": "docs", + "doc_path": "docs/source", } + +# These pages are valid in a browser but reject or cannot be verified by the +# automated link checker. Keep this list narrow so other external links remain +# covered by the scheduled documentation check. +linkcheck_ignore = [ + r"https://www\.diamond\.ac\.uk/$", + r"https://www\.diamond\.ac\.uk/Home/About/Vision/Diamond-II\.html$", + r"https://www\.silx\.org/doc/silx/latest/applications/view\.html$", +] diff --git a/docs/source/developers/add_own_method.rst b/docs/source/developers/add_own_method.rst new file mode 100644 index 000000000..d74be2d15 --- /dev/null +++ b/docs/source/developers/add_own_method.rst @@ -0,0 +1,25 @@ +.. _developers_add_own_method: + +Adding a processing method +************************** + +Adding a method spans a processing library, ``httomo-backends`` and HTTomo. +Use the following workflow to keep each part in the correct project: + +.. figure:: ../_static/add_method_workflow.svg + :align: center + :alt: Four-step workflow for implementing, registering and validating a new HTTomo processing method + :width: 100% + + Implement and test the function first; describe it for HTTomo only after its + interface is stable. + +HTTomo usually exposes processing functions supplied by a separate scientific +library. Functions from modular packages can often be integrated directly when +they accept an array and explicit processing parameters. Functions with more +complex interfaces may require an HTTomo :ref:`wrapper `. + +After integrating and testing the function in its backend library, register it +in ``httomo-backends``. That package supplies the execution metadata HTTomo +needs for sectioning and memory management, and generates the user-facing YAML +template. Follow :ref:`developers_httomo_backends` for the complete workflow. diff --git a/docs/source/developers/api.rst b/docs/source/developers/api.rst index 1a244ab8b..5cb92fa1b 100644 --- a/docs/source/developers/api.rst +++ b/docs/source/developers/api.rst @@ -1,6 +1,104 @@ +.. _api: + API === +This reference is generated from HTTomo's Python modules. Most of these +modules support the framework's internal execution machinery rather than the +scientific methods used in a YAML pipeline. Backend method developers should +normally begin with :ref:`developers_add_own_method` and +:ref:`developers_httomo_backends`; use this API when extending HTTomo itself or +integrating with its runner interfaces. + +Module overview +--------------- + +.. list-table:: + :header-rows: 1 + :widths: 28 72 + + * - Module + - Responsibility + * - ``httomo.base_block`` + - Provides ``BaseBlock``, the default implementation of block data access + and CPU/GPU transfer behaviour, including access to angles, darks and + flats. + * - ``httomo.block_interfaces`` + - Defines the protocols that a processable block must satisfy: data and + auxiliary-data access, global indexing, and transfer between CPU and GPU + memory. + * - ``httomo.cli`` + - Implements the ``check``, ``memory-check`` and ``run`` commands, converts + command-line options, prepares the output directory and starts the + appropriate runner. + * - ``httomo.cli_utils`` + - Contains command-line support functions, principally detection of + parameter sweeps in YAML files, JSON strings and parsed configurations. + * - ``httomo.darks_flats`` + - Describes where dark-field and flat-field images are stored and loads or + selects those calibration images for a loader. + * - ``httomo.data`` + - Contains RAM- and HDF5-backed dataset stores, MPI redistribution helpers, + padding operations and the stores used during parameter sweeps. + * - ``httomo.globals`` + - Holds process-wide runtime settings populated by the CLI, such as the + output directory, selected GPU, chunking, compression and reconstruction + filename options. + * - ``httomo.loaders`` + - Creates loader implementations from pipeline configuration. The standard + tomography loader reads projections, angles, darks and flats from + HDF5/NeXus input. + * - ``httomo.logger`` + - Configures concise terminal and ``user.log`` output, detailed + ``debug.log`` output and optional GELF syslog reporting. + * - ``httomo.methods`` + - Provides framework-owned pipeline operations, including global + statistics calculation and block-wise writing of intermediate HDF5 + datasets. + * - ``httomo.method_wrappers`` + - Selects and constructs the adapter that invokes each backend function. + Specialised wrappers handle reconstruction, rotation, image output, + statistics and other non-generic method behaviour. + * - ``httomo.monitors`` + - Constructs summary and benchmark monitors and combines multiple monitors + into one reporting interface. + * - ``httomo.preview`` + - Represents input cropping selections, checks their bounds and calculates + the selected indices and resulting global data shape. + * - ``httomo.runner`` + - Contains the main execution engine and its contracts: pipelines, + sections, block splitting, dataset blocks and stores, loaders, method + wrappers, monitoring, side-output references and GPU utilities. + * - ``httomo.sweep_runner`` + - Parses and executes single-parameter sweeps, divides sweep values among + MPI ranks, manages sweep stages and side outputs, and stores sweep + results. + * - ``httomo.transform_layer`` + - Rewrites an executable pipeline before it runs by inserting framework + operations such as data reduction, data checking, intermediate saving + and sweep image output. + * - ``httomo.transform_loader_params`` + - Converts loader values parsed from YAML or JSON into validated internal + configurations for previews, angles, calibration images and continuous + scan subsets. + * - ``httomo.types`` + - Defines shared type aliases, notably the generic array type used for + either NumPy CPU arrays or CuPy GPU arrays. + * - ``httomo.ui_layer`` + - Converts a user-facing YAML or JSON pipeline into the internal immutable + ``Pipeline``, constructing its loader and method wrappers and resolving + output references. + * - ``httomo.utils`` + - Supplies shared array, timing, logging, error-handling, snapshot and + block-size helpers, as well as NumPy/CuPy backend selection. + * - ``httomo.yaml_checker`` + - Validates pipeline YAML structure, loader position, methods, parameters, + side-output references and, when input data is supplied, referenced HDF5 + paths. + +Generated reference +------------------- + .. autosummary:: :recursive: :toctree: generated diff --git a/docs/source/developers/architecture.rst b/docs/source/developers/architecture.rst new file mode 100644 index 000000000..ba2e16ee9 --- /dev/null +++ b/docs/source/developers/architecture.rst @@ -0,0 +1,55 @@ +.. _developer_architecture: + +Architecture and internals +========================== + +HTTomo separates orchestration from scientific processing. Backend libraries +implement algorithms, ``httomo-backends`` describes how those functions behave +inside a pipeline, and HTTomo validates, schedules and executes them. + +Execution path +-------------- + +#. The CLI loads and validates the pipeline configuration. +#. The UI and transform layers construct method wrappers and insert framework + operations such as intermediate saving. +#. The pipeline is divided into sections with compatible processing patterns. +#. The loader supplies data to the runner, which distributes each section into + per-rank chunks and memory-sized blocks. +#. Wrappers invoke backend methods and pass the resulting block onwards. +#. Dataset stores handle in-memory or disk-backed transitions between sections. + +Read :ref:`execution model ` and :ref:`Core concepts +` first. The implementation-oriented concepts are: + +.. toctree:: + :maxdepth: 1 + + ../introduction/concepts/wrappers + ../introduction/concepts/memory_estimators + +Repository responsibilities +--------------------------- + +``httomo`` + Pipeline validation, loading, orchestration, MPI distribution, data stores, + wrappers, monitoring and command-line behaviour. + +Backend library + The scientific implementation and its numerical tests. + +``httomo-backends`` + Execution metadata, generated method templates, memory estimation, padding + and output-shape information. See :ref:`developers_httomo_backends`. + +Most new scientific methods require changes to the backend library and +``httomo-backends`` only. Change HTTomo when orchestration or wrapper behaviour +must also change. + +Public entry points +------------------- + +The :ref:`api` is exhaustive and includes implementation modules. New code +should depend on the narrowest stable interface available, particularly the +pipeline, method-wrapper, loader, method-repository and monitoring interfaces. +Avoid importing from generated documentation paths or relying on private names. diff --git a/docs/source/developers/development_setup.rst b/docs/source/developers/development_setup.rst new file mode 100644 index 000000000..181f08a8e --- /dev/null +++ b/docs/source/developers/development_setup.rst @@ -0,0 +1,126 @@ +.. _developer_setup: +.. _run_tests: + +Development setup and testing +***************************** + +Create a development checkout +============================= + +Create and activate an environment using :ref:`installation_main`, then clone +HTTomo and enter the repository: + +.. code-block:: console + + $ git clone https://github.com/DiamondLightSource/httomo.git + $ cd httomo + +Install the checkout and development dependencies. Choose the extra that +matches the environment: + +.. code-block:: console + + $ python -m pip install --editable ".[dev-cpu]" + +For a CUDA development environment, use ``.[dev-gpu]`` instead. Confirm that +Python imports the checkout and that parallel HDF5 is available: + +.. code-block:: console + + $ python -c "import httomo; print(httomo.__file__)" + $ python -c "import h5py; print('Parallel HDF5:', h5py.get_config().mpi)" + +The first command should print a path inside the checkout, and the second must +print ``Parallel HDF5: True``. + +Run the test suite +================== + +Run commands in this section from the repository root. To test the same version +as an installed release, check out its corresponding Git tag rather than the +latest ``main`` branch. + +Run the default unit-test selection: + +.. code-block:: console + + $ python -m pytest tests/ + +Tests requiring CuPy, example datasets, performance testing, or full datasets +are skipped by default. + +GPU tests +========= + +On a system with a supported NVIDIA GPU and a working CuPy installation, run +the tests marked as requiring CuPy: + +.. code-block:: console + + $ python -m pytest tests/ --cupy + +The ``--cupy`` option selects only the CuPy tests. Run both this command and +the default test command to exercise both selections. + +Small-data pipeline tests +========================= + +Generate the example pipeline files from the directives installed with +``httomo-backends``: + +.. code-block:: console + + $ mkdir -p docs/source/pipelines_full + $ python docs/source/scripts/execute_pipelines_build.py \ + --output docs/source/pipelines_full/ + +Run the CPU/TomoPy small-data tests with: + +.. code-block:: console + + $ python -m pytest tests/ --small_data + +These tests require TomoPy. Tests for GPU pipelines are skipped. + +On a CUDA-enabled system with CuPy, TomoBAR, and ``httomolibgpu`` installed, +run the GPU small-data tests with both selection options: + +.. code-block:: console + + $ python -m pytest tests/ --small_data --cupy + +Using both options selects tests marked as both ``small_data`` and ``cupy``. + +Run code-quality checks +======================= + +Install the repository's pre-commit hooks, then run them before opening a pull +request: + +.. code-block:: console + + $ pre-commit install + $ pre-commit run --all-files + +Build the documentation +======================= + +Create the documentation environment, install the same ``httomo-backends`` +release pinned by the documentation workflow, generate the pipeline examples +and build with warnings treated as errors: + +.. code-block:: console + + $ micromamba create --file docs/source/doc-conda-requirements.yml + $ micromamba activate httomo-docs + $ python -m pip install --no-deps \ + -r docs/source/doc-pip-requirements.txt + $ python docs/source/scripts/execute_pipelines_build.py \ + --output docs/source/pipelines_full/ + $ sphinx-build -W --keep-going -a -E -b html \ + docs/source docs/build + +Open ``docs/build/index.html`` to inspect the result. Update +``docs/source/doc-pip-requirements.txt`` when the complete pipeline examples +must use a new ``httomo-backends`` release; local and CI builds both consume +that file. diff --git a/docs/source/developers/how_to_contribute.rst b/docs/source/developers/how_to_contribute.rst index 77603d8c5..c513153b9 100644 --- a/docs/source/developers/how_to_contribute.rst +++ b/docs/source/developers/how_to_contribute.rst @@ -3,34 +3,68 @@ How to contribute ***************** -For those who are interested in contributing to HTTomo, we provide here steps to follow. All additional enquires can be left in -the issues section on `HTTomo's Github page `_. +Use this workflow to expose a new processing method through HTTomo. For changes +to HTTomo itself, open a pull request with focused tests and documentation. If +the scope is unclear, discuss it first in the `HTTomo issue tracker +`_. -1. Write a new data processing method in Python. +1. Put the code in the right project +==================================== - One needs to write a method and make it accessible in either a separate library or integrating - it into the list of already :ref:`backends_list`. The latter option is preferred as some of the packages are - maintained by HTTomo developers which will provide support during the integration. +Implement the processing function in a backend library, such as HTTomolib, +HTTomolibGPU or TomoPy. Keep HTTomo responsible for orchestration rather than +scientific algorithms. See :ref:`backends_list` for the supported libraries. -2. Expose the method in the library file in HTTomo +Add tests and make the function importable from its public module before +starting the HTTomo integration. - Then one needs to expose that method to HTTomo by editing the :ref:`pl_library`. You would need to specify - the main static descriptors of that method, such as, :code:`pattern`, :code:`implementation`, etc. If the implementation is :code:`cpu` only, - then :code:`memory_gpu` must be set to :code:`None`. However, if the method requires GPU, then you would need to provide more information so - that HTTomo's framework would account for the memory use on the device. See :code:`HTTomolibgpu` library file for that. - In a simple case, one can calculate the memory directly by providing multipliers in the library file. When memory - calculation is more complicated, one needs to add a Python script that does this calculation. See more in :ref:`developers_memorycalc`. +2. Confirm that HTTomo can call it +================================== -3. Check the wrapper type +Prefer a function that accepts an array and explicit parameters and returns the +processed array. Check that it fits an existing :ref:`wrapper `. +Only add or modify a wrapper when the function's interface cannot use an +existing one. See :ref:`developers_add_own_method` for the basic method +requirements. - Every method is executed by using :ref:`info_wrappers`. Check that the method fits the existing wrapper type, and if not, then possibly more work required - to accommodate it. In most of the cases the method should fit the existing types. +3. Register it in ``httomo-backends`` +===================================== -4. Generate the Yaml template +Add the method's execution metadata, any required memory or shape calculations, +and its generated pipeline template. Follow +:ref:`developers_httomo_backends` for the complete procedure. - HTTomo's UI requires :ref:`reference_templates` to execute the created method. One can either construct that YAML template manually or employ - `YAML generator `_. +Do not add scientific processing code to ``httomo-backends``. That repository +describes how HTTomo should execute the backend function; it does not implement +the function itself. +4. Test the integration +======================= +Before submitting the changes: +* run the backend library's tests; +* test every new ``httomo-backends`` metadata or supporting-function entry; +* review the generated YAML template; +* validate a representative pipeline with + ``python -m httomo check pipeline.yaml input.nxs``; and +* run that pipeline on a small representative dataset. +Include CPU or GPU tests appropriate to the implementation. Memory estimates, +padding and output-shape calculations should cover boundary values, not only a +single typical input. + +5. Submit linked changes +======================== + +Submit changes to the backend library before, or together with, the associated +``httomo-backends`` change. Link the pull requests and identify any minimum +compatible package versions. + +Update HTTomo itself only when the new method requires a framework change, such +as a new wrapper capability. Include the relevant tests and documentation in +that pull request. + +A contribution is complete when the backend method is tested and importable, +its metadata and template are available from ``httomo-backends``, and a small +HTTomo pipeline validates and runs successfully. diff --git a/docs/source/developers/httomo_backends.rst b/docs/source/developers/httomo_backends.rst new file mode 100644 index 000000000..840fdafb5 --- /dev/null +++ b/docs/source/developers/httomo_backends.rst @@ -0,0 +1,312 @@ +.. _developers_httomo_backends: + +Integrating methods with ``httomo-backends`` +********************************************* + +After writing and testing a processing function in a backend library, describe +the function in `httomo-backends +`_ before trying to use +it in an HTTomo pipeline. This page explains that integration step. + +Where ``httomo-backends`` fits +============================== + +The three projects have different responsibilities: + +* a backend library, such as HTTomolib, HTTomolibGPU or TomoPy, implements the + processing function; +* ``httomo-backends`` describes how HTTomo should execute that function and + provides its pipeline template; and +* HTTomo imports the function, divides the data into suitable blocks and calls + it with the parameters supplied by the pipeline. + +Adding a function to a backend library therefore does not automatically make it +available in HTTomo. The corresponding ``httomo-backends`` change is what makes +the function discoverable and safe for HTTomo to schedule. + +``httomo-backends`` supplies two related descriptions of every supported +method: + +Runtime metadata + The method database records the data pattern, implementation type, output + shape behaviour, padding and memory requirements. HTTomo reads this + information while constructing and executing pipeline sections. + +Pipeline template + A generated YAML file exposes the method name, module path and configurable + parameters to pipeline authors. These files form the + :ref:`reference_templates` catalogue. + +For background on this separation, see the `upstream method-information guide +`_. + +Before starting +=============== + +Complete the backend-library work first. The new function should: + +* be importable from its intended public module; +* be included in that module's ``__all__`` list, because the template generator + inspects ``__all__``; +* accept the data array and processing options as explicit function arguments; +* have meaningful defaults for optional arguments; and +* have tests in the backend library. + +Install the backend library and ``httomo-backends`` from their development +checkouts in the same environment. This allows the generator and tests to +inspect the newly added function rather than an older installed release. + +Integration workflow +==================== + +The examples below use a function called ``new_filter`` in +``httomolibgpu.prep.stripe``. Substitute the real package, module and function +names. + +1. Register a new module when necessary +--------------------------------------- + +If the function is in a module that is already listed, skip this step. +Otherwise add its full import path to the backend's module list: + +.. code-block:: text + + httomo_backends/methods_database/packages/backends/ + └── httomolibgpu/ + └── httomolibgpu_modules.yaml + +For example: + +.. code-block:: yaml + + - httomolibgpu.prep.stripe + +The module list tells the template generator which modules to inspect. It does +not contain method metadata. + +2. Add the method to the method database +---------------------------------------- + +Edit the library file for the backend: + +.. code-block:: text + + httomo_backends/methods_database/packages/backends/ + ├── httomolib/httomolib.yaml + ├── httomolibgpu/httomolibgpu.yaml + └── tomopy/tomopy.yaml + +Mirror the Python module hierarchy below the package name. For +``httomolibgpu.prep.stripe.new_filter``, add an entry under ``prep``, then +``stripe``: + +.. code-block:: yaml + + prep: + stripe: + new_filter: + pattern: sinogram + output_dims_change: false + implementation: gpu_cupy + save_result_default: false + padding: false + memory_gpu: + multiplier: 2.0 + method: direct + +Choose each value from the behaviour of the function, not from a similar +method's name. + +.. list-table:: Method metadata + :header-rows: 1 + :widths: 24 76 + + * - Field + - Meaning + * - ``pattern`` + - ``projection`` processes projection slices, ``sinogram`` processes + sinogram slices, and ``all`` can use the current data orientation. + * - ``output_dims_change`` + - Set to ``true`` when the two non-slice dimensions can change. A + supporting function must then calculate the output dimensions. + * - ``implementation`` + - ``cpu`` runs with NumPy arrays, ``gpu`` manages its own GPU transfer, + and ``gpu_cupy`` accepts and returns CuPy arrays. + * - ``save_result_default`` + - Whether HTTomo should save this method's result when the pipeline entry + does not provide ``save_result``. This is normally ``false`` for + intermediate processing and ``true`` for a final reconstruction. + * - ``padding`` + - Set to ``true`` when independently processed blocks need overlapping + slices. A supporting function must calculate the overlap. + * - ``memory_gpu`` + - Use ``None`` for CPU methods. GPU methods use either a direct per-slice + multiplier or a supporting function for a parameter-dependent or + iterative calculation. + +The metadata is operational: an incorrect pattern can introduce the wrong +re-slicing behaviour, while an underestimated memory requirement can cause a +GPU out-of-memory failure. Add a test for every value that affects scheduling +or memory estimation. + +3. Add supporting functions when metadata is not enough +------------------------------------------------------- + +Simple properties belong in the library YAML file. Calculations that depend on +array shape, data type or method parameters belong in Python supporting +functions. Their package and module hierarchy must match the backend function: + +.. code-block:: text + + httomo_backends/methods_database/packages/backends/httomolibgpu/ + └── supporting_funcs/prep/stripe.py + +HTTomo locates these functions by name. Implement the functions required by +the metadata: + +.. list-table:: Supporting-function names + :header-rows: 1 + :widths: 36 64 + + * - Situation + - Required name + * - Parameter-dependent GPU memory + - ``_calc_memory_bytes_`` + * - Iterative GPU memory search + - ``_calc_memory_bytes_for_slices_`` + * - ``output_dims_change: true`` + - ``_calc_output_dim_`` + * - ``padding: true`` + - ``_calc_padding_`` + +For example, a method whose output width is controlled by ``new_width`` could +provide: + +.. code-block:: python + + def _calc_output_dim_new_filter(non_slice_dims_shape, **kwargs): + return non_slice_dims_shape[0], kwargs["new_width"] + +Follow the signatures and return types of an existing supporting function with +the same purpose. For memory calculations, test the estimate against measured +peak allocation over representative shapes and parameter values. See +:ref:`developers_memorycalc` for the HTTomo memory model. + +4. Generate the pipeline template +--------------------------------- + +Run the template generator from the root of the ``httomo-backends`` checkout. +For HTTomolibGPU, use: + +.. code-block:: console + + $ python httomo_backends/scripts/yaml_templates_generator.py \ + -i httomo_backends/methods_database/packages/backends/httomolibgpu/httomolibgpu_modules.yaml \ + -o httomo_backends/yaml_templates/httomolibgpu + +Use the equivalent ``httomolib`` or ``tomopy`` paths for another backend. The +backend package must be installed in the active environment. + +The expected file for the example is: + +.. code-block:: text + + httomo_backends/yaml_templates/httomolibgpu/ + └── httomolibgpu.prep.stripe/new_filter.yaml + +The generator obtains parameters and defaults from the Python signature. Review +the result rather than assuming it is ready: + +* the ``method`` and ``module_path`` must identify the new function; +* the input data argument must not appear under ``parameters``; +* optional parameters should retain useful defaults; +* mandatory user parameters should be marked ``REQUIRED``; and +* any side outputs and references must follow the HTTomo pipeline format. + +If the generator handles a signature incorrectly, update the generator rules +or the function interface as appropriate, then regenerate. Avoid maintaining a +manual template that will be overwritten during a later documentation or +release build. See the `template-generator documentation +`_ +for more detail. + +5. Test the integration +----------------------- + +At minimum, query every new metadata property in a unit test. The following +example catches misspelled module paths and incorrectly nested library entries: + +.. code-block:: python + + from httomo_backends.methods_database.query import ( + MethodsDatabaseQuery, + Pattern, + ) + + + def test_new_filter_metadata(): + query = MethodsDatabaseQuery( + "httomolibgpu.prep.stripe", "new_filter" + ) + assert query.get_pattern() == Pattern.sinogram + assert query.get_implementation() == "gpu_cupy" + assert query.get_output_dims_change() is False + +Run the method-database tests and the relevant backend tests: + +.. code-block:: console + + $ pytest tests/test_method_query.py + $ pytest tests/test_httomolibgpu.py + +GPU memory tests require a suitable GPU environment. Test any new supporting +function directly, including boundary shapes and parameters that maximise its +memory or padding requirement. + +Finally, install the modified ``httomo-backends`` checkout alongside HTTomo, +put the generated method entry into a small pipeline, and validate it: + +.. code-block:: console + + $ python -m httomo check pipeline.yaml input.nxs + +Run the pipeline on a small representative dataset as an end-to-end check. For +a GPU method, include enough variation in shape and parameters to exercise its +memory estimate. + +6. Submit and release in dependency order +----------------------------------------- + +The backend function must be available before the ``httomo-backends`` change +that imports and describes it can be released. Keep the two changes linked in +their pull-request descriptions and state the minimum compatible backend +version when needed. + +A method is ready for HTTomo users when all of the following are true: + +* the backend function is released and importable; +* its runtime metadata and any supporting functions are tested; +* its generated YAML template is present and correct; +* a representative pipeline passes ``httomo check``; and +* the compatible ``httomo-backends`` version is available to HTTomo. + +Common mistakes +=============== + +Method is present in the library but has no template + Confirm that its module is in ``_modules.yaml`` and that the + function is exported through the module's ``__all__`` list, then regenerate + the templates. + +``KeyError`` when HTTomo constructs the method + Check that the hierarchy in ``.yaml`` exactly matches the full + module path and method name, including capitalisation. + +Supporting function cannot be imported + Check both the directory hierarchy and the required function-name prefix. + They are derived from the backend module path and method name. + +Pipeline validates but fails on larger data + Revisit the GPU memory estimate, padding and output-dimension calculation. + Template validation checks the interface; it cannot prove that runtime + resource estimates are conservative. diff --git a/docs/source/developers/memory_calculation.rst b/docs/source/developers/memory_calculation.rst index 85650c0b2..3124072fa 100644 --- a/docs/source/developers/memory_calculation.rst +++ b/docs/source/developers/memory_calculation.rst @@ -1,83 +1,313 @@ .. _developers_memorycalc: -GPU memory calculations -*********************** - -The ``calc_max_slices`` function must have the following signature:: - - def calc_max_slices(slice_dim: int, - other_dims: Tuple[int, int], - dtype: np.dtype, - available_memory: int, - **kwargs) -> Tuple[int, np.dtype]: ... - -The ``httomo`` package will call this function, passing in the dimension along which it will slice -(``0`` for projection, ``1`` for sinogram), the other dimensions of the data array shape, -the data type for the input, and the available memory on the GPU for method execution. -Additionally it passes all other parameters of the method in the ``kwargs`` argument, -which can be used by the function in case parameters determine the memory consumption. -The function should calculate how many slices along the slicing dimension it can fit into the given memory. -Further, it returns the output datatype of the method (given the input ``dtype`` argument), -which ``httomo`` will use for calling subsequent functions. - -**Example (All Patterns):** - -The ``httomo`` package will have to determine the number of slices along the projection dimension -it can fit, given the other two dimension sizes. For example: - -* ``data.shape`` is ``[x, 10, 20]``, and ``httomo`` needs to determine the value for ``x`` -* The data type for ``data`` is ``float32`` (which means 4 bytes are needed per element) -* Assume the available free memory on the GPU is 14,450 bytes -* It will call the function as:: - - max_slices, outdtype = my_method.meta.calc_max_slices(0, (10, 20), np.float32(), 14450, **method_args) - -* The developer of the given method needs to provide this function implementation, - and it needs to calculate the maximum number of slices it can fit. -* Assuming that the method is very simple and does not need any local temporary memory, - requiring only space for the input and output array, it could be implemented as follows:: - - def _my_calc_max_slices(slice_dim: int, - other_dims: Tuple[int, int], - dtype: np.dtype, - available_memory: int, - **kwargs) -> int: - input_mem_per_slice = other_dims[0] * other_dims[1] * dtype.nbytes - output_mem_per_slice = input_mem_per_slice - max_slices = available_memory // (input_mem_per_slice + output_mem_per_slice) - return max_slices, dtype - - (note that `//` denotes integer division, which rounds towards zero) -* If the method needs extra memory internally, this should be taken into account in the implementation. - There are several methods in this respository which can serve as an example. - -* With the example data given above, the function would determine that 9 slices can fit: - - * call ``calc_max_slices(0, (10, 20), np.float32(), 14450, **method_args)`` - * ``input_mem_per_slice = 800`` - * ``output_mem_per_slice = 800`` - * => ``max_slices = 14450 // 1600 = 9`` - - -Max Slices Tests +Memory estimation and block sizing +********************************** + +HTTomo processes a section in blocks. For a section containing GPU methods, +the block length is limited by the peak GPU memory required by every method in +that section. A correct estimate must be conservative enough to prevent an +out-of-memory failure without making blocks unnecessarily small. + +Memory-estimation metadata and helper functions belong in +``httomo-backends``, not in the processing library or HTTomo itself. Read +:ref:`developers_httomo_backends` before adding an estimator. + +.. important:: + + The former ``calc_max_slices`` extension API is no longer used. Do not add a + ``calc_max_slices`` function to a processing method. Estimators now report + memory in bytes through ``httomo-backends``; HTTomo calculates the block + length from those values. Estimators also no longer return an output data + type. + +How HTTomo selects a block length +================================= + +Before executing a section, HTTomo: + +#. clears the CuPy memory pool and FFT plan cache; +#. queries the available memory on the selected GPU; +#. reserves a 10% safety margin; +#. applies ``--max-memory`` as an additional per-process upper limit when that + option is non-zero; +#. asks each GPU method with ``memory_gpu`` metadata how many slices fit; and +#. uses the smallest result for the whole section, capped by the number of + slices in the process's chunk. + +The non-slice dimensions are passed through the methods in execution order. If +a method changes the output dimensions, the next estimator receives those +updated dimensions. This is why an output-dimension helper is as important as +the memory estimator for a size-changing method. + +CPU methods do not use this GPU estimator. A CPU-only section uses the +``--max-cpu-slices`` limit instead. In a mixed section, the registered GPU +methods determine the memory-based limit. + +If a section uses overlap padding, at least one core slice plus the leading and +trailing padding must fit. HTTomo stops before execution if the calculated +block is smaller than that minimum. + +Where the configuration lives +============================== + +Every supported method has an entry in its backend library file, for example: + +.. code-block:: text + + httomo_backends/methods_database/packages/backends/ + └── httomolibgpu/httomolibgpu.yaml + +GPU methods define ``memory_gpu``. CPU methods use ``memory_gpu: None``. + +.. code-block:: yaml + + prep: + normalize: + minus_log: + pattern: all + output_dims_change: false + implementation: gpu_cupy + save_result_default: false + padding: false + memory_gpu: + multiplier: 3.0 + method: direct + +Do not leave ``memory_gpu`` unset for a GPU method. HTTomo does not apply a +method-specific block-size constraint when this metadata is absent. + +Choose an estimation strategy +============================== + +``memory_gpu.method`` selects one of three strategies. + +.. list-table:: GPU memory-estimation strategies + :header-rows: 1 + :widths: 18 34 48 + + * - Strategy + - Use it when + - Estimator contract + * - ``direct`` + - Peak memory is proportional to the number of elements in one input + slice. + - Store a conservative multiplier directly in the library YAML file. + * - ``module`` + - Memory is still linear in the number of slices, but the per-slice or + fixed cost depends on shape, data type or method parameters. + - Implement ``_calc_memory_bytes_`` in the matching supporting + module. + * - ``iterative`` + - Total memory is non-linear in the number of slices, or a backend already + provides a whole-block peak-memory calculation. + - Implement ``_calc_memory_bytes_for_slices_`` in the matching + supporting module. + +The supporting-module hierarchy must mirror the processing function's module +path. For example, helpers for ``httomolibgpu.prep.phase.paganin_filter`` live +in: + +.. code-block:: text + + httomo_backends/methods_database/packages/backends/httomolibgpu/ + └── supporting_funcs/prep/phase.py + +Direct estimates +================ + +Use ``direct`` when a single multiplier accurately describes the complete peak +allocation per input slice: + +.. code-block:: yaml + + memory_gpu: + multiplier: 3.0 + method: direct + +HTTomo calculates: + +.. code-block:: python + + bytes_per_slice = ( + multiplier + * non_slice_dims_shape[0] + * non_slice_dims_shape[1] + * dtype.itemsize + ) + max_slices = available_memory // bytes_per_slice + +The multiplier represents the full peak allocation, not only temporary +workspace. Include the input, output and all arrays that coexist at the peak. +For example, an in-place operation with no additional allocation may use a +multiplier near ``1.0``; a method holding the input, output and another +full-sized temporary array needs at least ``3.0``. + +Use a supporting function instead if the ratio changes with dimensions, +parameters or fixed-size allocations. + +Module estimates +================ + +Use ``module`` for a memory model consisting of a per-slice cost and a fixed +cost: + +.. code-block:: yaml + + memory_gpu: + multiplier: None + method: module + +Implement this function in the matching supporting module: + +.. code-block:: python + + def _calc_memory_bytes_new_filter( + non_slice_dims_shape: tuple[int, int], + dtype: np.dtype, + **kwargs, + ) -> tuple[int, int]: + """Return (bytes_per_slice, fixed_bytes).""" + input_bytes = np.prod(non_slice_dims_shape) * dtype.itemsize + output_bytes = input_bytes + fixed_bytes = 8 * 1024**2 + return int(input_bytes + output_bytes), fixed_bytes + +HTTomo then calculates: + +.. code-block:: python + + max_slices = ( + available_memory - fixed_bytes + ) // bytes_per_slice + +``non_slice_dims_shape`` contains the two dimensions that do not form the +current block length. ``kwargs`` contains the configured method parameters, +with side-output references resolved to their runtime values. + +``bytes_per_slice`` must include allocations that grow linearly with the block +length. ``fixed_bytes`` covers allocations that occur once per method call, +such as a filter or other workspace independent of the number of slices. Both +values describe the peak, so do not add allocations that cannot coexist. + +Iterative estimates +=================== + +Use ``iterative`` when a single per-slice value is not valid: + +.. code-block:: yaml + + memory_gpu: + multiplier: None + method: iterative + +Implement a whole-block estimator: + +.. code-block:: python + + def _calc_memory_bytes_for_slices_new_filter( + dims_shape: tuple[int, int, int], + dtype: np.dtype, + **kwargs, + ) -> int: + """Return peak bytes for the complete candidate block.""" + return calculate_peak_bytes(dims_shape, dtype=dtype, **kwargs) + +HTTomo inserts a candidate block length into ``dims_shape`` at the section's +slicing dimension and calls the estimator repeatedly. It first makes a linear +approximation and then searches for a safe value. The search can stop once an +estimate uses at least 90% of the available memory, so the result is a safe +approximation rather than necessarily the exact largest possible block. + +An iterative estimator must therefore: + +* return the total peak bytes for the complete candidate block; +* be monotonically non-decreasing as the candidate slice count increases; +* be deterministic and safe to call repeatedly; and +* account for the supplied dimensions, data type and relevant parameters. + +If the estimator raises an exception for a candidate size, HTTomo treats that +candidate as too large. Do not use exceptions as the normal calculation path. + +Data types and allocations +========================== + +During section sizing, HTTomo currently passes ``float32`` to GPU memory +estimators because input data is converted to floating point after loading. +The old estimator API propagated an output dtype; the current one does not. + +If a method creates arrays with another dtype, include their actual byte sizes +inside its multiplier or supporting function. Use ``np.dtype(...).itemsize`` +rather than assuming four bytes per element. + +Account for every allocation that can be live at the peak, including where +applicable: + +* the input and output arrays; +* dtype conversions and contiguous copies; +* padded arrays and overlap regions; +* temporary workspaces; +* FFT plans and FFT outputs; +* reconstruction buffers; and +* parameter-dependent fixed arrays. + +Follow the allocation lifetime through the backend implementation. Summing all +allocations made during the function can substantially overestimate the peak +when earlier arrays are released before later ones are created. + +Testing an estimator +==================== + +Test the metadata and the estimate separately. + +Metadata test +------------- + +Query the method through ``MethodsDatabaseQuery`` so that an incorrect YAML +hierarchy, method name or strategy is detected: + +.. code-block:: python + + from httomo_backends.methods_database.query import MethodsDatabaseQuery + + + def test_new_filter_memory_metadata(): + query = MethodsDatabaseQuery( + "httomolibgpu.prep.stripe", "new_filter" + ) + requirement = query.get_memory_gpu_params() + assert requirement is not None + assert requirement.method == "module" + assert requirement.multiplier is None + +Peak-memory test ---------------- -In order to test that the slice calculation function is reflecting reality, each method has a -unit test implemented that verifies that the calculation is right (within bounds). -That is, it tests that the estimated slices are between 80% and 100% of the actually used slices. -These tests also help to keep the memory estimation functions in sync with the implementation. +Measure the processing function's peak GPU allocation with representative +inputs, then compare it with the estimate. For a module estimator: + +.. code-block:: python -The strategy for testing is the other way around: + bytes_per_slice, fixed_bytes = _calc_memory_bytes_new_filter( + data.shape[1:], data.dtype, **parameters + ) + estimated_peak = data.shape[0] * bytes_per_slice + fixed_bytes -* We first run the actual method, given a specific data set, and record the maximum memory actually - used by the method. -* Then, retrospectively, we call the ``calc_max_slices`` estimator function and pass in this memory - as the ``available_memory`` argument. So we're asking the estimation function to assume that - the memory available is the actually used memory in the method call. -* The estimated number of slices should then be less or equal to the actual slices used earlier. -* To make sure the function is not too conservative, we're checking that it returns at least 80% - of the slices that actually fit +The estimate must not be below the measured peak. Also keep it close enough to +the measurement to avoid unnecessarily small blocks. Existing estimators often +use a tolerance around 20%, but the appropriate tolerance depends on the +method and must be stated explicitly in its test. +Cover more than one block length and include the dimensions, dtypes and +parameters that change memory use. Test boundary cases such as padding, +optional workspaces and the largest supported reconstruction shape. +For an iterative estimator, additionally verify that: +* the estimate is monotonic over the tested slice counts; +* the block length returned by HTTomo fits in the available memory; and +* either the next slice count does not fit or the selected block already uses + at least 90% of the available memory. +Run the relevant ``httomo-backends`` unit and GPU tests, followed by a small +end-to-end HTTomo pipeline. A template or metadata test alone cannot detect an +underestimated runtime allocation. diff --git a/docs/source/developers/profiling_tracing.rst b/docs/source/developers/profiling_tracing.rst new file mode 100644 index 000000000..98f69eb46 --- /dev/null +++ b/docs/source/developers/profiling_tracing.rst @@ -0,0 +1,52 @@ +.. _profiling_tracing: + +Profiling and tracing +********************* + +`VizTracer`_ can record CPU traces and CPU and memory usage statistics for +investigating HTTomo performance. + +Record a serial run +=================== + +.. code-block:: console + + $ python -m viztracer \ + --plugin "vizplugins --cpu_usage --memory_usage" \ + --output_file output.json \ + -m httomo run --no-standalone \ + data.nxs pipeline.yaml output_directory + +Record an MPI run +================= + +Create a separate trace for each MPI rank: + +.. code-block:: console + + $ mpirun -n 4 bash -c \ + 'python -m viztracer \ + --plugin "vizplugins --cpu_usage --memory_usage" \ + --output_file output_rank_${OMPI_COMM_WORLD_RANK}.json \ + -m httomo run --no-standalone \ + data.nxs pipeline.yaml output_directory' + +Combine and inspect traces +========================== + +Combine the rank-specific traces: + +.. code-block:: console + + $ viztracer --combine output_rank_*.json -o output.json + +View the result locally: + +.. code-block:: console + + $ vizviewer output.json + +VizTracer output can also be opened using the online `Perfetto`_ viewer. + +.. _VizTracer: https://viztracer.readthedocs.io/ +.. _Perfetto: https://ui.perfetto.dev/ diff --git a/docs/source/doc-conda-requirements.yml b/docs/source/doc-conda-requirements.yml index 3c1ac82a4..df7779f89 100644 --- a/docs/source/doc-conda-requirements.yml +++ b/docs/source/doc-conda-requirements.yml @@ -1,21 +1,21 @@ name: httomo-docs channels: - - conda-forge + - conda-forge dependencies: - - python>=3.10,<3.13 - - numpy - - sphinx>=8.0,<8.2.0 - - sphinx-book-theme + - python=3.12 + - numpy=2.4 + - sphinx>=8,<10 + - sphinx-book-theme>=1.2,<2 - pandoc - - jinja2 - - nbsphinx - - sphinx-design - - sphinx-copybutton + - jinja2>=3.1,<4 + - nbsphinx>=0.9,<1 + - sphinx-design>=0.6,<1 + - sphinx-copybutton>=0.5,<1 - pyyaml - ipython - ghp-import - loguru - graypy - tqdm - - ruamel.yaml>=0.18 + - ruamel.yaml>=0.18,<1 - pip diff --git a/docs/source/doc-pip-requirements.txt b/docs/source/doc-pip-requirements.txt new file mode 100644 index 000000000..83b20851b --- /dev/null +++ b/docs/source/doc-pip-requirements.txt @@ -0,0 +1,2 @@ +# Backend metadata used to generate the complete example pipelines. +httomo-backends==1.2.0 diff --git a/docs/source/explanation/faq.rst b/docs/source/explanation/faq.rst deleted file mode 100644 index 9ae89ba9a..000000000 --- a/docs/source/explanation/faq.rst +++ /dev/null @@ -1,86 +0,0 @@ - -.. raw:: html - - - -.. role:: blue - -Frequently Asked Questions ---------------------------- - -.. dropdown:: How can I use a template? - - Template can be used to build a process list or a pipeline of methods. Normally the content of a template (a YAML file) is copied to build a process list file. Please see :ref:`howto_process_list`. - -.. dropdown:: Can I create a template? - - You can create a template manually if you want to run a method from the external software. See more on :ref:`backends_list`. - -.. dropdown:: How can I configure a multi-task pipeline? - - The multi-task pipeline is build from the available :ref:`reference_templates` by stacking them together. Please see :ref:`howto_process_list`. - -.. dropdown:: How can I run HTTomo? - - Please see :ref:`howto_run`. - -.. dropdown:: I have a Python method, can it be used with HTTomo? - - There is a high chance that it can be used. The method needs to be accessible in your Python environment and you will need a YAML template for it. See more on what kind of :ref:`backends_list` can be used with HTTomo. It is also recommended if you integrate your method in a library first. See :ref:`developers_content`. - -.. dropdown:: How can I contribute to HTTomo? - - You can contribute by adding new methods to :ref:`backends_list` or by contributing to the source base of the `HTTomo project `_. - -.. _faq_workstation: - -Working from a workstation at Diamond Light Source -************************************************** - -.. _`terminal`: - -.. dropdown:: What is a terminal? - - A terminal could also be referred to as a console, shell, command - prompt or command line. - - It is a program on your computer which can take in text based - instructions and complete them. For example, navigating to a particular file - or directory. It can also perform more complex tasks relating to - software installation. - - It doesn't have a graphical interface, and it allows access to a wide - range of commands quickly. - -.. dropdown:: How can I use HTTomo inside the terminal? - - 1. Using the module system, ``module load`` allows you to obtain an access to the installed HTTomo at Diamond computing systems. - You can check which versions of HTTomo are installed with the command: :code:`module avail httomo`. You can either load a specific version - with ``module load httomo/*httomo_version*`` or the default (recommended) version by executing: - - .. code-block:: console - - $ module load httomo - - This will add all of the related packages and files into your path, meaning - that you will have an access to these packages from your loaded Python environment. - - 2. Configure your pipeline using the templates as shown previously - and run HTTomo. - - -.. dropdown:: What is ``module load`` doing? - - It is modifying the users environment, by including the path to certain - environment modules. In case of HTTomo it enables a specific conda environment with Python. - - You can read more about how module works at `modules.readthedocs.io `_ - -.. dropdown:: What do I do if I have module loaded the wrong version of HTTomo? - - You can use repeat the module command, replacing ``load`` with ``unload`` - - .. code-block:: console - - $ module unload httomo/*httomo_old_version* # unload old version first - $ module load httomo/*httomo_version* # load the correct one diff --git a/docs/source/explanation/process_list.rst b/docs/source/explanation/process_list.rst deleted file mode 100644 index e2ef78604..000000000 --- a/docs/source/explanation/process_list.rst +++ /dev/null @@ -1,9 +0,0 @@ -.. _explanation_process_list: - -What is a process list? ------------------------- - -A process list is a YAML file (see :ref:`explanation_yaml`), which is required to execute the processing of data in HTTomo. The process list file consists of methods exposed as YAML templates stacked together -to form a **serially processed sequence of methods**. Each YAML template represents a standalone method or a loader which can be chained together with other templates to form a process list. - -Please check how :ref:`howto_process_list`. diff --git a/docs/source/explanation/templates.rst b/docs/source/explanation/templates.rst deleted file mode 100644 index 2db04ef2c..000000000 --- a/docs/source/explanation/templates.rst +++ /dev/null @@ -1,31 +0,0 @@ -.. _explanation_templates: - -What is a template? ------------------------- - -A YAML template (see :ref:`explanation_yaml`) is a textual interface to a method, which can be executed in HTTomo. -The template provides a communication with a chosen method by setting its input/output entries and also additional parameters, if required. - -The combination of YAML templates results in a processing list, also called a pipeline. See more on :ref:`explanation_process_list` - -As a simple template example, let's consider the template for the median filter from the `TomoPy `_ package. - -.. code-block:: yaml - - - method: median_filter3d - module_path: tomopy.misc.corr - parameters: - size: 3 - -The first two lines are self-explanatory: the method's name and path to the module. HTTomo will interpret this in Python -by importing the method as: - -.. code-block:: python - - from tomopy.misc.corr import median_filter3d - -The set of parameter values for that method is given in the *parameters* field. - -.. note:: Please note that in the original TomoPy's method, there is also the :code:`arr` parameter. This parameter is not exposed in the template because HTTomo will deal with all I/O aspects behind the scenes by using special wrappers. More on that in :ref:`detailed_about`. `YAML generator `_ will automatically generate the ready-to-be-used templates. - -Please see the list of all supported :ref:`reference_templates`. diff --git a/docs/source/faq/faq.rst b/docs/source/faq/faq.rst new file mode 100644 index 000000000..e4722ded9 --- /dev/null +++ b/docs/source/faq/faq.rst @@ -0,0 +1,120 @@ +.. _faq: + +Frequently asked questions +========================== + +YAML and pipelines +------------------ + +.. dropdown:: What is YAML and how does HTTomo use it? + + YAML is a human-readable format for structured configuration files. HTTomo + uses YAML to describe a :ref:`pipeline + ` containing the loader and processing + methods. + + See the :ref:`pipeline_file_reference` for YAML formatting, supported method + fields, side-output references and parameter-sweep syntax. + +.. dropdown:: What is a YAML template? + + A YAML template configures one loader or processing method. It identifies the + method, its Python module and its configurable parameters. + + Templates can be copied from :ref:`reference_templates` and adapted for a + particular dataset or processing task. + +.. dropdown:: What is a pipeline? + + A pipeline is a YAML file containing an ordered sequence of method + templates. Older documentation may call it a *process list*. + + The first entry must be a loader. Subsequent methods are executed in order + from top to bottom, with each method receiving the data produced by the + preceding operation. + + See :ref:`explanation_pipelines_templates` for an introduction to pipelines + and templates. + +.. dropdown:: How do I build a pipeline? + + Start with a compatible loader template, then add processing templates in + execution order. Edit the parameters for your data and processing + requirements. + + The recommended workflow is: + + #. Select methods from :ref:`reference_templates`. + #. Copy their templates into one YAML file. + #. Place the loader first. + #. Configure the method parameters. + #. Validate the completed pipeline. + #. Run the pipeline. + + See :ref:`howto_process_list` for detailed instructions and + :ref:`tutorials_pl_templates` for complete examples. + +.. dropdown:: How do I validate a pipeline? + + Use the HTTomo YAML checker before running the pipeline: + + .. code-block:: console + + python -m httomo check pipeline.yaml + + You can optionally provide the input data file so that HTTomo also validates + its dataset paths: + + .. code-block:: console + + python -m httomo check pipeline.yaml input.nxs + + The checker detects malformed YAML, unknown methods, invalid parameters and + incompatible dataset paths. See :ref:`utilities_yamlchecker` for details. + +.. dropdown:: Can I create a method template? + + Yes. A template can be written manually or generated from a supported backend + method. It must provide the correct ``method``, ``module_path`` and + ``parameters`` fields. + + For reusable integration, add the method and its metadata to + ``httomo-backends``. See :ref:`developers_add_own_method`. + +Using and extending HTTomo +-------------------------- + +.. dropdown:: How do I run HTTomo? + + First install or load HTTomo, prepare a validated pipeline and select the + input data. + + See :ref:`How to run HTTomo ` or + :ref:`howto_run_at_diamond` when working at Diamond Light Source. + +.. dropdown:: Can HTTomo run my own Python method? + + Usually, provided that: + + - the method is importable in the HTTomo environment + - its execution properties are described in ``httomo-backends`` + - a YAML template is available + - its inputs and outputs are compatible with an existing + :ref:`wrapper ` + + A new wrapper may be required for methods with unusual inputs or outputs. See + :ref:`developers_add_own_method` for the integration procedure. + +.. dropdown:: How can I contribute? + + Contributions can add backend methods, improve documentation or modify the + `HTTomo source code + `_. + + See :ref:`developers_howtocontribute` for development guidance. + +.. dropdown:: Where should I start when a run fails? + + Read ``user.log`` and then ``debug.log`` in the run directory. Validate the + pipeline against the input file with ``python -m httomo check`` and consult + :ref:`troubleshooting` for data, MPI, HDF5, CUDA and memory problems. diff --git a/docs/source/getting_started/at_diamond.rst b/docs/source/getting_started/at_diamond.rst new file mode 100644 index 000000000..7b888d2d0 --- /dev/null +++ b/docs/source/getting_started/at_diamond.rst @@ -0,0 +1,99 @@ +.. _howto_run_at_diamond: + +Running HTTomo at Diamond +========================= + +HTTomo can be run at Diamond in two ways: + +* in parallel on the ``wilson`` compute cluster, using the ``httomo_mpi`` + launcher; or +* serially on a Diamond workstation, using the ``httomo run`` command. + +Parallel execution on the compute cluster is the recommended and most common +way to process tomography data at Diamond. + +Commands are entered in a terminal, also called a shell or command line. Before +running HTTomo, load its software module: + +.. code-block:: console + + $ module load httomo + +The module configures executable and library paths for the current terminal. +List the installed versions with ``module avail httomo``. To select a specific +version, unload the current one and load the required version: + +.. code-block:: console + + $ module unload httomo + $ module load httomo/ + +Running HTTomo in parallel +++++++++++++++++++++++++++ + +Parallel HTTomo jobs run on the ``wilson`` production compute cluster. The +``httomo_mpi`` launcher is integrated with the SLURM workload manager and +submits the requested processing job to the cluster. + +Submitting from a Diamond workstation +##################################### + +On a Diamond workstation, load the HTTomo environment if it is not already +loaded: + +.. code-block:: console + + $ module load httomo + +Then submit the processing job: + +.. code-block:: console + + $ httomo_mpi IN_FILE YAML_CONFIG OUT_DIR + +Alternatively, log in to ``wilson``, load the HTTomo module and submit the job +from there: + +.. code-block:: console + + $ ssh wilson + $ module load httomo + $ httomo_mpi IN_FILE YAML_CONFIG OUT_DIR + +The command takes the following arguments: + +``IN_FILE`` + The path to the HDF5 file containing the input tomography data. + +``YAML_CONFIG`` + The path to the YAML file that defines the processing pipeline. + +``OUT_DIR`` + The directory in which HTTomo will write its output. + +To see the available launcher options, run: + +.. code-block:: console + + $ httomo_mpi --help + +Running HTTomo serially on workstation +++++++++++++++++++++++++++++++++++++++ + +For smaller jobs or testing pipelines, HTTomo can be run serially on a Diamond +workstation. + +First, load the HTTomo environment: + +.. code-block:: console + + $ module load httomo + +Then run the pipeline locally: + +.. code-block:: console + + $ httomo run IN_FILE YAML_CONFIG OUT_DIR + +This command runs HTTomo on the workstation itself and does not submit a job to +the compute cluster. diff --git a/docs/source/getting_started/quickstart.rst b/docs/source/getting_started/quickstart.rst new file mode 100644 index 000000000..9694a6f1d --- /dev/null +++ b/docs/source/getting_started/quickstart.rst @@ -0,0 +1,73 @@ +.. _quickstart: + +10-minute quickstart +==================== + +This example downloads HTTomo's small standard test dataset, validates a CPU +pipeline and reconstructs the data with TomoPy. It does not require a GPU. + +Prerequisites +------------- + +Install HTTomo by following :ref:`installation_cpu_only`. The example pipeline uses +TomoPy and HTTomolib. + + +Download the data and pipeline +------------------------------ + +Create an empty working directory and download the test dataset and CPU +pipeline from the current HTTomo version: + +.. code-block:: console + + $ mkdir httomo-quickstart + $ cd httomo-quickstart + $ curl -L -O \ + https://raw.githubusercontent.com/DiamondLightSource/httomo/main/tests/test_data/tomo_standard.nxs + $ curl -L -O \ + https://raw.githubusercontent.com/DiamondLightSource/httomo/main/docs/source/pipelines_full/tomopy_gridrec.yaml + +The `tomo_standard.nxs test dataset +`_ +is approximately 9 MB. It contains 180 projections plus flat and dark fields, +with detector frames of 128 by 160 pixels. ``tomopy_gridrec.yaml`` performs +normalisation, automatic centre finding, gridrec reconstruction, rescaling and +TIFF output. + +Validate the pipeline +--------------------- + +Check both the pipeline syntax and the dataset paths before processing: + +.. code-block:: console + + $ python -m httomo check tomopy_gridrec.yaml tomo_standard.nxs + +A successful check finishes without validation errors. + +Run the reconstruction +---------------------- + +Create the parent output directory, then run HTTomo: + +.. code-block:: console + + $ mkdir output + $ python -m httomo run \ + tomo_standard.nxs tomopy_gridrec.yaml output + +HTTomo creates a timestamped directory below ``output``. The run reconstructs +128 slices of 160 by 160 pixels and writes them to the ``images8bit_tif`` +subdirectory as TIFF files. The run directory also contains an intermediate HDF5 +reconstruction, ``tomopy_gridrec.yaml``, ``user.log`` and ``debug.log``. + +Open one of the TIFF files to confirm that the reconstruction completed. The +centre reported in ``user.log`` should be close to 79.5 pixels. + +Next steps +---------- + +* Use :ref:`howto_run` with your own input data. +* Consult :ref:`troubleshooting` if validation or execution fails. +* Check :ref:`versioned_downloads` when using another HTTomo release. diff --git a/docs/source/howto/how_to_run/at_diamond.rst b/docs/source/howto/how_to_run/at_diamond.rst index d1eaeb1a3..09dbcddf3 100644 --- a/docs/source/howto/how_to_run/at_diamond.rst +++ b/docs/source/howto/how_to_run/at_diamond.rst @@ -1,52 +1,10 @@ -.. _howto_run_at_diamond: +:orphan: -Inside Diamond -++++++++++++++ +Diamond guide moved +=================== -If you're not familiar what the module system is, please check :ref:`faq_workstation` guide. +The Diamond-specific guide is now at :ref:`howto_run_at_diamond`. -Cluster -####### +.. raw:: html -This will be the most common way to use HTTomo at Diamond, it submits a job to the -production compute cluster at Diamond that will run HTTomo. - -In a terminal, the commands to log onto the compute cluster and submit an HTTomo -job are the following: - -.. code-block:: console - - $ ssh wilson - $ module load httomo - $ httomo_mpi IN_FILE YAML_CONFIG OUT_DIR - -Workstation -########### - -Serial -~~~~~~ - -HTTomo can be loaded on a Diamond workstation by doing :code:`module load httomo`. -This will allow HTTomo to be run on the local machine like so: - -.. code-block:: console - - $ httomo run IN_FILE YAML_CONFIG OUT_DIR - -Parallel -~~~~~~~~ - -Parallel execution of HTTomo at Diamond is typically performed on a compute cluster. -The HTTomo launcher is integrated with the SLURM workload manager via REST APIs and is -configured to submit jobs to the :code:`wilson` compute cluster. - -To run HTTomo, either log in to a Wilson compute node directly or load the HTTomo environment on a workstation -with :code:`module load httomo`. - -Once the environment is loaded, HTTomo jobs can be submitted and executed in parallel as described below. - -.. code-block:: console - - $ httomo_mpi IN_FILE YAML_CONFIG OUT_DIR -` -Also look for help with :code:`httomo_mpi --help`. \ No newline at end of file + diff --git a/docs/source/howto/how_to_run/outside_diamond.rst b/docs/source/howto/how_to_run/outside_diamond.rst deleted file mode 100644 index b48a1b43e..000000000 --- a/docs/source/howto/how_to_run/outside_diamond.rst +++ /dev/null @@ -1,30 +0,0 @@ -.. _howto_run_outside_diamond: - -Outside Diamond -+++++++++++++++ - -Make sure to activate the conda environment that has HTTomo installed in it. - -Serial -###### - -This is the simplest case: - -.. code-block:: console - - $ python -m httomo run IN_FILE YAML_CONFIG OUT_DIR - -Parallel -######## - -HTTomo's parallel processing capability has been implemented with :code:`mpi4py` -and :code:`h5py`. Therefore, HTTomo is intended to be run in parallel by using -the :code:`mpirun` command (or equivalent, such as :code:`srun` for SLURM -clusters): - -.. code-block:: console - - $ mpirun -np N python -m httomo run IN_FILE YAML_CONFIG OUT_DIR - -where :code:`N` is the number of parallel processes to launch. - diff --git a/docs/source/howto/how_to_run/real_data_example.rst b/docs/source/howto/how_to_run/real_data_example.rst deleted file mode 100644 index ec81314e3..000000000 --- a/docs/source/howto/how_to_run/real_data_example.rst +++ /dev/null @@ -1,80 +0,0 @@ -.. _real-data-example: - -Real data processing -==================== - -This section presents an example of processing real experimental data using HTTomo from the `TomoBank`_ data archive. - -.. list-table:: - - - * - .. figure:: ../../_static/real_data/sino_tomo088.jpg - - Dark/Flat field corrected sinogram of the `Lorentz data set`_. - - - .. figure:: ../../_static/real_data/recon_tomo088.jpg - - Reconstructed slice using FBP method - - -Before starting, we assume that HTTomo has been successfully installed. If you have not installed HTTomo yet, please follow the -:ref:`installation_main`. - -We also recommend running the :ref:`run_tests` to verify that all required dependencies are installed and that the framework is -functioning correctly. - -For this example, we will use raw data from the `TomoBank`_ data archive. Please download the `Lorentz data set`_. The dataset is -hosted using the Globus file management system, which requires authentication. You can sign in using your GitHub credentials. - -.. _TomoBank: https://tomobank.readthedocs.io/en/latest/ - -.. _Lorentz data set: https://tomobank.readthedocs.io/en/latest/source/data/docs.data.lorentz.html - -Once the dataset has been downloaded, you should have the file :code:`tomo_00088.h5` on your disk. You can then proceed with running a simple HTTomo pipeline. - -TomoPy (CPU) pipeline -+++++++++++++++++++++ - -This pipeline uses the CPU implementation of the TomoPy library. TomoPy must be installed before running the pipeline. -See :ref:`backends_list`. - -Running this pipeline requires TomoPy package to be installed, see :ref:`backends_list`. Copy the following pipeline into a YAML file, -for example: :code:`tomopy_tomo_00088.yaml`. - -.. dropdown:: Standard 180 degrees pipeline using TomoPy (CPU) for tomo_00088.h5 dataset - - .. literalinclude:: ../../pipelines_full/tomopy_tomobank.yaml - :language: yaml - - -Then run HTTomo according to :ref:`howto_run_outside_diamond` documentation. In particular, provide the path to the input dataset, the pipeline YAML file, and the output directory. - -.. code-block:: console - - $ python -m httomo run path/to/tomo_00088.h5 tomopy_tomo_00088.yaml /path/to/output_folder - -GPU pipeline -++++++++++++ - -If a CUDA-enabled GPU is available, the same dataset can be processed using GPU-accelerated libraries. -This can significantly reduce the processing time for suitable pipelines. Run the pipeline bellow in a similar way as explained above. - -.. dropdown:: GPU-enabled processing for tomo_00088.h5 dataset - - .. literalinclude:: ../../pipelines_full/FBP3d_tomobar_tomobank.yaml - :language: yaml - -Output results -++++++++++++++ - -In the output folder you will find: - -1. Copied YAML file with the executed pipeline. -2. Debug log and the user log, see more :ref:`info_logger`. -3. The result of the reconstruction saved as HDF5 file. This file can be open, for instance with `Dawn`_ software. -4. Saved tiff files of the reconstructed image. You can use `ImageJ`_ or `ImageJ.JS`_ to visualise. - - -.. _Dawn: https://dawnsci.org/ -.. _ImageJ: https://imagej.net/ij/ -.. _ImageJ.JS: https://ij.imjoy.io/ \ No newline at end of file diff --git a/docs/source/howto/how_to_run/run_in_depth.rst b/docs/source/howto/how_to_run/run_in_depth.rst index fc9f8677b..733245382 100644 --- a/docs/source/howto/how_to_run/run_in_depth.rst +++ b/docs/source/howto/how_to_run/run_in_depth.rst @@ -1,640 +1,10 @@ -.. _run-httomo-indepth: +:orphan: -In-depth guide -============== +Command-line reference moved +============================ -Interacting with HTTomo through the command line interface (CLI) -++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ +The command-line reference is now at :ref:`run-httomo-indepth`. -The way to interact with the HTTomo software is through its "command line -interface" (CLI). +.. raw:: html -As mentioned earlier, the preliminary step to accessing installed HTTomo software -depends on if you are using a Diamond machine or not: - -- not on a Diamond machine: activate the conda environment that HTTomo was - installed into (please refer to :ref:`installation_main` for instructions on how to - install HTTomo) - -- on a Diamond machine: run the command :code:`module load httomo` - -Once the appropriate step has been done, you will have access to the HTTomo CLI: - -.. code-block:: console - - $ python -m httomo --help - Usage: python -m httomo [OPTIONS] COMMAND [ARGS]... - - httomo: High Throughput Tomography. - - Options: - --version Show the version and exit. - --help Show this message and exit. - - Commands: - check Check a YAML pipeline file for errors. - memory-check Estimate CPU memory requirements for processing input... - run Run a processing pipeline defined in YAML on input data. - -As can be seen from the output above, there are three HTTomo commands -available: :code:`check`, :code:`memory-check`, and :code:`run`. - -The :code:`check` command is used for checking a YAML process list file for -errors, and is highly recommended to be run before attempting to run the -pipeline. Please see :ref:`utilities_yamlchecker` for more information about -the checks being performed, the help information that is printed, etc. - -The :code:`memory-check` command is for estimating the CPU memory requirements -for processing the input data with a given pipeline and number of processes. -The underlying functionality was primarily developed for use with the DLS -HTTomo launcher, and has been exposed as a CLI command for convenience. - -The :code:`run` command is used for running HTTomo with a pipeline on the given -HDF5 input data. - -Both commands have arguments that are necessary to provide, arguments that are -optional, as well as several options/flags to customise their behaviour. - -Condensed information regarding the arguments that the commands take, as well as -the options for both commands, can be found directly from the command line by -using the :code:`--help` flag, such as :code:`python -m httomo check --help`. - -However, the next sections will describe each command in more detail, providing -supplementary material to the information in the CLI. - -.. note:: Diamond users will be able to use :code:`httomo` as a shortcut for - :code:`python -m httomo` - -The :code:`check` command -+++++++++++++++++++++++++ - -.. code-block:: console - - $ python -m httomo check --help - Usage: python -m httomo check [OPTIONS] YAML_CONFIG [IN_DATA] - - Check a YAML pipeline file for errors. - - Options: - --help Show this message and exit. - -.. note:: While HTTomo does support running pipelines in both YAML and JSON format, currently - the check functionality is only supported for YAML pipelines. - -Arguments -######### - -For :code:`check`, there is one *required* argument :code:`YAML_CONFIG`, and one -*optional* argument :code:`IN_DATA`. - -:code:`YAML_CONFIG` (required) -~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ - -This is the filepath to the YAML process list file that is to be checked. - -:code:`IN_DATA` (optional) -~~~~~~~~~~~~~~~~~~~~~~~~~~ - -This is the filepath to the HDF5 input data that you are intending to run the -YAML process list file on. - -This is useful to provide because the configuration of the loader in the YAML -process list file will have some references to the internal paths within the -HDF5 file, which must be typed correctly otherwise HTTomo will fail to access -the intended dataset within the HDF5 file. - -Providing the filepath to the HDF5 input data will perform a check of the loader -configuration in the YAML process list, determining if the paths mentioned in it -exist or not in the accompanying HDF5 file. - -Options/flags -############# - -The :code:`check` command has *no* options/flags. - -The :code:`memory-check` command -++++++++++++++++++++++++++++++++ - -.. code-block:: console - - $ python -m httomo memory-check --help - Usage: python -m httomo memory-check [OPTIONS] IN_DATA_FILE PIPELINE NPROCS - - Estimate CPU memory requirements for processing input data with a given - pipeline and number of processes - - Options: - --help Show this message and exit. - - -Arguments -######### - -For :code:`memory-check` there are three *required* arguments: -:code:`IN_DATA_FILE`, :code:`PIPELINE`, and :code:`NPROCS`, and zero *optional* -arguments. - -:code:`IN_DATA_FILE` (required) -~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ - -This is the filepath to the HDF5 input data that is intended to be processed. -This is required primarily for querying the size of the input data. - -:code:`PIPELINE` (required) -~~~~~~~~~~~~~~~~~~~~~~~~~~~ - -This is the filepath to the YAML process list file that defines the processing -to be applied to the input data. - -This is required for several reasons: - -- any cropping of the data via the loader's :code:`preview` parameter will - affect the size of the data being processed -- any methods requiring padding will affect the size of the data being - processed -- any :ref:`info_reslice` in the pipeline will affect the amount of memory - required - -:code:`NPROCS` (required) -~~~~~~~~~~~~~~~~~~~~~~~~~~~ - -This is the number of processes the input data is intended to be processed -with. - -This is required primarily due to the number of processes affecting how the -input data is split up, and thus affects the allocations required for -processing the subsets of data. - -.. note:: The value of :code:`NPROCS` must be >= 1. - -Options/flags -############# - -The :code:`memory-check` command has *zero* options/flags. - -The :code:`run` command -+++++++++++++++++++++++ - -.. code-block:: console - - $ python -m httomo run --help - Usage: python -m httomo run [OPTIONS] IN_DATA_FILE PIPELINE OUT_DIR - - Run a pipeline on input data. - - Options: - --output-folder-name DIRECTORY Define the name of the output folder created - by HTTomo - --save-all BOOL Save intermediate datasets for all tasks in - the pipeline. Set to True or False. - --gpu-id INTEGER The GPU ID of the device to use. - --reslice-dir DIRECTORY Directory for temporary files potentially - needed for reslicing (defaults to output - dir) - --max-cpu-slices INTEGER Maximum number of slices to use for a block - for CPU-only sections (default: 64) - --max-memory TEXT Limit the amount of memory used by the - pipeline to the given memory (supports - strings like 3.2G or bytes) - --save-snapshots BOOL Save intermediate images (snapshots) from - some methods in the pipeline. Set to True or - False. - --bits-sweep-images INTEGER Change the bit depth of saved tiff images in - the sweep run from default 32 bit to 16 or 8 - bit tiffs. - --monitor TEXT Add monitor to the runner (can be given - multiple times). Available monitors: bench, - summary - --monitor-output FILENAME File to store the monitoring output. - Defaults to '-', which denotes stdout - --intermediate-format [hdf5] Write intermediate data in hdf5 format - --compress-intermediate Write intermediate data in chunked format - with BLOSC compression applied - --syslog-host TEXT Host of the syslog server - --syslog-port INTEGER Port on the host the syslog server is - running on - --frames-per-chunk INTEGER RANGE - Number of frames per-chunk in intermediate - data (0 = write as contiguous, -1 = decide - automatically) [x>=-1] - --recon-filename-stem TEXT Name of output recon file without file - extension (assumes `.h5`) - --pipeline-format [yaml|json] Format of the pipeline input (YAML or JSON) - --mpi-abort-hook Enable hook that invokes MPI abort if an - unhandled exception is encountered - --help Show this message and exit. - -Arguments -######### - -For :code:`run`, there are three *required* arguments: - -- :code:`IN_FILE` -- :code:`PIPELINE` -- :code:`OUT_DIR` - -and zero *optional* arguments. - -:code:`IN_FILE` (required) -~~~~~~~~~~~~~~~~~~~~~~~~~~ - -This is the filepath to the HDF5 input data that you are intending to process. - -:code:`PIPELINE` (required) -~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ - -Most commonly, this is the filepath to the YAML process list file that contains -the desired processing pipeline. - -HTTomo also supports pipelines in JSON format provided as a string. See -:ref:`pipeline-format` for specifying the pipeline format. - -:code:`OUT_DIR` (required) -~~~~~~~~~~~~~~~~~~~~~~~~~~ - -This is the path to a directory which HTTomo will create its output directory -inside. - -The output directory created by HTTomo contains a date and timestamp in the -following format: :code:`{DAY}-{MONTH}-{YEAR}_{HOUR}_{MIN}_{SEC}_output/`. For -example, the output directory created for an HTTomo run on 1st May 2023 at -15:30:45 would be :code:`01-05-2023_15_30_45_output/`. If the :code:`OUT_DIR` -path provided was :code:`/home/myuser/`, then the absolute path to the output -directory created by HTTomo would be -:code:`/home/myuser/01-05-2023_15_30_45_output/`. - -Options/flags -############# - -The :code:`run` command has 19 options/flags: - -- :code:`--output-folder-name` -- :code:`--save-all` -- :code:`--gpu-id` -- :code:`--reslice-dir` -- :code:`--max-cpu-slices` -- :code:`--max-memory` -- :code:`--save-snapshots` -- :code:`--bits-sweep-images` -- :code:`--monitor` -- :code:`--monitor-output` -- :code:`--intermediate-format` -- :code:`--compress-intermediate` -- :code:`--syslog-host` -- :code:`--syslog-port` -- :code:`--frames-per-chunk` -- :code:`--recon-filename-stem` -- :code:`--pipeline-format` -- :code:`--mpi-abort-hook` -- :code:`--continuous-scan-subset` - -:code:`--output-folder-name` -~~~~~~~~~~~~~~~~~~~~~~~~~~~~ - -As described in the documentation for the :code:`OUT_DIR` argument, the default name of the -output directory created by HTTomo consists primarily of a timestamp. If one wishes to provide -a name for the directory created by HTTomo instead of using the default timestamp name, then -the :code:`--output-folder-name` flag may be used to achieve this. - -For example, if the :code:`OUT_DIR` path provided was :code:`/home/myuser`, and -:code:`--output-folder-name=test-1` was given, then the absolute path of the output directory -created by HTTomo would be :code:`/home/myuser/test-1/`. - -.. _httomo-saving: - -:code:`--save-all` -~~~~~~~~~~~~~~~~~~ - -Regarding the output of methods, HTTomo's default behaviour is to *not* write -the output of a method to a file in the output directory unless one of the -following conditions is satisfied: - -- the method is the last one in the processing pipeline -- the method is a reconstruction method -- the :code:`save_result` parameter has been provided a value of :code:`true` in - a method's YAML configuration (see :ref:`save-result-examples` for more info - on the :code:`save_result` parameter) - -However, there are certain cases such as debugging, where saving the output of -all methods to files in the output directory is beneficial. This flag is a quick -way of doing so. - -:code:`--reslice-dir` -~~~~~~~~~~~~~~~~~~~~~ - -This is related to the :code:`--file-based-reslice` flag. - -By default, the directory that the file being used for the re-slice operation is -the output directory that HTTomo creates. - -If this output directory is on a network-mounted disk, then read/write -operations to such a disk will in general be much slower compared to a local -disk. In particular, this means that the re-slice operation will be much slower -if the output directory is on a network-mounted disk rather than on a local -disk. - -This flag can be used to specify a different directory inside which the file -used for re-slicing should reside. - -In particular, if performing the re-slice with a file and the output directory is -on a *network-mounted disk*, it is recommended to use this flag to choose an -output directory that is on a *local disk* where possible. This will -*drastically* improve performance, compared to performing the re-slice with a -file on a network-mounted disk. - -.. note:: If running HTTomo across multiple machines, using a single local disk - to contain the file used for re-slicing is not possible. - -Below is a summary of the different re-slicing approaches and their relative -performances: - -============================ ========= -Re-slice type Speed -============================ ========= -In-memory Very fast -File w/ local disk Fast -File w/ network-mounted disk Very slow -============================ ========= - -:code:`--max-cpu-slices` -~~~~~~~~~~~~~~~~~~~~~~~~ - -This flag is only relevant only for runs which are using a pipeline that contains -1 or more sections that are composed of purely CPU methods. - -Understanding this flag's usage is dependent on knowledge of the concept of -"chunks", "blocks", and "sections" within HTTomo's framework, so please refer to -:ref:`detailed_about` for information on these concepts. - -The notion of a block is fully utilised to increase performance when a sequence of -two or more GPU methods are being executed. When two or more CPU methods are -executed in sequence, the notion of a block plays a less significant role in -performance. The number of slices in a block is driven by the memory capacity of -the GPU, but if no GPU is being used for executing a sequence of methods in the -pipeline, there is no obvious way to choose the number of slices in a block (the -"block size"). - -In such cases the user may wish to tweak the block size to explore if a specific -block size happens to improve performance for the CPU-only section(s). - -:code:`--max-memory` -~~~~~~~~~~~~~~~~~~~~ - -HTTomo supports execution on both: - -- compute clusters, where large amounts of system RAM are typically available -- personal workstations or laptops, where available RAM is often more limited - -To accommodate these different environments, HTTomo dynamically manages intermediate -data during pipeline execution. Whenever sufficient RAM is available, data is kept -in memory to maximise performance. If memory requirements exceed the available -RAM, HTTomo automatically stores intermediate data on disk and reloads it as needed. - -The :code:`--max-memory` option specifies the amount of RAM available to HTTomo. -It sets the limit in terms of the maximum amount of the CPU memory per process available -on the system. For instance, if you're running HTTomo on 1 process serially, you need -to pass the value of the CPU memory available on your system, e.g., :code:`--max-memory 32G` would -set it to 32 Gigabyte. Providing an accurate value allows HTTomo to optimise memory usage -while avoiding out-of-memory errors. - - -:code:`--save-snapshots` -~~~~~~~~~~~~~~~~~~~~~~~~ - -When this flag is enabled, the pipeline saves image snapshots at specific execution points. -These snapshots are captured during selected methods - typically when a section boundary -is reached and data is transferred to the CPU. At which time a slice of the data is saved for -inspection. - -This feature is particularly useful for complex pipelines (e.g. 360 degrees with stitching and phase contrast), -where intermediate processing steps involved in reconstruction may unintentionally alter -the data. By reviewing these snapshot images (JPEGs), users can more easily pinpoint -where issues are introduced in the pipeline. - -Enabling snapshots incurs almost no additional computational cost, unlike the :code:`--save-all` -flag, which requires saving the entire dataset into a file for each method. - -:code:`--bits-sweep-images` -~~~~~~~~~~~~~~~~~~~~~~~~ - -This flag allows to set the bit depth for tiff images saved during :ref:`parameter_sweeping`. By default, -HTTomo will use the bit-depth of the data, select here 8, 16 or 32-bit (default). - -:code:`--monitor` -~~~~~~~~~~~~~~~~~ - -HTTomo has the capability of reporting information about the performance of the -various methods involved in the specific pipeline that will be executed. -Specifically: - -- time taken for methods to execute on the CPU/GPU -- transfer time to and from the GPU -- time taken to write to files (if HTTomo uses a file instead of RAM to hold data - during pipeline execution) - -There are two options for this flag, :code:`summary` and :code:`bench`. - -:code:`--monitor=summary` -^^^^^^^^^^^^^^^^^^^^^^^^^ - -The :code:`summary` option will produce a brief summary of the time taken for each -method to execute in the pipeline, which will look something like the following: - -.. code-block:: console - - Summary Statistics (aggregated across 1 processes): - Total methods CPU time: 19.376s - Total methods GPU time: 19.042s - Total host2device time: 0.013s - Total device2host time: 0.548s - Total sources time : 0.063s - Total sinks time : 0.028s - Other overheads : 0.362s - --------------------------------------- - Total pipeline time : 19.829s - Total wall time : 19.829s - --------------------------------------- - Method breakdowns: - data_reducer : 0.001s ( 0.0%) - find_center_vo : 11.586s (58.4%) - remove_outlier : 3.312s (16.7%) - normalize : 0.334s ( 1.7%) - remove_stripe_based_sorting : 2.987s (15.1%) - FBP : 0.966s ( 4.9%) - save_intermediate_data : 0.019s ( 0.1%) - save_to_images : 0.171s ( 0.9%) - -:code:`--monitor=bench` -^^^^^^^^^^^^^^^^^^^^^^^ - -The :code:`bench` option (short for "benchmark") provides a much more in-depth -breakdown of the time taken for each method to execute (dividing it into time -taken on CPU vs. GPU, data transfer times to and from the GPU), and providing this -information for all processes involved in the run. - -This output is very verbose, but can provide some insight if, for example, wanting -to see what parts of the pipeline may be slower than expected. - -:code:`--monitor-output` -~~~~~~~~~~~~~~~~~~~~~~~~ - -By default the output of any usage of the :code:`--monitor` flag will be written -to :code:`stdout` (ie, printed to the terminal). However, there are times when -it's useful to write the monitoring output to a file, such as for performance -analysis. - -HTTomo supports writing the monitoring results in CSV format, and so any given -filepath to the :code:`--monitor-output` flag will produce a file with the -benchmarking results written in CSV format. - -:code:`--intermediate-format` -~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ - -TODO - -:code:`--compress-intermediate` -~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ - -TODO - -:code:`--syslog-host` -~~~~~~~~~~~~~~~~~~~~~ - -TODO - -:code:`--syslog-port` -~~~~~~~~~~~~~~~~~~~~~ - -TODO - -:code:`--frames-per-chunk` -~~~~~~~~~~~~~~~~~~~~~~~~~~ - -This flag sets the number of frames in a chunk for the intermediate file. - -- -1 (the default), it will be decided automatically; -- 0, contiguous storage will be used (no chunk storage); -- >= 1, the number of frames in a chunk. - -For most cases the default -1 should be sufficient as the actual number of -frames in a chunk is optimised by considering the saturation bandwidth of the -filesystem. - -:code:`--recon-filename-stem` -~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ - -By default, if the output of a method is saved to a file, the filename is of -the form :code:`task_{N}-{PACKAGE_NAME}-{METHOD_NAME}.h5`, where: - -- :code:`N` is the index of the method in the pipeline (zero-indexing) -- :code:`PACKAGE_NAME` is the name of the package that the method comes from -- :code:`METHOD_NAME` is the name of the method - -For the output of a reconstruction method specifically, if a filename different -to the above format is desired, this can be provided using the -:code:`--recon-filename-stem` flag. The files created will always be hdf5 -files, so the file extension should always be :code:`h5`. Therefore, only the -"stem" of the desired filename needs to be provided (the part of the filename -before the file extension). - -For example, if the desired reconstruction filename is :code:`my-recon.h5`, -then the flag should be used as :code:`--recon-filename-stem=my-recon`. - -.. _pipeline-format: - -:code:`--pipeline-format` -~~~~~~~~~~~~~~~~~~~~~~~~~ - -HTTomo supports running pipelines defined in YAML and JSON format. The format -of a given pipeline can be specified with this flag by providing a string -stating either YAML or JSON, where the string is case-insensitive. - -The default setting is YAML, so this flag can be omitted if one wishes to run a -YAML pipeline. - -.. note:: HTTomo currently only accepts YAML pipelines as files and JSON - pipelines as strings. Ie, both YAML pipelines provided as strings and JSON - pipelines provided as files are not currently supported. - -:code:`--mpi-abort-hook` -~~~~~~~~~~~~~~~~~~~~~~~~ - -.. note:: This is a flag primarily for debugging. - -When running HTTomo under MPI, there are cases when an exception can be raised in one or more -processes but not in all processes. If execution of the alive process(es) reaches a point which -involves communication with the other process(es) (for example, an MPI gather), the alive -process(es) will be stuck indefinitely, waiting for the dead process(es) to communicate to -them. - -To avoid a deadlock in such cases, a hook into python's exception handling can be used to -invoke MPI abort if any of the python processes encounter an unhandled exception. This results -in the following behaviour: if one process encounters an unhandled exception, all processes are -guaranteed to be terminated, and thus no deadlock will be encountered. - -Something to be aware about with this aproach is that the MPI abort occurs not at the python -layer, but rather, at the MPI implementation layer. In particular, this means that python's -writing to stdout/stderr is not guaranteed to complete before the MPI abort is invoked. -Meaning, while printing of the traceback of the unhandled exception that triggered the MPI -abort does exist in the python code, there is no guarantee that this printing will be complete -before the MPI abort mechanism begins to terminate the python processes. Thus, the output in a -terminal when MPI abort is invoked may only contain partial information about the exception -that triggered the MPI abort. - -:code:`--continuous-scan-subset` -~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ - -This is an alternative to the :code:`continuous_scan_subset` loader parameter, see -:ref:`continuous_scan_subset_selection` for more details. - -This flag takes two values, the first one being the start index and the second one being the -stop index. For example, an equivalent of the following config: - -.. literalinclude:: ../../../tests/samples/pipeline_template_examples/testing/loader_with_offset_param.yaml - :language: yaml - :emphasize-lines: 7-9 - -can be achieved with the flag via :code:`--continuous-scan-subset 90 180` - -.. note:: This flag overrides the :code:`continuous_scan_subset` parameter in the YAML - config. Meaning, if the :code:`continuous_scan_subset` parameter is present in the loader's - config in the YAML pipeline but the :code:`--continuous-scan-subset` flag is used, then the - values given in the YAML config are ignored and the values given to the flag take - precedence. If this occurs, it will be logged to the :code:`debug.log` file that HTTomo - produces. - -Developer options -+++++++++++++++++ - -Tracing -####### - -One tool that can be used to record CPU traces, along with CPU and memory usage statistics is `VizTracer`_. - -.. _VizTracer: https://viztracer.readthedocs.io - -To enable CPU and memory usage statistic recording in the traces use the :code:`--no-standalone` option of httomo. - -VizTracer -######### - -Recording a trace with viztracer: - -:code:`python -m viztracer --plugin "vizplugins --cpu_usage --memory_usage" --output_file output.json -m httomo run --no-standalone data.nxs pipeline.yaml output_directory` - -Recording a trace for multiple ranks: - -:code:`mpirun -n 4 bash -c 'python -m viztracer --plugin "vizplugins --cpu_usage --memory_usage" --output_file output_rank_${OMPI_COMM_WORLD_RANK}.json -m httomo run --no-standalone data.nxs pipeline.yaml output_directory'` - -A separate file is created for each rank. -VizTracer can combine them into a single json file: - -:code:`viztracer --combine output_rank_*.json -o output.json` - -VizTracer's output json files can be viewed by vizviewer offline: - -:code:`vizviewer output.json` - -Or by using the online version of `Perfetto`_. - -.. _Perfetto: https://ui.perfetto.dev \ No newline at end of file + diff --git a/docs/source/howto/httomo_features.rst b/docs/source/howto/httomo_features.rst index f5a8fb03b..02bc9749b 100644 --- a/docs/source/howto/httomo_features.rst +++ b/docs/source/howto/httomo_features.rst @@ -1,12 +1,21 @@ -HTTomo Features -********************** +.. _httomo_features: + +Pipeline features +***************** + +This section guides you from the basics of configuring a pipeline to advanced +features such as parameter sweeps, saving results and managing memory. It also +explains side outputs and padding. + .. toctree:: :maxdepth: 2 - httomo_features/previewing + httomo_features/how_configure_pipeline + httomo_features/optimise_pipeline + httomo_features/memory_and_performance + httomo_features/side_out httomo_features/centering httomo_features/parameter_sweeping + httomo_features/save_results httomo_features/padding - - diff --git a/docs/source/howto/httomo_features/centering.rst b/docs/source/howto/httomo_features/centering.rst index f26467c08..7ea92a60d 100644 --- a/docs/source/howto/httomo_features/centering.rst +++ b/docs/source/howto/httomo_features/centering.rst @@ -1,65 +1,63 @@ .. default-role:: math .. _centering: -Centre of Rotation (CoR) -^^^^^^^^^^^^^^^^^^^^^^^^ - -What is it? -=========== -Identifying the optimal Centre of Rotation (CoR) parameter is an important -procedure to ensure the correctness of the reconstruction. It is crucial to find it -as precise as possible, as the incorrect value can lead to distortions in reconstructions, therefore making post processing and -quantification invalid. - -The required CoR parameter places the object (a scanned sample) into a coordinate system of the scanning device to ensure -that the object is centered and rotates around its axis in this system, see :numref:`fig_centerscheme`. This is essential for -valid reconstruction as the back projection model assumes a central placing of a sample with respect to the detector's axis -(usually horizontal for synchrotrons). - -The CoR estimation problem is also sometimes referred to the centered sinogram. -If the sinogram is not centered and there is an offset (`d` in :numref:`fig_centerscheme`), the reconstruction -will result in strong arching artefacts at the boundaries of reconstructed objects, see :numref:`fig_center_find`. -Further from the optimal CoR value (here `d=0`), one should expect more pronounced arching. -Therefore the optimisation problem usually involves minimising the artefacts in the reconstructed images -by varying the CoR value. +Centre of Rotation +^^^^^^^^^^^^^^^^^^ + +The Centre of Rotation (CoR) aligns the sample's rotation axis with the +acquisition coordinate system, as shown in :numref:`fig_centerscheme`. +Reconstruction assumes this alignment, so an inaccurate CoR can distort the +result and invalidate later analysis. + +An offset sinogram (`d` in :numref:`fig_centerscheme`) produces arching +artefacts around object boundaries, as shown in :numref:`fig_center_find`. +These artefacts increase with the distance from the correct CoR (`d=0`), so +CoR estimation typically searches for the value that minimises them. .. _fig_centerscheme: .. figure:: ../../_static/cor/CoR.svg :scale: 55 % :alt: CoR scheme for tomography - The CoR parameter can be defined as a distance `d` that translates the coordinate system `(x,y)` of the scanned object to the coordinate system `(s,p)` of the acquisition device. This leads to a simple linear mapping `(s = x +- d, p = y)`. + The CoR offset `d` translates the sample coordinates `(x,y)` into the + acquisition coordinates `(s,p)`: `(s = x +- d, p = y)`. .. _fig_center_find: .. figure:: ../../_static/cor/corr_select.png :scale: 85 % :alt: Finding CoR - The reconstructions using different CoR values. Incorrectly centered sinogram results in strong arching artifacts on the boundaries of the reconstructed object. Note how the arching is being reduced when `d` is closer to the correct value. + Reconstructions with different CoR values. Boundary artefacts decrease as + `d` approaches the correct value. CoR in HTTomo ============= -The CoR parameter is present in any reconstruction template given as the :code:`center` parameter. It can be configured either -automatically (see :ref:`centering_auto`) or manually (see :ref:`centering_manual`). +Every reconstruction template provides a :code:`center` parameter. Set it +automatically (see :ref:`centering_auto`) or manually (see +:ref:`centering_manual`). .. _centering_auto: Auto-centering =============== -There is a `variety `_ of -methods to estimate CoR automatically. At DLS, we frequently use the centering method which -was developed by Nghia Vo and it relies on the Fourier analysis of a sinogram, see the `paper `_. -This method is implemented in both TomoPy and HTTomolibgpu libraries and -available for HTTomo as *find_center_vo* template, see :ref:`reference_templates`. +Several methods can estimate the CoR automatically. DLS commonly uses Nghia +Vo's Fourier-based sinogram method (`paper`_), implemented by TomoPy and +HTTomolibGPU. In HTTomo it is available as the ``find_center_vo`` template; +see :ref:`reference_templates`. If one automatic method fails, try another +`HTTomolibGPU centring method`_. + +.. _paper: https://doi.org/10.1364/OE.22.019078 +.. _HTTomolibGPU centring method: https://diamondlightsource.github.io/httomolibgpu/api/httomolibgpu.recon.rotation.html -Here are the steps to enable the auto-centering and then use the estimated value in the reconstruction: +To use automatic centering: -1. The auto-centering method should be added to the process list before the reconstruction method. -2. It is recommended to position any auto-centering method right after the loader, see :ref:`pl_conf_order`. -3. The calculated CoR value with be stored in :ref:`side_output`. -4. In the reconstruction module we refer to that value by placing the reference into the :code:`center` parameter. +1. Add the centering method before reconstruction, preferably immediately + after the loader; see :ref:`pl_conf_order`. +2. Store its calculated CoR as a :ref:`side_output`. +3. Reference that output in the reconstruction method's :code:`center` + parameter. .. code-block:: yaml :emphasize-lines: 13,17 @@ -85,20 +83,18 @@ Here are the steps to enable the auto-centering and then use the estimated value recon_size: null recon_mask_radius: null -.. note:: When one auto-centering method fails it is recommended to try other available methods as they can still provide the correct or close to the correct CoR value. .. _centering_manual: Manual Centering ================= -Unfortunately, there could be various cases when :ref:`centering_auto` fails, e.g., the projection data is corrupted, incomplete, the object is outside the field of view of the detector, and possibly other issues. -In that case, it is recommended to find the center of rotation manually. :ref:`parameter_sweeping` can simplify such search significantly. - -To enable manual centering without sweeping, one would need to do the following steps: - -1. Ensure that the auto centering estimation method is not in the process list (remove or comment it). -2. Modify the centre of rotation value :code:`center` in the reconstruction plugin by substituting a number instead of the reference to side outputs. - +Automatic centering can fail when projection data is corrupt or incomplete, +or when the sample extends beyond the detector's field of view. In these cases, +set the CoR manually. :ref:`parameter_sweeping` can help identify the value. +To set it without parameter sweeping: +1. Remove or comment out the automatic centering method. +2. Replace the side-output reference in the reconstruction method's + :code:`center` parameter with a numeric value. diff --git a/docs/source/howto/httomo_features/how_configure_pipeline.rst b/docs/source/howto/httomo_features/how_configure_pipeline.rst new file mode 100644 index 000000000..82ac33d06 --- /dev/null +++ b/docs/source/howto/httomo_features/how_configure_pipeline.rst @@ -0,0 +1,96 @@ +.. _howto_process_list: +.. _how_to_configure_pipeline: + +Configure a pipeline +******************** + +An HTTomo pipeline is an ordered sequence of loading and processing operations +defined in YAML. See +:ref:`explanation_process_list` for an introduction to pipelines and +:ref:`explanation_templates` for the structure of method templates. + +Choose an editor +---------------- + +Use a text editor with YAML syntax highlighting and indentation support. +Suitable editors include: + +* `VS Code for the Web `_ Online YAML editing +* `Visual Studio Code `_ +* `PyCharm `_ +* `Sublime Text `_ +* `Notepad++ `_ for Windows +* `Vim `_ or `Neovim `_ for + terminal-based editing + +Build the pipeline +------------------ + +#. Start with an :ref:`HTTomo loader `. +#. Copy the required processing methods from + :ref:`reference_templates` into the same YAML file. +#. Arrange the methods in execution order. HTTomo runs them sequentially from + top to bottom. +#. Edit each method's parameters for the input data and processing task. Consult + the relevant library documentation for parameter details. + + +.. _utilities_yamlchecker: + +Validate the pipeline +--------------------- + +.. note:: + + Pipeline validation is integrated into the launcher at Diamond Light Source. + See :ref:`howto_run_at_diamond`. + +Check the pipeline regularly while editing it: + +.. code-block:: console + + $ python -m httomo check pipeline.yaml + +Supplying the input file also validates the dataset paths used by the loader: + +.. code-block:: console + + $ python -m httomo check pipeline.yaml input.nxs + +Always validate the completed pipeline before running it. See +:ref:`run-httomo-indepth` for the complete ``check`` command syntax. + +What validation checks +++++++++++++++++++++++ + +The checker reports an error when: + +* the YAML syntax or indentation is invalid; +* the first pipeline entry is not an HTTomo loader; +* a method or module path is unknown; +* a parameter name or value has the wrong type; +* a required parameter is missing; +* a side-output reference is invalid; or +* a referenced HDF5 dataset does not exist when an input file is supplied. + +When a loader parameter is set to ``auto``, supplying the input file also +checks that HTTomo can find an NXtomo entry. See :ref:`nxtomo_discovery`. + +Common validation errors +++++++++++++++++++++++++ + +.. list-table:: + :header-rows: 1 + :widths: 38 62 + + * - Message or symptom + - What to check + * - YAML cannot be parsed + - Use spaces rather than tabs and align fields at the same nesting level. + * - Method is not valid + - Copy its ``method`` and ``module_path`` from :ref:`reference_templates`. + * - Parameter is unknown or has the wrong type + - Compare the ``parameters`` mapping with the method template. + * - Dataset path is not valid + - Check the loader paths in the input HDF5 file, or use NXtomo automatic + discovery where supported. diff --git a/docs/source/howto/httomo_features/memory_and_performance.rst b/docs/source/howto/httomo_features/memory_and_performance.rst new file mode 100644 index 000000000..b03c12e3c --- /dev/null +++ b/docs/source/howto/httomo_features/memory_and_performance.rst @@ -0,0 +1,77 @@ +.. _memory_and_performance: + +Memory and performance +====================== + +HTTomo divides each :term:`section` between MPI processes and then processes +each :term:`chunk` in memory-sized :term:`blocks `. Two controls help +when planning a run: ``memory-check`` estimates host-memory demand before the +run, while ``--max-memory`` limits memory use by each process at runtime. + +Estimate CPU memory before a run +-------------------------------- + +Run the estimate with the same input, pipeline and process count that will be +used for processing: + +.. code-block:: console + + $ python -m httomo memory-check INPUT.nxs PIPELINE.yaml NPROCS + +The result is an estimated peak across all ``NPROCS`` processes. Divide it by +``NPROCS`` for an approximate per-process value. The calculation includes the +input data type and preview, section padding, shape changes and memory required +for a :term:`re-slice`. + +Leave additional capacity for Python, MPI, HDF5, the operating system and other +jobs. The command estimates HTTomo's main section storage; it is not a guarantee +that the complete process will remain below the reported value. + +Choose a process count +---------------------- + +More processes divide the section data into smaller chunks, but every process +has runtime overhead and may allocate method-specific buffers. For GPU runs, +start with one process per GPU. For CPU runs, increase the process count only +while the machine has sufficient memory and I/O bandwidth. + +Use a runtime memory ceiling +---------------------------- + +``--max-memory`` is a **per-process** ceiling, unlike the total reported by +``memory-check``: + +.. code-block:: console + + $ python -m httomo run INPUT.nxs PIPELINE.yaml OUTPUT \ + --max-memory 32G + +When a section's estimated host-memory requirement reaches the ceiling, HTTomo +uses a temporary HDF5-backed store instead of keeping the section data in RAM. +For GPU sections, the same value caps the memory budget used to choose block +sizes; available device memory still provides an upper bound. A value of ``0`` +disables the user ceiling. + +Disk-backed sections protect memory at the cost of extra I/O. The warning +``Chunk does not fit in memory - using a file-based store`` indicates that this +path was selected. Put ``--reslice-dir`` on fast storage that every +participating process can access. + +Reduce resource use +------------------- + +If an estimate or run is too large: + +* crop unused detector regions with :term:`preview`; +* process a smaller angular range while developing a pipeline; +* reduce the number of simultaneous processes if aggregate host memory is the + limit; +* lower ``--max-cpu-slices`` for CPU-only sections; +* lower ``--max-memory`` to reduce GPU blocks or select disk-backed storage; +* avoid ``--save-all`` unless every intermediate result is needed; and +* use ``--compress-intermediate`` when storage capacity matters more than + compression overhead. + +For the complete option definitions, see :ref:`run-httomo-indepth`. Developers +implementing or correcting method estimates should read +:ref:`developers_memorycalc`. diff --git a/docs/source/howto/httomo_features/optimise_pipeline.rst b/docs/source/howto/httomo_features/optimise_pipeline.rst new file mode 100644 index 000000000..140856f2d --- /dev/null +++ b/docs/source/howto/httomo_features/optimise_pipeline.rst @@ -0,0 +1,71 @@ +.. _optimise_pipeline: + +Optimise pipeline performance +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +Pipeline performance is influenced mainly by method order, CPU/GPU data +transfers and intermediate file output. Understanding :ref:`info_sections` and +:ref:`info_reslice` can help when applying the guidance below. + +.. _pl_conf_order: + +Group methods by data pattern +============================= + +HTTomo runs pipeline methods sequentially from top to bottom. Each method has +one of three data patterns: + +``projection`` + Data is sliced by projection. + +``sinogram`` + Data is sliced by sinogram. + +``all`` + The method inherits the pattern of the preceding method. + +Changing between ``projection`` and ``sinogram`` requires a potentially costly +:ref:`re-slice `. Where processing requirements allow, group +methods with the same pattern to reduce the number of re-slices. + +HTTomo loaders use the ``projection`` pattern, so start with projection-based +methods where possible. Centre-finding methods should normally be placed near +the start of the pipeline. + +.. _pl_library: + +Method metadata +=============== + +HTTomo obtains each method's pattern, implementation type and memory +requirements from :ref:`httomo-backends `. See the +`httomo-backends method metadata documentation +`_ +for details. + +Group GPU methods +================= + +Supported methods use one of three implementation types: + +``cpu`` + Runs on the CPU. + +``gpu`` + Runs on a GPU but receives its input as a NumPy array in CPU memory. + +``gpu_cupy`` + Runs on a GPU using CuPy arrays. Data remains in GPU memory between + consecutive ``gpu_cupy`` methods. + +When a GPU is available, prefer GPU implementations where appropriate and keep +``gpu_cupy`` methods together to reduce transfers between CPU and GPU memory. +Available implementations are listed in :ref:`backends_list`. + +Minimise writing to disk +======================== + +Writing intermediate datasets can significantly slow a pipeline and consume +substantial disk space. Use ``save_result`` or ``--save-all`` only when those +intermediate results are needed. Use ``--save-snapshots`` to capture lightweight +diagnostic snapshots instead. See :ref:`save-result-examples` for details. diff --git a/docs/source/howto/httomo_features/padding.rst b/docs/source/howto/httomo_features/padding.rst index e4dc67fbd..9c220764c 100644 --- a/docs/source/howto/httomo_features/padding.rst +++ b/docs/source/howto/httomo_features/padding.rst @@ -4,28 +4,35 @@ Padding ^^^^^^^ -Padding is an important feature of HTTomo when performing computations on :ref:`chunks_data` and :ref:`blocks_data`. -If the method is a 2D method (work with 2D frames), e.g., a denoising filter, then the data does not require any padding as -the boundary conditions between blocks will not be violated. However, when the method is a 3D method and it works -with 3D volumes, then in order to satisfy boundary conditions and avoid artefacts, one needs to pad blocks. +HTTomo processes data as :ref:`chunks_data` and :ref:`blocks_data`. Methods +that operate on independent 2D frames, such as 2D denoising filters, do not +need :term:`padding`. Methods that operate on 3D volumes need padded blocks to preserve +boundary conditions and prevent artefacts. How this can be useful? ======================= -It is useful because when padding feature exists one can use true fully 3D methods which provide a consistent resolution in -all three dimensions, see the image below. Because of the access to 3D data, one can perform better in removing artefacts, -improving contrast, etc. +Padding enables fully 3D methods, which can provide consistent resolution in +all dimensions, improve contrast, and remove artefacts more effectively. -.. list-table:: +.. list-table:: * - .. figure:: ../../_static/padding/denoising2d.jpg :scale: 20 % - 2D denoising applied to 3D data, note the resolution inconsistency in the vertical direction. + 2D denoising produces inconsistent vertical resolution. - - .. figure:: ../../_static/padding/denoising3d_pad5.jpg + - .. figure:: ../../_static/padding/denoising3d_pad5.jpg :scale: 20 % - 3D denoising applied to 3D data. The resolution is consistent in all three dimensions. + 3D denoising produces consistent resolution in all dimensions. -.. note:: With padding enabled, HTTomo can perform more state-of-the-art filtering techniques as well as advanced iterative reconstruction in 3D. \ No newline at end of file +.. note:: Padding supports advanced 3D filters and iterative reconstruction + methods. + +How to use +========== + +There is no need to do anything specific to enable this feature as it will +be switched on automatically when the method that requires it added to the +pipeline. diff --git a/docs/source/howto/httomo_features/parameter_sweeping.rst b/docs/source/howto/httomo_features/parameter_sweeping.rst index 85e72fb3d..9e0a11d06 100644 --- a/docs/source/howto/httomo_features/parameter_sweeping.rst +++ b/docs/source/howto/httomo_features/parameter_sweeping.rst @@ -3,57 +3,45 @@ Parameter Sweeping ^^^^^^^^^^^^^^^^^^ -What is it? -=========== - -Parameter sweeping refers to providing multiple values for a specific parameter -of a method, and then running that method on its input data with the different -values for that parameter. +A :term:`parameter sweep` provides multiple values for one parameter and runs +the method once for each value. How would this be useful when processing data? ============================================== -This feature is typically used when prototyping a process list and it is -difficult to guess a reasonable value of a method's parameter. There could be -many reasons for this situation, such as being unfamiliar with the method, or -working with unfamiliar data, etc. +Use a parameter sweep when prototyping a pipeline to compare possible +values, especially for an unfamiliar method or dataset. -How the output looks like? -========================== +What does the output look like? +=============================== .. _fig_centergif: .. figure:: ../../_static/sweep/sweep_cor.gif :scale: 55 % :alt: Sweep for cor - The result of sweeping applied to find the correct :ref:`centering`. The images with the CoR values printed on them are saved in the folder. The user can find the correct value by looking through the images and then input the value manually into the pipeline, see :ref:`centering_manual`. - + A sweep used to find the correct :ref:`centering`. Each saved image shows + its CoR value. Inspect the images, then enter the best value in the pipeline; + see :ref:`centering_manual`. -How are parameter sweeps defined in the process list YAML file? -=============================================================== -There are two ways of specifying the values that a parameter sweep should be -performed across: +How are parameter sweeps defined in the pipeline YAML file? +=========================================================== -1. Specifying a range of values via start, stop and step values +Specify sweep values in either of two ways: -2. Manually specifying each value +1. A range defined by start, stop, and step values. +2. An explicit list of values. -.. note:: A pipeline can only have 1 parameter sweep in it at a time. Any - pipelines with more than 1 parameter sweep defined in it will not be - executed, and an error message will be displayed. +.. note:: A pipeline can contain only one parameter sweep. HTTomo reports an + error and does not run a pipeline containing more than one. Specifying a range ++++++++++++++++++ -The first way is done by providing a start, stop and step value. Along with this -information, a special phrase :code:`!SweepRange` is used to "mark" in the YAML -that the start, stop and step values are for defining a parameter sweep. - -The snippet below is defining a parameter sweep for the :code:`center` parameter -of a reconstruction method, where the sweep starts at :code:`10`, ends at -:code:`40` (similar to python slicing, the end value is not included), with steps -of :code:`10` inbetween: +Use :code:`!SweepRange` with :code:`start`, :code:`stop`, and :code:`step`. +Like a Python range, it excludes the stop value. This example sweeps the +reconstruction :code:`center` parameter from 10 to 40 in steps of 10: .. code-block:: yaml @@ -65,13 +53,8 @@ of :code:`10` inbetween: Specifying each value +++++++++++++++++++++ -The second way is done by providing a list of values for a parameter, and again -"marking" the list with a special phrase to denote that this list of values is -defining a parameter sweep. The phrase in this case is :code:`!Sweep`. - -The snippet below is defining a parameter sweep for the :code:`size` parameter -of a median filter method, where the sweep is across the two values :code:`3` -and :code:`5`: +Use :code:`!Sweep` to provide explicit values. This example sweeps a median +filter's :code:`size` parameter over :code:`3` and :code:`5`: .. code-block:: yaml @@ -82,52 +65,38 @@ and :code:`5`: Example +++++++ -Below, :code:`!Sweep` is used in the context of a fully working minimal -pipeline to sweep over the :code:`size` parameter of a median filter: +This minimal pipeline uses :code:`!Sweep` for a median filter's :code:`size` +parameter: .. literalinclude:: ../../../../tests/samples/pipeline_template_examples/testing/sweep_manual.yaml :language: yaml -.. note:: There is no need to add image saving method after the `sweep` method. The result of the `sweep` method will be saved into images automatically. +.. note:: Sweep results are saved as images automatically; no image-saving + method is required. How big should the input data be? ================================= -Due to the goal of parameter sweeps being to provide quick feedback to optimise -a parameter value, it is typical to run a parameter sweep on a few sinograms, -rather than the full data. - -As such, a parameter sweep run in HTTomo is constrained to run on data previewed -to contain 7 sinogram slices or less. Meaning, in order to perform a parameter -sweep in a pipeline, the input data must be cropped to 7 sinogram slices or less -using the :code:`preview` parameter of the loader (see :ref:`previewing` for -more details), otherwise the parameter sweep run will not execute and an error -message will be displayed. +Sweeps are intended to give quick feedback on a small input. Use the loader's +:code:`preview` parameter to select no more than seven sinogram slices; see +:ref:`previewing`. HTTomo reports an error and stops if the preview contains +more slices. What structure does the output data of a parameter sweep have? ============================================================== -When a parameter sweep is executed, the output of the method will be the set of -middle slices from each individual result of the sweep (sinogram slices or recon -slices), collected along the middle dimension. +The sweep output contains the middle slice from each run, concatenated along +the middle dimension. This applies to both sinogram and reconstruction slices. For example, suppose: -- the input data is previewed to 3 sinogram slices and has shape - :code:`(1801, 3, 2560)` +- the input contains three sinogram slices with shape + :code:`(1801, 3, 2560)`; and - a parameter sweep is performed on the :code:`center` parameter of a - reconstruction method in the pipeline, across 10 different CoR values - -In this case, each execution of the reconstruction method will produce 3 slices, -as an array of shape :code:`(2560, 3, 2560)`. So, 10 arrays of shape -:code:`(2560, 3, 2560)` will be produced. - -The middle slice from each array of 3 slices will be taken, resulting in 10 -reconstructed slices altogether. These 10 reconstructed slices will then be -concatenated along the middle dimension and put into a separate array, resulting -in the final data shape of :code:`(2560, 10, 2560)`. + reconstruction method using ten CoR values. -This output containing 10 slices will then be passed onto the next method in the -pipeline; for example, a method to save the 10 slices as images for quick -inspection. +Each reconstruction produces an array with shape :code:`(2560, 3, 2560)`. +HTTomo takes the middle slice from each of the ten arrays and concatenates them +into an output with shape :code:`(2560, 10, 2560)`. It passes this output to the +next pipeline method, such as one that saves the slices for inspection. diff --git a/docs/source/howto/httomo_features/previewing.rst b/docs/source/howto/httomo_features/previewing.rst deleted file mode 100644 index 03521e4b6..000000000 --- a/docs/source/howto/httomo_features/previewing.rst +++ /dev/null @@ -1,240 +0,0 @@ -.. default-role:: math -.. _previewing: - -Previewing -^^^^^^^^^^ - -Previewing is the way to change the dimensions of the input data by reducing them. -It also can be interpreted as a data cropping or data slicing operation. - -Reduction of the input data is often done to remove unnecessary/useless -information, and to accelerate the processing time. It is also recommended to use -when searching for optimal parameter values, see :ref:`parameter_sweeping`. Skip to -:ref:`previewing_enable` for information about how to use it in HTTomo. - -Previewing in the loader -======================== - -Previewing is an important part of the loader (see :ref:`reference_loaders`). Here, -a brief explanation is given on how to use the :code:`preview` parameter in the -:code:`standard_tomo` loader. - -.. note:: HTTomo assumes that the input data is a three dimensional (3D) array, - where the 1st axis is the *angular* dimension, the 2nd axis is the *vertical* - `Y`-detector dimension and the 3rd axis is the *horizontal* `X`-detector - dimension (see :numref:`fig_dimsdata`). - -.. _fig_dimsdata: -.. figure:: ../../_static/preview/dims_prev.svg - :scale: 55 % - :alt: 3D data - - 3D projection data and their axes - - -Structure of the :code:`preview` parameter value -================================================ - -The value of the :code:`preview` parameter has three fields, one for each axis in -the 3D input data, and also a :code:`start` and :code:`stop` field for each -dimension: - -.. code-block:: yaml - - preview: - angles: - start: - stop: - detector_y: - start: - stop: - detector_x: - start: - stop: - -.. warning:: Note that previewing in the :code:`angles` dimension is not yet - supported by loaders in HTTomo, but ignoring data along this dimension is a - feature that will be coming in a future release. - -Full data preview -================= - -If the :code:`preview` parameter is omitted entirely in the loader configuration, -then the full data will be selected and no cropping/previewing will be applied. Ie, -previewing is disabled in this case. - -.. _previewing_enable: - -Enabling data preview -===================== - -In order to change the input data dimensions and accelerate the processing -pipeline, one can do two of the following operations. - -.. note:: Although this is optional, by doing this the size of the reconstructed - volume is reduced without any detriment to the data, which can result in a - significant speedup in post-processing analysis time. - -In the figure below the projections have been cropped vertically and horizontally. - -Before cropping |pic1| and after |pic2| - -.. |pic1| image:: ../../_static/preview/uncropped.gif - :width: 44% - -.. |pic2| image:: ../../_static/preview/cropped.gif - :width: 27% - - -1. Reduce the size of the vertical dimension (detector- `Y`) by removing blank regions in your data (top and bottom cropping), - see :numref:`fig_dimsdataY`. The blank areas, if any, can be established by looking through the sequence of raw projections. - - .. code-block:: yaml - - preview: - detector_y: - start: 200 - stop: 1800 - - This will crop the data starting at slice 200 and finishing at slice 1800, - therefore resulting in the data with the vertical dimension equal to 1600 pixels. - In Python this will be interpreted as :code:`[:,200:1800,:]`. - -.. _fig_dimsdataY: -.. figure:: ../../_static/preview/dims_prevY.svg - :scale: 55 % - :alt: 3D data, Y slicing - - Cropping detector- `Y` dimension of 3D projection data - -2. Reduce the size of the horizontal dimension (detector- `X`) by removing blank regions in your data (cropping the left and right sides), - see :numref:`fig_dimsdataX`. - - .. warning:: - Please be aware that cropping this dimension can create issues with the automatic centering - and potentially lead to reconstruction artefacts, especially if iterative methods are used. - It is general practice to be more conservative with the cropping of the `X` - detector dimension. - - .. code-block:: yaml - - preview: - detector_x: - start: 100 - stop: 2000 - - In Python this will be interpreted as :code:`[:,:,100:2000]`. - -.. _fig_dimsdataX: -.. figure:: ../../_static/preview/dims_prevX.svg - :scale: 55 % - :alt: 3D data, X slicing - - Cropping detector- `X` dimension of 3D projection data - -One can combine vertical and horizontal cropping with: - -.. code-block:: yaml - - preview: - detector_y: - start: 200 - stop: 1800 - detector_x: - start: 100 - stop: 2000 - -Using :code:`begin`, :code:`mid` and :code:`end` keywords with offsets -====================================================================== - -The :code:`begin, mid, end` keywords can be used in setting up :code:`start, stop` preview parameters. The main purpose of those keywords is -to help a user to set the preview without using actual data indices. It can be a convenient and a quick way to set the preview without any prior -knowledge about the input data sizes. -To unlock the full potential of this feature, we recommend using the :code:`begin, mid, end` keywords together with the *offset* parameters. -The preview parameters :code:`start_offset` and :code:`stop_offset` can be used -to compliment :code:`start, stop` preview parameters respectively. - -Here is an example how the keywords and offsets can be used in the loader's preview: - -.. code-block:: yaml - - preview: - detector_x: - start: begin - start_offset : 100 - stop: end - stop_offset : -100 - detector_y: - start: mid - start_offset : -50 - stop: mid - stop_offset : 50 - -In the example above, we crop both of data dimensions :code:`detector_x` and :code:`detector_y`. With :code:`detector_x`, the data is cropped with +100 pixels offset from the first index (:code:`begin`) and -100 pixels offset from the last index :code:`end`. -In Python this would be an equivalent to perform the following slicing of that dimension :code:`[100:-100]`. And with :code:`detector_y` dimension, we slice by taking -a 100 pixels wide chunk centered around the middle (:code:`mid`) index of that dimension. - -.. note:: The :code:`begin` parameter defines the first index of the chosen dimension, :code:`mid` defines the middle index, and :code:`end` defines the last index. - - -The sole use of :code:`mid` value -================================= - -The :code:`detector_y` and :code:`detector_x` dimension fields also support the -value :code:`mid` in addition to the :code:`start` and/or :code:`stop` fields. For instance, -one can extract the the middle slice of :code:`detector_y` with: - -.. code-block:: yaml - - preview: - detector_y: - mid - -Specifying :code:`mid` for either of these dimensions will result in the middle -three slices of that dimension being selected. - -.. warning:: The :code:`angles` dimension field doesn't support the value - :code:`mid` - -Rules for omitting fields in the :code:`preview` parameter value -================================================================ - -One may have noticed that, in many of the :code:`preview` parameter value examples -above, some fields were omitted. It's infrequently needed to crop all three -dimensions, and sometimes when cropping, only either the start or end is of -interest. - -With these in mind, along the general notion that anything is more readable -when unnecessary information is omitted, there are several ways in which the - -- dimension fields -- start/stop fields - -in the :code:`preview` parameter value can be omitted in the process list, and -still achieve the desired cropping behavior. - -Omitting one or more dimension fields -------------------------------------- - -If any of the three top-level dimension fields are omitted, then no cropping will -be applied to the omitted dimension(s). - -If a top-level dimension is provided but given no value, then no cropping will be -applied to that dimension either. Ie, the following configuration will select the -entire input data and apply no cropping/previewing: - -.. code-block:: yaml - - preview: - angles: - detector_y: - detector_x: - -Omitting the :code:`start` or :code:`stop` fields -------------------------------------------------- - -For a given dimension field: - -- if the :code:`start` field is omitted, then the start value is assumed to be 0 -- if the :code:`stop` field is omitted, then the stop value is assumed to be the - very last element in that dimension diff --git a/docs/source/howto/httomo_features/save_results.rst b/docs/source/howto/httomo_features/save_results.rst new file mode 100644 index 000000000..5916460a3 --- /dev/null +++ b/docs/source/howto/httomo_features/save_results.rst @@ -0,0 +1,41 @@ +.. _save-result-examples: + +Save intermediate datasets +++++++++++++++++++++++++++ + +Use the top-level ``save_result`` field to control whether a method's processed +dataset is written to an intermediate HDF5 file. Its value is a YAML Boolean: +``true`` or ``false`` without quotation marks. + +.. warning:: + + Place ``save_result`` beside ``method``, ``module_path`` and ``parameters``, + not inside the ``parameters`` mapping. + +For example, save the result of ``dark_flat_field_correction`` as follows: + +.. code-block:: yaml + :emphasize-lines: 6 + + - method: dark_flat_field_correction + module_path: httomolibgpu.prep.normalize + parameters: + flats_multiplier: 1.0 + darks_multiplier: 1.0 + save_result: true + +If ``save_result`` is omitted, the method's default setting is used. To save +intermediate datasets for every applicable task, run HTTomo with +``--save-all``. When that option is enabled, ``save_result: false`` *does not +disable* saving for an individual method. + +.. note:: + + Saving intermediate datasets can substantially increase disk use and + execution time, so enable it only for results that need to be inspected or + reused. Reconstruction results are saved to an intermediate file by + default. + +For other output choices, use a ``save_to_images`` method to write images or +``--save-snapshots`` to capture lightweight diagnostic snapshots. See +:ref:`reference_templates` and :ref:`run-httomo-indepth` respectively. diff --git a/docs/source/howto/httomo_features/side_out.rst b/docs/source/howto/httomo_features/side_out.rst new file mode 100644 index 000000000..537330654 --- /dev/null +++ b/docs/source/howto/httomo_features/side_out.rst @@ -0,0 +1,62 @@ +.. _side_output: + +Side outputs +++++++++++++ + +Methods normally pass their processed dataset to the next method. Some methods +also produce supplementary values, called *side outputs*, which can be used as +parameters by methods later in the pipeline. A common example is passing a +calculated centre of rotation to a reconstruction method. + +.. figure:: ../../_static/side_output_reference.svg + :width: 100% + :alt: Main dataset and side-output flows between two pipeline methods + + The processed dataset follows the main pipeline, while the named side output + is passed to a later method parameter. + +Define and reference a side output +################################## + +The producing method requires: + +* a unique ``id``; +* a ``side_outputs`` mapping from the method's output name to a pipeline name. + +The consuming method refers to the value using +``${{id.side_outputs.name}}``: + +.. code-block:: yaml + :emphasize-lines: 11-13,18 + + - method: find_center_vo + module_path: httomolibgpu.recon.rotation + parameters: + ind: null + smin: -50 + smax: 50 + srad: 6.0 + step: 0.25 + ratio: 0.5 + drop: 20 + id: centering + side_outputs: + cor: centre_of_rotation + + - method: FBP3d_tomobar + module_path: httomolibgpu.recon.algorithm + parameters: + center: ${{centering.side_outputs.centre_of_rotation}} + filter_freq_cutoff: 1.0 + recon_size: null + recon_mask_radius: 0.95 + +The referenced method must appear *before* the method that uses its output, and +each explicit ``id`` must be *unique* within the pipeline. + +.. note:: + + Side outputs and their references are generated automatically by the + `YAML generator + `_. + They normally do not need to be changed when adapting a generated pipeline. diff --git a/docs/source/howto/installation.rst b/docs/source/howto/installation.rst index 63941e3d5..b0e68eac1 100644 --- a/docs/source/howto/installation.rst +++ b/docs/source/howto/installation.rst @@ -1,63 +1,130 @@ .. _installation_main: -Installation Guide -****************** +Installation +************ + +HTTomo is available from PyPI. We recommend installing it in a Conda +environment because HTTomo depends on MPI and parallel HDF5; GPU installations +also require compatible CUDA libraries. A Python virtual environment can be +used when these system dependencies are already available. + +.. note:: + + The primary recipe assumes Linux and a CUDA-compatible GPU. A Linux CPU-only + recipe is also provided. For Windows or macOS, see + :ref:`installation_other`. + +Choose an installation path +=========================== + +.. list-table:: + :header-rows: 1 + :widths: 22 28 50 + + * - Platform + - Processing support + - Recommended path + * - Linux with NVIDIA GPU + - CPU and CUDA GPU methods + - Use the Conda environment below. + * - Linux without a GPU + - CPU methods + - Use the CPU-only Conda environment below. + * - Windows + - CPU and supported NVIDIA GPUs + - Install Linux under WSL 2, then follow the Linux instructions. + * - macOS on Apple Silicon + - CPU methods only + - Follow :ref:`installation_mac`. + +The commands below use Python 3.12, NumPy 2.4, CuPy 14.2 and OpenMPI 4.1.6. +TomoPy 1.15.3 is optional unless the selected pipeline uses TomoPy. See +:ref:`compatibility` before changing these versions. + + +Conda environment with GPU support +================================== -HTTomo is available on PyPI, so it can be installed into either a virtual environment or a -conda environment. - -However, there are certain constraints under which a virtual environment can be used, due to -the dependence on an MPI implementation, the hdf5 library, CUDA libraries, and whether the user -requires using :code:`tomopy` methods in pipelines. - -.. note:: - These instructions assume a Linux OS with a CUDA-compatible GPU. - If you are using Windows or macOS, see :ref:`installation_other` for platform-specific guidance. +.. code-block:: console + $ conda create --name httomo --channel conda-forge \ + python=3.12 "numpy==2.4.*" "cupy==14.2.*" \ + openmpi==4.1.6 mpi4py "h5py[build=*openmpi*]" \ + astra-toolbox aiofiles click graypy loguru nvtx pillow pyyaml \ + scikit-image scipy tqdm hdf5plugin pip pywavelets + $ conda activate httomo + $ conda install --channel conda-forge tomopy==1.15.3 # Optional + $ pip install --no-deps \ + httomo httomo-backends httomolib httomolibgpu tomobar -Conda environment -================= -By default the :code:`cupy` installation will install the latest :code:`cuda-cudart`. This can result in CUDA versions higher than the supported by the GPU device of the system. One can specify the compatible to their system CUDA package, e.g., :code:`cuda-cudart==12.9.79`. +.. note:: -.. code-block:: console + Conda may select a ``cuda-cudart`` version that is newer than the installed + NVIDIA driver supports. If necessary, add a compatible CUDA runtime to the + create command, for example ``cuda-cudart==12.9.79``. - $ conda create --name httomo - $ conda activate httomo - $ conda install -c conda-forge cupy==14.2 openmpi==4.1.6 h5py[build=*openmpi*] python numpy astra-toolbox aiofiles click graypy loguru nvtx pillow pyyaml scikit-image scipy tqdm hdf5plugin pip pywavelets - $ conda install -c conda-forge tomopy==1.15.3 # optional - $ pip install httomo httomo-backends httomolib httomolibgpu tomobar --no-deps +.. _installation_cpu_only: -Setup HTTomo development environment: -====================================================== +CPU-only Conda environment +========================== -Development mode requires git cloning the HTTomo's repository and pip installing from the source as below. Note that all other dependencies, apart from :code:`httomo`, must be satisfied as above. +Use this environment on systems without a CUDA-capable GPU. It omits CuPy, +HTTomolibGPU and TomoBAR but retains MPI and parallel HDF5: .. code-block:: console - $ pip install -e .[dev] # development mode + $ conda create --name httomo-cpu --channel conda-forge \ + python=3.12 "numpy==2.4.*" openmpi==4.1.6 mpi4py \ + "h5py[build=*openmpi*]" astra-toolbox aiofiles click graypy loguru \ + pillow pyyaml scikit-image scipy tqdm hdf5plugin pip pywavelets + $ conda activate httomo-cpu + $ conda install --channel conda-forge tomopy==1.15.3 + $ pip install --no-deps httomo httomo-backends httomolib + Virtual environment =================== -A virtual environment can be used if the following conditions are met: +A Python virtual environment can be used when: + +- an MPI implementation, such as OpenMPI, is installed; +- a parallel build of HDF5 is installed; +- the required CUDA libraries or CUDA Toolkit are installed; and +- TomoPy methods are not required in HTTomo pipelines. -- an MPI implementation is installed on the system (ie, OpenMPI) -- the hdf5 library is installed on the system -- CUDA libraries or CUDA toolkit are installed on the system -- methods from :code:`tomopy` are not required to be used in pipelines +The exact installation commands depend on the locally installed MPI, HDF5, +CUDA, and NVIDIA driver versions. .. code-block:: console - $ python -m venv httomo + $ python3.12 -m venv httomo $ source httomo/bin/activate - $ MPICC=$(type -p mpicc) pip install mpi4py==3.1.6 - $ pip install cython numpy pkgconfig setuptools # build dependencies of h5py - $ CC=$(type -p mpicc) HDF5_MPI="ON" HDF5_DIR=/path/to/parallel-hdf5 pip install --no-build-isolation --no-binary=h5py h5py - $ pip install cupy-cuda14x # install cupy-cuda14x if CUDA library/CUDA toolkit version is 14.x - $ pip install aiofiles astra-toolbox click graypy hdf5plugin loguru nvtx pillow pyyaml scikit-image scipy tqdm + $ MPICC=$(type -p mpicc) pip install mpi4py + $ pip install cython "numpy==2.4.*" pkgconfig setuptools # h5py build dependencies + $ CC=$(type -p mpicc) HDF5_MPI="ON" \ + HDF5_DIR=/path/to/parallel-hdf5 \ + pip install --no-build-isolation --no-binary=h5py h5py + $ pip install "cupy-cuda14x==14.2.*" # For a CUDA 14.x runtime + $ pip install aiofiles astra-toolbox click graypy hdf5plugin loguru \ + nvtx pillow pyyaml scikit-image scipy tqdm $ pip install --no-deps httomo httomolib httomolibgpu httomo-backends tomobar +Verify the installation +======================= + +Check the command-line entry point and confirm that h5py has parallel HDF5 +support: + +.. code-block:: console + + $ python -m httomo --version + $ python -m httomo --help + $ python -c "import h5py; print('Parallel HDF5:', h5py.get_config().mpi)" + +The final command must print ``Parallel HDF5: True``. Developers working from +a source checkout should instead follow :ref:`developer_setup`. + .. _installation_other: Installation on Other Platforms @@ -68,13 +135,3 @@ Installation on Other Platforms installation_variants/installation_windows installation_variants/installation_mac - -.. _running_tests: - -Run tests (optional) -==================== - -.. toctree:: - :maxdepth: 2 - - running_tests diff --git a/docs/source/howto/installation_variants/installation_mac.rst b/docs/source/howto/installation_variants/installation_mac.rst index db7627dea..d34a66c3d 100644 --- a/docs/source/howto/installation_variants/installation_mac.rst +++ b/docs/source/howto/installation_variants/installation_mac.rst @@ -4,6 +4,7 @@ macOS (Apple Silicon) ********************* .. note:: + HTTomo's GPU-accelerated methods (``httomolibgpu``) depend on `CuPy `_, which requires an NVIDIA CUDA GPU. Apple Silicon Macs (M1/M2/M3/M4) have no CUDA support, so this path installs HTTomo in @@ -29,75 +30,57 @@ everything below runs emulated under Rosetta and is significantly slower: 2. Create the environment -Skip ``cupy`` entirely — there is no arm64/macOS build, and it cannot be -installed on Apple Silicon. ``astra-toolbox`` and ``tomopy`` do have -osx-arm64 conda-forge builds (CPU-only algorithms), which is all a CPU -pipeline needs. ``mpi4py`` is required even for a single-process run, since -HTTomo's CLI unconditionally imports ``mpi4py.MPI``. -Replace `conda` with `mamba` below if it's available in the environment, for a faster package resolution. - -.. code-block:: bash - - conda create --name httomo python=3.11 - conda activate httomo - - # numpy must stay below 2.0 — HTTomo's CPU/GPU array-type detection - # (block.is_gpu) relies on numpy.ndarray *not* having a `.device` - # attribute, an assumption NumPy 2.0's Array API support breaks. - conda install -c conda-forge "numpy<2" mpi4py openmpi==4.1.6 \ - "h5py=*=mpi_openmpi*" astra-toolbox tomopy==1.15.3 \ - aiofiles click graypy loguru nvtx pillow pyyaml \ - scikit-image scipy tqdm hdf5plugin pywavelets - - # compilers needed to build httomolib's OpenMP-based C extension — - # macOS's system clang has no -fopenmp support - conda install -c conda-forge compilers llvm-openmp - -3. Install HTTomo - -.. code-block:: bash - - pip install httomo httomo-backends httomolib tomobar --no-deps - -Verify: - -.. code-block:: bash - - python -m httomo --help +HTTomo requires Python 3.12 or later and NumPy 2.4. CuPy and +``httomolibgpu`` are not installed because they require an NVIDIA CUDA GPU. -4. Known issues on this path (as of httomo 3.0 / httomolib 4.0.1) +``mpi4py`` is required even for a single-process run because HTTomo imports +``mpi4py.MPI`` when its command-line interface starts. -If ``h5py`` ever gets silently swapped back to a non-MPI build by a later ``conda install`` (check with -``python -c "import h5py; print(h5py.get_config().mpi)"``), pin it: +.. code-block:: console - .. code-block:: bash + $ conda create --name httomo --channel conda-forge \ + python=3.12 "numpy==2.4.*" \ + mpi4py openmpi==4.1.6 "h5py=*=mpi_openmpi*" \ + tomopy==1.15.3 astra-toolbox \ + aiofiles click graypy loguru nvtx pillow pyyaml \ + scikit-image scipy tqdm hdf5plugin pywavelets \ + compilers llvm-openmp pip + $ conda activate httomo - conda config --env --append pinned_packages 'h5py=*=mpi_openmpi*' +NumPy 2.x is required by the current HTTomo implementation. In particular, +HTTomo uses the ``numpy.ndarray.device`` attribute introduced in NumPy 2.0 to +identify CPU arrays. -5. Optional step. :ref:`run_tests` to make sure that everything works correctly. +The ``compilers`` and ``llvm-openmp`` packages are needed when building +HTTomolib's OpenMP-based extension because the system Clang compiler supplied +by macOS does not provide OpenMP support by default. -6. Running a CPU pipeline +3. Install HTTomo -Always validate the pipeline first: +Install only the CPU backend packages. ``--no-deps`` is required because the +published package metadata currently includes CUDA-only dependencies that +cannot be installed on Apple Silicon. -.. code-block:: bash +.. code-block:: console - python -m httomo check pipeline.yaml data.nxs + $ python -m pip install --no-deps \ + httomo httomo-backends httomolib -Run serially: +Do not install ``httomolibgpu`` or ``tomobar`` in this environment. Both are +GPU-oriented packages with CUDA dependencies. -.. code-block:: bash +4. Verify the installation - python -m httomo run data.nxs pipeline.yaml ./output --max-memory 10G +Confirm the Python and NumPy versions and verify that parallel HDF5 is enabled: -Or across multiple CPU cores with MPI (``--max-memory`` is per-process, -so divide your budget by the process count): +.. code-block:: console -.. code-block:: bash + $ python -c "import sys, numpy; print(sys.version); print(numpy.__version__)" + $ python -c "import h5py; print('Parallel HDF5:', h5py.get_config().mpi)" + $ python -m httomo --help - mpirun -np 4 python -m httomo run data.nxs pipeline.yaml ./output --max-memory 2G +The first command should report Python 3.12 or later and NumPy 2.4.x. The +second command should print ``Parallel HDF5: True``. -.. note:: - With 16GB of unified memory shared with macOS itself, keep - ``--max-memory`` well under the physical total (8–10G total budget is a - safe starting point) to avoid swapping. +Developers who need to run the source test suite should follow +:ref:`developer_setup`. diff --git a/docs/source/howto/installation_variants/installation_windows.rst b/docs/source/howto/installation_variants/installation_windows.rst index 83aa4dad8..95196050e 100644 --- a/docs/source/howto/installation_variants/installation_windows.rst +++ b/docs/source/howto/installation_variants/installation_windows.rst @@ -3,37 +3,99 @@ Windows ******* -Although the libraries (backends) can be installed on Windows natively, the HTTomo framework requires Linux. For Windows 10 and newer one can install Linux through `WSL `_ and run HTTomo there. -WSL supports also CUDA-compatible GPU devices, so the GPU methods will also be working smoothly. +HTTomo requires Linux and cannot run directly on Windows. On supported +versions of Windows, HTTomo can be run using Windows Subsystem for Linux 2 +(WSL 2). + +These instructions require Windows 10 version 2004 (build 19041) or later, or +Windows 11. GPU acceleration additionally requires a supported NVIDIA GPU and +an NVIDIA Windows driver that supports CUDA in WSL. + +See the following documentation before continuing: + +- `Install WSL `_ +- `Enable NVIDIA CUDA on WSL + `_ Installation steps ================== -Steps 1-3 are following the official WSL installation provided `here `_. +1. Open PowerShell or Windows Terminal as an administrator. + +2. Install WSL and its default Ubuntu distribution: + + .. code-block:: powershell + + wsl --install + +3. Restart Windows when prompted. Then open Ubuntu from the Start menu and + complete the initial Linux user setup. + +4. Confirm that the distribution is using WSL 2: + + .. code-block:: powershell + + wsl --list --verbose + + If necessary, replace ``Ubuntu`` below with the distribution name shown by + the preceding command: + + .. code-block:: powershell + + wsl --set-version Ubuntu 2 + +5. Inside the WSL terminal, update the package index and install the required + build tools: + + .. code-block:: console + + $ sudo apt update + $ sudo apt install build-essential wget + +6. Download and install Miniforge inside WSL: + + .. code-block:: console + + $ wget https://github.com/conda-forge/miniforge/releases/latest/download/Miniforge3-Linux-x86_64.sh + $ bash Miniforge3-Linux-x86_64.sh + $ source ~/.bashrc + + The installer shown above is for x86-64 systems. Select a different + `Miniforge installer + `_ + when using another architecture. + +7. Follow the Conda instructions in :ref:`installation_main` to create an + environment and install HTTomo. + +8. Run the verification commands in :ref:`installation_main`. + +.. dropdown:: Troubleshooting: WSL has no network connection -1. Open Terminal as Admin: Right-click Start, select "Windows PowerShell (Admin)" or "Terminal (Admin)". -2. Run Install Command: Type :code:`wsl --install` and hit Enter. Reboot PC. -3. Open Promt and type :code:`wsl` to start WSL. + Follow Microsoft's + `WSL troubleshooting guidance + `_. -.. dropdown:: Troubleshooting: No Internet connection in WSL. +.. dropdown:: Troubleshooting: A compiler is missing - Perform the steps as described in `here `_. + Install the standard Ubuntu build tools: -4. Get the latest Miniforge for Linux using this `link `_. -5. Get into the **base** environment of conda by :code:`source /path/to/.bashrc`. -6. Install HTTomo and dependencies by following the :ref:`installation_main` notes for Linux. -7. Optional step. :ref:`run_tests` to make sure that everything works correctly. + .. code-block:: console -.. dropdown:: Troubleshooting: HTTomoLib library requires a :code:`gcc` compiler. + $ sudo apt update + $ sudo apt install build-essential - Install the :code:`gcc` compiler with :code:`sudo apt-get install gcc`. +.. dropdown:: Troubleshooting: HTTomo cannot use the GPU -.. dropdown:: Troubleshooting: HTTomo doesn't run on a GPU with an older architecture. + First confirm that the GPU is visible inside WSL: - If there is a GPU with an older GPU architecture then try: + .. code-block:: console - a. :code:`conda install -c conda-forge cupy==12.3.6 openmpi==4.1.6 h5py[build=*openmpi*] python>=3.10 numpy astra-toolbox aiofiles click graypy loguru nvtx pillow pyyaml scikit-image scipy tqdm hdf5plugin pip pywavelets` + $ nvidia-smi - b. :code:`pip install tomobar httomolib httomolibgpu httomo-backends --no deps` + If this command fails, update the NVIDIA driver installed on Windows and + follow Microsoft's + `CUDA on WSL guidance + `_. - c. Go to :code:`numpy1` branch on the cloned HTTomo repository and run :code:`pip install . --no-deps` to install a compatible older version. \ No newline at end of file + Do not install a Linux NVIDIA display driver inside WSL. diff --git a/docs/source/howto/interpret_logger.rst b/docs/source/howto/interpret_logger.rst index 4175662f7..c954c06f0 100644 --- a/docs/source/howto/interpret_logger.rst +++ b/docs/source/howto/interpret_logger.rst @@ -1,43 +1,10 @@ -.. _info_logger: +:orphan: -Interpret Log File -====================== +Run output reference moved +========================== -This section contains information on how to interpret the log file created by HTTomo. +The output and logging reference is now at :ref:`info_logger`. -HTTomo uses :code:`loguru` software to unify and simplify the logging system. During the job execution, the concise information -goes to the terminal (see :numref:`fig_log`) and also to the :code:`user.log` file. More verbose information, that is usually -needed to debug the run, is saved into the :code:`debug.log` file. Let us explain few main elements of the :code:`user.log` -file and also stdout. +.. raw:: html -.. _fig_log: -.. figure:: ../_static/log/log_screenshot.png - :scale: 40 % - :alt: HTTomo log screenshot - - The screenshot of the terminal output (AKA stdout) which also goes into the :code:`user.log` file. - - -* :code:`Pipeline has been separated into N sections` - This means that `N` :ref:`info_sections` created for this pipeline and each section contains a certain amount of methods grouped together to work on :ref:`blocks_data`. The progress can be seen in every - section processing all of the input data divided into :ref:`chunks_data` and :ref:`blocks_data`, before continue to the next section. - -* :code:`Running loader` - The loader does not belong to sections and always at the start of the pipeline. Note that the loader - loads the data using the specific :code:`pattern=projection` (See more :ref:`info_reslice`). The same pattern is used by the - following section. - -* :code:`Section N with the following methods` - Each section contains a number of methods that run sequentially for each :ref:`blocks_data` - of data. When all blocks are processed, the user will see the message :code:`Finished processing the last block`. This means that all of the - input data have been processed in this section and the pipeline moves to the next section, if it exists. - -* :code:`50%|##### | 1/2 [00:02<00:02, 2.52s/block]` - These are the progress bars showing how much data is being processed in every section. - The percentage progress bar demonstrates how many blocks have been processed by the `M` number of methods of the current section. Specifically in this case - we have :code:`1/2`, which means that one of two blocks completed (hence `50%`). Then :code:`00:02<00:02` shows the time in seconds to - reach the current block (time elapsed) and the remaining time to complete all iterations over blocks. The :code:`2.52s/block` part is an - estimation of how much time it's taking per block. When the time per block is less than one second then this can be presented as :code:`block/s` instead. - See :code:`save_to_images` progress report, for instance. - -.. note:: When interpreting progress bars, one possible misunderstanding can be an association of the progress with the methods completed. Because each piece of data (a block) can be processed by multiple methods, we report on how many blocks have been processed instead. \ No newline at end of file + diff --git a/docs/source/howto/loading_data.rst b/docs/source/howto/loading_data.rst new file mode 100644 index 000000000..a1b878383 --- /dev/null +++ b/docs/source/howto/loading_data.rst @@ -0,0 +1,20 @@ +.. _reference_loaders: +.. _loading_data: + +Loading data +************ + +HTTomo's standard tomography loader reads projection data, darks, flats, and +rotation angles from HDF5/NeXus files. It can discover NXtomo datasets +automatically, combine data from separate files, crop the input, and select an +individual scan from a continuous acquisition. + +.. toctree:: + :maxdepth: 2 + + loading_data/standard_loader + loading_data/darks_flats + loading_data/rotation_angles + loading_data/previewing + loading_data/continuous_scan_subset + loading_data/create_nxtomo diff --git a/docs/source/howto/loading_data/continuous_scan_subset.rst b/docs/source/howto/loading_data/continuous_scan_subset.rst new file mode 100644 index 000000000..dd43c8178 --- /dev/null +++ b/docs/source/howto/loading_data/continuous_scan_subset.rst @@ -0,0 +1,21 @@ +.. _continuous_scan_subset_selection: + +Continuous-scan subsets +^^^^^^^^^^^^^^^^^^^^^^^ + +A single 3D HDF5 dataset can contain several tomography scans arranged along +the angular dimension. Use :code:`continuous_scan_subset` to load one scan by +specifying its start and stop indices. As with Python slicing, the stop index is +excluded. + +This example selects indices 90 through 179: + +.. literalinclude:: ../../../../tests/samples/pipeline_template_examples/testing/loader_with_offset_param.yaml + :language: yaml + :emphasize-lines: 7-9 + +The option can be combined with :ref:`previewing` to crop the detector +dimensions and with :ref:`darks_flats` to load external darks or flats. +Its start and stop values replace ``preview.angles``. If +``--continuous-scan-subset`` is also supplied on the command line, the +command-line values replace the values in the pipeline. diff --git a/docs/source/howto/loading_data/create_nxtomo.rst b/docs/source/howto/loading_data/create_nxtomo.rst new file mode 100644 index 000000000..f8d295131 --- /dev/null +++ b/docs/source/howto/loading_data/create_nxtomo.rst @@ -0,0 +1,178 @@ +.. _create_nxtomo: + +Creating an NXtomo file +^^^^^^^^^^^^^^^^^^^^^^^ + +HTTomo can automatically find tomography data in a `NeXus NXtomo file +`_. This +tutorial shows how to pack either NumPy arrays or stacks of TIFF images into +that format. + +:download:`Download the complete create_nxtomo.py script +<../../scripts/create_nxtomo.py>` before following the examples. The script is +self-contained and may also be imported as a Python module. + +Install the packages used by the script in the environment where the packing +will run: + +.. code-block:: console + + python -m pip install numpy h5py tifffile + +``tifffile`` is only needed for the TIFF example. + +Input layout +============ + +Projection, flat-field, and dark-field data must have the axis order +``(frames, detector_y, detector_x)``. For projections, ``frames`` is the +rotation-angle axis. The angle array must be one-dimensional, measured in +degrees, and contain exactly one value per projection. + +A single flat or dark image may be supplied as a two-dimensional array. The +script adds its frame axis automatically. Flats and darks are optional, but +operations that perform flat/dark correction need meaningful calibration +images. + +From NumPy arrays +================= + +Import :func:`write_nxtomo` from the downloaded script. Arrays can come from +``numpy.load``, another Python library, or calculations performed in the same +program: + +.. code-block:: python + + import numpy as np + + from create_nxtomo import write_nxtomo + + projections = np.load("projections.npy") # (n_angles, detector_y, detector_x) + angles = np.load("angles.npy") # (n_angles,), in degrees + flats = np.load("flats.npy") # (n_flats, detector_y, detector_x) + darks = np.load("darks.npy") # (n_darks, detector_y, detector_x) + + write_nxtomo( + "scan.nxs", + projections, + angles, + flats=flats, + darks=darks, + sample_name="my sample", + compression="gzip", + ) + +Use ``overwrite=True`` only when an existing output file should be replaced. +Set ``compression=None`` for faster writing and a larger file, or +``compression="lzf"`` for lightweight compression. The default is +``compression="gzip"``. + +From TIFF stacks +================ + +Keep each image type in its own directory, for example: + +.. code-block:: text + + scan/ + ├── projections/ + │ ├── projection_0000.tif + │ ├── projection_0001.tif + │ └── ... + ├── flats/ + │ ├── flat_0000.tif + │ └── ... + ├── darks/ + │ ├── dark_0000.tif + │ └── ... + └── angles.npy + +Then run the script. Quote each glob so that the script, rather than the +shell, receives and sorts the complete file list: + +.. code-block:: console + + python create_nxtomo.py scan.nxs \ + --projections "scan/projections/*.tif" \ + --angles scan/angles.npy \ + --flats "scan/flats/*.tif" \ + --darks "scan/darks/*.tif" \ + --sample-name "my sample" + +The angles may instead be stored as a one-column ``.txt`` file or a +comma-separated ``.csv`` file. Omit ``--flats`` or ``--darks`` when that image +type is unavailable. Use ``--overwrite`` to replace an existing output file. + +The TIFF loader sorts filenames naturally, so ``projection_2.tif`` precedes +``projection_10.tif``. The filenames still need to encode the acquisition +order correctly. Every matched TIFF must be a single two-dimensional grayscale +image, and all images must have the same height and width. + +What the script writes +====================== + +The output stores darks, then flats, then projections in one three-dimensional +detector dataset. The ``image_key`` gives every frame its NXtomo meaning: + +* ``0`` -- projection; +* ``1`` -- flat field; and +* ``2`` -- dark field. + +Calibration frames are assigned the first projection angle because NXtomo +requires one rotation angle per frame. HTTomo uses ``image_key`` to select only +the projection frames and their corresponding angles. + +The important part of the resulting file is: + +.. code-block:: text + + /entry NXentry + ├── definition "NXtomo" + ├── instrument NXinstrument + │ └── detector NXdetector + │ ├── data (frames, detector_y, detector_x) + │ └── image_key (frames,) + ├── sample NXsample + │ └── rotation_angle (frames,), units="deg" + └── data NXdata + ├── data link to detector/data + ├── image_key link to detector/image_key + └── rotation_angle link to sample/rotation_angle + +The links in ``/entry/data`` do not duplicate the arrays. They expose the +standard NXtomo paths that HTTomo's automatic discovery uses. + +Loading the result in HTTomo +============================ + +Use the standard tomography loader and set all discoverable paths to +``auto``: + +.. code-block:: yaml + + - method: standard_tomo + module_path: httomo.data.hdf.loaders + parameters: + data_path: auto + image_key_path: auto + rotation_angles: auto + +Pass ``scan.nxs`` as the input data file when running HTTomo. If the file has +no flats or darks, HTTomo supplies dummy calibration arrays. Alternatively, +set ``flats: ignore`` or ``darks: ignore`` explicitly when the corresponding +correction should not use stored calibration images; see :ref:`darks_flats`. + +Reusable NeXus writer code +========================== + +The core NumPy writer is included below for reference. The downloadable script +also contains TIFF loading, validation, filename sorting, and its command-line +interface. + +.. dropdown:: Reusable NeXus writer code + + .. literalinclude:: ../../scripts/create_nxtomo.py + :language: python + :start-after: # [write-nxtomo-start] + :end-before: # [write-nxtomo-end] + :linenos: diff --git a/docs/source/howto/loading_data/darks_flats.rst b/docs/source/howto/loading_data/darks_flats.rst new file mode 100644 index 000000000..d79c32b62 --- /dev/null +++ b/docs/source/howto/loading_data/darks_flats.rst @@ -0,0 +1,94 @@ +.. _darks_flats: + +Darks and flats +^^^^^^^^^^^^^^^ + +The standard loader supports dark and flat images that are: + +* stored with the projections; +* stored in separate files or datasets; +* absent; or +* present but intentionally ignored. + +Stored with the projections +=========================== + +When projections, darks, and flats share one dataset, use +:code:`image_key_path` to identify their image types. This is the standard +configuration shown in :doc:`standard_loader`. + +Separate datasets without image keys +===================================== + +If a separate dataset contains only darks or only flats, specify its +:code:`file` and :code:`data_path`: + +.. code-block:: yaml + :emphasize-lines: 5,6,8,9 + + - method: standard_tomo + module_path: httomo.data.hdf.loaders + parameters: + darks: + file: path/to/darks.nxs + data_path: /entry1/tomo_entry/data/data + flats: + file: path/to/flats.nxs + data_path: /entry1/tomo_entry/data/data + +Use :code:`input_data` when the separate datasets are in the main input file: + +.. code-block:: yaml + :emphasize-lines: 5,8 + + - method: standard_tomo + module_path: httomo.data.hdf.loaders + parameters: + darks: + file: input_data + data_path: /exchange/darks + flats: + file: input_data + data_path: /exchange/flats + +Separate data with image keys +============================= + +If a specified dataset contains several image types, also provide its +:code:`image_key_path`: + +.. code-block:: yaml + :emphasize-lines: 7,11 + + - method: standard_tomo + module_path: httomo.data.hdf.loaders + parameters: + darks: + file: path/to/darks.nxs + data_path: /entry1/tomo_entry/data/data + image_key_path: /entry1/tomo_entry/instrument/detector/image_key + flats: + file: path/to/flats.nxs + data_path: /entry1/tomo_entry/data/data + image_key_path: /entry1/tomo_entry/instrument/detector/image_key + +Missing darks or flats +====================== + +No additional configuration is required when the data does not contain darks +or flats. HTTomo handles their absence automatically. + +Ignoring darks or flats +======================= + +Set either parameter to :code:`ignore` to exclude images that are present in +the dataset: + +.. code-block:: yaml + :emphasize-lines: 4,5 + + - method: standard_tomo + module_path: httomo.data.hdf.loaders + parameters: + darks: ignore + flats: ignore diff --git a/docs/source/howto/loading_data/previewing.rst b/docs/source/howto/loading_data/previewing.rst new file mode 100644 index 000000000..6f48471aa --- /dev/null +++ b/docs/source/howto/loading_data/previewing.rst @@ -0,0 +1,218 @@ +.. default-role:: math +.. _previewing: + +Previewing +^^^^^^^^^^ + +Previewing crops, or slices, the input data. It can remove unused regions and +reduce processing time, particularly while :ref:`sweeping parameters +`. See :ref:`previewing_enable` to start configuring it. + +Previewing in the loader +======================== + +The :doc:`standard_loader` provides a :code:`preview` parameter for selecting +part of the input data. + +.. note:: HTTomo assumes a three-dimensional array whose axes are the *angular* + dimension, the vertical detector (`Y`) and the horizontal detector (`X`), + in that order (see :numref:`fig_dimsdata`). + +.. _fig_dimsdata: +.. figure:: ../../_static/preview/dims_prev.svg + :scale: 55 % + :alt: 3D data + + 3D projection data and their axes + + +The :code:`preview` parameter +============================= + +The :term:`preview` parameter has one field per axis. Each field accepts +:code:`start` and :code:`stop` values: + +.. code-block:: yaml + + preview: + angles: + start: + stop: + detector_y: + start: + stop: + detector_x: + start: + stop: + +The stop value is excluded, as in a Python slice. For example, the following +selection loads projections 20 through 99: + +.. code-block:: yaml + + preview: + angles: + start: 20 + stop: 100 + +.. note:: + + ``continuous_scan_subset`` also selects the angular range. When it is set, + it replaces ``preview.angles``. The command-line + ``--continuous-scan-subset`` option takes precedence over both values. See + :ref:`continuous_scan_subset_selection`. + +Using the full dataset +====================== + +Omitting :code:`preview` selects the full dataset without cropping. + +.. _previewing_enable: + +Enabling data preview +===================== + +Crop either or both detector dimensions to reduce the data size and accelerate +processing. + +.. note:: Removing blank detector regions reduces the reconstructed volume and + can also accelerate post-processing. + +The following projections show vertical and horizontal cropping. + +Before cropping |pic1| and after |pic2| + +.. |pic1| image:: ../../_static/preview/uncropped.gif + :width: 44% + +.. |pic2| image:: ../../_static/preview/cropped.gif + :width: 27% + + +1. Crop blank regions from the top and bottom of the vertical detector (`Y`), + as shown in :numref:`fig_dimsdataY`. Inspect the raw projections to identify + regions that remain blank throughout the scan. + + .. code-block:: yaml + + preview: + detector_y: + start: 200 + stop: 1800 + + This selects slices 200 to 1799, producing a vertical dimension of 1600 + pixels. The equivalent Python slice is :code:`[:, 200:1800, :]`. + +.. _fig_dimsdataY: +.. figure:: ../../_static/preview/dims_prevY.svg + :scale: 55 % + :alt: 3D data, Y slicing + + Cropping detector- `Y` dimension of 3D projection data + +2. Crop blank regions from the left and right of the horizontal detector (`X`), + as shown in :numref:`fig_dimsdataX`. + + .. warning:: + Horizontal cropping can disrupt automatic centering and introduce + reconstruction artefacts, particularly with iterative methods. Crop the + `X` dimension conservatively. + + .. code-block:: yaml + + preview: + detector_x: + start: 100 + stop: 2000 + + The equivalent Python slice is :code:`[:, :, 100:2000]`. + +.. _fig_dimsdataX: +.. figure:: ../../_static/preview/dims_prevX.svg + :scale: 55 % + :alt: 3D data, X slicing + + Cropping detector- `X` dimension of 3D projection data + +Combine both operations as follows: + +.. code-block:: yaml + + preview: + detector_y: + start: 200 + stop: 1800 + detector_x: + start: 100 + stop: 2000 + +Using :code:`begin`, :code:`mid` and :code:`end` with offsets +================================================================ + +Use :code:`begin`, :code:`mid` and :code:`end` instead of absolute indices +when the input dimensions are unknown. They may be used in the angular range +as well as the detector ranges. Adjust them with :code:`start_offset` and +:code:`stop_offset`: + +.. code-block:: yaml + + preview: + detector_x: + start: begin + start_offset: 100 + stop: end + stop_offset: -100 + detector_y: + start: mid + start_offset: -50 + stop: mid + stop_offset: 50 + +This removes 100 pixels from each end of :code:`detector_x`, equivalent to +:code:`[100:-100]`, and selects 100 pixels centred on :code:`detector_y`. + +.. note:: :code:`begin`, :code:`mid` and :code:`end` identify the first, + middle and last indices of a dimension, respectively. + + +Using :code:`mid` by itself +=========================== + +The :code:`detector_y` and :code:`detector_x` fields also accept :code:`mid` +without :code:`start` or :code:`stop`: + +.. code-block:: yaml + + preview: + detector_y: + mid + +This selects the middle three slices of the specified dimension. + +.. warning:: The :code:`angles` field does not support :code:`mid`. + +Omitting :code:`preview` fields +=============================== + +You may omit unused dimension fields and :code:`start` or :code:`stop` values. + +Omitting one or more dimension fields +------------------------------------- + +An omitted or empty dimension field selects that entire dimension. The following +configuration therefore selects the full dataset: + +.. code-block:: yaml + + preview: + angles: + detector_y: + detector_x: + +Omitting the :code:`start` or :code:`stop` fields +------------------------------------------------- + +For each dimension: + +- Omitting :code:`start` begins at index 0. +- Omitting :code:`stop` continues to the end of the dimension. diff --git a/docs/source/howto/loading_data/rotation_angles.rst b/docs/source/howto/loading_data/rotation_angles.rst new file mode 100644 index 000000000..393b05f08 --- /dev/null +++ b/docs/source/howto/loading_data/rotation_angles.rst @@ -0,0 +1,31 @@ +.. _user_defined_angles: + +Rotation angles +^^^^^^^^^^^^^^^ + +The standard loader normally reads rotation angles from the input file: + +.. code-block:: yaml + + rotation_angles: + data_path: /entry1/tomo_entry/data/rotation_angle + +If this dataset is absent or unsuitable, define an evenly spaced angle array +using its start, stop, and total number of angles: + +.. code-block:: yaml + :emphasize-lines: 8-10 + + - method: standard_tomo + module_path: httomo.data.hdf.loaders + parameters: + data_path: /1-TempPlugin-tomo/data + image_key_path: /entry1/tomo_entry/instrument/detector/image_key + rotation_angles: + user_defined: + start_angle: 0 + stop_angle: 180 + angles_total: 724 + +``start_angle`` and ``stop_angle`` are measured in degrees. +``angles_total`` specifies how many equally spaced angles HTTomo generates. diff --git a/docs/source/howto/loading_data/standard_loader.rst b/docs/source/howto/loading_data/standard_loader.rst new file mode 100644 index 000000000..ca4dbd7f1 --- /dev/null +++ b/docs/source/howto/loading_data/standard_loader.rst @@ -0,0 +1,58 @@ +.. _standard_tomo_loader: + +Standard tomography loader +^^^^^^^^^^^^^^^^^^^^^^^^^^ + +HTTomo provides the :code:`standard_tomo` :term:`loader` for parallel-beam +tomography data stored in HDF5/NeXus files. A basic configuration specifies the +projection data, :term:`image key`, and rotation angles: + +.. code-block:: yaml + + - method: standard_tomo + module_path: httomo.data.hdf.loaders + parameters: + data_path: /entry1/tomo_entry/data/data + image_key_path: /entry1/tomo_entry/instrument/detector/image_key + rotation_angles: + data_path: /entry1/tomo_entry/data/rotation_angle + +``data_path`` + The dataset containing the projection data. It commonly also contains the + dark and flat images. + +``image_key_path`` + The dataset identifying each image as a projection (0), flat (1), or dark + (2). + +``rotation_angles`` + The location or definition of the rotation angles. See + :ref:`user_defined_angles` when the input does not contain a usable angle + dataset. + +See :ref:`darks_flats` for other dark and flat layouts, and :ref:`previewing` +for loading only part of the dataset. + +.. _nxtomo_discovery: + +Automatic NXtomo discovery +========================== + +If the input contains a valid `NXtomo entry +`_, set the +following parameters to :code:`auto`: + +.. code-block:: yaml + + - method: standard_tomo + module_path: httomo.data.hdf.loaders + parameters: + data_path: auto + image_key_path: auto + rotation_angles: auto + +HTTomo then discovers the projection data, image keys and rotation angles. Use +the script in :ref:`create_nxtomo` to create compatible :term:`NXtomo` data. + +.. note:: Automatic NXtomo discovery is unavailable when darks or flats are + loaded from separate files. diff --git a/docs/source/howto/process_lists/httomo_parameters.rst b/docs/source/howto/process_lists/httomo_parameters.rst deleted file mode 100644 index ff5ed5fc1..000000000 --- a/docs/source/howto/process_lists/httomo_parameters.rst +++ /dev/null @@ -1,24 +0,0 @@ -.. _howto_proc_httomo_params: - -========================== -HTTomo-specific parameters -========================== - -Method Parameters vs. HTTomo Method Parameters ----------------------------------------------- - -In addition to parameters for a method that influence the processing that the -method performs, there are some parameters that can be used in a method's YAML -configuration that modify in some way how HTTomo handles the method (for -example, whether or not to write the output of the method to a file). - -Since these parameters are *not* parameters for the method itself and are -instead specific to HTTomo, the sections below demonstrate these extra -parameters, explaining what they are for and giving examples of how they can be -used. - -.. toctree:: - :maxdepth: 1 - - side_outputs/side_out - save_results/save_results diff --git a/docs/source/howto/process_lists/process_list_configure.rst b/docs/source/howto/process_lists/process_list_configure.rst deleted file mode 100644 index 99e62b7cb..000000000 --- a/docs/source/howto/process_lists/process_list_configure.rst +++ /dev/null @@ -1,94 +0,0 @@ -.. _howto_process_list: - -Configure efficient pipelines -============================= - -Here we focus on several important aspects which can be helpful while configuring a -process list. In order to construct more efficient pipelines one needs to be -familiar with :ref:`pl_conf_order`, :ref:`info_reslice`, and :ref:`info_sections`. - -.. _pl_conf_order: - -Method pattern and method order -------------------------------- - -An HTTomo pipeline consists of multiple methods ordered sequentially and is -executed in the given serial order (meaning that there is no branching in HTTomo -pipelines). Behind the scenes HTTomo will take care of providing the input data -for each method, and passing the output data of each method to the next method. - -Different methods require data to be provided in different orientations (ie, the -direction of slicing an array). In order to satisfy those requirements, the notion -of a method having a *pattern* was introduced in HTTomo, i.e., every method has a -pattern associated with it. So far HTTomo supports three types of patterns: -:code:`projection`, :code:`sinogram`, and :code:`all`. - -.. note:: Transitioning between methods that change the pattern from - :code:`projection` to :code:`sinogram` or vice versa will trigger a costly - :ref:`info_reslice` operation. Methods with pattern :code:`all` inherit the - pattern of the previous method. - -In order to minimise the amount of reslice operations it is best to group methods -together based on the pattern. For example, putting methods that work with -projections in one group, and methods that work with sinograms in another group. It -may not always be possible to group the methods in such way, especially with longer -pipelines. However, it's useful to keep this in mind if one seeks the most -computationally efficient pipeline. - -The pattern of any supported method can be found in :ref:`pl_library`. - -.. note:: Currently, HTTomo loaders use the :code:`projection` pattern by default, - therefore it's best for efficiency purposes that the first method after the - loader has the :code:`projection` pattern. It is also recommended to place - :ref:`centering` methods right after the loader. - -.. _pl_library: - -Library files -------------- - -In order for HTTomo to execute a method it requires certain information about the method, such -as its pattern, and (if it's a GPU method) the amount of GPU memory required per-slice. HTTomo -uses a package called :code:`httomo-backends` to get this information. - -See the `httomo-backends -`_ -documentation for more details on how it provides the required information through "library -files" and "supporting functions". - -.. _pl_grouping: - -Grouping CPU/GPU methods ------------------------- - -There are different implementations of methods in :ref:`backends_list`, and can be -classified into three categories: - -- :code:`cpu` methods. These are traditional CPU implementations in Python or other - compiled languages. The exposed TomoPy functions are mostly pure CPU. -- :code:`gpu` methods. These are methods that use GPU devices and require an input - array in CPU memory (e.g. Numpy ndarray). -- :code:`gpu_cupy` methods. These are a special group of methods, mostly from the - `HTTomolibgpu `_ library, - that are executed on GPU devices using the CuPy API. The main difference between - :code:`gpu_cupy` methods and :code:`gpu` methods is that :code:`gpu_cupy` methods - require CuPy arrays as input instead of Numpy arrays. The CuPy arrays are then - kept in GPU memory across any consecutive :code:`gpu_cupy` methods until they are - requested back on the CPU. This approach allows more flexibility with the - sequences of GPU methods, as they can be chained together for more efficient - processing. - -.. note:: If GPUs are available to the user, it is recommended to use - :code:`gpu_cupy` or :code:`gpu` methods in process lists. The methods themselves - are usually optimised for performance and HTTomo will take care of chaining the - methods together to avoid unnecessary CPU-GPU data transfers. - -The implementation of any supported method can be found in :ref:`pl_library`. - -Minimise writing to disk ------------------------- - -HTTomo does not require :ref:`save-result-examples` by default. If the result of a -method is not needed as a separate file, then there is no reason for it to be -written to disk. This is because saving intermediate files can significantly slow -down the execution time. diff --git a/docs/source/howto/process_lists/save_results/save_results.rst b/docs/source/howto/process_lists/save_results/save_results.rst deleted file mode 100644 index d12dbd180..000000000 --- a/docs/source/howto/process_lists/save_results/save_results.rst +++ /dev/null @@ -1,52 +0,0 @@ -.. _save-result-examples: - -Saving intermediate files -+++++++++++++++++++++++++ - -HTTomo, by default, will *not* write the output of a method to a file unless under certain conditions (please see more in :ref:`httomo-saving`). - -HTTomo can be informed to write or not write the output of a method to a file -with the :code:`save_result` parameter. Its value is a boolean, so either -:code:`true` or :code:`false` are valid values for it. - - -Example 1: save output of a specific method -########################################### - -Suppose we wanted to save the output of the normalisation function :code:`normalize`. Then we -should add :code:`save_result: true` to the list of the function parameters, but NOT the method's parameters: - -.. code-block:: yaml - :emphasize-lines: 8 - - - method: normalize - module_path: httomolibgpu.prep.normalize - parameters: - cutoff: 10.0 - minus_log: true - nonnegativity: false - remove_nans: false - save_result: true - -Example 2: using :code:`--save_all` and :code:`save_result` together -#################################################################### - -When the :code:`--save_all` option/flag is provided, the :code:`save_result` -parameter can be used to override individual method's to *not* save their -output. - -In contrast to the previous example, suppose we had a process list where we -would like to save the output of all methods using :code:`--save_all`, *apart* from the -:code:`normalize` method. - -.. code-block:: yaml - :emphasize-lines: 8 - - - method: normalize - module_path: httomolibgpu.prep.normalize - parameters: - cutoff: 10.0 - minus_log: true - nonnegativity: false - remove_nans: false - save_result: false \ No newline at end of file diff --git a/docs/source/howto/process_lists/side_outputs/side_out.rst b/docs/source/howto/process_lists/side_outputs/side_out.rst deleted file mode 100644 index f5253f33b..000000000 --- a/docs/source/howto/process_lists/side_outputs/side_out.rst +++ /dev/null @@ -1,116 +0,0 @@ -.. _side_output: - -Side outputs -++++++++++++ - -There are cases where the output of a method is needed as the value of a parameter -for a method further down the pipeline. For example, the output of a method that -calculates the :ref:`centering`, that is required for a reconstruction method. - -HTTomo provides a special syntax (loosely based on the syntax for references in -`GitHub Actions -`_) -for how such an output of a method needs to be defined and how to refer to that -special output later. - -Specifying the side output -########################## - -The output of some methods isn't processed data, but rather supplementary -information to be used later in the pipeline. The given term for that supplementary -data is "side outputs", and is what the :code:`side_outputs` parameter is for. As -an example, let us consider the following centering algorithm: - -.. code-block:: yaml - :emphasize-lines: 11,12,13 - - - method: find_center_vo - module_path: httomolibgpu.recon.rotation - parameters: - ind: null - smin: -50 - smax: 50 - srad: 6.0 - step: 0.25 - ratio: 0.5 - drop: 20 - id: centering - side_outputs: - cor: centre_of_rotation - -One can see that :code:`side_outputs` here includes a single value :code:`cor` with -the :code:`centre_of_rotation` reference. The :code:`id` parameter here is needed to -refer to the method later. - -Referring to the side output -############################ - -The purpose of :code:`side_outputs` is to refer to it later, when some method(s) -require the contained information in the reference. Consider this example where the -reconstruction method refers to the centering method's side outputs. The required -information of :ref:`centering` is stored in the reference -:code:`${{centering.side_outputs.centre_of_rotation}}`. - -.. code-block:: yaml - :emphasize-lines: 4 - - - method: FBP3d_tomobar - module_path: httomolibgpu.recon.algorithm - parameters: - center: ${{centering.side_outputs.centre_of_rotation}} - filter_freq_cutoff: 1.1 - recon_size: null - recon_mask_radius: null - - -There could be various configurations when this reference is required from other methods as well. We present more verbose :ref:`side_output_example` below. - -.. note:: Side outputs and references to them are generated automatically with the - `YAML generator `_. Usually there is no need to modify them when - editing a process list. - -.. _side_output_example: - -Example of side outputs -####################### - -Pipeline overview -================= - -This pipeline is for reconstructing DFoV data which needs to be stitched into the -traditional 180 degrees data. It demonstrates three cases where a method produces -one or more side outputs, and a method later in the pipeline references them. - -A rough outline of the three side outputs being used is: - -1. the 360 centering method produces an "overlap" value as a side output, for the - stitching method to use -2. the 360 centering method also produces a CoR value as a side output, for the - recon method to use -3. the stats calculator method produces a global stats value as a side output, for - the image saver method to use - -Detailed look at side outputs usage -=================================== - -Parameters for stitching are generated by :code:`find_center_360`, stored in side -outputs, and then used later in the :code:`sino_360_to_180` method. The -reconstruction method :code:`FBP3d_tomobar` then refers to the found :ref:`centering` for the -stitched dataset produced by :code:`find_center_360`. Finally, we also need to -extract the global statistics for normalisation of the data using the -:code:`calculate_stats` method when saving into images with the -:code:`save_to_images` method. - -.. literalinclude:: ../../../pipelines_full/deg360_distortion_FBP3d_tomobar.yaml - :language: yaml - :caption: Pipeline for 360 degrees scan, a double field of view (DFoV) data. - -Please see below for a concise description of how each of the individual side -outputs of methods are connected to the method using them: - -1. :code:`find_center_360` produces side outputs called :code:`overlap` and :code:`side` that the - :code:`sino_360_to_180` method uses. -2. :code:`find_center_360` also produces a side output called - :code:`centre_of_rotation` that the :code:`FBP3d_tomobar` method uses. -3. :code:`calculate_stats` produces a side output called :code:`glob_stats` that - the :code:`rescale_to_int` method uses. diff --git a/docs/source/howto/process_lists_guide.rst b/docs/source/howto/process_lists_guide.rst deleted file mode 100644 index 4888ac236..000000000 --- a/docs/source/howto/process_lists_guide.rst +++ /dev/null @@ -1,48 +0,0 @@ -Process Lists Guide -******************** - -In this section we describe how a process list (aka pipeline) can be configured. We -explain how to begin building and editing a process list in general using the -pre-existing templates, and how to :ref:`howto_process_list`. -:ref:`howto_proc_httomo_params` also play a role in defining a process list, so -they are introduced here too. - -Editing process lists ---------------------- - -This section explains how to build a process list (see more on :ref:`explanation_process_list`) from YAML templates -(see more on :ref:`explanation_templates`). - -Given time working with HTTomo, a user will likely settle on a workflow for -defining process list YAML files that suits their individual needs. For editing -YAML files, we can recommend Visual Studio Code, Atom, and Notepad++ as editors -that recognise YAML syntax out-of-the-box. - -As a starting point, the general process of building the pipeline can be the following: - -- copy+paste templates for the desired methods from the - :ref:`reference_templates` section -- manually edit the parameter values within the copied template as needed. The user might want - to check the documentation for the relevant method in the library itself. -- intermittently run the :ref:`YAML checker ` during - editing of the YAML file to detect any errors early on. It is strongly recommended to run - the checker at least once when the YAML pipeline is configured and ready to be run. - -Methods order -------------- - -Some general rules for building a process list from individual methods are the -following: - -* Any process list needs to start with an :ref:`HTTomo loader`, - which are provided as :ref:`reference_templates`. -* The execution order of the methods in the process list is **sequential** starting - from the top and ending at the bottom. -* The exchange of additional data between methods is performed using - :ref:`howto_proc_httomo_params`. - -.. toctree:: - :maxdepth: 2 - - process_lists/httomo_parameters - process_lists/process_list_configure diff --git a/docs/source/howto/run_httomo.rst b/docs/source/howto/run_httomo.rst index 6b77d71d1..ef3e60fc6 100644 --- a/docs/source/howto/run_httomo.rst +++ b/docs/source/howto/run_httomo.rst @@ -1,43 +1,118 @@ .. _howto_run: -Running HTTomo --------------- +Run your first pipeline +======================= -The next section gives an overview of the commands to quickly get started running -HTTomo. +This guide takes you through choosing, checking and running an existing HTTomo +pipeline. If HTTomo is not installed yet, follow :ref:`installation_main` +first. -For those interested in learning about the different ways HTTomo can be configured -to run, there is the :ref:`run-httomo-indepth` section. +.. admonition:: HTTomo at Diamond Light Source + :class: note -Quick Overview of Running HTTomo -================================ + If you are running HTTomo at Diamond, see the dedicated page on + :ref:`howto_run_at_diamond`. Required inputs -+++++++++++++++ +--------------- -In order to run HTTomo you require a data file (an HDF5 file) and a YAML process -list file that describes the desired processing pipeline. For information on -getting started creating this YAML file, please see :ref:`howto_process_list` -and also ready-to-be-used :ref:`tutorials_pl_templates`. +You need: -Running HTTomo Inside or Outside of Diamond -+++++++++++++++++++++++++++++++++++++++++++ +* an HDF5 file containing the input tomography data; +* a YAML pipeline describing the processing; +* a directory in which HTTomo can create its output. -As HTTomo was developed at the Diamond Light Source, there have been some extra -efforts to accommodate the users at Diamond (for example, aliases for commands -and launcher scripts). As such, there are some differences as to how one would run -HTTomo at Diamond vs. outside of Diamond, and the guidance on running HTTomo has -been split into two sections accordingly. +The input data should normally follow the `NXtomo application definition +`_. HTTomo's +standard loader can automatically locate the projections, image keys and +rotation angles in an NXtomo file. Other HDF5 layouts can be used by setting +the dataset paths explicitly in the loader configuration. -Additionally, HTTomo is able to run in serial or in parallel depending on what -computer hardware is available to the user, so some sections have been further -split into these two subsections where relevant. +If you need sample data, the :ref:`synthetic-data-example` page explains how to +generate an NXtomo file. :ref:`Real data ` can also be used +directly. -.. toctree:: - :maxdepth: 2 +Choose an example pipeline +-------------------------- - how_to_run/at_diamond - how_to_run/outside_diamond - how_to_run/run_in_depth - how_to_run/real_data_example +Use :ref:`choose_pipeline` to select one of the +:ref:`ready-to-use pipelines `. You can also find data +and associated pipelines in :ref:`data_tutorials`. Choose a GPU pipeline when +a CUDA-enabled GPU and its required processing libraries are available. +Otherwise, choose a CPU pipeline and ensure that its backend library, such as +TomoPy, is installed. +Copy the selected pipeline into a local YAML file. Check its loader entry and +adapt the data paths if the input is not an NXtomo file. For guidance on +changing methods or parameters, see :ref:`how_to_configure_pipeline`. + +Validate the pipeline +--------------------- + +Before running the pipeline, check that its configuration is valid and that its +data paths exist in the input file: + +.. code-block:: console + + $ python -m httomo check PIPELINE.yaml INPUT.h5 + +Replace ``PIPELINE.yaml`` and ``INPUT.h5`` with the paths to the chosen +pipeline and input data. A successful check confirms that the YAML structure, +methods, parameters and referenced HDF5 paths are valid. Correct any reported +errors before continuing. See :ref:`utilities_yamlchecker` for details of the +checks performed. + +Run HTTomo +---------- + +Then run the pipeline: + +.. code-block:: console + + $ python -m httomo run INPUT.h5 PIPELINE.yaml OUTPUT_DIR + +Replace the example arguments as follows: + +``INPUT.h5`` + The path to the HDF5 file containing the input data. + +``PIPELINE.yaml`` + The path to the YAML file that defines the processing pipeline. + +``OUTPUT_DIR`` + The directory in which HTTomo will create the processing output. + +HTTomo creates a timestamped run directory inside ``OUTPUT_DIR``. For all +available commands and options, see :ref:`run-httomo-indepth`. + +Run in parallel +--------------- + +To divide the input data between multiple processes, launch HTTomo using MPI: + +.. code-block:: console + + $ mpirun -np N python -m httomo run INPUT.h5 PIPELINE.yaml OUTPUT_DIR + +Here, ``N`` is the number of parallel processes to launch. Each process +normally requires access to a GPU when the pipeline contains GPU methods. + +At Diamond, use the ``httomo_mpi`` launcher instead; see +:ref:`howto_run_at_diamond`. + +Find and inspect the output +--------------------------- + +Open the newly created run directory inside ``OUTPUT_DIR``. It contains: + +* a copy of the pipeline, retaining its source filename; +* ``user.log``, containing the concise progress information shown in the + terminal; +* ``debug.log``, containing more detailed diagnostic information; +* requested intermediate or reconstruction results as HDF5 files; and +* snapshots or image files when the corresponding output options or pipeline + methods were enabled. + +Inspect HDF5 results with an HDF5-compatible viewer such as DAWN, HDFView or +silx. See :ref:`info_logger` for help interpreting the logs, progress +bars and other output files. diff --git a/docs/source/howto/running_tests.rst b/docs/source/howto/running_tests.rst index 0d3e0abc2..2af2defc8 100644 --- a/docs/source/howto/running_tests.rst +++ b/docs/source/howto/running_tests.rst @@ -1,17 +1,11 @@ -.. _run_tests: +:orphan: -Run HTTomo tests ----------------- +Testing guide moved +=================== -After installing HTTomo, you can quickly verify that the installation and all required dependencies are working correctly by running the test suite. +Source-checkout setup and testing are now described in +:ref:`developer_setup`. -1. Git clone the HTTomo repository :code:`git clone https://github.com/DiamondLightSource/httomo.git`. - -2. Install the testing dependencies: :code:`conda install -c conda-forge pytest pytest-cov pytest-xdist pytest-mock plumbum`. - -3. **Run the CPU test suite.** Navigate to the root directory of the HTTomo repository and run: :code:`pytest tests/`. This executes the CPU-only tests for the HTTomo framework. - -4. **Run the GPU test suite (CUDA-enabled systems only).** If you have a CUDA-compatible GPU, run: :code:`pytest tests/ --cupy`. - -5. **Run the small dataset pipeline tests.** Get full YAML `pipelines `_ and place/unzip them into the :code:`/docs/source/pipelines_full` folder of your cloned HTTomo repository. Then you can run :code:`pytest tests/ --small_data`. These tests execute example pipelines using a small test dataset. On systems without a CUDA-compatible GPU, some tests are expected to fail. However, if TomoPy is installed, the TomoPy pipeline test should pass successfully. +.. raw:: html + diff --git a/docs/source/howto/troubleshooting.rst b/docs/source/howto/troubleshooting.rst new file mode 100644 index 000000000..ea15a1dc7 --- /dev/null +++ b/docs/source/howto/troubleshooting.rst @@ -0,0 +1,106 @@ +.. _troubleshooting: + +Troubleshooting +=============== + +Start by reading ``user.log`` in the run directory. If it does not explain the +failure, inspect ``debug.log`` and rerun pipeline validation with the input +file: + +.. code-block:: console + + $ python -m httomo check pipeline.yaml input.nxs + +Pipeline validation fails +------------------------- + +Check YAML indentation, method names, parameter names and required values. +When an input file is supplied, HTTomo also verifies explicit HDF5 dataset +paths. See :ref:`utilities_yamlchecker` and :ref:`pipeline_file_reference`. + +HTTomo cannot find NXtomo data +------------------------------ + +Inspect the input hierarchy and confirm that the NXtomo ``NX_class`` and +``definition`` attributes are present. If automatic discovery is not suitable, +set ``data_path``, ``image_key_path`` and ``rotation_angles`` explicitly. See +:ref:`nxtomo_discovery` and :ref:`create_nxtomo`. + +MPI or parallel HDF5 fails at startup +------------------------------------- + +Confirm that HTTomo, ``mpi4py`` and ``h5py`` use compatible MPI libraries and +that h5py reports parallel support: + +.. code-block:: console + + $ python -c "import h5py; print(h5py.get_config().mpi)" + +The command must print ``True``. Inconsistent MPI installations commonly cause +import errors, immediate termination or hangs during file access. + +CUDA or CuPy cannot see a GPU +----------------------------- + +Check the NVIDIA driver and CuPy runtime before running HTTomo: + +.. code-block:: console + + $ nvidia-smi + $ python -c "import cupy; print(cupy.cuda.runtime.getDeviceCount())" + +The installed CuPy CUDA package must be compatible with the system driver. Use +a CPU pipeline when no CUDA-capable GPU is available. + +GPU memory is exhausted +----------------------- + +An out-of-memory error can be caused by either the pipeline block size or other +processes already using the device. + +#. Run ``nvidia-smi`` and stop unrelated jobs using the selected GPU. +#. In a multi-process run, assign one MPI rank to each GPU. Do not allow several + ranks to select the same device unless the pipeline and hardware were sized + for that arrangement. +#. Reduce the input :term:`preview`, particularly ``detector_y`` while testing. +#. Set a smaller per-process ceiling, for example ``--max-memory 12G``. For GPU + sections this caps the budget used to calculate the block size. It may also + cause section data to use slower disk-backed storage. +#. If one method still fails while others fit, its GPU memory estimator may be + inaccurate. Record the method, parameters, input shape and ``debug.log`` when + reporting the problem. + +See :ref:`memory_and_performance` for user guidance and +:ref:`developers_memorycalc` when diagnosing an estimator. + +Pipeline and method templates do not match +------------------------------------------ + +Errors about an unknown method or parameter often mean that a pipeline was +generated for a different HTTomo or ``httomo-backends`` version. Use +:ref:`versioned_downloads` for a released HTTomo version and check the +:ref:`compatibility` notes before updating one component independently. + +The output directory cannot be created or written +-------------------------------------------------- + +HTTomo creates a run directory below ``OUT_DIR``. Confirm that the parent +directory exists where required by the scheduler, is writable by every MPI +rank and has enough free space for requested intermediate files. With a +multi-node run, use storage visible from every node. The directory supplied to +``--reslice-dir`` must already exist and be writable. + +Re-slicing is unexpectedly slow +------------------------------- + +A change between projection and sinogram processing can require temporary +disk-backed data. Put ``--reslice-dir`` on fast storage accessible to every +participating process. See :ref:`info_reslice` and the +:ref:`command-line reference `. + +Finding more information +------------------------ + +The :ref:`info_logger` page explains common progress messages and generated +files. When reporting a reproducible problem, include the HTTomo version, +pipeline, relevant log excerpt, execution command and hardware configuration. diff --git a/docs/source/howto/tutorial.rst b/docs/source/howto/tutorial.rst new file mode 100644 index 000000000..77bb01afd --- /dev/null +++ b/docs/source/howto/tutorial.rst @@ -0,0 +1,13 @@ +.. _data_tutorials: + +Tutorials +========= + +These tutorials cover synthetic data generation and the processing of synthetic +and experimental tomography data with HTTomo. + +.. toctree:: + :maxdepth: 2 + + tutorial/real_data_example + tutorial/synthetic_data diff --git a/docs/source/howto/tutorial/data/artefacts.json b/docs/source/howto/tutorial/data/artefacts.json new file mode 100644 index 000000000..89228d8bc --- /dev/null +++ b/docs/source/howto/tutorial/data/artefacts.json @@ -0,0 +1,10 @@ +{ + "zingers_percentage": 0.05, + "zingers_modulus": 10, + "stripes_percentage": 3.0, + "stripes_maxthickness": 3, + "stripes_intensity": 0.05, + "stripes_type": "full", + "stripes_variability": 0.002, + "verbose": true +} diff --git a/docs/source/howto/tutorial/data/flat_settings.json b/docs/source/howto/tutorial/data/flat_settings.json new file mode 100644 index 000000000..1a60c6b73 --- /dev/null +++ b/docs/source/howto/tutorial/data/flat_settings.json @@ -0,0 +1,9 @@ +{ + "detectors_miscallibration": 0.05, + "variations_number": 3, + "arguments_Bessel": [1, 25], + "specklesize": 2, + "kbar": 2, + "sigmasmooth": 3, + "jitter_projections": 0.0 +} diff --git a/docs/source/howto/tutorial/examples/lorentz_data_example.rst b/docs/source/howto/tutorial/examples/lorentz_data_example.rst new file mode 100644 index 000000000..93126bd5f --- /dev/null +++ b/docs/source/howto/tutorial/examples/lorentz_data_example.rst @@ -0,0 +1,55 @@ +.. _real-data-lorentz: + +Lorentz data +============ + +This example uses raw data from the `TomoBank`_ archive. + +.. list-table:: + + * - .. figure:: ../../../_static/real_data/sino_tomo088.jpg + :width: 70% + :align: center + + Dark/flat-field-corrected sinogram of the `Lorentz data set`_. + + - .. figure:: ../../../_static/real_data/recon_tomo088.jpg + :width: 70% + :align: center + + Reconstructed slice using the FBP method. + +Download the `Lorentz data set`_. It is hosted using the Globus file management +system, which requires authentication. You can sign in using GitHub +credentials. + +.. _TomoBank: https://tomobank.readthedocs.io/en/latest/ + +.. _Lorentz data set: https://tomobank.readthedocs.io/en/latest/source/data/docs.data.lorentz.html + +After downloading the dataset, confirm that ``tomo_00088.h5`` is available, +then run one of the pipelines below. + +TomoPy (CPU) pipeline ++++++++++++++++++++++ + +This pipeline uses TomoPy on the CPU, so TomoPy must be installed. See +:ref:`backends_list`. Copy the pipeline into a YAML file and +:ref:`run HTTomo `. + +.. dropdown:: Standard 180 degrees pipeline using TomoPy (CPU) for tomo_00088.h5 dataset + + .. literalinclude:: ../../../pipelines_full/tomopy_tomobank.yaml + :language: yaml + +GPU pipeline +++++++++++++ + +If a CUDA-enabled GPU is available, the same dataset can be processed using +GPU-accelerated libraries. This can significantly reduce processing time for +suitable pipelines. Run the pipeline below in the same way as the CPU example. + +.. dropdown:: GPU-enabled processing for tomo_00088.h5 dataset + + .. literalinclude:: ../../../pipelines_full/FBP3d_tomobar_tomobank.yaml + :language: yaml diff --git a/docs/source/howto/tutorial/examples/sandstone_data_example.rst b/docs/source/howto/tutorial/examples/sandstone_data_example.rst new file mode 100644 index 000000000..03acbf175 --- /dev/null +++ b/docs/source/howto/tutorial/examples/sandstone_data_example.rst @@ -0,0 +1,77 @@ +.. _real-data-sandstone: + +Sandstone data +============== + +This example reconstructs experimental tomography data from a sandstone rock +sample collected at the I12 beamline at Diamond Light Source. The dataset is +available from the `Sandstone rock tomographic data Zenodo record`_. + +.. _fig_sandstone: + +.. figure:: ../../../_static/real_data/recon_sandstone.png + :scale: 80 % + :alt: sandstone reconstruction + + Reconstructed slice of the sandstone dataset + + +.. _Sandstone rock tomographic data Zenodo record: https://doi.org/10.5281/zenodo.10033401 + +Download the data ++++++++++++++++++ + +The Zenodo record provides two datasets: + +* ``dataset_sandstone1.zip`` (17.9 GB); and +* ``dataset_sandstone2.zip`` (17.8 GB). + +Download either archive from Zenodo and extract it to a directory with enough +space for both the archive and its extracted contents. The instructions below +apply to either dataset. + +.. note:: + + These are large downloads. Make sure that sufficient storage is also + available for the reconstruction and TIFF images produced by HTTomo. + +After extraction, identify the HDF5 or NeXus tomography file that will be +passed to HTTomo as ``INPUT_FILE``. + +GPU reconstruction pipeline ++++++++++++++++++++++++++++ + +The :ref:`LPRec3d pipeline ` uses GPU-accelerated +methods for correction, centre finding, stripe removal and reconstruction. It +reconstructs the volume with ``LPRec3d_tomobar`` and saves the result as TIFF +images. + +This pipeline requires a CUDA-enabled GPU and the HTTomolibGPU and ToMoBAR +backends. See :ref:`backends_list` for the available processing libraries. + +Copy the following pipeline into a file named +``LPRec3d_tomobar.yaml``: + +.. dropdown:: LPRec3d GPU pipeline for the sandstone dataset + + .. literalinclude:: ../../../pipelines_full/LPRec3d_tomobar.yaml + :language: yaml + +The standard loader in this pipeline uses automatic dataset discovery. If the +downloaded file does not follow the NXtomo layout, replace the loader's +``data_path``, ``image_key_path`` and ``rotation_angles`` values with the +corresponding paths in the input file. See :ref:`reference_loaders` for loader +configuration details. + +.. warning:: + + On a memory-limited workstation, use :ref:`previewing` to reconstruct a + small vertical range first. For example, select ten detector rows in the + loader: + + .. code-block:: yaml + + preview: + detector_y: + start: 1000 + stop: 1010 diff --git a/docs/source/howto/tutorial/examples/stripes_data_example.rst b/docs/source/howto/tutorial/examples/stripes_data_example.rst new file mode 100644 index 000000000..7d2de39f0 --- /dev/null +++ b/docs/source/howto/tutorial/examples/stripes_data_example.rst @@ -0,0 +1,136 @@ +.. _real-data-stripes: + +Stripe-removal data +=================== + +This example processes the ``68067.nxs`` tomography dataset collected at the +I12 beamline at Diamond Light Source. The data contain full, partial, +unresponsive, fluctuating and blurry stripes, which appear as ring artefacts in +reconstructed images. They accompanied the paper `Superior techniques for +eliminating ring artifacts in X-ray micro-tomography`_ and are available from +the `stripe-removal data Zenodo record`_. + +.. _fig_stripes: + +.. figure:: ../../../_static/real_data/stripe_vo_recon.png + :scale: 80 % + :alt: stripes data reconstruction + + Reconstructed slice of the dataset + +.. _Superior techniques for eliminating ring artifacts in X-ray micro-tomography: https://doi.org/10.1364/OE.26.028396 +.. _stripe-removal data Zenodo record: https://doi.org/10.5281/zenodo.1443568 + +Download the data ++++++++++++++++++ + +Download ``Datasets.zip`` (26.7 GB) from Zenodo and extract it. The raw data +used in this example are in ``Datasets/Data_Fig25/Raw_data``: + +* ``68067.nxs`` (422.3 kB) contains the scan metadata; and +* ``pco1-68067.hdf`` (21.5 GB) contains the projections, flat fields and dark + fields. + +Keep these two files together because the NeXus file links to the HDF5 file. +Pass ``68067.nxs``, rather than ``pco1-68067.hdf``, to HTTomo as the +``INPUT_FILE``. + +.. note:: + + This is a large download. Allow enough space for the archive, its extracted + contents, the reconstruction and any TIFF images produced by HTTomo. + +GPU reconstruction pipeline +++++++++++++++++++++++++++++ + +The :ref:`LPRec3d pipeline ` uses the +``remove_all_stripe`` method from HTTomolibGPU. This method combines the +sorting, large-stripe and dead-stripe techniques described in the paper, making +it a suitable starting point for a dataset containing several types of stripe. +The pipeline then reconstructs the corrected data with ``LPRec3d_tomobar`` and +saves the result as TIFF images. + +This pipeline requires a CUDA-enabled GPU and the HTTomolibGPU and ToMoBAR +backends. See :ref:`backends_list` for the available processing libraries. + +Copy the following pipeline into a file named +``LPRec3d_tomobar.yaml``: + +.. dropdown:: LPRec3d GPU pipeline for the stripe-removal dataset + + .. literalinclude:: ../../../pipelines_full/LPRec3d_tomobar.yaml + :language: yaml + +This dataset follows NXtomo, so the standard loader can discover its datasets +automatically. + +.. warning:: + + Preview a small range of the vertical detector before reconstructing the + full dataset. For example, use ten slices while tuning the stripe-removal + parameters: + + .. code-block:: yaml + + preview: + detector_y: + start: 1000 + stop: 1010 + +Comparing stripe-removal methods +++++++++++++++++++++++++++++++++ + +The stripe-removal stage is located after dark/flat-field correction and before +``minus_log`` in the pipeline. To compare filters, replace +``remove_all_stripe`` with one of the stages below and write each run to a +different output directory. Also run the pipeline once without a stripe-removal +stage to provide an uncorrected reference. + +Sorting-based removal +--------------------- + +``remove_stripe_based_sorting`` is particularly effective for full and partial +stripes. A larger median-filter window removes broader stripes but can also +smooth genuine detector-direction features. + +.. code-block:: yaml + + - method: remove_stripe_based_sorting + module_path: httomolibgpu.prep.stripe + parameters: + size: 21 + dim: 1 + +Fourier-wavelet removal +----------------------- + +``remove_stripe_fw`` suppresses stripe components using a Fourier-wavelet +filter. Increase ``sigma`` for stronger damping and compare the reconstruction +with the sorting-based result to check that sample features have been +preserved. + +.. code-block:: yaml + + - method: remove_stripe_fw + module_path: httomolibgpu.prep.stripe + parameters: + sigma: 2 + wname: db5 + level: null + +Titarenko removal +----------------- + +``remove_stripe_ti`` corrects detector-channel variations using the Titarenko +method. Lower ``beta`` values apply stronger filtering. + +.. code-block:: yaml + + - method: remove_stripe_ti + module_path: httomolibgpu.prep.stripe + parameters: + beta: 0.1 + +Use the same detector preview and reconstruction settings for every run. Compare +both the remaining rings and the preservation of fine sample features; the +strongest-looking correction is not necessarily the most faithful one. diff --git a/docs/source/howto/tutorial/real_data_example.rst b/docs/source/howto/tutorial/real_data_example.rst new file mode 100644 index 000000000..8ec55fd13 --- /dev/null +++ b/docs/source/howto/tutorial/real_data_example.rst @@ -0,0 +1,14 @@ +.. _real-data-example: + +Real data processing +==================== + +This section presents an example of processing real experimental data using HTTomo. + + +.. toctree:: + :maxdepth: 2 + + examples/sandstone_data_example + examples/stripes_data_example + examples/lorentz_data_example diff --git a/docs/source/howto/tutorial/synthetic_data.rst b/docs/source/howto/tutorial/synthetic_data.rst new file mode 100644 index 000000000..c1c434809 --- /dev/null +++ b/docs/source/howto/tutorial/synthetic_data.rst @@ -0,0 +1,133 @@ +.. _synthetic-data-example: + +Generate synthetic data +======================= + +This tutorial uses `TomoPhantom `_ to +create a synthetic tomography dataset that can be processed directly by HTTomo. +Synthetic data can include controlled artefacts such as zingers, stripes, +noise and misalignment. This makes it useful for testing the robustness of +processing methods. + +The data output follows the `NXtomo application definition +`_ and contains +projections, flat-field images, dark-field images and rotation angles. See +:ref:`create_nxtomo` for more information about the NXtomo format. + +Install TomoPhantom +------------------- + +Install TomoPhantom and the generator's optional dependencies in the same Conda +environment as HTTomo: + +.. code-block:: console + + $ conda install -c httomo -c conda-forge "tomophantom>=3.1.5" psutil scikit-image + +Prepare the detector settings +----------------------------- + +Download the following example configurations and place them in a new working +directory: + +* :download:`artefacts.json ` adds stripes and zingers to + the projections. +* :download:`flat_settings.json ` configures the + simulated flat-field images and detector response. + +The files are optional and can be edited to create different detector +conditions. See the `TomoPhantom artefacts documentation +`_ for the available +settings. + +Generate the dataset +-------------------- + +From the directory containing the JSON files, run: + +.. code-block:: console + + $ python -m tomophantom.scripts.nxs_generator \ + --realistic \ + --model-number 18 \ + --sinogram-shape 256 512 740 \ + --flats 20 \ + --darks 10 \ + --source-intensity 20000 \ + --artefacts artefacts.json \ + --flat-settings flat_settings.json \ + --seed 1 \ + --output-path tomodata_synth.nxs + +The main options are: + +``--model-number`` + The model selected from the TomoPhantom 3D phantom library. + +``--sinogram-shape`` + The detector height, number of projection angles and detector width. + +``--flats`` and ``--darks`` + The number of flat-field and dark-field images to generate. + +``--source-intensity`` + The simulated source intensity used by the detector noise model. + +``--seed`` + The random seed used to make the simulation reproducible. + +For a faster test, reduce the shape to ``128 256 362``. To generate the 3D +Shepp--Logan phantom, use ``--model-number 13``. Run the following command to +see all generator options: + +.. code-block:: console + + $ python -m tomophantom.scripts.nxs_generator --help + +Inspect the result +------------------ + +.. _fig_synth_data: + +.. figure:: ../../_static/synth_data_screenshot.png + :alt: Synthetic data generated + :align: center + :width: 70% + + Visualising the generated synthetic data in `myHDF5 viewer + `_. + +The command creates ``tomodata_synth.nxs`` with 10 darks, 20 flats and 512 +projections. You can inspect its hierarchy and datasets with `DAWN +`_, `HDFView +`_, `silx view +`_ or the browser-based +`myHDF5 viewer `_. + +Use the data with HTTomo +------------------------ + +Because the generated file is NXtomo-compliant, the standard loader can locate +its data, image keys and rotation angles automatically: + +.. code-block:: yaml + + - method: standard_tomo + module_path: httomo.data.hdf.loaders + parameters: + data_path: auto + image_key_path: auto + rotation_angles: auto + preview: + detector_x: + start: null + stop: null + detector_y: + start: null + stop: null + darks: null + flats: null + +Add the required processing methods after the loader, then :ref:`check and run +the pipeline `. See :ref:`nxtomo_discovery` for more information +about automatic NXtomo discovery. diff --git a/docs/source/index.rst b/docs/source/index.rst index e1fc25491..1be422048 100644 --- a/docs/source/index.rst +++ b/docs/source/index.rst @@ -1,63 +1,97 @@ -.. include:: ../../README.rst +HTTomo documentation +==================== + +HTTomo is a Python framework for high-performance tomographic data processing. +It coordinates distributed I/O and CPU/GPU processing through MPI and runs +pipelines described in readable YAML files. + +Where should I start? +--------------------- + +.. grid:: 1 2 2 2 + :gutter: 2 + + .. grid-item-card:: New to HTTomo? + + Follow the :ref:`quickstart` to download a small test dataset, validate + a pipeline and run it from start to finish. + + .. grid-item-card:: Preparing your own data? + + Read :ref:`loading_data`, including how to create an + :ref:`HTTomo-compatible NXtomo file `. + + .. grid-item-card:: Building a pipeline? + + Browse the :ref:`ready-to-use pipelines + `, or configure one from + :ref:`available method templates `. + + .. grid-item-card:: Developing HTTomo or a method? + + Start with :ref:`developer_architecture`, then follow the contribution + path that matches your change. .. _intro_content: .. toctree:: :caption: Introduction :maxdepth: 2 - :glob: introduction/about - explanation/templates - explanation/process_list - explanation/faq + introduction/execution_model + introduction/data_proc_concepts -.. _how_to_content: +.. _getting_started: .. toctree:: - :caption: How To's + :caption: Getting started :maxdepth: 2 howto/installation + getting_started/quickstart howto/run_httomo - howto/process_lists_guide - howto/httomo_features - howto/interpret_logger + getting_started/at_diamond -.. _reference_content: +.. _how_to_content: +.. _tutorials_content: .. toctree:: - :caption: Reference guides + :caption: User guide :maxdepth: 2 - reference/yaml - reference/loaders + howto/loading_data + howto/httomo_features + howto/troubleshooting + howto/tutorial + faq/faq +.. _pipelines_methods_content: .. _backends_content: .. toctree:: - :caption: Data processing + :caption: Pipelines and methods :maxdepth: 2 - backends/list + pipelines/yaml + pipelines/versioned_downloads + pipelines/choose_pipeline backends/templates + backends/list + pipelines/reconstruction_ecosystem -.. _tutorials_content: +.. _reference_content: .. toctree:: - :caption: Ready-to-use pipelines + :caption: Reference :maxdepth: 2 - :glob: - - pipelines/yaml -.. _utilities_content: - -.. toctree:: - :caption: Utilities - :maxdepth: 2 + reference/cli + reference/pipeline_file + reference/run_output + reference/compatibility + reference/glossary - utilities/yaml_checker .. _developers_content: @@ -65,6 +99,11 @@ :caption: Developers :maxdepth: 2 + developers/architecture + developers/development_setup developers/how_to_contribute + developers/add_own_method + developers/httomo_backends developers/memory_calculation + developers/profiling_tracing developers/api diff --git a/docs/source/introduction/about.rst b/docs/source/introduction/about.rst index f72913776..9afcdaab7 100644 --- a/docs/source/introduction/about.rst +++ b/docs/source/introduction/about.rst @@ -1,29 +1,24 @@ -HTTomo concept -******************* - -HTTomo stands for High Throughput Tomography pipeline for processing and reconstruction of parallel-beam tomography data. -The `HTTomo project `_ was initiated in 2022 at `Diamond Light source `_ by the Data Analysis Group and it is written in Python. -With the `Diamond-II `_ upgrade approaching, there is a -need to be able to process bigger data in larger quantities and with high fidelity. With the support of modern developments in -the field of High Performance Computing and multi-GPU processing, it is possible to enable faster data streaming and higher throughput for big data. - -The main concept of HTTomo is to split the data into three-dimensional (3D) chunks/blocks and process them in parallel. The speed is gained using -the optimised I/O modules, in-memory reslicing operations using MPI protocols, GPU-accelerated processing routines, and a capability of device-to-device GPU processing using the `CuPy `_ library. -HTTomo orchestrates the optimal data splitting driven by the available GPU memory, which makes possible processing and reconstruction of big data even on smaller GPU cards. - -.. figure:: ../_static/3d_setup.png - :scale: 40 % - :alt: Simple tomographic pipeline - - HTTomo is tailored to work with 3D data, here 3D parallel-beam tomographic projection data is split and sent to a cluster with multiple GPUs for processing and reconstruction. Serial processing of data is also possible. - -HTTomo is a User Interface (UI) package and does not contain any data processing methods, but rather utilises other libraries as `backends `_. -Please see the list of currently supported packages by HTTomo in :ref:`backends_list`. It should be relatively simple to integrate any other modular -CPU/GPU Python library for data processing methods in HTTomo, please see more on that in :ref:`developers_content` section. - -A complex data analysis pipelines can be built by stacking together provided :ref:`reference_templates` and :ref:`tutorials_pl_templates` are also provided. - -.. toctree:: - :maxdepth: 2 - - indepth/detailed_about +About HTTomo +************ + +HTTomo is a Python framework for high-throughput processing and reconstruction +of parallel-beam tomography data. It is developed at `Diamond Light Source +`_ to support increasing data rates, including +those anticipated from the `Diamond-II upgrade +`_. + +HTTomo can divide large three-dimensional datasets into memory-sized chunks and +process them in parallel across CPUs or multiple GPUs. Performance comes from +distributed I/O, MPI-based in-memory operations, GPU-accelerated methods and +CuPy device-to-device processing. + +The framework orchestrates methods supplied by external CPU and GPU processing +libraries rather than implementing the scientific methods itself. Users combine +those methods in reusable YAML :term:`pipelines `. + +.. figure:: ../_static/httomo-workflow-dark.png + :scale: 40 % + :alt: Simple tomographic pipeline + + Parallel-beam projection data can be divided between several workers for + processing and reconstruction. diff --git a/docs/source/introduction/concepts/chunks_blocks.rst b/docs/source/introduction/concepts/chunks_blocks.rst new file mode 100644 index 000000000..3126ebc6b --- /dev/null +++ b/docs/source/introduction/concepts/chunks_blocks.rst @@ -0,0 +1,133 @@ +.. _chunks_blocks_data: + +Chunks and blocks +================= + +HTTomo divides data at two levels. The dataset is first distributed across MPI +processes as *chunks*. Each chunk is then divided into memory-sized *blocks* for +processing. + +.. _chunks_data: + +Chunks +------ + +A *chunk* is the part of the dataset assigned to one MPI process. Distributing +chunks allows multiple processes to work on the data in parallel. + + +.. admonition:: on "chunk" terminology + :class: note + + *chunk* is purely an HTTomo term and is unrelated to HDF5 chunks. + + +.. _fig_chunks: + +.. figure:: ../../_static/blocks_chunks/chunks.png + :alt: Tomographic data distributed as chunks between two MPI processes + :align: center + :width: 90% + + Data with shape :code:`(180, 128, 160)` distributed between two MPI + processes. Each process receives a chunk of shape + :code:`(90, 128, 160)`. + +Chunk size +~~~~~~~~~~ + +Chunk size depends on the full dataset shape, its current projection or sinogram +orientation, and the number of MPI processes. + +HTTomo divides the data as evenly as possible. If an equal division is not +possible, the highest-rank MPI process receives the differently sized chunk. + +.. _blocks_data: + +Blocks +------ + +A *block* is a smaller piece of a :ref:`chunk ` or equal to the size of chunk. HTTomo processes +blocks individually so that the data fits within the available CPU or GPU memory. + +Block size +~~~~~~~~~~ + +HTTomo calculates block size at runtime using: + +- the available memory +- the memory requirements of every method in the current + :ref:`section ` + +Block size may change between sections. If sufficient memory is available, a +single block can contain the entire chunk. + +.. _fig_blocks: + +.. figure:: ../../_static/blocks_chunks/blocks.png + :alt: An HTTomo chunk divided into smaller blocks + :align: center + :width: 90% + + A chunk of shape :code:`(90, 128, 160)` divided into two blocks of shape + :code:`(45, 128, 160)`. Each block contains 45 projections and is processed + individually. + +Processing blocks +~~~~~~~~~~~~~~~~~ + +Blocks are HTTomo's main processing unit. Loaders produce blocks, methods process +them and the resulting blocks continue through the pipeline. + +When a chunk contains multiple blocks, they are processed sequentially. + +.. dropdown:: More details about block processing + + **Notes on the framework's approach to data** + + HTTomo's framework has been written with GPUs in mind. More specifically, + HTTomo aims to use as much available GPU memory as possible while remaining + within a safe limit. + + Even after data is divided into chunks, a chunk may not fit into GPU memory. + Similarly, the full dataset may not fit into CPU memory. This can occur on both + compute clusters and personal machines. + + **Why split a chunk into smaller pieces?** + + Each MPI process works with one chunk and is typically associated with one GPU. + HTTomo cannot assume that the entire chunk will fit into that GPU's memory. + + For example, dividing a 20 GB dataset among four MPI processes produces chunks + of approximately 5 GB. A cluster GPU may have enough memory for an entire chunk, + while a personal GPU with 4 GB of memory would not. + + HTTomo therefore divides each chunk into smaller pieces called *blocks*. + + **How are block shapes calculated?** + + HTTomo calculates block sizes during pipeline execution. Blocks within a + sequence of methods have approximately the same shape, but their shape may + change between pipeline stages. + + The calculation uses information from the :ref:`library files ` and + is mainly based on: + + - the GPU memory available to the process + - the memory requirements of the methods in the current + :ref:`section ` + + Block size is expressed as a number of slices. For projection data, each process + owns a chunk of projections and divides it into blocks containing a suitable + number of projection slices. The same principle applies to sinogram data. + + **Blocks as the fundamental data quantity** + + Blocks are HTTomo's main processing unit: + + #. Loaders produce individual blocks. + #. Methods receive individual blocks as input. + #. Methods produce individual blocks as output. + + If an entire chunk fits into memory, a single block may span the whole chunk. + HTTomo still treats it as a block. \ No newline at end of file diff --git a/docs/source/introduction/concepts/memory_estimators.rst b/docs/source/introduction/concepts/memory_estimators.rst new file mode 100644 index 000000000..47b23b6cd --- /dev/null +++ b/docs/source/introduction/concepts/memory_estimators.rst @@ -0,0 +1,41 @@ +.. _info_memory_estimators: + +GPU memory estimation +===================== + +HTTomo uses GPU memory estimators to determine how many data slices can be processed +at once. This defines the size of the :ref:`blocks ` used within each +:ref:`section `. + +.. _fig_gpu_memory_estimation: + +.. figure:: ../../_static/memory_estimators/gpu_memory_estimation_vector.png + :alt: GPU memory estimation and block-size selection in HTTomo + :align: center + :width: 100% + + HTTomo selects a safe block size that satisfies the memory requirements of + every method in a section. + +Selecting a block size +~~~~~~~~~~~~~~~~~~~~~~ + +HTTomo first determines the available memory on the selected GPU. For a candidate +block size, each GPU method in the section estimates its peak memory use, including +its inputs, outputs and temporary allocations. + +The most memory-demanding method limits the block size. HTTomo selects a block +that fits the available memory for every method, then uses that size throughout +the section. Iterative estimators may return a safe near-maximum value rather +than the exact largest possible block. + +Why block size matters +~~~~~~~~~~~~~~~~~~~~~~ + +Larger blocks use the GPU more efficiently and reduce the number of data transfers +and method calls. The memory estimators maximise block size while preventing +out-of-memory failures. + +Memory requirements vary between methods and may depend on the data shape, data +type and method parameters. See :ref:`developers_memorycalc` for details about +implementing and testing GPU memory estimators. diff --git a/docs/source/introduction/concepts/process_templates.rst b/docs/source/introduction/concepts/process_templates.rst new file mode 100644 index 000000000..b6e8832c4 --- /dev/null +++ b/docs/source/introduction/concepts/process_templates.rst @@ -0,0 +1,84 @@ +.. _explanation_pipelines_templates: + +Pipelines and templates +======================= + +An HTTomo *pipeline* is an ordered sequence of data-loading and processing +operations described using :ref:`YAML ` syntax. Older +material may call a pipeline a *process list*. + +The pipeline is assembled from reusable *YAML templates*. Each template +configures one loader or a processing method. + +.. _explanation_templates: + +YAML templates +-------------- + +A method template contains: + +- ``method``: the function to execute +- ``module_path``: the Python module containing the function +- ``parameters``: the values passed to the function + +For example: + +.. code-block:: yaml + + - method: median_filter3d + module_path: tomopy.misc.corr + parameters: + size: 3 + +This entry tells HTTomo to import ``median_filter3d`` from ``tomopy.misc.corr`` +and run it with ``size=3``. + +Data parameters such as the input array are not included because HTTomo manages +data flow through its :ref:`wrapper layer `. + +Ready-to-use templates are available in :ref:`reference_templates` and are +generated by the `HTTomo-backends YAML generator +`_. +See :ref:`developers_httomo_backends` for the developer workflow. + +.. _explanation_process_list: + +Pipelines +--------- + +A pipeline combines templates in execution order. It must begin with a loader; +the remaining methods receive the output of the preceding operation. + +For example: + +.. code-block:: yaml + + - method: standard_tomo + module_path: httomo.data.hdf.loaders + parameters: + data_path: entry1/tomo_entry/data/data + image_key_path: entry1/tomo_entry/instrument/detector/image_key + rotation_angles: + data_path: entry1/tomo_entry/data/rotation_angle + + - method: normalize + module_path: tomopy.prep.normalize + parameters: + cutoff: null + averaging: mean + + - method: minus_log + module_path: tomopy.prep.normalize + parameters: {} + +HTTomo reads this pipeline from top to bottom: it loads the data, normalises +it and then applies the negative logarithm. + +Next steps +---------- + +- See :ref:`howto_process_list` to build and configure a pipeline. +- Browse :ref:`reference_templates` for supported method templates. +- See :ref:`explanation_yaml` for YAML syntax. +- Use the :ref:`YAML checker ` before running a pipeline. +- Browse :ref:`tutorials_pl_templates` for complete example pipelines. diff --git a/docs/source/introduction/concepts/reslice.rst b/docs/source/introduction/concepts/reslice.rst new file mode 100644 index 000000000..08a711b5b --- /dev/null +++ b/docs/source/introduction/concepts/reslice.rst @@ -0,0 +1,37 @@ +.. _info_reslice: + +Re-slicing +========== + +A *re-slice* changes how tomographic data is divided and accessed. It is required +when consecutive :ref:`sections ` use different data patterns, +typically when switching between projections and sinograms. + +.. _fig_reslice: + +.. figure:: ../../_static/reslice_gather/httomo_reslicing_dark.png + :alt: Projection data re-sliced and redistributed into sinogram-oriented chunks + :align: center + :width: 100% + + Projection-oriented chunks are re-sliced and redistributed into + sinogram-oriented chunks. + +How re-slicing works +~~~~~~~~~~~~~~~~~~~~ + +Each sinogram is formed by selecting the same detector row from every projection +angle. HTTomo redistributes these rows among the MPI processes to create +sinogram-oriented chunks. + +Re-slicing is performed in CPU memory, so data held on GPUs must first be +transferred back to the host. If sufficient CPU memory is unavailable, HTTomo uses +temporary disk storage instead. + +Performance considerations +~~~~~~~~~~~~~~~~~~~~~~~~~~ + +Re-slicing can be expensive for large datasets, particularly when temporary disk +storage is required. Pipelines should therefore group methods by data pattern to +minimise pattern changes. A typical preprocessing and reconstruction pipeline +requires only one transition from projections to sinograms. \ No newline at end of file diff --git a/docs/source/introduction/concepts/sections.rst b/docs/source/introduction/concepts/sections.rst new file mode 100644 index 000000000..68b146922 --- /dev/null +++ b/docs/source/introduction/concepts/sections.rst @@ -0,0 +1,51 @@ +.. _info_sections: + +Sections +======== + +A *section* is a consecutive group of pipeline methods that operate on the same +data pattern or slicing orientation. HTTomo creates sections automatically to +organise data processing and movement efficiently. + +.. _fig_sections_concept: + +.. figure:: ../../_static/sections/sections_concept.png + :alt: An HTTomo pipeline divided into sections + :align: center + :width: 100% + + Methods are grouped by data pattern. A pattern change, data saving or + side-output dependency introduces a new section. + +Processing sections +~~~~~~~~~~~~~~~~~~~ + +HTTomo processes one section at a time. It calculates a :ref:`blocks_data` +size that satisfies the memory requirements of every method in the section, +using :ref:`info_memory_estimators`, then passes each block through those +methods in sequence. + +A new section begins when: + +- the data pattern changes between projections and sinograms; +- a method depends on a side output produced by an earlier method; +- a method's output needs to be saved to disk; or +- another method requiring padded data is encountered. + +When the data pattern changes, HTTomo also :ref:`re-slices ` the +data before processing the next section. + +Example +~~~~~~~ + +In :numref:`fig_sections_concept`, Methods 1 and 2 operate on projections and +form the first section. Method 3 requires sinograms, so HTTomo re-slices the +data and starts a second section. Method 4 depends on a side output from Method +3, creating a third section that also contains Method 5. + +Performance considerations +~~~~~~~~~~~~~~~~~~~~~~~~~~ + +Section boundaries may require process synchronisation, data redistribution or +temporary storage. Pipelines with fewer sections are therefore generally more +efficient, although some boundaries are required by the selected methods. diff --git a/docs/source/introduction/concepts/wrappers.rst b/docs/source/introduction/concepts/wrappers.rst new file mode 100644 index 000000000..31705cc1d --- /dev/null +++ b/docs/source/introduction/concepts/wrappers.rst @@ -0,0 +1,39 @@ +.. _info_wrappers: + +Method wrappers +=============== + +HTTomo uses processing methods from external :ref:`backend libraries +`. Because these methods accept different inputs and produce +different outputs, HTTomo uses *wrappers* to provide a consistent interface between +the pipeline steps and each method action. + +Wrappers prepare method inputs, handle CPU/GPU data transfers, invoke the backend +method and manage its outputs. Supporting information from :ref:`developers_httomo_backends` +describes requirements such as the data pattern, implementation, memory usage and +padding. + +.. _fig_wrappers: + +.. figure:: ../../_static/wrappers/httomo_wrappers.png + :alt: HTTomo wrapper layer and its five main wrapper types + :align: center + :width: 100% + + The HTTomo wrapper layer connects backend methods to the processing pipeline and + handles their method-specific inputs and outputs. + +Wrapper types +~~~~~~~~~~~~~ + +HTTomo provides five main wrapper types: + +- **Generic:** passes data to standard processing methods and returns the result. +- **Normalisation:** supplies projection data, flats and darks, then returns + normalised data. +- **Centering:** prepares the required projections or sinograms and produces a + centre-of-rotation value. +- **Reconstruction:** supplies the data, projection angles and reconstruction + parameters, including the centre of rotation. +- **Image saver:** transfers data to the CPU when necessary and writes images to + storage without changing the pipeline data. \ No newline at end of file diff --git a/docs/source/introduction/data_proc_concepts.rst b/docs/source/introduction/data_proc_concepts.rst new file mode 100644 index 000000000..97e2c2861 --- /dev/null +++ b/docs/source/introduction/data_proc_concepts.rst @@ -0,0 +1,16 @@ +.. _detailed_about: + +Core concepts +============= + +These pages explain the terms used to describe a pipeline and how HTTomo moves +data through it. Implementation details about wrappers and memory-estimator +interfaces belong to :ref:`developer_architecture`. + +.. toctree:: + :maxdepth: 2 + + concepts/process_templates + concepts/chunks_blocks + concepts/sections + concepts/reslice diff --git a/docs/source/introduction/execution_model.rst b/docs/source/introduction/execution_model.rst new file mode 100644 index 000000000..0d21e215e --- /dev/null +++ b/docs/source/introduction/execution_model.rst @@ -0,0 +1,49 @@ +How HTTomo runs a pipeline +========================== + +HTTomo reads the YAML pipeline, validates its methods and parameters, and +constructs an executable pipeline. Method wrappers adapt backend functions to +HTTomo's common data-processing interface. + +.. _fig_execution_model: + +.. figure:: ../_static/execution_model.svg + :width: 100% + :alt: Diagram of the HTTomo pipeline execution model + + HTTomo execution from a YAML pipeline to output. The pipeline is divided + into sections; within each section, data is distributed into chunks and + processed block by block. + +Creating sections +~~~~~~~~~~~~~~~~~ + +HTTomo groups compatible methods into :ref:`sections `. For each +section, it calculates a safe block size from the available memory and the +requirements of its methods. + +Executing a section +~~~~~~~~~~~~~~~~~~~ + +Within each section, the dataset is divided among the available MPI processes. +Each process receives one :ref:`chunk ` and normally operates +independently on its assigned data. + +Each chunk is then divided into :ref:`blocks ` using the block size +calculated for the section. A block passes through every method in the section +before the next block is processed. + +Moving between sections +~~~~~~~~~~~~~~~~~~~~~~~ + +At a section boundary, HTTomo may synchronise the MPI processes, save +intermediate data or redistribute the dataset. If the data pattern changes +between projections and sinograms, HTTomo performs a +:ref:`re-slice `. + +Completing the pipeline +~~~~~~~~~~~~~~~~~~~~~~~ + +Processing continues section by section until every block has passed through +the pipeline. HTTomo then writes the requested results and monitoring +information. diff --git a/docs/source/introduction/indepth/blocks.rst b/docs/source/introduction/indepth/blocks.rst deleted file mode 100644 index 9a2178db4..000000000 --- a/docs/source/introduction/indepth/blocks.rst +++ /dev/null @@ -1,127 +0,0 @@ -.. _blocks_data: - -Blocks -====== - -Definition -~~~~~~~~~~ - -When a chunk (see :ref:`chunks_data`) is split into smaller pieces, these smaller pieces are called -*blocks*. - -.. _fig_blocks1: -.. figure:: ../../_static/blocks_chunks/chunks_blocks.png - :scale: 45 % - :alt: Chunks and blocks - - An example of the data being divided into 4 parallel processes which are associated with 4 :ref:`chunks_data` of data. In each chunk there could be one or a number of blocks, depending on the information provided by :ref:`info_memory_estimators`. - - -Motivation -~~~~~~~~~~ - -Notes on the framework's approach to data -''''''''''''''''''''''''''''''''''''''''' - -HTTomo's framework has been written with the use of GPUs in mind. More -specifically, HTTomo is geared towards filling up the available GPU memory with as -much data as possible, within a reasonable tolerance to the limit. - -HTTomo's framework has also been written with the idea in mind that data, even -after splitting into chunks, may or may not fit into GPU memory. This applies to -both most commonly used hardware setups: compute clusters and personal machines. - -Similarly, data may or may not even fit into CPU memory (RAM) without splitting it -into smaller pieces and holding only one piece in memory at a time. This also -applies to both most commonly used hardware setups (as surprising as it may first -sound, there is indeed the case where data not fitting in RAM can happen even with -nodes in compute clusters!). - -Why split a "chunk" into smaller pieces? -'''''''''''''''''''''''''''''''''''''''' - -Each MPI process is associated with one GPU, and earlier it was mentioned that each -MPI process has one chunk to work with. So, each MPI process has the task of trying -to fit as much of its chunk into the memory of the GPU it is using, for every -method in the pipeline. - -Highlighting one use-case, in order for HTTomo to be usable on personal machines as -well as compute clusters, it's necessary to *not assume* that a single chunk can -fit into GPU memory. For example, data of size 20GB when running with four MPI -processes would result in chunks of size ~5GB. Compute clusters equipped with GPUs -would likely have GPU models with memory far exceeding 5GB, and so could easily fit -the entire 5GB chunk into GPU memory. However, a personal machine with a discrete -GPU may only have, say, 4GB of memory; in which case, a 5GB chunk wouldn't fit in -GPU memory. - -To generically handle the possibility that a single chunk may not fit into GPU -memory (among other reasons), a chunk is split into smaller pieces. The pieces that -a chunk is split into are called *blocks*. - -How are block shapes calculated? -~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ - -The size of a block when splitting a chunk varies on a case-by-case basis, and is a -calculation that the HTTomo framework performs during pipeline execution. - -More specifically, when a chunk is split into multiple blocks for a sequence of -methods, each block is roughly the same shape. However, at another stage in the -pipeline with a different sequence of methods, a chunk may be split into multiple -blocks where the block shape is *different* to the previous block shape. - -The block size calculation uses information in the :ref:`library -files`, as well as the GPU memory that is available at the time of -calculation. - -The size of a block is mainly based on: - -- the GPU memory available to the process -- a method's memory requirements (more precisely, the memory requirements of the - GPU methods in a section, see :ref:`info_sections` for information on what - "sections" are) - -At a high-level, the size of a block is given in terms of the number of slices that -it contains. For example, if the data was split and distributed among the MPI -processes as projections, then: - -- each process has a *chunk* of projections -- when each process splits its chunk into *blocks*, each block will contain a - certain number of projection slices - -The analagous explanation for the sinogram case, replacing "projections" with -"sinograms" in the above, holds true too. - -Blocks as the fundamental data quantity in HTTomo -~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ - -Blocks are the level at which most objects in HTTomo interact with, for example: - -1. loaders load individual blocks -2. methods take individual blocks as input -3. methods produce individual blocks as output - -There are indeed edge cases where a block can span the entire chunk from which it -comes from, in which case the chunk an MPI process has and the single block from it -have the same shape. However, as far as the framework knows, it is dealing with -blocks. - -Therefore, a block could be considered as the fundamental data quantity in HTTomo's -framework. - -Example -~~~~~~~ - -The example given in the chunks section had started with: - -- the input 3D data has shape :code:`(180, 128, 160)` -- HTTomo has been executed with two MPI processes -- each chunk has shape :code:`(90, 128, 160)` - -Continuing on from this, suppose that the GPU for both processes was such that only -half of the stack of projections could fit into the GPU's memory at one time. Ie, a -:code:`(45, 128, 160)` subset of a chunk could fit into GPU memory and no more. - -Each MPI process would split its chunk of shape :code:`(90, 128, 160)` into two -*blocks* of shape :code:`(45, 128, 160)`. Very roughly speaking, after this -splitting, each MPI process would then proceed to pass each :code:`(45, 128, 160)` -block into the current method. diff --git a/docs/source/introduction/indepth/chunks.rst b/docs/source/introduction/indepth/chunks.rst deleted file mode 100644 index 0fefdd966..000000000 --- a/docs/source/introduction/indepth/chunks.rst +++ /dev/null @@ -1,49 +0,0 @@ -.. _chunks_data: - -Chunks -====== - -Definition -~~~~~~~~~~ - -When data is split into pieces and distributed among the MPI processes (one piece -per process), these pieces are called *chunks*. - -Motivation: distributing data across MPI processes -~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ - -HTTomo is able to run with multiple processes using MPI. The main idea is that -HTTomo is given input data, and each process gets a subset of the input data. Thus, -the input data is split, and each MPI process gets one piece. The pieces that the -input data is split into are called *chunks*. So, in this terminology, each MPI -process has one chunk to work with. (Again, the term "chunk" here shouldn't be -confused with the hdf5 notion of a chunk!) - -How are chunk shapes calculated? -~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ - -The chunk shape calculation is simple and is based on: - -- the number of MPI processes HTTomo is launched with -- the shape of the "full data" - -The data is split such that each chunk is as close to the same shape as other -chunks. If the data is being split as projections, then each MPI process gets a -chunk with roughly the same number of projections. Similarly, if the data is being -split as sinograms, then each MPI process gets a chunk with roughly the same number -of sinograms. - -.. note:: If the data doesn't split evenly, then the MPI process with the largest - rank is the one that gets a chunk with the shape that's the "odd one out" - -Example -~~~~~~~ - -Consider 3D input data with shape :code:`(180, 128, 160)` (ie, 180 projections, -where each projection has dimensions :code:`(128, 160)`), and running HTTomo with -two MPI processes. - -If evenly splitting the data along the first axis of length :code:`180`, this -results in two pieces, each with shape :code:`(90, 128, 160)`. Each piece would be -referred to as a "chunk" in HTTomo, and each MPI process would get one of these -:code:`(90, 128, 160)` shaped chunks. diff --git a/docs/source/introduction/indepth/detailed_about.rst b/docs/source/introduction/indepth/detailed_about.rst deleted file mode 100644 index cf5949c71..000000000 --- a/docs/source/introduction/indepth/detailed_about.rst +++ /dev/null @@ -1,42 +0,0 @@ -.. _detailed_about: - -Detailed concepts -+++++++++++++++++ - -Here we present more detailed concepts of HTTomo's framework, such as, -:ref:`info_sections`, :ref:`info_reslice`, :ref:`info_data` (:ref:`chunks_data`, :ref:`blocks_data`), :ref:`info_wrappers`, -:ref:`info_memory_estimators`, and others. - -.. toctree:: - :maxdepth: 2 - - sections - reslice - wrappers - memory_estimators - -.. _info_data: - -Data terminology ----------------- - -HTTomo's framework deals with handling all the data involved in executing the -pipeline. "Data handling" in HTTomo involves operations such as splitting data up -into pieces, passing the pieces of data into methods, gathering data up and -re-splitting into a different set of pieces, and more. - -The concept of data being split into smaller pieces is a common theme in HTTomo, -and there naturally arose two main "levels" of data splitting. These two levels -have been given names, as an easy way to give some context to a piece of data when -referring to it. - -One level of data splitting produces a piece of data called a *chunk* (not to be -confused with the term "chunk" used in the hdf5 data format), and the other level -of data splitting produces a piece of data called a *block*. - - -.. toctree:: - :maxdepth: 2 - - chunks - blocks \ No newline at end of file diff --git a/docs/source/introduction/indepth/memory_estimators.rst b/docs/source/introduction/indepth/memory_estimators.rst deleted file mode 100644 index 2560a6992..000000000 --- a/docs/source/introduction/indepth/memory_estimators.rst +++ /dev/null @@ -1,30 +0,0 @@ -.. _info_memory_estimators: - -Memory Estimators ------------------ - -Memory estimators is the integral part of HTTomo when it comes to the GPU processing and efficient device-to-device transfers. - -.. note:: One needs to know how much of the data safely fits a GPU card. This is the main purpose of the GPU memory estimators in HTTomo. - -Based on the free memory available in the GPU device and memory requirements of a particular method, GPU memory estimators will initiate -:ref:`developers_memorycalc` to decide how much of the data can be safely transferred to the GPU for processing. The size of the data -that can be fitted in the GPU card for computations will define the size of the :ref:`blocks_data`. - -.. _fig_memest1: -.. figure:: ../../_static/memory_estimators/memoryest_1.png - :scale: 40 % - :alt: Memory Estimators - - Free memory of the GPU device and the method's memory requirement is the information that is required for memory estimators to define the size of the :ref:`blocks_data` in the :ref:`info_sections`. - -As one would like to minimise device-to-device transfers while doing GPU computations, HTTomo tries to chain the methods together in one Section. -Therefore if there are multiple methods chained together, the size of the block will be defined by the most memory-demanding method. -Memory estimators will inspect every method in that methods' chain in order to decide which one dominates. - -.. _fig_memest2: -.. figure:: ../../_static/memory_estimators/memoryest_2.png - :scale: 40 % - :alt: Memory Estimators - - A chain of GPU methods will be executed in loop over blocks. GPU memory estimators will make sure that the memory requirements will be satisfied for all of them. \ No newline at end of file diff --git a/docs/source/introduction/indepth/reslice.rst b/docs/source/introduction/indepth/reslice.rst deleted file mode 100644 index 6147093aa..000000000 --- a/docs/source/introduction/indepth/reslice.rst +++ /dev/null @@ -1,28 +0,0 @@ -.. _info_reslice: - -Re-slicing ----------- -The re-slicing of data happens when we need to access a slice which is orthogonal to the current one. -In tomography, we normally work in the space of projections or in the space of sinograms. Different methods require different slicing -orientations, or, as we call it, a *pattern*. The change of the pattern is a **re-slice** operation or a transformation of an array by -re-slicing in a particular direction. For instance, from the projection space/pattern to the sinogram space/patterns, as in the figure below. - -.. _fig_reslice: -.. figure:: ../../_static/reslice_gather/reslice.png - :scale: 40 % - :alt: Reslicing procedure - - The re-slicing operation for tomographic data. Here the data is resliced from the stack of projections to the stack of sinograms. - -In HTTomo, the re-slicing operation is performed on the CPU as we need to access all the data. Even if the pipeline consists of only GPU methods stacked together, -the re-slicing step will transfer the data from the GPU device to the CPU memory first. This operation can be costly for big datasets and we recommend to minimise the number of -re-slicing operations in your pipeline. Normally for tomographic pre-processing and reconstruction there is just one re-slice needed, please see how :ref:`howto_process_list`. - -.. _fig_reslice2: -.. figure:: ../../_static/reslice_gather/gather_chunks.png - :scale: 30 % - :alt: Reslicing procedure - - Data needs to be gathered from the GPU devices to transfer to the CPU memory (or a disk) in order to perform the re-slicing procedure. - -.. note:: Note that when the CPU memory is not enough to perform re-slicing operation, the operation will be performed through the disk. This is substantially slower and the processing becomes heavily I/O bounded. diff --git a/docs/source/introduction/indepth/sections.rst b/docs/source/introduction/indepth/sections.rst deleted file mode 100644 index 3da7e5835..000000000 --- a/docs/source/introduction/indepth/sections.rst +++ /dev/null @@ -1,64 +0,0 @@ -.. _info_sections: - -Sections --------- - -Sections is the fundamental concept of the HTTomo's framework which is related to how the I/O operations and processing of data is organised. - -.. note:: The main purpose of a section is to organise the data input/output workflow, as well as, chain together the methods so that the constructed pipeline is computationally efficient. - -To better understand the purpose of the section it is also useful to read information about :ref:`chunks_data`, :ref:`blocks_data` and :ref:`info_memory_estimators`. - -Bellow we present different situations that can lead to the sections being organised in a specific manner. - -.. _fig_sec1: -.. figure:: ../../_static/sections/sections1.png - :scale: 40 % - :alt: Sections in pipelines - - Here is a typical pipeline with a loader (`L`), 5 methods (`M`), and 4 data transfer operations (`T`) between methods. - -Sections are created when: - -1. :ref:`info_reslice` is needed, which is related to the change of pattern. -2. The output of the method needs to be saved to the disk. -3. The :ref:`side_output` is required by one of the methods. - -Example 1: Sections with re-slice -================================= - -.. _fig_sec2: -.. figure:: ../../_static/sections/sections2.png - :scale: 50 % - :alt: Sections in pipelines - - Let us say that the pattern in methods `M`\ :sub:`1-3` is *projection* and methods in `M`\ :sub:`4-5` belong to *sinogram* pattern. - This will result in two sections created and also :ref:`info_reslice` operation in the data transfer `T`\ :sub:`3` layer. - -Example 2 : Sections with re-slice and data saving -================================================== - -.. _fig_sec3: -.. figure:: ../../_static/sections/sections3.png - :scale: 50 % - :alt: Sections in pipelines - - In addition Example 1 situation, let us assume that we want to save the result of `M`\ :sub:`2` method to the disk. - This means that even though `M`\ :sub:`1-3` methods can be performed on the GPU, the data will be transferred to CPU. - The pipeline will be further fragmented to introduce another section, so that the data transfer `T`\ :sub:`2` layer also saves the data on the - disk, as well as, taking care to return the data back on the GPU for the method `M`\ :sub:`3`. - -Example 3 : Sections with side outputs -====================================== - -.. _fig_sec4: -.. figure:: ../../_static/sections/sections4.png - :scale: 50 % - :alt: Sections in pipelines - - Consider, again, Example 1. Here, however, `M`\ :sub:`5` requests the :ref:`side_output` of the method `M`\ :sub:`4`. - This, for example, can be because the reconstruction method `M`\ :sub:`5` requires the :ref:`centering` value of `M`\ :sub:`4`, where - this value is calculated. This divides `M`\ :sub:`4` and `M`\ :sub:`5` into separate sections. Also notice that `M`\ :sub:`1` needs the data - to be saved on disk, so in total, it is a pipeline with 4 sections in it. - -.. note:: It can be seen that creating more sections in pipelines is to be avoided when building an efficient pipeline. Creating a section usually leads to synchronisation of all processes on the CPU and potentially, if not enough memory, through-disk operations. diff --git a/docs/source/introduction/indepth/wrappers.rst b/docs/source/introduction/indepth/wrappers.rst deleted file mode 100644 index 9a102cb56..000000000 --- a/docs/source/introduction/indepth/wrappers.rst +++ /dev/null @@ -1,26 +0,0 @@ -.. _info_wrappers: - -Method Wrappers -=============== - -HTTomo does not contain image processing methods and uses external libraries for processing, see :ref:`backends_list`. -However, because methods process data in a different manner, there's a need to tell the framework what kind of processing that is. - -HTTomo currently has the following wrappers: - -1. The generic wrapper. Suitable when the method does not change the dimensions of the data for its output and requires one input and one output. This also can be called a filter. - -2. Normalisation wrapper is to deal with normalisation methods as they have supplementary data as an input (e.g. flats/darks) and one dataset as an output. - -3. Rotation or Centering wrapper. These are special methods to estimate :ref:`centering` automatically. The input can be quite specific, e.g. normalised sinogram or selected projections, and the output is usually a scalar. - -4. Reconstruction wrapper. Reconstruction needs an additional information to be passed like angles and the centre of rotation. This is all handled by the framework in the wrapper. - -5. Image Saving wrapper. This wrapper is needed because no output in HTTomo is produced, we just save images to the disk. - -.. _fig_wrappers: -.. figure:: ../../_static/wrappers/wrappers.png - :scale: 40 % - :alt: Different wrappers of HTTomo - - Wrappers of HTTomo deal with the i/o differences in methods from the external libraries. diff --git a/docs/source/pipelines/choose_pipeline.rst b/docs/source/pipelines/choose_pipeline.rst new file mode 100644 index 000000000..dc8bb386e --- /dev/null +++ b/docs/source/pipelines/choose_pipeline.rst @@ -0,0 +1,80 @@ +.. _choose_pipeline: + +Choose a pipeline +================= + +Start with the pipeline that most closely matches the hardware, scan geometry +and reconstruction goal. Then update its loader paths and processing +parameters for the input data. The table below describes the examples on +:ref:`tutorials_pl_templates`; it is guidance rather than a performance +ranking. + +.. list-table:: + :header-rows: 1 + :widths: 22 11 14 24 29 + + * - Starting pipeline + - Hardware + - Scan or mode + - Best starting point for + - Main dependencies + * - ``FBP3d_tomobar`` + - GPU + - 180°, analytical + - Routine reconstruction with centring and stripe removal + - HTTomolibGPU, TomoBAR and HTTomolib + * - ``titaren_center_pc_FBP3d_resample`` + - GPU + - 180°, analytical + - Phase-correlation centring and downsampled output + - HTTomolibGPU, TomoBAR and HTTomolib + * - ``LPRec3d_tomobar`` + - GPU + - 180°, analytical + - Trying LPRec3d on compatible parallel-beam data + - HTTomolibGPU, TomoBAR and HTTomolib + * - ``FBP3d_tomobar_denoising`` + - GPU + - 180°, analytical + - FBP followed by total-variation denoising + - HTTomolibGPU, TomoBAR and HTTomolib + * - ``FISTA3d_tomobar`` + - GPU + - 180°, iterative + - Noisy or undersampled data where regularisation is useful + - HTTomolibGPU, TomoBAR and HTTomolib + * - ``deg360_paganin_FBP3d_tomobar`` + - GPU + - 360°, analytical + - Overlap finding, conversion to 180° and Paganin filtering + - HTTomolibGPU, TomoBAR and HTTomolib + * - ``deg360_distortion_FBP3d_tomobar`` + - GPU + - 360°, analytical + - Optical-distortion correction before 360° conversion + - HTTomolibGPU, TomoBAR and HTTomolib + * - ``tomopy_gridrec`` + - CPU + - 180°, analytical + - A small CPU run or a system without a CUDA-capable GPU + - TomoPy and HTTomolib + * - Sweep examples + - GPU + - Parameter search + - Comparing centre-of-rotation or Paganin values + - HTTomolibGPU, TomoBAR and HTTomolib + +Before running an example: + +#. Confirm that its libraries are installed; see :ref:`backends_list`. +#. Replace loader paths or use NXtomo automatic discovery; see + :ref:`loading_data`. +#. Review method parameters against :ref:`reference_templates`. +#. Validate the result with ``python -m httomo check PIPELINE INPUT``. +#. Use the archives in :ref:`versioned_downloads` for a tagged HTTomo release. + +Choose analytical reconstruction for a fast baseline. Iterative reconstruction +is more computationally expensive but can be valuable for noisy or incomplete +data. Actual speed depends on data dimensions, parameters, hardware, process +count and storage, so benchmark representative data rather than relying on a +general ranking. diff --git a/docs/source/pipelines/reconstruction_ecosystem.rst b/docs/source/pipelines/reconstruction_ecosystem.rst new file mode 100644 index 000000000..5cf8470cc --- /dev/null +++ b/docs/source/pipelines/reconstruction_ecosystem.rst @@ -0,0 +1,138 @@ +.. _reconstruction_ecosystem: + +Reconstruction ecosystem +======================== + +HTTomo exposes two main reconstruction pathways: a CPU pathway through +TomoPy and a GPU pathway through HTTomolibGPU and TomoBAR. Their pipeline +entries have the same ``method`` and ``module_path`` structure, but the +packages below those entries have different responsibilities. + +The ``module_path`` identifies the public function called by HTTomo. It does +not necessarily identify the library that performs every numerical operation. +For example, a pipeline calls ``httomolibgpu.recon.algorithm.FBP3d_tomobar``; +HTTomolibGPU prepares the call, while TomoBAR and ASTRA perform parts of the +reconstruction. + +Pathways at a glance +-------------------- + +.. list-table:: CPU and GPU reconstruction pathways + :header-rows: 1 + :widths: 14 23 12 27 24 + + * - Pathway + - Pipeline entry + - Execution + - Implementation + - Main dependencies + * - TomoPy + - ``tomopy.recon.algorithm.recon`` + - CPU, using NumPy arrays + - TomoPy selects the reconstruction implementation from its + ``algorithm`` parameter; ``gridrec`` is used by the CPU example + pipeline. + - TomoPy and its numerical/compiled dependencies; HTTomo distributes + work between MPI processes. + * - TomoBAR + - ``httomolibgpu.recon.algorithm.`` + - NVIDIA GPU; most 3D methods use CuPy arrays + - HTTomolibGPU provides the public pipeline methods. TomoBAR supplies + direct and iterative reconstruction, using ASTRA or its CuPy Fourier + implementation according to the method. + - HTTomolibGPU, TomoBAR, ASTRA Toolbox, CuPy, CUDA and a compatible + NVIDIA driver. + +``httomo-backends`` supports both pathways, but it is not a numerical +reconstruction library. It provides HTTomo with method metadata, such as the +processing pattern, memory estimator, padding and output dimensions, and it +generates the :ref:`method templates `. + +The GPU pathway +--------------- + +The diagram shows how HTTomolibGPU presents one pipeline-facing API above +the GPU reconstruction components. ASTRA supplies projection and +backprojection operators, TomoBAR supplies reconstruction algorithms and +CuPy provides CUDA-compatible arrays and kernels. Not every method uses every +component: the log-polar method, for example, follows TomoBAR's Fourier/CuPy +route rather than the ASTRA route. + +.. figure:: ../_static/gpu_reconstruction_httomo.svg + :alt: HTTomolibGPU above ASTRA Toolbox, TomoBAR and CuPy. ASTRA supplies GPU projection operators, TomoBAR supplies reconstruction algorithms, and CuPy supplies device arrays and CUDA kernels. + :width: 100% + :align: center + + The GPU reconstruction layers used by HTTomo. The boxes link to the + corresponding project documentation. + +.. list-table:: GPU method implementations + :header-rows: 1 + :widths: 24 14 20 42 + + * - HTTomolibGPU method + - Type + - Data and execution model + - Numerical implementation + * - ``FBP2d_astra`` + - Analytical FBP + - GPU, reconstructed slice by slice; NumPy input and output + - TomoBAR's two-dimensional direct-method wrapper calls ASTRA's + ``FBP_CUDA`` implementation. + * - ``FBP3d_tomobar`` + - Analytical FBP + - GPU volume using CuPy arrays + - TomoBAR applies CuPy-based filtering and uses ASTRA for GPU + backprojection. + * - ``LPRec3d_tomobar`` + - Analytical log-polar + - GPU volume using CuPy arrays + - TomoBAR performs Fourier inversion on log-polar grids using CuPy; this + computational route does not use ASTRA projection operators. + * - ``SIRT3d_tomobar`` and ``CGLS3d_tomobar`` + - Iterative + - GPU volume using CuPy arrays + - TomoBAR implements the iteration and uses ASTRA's GPU forward and + backprojection operators. + * - ``FISTA3d_tomobar``, ``ADMM3d_tomobar`` and ``OSEM3d_tomobar`` + - Regularised iterative + - GPU volume using CuPy arrays + - TomoBAR combines data-fidelity iterations, ASTRA projection operators + and CuPy regularisers. + +ASTRA remains a package dependency of TomoBAR even when a particular method, +such as ``LPRec3d_tomobar``, does not use ASTRA in its computational path. +The method name therefore describes the pipeline-facing implementation, not +the complete dependency graph. + +The CPU pathway +--------------- + +The TomoPy pathway is shorter: HTTomo passes NumPy data to +``tomopy.recon.algorithm.recon`` and the ``algorithm`` parameter selects the +TomoPy reconstruction implementation. TomoPy may use compiled CPU kernels and +local threads, while HTTomo remains responsible for MPI distribution, +pipeline ordering and I/O. + +Use this pathway for a CPU-only system, a small reconstruction, or when a +TomoPy algorithm is specifically required. See the ``tomopy_gridrec`` entry +in :ref:`choose_pipeline` for a complete CPU example. + +Choosing and troubleshooting a pathway +--------------------------------------- + +Choose the algorithm first, then confirm that the required execution stack +is available: + +* use an analytical method for a fast baseline and an iterative method when + the data or reconstruction objective needs it; +* use the TomoPy pathway when CUDA is unavailable; +* use a TomoBAR pathway for the GPU volume methods and regularisation options; +* check :ref:`compatibility` before changing HTTomo, HTTomolibGPU, TomoBAR, + ASTRA or CuPy independently; and +* validate the finished pipeline with + ``python -m httomo check PIPELINE INPUT`` before a production run. + +For parameter names and defaults, use :ref:`reference_templates`. For the +role of the processing libraries outside reconstruction, see +:ref:`backends_list`. diff --git a/docs/source/pipelines/versioned_downloads.rst b/docs/source/pipelines/versioned_downloads.rst new file mode 100644 index 000000000..df68b6f28 --- /dev/null +++ b/docs/source/pipelines/versioned_downloads.rst @@ -0,0 +1,47 @@ +.. _versioned_downloads: +.. _archived_templates: +.. _full_pipelines_archived: + +Versioned downloads +=================== + +Method templates and complete pipelines must match the installed HTTomo +release. Download both archives from the same row. Patch releases within a +minor series use the same archive unless their release notes state otherwise; +for example, HTTomo 3.3.2 uses the 3.3 archives. + +.. only:: builder_html + + .. list-table:: + :header-rows: 1 + :widths: 20 40 40 + + * - HTTomo version + - Method templates + - Complete pipelines + * - 3.3 + - :download:`Download templates + <../templates_archive/httomo_ver3_3_yaml_templates.zip>` + - :download:`Download pipelines + <../templates_archive/httomo_ver3_3_full_yaml_pipelines.zip>` + * - 3.2 + - :download:`Download templates + <../templates_archive/httomo_ver3_2_yaml_templates.zip>` + - :download:`Download pipelines + <../templates_archive/httomo_ver3_2_full_yaml_pipelines.zip>` + * - 3.1 + - :download:`Download templates + <../templates_archive/httomo_ver3_1_yaml_templates.zip>` + - :download:`Download pipelines + <../templates_archive/httomo_ver3_1_full_yaml_pipelines.zip>` + * - 3.0 + - :download:`Download templates + <../templates_archive/httomo_ver3_0_yaml_templates.zip>` + - :download:`Download pipelines + <../templates_archive/httomo_ver3_0_full_yaml_pipelines.zip>` + +For the current development version, use the generated +:ref:`reference_templates` and the latest :ref:`tutorials_pl_templates`. +See :ref:`compatibility` for runtime package requirements and the +`HTTomo releases page `_ +for changes between releases. diff --git a/docs/source/pipelines/yaml.rst b/docs/source/pipelines/yaml.rst index b5d1befb1..3ee8ccc0d 100644 --- a/docs/source/pipelines/yaml.rst +++ b/docs/source/pipelines/yaml.rst @@ -1,114 +1,120 @@ .. _tutorials_pl_templates: -Full YAML pipelines -============================== +Ready-to-use pipelines +====================== -This is a collection of ready to be used full pipelines or process lists for HTTomo. -See more on :ref:`explanation_process_list` and how to :ref:`howto_process_list`. +These complete HTTomo pipelines are starting points for common workflows. +Select one using :ref:`choose_pipeline`, then adapt its loader and method +parameters to the input data. See :ref:`explanation_process_list` for the +underlying concepts and :ref:`howto_process_list` for configuration guidance. -HTTomo mainly targets GPU computations, therefore the use of :ref:`tutorials_pl_templates_gpu` is -preferable. However, when the GPU device is not available or a GPU method is not implemented, the use of -:ref:`tutorials_pl_templates_cpu` is possible. +.. warning:: -.. _full_pipelines_archived: + These examples track the current HTTomo development version. For a tagged + release, use the matching pipeline from :ref:`versioned_downloads`. -Pipelines for released HTTomo versions --------------------------------------- +.. _tutorials_pl_templates_gpu: -These are archived full YAML pipelines that can be used with already released and tagged version of HTTomo. They are built using the :ref:`archived_templates`. +GPU pipelines +------------- -.. only:: builder_html - - :download:`HTTomo version 3.1 full YAML pipelines <../templates_archive/httomo_ver3_1_full_yaml_pipelines.zip>` +These pipelines combine GPU methods from HTTomolibGPU with CPU output methods +from HTTomolib. Reconstruction methods also require TomoBAR. See +:ref:`backends_list` for the role of each library. - :download:`HTTomo version 3.2 full YAML pipelines <../templates_archive/httomo_ver3_2_full_yaml_pipelines.zip>` - - :download:`HTTomo version 3.3 full YAML pipelines <../templates_archive/httomo_ver3_3_full_yaml_pipelines.zip>` +.. dropdown:: FBP3d with find_center_vo centring and image output + .. literalinclude:: ../pipelines_full/FBP3d_tomobar.yaml + :language: yaml -.. warning:: At DLS, the templates below should work with the :code:`httomo/latest` module, however, for production please use :ref:`full_pipelines_archived`. +.. dropdown:: FBP3d with phase-correlation centring and downsampling -.. _tutorials_pl_templates_gpu: + .. literalinclude:: ../pipelines_full/titaren_center_pc_FBP3d_resample.yaml + :language: yaml + +.. dropdown:: LPRec3d reconstruction -Pipelines using HTTomo libraries --------------------------------- + .. literalinclude:: ../pipelines_full/LPRec3d_tomobar.yaml + :language: yaml -Those pipelines consist of methods from HTTomolibgpu (GPU) and HTTomolib (CPU) backends :ref:`backends_list`. Those libraries are supported directly by the HTTomo development team and pipelines are built in computationally efficient way. +.. dropdown:: FBP3d followed by total-variation denoising -.. dropdown:: Using :code:`find_center_vo` auto-centering and :code:`FBP3d_tomobar` reconstruction method, then save the result into images. + .. literalinclude:: ../pipelines_full/FBP3d_tomobar_denoising.yaml + :language: yaml - .. literalinclude:: ../pipelines_full/FBP3d_tomobar.yaml - :language: yaml +.. dropdown:: FISTA3d with total-variation regularisation -.. dropdown:: Using :code:`find_center_pc` auto-centering, FBP reconstruction and downsampling the result before saving the images. + This iterative example is intended for noisy or undersampled data. - .. literalinclude:: ../pipelines_full/titaren_center_pc_FBP3d_resample.yaml - :language: yaml + .. literalinclude:: ../pipelines_full/FISTA3d_tomobar.yaml + :language: yaml -.. dropdown:: Using :code:`LPRec3d_tomobar` reconstruction, which is the fastest from all available reconstruction methods. +.. _tutorials_pipelines: - .. literalinclude:: ../pipelines_full/LPRec3d_tomobar.yaml - :language: yaml +Tutorial pipelines +------------------ -.. dropdown:: Applying Total Variation denoising :code:`total_variation_PD` to the result of the FBP reconstruction. +These pipelines accompany :ref:`data_tutorials`, where the input data is +provided or generated. - .. literalinclude:: ../pipelines_full/FBP3d_tomobar_denoising.yaml - :language: yaml +.. dropdown:: TomoPy CPU pipeline for the Lorentz dataset -.. dropdown:: Using advanced iterative reconstruction :code:`FISTA3d_tomobar` with Total Variation regularisation. Recommended for undersampled and/or noisy data. + .. literalinclude:: ../pipelines_full/tomopy_tomobank.yaml + :language: yaml - .. literalinclude:: ../pipelines_full/FISTA3d_tomobar.yaml - :language: yaml +.. dropdown:: GPU pipeline for the Lorentz dataset + + .. literalinclude:: ../pipelines_full/FBP3d_tomobar_tomobank.yaml + :language: yaml .. _tutorials_pl_templates_dls: -DLS-specific pipelines ----------------------- +Diamond-specific pipelines +-------------------------- -These pipelines are specific to Diamond Light Source processing strategies and can vary between different tomographic beamlines. +These examples implement processing strategies used at Diamond Light Source. +Required parameters and calibration files can differ between beamlines. -.. dropdown:: Reconstructing 360-degrees data with automatic CoR/overlap finding and stitching to 180-degrees data. Paganin filter is applied to the data. +.. dropdown:: Convert a 360° scan to 180°, apply Paganin filtering and reconstruct - .. literalinclude:: ../pipelines_full/deg360_paganin_FBP3d_tomobar.yaml - :language: yaml + .. literalinclude:: ../pipelines_full/deg360_paganin_FBP3d_tomobar.yaml + :language: yaml -.. dropdown:: Using distortion correction module as a part of the pipeline with 360-degrees data. +.. dropdown:: Correct distortion, convert a 360° scan and reconstruct - .. literalinclude:: ../pipelines_full/deg360_distortion_FBP3d_tomobar.yaml - :language: yaml + .. literalinclude:: ../pipelines_full/deg360_distortion_FBP3d_tomobar.yaml + :language: yaml .. _tutorials_pl_templates_sweeps: -Pipelines with parameter sweeps -------------------------------- - -Here we demonstrate how to perform a sweep across multiple values of a single parameter (see :ref:`parameter_sweeping` for more details). +Parameter-sweep pipelines +------------------------- -.. note:: There is no need to add image saving plugin for sweep runs as it will be added automatically. +Sweep runs automatically add image output after each swept method. Do not add a +separate image-saving method there. See :ref:`parameter_sweeping` for syntax +and execution details. -.. dropdown:: Parameter sweep using the :code:`!SweepRange` tag to do a sweep over several CoR values of the :code:`center` parameter in the reconstruction method. +.. dropdown:: Sweep centre-of-rotation values with !SweepRange .. literalinclude:: ../pipelines_full/sweep_center_FBP3d_tomobar.yaml - :language: yaml - :emphasize-lines: 36-39 + :language: yaml + :emphasize-lines: 36-39 -.. dropdown:: Parameter sweep using the :code:`!Sweep` tag over several particular values (not a range) of the :code:`ratio_delta_beta` parameter for the Paganin filter. +.. dropdown:: Sweep selected Paganin ratio values with !Sweep .. literalinclude:: ../pipelines_full/sweep_paganin_FBP3d_tomobar.yaml - :language: yaml - :emphasize-lines: 51-54 - + :language: yaml + :emphasize-lines: 51-54 .. _tutorials_pl_templates_cpu: -Pipelines using TomoPy library ------------------------------- - -One can build CPU-only pipelines by using mostly TomoPy methods. +CPU pipeline +------------ -.. note:: Methods from TomoPy are expected to be slower than the GPU-accelerated methods from the libraries above. +Use the TomoPy example when a CUDA-capable GPU is unavailable. Performance +depends on the input, method parameters, CPU resources and process count. -.. dropdown:: CPU pipeline using auto-centering and the gridrec reconstruction method from TomoPy. +.. dropdown:: TomoPy gridrec with automatic centring - .. literalinclude:: ../pipelines_full/tomopy_gridrec.yaml - :language: yaml + .. literalinclude:: ../pipelines_full/tomopy_gridrec.yaml + :language: yaml diff --git a/docs/source/pipelines_full/FBP2d_astra.yaml b/docs/source/pipelines_full/FBP2d_astra.yaml new file mode 100644 index 000000000..b5d97f83a --- /dev/null +++ b/docs/source/pipelines_full/FBP2d_astra.yaml @@ -0,0 +1,59 @@ +# This pipeline should be supported by the latest developments of HTTomo. Use module load httomo/latest module at Diamond. +# --- Standard tomography loader for NeXus files. --- +- method: standard_tomo + module_path: httomo.data.hdf.loaders + parameters: + data_path: auto + image_key_path: auto + rotation_angles: auto + preview: + detector_x: # horizontal data previewing/cropping. + # when null, the full data dimension is used, i.e., no previewing + start: null + stop: null + detector_y: # vertical data previewing/cropping. + # when null, the full data dimension is used, i.e., no previewing + start: null + stop: null + darks: null + flats: null + continuous_scan_subset: null +# --- Flat-field and dark-field projection correction. --- +- method: dark_flat_field_correction + module_path: httomolibgpu.prep.normalize + parameters: + flats_multiplier: 1.0 + darks_multiplier: 1.0 + upper_bound: null + lower_bound: null +# --- Center of Rotation auto-finding. Required for reconstruction below. --- +- method: find_center_vo + module_path: httomolibgpu.recon.rotation + parameters: + ind: null # A vertical slice (sinogram) index to calculate CoR, 'mid' can be used for middle + average_radius: 0 # Average several sinograms to improve SNR, one can try 3-5 range + cor_initialisation_value: null # Use if an approximate CoR is known + smin: -50 + smax: 50 + srad: 6.0 + step: 0.5 + ratio: 0.5 + drop: 20 + id: centering + side_outputs: + cor: centre_of_rotation # An estimated CoR value provided as a side output +# --- Negative log is required for reconstruction to convert raw intensity measurements into the line integrals of attenuation. --- +- method: minus_log + module_path: httomolibgpu.prep.normalize + parameters: {} +# --- Reconstruction method. --- +- method: FBP2d_astra + module_path: httomolibgpu.recon.algorithm + parameters: + center: ${{centering.side_outputs.centre_of_rotation}} # Reference to center of rotation side output, a float number or 'null' for a middle of the horizontal dimension. + detector_pad: false # Horizontal detector padding to minimise circle/arc-type artifacts in the reconstruction. Set to 'true' to enable automatic padding or an integer + filter_type: ram-lak + filter_parameter: null + filter_d: null + recon_size: null + recon_mask_radius: 0.95 # Zero pixels outside the mask-circle radius. Make radius equal to 2.0 to remove the mask effect. diff --git a/docs/source/pipelines_full/FBP3d_tomobar.yaml b/docs/source/pipelines_full/FBP3d_tomobar.yaml new file mode 100644 index 000000000..b2c903b54 --- /dev/null +++ b/docs/source/pipelines_full/FBP3d_tomobar.yaml @@ -0,0 +1,95 @@ +# This pipeline should be supported by the latest developments of HTTomo. Use module load httomo/latest module at Diamond. +# --- Standard tomography loader for NeXus files. --- +- method: standard_tomo + module_path: httomo.data.hdf.loaders + parameters: + data_path: auto + image_key_path: auto + rotation_angles: auto + preview: + detector_x: # horizontal data previewing/cropping. + # when null, the full data dimension is used, i.e., no previewing + start: null + stop: null + detector_y: # vertical data previewing/cropping. + # when null, the full data dimension is used, i.e., no previewing + start: null + stop: null + darks: null + flats: null + continuous_scan_subset: null +# --- Removing unresponsive/dead pixels in the data, aka zingers. Use if sharp streaks are present in the reconstruction. To be applied before normalisation. --- +- method: remove_outlier + module_path: httomolibgpu.misc.corr + parameters: + kernel_size: 3 # The size of the 3D neighbourhood surrounding the voxel. Odd integer. + dif: 1000 # A difference between the outlier value and the median value of neighbouring pixels. +# --- Flat-field and dark-field projection correction. --- +- method: dark_flat_field_correction + module_path: httomolibgpu.prep.normalize + parameters: + flats_multiplier: 1.0 + darks_multiplier: 1.0 + upper_bound: null + lower_bound: null +# --- Center of Rotation auto-finding. Required for reconstruction below. --- +- method: find_center_vo + module_path: httomolibgpu.recon.rotation + parameters: + ind: null # A vertical slice (sinogram) index to calculate CoR, 'mid' can be used for middle + average_radius: 0 # Average several sinograms to improve SNR, one can try 3-5 range + cor_initialisation_value: null # Use if an approximate CoR is known + smin: -50 + smax: 50 + srad: 6.0 + step: 0.5 + ratio: 0.5 + drop: 20 + id: centering + side_outputs: + cor: centre_of_rotation # An estimated CoR value provided as a side output +# --- Method to remove stripe artefacts in the data that lead to ring artefacts in the reconstruction. --- +- method: remove_all_stripe + module_path: httomolibgpu.prep.stripe + parameters: + snr: 3.0 + la_size: 61 + sm_size: 21 + dim: 1 + normalize: false +# --- Negative log is required for reconstruction to convert raw intensity measurements into the line integrals of attenuation. --- +- method: minus_log + module_path: httomolibgpu.prep.normalize + parameters: {} +# --- Reconstruction method. --- +- method: FBP3d_tomobar + module_path: httomolibgpu.recon.algorithm + parameters: + center: ${{centering.side_outputs.centre_of_rotation}} # Reference to center of rotation side output, a float number or 'null' for a middle of the horizontal dimension. + detector_pad: false # Horizontal detector padding to minimise circle/arc-type artifacts in the reconstruction. Set to 'true' to enable automatic padding or an integer + filter_freq_cutoff: 0.35 + recon_size: null + recon_mask_radius: 0.95 # Zero pixels outside the mask-circle radius. Make radius equal to 2.0 to remove the mask effect. +# --- Calculate global statistics on the reconstructed volume, required for data rescaling. --- +- method: calculate_stats + module_path: httomo.methods + parameters: {} + id: statistics + side_outputs: + glob_stats: glob_stats +# --- Rescaling the data using min/max obtained from `calculate_stats`. --- +- method: rescale_to_int + module_path: httomolib.misc.rescale + parameters: + perc_range_min: 0.0 + perc_range_max: 100.0 + bits: 8 + glob_stats: ${{statistics.side_outputs.glob_stats}} +# --- Saving data into images. --- +- method: save_to_images + module_path: httomolib.misc.images + parameters: + subfolder_name: images + axis: auto + file_format: tif # `tif` or `jpeg` can be used. + asynchronous: true diff --git a/docs/source/pipelines_full/FBP3d_tomobar_denoising.yaml b/docs/source/pipelines_full/FBP3d_tomobar_denoising.yaml new file mode 100644 index 000000000..cca06c340 --- /dev/null +++ b/docs/source/pipelines_full/FBP3d_tomobar_denoising.yaml @@ -0,0 +1,91 @@ +- method: standard_tomo + module_path: httomo.data.hdf.loaders + parameters: + data_path: auto + image_key_path: auto + rotation_angles: auto + preview: + detector_x: + start: null + stop: null + detector_y: + start: null + stop: null + darks: null + flats: null + continuous_scan_subset: null +- method: remove_outlier + module_path: httomolibgpu.misc.corr + parameters: + kernel_size: 3 + dif: 1000 +- method: dark_flat_field_correction + module_path: httomolibgpu.prep.normalize + parameters: + flats_multiplier: 1.0 + darks_multiplier: 1.0 + upper_bound: null + lower_bound: null +- method: find_center_vo + module_path: httomolibgpu.recon.rotation + parameters: + ind: null + average_radius: 0 + cor_initialisation_value: 80.0 + smin: -20 + smax: 20 + srad: 6.0 + step: 0.5 + ratio: 0.5 + drop: 20 + id: centering + side_outputs: + cor: centre_of_rotation +- method: remove_all_stripe + module_path: httomolibgpu.prep.stripe + parameters: + snr: 3.0 + la_size: 61 + sm_size: 21 + dim: 1 + normalize: false +- method: minus_log + module_path: httomolibgpu.prep.normalize + parameters: {} +- method: FBP3d_tomobar + module_path: httomolibgpu.recon.algorithm + parameters: + center: ${{centering.side_outputs.centre_of_rotation}} + detector_pad: false + filter_freq_cutoff: 0.35 + recon_size: null + recon_mask_radius: 0.95 +- method: total_variation_PD + module_path: httomolibgpu.misc.denoise + parameters: + regularisation_parameter: 1.0e-05 + iterations: 200 + isotropic: true + nonnegativity: false + lipschitz_const: 8.0 + half_precision: false +- method: calculate_stats + module_path: httomo.methods + parameters: {} + id: statistics + side_outputs: + glob_stats: glob_stats +- method: rescale_to_int + module_path: httomolib.misc.rescale + parameters: + perc_range_min: 0.0 + perc_range_max: 100.0 + bits: 8 + glob_stats: ${{statistics.side_outputs.glob_stats}} +- method: save_to_images + module_path: httomolib.misc.images + parameters: + subfolder_name: images + axis: auto + file_format: tif + asynchronous: true diff --git a/docs/source/pipelines_full/FBP3d_tomobar_noimagesaving.yaml b/docs/source/pipelines_full/FBP3d_tomobar_noimagesaving.yaml new file mode 100644 index 000000000..c142e4445 --- /dev/null +++ b/docs/source/pipelines_full/FBP3d_tomobar_noimagesaving.yaml @@ -0,0 +1,69 @@ +# This pipeline should be supported by the latest developments of HTTomo. Use module load httomo/latest module at Diamond. +# --- Standard tomography loader for NeXus files. --- +- method: standard_tomo + module_path: httomo.data.hdf.loaders + parameters: + data_path: auto + image_key_path: auto + rotation_angles: auto + preview: + detector_x: # horizontal data previewing/cropping. + # when null, the full data dimension is used, i.e., no previewing + start: null + stop: null + detector_y: # vertical data previewing/cropping. + # when null, the full data dimension is used, i.e., no previewing + start: null + stop: null + darks: null + flats: null + continuous_scan_subset: null +# --- Removing unresponsive/dead pixels in the data, aka zingers. Use if sharp streaks are present in the reconstruction. To be applied before normalisation. --- +- method: remove_outlier + module_path: httomolibgpu.misc.corr + parameters: + kernel_size: 3 # The size of the 3D neighbourhood surrounding the voxel. Odd integer. + dif: 1000 # A difference between the outlier value and the median value of neighbouring pixels. +# --- Flat-field and dark-field projection correction. --- +- method: dark_flat_field_correction + module_path: httomolibgpu.prep.normalize + parameters: + flats_multiplier: 1.0 + darks_multiplier: 1.0 + upper_bound: null + lower_bound: null +# --- Center of Rotation auto-finding. Required for reconstruction below. --- +- method: find_center_vo + module_path: httomolibgpu.recon.rotation + parameters: + ind: null # A vertical slice (sinogram) index to calculate CoR, 'mid' can be used for middle + average_radius: 0 # Average several sinograms to improve SNR, one can try 3-5 range + cor_initialisation_value: null # Use if an approximate CoR is known + smin: -50 + smax: 50 + srad: 6.0 + step: 0.5 + ratio: 0.5 + drop: 20 + id: centering + side_outputs: + cor: centre_of_rotation # An estimated CoR value provided as a side output +# --- Method to remove stripe artefacts in the data that lead to ring artefacts in the reconstruction. --- +- method: remove_stripe_based_sorting + module_path: httomolibgpu.prep.stripe + parameters: + size: 11 + dim: 1 +# --- Negative log is required for reconstruction to convert raw intensity measurements into the line integrals of attenuation. --- +- method: minus_log + module_path: httomolibgpu.prep.normalize + parameters: {} +# --- Reconstruction method. --- +- method: FBP3d_tomobar + module_path: httomolibgpu.recon.algorithm + parameters: + center: ${{centering.side_outputs.centre_of_rotation}} # Reference to center of rotation side output, a float number or 'null' for a middle of the horizontal dimension. + detector_pad: false # Horizontal detector padding to minimise circle/arc-type artifacts in the reconstruction. Set to 'true' to enable automatic padding or an integer + filter_freq_cutoff: 0.35 + recon_size: null + recon_mask_radius: 0.95 # Zero pixels outside the mask-circle radius. Make radius equal to 2.0 to remove the mask effect. diff --git a/docs/source/pipelines_full/FBP3d_tomobar_tomobank.yaml b/docs/source/pipelines_full/FBP3d_tomobar_tomobank.yaml new file mode 100644 index 000000000..873ae1b0a --- /dev/null +++ b/docs/source/pipelines_full/FBP3d_tomobar_tomobank.yaml @@ -0,0 +1,89 @@ +# This pipeline should be supported by the latest developments of HTTomo. Use module load httomo/latest module at Diamond. +# --- Standard tomography loader for NeXus files. --- +- method: standard_tomo + module_path: httomo.data.hdf.loaders + parameters: + data_path: auto + image_key_path: auto + rotation_angles: auto + preview: + detector_x: # horizontal data previewing/cropping. + # when null, the full data dimension is used, i.e., no previewing + start: null + stop: null + detector_y: # vertical data previewing/cropping. + # when null, the full data dimension is used, i.e., no previewing + start: null + stop: null + darks: null + flats: null + continuous_scan_subset: null +# --- Flat-field and dark-field projection correction. --- +- method: dark_flat_field_correction + module_path: httomolibgpu.prep.normalize + parameters: + flats_multiplier: 1.0 + darks_multiplier: 1.0 + upper_bound: null + lower_bound: null +# --- Center of Rotation auto-finding. Required for reconstruction below. --- +- method: find_center_vo + module_path: httomolibgpu.recon.rotation + parameters: + ind: null # A vertical slice (sinogram) index to calculate CoR, 'mid' can be used for middle + average_radius: 0 # Average several sinograms to improve SNR, one can try 3-5 range + cor_initialisation_value: null # Use if an approximate CoR is known + smin: -50 + smax: 50 + srad: 6.0 + step: 0.5 + ratio: 0.5 + drop: 20 + id: centering + side_outputs: + cor: centre_of_rotation # An estimated CoR value provided as a side output +# --- Method to remove stripe artefacts in the data that lead to ring artefacts in the reconstruction. --- +- method: remove_all_stripe + module_path: httomolibgpu.prep.stripe + parameters: + snr: 3.0 + la_size: 61 + sm_size: 21 + dim: 1 + normalize: false +# --- Negative log is required for reconstruction to convert raw intensity measurements into the line integrals of attenuation. --- +- method: minus_log + module_path: httomolibgpu.prep.normalize + parameters: {} +# --- Reconstruction method. --- +- method: FBP3d_tomobar + module_path: httomolibgpu.recon.algorithm + parameters: + center: ${{centering.side_outputs.centre_of_rotation}} # Reference to center of rotation side output, a float number or 'null' for a middle of the horizontal dimension. + detector_pad: false # Horizontal detector padding to minimise circle/arc-type artifacts in the reconstruction. Set to 'true' to enable automatic padding or an integer + filter_freq_cutoff: 0.35 + recon_size: null + recon_mask_radius: 0.95 # Zero pixels outside the mask-circle radius. Make radius equal to 2.0 to remove the mask effect. +# --- Calculate global statistics on the reconstructed volume, required for data rescaling. --- +- method: calculate_stats + module_path: httomo.methods + parameters: {} + id: statistics + side_outputs: + glob_stats: glob_stats +# --- Rescaling the data using min/max obtained from `calculate_stats`. --- +- method: rescale_to_int + module_path: httomolib.misc.rescale + parameters: + perc_range_min: 0.0 + perc_range_max: 100.0 + bits: 8 + glob_stats: ${{statistics.side_outputs.glob_stats}} +# --- Saving data into images. --- +- method: save_to_images + module_path: httomolib.misc.images + parameters: + subfolder_name: images + axis: auto + file_format: tif # `tif` or `jpeg` can be used. + asynchronous: true diff --git a/docs/source/pipelines_full/FISTA3d_tomobar.yaml b/docs/source/pipelines_full/FISTA3d_tomobar.yaml new file mode 100644 index 000000000..a83668411 --- /dev/null +++ b/docs/source/pipelines_full/FISTA3d_tomobar.yaml @@ -0,0 +1,87 @@ +# This pipeline should be supported by the latest developments of HTTomo. Use module load httomo/latest module at Diamond. +# --- Standard tomography loader for NeXus files. --- +- method: standard_tomo + module_path: httomo.data.hdf.loaders + parameters: + data_path: auto + image_key_path: auto + rotation_angles: auto + preview: + detector_x: # horizontal data previewing/cropping. + # when null, the full data dimension is used, i.e., no previewing + start: null + stop: null + detector_y: # vertical data previewing/cropping. + # when null, the full data dimension is used, i.e., no previewing + start: null + stop: null + darks: null + flats: null + continuous_scan_subset: null +# --- Flat-field and dark-field projection correction. --- +- method: dark_flat_field_correction + module_path: httomolibgpu.prep.normalize + parameters: + flats_multiplier: 1.0 + darks_multiplier: 1.0 + upper_bound: null + lower_bound: null +# --- Center of Rotation auto-finding. Required for reconstruction below. --- +- method: find_center_vo + module_path: httomolibgpu.recon.rotation + parameters: + ind: null # A vertical slice (sinogram) index to calculate CoR, 'mid' can be used for middle + average_radius: 0 # Average several sinograms to improve SNR, one can try 3-5 range + cor_initialisation_value: null # Use if an approximate CoR is known + smin: -50 + smax: 50 + srad: 6.0 + step: 0.5 + ratio: 0.5 + drop: 20 + id: centering + side_outputs: + cor: centre_of_rotation # An estimated CoR value provided as a side output +# --- Negative log is required for reconstruction to convert raw intensity measurements into the line integrals of attenuation. --- +- method: minus_log + module_path: httomolibgpu.prep.normalize + parameters: {} +# --- Reconstruction method. --- +- method: FISTA3d_tomobar + module_path: httomolibgpu.recon.algorithm + parameters: + center: ${{centering.side_outputs.centre_of_rotation}} # Reference to center of rotation side output, a float number or 'null' for a middle of the horizontal dimension. + detector_pad: false # Horizontal detector padding to minimise circle/arc-type artifacts in the reconstruction. Set to 'true' to enable automatic padding or an integer + recon_size: null + recon_mask_radius: 0.95 # Zero pixels outside the mask-circle radius. Make radius equal to 2.0 to remove the mask effect. + iterations: 20 + subsets_number: 6 + data_fidelity: LS + regularisation_type: PD_TV + regularisation_parameter: 1.0e-06 + regularisation_iterations: 50 + regularisation_half_precision: true + nonnegativity: true +# --- Calculate global statistics on the reconstructed volume, required for data rescaling. --- +- method: calculate_stats + module_path: httomo.methods + parameters: {} + id: statistics + side_outputs: + glob_stats: glob_stats +# --- Rescaling the data using min/max obtained from `calculate_stats`. --- +- method: rescale_to_int + module_path: httomolib.misc.rescale + parameters: + perc_range_min: 0.0 + perc_range_max: 100.0 + bits: 8 + glob_stats: ${{statistics.side_outputs.glob_stats}} +# --- Saving data into images. --- +- method: save_to_images + module_path: httomolib.misc.images + parameters: + subfolder_name: images + axis: auto + file_format: tif # `tif` or `jpeg` can be used. + asynchronous: true diff --git a/docs/source/pipelines_full/LPRec3d_tomobar.yaml b/docs/source/pipelines_full/LPRec3d_tomobar.yaml new file mode 100644 index 000000000..39c044355 --- /dev/null +++ b/docs/source/pipelines_full/LPRec3d_tomobar.yaml @@ -0,0 +1,96 @@ +# This pipeline should be supported by the latest developments of HTTomo. Use module load httomo/latest module at Diamond. +# --- Standard tomography loader for NeXus files. --- +- method: standard_tomo + module_path: httomo.data.hdf.loaders + parameters: + data_path: auto + image_key_path: auto + rotation_angles: auto + preview: + detector_x: # horizontal data previewing/cropping. + # when null, the full data dimension is used, i.e., no previewing + start: null + stop: null + detector_y: # vertical data previewing/cropping. + # when null, the full data dimension is used, i.e., no previewing + start: null + stop: null + darks: null + flats: null + continuous_scan_subset: null +# --- Removing unresponsive/dead pixels in the data, aka zingers. Use if sharp streaks are present in the reconstruction. To be applied before normalisation. --- +- method: remove_outlier + module_path: httomolibgpu.misc.corr + parameters: + kernel_size: 3 # The size of the 3D neighbourhood surrounding the voxel. Odd integer. + dif: 1000 # A difference between the outlier value and the median value of neighbouring pixels. +# --- Flat-field and dark-field projection correction. --- +- method: dark_flat_field_correction + module_path: httomolibgpu.prep.normalize + parameters: + flats_multiplier: 1.0 + darks_multiplier: 1.0 + upper_bound: null + lower_bound: null +# --- Center of Rotation auto-finding. Required for reconstruction below. --- +- method: find_center_vo + module_path: httomolibgpu.recon.rotation + parameters: + ind: null # A vertical slice (sinogram) index to calculate CoR, 'mid' can be used for middle + average_radius: 0 # Average several sinograms to improve SNR, one can try 3-5 range + cor_initialisation_value: null # Use if an approximate CoR is known + smin: -50 + smax: 50 + srad: 6.0 + step: 0.5 + ratio: 0.5 + drop: 20 + id: centering + side_outputs: + cor: centre_of_rotation # An estimated CoR value provided as a side output +# --- Method to remove stripe artefacts in the data that lead to ring artefacts in the reconstruction. --- +- method: remove_all_stripe + module_path: httomolibgpu.prep.stripe + parameters: + snr: 3.0 + la_size: 61 + sm_size: 21 + dim: 1 + normalize: false +# --- Negative log is required for reconstruction to convert raw intensity measurements into the line integrals of attenuation. --- +- method: minus_log + module_path: httomolibgpu.prep.normalize + parameters: {} +# --- Reconstruction method. --- +- method: LPRec3d_tomobar + module_path: httomolibgpu.recon.algorithm + parameters: + center: ${{centering.side_outputs.centre_of_rotation}} # Reference to center of rotation side output, a float number or 'null' for a middle of the horizontal dimension. + detector_pad: false # Horizontal detector padding to minimise circle/arc-type artifacts in the reconstruction. Set to 'true' to enable automatic padding or an integer + filter_type: shepp + filter_freq_cutoff: 1.0 + recon_size: null + recon_mask_radius: 0.95 # Zero pixels outside the mask-circle radius. Make radius equal to 2.0 to remove the mask effect. +# --- Calculate global statistics on the reconstructed volume, required for data rescaling. --- +- method: calculate_stats + module_path: httomo.methods + parameters: {} + id: statistics + side_outputs: + glob_stats: glob_stats +# --- Rescaling the data using min/max obtained from `calculate_stats`. --- +- method: rescale_to_int + module_path: httomolib.misc.rescale + parameters: + perc_range_min: 0.0 + perc_range_max: 100.0 + bits: 8 + glob_stats: ${{statistics.side_outputs.glob_stats}} +# --- Saving data into images. --- +- method: save_to_images + module_path: httomolib.misc.images + parameters: + subfolder_name: images + axis: auto + file_format: tif # `tif` or `jpeg` can be used. + asynchronous: true diff --git a/docs/source/pipelines_full/angles_averaging.yaml b/docs/source/pipelines_full/angles_averaging.yaml new file mode 100644 index 000000000..a71a35420 --- /dev/null +++ b/docs/source/pipelines_full/angles_averaging.yaml @@ -0,0 +1,92 @@ +# This pipeline should be supported by the latest developments of HTTomo. Use module load httomo/latest module at Diamond. +# --- Standard tomography loader for NeXus files. --- +- method: standard_tomo + module_path: httomo.data.hdf.loaders + parameters: + data_path: auto + image_key_path: auto + rotation_angles: auto + preview: + detector_x: # horizontal data previewing/cropping. + # when null, the full data dimension is used, i.e., no previewing + start: null + stop: null + detector_y: # vertical data previewing/cropping. + # when null, the full data dimension is used, i.e., no previewing + start: null + stop: null + darks: null + flats: null + continuous_scan_subset: null +# --- Removing unresponsive/dead pixels in the data, aka zingers. Use if sharp streaks are present in the reconstruction. To be applied before normalisation. --- +- method: remove_outlier + module_path: httomolibgpu.misc.corr + parameters: + kernel_size: 3 # The size of the 3D neighbourhood surrounding the voxel. Odd integer. + dif: 1000 # A difference between the outlier value and the median value of neighbouring pixels. +# --- Flat-field and dark-field projection correction. --- +- method: dark_flat_field_correction + module_path: httomolibgpu.prep.normalize + parameters: + flats_multiplier: 1.0 + darks_multiplier: 1.0 + upper_bound: null + lower_bound: null +# --- Center of Rotation auto-finding. Required for reconstruction below. --- +- method: find_center_vo + module_path: httomolibgpu.recon.rotation + parameters: + ind: null # A vertical slice (sinogram) index to calculate CoR, 'mid' can be used for middle + average_radius: 0 # Average several sinograms to improve SNR, one can try 3-5 range + cor_initialisation_value: null # Use if an approximate CoR is known + smin: -50 + smax: 50 + srad: 6.0 + step: 0.5 + ratio: 0.5 + drop: 20 + id: centering + side_outputs: + cor: centre_of_rotation # An estimated CoR value provided as a side output +# --- Apply averaging of projection data in the angular dimension. --- +- method: average_projection_frames + module_path: httomolibgpu.misc.morph + parameters: + projection_averaging_factor: 2 +# --- Negative log is required for reconstruction to convert raw intensity measurements into the line integrals of attenuation. --- +- method: minus_log + module_path: httomolibgpu.prep.normalize + parameters: {} +# --- Reconstruction method. --- +- method: LPRec3d_tomobar + module_path: httomolibgpu.recon.algorithm + parameters: + center: ${{centering.side_outputs.centre_of_rotation}} # Reference to center of rotation side output, a float number or 'null' for a middle of the horizontal dimension. + detector_pad: false # Horizontal detector padding to minimise circle/arc-type artifacts in the reconstruction. Set to 'true' to enable automatic padding or an integer + filter_type: shepp + filter_freq_cutoff: 1.0 + recon_size: null + recon_mask_radius: 0.95 # Zero pixels outside the mask-circle radius. Make radius equal to 2.0 to remove the mask effect. +# --- Calculate global statistics on the reconstructed volume, required for data rescaling. --- +- method: calculate_stats + module_path: httomo.methods + parameters: {} + id: statistics + side_outputs: + glob_stats: glob_stats +# --- Rescaling the data using min/max obtained from `calculate_stats`. --- +- method: rescale_to_int + module_path: httomolib.misc.rescale + parameters: + perc_range_min: 0.0 + perc_range_max: 100.0 + bits: 8 + glob_stats: ${{statistics.side_outputs.glob_stats}} +# --- Saving data into images. --- +- method: save_to_images + module_path: httomolib.misc.images + parameters: + subfolder_name: images + axis: auto + file_format: tif # `tif` or `jpeg` can be used. + asynchronous: true diff --git a/docs/source/pipelines_full/deg360_distortion_FBP3d_tomobar.yaml b/docs/source/pipelines_full/deg360_distortion_FBP3d_tomobar.yaml new file mode 100644 index 000000000..5d9a4d877 --- /dev/null +++ b/docs/source/pipelines_full/deg360_distortion_FBP3d_tomobar.yaml @@ -0,0 +1,102 @@ +# This pipeline should be supported by the latest developments of HTTomo. Use module load httomo/latest module at Diamond. +# --- Standard tomography loader for NeXus files. --- +- method: standard_tomo + module_path: httomo.data.hdf.loaders + parameters: + data_path: auto + image_key_path: auto + rotation_angles: auto + preview: + detector_x: # horizontal data previewing/cropping. + # when null, the full data dimension is used, i.e., no previewing + start: null + stop: null + detector_y: # vertical data previewing/cropping. + # when null, the full data dimension is used, i.e., no previewing + start: null + stop: null + darks: null + flats: null + continuous_scan_subset: null +# --- Applying optical distortion correction to projections. --- +- method: distortion_correction_proj_discorpy + module_path: httomolibgpu.prep.alignment + parameters: + metadata_path: null # Provide an absolute path to the text file with distortion coefficients. + xcenter: null + ycenter: null + list_fact: null + order: 3 + mode: constant +# --- Flat-field and dark-field projection correction. --- +- method: dark_flat_field_correction + module_path: httomolibgpu.prep.normalize + parameters: + flats_multiplier: 1.0 + darks_multiplier: 1.0 + upper_bound: null + lower_bound: null +# --- Center of Rotation auto-finding. Required for reconstruction below. --- +- method: find_center_360 + module_path: httomolibgpu.recon.rotation + parameters: + ind: null # A vertical slice (sinogram) index to calculate CoR, 'mid' can be used for middle + win_width: 10 + side: null # Choose 'left' to stitch to the left side, 'right' to the right side, or null for automated determination + denoise: true + norm: false + use_overlap: false + id: centering + side_outputs: + cor: centre_of_rotation # An estimated CoR value provided as a side output + overlap: overlap # An overlap to use for converting 360 degrees scan to 180 degrees scan. + side: side + overlap_position: overlap_position +# --- Using the overlap and side provided, converting 360 degrees scan to 180 degrees scan. --- +- method: sino_360_to_180 + module_path: httomolibgpu.misc.morph + parameters: + overlap: ${{centering.side_outputs.overlap}} + side: ${{centering.side_outputs.side}} +# --- Method to remove stripe artefacts in the data that lead to ring artefacts in the reconstruction. --- +- method: remove_stripe_based_sorting + module_path: httomolibgpu.prep.stripe + parameters: + size: 11 + dim: 1 +# --- Negative log is required for reconstruction to convert raw intensity measurements into the line integrals of attenuation. --- +- method: minus_log + module_path: httomolibgpu.prep.normalize + parameters: {} +# --- Reconstruction method. --- +- method: FBP3d_tomobar + module_path: httomolibgpu.recon.algorithm + parameters: + center: ${{centering.side_outputs.centre_of_rotation}} # Reference to center of rotation side output, a float number or 'null' for a middle of the horizontal dimension. + detector_pad: false # Horizontal detector padding to minimise circle/arc-type artifacts in the reconstruction. Set to 'true' to enable automatic padding or an integer + filter_freq_cutoff: 0.35 + recon_size: null + recon_mask_radius: 0.95 # Zero pixels outside the mask-circle radius. Make radius equal to 2.0 to remove the mask effect. +# --- Calculate global statistics on the reconstructed volume, required for data rescaling. --- +- method: calculate_stats + module_path: httomo.methods + parameters: {} + id: statistics + side_outputs: + glob_stats: glob_stats +# --- Rescaling the data using min/max obtained from `calculate_stats`. --- +- method: rescale_to_int + module_path: httomolib.misc.rescale + parameters: + perc_range_min: 0.0 + perc_range_max: 100.0 + bits: 8 + glob_stats: ${{statistics.side_outputs.glob_stats}} +# --- Saving data into images. --- +- method: save_to_images + module_path: httomolib.misc.images + parameters: + subfolder_name: images + axis: auto + file_format: tif # `tif` or `jpeg` can be used. + asynchronous: true diff --git a/docs/source/pipelines_full/deg360_paganin_FBP3d_tomobar.yaml b/docs/source/pipelines_full/deg360_paganin_FBP3d_tomobar.yaml new file mode 100644 index 000000000..1b45092f0 --- /dev/null +++ b/docs/source/pipelines_full/deg360_paganin_FBP3d_tomobar.yaml @@ -0,0 +1,98 @@ +# This pipeline should be supported by the latest developments of HTTomo. Use module load httomo/latest module at Diamond. +# --- Standard tomography loader for NeXus files. --- +- method: standard_tomo + module_path: httomo.data.hdf.loaders + parameters: + data_path: auto + image_key_path: auto + rotation_angles: auto + preview: + detector_x: # horizontal data previewing/cropping. + # when null, the full data dimension is used, i.e., no previewing + start: null + stop: null + detector_y: # vertical data previewing/cropping. + # when null, the full data dimension is used, i.e., no previewing + start: null + stop: null + darks: null + flats: null + continuous_scan_subset: null +# --- Flat-field and dark-field projection correction. --- +- method: dark_flat_field_correction + module_path: httomolibgpu.prep.normalize + parameters: + flats_multiplier: 1.0 + darks_multiplier: 1.0 + upper_bound: null + lower_bound: null +# --- Center of Rotation auto-finding. Required for reconstruction below. --- +- method: find_center_360 + module_path: httomolibgpu.recon.rotation + parameters: + ind: null # A vertical slice (sinogram) index to calculate CoR, 'mid' can be used for middle + win_width: 10 + side: null # Choose 'left' to stitch to the left side, 'right' to the right side, or null for automated determination + denoise: true + norm: false + use_overlap: false + id: centering + side_outputs: + cor: centre_of_rotation # An estimated CoR value provided as a side output + overlap: overlap # An overlap to use for converting 360 degrees scan to 180 degrees scan. + side: side + overlap_position: overlap_position +# --- Using the overlap and side provided, converting 360 degrees scan to 180 degrees scan. --- +- method: sino_360_to_180 + module_path: httomolibgpu.misc.morph + parameters: + overlap: ${{centering.side_outputs.overlap}} + side: ${{centering.side_outputs.side}} +# --- Method to remove stripe artefacts in the data that lead to ring artefacts in the reconstruction. --- +- method: remove_stripe_based_sorting + module_path: httomolibgpu.prep.stripe + parameters: + size: 11 + dim: 1 +# --- Apply a phase contrast filter to improve image contrast. --- +- method: paganin_filter + module_path: httomolibgpu.prep.phase + parameters: + pixel_size: 1.28 # Detector pixel size (resolution) in MICRON units. + distance: 1.0 # Propagation distance of the wavefront from sample to detector in METRE units. + energy: 53.0 # Beam energy in keV. + ratio_delta_beta: 250 # The ratio of delta/beta for filter strength control. Larger values lead to more smoothing. + calculate_padding_value_method: next_power_of_2 # Select type of padding from 'next_power_of_2', 'next_fast_length' and 'use_pad_x_y'. + pad_x_y: null # Manual padding is enabled when 'calculate_padding_value_method' is set to 'use_pad_x_y'. +# --- Reconstruction method. --- +- method: FBP3d_tomobar + module_path: httomolibgpu.recon.algorithm + parameters: + center: ${{centering.side_outputs.centre_of_rotation}} # Reference to center of rotation side output, a float number or 'null' for a middle of the horizontal dimension. + detector_pad: false # Horizontal detector padding to minimise circle/arc-type artifacts in the reconstruction. Set to 'true' to enable automatic padding or an integer + filter_freq_cutoff: 0.35 + recon_size: null + recon_mask_radius: 0.95 # Zero pixels outside the mask-circle radius. Make radius equal to 2.0 to remove the mask effect. +# --- Calculate global statistics on the reconstructed volume, required for data rescaling. --- +- method: calculate_stats + module_path: httomo.methods + parameters: {} + id: statistics + side_outputs: + glob_stats: glob_stats +# --- Rescaling the data using min/max obtained from `calculate_stats`. --- +- method: rescale_to_int + module_path: httomolib.misc.rescale + parameters: + perc_range_min: 0.0 + perc_range_max: 100.0 + bits: 8 + glob_stats: ${{statistics.side_outputs.glob_stats}} +# --- Saving data into images. --- +- method: save_to_images + module_path: httomolib.misc.images + parameters: + subfolder_name: images + axis: auto + file_format: tif # `tif` or `jpeg` can be used. + asynchronous: true diff --git a/docs/source/pipelines_full/no_loader_FBP3d_tomobar.yaml b/docs/source/pipelines_full/no_loader_FBP3d_tomobar.yaml new file mode 100644 index 000000000..b2c903b54 --- /dev/null +++ b/docs/source/pipelines_full/no_loader_FBP3d_tomobar.yaml @@ -0,0 +1,95 @@ +# This pipeline should be supported by the latest developments of HTTomo. Use module load httomo/latest module at Diamond. +# --- Standard tomography loader for NeXus files. --- +- method: standard_tomo + module_path: httomo.data.hdf.loaders + parameters: + data_path: auto + image_key_path: auto + rotation_angles: auto + preview: + detector_x: # horizontal data previewing/cropping. + # when null, the full data dimension is used, i.e., no previewing + start: null + stop: null + detector_y: # vertical data previewing/cropping. + # when null, the full data dimension is used, i.e., no previewing + start: null + stop: null + darks: null + flats: null + continuous_scan_subset: null +# --- Removing unresponsive/dead pixels in the data, aka zingers. Use if sharp streaks are present in the reconstruction. To be applied before normalisation. --- +- method: remove_outlier + module_path: httomolibgpu.misc.corr + parameters: + kernel_size: 3 # The size of the 3D neighbourhood surrounding the voxel. Odd integer. + dif: 1000 # A difference between the outlier value and the median value of neighbouring pixels. +# --- Flat-field and dark-field projection correction. --- +- method: dark_flat_field_correction + module_path: httomolibgpu.prep.normalize + parameters: + flats_multiplier: 1.0 + darks_multiplier: 1.0 + upper_bound: null + lower_bound: null +# --- Center of Rotation auto-finding. Required for reconstruction below. --- +- method: find_center_vo + module_path: httomolibgpu.recon.rotation + parameters: + ind: null # A vertical slice (sinogram) index to calculate CoR, 'mid' can be used for middle + average_radius: 0 # Average several sinograms to improve SNR, one can try 3-5 range + cor_initialisation_value: null # Use if an approximate CoR is known + smin: -50 + smax: 50 + srad: 6.0 + step: 0.5 + ratio: 0.5 + drop: 20 + id: centering + side_outputs: + cor: centre_of_rotation # An estimated CoR value provided as a side output +# --- Method to remove stripe artefacts in the data that lead to ring artefacts in the reconstruction. --- +- method: remove_all_stripe + module_path: httomolibgpu.prep.stripe + parameters: + snr: 3.0 + la_size: 61 + sm_size: 21 + dim: 1 + normalize: false +# --- Negative log is required for reconstruction to convert raw intensity measurements into the line integrals of attenuation. --- +- method: minus_log + module_path: httomolibgpu.prep.normalize + parameters: {} +# --- Reconstruction method. --- +- method: FBP3d_tomobar + module_path: httomolibgpu.recon.algorithm + parameters: + center: ${{centering.side_outputs.centre_of_rotation}} # Reference to center of rotation side output, a float number or 'null' for a middle of the horizontal dimension. + detector_pad: false # Horizontal detector padding to minimise circle/arc-type artifacts in the reconstruction. Set to 'true' to enable automatic padding or an integer + filter_freq_cutoff: 0.35 + recon_size: null + recon_mask_radius: 0.95 # Zero pixels outside the mask-circle radius. Make radius equal to 2.0 to remove the mask effect. +# --- Calculate global statistics on the reconstructed volume, required for data rescaling. --- +- method: calculate_stats + module_path: httomo.methods + parameters: {} + id: statistics + side_outputs: + glob_stats: glob_stats +# --- Rescaling the data using min/max obtained from `calculate_stats`. --- +- method: rescale_to_int + module_path: httomolib.misc.rescale + parameters: + perc_range_min: 0.0 + perc_range_max: 100.0 + bits: 8 + glob_stats: ${{statistics.side_outputs.glob_stats}} +# --- Saving data into images. --- +- method: save_to_images + module_path: httomolib.misc.images + parameters: + subfolder_name: images + axis: auto + file_format: tif # `tif` or `jpeg` can be used. + asynchronous: true diff --git a/docs/source/pipelines_full/sweep_center_FBP3d_tomobar.yaml b/docs/source/pipelines_full/sweep_center_FBP3d_tomobar.yaml new file mode 100644 index 000000000..6c5c2afb7 --- /dev/null +++ b/docs/source/pipelines_full/sweep_center_FBP3d_tomobar.yaml @@ -0,0 +1,44 @@ +# This pipeline should be supported by the latest developments of HTTomo. Use module load httomo/latest module at Diamond. +# --- Standard tomography loader for NeXus files. --- +- method: standard_tomo + module_path: httomo.data.hdf.loaders + parameters: + data_path: auto + image_key_path: auto + rotation_angles: auto + preview: + detector_x: # horizontal data previewing/cropping. + # when null, the full data dimension is used, i.e., no previewing + start: null + stop: null + detector_y: # vertical data previewing/cropping. + # when null, the full data dimension is used, i.e., no previewing + start: null + stop: null + darks: null + flats: null + continuous_scan_subset: null +# --- Flat-field and dark-field projection correction. --- +- method: dark_flat_field_correction + module_path: httomolibgpu.prep.normalize + parameters: + flats_multiplier: 1.0 + darks_multiplier: 1.0 + upper_bound: null + lower_bound: null +# --- Negative log is required for reconstruction to convert raw intensity measurements into the line integrals of attenuation. --- +- method: minus_log + module_path: httomolibgpu.prep.normalize + parameters: {} +# --- Reconstruction method. --- +- method: FBP3d_tomobar + module_path: httomolibgpu.recon.algorithm + parameters: + center: !SweepRange # Reference to center of rotation side output, a float number or 'null' for a middle of the horizontal dimension. + start: 1100 + stop: 1300 + step: 25 + detector_pad: false # Horizontal detector padding to minimise circle/arc-type artifacts in the reconstruction. Set to 'true' to enable automatic padding or an integer + filter_freq_cutoff: 0.35 + recon_size: null + recon_mask_radius: 0.95 # Zero pixels outside the mask-circle radius. Make radius equal to 2.0 to remove the mask effect. diff --git a/docs/source/pipelines_full/sweep_paganin_FBP3d_tomobar.yaml b/docs/source/pipelines_full/sweep_paganin_FBP3d_tomobar.yaml new file mode 100644 index 000000000..7be66c89f --- /dev/null +++ b/docs/source/pipelines_full/sweep_paganin_FBP3d_tomobar.yaml @@ -0,0 +1,66 @@ +# This pipeline should be supported by the latest developments of HTTomo. Use module load httomo/latest module at Diamond. +# --- Standard tomography loader for NeXus files. --- +- method: standard_tomo + module_path: httomo.data.hdf.loaders + parameters: + data_path: auto + image_key_path: auto + rotation_angles: auto + preview: + detector_x: # horizontal data previewing/cropping. + # when null, the full data dimension is used, i.e., no previewing + start: null + stop: null + detector_y: # vertical data previewing/cropping. + # when null, the full data dimension is used, i.e., no previewing + start: null + stop: null + darks: null + flats: null + continuous_scan_subset: null +# --- Flat-field and dark-field projection correction. --- +- method: dark_flat_field_correction + module_path: httomolibgpu.prep.normalize + parameters: + flats_multiplier: 1.0 + darks_multiplier: 1.0 + upper_bound: null + lower_bound: null +# --- Center of Rotation auto-finding. Required for reconstruction below. --- +- method: find_center_vo + module_path: httomolibgpu.recon.rotation + parameters: + ind: null # A vertical slice (sinogram) index to calculate CoR, 'mid' can be used for middle + average_radius: 0 # Average several sinograms to improve SNR, one can try 3-5 range + cor_initialisation_value: null # Use if an approximate CoR is known + smin: -50 + smax: 50 + srad: 6.0 + step: 0.5 + ratio: 0.5 + drop: 20 + id: centering + side_outputs: + cor: centre_of_rotation # An estimated CoR value provided as a side output +# --- Apply a phase contrast filter to improve image contrast. --- +- method: paganin_filter + module_path: httomolibgpu.prep.phase + parameters: + pixel_size: 1.28 # Detector pixel size (resolution) in MICRON units. + distance: 1.0 # Propagation distance of the wavefront from sample to detector in METRE units. + energy: 53.0 # Beam energy in keV. + ratio_delta_beta: !Sweep # The ratio of delta/beta for filter strength control. Larger values lead to more smoothing. + - 10 + - 150 + - 350 + calculate_padding_value_method: next_power_of_2 # Select type of padding from 'next_power_of_2', 'next_fast_length' and 'use_pad_x_y'. + pad_x_y: null # Manual padding is enabled when 'calculate_padding_value_method' is set to 'use_pad_x_y'. +# --- Reconstruction method. --- +- method: FBP3d_tomobar + module_path: httomolibgpu.recon.algorithm + parameters: + center: ${{centering.side_outputs.centre_of_rotation}} # Reference to center of rotation side output, a float number or 'null' for a middle of the horizontal dimension. + detector_pad: false # Horizontal detector padding to minimise circle/arc-type artifacts in the reconstruction. Set to 'true' to enable automatic padding or an integer + filter_freq_cutoff: 0.35 + recon_size: null + recon_mask_radius: 0.95 # Zero pixels outside the mask-circle radius. Make radius equal to 2.0 to remove the mask effect. diff --git a/docs/source/pipelines_full/titaren_center_pc_FBP3d_resample.yaml b/docs/source/pipelines_full/titaren_center_pc_FBP3d_resample.yaml new file mode 100644 index 000000000..c0770158c --- /dev/null +++ b/docs/source/pipelines_full/titaren_center_pc_FBP3d_resample.yaml @@ -0,0 +1,87 @@ +# This pipeline should be supported by the latest developments of HTTomo. Use module load httomo/latest module at Diamond. +# --- Standard tomography loader for NeXus files. --- +- method: standard_tomo + module_path: httomo.data.hdf.loaders + parameters: + data_path: auto + image_key_path: auto + rotation_angles: auto + preview: + detector_x: # horizontal data previewing/cropping. + # when null, the full data dimension is used, i.e., no previewing + start: null + stop: null + detector_y: # vertical data previewing/cropping. + # when null, the full data dimension is used, i.e., no previewing + start: null + stop: null + darks: null + flats: null + continuous_scan_subset: null +# --- Flat-field and dark-field projection correction. --- +- method: dark_flat_field_correction + module_path: httomolibgpu.prep.normalize + parameters: + flats_multiplier: 1.0 + darks_multiplier: 1.0 + upper_bound: null + lower_bound: null +# --- Center of Rotation auto-finding. Required for reconstruction below. --- +- method: find_center_pc + module_path: httomolibgpu.recon.rotation + parameters: + proj1: auto + proj2: auto + tol: 0.5 + rotc_guess: null + id: centering + side_outputs: + cor: centre_of_rotation # An estimated CoR value provided as a side output +# --- Method to remove stripe artefacts in the data that lead to ring artefacts in the reconstruction. --- +- method: remove_stripe_ti + module_path: httomolibgpu.prep.stripe + parameters: + beta: 0.1 +# --- Negative log is required for reconstruction to convert raw intensity measurements into the line integrals of attenuation. --- +- method: minus_log + module_path: httomolibgpu.prep.normalize + parameters: {} +# --- Reconstruction method. --- +- method: FBP3d_tomobar + module_path: httomolibgpu.recon.algorithm + parameters: + center: ${{centering.side_outputs.centre_of_rotation}} # Reference to center of rotation side output, a float number or 'null' for a middle of the horizontal dimension. + detector_pad: false # Horizontal detector padding to minimise circle/arc-type artifacts in the reconstruction. Set to 'true' to enable automatic padding or an integer + filter_freq_cutoff: 0.35 + recon_size: null + recon_mask_radius: 0.95 # Zero pixels outside the mask-circle radius. Make radius equal to 2.0 to remove the mask effect. +# --- Down/up sampling the data. --- +- method: data_resampler + module_path: httomolibgpu.misc.morph + parameters: + newshape: REQUIRED # Provide a new shape for a 2D slice, e.g. [256, 256]. + axis: auto + interpolation: linear +# --- Calculate global statistics on the reconstructed volume, required for data rescaling. --- +- method: calculate_stats + module_path: httomo.methods + parameters: {} + id: statistics + side_outputs: + glob_stats: glob_stats +# --- Rescaling the data using min/max obtained from `calculate_stats`. --- +- method: rescale_to_int + module_path: httomolib.misc.rescale + parameters: + perc_range_min: 0.0 + perc_range_max: 100.0 + bits: 8 + glob_stats: ${{statistics.side_outputs.glob_stats}} +# --- Saving data into images. --- +- method: save_to_images + module_path: httomolib.misc.images + parameters: + subfolder_name: images + axis: auto + file_format: tif # `tif` or `jpeg` can be used. + asynchronous: true diff --git a/docs/source/pipelines_full/tomopy_gridrec.yaml b/docs/source/pipelines_full/tomopy_gridrec.yaml new file mode 100644 index 000000000..337dce71b --- /dev/null +++ b/docs/source/pipelines_full/tomopy_gridrec.yaml @@ -0,0 +1,82 @@ +# This pipeline should be supported by the latest developments of HTTomo. Use module load httomo/latest module at Diamond. +# --- Standard tomography loader for NeXus files. --- +- method: standard_tomo + module_path: httomo.data.hdf.loaders + parameters: + data_path: auto + image_key_path: auto + rotation_angles: auto + preview: + detector_x: # horizontal data previewing/cropping. + # when null, the full data dimension is used, i.e., no previewing + start: null + stop: null + detector_y: # vertical data previewing/cropping. + # when null, the full data dimension is used, i.e., no previewing + start: null + stop: null + darks: null + flats: null + continuous_scan_subset: null +# --- Removing unresponsive/dead pixels in the data, aka zingers. Use if sharp streaks are present in the reconstruction. To be applied before normalisation. --- +- method: remove_outlier + module_path: tomopy.misc.corr + parameters: + dif: 0.1 # A difference between the outlier value and the median value of neighbouring pixels. + size: 3 + axis: auto +# --------------------------------------------------------# +- method: normalize + module_path: tomopy.prep.normalize + parameters: + cutoff: null + averaging: mean +# --- Negative log is required for reconstruction to convert raw intensity measurements into the line integrals of attenuation. --- +- method: minus_log + module_path: tomopy.prep.normalize + parameters: {} +# --- Center of Rotation auto-finding. Required for reconstruction below. --- +- method: find_center_vo + module_path: tomopy.recon.rotation + parameters: + ind: null # A vertical slice (sinogram) index to calculate CoR, 'mid' can be used for middle + smin: -50 + smax: 50 + srad: 6 + step: 0.25 + ratio: 0.5 + drop: 20 + id: centering + side_outputs: + cor: centre_of_rotation # An estimated CoR value provided as a side output +# --- Reconstruction method. --- +- method: recon + module_path: tomopy.recon.algorithm + parameters: + center: ${{centering.side_outputs.centre_of_rotation}} # Reference to center of rotation side output, a float number or 'null' for a middle of the horizontal dimension. + sinogram_order: false + algorithm: gridrec # Select the required algorithm, e.g. 'gridrec' + init_recon: null +# --- Calculate global statistics on the reconstructed volume, required for data rescaling. --- +- method: calculate_stats + module_path: httomo.methods + parameters: {} + id: statistics + side_outputs: + glob_stats: glob_stats +# --- Rescaling the data using min/max obtained from `calculate_stats`. --- +- method: rescale_to_int + module_path: httomolib.misc.rescale + parameters: + perc_range_min: 0.0 + perc_range_max: 100.0 + bits: 8 + glob_stats: ${{statistics.side_outputs.glob_stats}} +# --- Saving data into images. --- +- method: save_to_images + module_path: httomolib.misc.images + parameters: + subfolder_name: images + axis: auto + file_format: tif # `tif` or `jpeg` can be used. + asynchronous: true diff --git a/docs/source/pipelines_full/tomopy_tomobank.yaml b/docs/source/pipelines_full/tomopy_tomobank.yaml new file mode 100644 index 000000000..c55a71a63 --- /dev/null +++ b/docs/source/pipelines_full/tomopy_tomobank.yaml @@ -0,0 +1,83 @@ +# This pipeline should be supported by the latest developments of HTTomo. Use module load httomo/latest module at Diamond. +# --- Standard tomography loader for NeXus files. --- +- method: standard_tomo + module_path: httomo.data.hdf.loaders + parameters: + data_path: auto + image_key_path: auto + rotation_angles: auto + preview: + detector_x: # horizontal data previewing/cropping. + # when null, the full data dimension is used, i.e., no previewing + start: null + stop: null + detector_y: # vertical data previewing/cropping. + # when null, the full data dimension is used, i.e., no previewing + start: null + stop: null + darks: null + flats: null + continuous_scan_subset: null +# --------------------------------------------------------# +- method: normalize + module_path: tomopy.prep.normalize + parameters: + cutoff: null + averaging: mean +# --- Center of Rotation auto-finding. Required for reconstruction below. --- +- method: find_center_vo + module_path: tomopy.recon.rotation + parameters: + ind: null # A vertical slice (sinogram) index to calculate CoR, 'mid' can be used for middle + smin: -50 + smax: 50 + srad: 6 + step: 0.25 + ratio: 0.5 + drop: 20 + id: centering + side_outputs: + cor: centre_of_rotation # An estimated CoR value provided as a side output +# --- Method to remove stripe artefacts in the data that lead to ring artefacts in the reconstruction. --- +- method: remove_all_stripe + module_path: tomopy.prep.stripe + parameters: + snr: 3 + la_size: 61 + sm_size: 21 + dim: 1 +# --- Negative log is required for reconstruction to convert raw intensity measurements into the line integrals of attenuation. --- +- method: minus_log + module_path: tomopy.prep.normalize + parameters: {} +# --- Reconstruction method. --- +- method: recon + module_path: tomopy.recon.algorithm + parameters: + center: ${{centering.side_outputs.centre_of_rotation}} # Reference to center of rotation side output, a float number or 'null' for a middle of the horizontal dimension. + sinogram_order: false + algorithm: gridrec # Select the required algorithm, e.g. 'gridrec' + init_recon: null +# --- Calculate global statistics on the reconstructed volume, required for data rescaling. --- +- method: calculate_stats + module_path: httomo.methods + parameters: {} + id: statistics + side_outputs: + glob_stats: glob_stats +# --- Rescaling the data using min/max obtained from `calculate_stats`. --- +- method: rescale_to_int + module_path: httomolib.misc.rescale + parameters: + perc_range_min: 0.0 + perc_range_max: 100.0 + bits: 8 + glob_stats: ${{statistics.side_outputs.glob_stats}} +# --- Saving data into images. --- +- method: save_to_images + module_path: httomolib.misc.images + parameters: + subfolder_name: images + axis: auto + file_format: tif # `tif` or `jpeg` can be used. + asynchronous: true diff --git a/docs/source/reference/cli.rst b/docs/source/reference/cli.rst new file mode 100644 index 000000000..ff5a51a7f --- /dev/null +++ b/docs/source/reference/cli.rst @@ -0,0 +1,278 @@ +.. _run-httomo-indepth: + +Command-line interface +====================== + +HTTomo provides a command-line interface (CLI) for validating and running +processing pipelines. + +Before using the CLI: + +* outside Diamond, activate the environment in which HTTomo is installed; +* at Diamond, run ``module load httomo``. + +Outside Diamond, invoke the CLI using ``python -m httomo``. Diamond users can +use ``httomo`` as a shortcut. + +To list the available commands, run: + +.. code-block:: console + + $ python -m httomo --help + +The main commands are: + +``check`` + Validate a YAML pipeline, optionally against an input HDF5 file. + +``memory-check`` + Estimate the peak CPU memory required to run a pipeline. + +``run`` + Run a pipeline on an input dataset. + +Use ``--help`` after any command to see its current arguments and options: + +.. code-block:: console + + $ python -m httomo run --help + + +The ``check`` command ++++++++++++++++++++++ + +Validate a YAML pipeline before running it: + +.. code-block:: console + + $ python -m httomo check PIPELINE [IN_DATA_FILE] + +``PIPELINE`` + Path to the YAML pipeline to validate. + +``IN_DATA_FILE`` + Optional path to the input HDF5 file. When supplied, HTTomo also checks that + the dataset paths referenced by the pipeline loader exist in the file. + +For details of the validation performed, see :ref:`utilities_yamlchecker`. + +.. note:: + + The ``check`` command accepts pipeline files only. Checking a pipeline + supplied as a string is not currently supported. + + +The ``memory-check`` command +++++++++++++++++++++++++++++ + +Estimate the peak CPU memory required to process a dataset: + +.. code-block:: console + + $ python -m httomo memory-check IN_DATA_FILE PIPELINE NPROCS + +``IN_DATA_FILE`` + Path to the input HDF5 file. + +``PIPELINE`` + Path to the pipeline that will process the data. + +``NPROCS`` + Number of processes that will run the pipeline. The value must be at least + one. + +The reported value is the estimated peak memory across all processes. It is +calculated from the estimated peak memory for one process multiplied by +``NPROCS``. + +The estimate accounts for the input data type and dimensions, loader previews, +padding, re-slicing and changes in data shape between pipeline sections. +See :ref:`memory_and_performance` for planning a process count and comparing +this total with the per-process runtime ceiling. + + +The ``run`` command ++++++++++++++++++++ + +Run a processing pipeline: + +.. code-block:: console + + $ python -m httomo run [OPTIONS] IN_DATA_FILE PIPELINE OUT_DIR + +Arguments +######### + +``IN_DATA_FILE`` + Path to an existing HDF5 input file. + +``PIPELINE`` + Path to a YAML pipeline. HTTomo can also accept a JSON pipeline supplied as + a string when ``--pipeline-format json`` is used. + +``OUT_DIR`` + Parent directory in which HTTomo creates the run output directory. + +By default, the output directory is named using the start time of the run: + +.. code-block:: text + + DD-MM-YYYY_HH_MM_SS_output + +For example, a run started at 15:30:45 on 1 May 2023 with ``OUT_DIR`` set to +``/home/myuser`` would write to: + +.. code-block:: text + + /home/myuser/01-05-2023_15_30_45_output/ + + +Options +####### + +Output and intermediate data +~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + +``--output-folder-name DIRECTORY`` + Use the given output-directory name instead of the timestamp-based default. + For example, ``--output-folder-name test-1`` creates ``OUT_DIR/test-1``. + +``--save-all`` + Save intermediate datasets for every task in the pipeline. Without this + option, datasets are saved only for tasks whose ``save_result`` setting is + enabled, either explicitly in the pipeline or by the method's default + configuration. + +``--save-snapshots`` + Save image snapshots at selected points in the pipeline. Snapshots are + useful for inspecting intermediate processing without saving every complete + intermediate dataset. + +``--intermediate-format hdf5`` + Store intermediate datasets in HDF5 format. This is currently the only + supported intermediate format and is selected by default. + +``--compress-intermediate`` + Store intermediate datasets in chunked HDF5 files with BLOSC compression. + +``--frames-per-chunk INTEGER`` + Set the number of frames per HDF5 chunk for intermediate data. The value + must be at least ``-1``: + + * ``-1`` selects the chunk size automatically and is the default; + * ``0`` uses contiguous storage; + * a positive value sets the number of frames per chunk. + + Compression requires chunked storage. If ``--compress-intermediate`` is + combined with ``--frames-per-chunk 0``, HTTomo changes the chunk setting to + ``-1`` and selects it automatically. + +``--recon-filename-stem NAME`` + Set the filename stem used for reconstruction output. HTTomo adds the + ``.h5`` extension. For example, ``--recon-filename-stem my-recon`` produces + ``my-recon.h5``. + + +Execution and resource use +~~~~~~~~~~~~~~~~~~~~~~~~~~ + +``--gpu-id INTEGER`` + Select the GPU device to use. The default is ``-1``, which does not + explicitly select a different CUDA device. + +``--max-memory SIZE`` + Set a per-process memory ceiling. Values may be supplied as bytes or with a + ``K``, ``M`` or ``G`` suffix, for example + ``--max-memory 32G``. + + When the estimated memory for a pipeline section reaches this limit, HTTomo + uses disk-backed intermediate storage. For GPU sections, the same value also + caps the memory budget used to calculate the block size; HTTomo uses the + smaller of this ceiling and the available GPU memory. The default is ``0``, + which disables the user-supplied ceiling. GPU block sizing still respects + the memory reported by the device. + + See :ref:`memory_and_performance` for practical sizing guidance. + +``--max-cpu-slices INTEGER`` + Set the maximum number of slices in a block for CPU-only pipeline sections. + The value must be at least one and defaults to ``64``. + + Adjusting this value may affect the performance of CPU-only processing. See + :ref:`detailed_about` for information about blocks, chunks and sections. + +``--reslice-dir DIRECTORY`` + Choose the directory used for temporary re-slicing files. The directory + must already exist and be writable. The run output directory is used by + default. + + When the output is on network-mounted storage, using a local temporary + directory can substantially improve file-based re-slicing performance. For + a multi-node run, the directory must be accessible to every participating + process. + +``--continuous-scan-subset START STOP`` + Select a subset of projections along the angular dimension. This option + overrides the ``continuous_scan_subset`` value in the pipeline loader + configuration. See :ref:`continuous_scan_subset_selection`. + +``--mpi-abort-hook`` + Abort all MPI processes when any process encounters an unhandled exception. + This prevents the remaining processes from waiting indefinitely for a failed + process. + + This option is mainly intended for debugging. Because termination occurs at + the MPI level, the exception traceback may be incomplete. + + +Pipeline format and parameter sweeps +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + +.. _pipeline-format: + +``--pipeline-format {yaml,json}`` + Select the pipeline format. The value is case-insensitive and defaults to + YAML. + + YAML pipelines must be provided as files. JSON pipelines must be provided + as strings. + +``--bits-sweep-images INTEGER`` + Set the bit depth of TIFF images produced by a + :ref:`parameter_sweeping` run. Use ``8``, ``16`` or ``32``. The default is + ``32``. + + The CLI currently accepts any integer, although the supported output bit + depths are 8, 16 and 32. + + +Monitoring +~~~~~~~~~~ + +``--monitor NAME`` + Enable a performance monitor. The available monitors are ``summary`` and + ``bench``. This option can be supplied more than once. + + ``summary`` + Report aggregate timings and a per-method breakdown. + + ``bench`` + Report detailed timings for every process, including CPU and GPU + execution, data transfers and file operations. + +``--monitor-output FILENAME`` + Write monitoring results to a file. By default, results are written to + standard output. + + The ``summary`` monitor produces human-readable text, while the ``bench`` + monitor produces CSV data. + + +System logging +~~~~~~~~~~~~~~ + +``--syslog-host HOST`` + Set the hostname of the syslog server. The default is ``localhost``. + +``--syslog-port PORT`` + Set the syslog server port. The default is ``514``. diff --git a/docs/source/reference/compatibility.rst b/docs/source/reference/compatibility.rst new file mode 100644 index 000000000..7b2c8466c --- /dev/null +++ b/docs/source/reference/compatibility.rst @@ -0,0 +1,72 @@ +.. _compatibility: + +Version compatibility +===================== + +Use the Python, NumPy and CuPy versions declared by the installed HTTomo +release. These packages affect the runtime interface and should not be upgraded +independently without testing the complete environment. + +.. list-table:: HTTomo 3.x runtime requirements + :header-rows: 1 + :widths: 18 20 22 20 20 + + * - HTTomo release + - Python + - NumPy + - CuPy + - Pipeline archive + * - 3.3.2 + - 3.12 or later + - 2.4.x + - 14.2.x + - 3.3 + * - 3.3.0--3.3.1 + - 3.12 or later + - 2.4.x + - 14.0.x + - 3.3 + * - 3.2.x + - 3.12 or later + - 2.4.x + - 14.0.x + - 3.2 + * - 3.1.x + - 3.12 or later + - 2.4.x + - 14.0.x + - 3.1 + +The current installation recipe uses OpenMPI 4.1.6, a parallel build of h5py, +and TomoPy 1.15.3 when TomoPy methods are required. ``mpi4py`` and h5py must be +built against compatible MPI libraries. CuPy's CUDA runtime must also be +supported by the installed NVIDIA driver. + +Backend packages +---------------- + +HTTomo deliberately does not pin HTTomolib, HTTomolibGPU, TomoBAR or +``httomo-backends`` to one version. Their interfaces and metadata evolve +together, so treat them as one tested environment: + +* use :ref:`versioned_downloads` for templates and pipelines matching a tagged + HTTomo release; +* do not combine generated templates from ``main`` with an older release; +* rerun ``python -m httomo check PIPELINE INPUT`` after changing a backend; +* test a representative small dataset before a production run; and +* report the versions of HTTomo, ``httomo-backends`` and all processing + libraries when requesting support. + +The documentation build for this branch uses the ``httomo-backends`` version +pinned in ``docs/source/doc-pip-requirements.txt`` to generate the complete +pipeline examples. The current method reference is published separately by +``httomo-backends``. Neither resource is a universal runtime compatibility +guarantee. + +Release information +------------------- + +Consult the `HTTomo releases page +`_ for changes and +upgrade notes. The authoritative requirements for any tag are its +``pyproject.toml`` and installation documentation. diff --git a/docs/source/reference/glossary.rst b/docs/source/reference/glossary.rst new file mode 100644 index 000000000..c2eae1b9e --- /dev/null +++ b/docs/source/reference/glossary.rst @@ -0,0 +1,104 @@ +.. _glossary: + +Glossary +======== + +.. glossary:: + + backend + A scientific processing library whose functions HTTomo can call, such as + HTTomolibGPU, HTTomolib or TomoPy. See :ref:`backends_list`; developers + should also read :ref:`developers_httomo_backends`. + + block + The memory-sized portion of one process's chunk passed through every + method in a section before the next block is processed. See + :ref:`blocks_data`. + + chunk + The portion of a section's dataset assigned to one MPI process. See + :ref:`chunks_data`. + + centre of rotation + The detector coordinate corresponding to the sample's rotation axis. + Accurate centring prevents characteristic reconstruction artefacts. See + :ref:`centering`. + + image key + A one-dimensional dataset that identifies each input frame as a + projection, flat or dark image. See :ref:`darks_flats` and + :ref:`create_nxtomo`. + + intermediate dataset + A complete volume saved after a pipeline method, usually as an HDF5 file + with the main array at ``/data``. See :ref:`info_logger`. + + loader + The first pipeline method, responsible for locating input arrays and + presenting projection data and auxiliary information to HTTomo. See + :ref:`standard_tomo_loader`. + + method + One loader, processing or output operation configured as an entry in a + pipeline. See :ref:`pipeline_file_reference` for the fields that define + a method entry and :ref:`reference_templates` for supported methods. + + method template + A YAML description of a supported method, its parameters and defaults, + generated from ``httomo-backends`` metadata. See + :ref:`explanation_templates` and :ref:`reference_templates`. + + monitor + Optional runtime instrumentation that reports aggregate or block-level + timings. See :ref:`info_logger` and :ref:`run-httomo-indepth`. + + NXtomo + The NeXus application definition for tomography data. An NXtomo entry + links projection, image-key and rotation-angle datasets in a standard + hierarchy that HTTomo can discover automatically. See + :ref:`create_nxtomo`. + + padding + Extra neighbouring slices supplied to a method so that operations near a + block boundary have sufficient context. See :ref:`padding`. + + parameter sweep + Repeated execution of a method for several candidate parameter values, + with images saved for comparison. See :ref:`parameter_sweeping`. + + pattern + The orientation in which methods consume data, principally projection or + sinogram order. Pattern changes determine section boundaries and can + require a re-slice; see :ref:`info_sections` and :ref:`info_reslice`. + + pipeline + The ordered sequence of methods that HTTomo executes. Older material may + call this a *process list*. See :ref:`explanation_process_list`, + :ref:`pipeline_file_reference` and :ref:`tutorials_pl_templates`. + + preview + A loader selection that crops the angular or detector dimensions before + processing. See :ref:`previewing`. + + rank + The identifier of one MPI process participating in a parallel run. Each + rank normally receives one chunk; see :ref:`chunks_data` and + :ref:`fig_execution_model`. + + re-slice + Redistribution or transposition of data when consecutive sections use + different processing patterns. See :ref:`info_reslice`. + + section + A consecutive group of compatible methods that use the same processing + pattern and are sized and executed together. See :ref:`info_sections` + and :ref:`fig_execution_model`. + + side output + A named value produced alongside the main dataset and referenced by a + later pipeline method. See :ref:`side_output` for syntax and examples. + + wrapper + HTTomo's adapter between a backend function and the common pipeline + execution interface. See :ref:`info_wrappers` and + :ref:`developer_architecture`. diff --git a/docs/source/reference/loaders.rst b/docs/source/reference/loaders.rst deleted file mode 100644 index 292ea976b..000000000 --- a/docs/source/reference/loaders.rst +++ /dev/null @@ -1,305 +0,0 @@ -.. _reference_loaders: - -HTTomo Loaders --------------- - -Available Loaders -================= - -HTTomo currently has one loader, :code:`standard_tomo_loader`, which is geared -towards loading "standard" tomography data collected at DLS beamlines. - -Basic Usage of the Standard Loader -================================== - -This loader has several parameters which are fairly self-explanatory. Other -parameters either require some more detail to use, or have extra capabilities which -are not obvious if one has seen only one or two simple loader configurations. - -The following YAML configuration of the loader shows all the parameters needed, and -is the standard case (ie, there's no configuration in it for special cases), so it -serves as a good starting point: - -.. code-block:: yaml - - - method: standard_tomo - module_path: httomo.data.hdf.loaders - parameters: - data_path: /entry1/tomo_entry/data/data - image_key_path: /entry1/tomo_entry/instrument/detector/image_key - rotation_angles: - data_path: /entry1/tomo_entry/data/rotation_angle - -.. note:: The input data being loaded is assumed to be in hdf5/NeXuS file format, - in accordance with the data typically collected at a DLS beamline. - -.. _nxtomo_discovery: - -Automatic `NXtomo` Discovery -++++++++++++++++++++++++++++ - -If the input file has a valid `NXtomo` entry (see the `NXtomo application -definition `_ -for more details) then the loader can be configured to automatically discover -it, without needing to explicitly specify values like the dataset path. - -This configuration is done by providing the :code:`auto` value to the following -parameters: - -- :code:`data_path` -- :code:`image_key_path` -- :code:`rotation_angles` - -For example: - -.. code-block:: yaml - - - method: standard_tomo - module_path: httomo.data.hdf.loaders - parameters: - data_path: auto - image_key_path: auto - rotation_angles: auto - -.. note:: Automatic :code:`NXtomo` discovery (and therefore the :code:`auto` value) is not - supported when the darks/flats are separate from the projection data. - -Manually Providing Dataset Paths -++++++++++++++++++++++++++++++++ - -:code:`data_path` -~~~~~~~~~~~~~~~~~ - -The :code:`data_path` parameter is the path to the dataset in the input hdf5/NeXuS -file containing the image data (usually, projections + darks + flats are in the -same dataset). - -:code:`image_key_path` -~~~~~~~~~~~~~~~~~~~~~~ - -The :code:`image_key_path` parameter is the path to the dataset in the input -hdf5/NeXuS file containing the so called "image key". - -.. note:: The "image key" is an array whose length is the same as the number of - images in the collected data. Each element has a value of 0, 1, or 2, to - indicate a projection (0), flat field image (1), or dark-field image (2). - -:code:`rotation_angles` -~~~~~~~~~~~~~~~~~~~~~~~ - -Typically, the rotation angle values are stored in a dataset within the input -hdf5/NeXuS file. This dataset is usually what is provided to serve as the -rotation angle values during processing. - -In such cases, the :code:`rotation_angles` parameter has two lines of -configuration. Specifying :code:`data_path` is meaning that the rotation angles are -indeed stored in a dataset within the input hdf5/NeXuS file, and the given path is -the path to that dataset within the input hdf5/NeXuS file. - -The reason the :code:`rotation_angles` parameter isn't simply one value (a path) -like the previously mentioned parameters is because there are situations when the -angles dataset in the input hdf5/NeXus file doesn't exist, or cannot be used. - -To configure the loader to handle such cases, please refer to -:ref:`user_defined_angles`. - - -Dealing with darks and flats -============================ - -HTTomo currently supports several options to deal with the flats and darks images: - -1. Darks, flats and projections are all stored in the same dataset in the same file -2. Darks and/or flats are stored in separate files from the file containing the projections, - and the dataset that contains the darks or flats contains **only** darks or flats - - .. note:: For example, the darks are in a separate file to the file containing the - projections, and the dataset within the separate file containing the darks - contains **only** darks and no other images. - - This means that no additional image key dataset is required for fetching the - darks, as there is no need to identify which indices within that dataset are what - kind of image. - -3. Darks and/or flats are stored in a separate file from the file containing the projections, - and the dataset that contains the darks or flats contains **multiple** kinds of images (ie, - the dataset contains darks, flats and projections) - - .. note:: For example, the darks are in a separate file to the file containing the - projections, and the dataset within the separate file containing the darks - contains darks, flats and projections. - - This means that an additional image key dataset is required for fetching the - darks, as there is a need to identify which indices within that dataset are what - kind of image, in order to extract the desired darks. - -4. Darks and/or flats do not exist, neither in the same file as the projections nor in separate - files -5. Darks and/or flats exist in the same dataset as the projections, but they need to be ignored - -Files that do not contain image keys -++++++++++++++++++++++++++++++++++++ - -These are the files without the image keys that contain only flats or darks in two separate files. -Here one needs to add :code:`darks` and :code:`flats` parameters to the loader parameters with the following fields (see the example below): - -- :code:`file`, the path to the hdf5/NeXus file containing the darks/flats. The :code:`input_data` keyword can be also used, see more info bellow. -- :code:`data_path`, the dataset within the hdf5/NeXus file that contains the - darks/flats - -.. code-block:: yaml - :emphasize-lines: 5,6,8,9 - - - method: standard_tomo - module_path: httomo.data.hdf.loaders - parameters: - darks: - file: path/to/new/file.nxs - data_path: /entry1/tomo_entry/data/data - flats: - file: path/to/new/file.nxs - data_path: /entry1/tomo_entry/data/data - -If darks and/or flats are in the same dataset as the input dataset, one can use the :code:`input_data` shortcut for the :code:`file` field. For instance, -in the situation when darks and flats are the part of the same input dataset: - -.. code-block:: yaml - :emphasize-lines: 5,8 - - - - method: standard_tomo - module_path: httomo.data.hdf.loaders - parameters: - darks: - file: input_data - data_path: /exchange/darks - flats: - file: input_data - data_path: /exchange/flats - - -Files with image keys -+++++++++++++++++++++ - -This can be the case when a new scan is performed, which contains the required image keys. Therefore the keys -in the older scan should be ignored. In this instance, we need to provide a parameter :code:`image_key_path` in addition to -:code:`file` and :code:`data_path` fields. - -.. code-block:: yaml - :emphasize-lines: 7,11 - - - - method: standard_tomo - module_path: httomo.data.hdf.loaders - parameters: - darks: - file: path/to/new/file.nxs - data_path: /entry1/tomo_entry/data/data - image_key_path: /entry1/tomo_entry/instrument/detector/image_key - flats: - file: path/to/new/file.nxs - data_path: /entry1/tomo_entry/data/data - image_key_path: /entry1/tomo_entry/instrument/detector/image_key - - -Data without darks/flats -++++++++++++++++++++++++ - -It is also possible to process the data that does not contain darks or flats, i.e., the pipeline runs without given darks or flats. -Nothing specific should be done about it in the loader, it will be handled automatically without any extra configuration needed. - -Ignore darks/flats -++++++++++++++++++ - -This is the case when darks or flats still present in the dataset, but one needs to ignore either of them or both of them. This can be done by providing -the keyword :code:`ignore` into the loader, like in the example below where both flats and darks are ignored: - -.. code-block:: yaml - :emphasize-lines: 4-5 - - - - method: standard_tomo - module_path: httomo.data.hdf.loaders - parameters: - darks: ignore - flats: ignore - -.. _user_defined_angles: - -Providing/Overriding Angles Data -================================ - -There are several situations in which overriding the angles dataset in the input -hdf5/NeXuS file, or generating an array due to the absence of an angles dataset, is -necessary. The loader offers the ability to specify an angles array via: - -- start angle -- stop angle -- total number of angles - -values by configuring the :code:`rotation_angles` parameter slightly differently -than shown earlier. - -The following is reusing the same example from the separate darks/flats example, -but is now drawing attention to the :code:`rotation_angles` parameter: - - -.. code-block:: yaml - :emphasize-lines: 8-10 - - - - method: standard_tomo - module_path: httomo.data.hdf.loaders - parameters: - data_path: /1-TempPlugin-tomo/data - image_key_path: /entry1/tomo_entry/instrument/detector/image_key - rotation_angles: - user_defined: - start_angle: 0 - stop_angle: 180 - angles_total: 724 - -It can be seen that :code:`user_defined` has been specified instead of -:code:`data_path`. Furthermore, there are then three fields provided: - -- :code:`start_angle`, which is the first angle (in degrees) -- :code:`stop_angle`, which is the last angle (in degrees) -- :code:`angles_total`, which is the number of angles on total to have in that - range, equally spaced - -to generate the desired angles array that HTTomo will use during pipeline -execution. - -Previewing -========== - -The data being loaded with the loader can be cropped/previewed prior to being -passed along to the first method. The loader has the :code:`preview` parameter -for configuring the cropping/previewing. Please see :ref:`previewing` for more -details on previewing. - -.. _continuous_scan_subset_selection: - -Continuous Scan Subset Selection -================================ - -Another data configuration that is supported is a single 3D hdf5 dataset -containing multiple tomography scans in sequence along the angular dimension. -HTTomo provides the ability to select a subset of the data along the angular -dimension, allowing the loading and processing of the individual tomography -scans within the 3D hdf5 dataset. - -This feature is configured using the parameter :code:`continuous_scan_subset`, -where a start and stop value describing the subset along the angular dimension -is required. The following shows an example of selecting the subset starting at -index 90 and ending at index 179 (similar to :code:`preview`, the :code:`stop` -index is excluded, which is why 180 is given): - - .. literalinclude:: ../../../tests/samples/pipeline_template_examples/testing/loader_with_offset_param.yaml - :language: yaml - :emphasize-lines: 7-9 - -This optional parameter can be used in conjunction with other optional -parameters, such as using :code:`preview` to crop the :code:`detector_x` and/or -:code:`detector_y` dimensions of the selected subset, or using the -:code:`darks` and :code:`flats` parameters to load external darks/flats. diff --git a/docs/source/reference/pipeline_file.rst b/docs/source/reference/pipeline_file.rst new file mode 100644 index 000000000..027713561 --- /dev/null +++ b/docs/source/reference/pipeline_file.rst @@ -0,0 +1,117 @@ +.. _pipeline_file_reference: +.. _explanation_yaml: + +Pipeline file reference +======================= + +An HTTomo pipeline file is a YAML sequence of method entries. HTTomo executes +the entries from top to bottom, passing the main dataset from one method to the +next. The first entry must be a loader. + +YAML formatting +--------------- + +YAML uses indentation to define structure. Use spaces rather than tabs, keep +indentation consistent and include a space after each colon. Comments begin +with ``#``. + +Common values include: + +``null`` + No value. This is converted to Python ``None``. + +``true`` and ``false`` + Boolean values all in small letters. These are converted to Python ``True`` and ``False``. + +``[value1, value2]`` + A list written on one line. Lists may also be written across multiple lines. + +Strings normally do not need quotation marks. Quote a string when it contains +YAML punctuation or might otherwise be interpreted as a number, Boolean or +``null`` value. + +Method entries +-------------- + +Each list entry configures one loader or processing method: + +.. code-block:: yaml + + - method: median_filter3d + module_path: tomopy.misc.corr + parameters: + size: 3 + +The supported fields are: + +``method`` + Required. The name of the Python function to run. + +``module_path`` + Required. The import path containing the function. + +``parameters`` + Required. A mapping of parameter names to values. Use ``parameters: {}`` + when the method has no configurable parameters. + +``id`` + Optional. A unique name used when another method references this entry's + side output. + +``side_outputs`` + Optional. Maps an output name supplied by the method to a pipeline name. See + :ref:`side_output`. + +``save_result`` + Optional Boolean. Controls whether the main dataset is saved after this + method. See :ref:`save-result-examples`. + +Use :ref:`reference_templates` for the supported method names, module paths, +parameters and defaults. + +Side-output references +---------------------- + +A later method can use a named side output with the following syntax: + +.. code-block:: yaml + + ${{method_id.side_outputs.output_name}} + +The producing method must appear earlier in the pipeline, and every explicit +``id`` must be unique. See :ref:`side_output` for a complete example. + +Parameter sweeps +---------------- + +Use ``!Sweep`` for an explicit list of parameter values and ``!SweepRange`` for +a start, stop and step. A pipeline can contain only one sweep. See +:ref:`parameter_sweeping` for syntax and output behaviour. + +Minimal pipeline +---------------- + +This example loads an NXtomo file, normalises the projections and applies the +negative logarithm: + +.. code-block:: yaml + + - method: standard_tomo + module_path: httomo.data.hdf.loaders + parameters: + data_path: auto + image_key_path: auto + rotation_angles: auto + + - method: normalize + module_path: tomopy.prep.normalize + parameters: + cutoff: null + averaging: mean + + - method: minus_log + module_path: tomopy.prep.normalize + parameters: {} + +Before running a pipeline, follow :ref:`utilities_yamlchecker` to validate its +structure, methods, parameters and input-data paths. diff --git a/docs/source/reference/run_output.rst b/docs/source/reference/run_output.rst new file mode 100644 index 000000000..6890a9032 --- /dev/null +++ b/docs/source/reference/run_output.rst @@ -0,0 +1,147 @@ +.. _info_logger: + +Run output, logs and monitoring +=============================== + +Unless ``--output-folder-name`` is supplied, HTTomo creates a timestamped run +directory named ``DD-MM-YYYY_HH_MM_SS_output`` below ``OUT_DIR``. Every run +contains: + +``user.log`` + The concise progress information also shown in the terminal. + +``debug.log`` + Detailed diagnostic information, including messages from individual ranks. + +Pipeline copy + A copy retaining the source filename, with omitted default parameters added + for ordinary YAML runs. JSON input is recorded as ``pipeline.json``. + +Depending on the pipeline and command-line options, the directory may also +contain intermediate HDF5 files, image directories, snapshots and monitoring +output. + +See :ref:`run-httomo-indepth` for the output-related command-line options. + +Intermediate HDF5 files ++++++++++++++++++++++++ + +``save_result: true`` saves the result after that method. ``--save-all`` adds +equivalent saves after every eligible method. Files normally use this pattern: + +.. code-block:: text + + TASK-ID-PACKAGE-METHOD[-ALGORITHM].h5 + +For reconstruction output, ``--recon-filename-stem NAME`` replaces that stem +and produces ``NAME.h5``. Each intermediate file stores the main volume at +``/data`` and also records ``/angles`` and +``/data_dims/detector_x_y``. + +By default, HTTomo selects an HDF5 chunk length automatically. +``--frames-per-chunk`` can set it explicitly, and +``--compress-intermediate`` enables BLOSC compression. Compression requires +chunked storage, so requesting contiguous storage together with compression +falls back to automatic chunking. + +Image output and sweeps ++++++++++++++++++++++++ + +The ``save_to_images`` method controls the image directory, format and bit +depth. The backend writer may append the bit depth and format to the configured +``subfolder_name``; for example, the quickstart's ``images`` configuration +produces ``images8bit_tif``. + +A :term:`parameter sweep` automatically saves images after each swept method; +do not add a separate ``save_to_images`` immediately after it. Sweep directories +use the method name and selected bit depth, for example +``images_sweep_FBP3d_tomobar32bit_tif``. + +Snapshots ++++++++++ + +``--save-snapshots`` writes representative JPEG images to +``pipeline_stages_snapshots``. Snapshots are intended for rapid inspection and +debugging, not as quantitative output. + +Monitoring output ++++++++++++++++++ + +Use ``--monitor summary`` for aggregate timing text or ``--monitor bench`` for +one CSV row per source, method, sink and total timing event. Direct output to a +file with ``--monitor-output FILE``; the default is standard output. More than +one ``--monitor`` option may be supplied. + +The benchmark CSV contains these fields: + +.. list-table:: + :header-rows: 1 + :widths: 30 70 + + * - Field + - Meaning + * - ``Type``, ``Rank`` + - Event type and MPI rank that produced it. + * - ``Name``, ``Task id``, ``Module`` + - Pipeline operation and its configured identity. + * - ``Slicing dim`` + - Axis along which the block was divided. + * - ``Block offset (chunk)``, ``Block offset (global)`` + - Block position within the rank's chunk and complete dataset. + * - ``Block dim z``, ``Block dim y``, ``Block dim x`` + - Shape of the recorded block. + * - ``CPU time`` + - Host elapsed time for the event. + * - ``GPU kernel time``, ``GPU H2D time``, ``GPU D2H time`` + - GPU execution and transfer times, or zero for CPU-only events. + + +.. _fig_log: + +.. figure:: ../_static/log/log_screenshot.png + :scale: 40 % + :alt: HTTomo terminal output + + Terminal output, which is also recorded in ``user.log``. + +Common log messages ++++++++++++++++++++ + +``Pipeline has been separated into N sections`` + The pipeline has been divided into ``N`` :ref:`info_sections`. Each section + groups methods that process the data as :ref:`chunks_data` and + :ref:`blocks_data`. A section processes all its input data before the next + section starts. + +``Running loader`` + The loader runs before the pipeline sections. It initially loads data using + the ``projection`` pattern, which is also used by the first section. See + :ref:`info_reslice`. + +``Section N with the following methods`` + The listed methods run sequentially on each block in the section. + ``Finished processing the last block`` indicates that the section has + processed all its input data. + +A progress bar may look like this: + +.. code-block:: text + + 50%|##### | 1/2 [00:02<00:02, 2.52s/block] + +It reports progress through the data blocks, not through individual methods: + +``50%`` and ``1/2`` + One of two blocks has been processed. + +``00:02<00:02`` + Two seconds have elapsed and approximately two seconds remain. + +``2.52s/block`` + The estimated processing time per block. For faster operations, this may + instead be displayed as blocks per second (``block/s``). + +.. note:: + + A block may be processed by several methods. The progress bar therefore + counts completed blocks rather than completed methods. diff --git a/docs/source/reference/yaml.rst b/docs/source/reference/yaml.rst deleted file mode 100644 index 0d60fcbb4..000000000 --- a/docs/source/reference/yaml.rst +++ /dev/null @@ -1,74 +0,0 @@ -.. _explanation_yaml: - -What is YAML? -------------- - -.. code-block:: yaml - - # What does YAML mean?​ - YAML:​ - -Y: YAML​ - -A: Ain't​ - -M: Markup​ - -L: Language - -.. note:: - - Markup Language is language which annotates text using tags or keywords - in order to define how content is displayed - - -* YAML documents are a collection of key-value pairs​ -* Indentation is used to denote structure -* The length of the indentation should not matter, as long as you are consistent inside the same file - -Format examples -=============== - -.. code-block:: yaml - - # A list of numbers using hyphens:​ - numbers:​ - - one​ - - two​ - - three​ - ​ - # The inline version:​ - numbers: [ one, two, three ] - - -* Indent with *spaces*, *not* tabs​. -* There must be spaces between elements​ - - -.. code-block:: - - # This is correct​ - name: tomo​ - - # This will fail​ - name:tomo - - -.. code-block:: yaml - - # Strings don't require quotes:​ - title: Introduction to YAML​ - ​ - # But you can still use them if you prefer:​ - title-with-quotes: 'Introduction to YAML'​ - - -Using a YAML file to specify a pipeline of functions for data processing is a common -and practical approach to creating a reproducible data analysis workflow. This approach -can help to ensure that your pipeline runs correctly and consistently over time, -regardless of the platform or environment in which it is executed. By using a YAML file -to define your pipeline, we are providing a user interface that is simple and intuitive for scientists -to use. This can be especially helpful for those who are less familiar with programming in general -or are new to the specific tools and libraries you are using. - -We have a :ref:`utilities_yamlchecker` that can help you to validate your YAML file. -Before running your pipeline, we highly recommend that you validate your YAML file using this utility. -The checker will help you to identify errors in your YAML file. - -We also recommend to use editors that `support `_ YAML format naturally, e.g.: Atom, Visual Studio Code, Notepad++, and others. diff --git a/docs/source/scripts/create_nxtomo.py b/docs/source/scripts/create_nxtomo.py new file mode 100644 index 000000000..6945c7696 --- /dev/null +++ b/docs/source/scripts/create_nxtomo.py @@ -0,0 +1,307 @@ +#!/usr/bin/env python3 +"""Create an NXtomo file that can be loaded by HTTomo. + +The module can be imported to pack NumPy arrays, or run as a command-line +program to pack stacks of TIFF files. Projection data must use the axis order +``(rotation_angle, detector_y, detector_x)``. +""" + +from __future__ import annotations + +import argparse +import glob +import re +from datetime import datetime, timezone +from pathlib import Path +from typing import Sequence + +import h5py +import numpy as np + + +def _nx_group(parent: h5py.Group, name: str, nx_class: str) -> h5py.Group: + """Create a NeXus group and set its ``NX_class`` attribute.""" + group = parent.create_group(name) + group.attrs["NX_class"] = nx_class + return group + + +def _as_frame_stack(array: np.ndarray, name: str) -> np.ndarray: + """Return a numeric array with shape ``(frames, detector_y, detector_x)``.""" + array = np.asarray(array) + if array.ndim == 2: + array = array[np.newaxis, ...] + if array.ndim != 3: + raise ValueError(f"{name} must be a 2D image or a 3D frame stack") + if 0 in array.shape: + raise ValueError(f"{name} must not contain an empty dimension") + if not ( + np.issubdtype(array.dtype, np.integer) + or np.issubdtype(array.dtype, np.floating) + ): + raise TypeError(f"{name} must have a real numeric dtype, not {array.dtype}") + if not np.isfinite(array).all(): + raise ValueError(f"{name} contains NaN or infinite values") + return array + + +# [write-nxtomo-start] +def write_nxtomo( + output_file: str | Path, + projections: np.ndarray, + angles: np.ndarray, + *, + flats: np.ndarray | None = None, + darks: np.ndarray | None = None, + sample_name: str = "sample", + compression: str | None = "gzip", + overwrite: bool = False, +) -> Path: + """Write NumPy arrays to an HTTomo-compatible NXtomo file. + + Parameters + ---------- + output_file + Destination ``.nxs`` or ``.h5`` file. + projections + Projection stack with shape ``(angles, detector_y, detector_x)``. + angles + One rotation angle in degrees per projection. + flats, darks + Optional calibration stacks. A single 2D image is also accepted. + sample_name + Descriptive sample name stored in the NXsample group. + compression + HDF5 compression: ``"gzip"``, ``"lzf"``, or ``None``. + overwrite + Replace ``output_file`` if it already exists. + """ + projections = _as_frame_stack(projections, "projections") + angles = np.asarray(angles) + if angles.ndim != 1 or len(angles) != len(projections): + raise ValueError("angles must be 1D with one value per projection") + if ( + not ( + np.issubdtype(angles.dtype, np.integer) + or np.issubdtype(angles.dtype, np.floating) + ) + or not np.isfinite(angles).all() + ): + raise ValueError("angles must contain finite numeric values") + + optional_stacks = [] + for name, stack in (("darks", darks), ("flats", flats)): + if stack is None: + optional_stacks.append(None) + continue + stack = _as_frame_stack(stack, name) + if stack.shape[1:] != projections.shape[1:]: + raise ValueError( + f"{name} frame shape {stack.shape[1:]} does not match " + f"projection frame shape {projections.shape[1:]}" + ) + optional_stacks.append(stack) + darks, flats = optional_stacks + + compression = None if compression == "none" else compression + if compression not in (None, "gzip", "lzf"): + raise ValueError("compression must be 'gzip', 'lzf', or None") + + stacks = [stack for stack in (darks, flats, projections) if stack is not None] + output_dtype = np.result_type(*(stack.dtype for stack in stacks)) + dark_count = 0 if darks is None else len(darks) + flat_count = 0 if flats is None else len(flats) + projection_count = len(projections) + frame_count = dark_count + flat_count + projection_count + detector_y, detector_x = projections.shape[1:] + + image_key = np.concatenate( + ( + np.full(dark_count, 2, dtype=np.int8), + np.full(flat_count, 1, dtype=np.int8), + np.zeros(projection_count, dtype=np.int8), + ) + ) + # NXtomo requires an angle for every frame. Calibration frames use the + # first projection angle; HTTomo selects projection angles via image_key. + frame_angles = np.concatenate( + (np.full(dark_count + flat_count, angles[0]), angles) + ).astype(np.float32, copy=False) + + output_file = Path(output_file) + output_file.parent.mkdir(parents=True, exist_ok=True) + mode = "w" if overwrite else "w-" + now = datetime.now(timezone.utc).isoformat() + + with h5py.File(output_file, mode) as nexus_file: + nexus_file.attrs.update( + default="entry", file_name=output_file.name, file_time=now + ) + + entry = _nx_group(nexus_file, "entry", "NXentry") + entry.attrs["default"] = "data" + entry.create_dataset("definition", data="NXtomo") + entry.create_dataset("title", data=sample_name) + entry.create_dataset("start_time", data=now) + + instrument = _nx_group(entry, "instrument", "NXinstrument") + detector = _nx_group(instrument, "detector", "NXdetector") + data = detector.create_dataset( + "data", + shape=(frame_count, detector_y, detector_x), + dtype=output_dtype, + chunks=(1, min(detector_y, 256), min(detector_x, 256)), + compression=compression, + ) + data.attrs.update( + interpretation="image", + axes="frame,detector_y,detector_x", + ) + key = detector.create_dataset("image_key", data=image_key) + key.attrs["meaning"] = "0=projection; 1=flat; 2=dark" + + first = 0 + for stack in (darks, flats, projections): + if stack is not None: + data[first : first + len(stack)] = stack + first += len(stack) + + sample = _nx_group(entry, "sample", "NXsample") + sample.create_dataset("name", data=sample_name) + rotation_angle = sample.create_dataset("rotation_angle", data=frame_angles) + rotation_angle.attrs["units"] = "deg" + + # NXdata contains hard links, not copies. These paths are also the + # locations used by HTTomo's automatic NXtomo discovery. + nx_data = _nx_group(entry, "data", "NXdata") + nx_data.attrs["signal"] = "data" + nx_data.attrs["axes"] = np.asarray( + ["rotation_angle", ".", "."], dtype=h5py.string_dtype() + ) + nx_data["data"] = data + nx_data["image_key"] = key + nx_data["rotation_angle"] = rotation_angle + + entry.create_dataset("end_time", data=datetime.now(timezone.utc).isoformat()) + + return output_file + + +# [write-nxtomo-end] + + +def _natural_sort_key(path: str) -> list[tuple[int, int | str]]: + """Sort numbered TIFF names in human order (for example, 2 before 10).""" + return [ + (0, int(part)) if part.isdigit() else (1, part.lower()) + for part in re.split(r"(\d+)", path) + ] + + +def load_tiff_stack(pattern: str) -> np.ndarray: + """Load the 2D TIFF files matched by a glob pattern into a 3D array.""" + paths = sorted(glob.glob(pattern), key=_natural_sort_key) + if not paths: + raise FileNotFoundError(f"No TIFF files match {pattern!r}") + + try: + import tifffile + except ImportError as error: + raise RuntimeError( + "Reading TIFF files requires tifffile: python -m pip install tifffile" + ) from error + + frames = [] + for path in paths: + frame = tifffile.imread(path) + if frame.ndim != 2: + raise ValueError(f"{path} is not a single 2D grayscale image") + frames.append(frame) + try: + return np.stack(frames) + except ValueError as error: + raise ValueError( + "All TIFF images in a stack must have the same shape" + ) from error + + +# [write-tiffs-start] +def write_nxtomo_from_tiffs( + output_file: str | Path, + projection_pattern: str, + angles: np.ndarray, + *, + flat_pattern: str | None = None, + dark_pattern: str | None = None, + **kwargs, +) -> Path: + """Load TIFF stacks and pass them to :func:`write_nxtomo`.""" + projections = load_tiff_stack(projection_pattern) + flats = load_tiff_stack(flat_pattern) if flat_pattern else None + darks = load_tiff_stack(dark_pattern) if dark_pattern else None + return write_nxtomo( + output_file, + projections, + angles, + flats=flats, + darks=darks, + **kwargs, + ) + + +# [write-tiffs-end] + + +def _load_angles(path: Path) -> np.ndarray: + """Load angles from a NumPy ``.npy`` file or a text/CSV file.""" + if path.suffix.lower() == ".npy": + return np.load(path) + delimiter = "," if path.suffix.lower() == ".csv" else None + return np.loadtxt(path, delimiter=delimiter) + + +def _parser() -> argparse.ArgumentParser: + parser = argparse.ArgumentParser( + description="Pack TIFF stacks into an HTTomo-compatible NXtomo file." + ) + parser.add_argument("output_file", type=Path) + parser.add_argument( + "--projections", + required=True, + metavar="GLOB", + help="Glob matching projection TIFF files (quote it in the shell)", + ) + parser.add_argument( + "--angles", + required=True, + type=Path, + help="One-dimensional angles file (.npy, .txt, or .csv), in degrees", + ) + parser.add_argument("--flats", metavar="GLOB", help="Glob matching flat TIFFs") + parser.add_argument("--darks", metavar="GLOB", help="Glob matching dark TIFFs") + parser.add_argument("--sample-name", default="sample") + parser.add_argument( + "--compression", choices=("none", "gzip", "lzf"), default="gzip" + ) + parser.add_argument("--overwrite", action="store_true") + return parser + + +def main(argv: Sequence[str] | None = None) -> None: + """Command-line entry point.""" + args = _parser().parse_args(argv) + output_file = write_nxtomo_from_tiffs( + args.output_file, + args.projections, + _load_angles(args.angles), + flat_pattern=args.flats, + dark_pattern=args.darks, + sample_name=args.sample_name, + compression=args.compression, + overwrite=args.overwrite, + ) + print(f"Wrote {output_file}") + + +if __name__ == "__main__": + main() diff --git a/docs/source/utilities/yaml_checker.rst b/docs/source/utilities/yaml_checker.rst deleted file mode 100644 index 97e73d972..000000000 --- a/docs/source/utilities/yaml_checker.rst +++ /dev/null @@ -1,140 +0,0 @@ -.. _utilities_yamlchecker: - -YAML Checker - Why use it? -************************** -YAML checker will help you to validate your process list (see :ref:`explanation_process_list`) -saved as a YAML file. Before running your pipeline with HTTomo, we highly recommend that you validate your process list using this utility. **The checker will help you to identify errors in your process list and avoid problems during the run**. - -Usage -===== - -.. code-block:: console - - $ python -m httomo check YAML_CONFIG IN_DATA - - -.. note:: - - - Use this :code:`check` command before you use the :code:`run` command to run your pipeline. - - The :code:`YAML_CONFIG` is the path to your YAML file and :code:`IN_DATA` is the path to your input data. - - :code:`IN_DATA` is optional, but if you provide it, the yaml checker will be checking that the paths - to the data and keys in the :code:`YAML_CONFIG` file match the paths and keys in the input file (:code:`IN_DATA`). - - -For example, if you have the following as a :code:`YAML_CONFIG` file saved as :code:`example.yaml`: - -.. literalinclude:: ../../../tests/samples/pipeline_template_examples/testing/example.yaml - :language: yaml - -And you run the YAML checker with: - -.. code-block:: console - - $ python -m httomo check example.yaml - - -You will get the following output: - -.. code-block:: console - - Checking that the YAML_CONFIG is properly indented and has valid mappings and tags... - Sanity check of the YAML_CONFIG was successfully done... - - Checking that the first method in the pipeline is a loader... - Loader check successful!! - - - YAML validation successful!! Please feel free to use the `run` command to run the pipeline. - - -The Yaml check was successful here because your yaml file was properly indented and had valid mappings and tags. -It also included valid parameters for each method used from TomoPy, HTTomolib, or other `backends `_. - - - -But if you had the following as a :code:`YAML_CONFIG` file saved as :code:`incorrect_method.yaml`: - -.. literalinclude:: ../../../tests/samples/pipeline_template_examples/testing/incorrect_method.yaml - :language: yaml - -And then you run the YAML checker, you get: - -.. code-block:: console - - $ python -m httomo check incorrect_method.yaml - Checking that the YAML_CONFIG is properly indented and has valid mappings and tags... - Sanity check of the YAML_CONFIG was successfully done... - - Checking that the first method in the pipeline is a loader... - Loader check successful!! - - 'tomopy.misc.corr/median_filters' is not a valid method. Please recheck the yaml file. - - -This is because :code:`median_filters` is not a valid method in TomoPy -- should be :code:`median_filter`. -To make sure you pass the correct method, refer to the documentation of the package you are using (TomoPy, HTTomoLib, etc.) - - -What else do we check with the YAML checker? -============================================ - -* We do a sanity check first, to make sure that the YAML_CONFIG is properly indented and has valid mappings. - -For instance, we cannot have the following in a YAML file: - -.. literalinclude:: ../../../tests/samples/pipeline_template_examples/testing/wrong_indentation_pipeline.yaml - :language: yaml - -This will raise a warning because :code:`data_path` is not at the same indentation level as the -other fields directly under the :code:`parameters` field. - -* We check that the first method in the pipeline is always a loader from :code:`'httomo.data.hdf.loaders'`. -* We check methods exist for the given module path. -* We check that the parameters for each method are valid. For example, :code:`find_center_vo` method from :code:`tomopy.recon.rotation` takes :code:`ratio` as a parameter with a float value. If you pass a string instead, it will raise an error. Again the trick is to refer the documentation always. -* We check the required parameters for each method are present. -* We check parameters that are omitted are not required (in particular, that omitted parameters - have a default value that can be assumed) -* If you pass :code:`IN_DATA` (path to the data) along with the yaml config, as: - -.. code-block:: console - - $ python -m httomo check config.yaml IN_DATA - -That will check that the paths to the data and keys in the :code:`YAML_CONFIG` file match the paths and keys in the input file (:code:`IN_DATA`). - -If you have the following loader in your yaml file: - -.. literalinclude:: ../../../tests/samples/pipeline_template_examples/testing/incorrect_path.yaml - :language: yaml - -And you provide that, together with the standard tomo data, it will raise an error because the image path does not match: - -.. code-block:: console - - Checking that the YAML_CONFIG is properly indented and has valid mappings and tags... - Sanity check of the YAML_CONFIG was successfully done... - - Checking that the first method in the pipeline is a loader... - Loader check successful!! - - Checking that the paths to the data and keys in the YAML_CONFIG file match the paths and keys in the input file (IN_DATA)... - 'entry1/tomo_entry/instrument/detect/image_key' is not a valid path to a dataset in YAML_CONFIG. Please recheck the yaml file. - -If the :code:`auto` value is used for any of the loader parameters that support it (see -:ref:`nxtomo_discovery` for more details), and an input data file is given, the YAML checker -will check if an :code:`NXtomo` entry can be found in the given input data. If no -:code:`NXtomo` entry can be found, the YAML check will fail with a message reporting this: - -.. code-block:: console - - Checking that the YAML_CONFIG is properly indented and has valid mappings and tags... - Sanity check of the YAML_CONFIG was successfully done... - - Checking that the first method in the pipeline is a loader... - Loader check successful!! - - Checking that the paths to the data and keys in the YAML_CONFIG file match the paths and keys in the input file (IN_DATA)... - No NXtomo entry detected in tests/test_data/i12/separate_flats_darks/i12_dynamic_start_stop180.nxs - -We have many other checks and are constantly improving the YAML checker to make it more robust, verbose, and user-friendly. -This is a user-interface so suggestions are always welcome. diff --git a/docs/sphinx-build.sh b/docs/sphinx-build.sh index c7b0361dd..db36ab6b3 100644 --- a/docs/sphinx-build.sh +++ b/docs/sphinx-build.sh @@ -12,16 +12,17 @@ DIR="$( cd "$( dirname "${BASH_SOURCE[0]}" )" && pwd )" # show-inheritance displays a list of base classes below the class signature # Remove directory for api so that there are no obsolete files -rm -rf $DIR/source/developers/generated/ -rm -rf $DIR/build/ +rm -rf "$DIR/source/developers/generated/" +rm -rf "$DIR/source/api/" +rm -rf "$DIR/build/" # sphinx-build [options] [filenames] # -a Write all output files. The default is to only write output files for new and changed source files. (This may not apply to all builders.) # -E Don’t use a saved environment (the structure caching all cross-references), but rebuild it completely. The default is to only read and parse source files that are new or have changed since the last run. # -b buildername, build pages of a certain file type -sphinx-build -a -E -b html $DIR/source/ $DIR/build/ +# -W treats warnings as errors; --keep-going reports all warnings in one run +sphinx-build -W --keep-going -a -E -b html "$DIR/source/" "$DIR/build/" echo "***********************************************************************" echo " End of script" echo "***********************************************************************" - diff --git a/docs/sphinx_build_setup.rst b/docs/sphinx_build_setup.rst index 348be201b..fd77b377f 100644 --- a/docs/sphinx_build_setup.rst +++ b/docs/sphinx_build_setup.rst @@ -1,39 +1,62 @@ -=================================== -How to build the html pages locally -=================================== +==================================== +How to build the HTML pages locally +==================================== -Create a conda environment +Create a Conda environment ========================== -Setup a new conda environment using the requirements file docs/source/doc-conda-requirements.yml -If you are using a diamond computer you can load python into your path to do this. +Create a documentation environment from the requirements file. On a Diamond +computer, first make Conda available by loading the Python module: - >>> module load python - >>> conda env create -f /path/to/HTTomo/docs/source/doc-conda-requirements.yml - >>> conda env create --prefix /path/to/env/doc-env --file /path/to/HTTomo/docs/source/doc-conda-requirements.yml +.. code-block:: console + $ module load python + $ conda env create --name httomo-docs \ + --file /path/to/HTTomo/docs/source/doc-conda-requirements.yml + $ conda activate httomo-docs -Update API documentation and build -================================== +Alternatively, create the environment at a specific path: -While inside your virtual environment, run the sphinx-build.sh script. +.. code-block:: console - >>> conda activate /path/to/env/doc-env - >>> source /path/to/HTTomo/docs/sphinx-build.sh + $ conda env create --prefix /path/to/env/httomo-docs \ + --file /path/to/HTTomo/docs/source/doc-conda-requirements.yml + $ conda activate /path/to/env/httomo-docs -The script will: +Build the documentation +======================= -1. Remove previous api and build directories. -2. Generate current HTTomo api files -3. Run a python script to add yaml file downloads for every function for every module rst file. -4. Run sphinx to create html documentation pages in docs/build. +Run the build script from anywhere after activating the environment: -You can view the completed pages by opening the HTTomo/docs/build/index.html page inside a browser. +.. code-block:: console -To conclude -=========== + $ bash /path/to/HTTomo/docs/sphinx-build.sh -When you have finished, you can deactivate the virtual environment and remove python from your path. +The script removes old generated API files and build output, then performs a +clean Sphinx HTML build. Sphinx regenerates the API summaries as part of the +build. Warnings are treated as errors, matching the documentation check in +continuous integration. - >>> conda deactivate - >>> module unload python +Open ``HTTomo/docs/build/index.html`` in a browser to view the result. + +Check external links +==================== + +The scheduled continuous-integration job also checks external links. Run the +same check locally when adding or changing links: + +.. code-block:: console + + $ sphinx-build -W --keep-going -a -E -b linkcheck \ + /path/to/HTTomo/docs/source /path/to/HTTomo/docs/linkcheck + +Finish +====== + +Deactivate the environment when finished. On a Diamond computer, unload the +Python module as well: + +.. code-block:: console + + $ conda deactivate + $ module unload python diff --git a/httomo/cli.py b/httomo/cli.py index 4f0fad9ce..0a291a1e5 100644 --- a/httomo/cli.py +++ b/httomo/cli.py @@ -172,7 +172,11 @@ def check(pipeline: Union[Path, str], in_data_file: Optional[Path] = None): "--max-memory", type=click.STRING, default="0", - help="The maximum amount of the CPU memory per process available on the system (supports strings like 3.2G or bytes)", + help=( + "Per-process memory ceiling used to select in-memory or disk-backed " + "storage and to cap GPU block sizing (supports values such as 3.2G or " + "bytes; 0 disables the user ceiling)" + ), ) @click.option( "--save-snapshots", diff --git a/tests/test_pipeline_small.py b/tests/test_pipeline_small.py index ef7d2a8e8..cc120fa60 100644 --- a/tests/test_pipeline_small.py +++ b/tests/test_pipeline_small.py @@ -58,6 +58,7 @@ def test_run_pipeline_tomopy_gridrec( @pytest.mark.small_data +@pytest.mark.cupy def test_run_pipeline_FBP3d_tomobar( get_files: Callable, cmd, standard_data, FBP3d_tomobar, output_folder ): @@ -105,6 +106,7 @@ def test_run_pipeline_FBP3d_tomobar( @pytest.mark.small_data +@pytest.mark.cupy def test_run_pipeline_FBP3d_tomobar_denoising( get_files: Callable, cmd,