mirror of
https://github.com/c-sooyoung/fold_slice.git
synced 2026-09-17 22:49:08 +09:00
103 lines
4.9 KiB
Matlab
103 lines
4.9 KiB
Matlab
% ASTRA_FIND_OPTIMAL_SPLIT Find optimal split of data and make the blocks sufficiently small for
|
||
% limited GPU memory
|
||
%
|
||
% split = ASTRA_find_optimal_split(cfg, num_gpu, angle_blocks, propagator)
|
||
%
|
||
% Inputs:
|
||
% **cfg - config structure generated by ASTRA_initialize
|
||
% **num_gpu - number of gpu to split the data
|
||
% **angle_blocks - number of angular blocks (ie in SART method or FSC)
|
||
% **propagator - which propagator should be assumed : FWD, BACK, both (default)
|
||
% Outputs:
|
||
% ++split - volume / angle split - [split_x,split_y,split_z,split_angles]
|
||
|
||
|
||
%*-----------------------------------------------------------------------*
|
||
%| |
|
||
%| Except where otherwise noted, this work is licensed under a |
|
||
%| Creative Commons Attribution-NonCommercial-ShareAlike 4.0 |
|
||
%| International (CC BY-NC-SA 4.0) license. |
|
||
%| |
|
||
%| Copyright (c) 2017 by Paul Scherrer Institute (http://www.psi.ch) |
|
||
%| |
|
||
%| Author: CXS group, PSI |
|
||
%*-----------------------------------------------------------------------*
|
||
% You may use this code with the following provisions:
|
||
%
|
||
% If the code is fully or partially redistributed, or rewritten in another
|
||
% computing language this notice should be included in the redistribution.
|
||
%
|
||
% If this code, or subfunctions or parts of it, is used for research in a
|
||
% publication or if it is fully or partially rewritten for another
|
||
% computing language the authors and institution should be acknowledged
|
||
% in written form in the publication: “Data processing was carried out
|
||
% using the “cSAXS matlab package” developed by the CXS group,
|
||
% Paul Scherrer Institut, Switzerland.”
|
||
% Variations on the latter text can be incorporated upon discussion with
|
||
% the CXS group if needed to more specifically reflect the use of the package
|
||
% for the published work.
|
||
%
|
||
% A publication that focuses on describing features, or parameters, that
|
||
% are already existing in the code should be first discussed with the
|
||
% authors.
|
||
%
|
||
% This code and subroutines are part of a continuous development, they
|
||
% are provided “as they are” without guarantees or liability on part
|
||
% of PSI or the authors. It is the user responsibility to ensure its
|
||
% proper use and the correctness of the results.
|
||
|
||
|
||
|
||
function split = ASTRA_find_optimal_split(cfg, num_gpu, angle_blocks, propagator)
|
||
|
||
if gpuDeviceCount == 0
|
||
split = 1;
|
||
return
|
||
end
|
||
|
||
gpu = gpuDevice;
|
||
if nargin < 2 || num_gpu == 0
|
||
num_gpu = 1;
|
||
end
|
||
if nargin < 3
|
||
angle_blocks = 1;
|
||
end
|
||
if nargin < 4
|
||
propagator = 'both';
|
||
end
|
||
|
||
Nangles = cfg.iProjAngles / angle_blocks;
|
||
if isfield(cfg, 'Grouping')
|
||
Nangles = min(Nangles, cfg.Grouping);
|
||
end
|
||
|
||
split = [1,1,1];
|
||
if ismember(lower(propagator), {'fwd','both'})
|
||
split = max(split, ceil([cfg.iVolX, cfg.iVolY, cfg.iVolZ] / 4096 )); % texture memory limit
|
||
split = max(split, [1,1,ceil( (cfg.iVolX*cfg.iVolY*cfg.iVolZ*4) / 1.024e9 / prod(split)/num_gpu)]); % texture memory limit
|
||
end
|
||
split = max(split, [1,1,split(3)*ceil( ((cfg.iVolX*cfg.iVolY*cfg.iVolZ*4)/ prod(split)/num_gpu) / (gpu.AvailableMemory/2 - min(1.1e9, cfg.iVolX*cfg.iVolY*cfg.iVolZ*4)) )]); % gpu memory limit
|
||
split = max(split, [1,1,split(3)*ceil( ((cfg.iVolX*cfg.iVolY*cfg.iVolZ )/ prod(split)/num_gpu) / double(intmax('int32')) )]); % maximal array on GPU limit
|
||
|
||
if ismember(lower(propagator), {'back','both'})
|
||
% if projection would be larger than 4096x4096 -> split the reconstruction volume
|
||
split = max(split, [cfg.iProjU, cfg.iProjU, cfg.iProjV]/4096 ); % texture memory limit
|
||
end
|
||
|
||
% projection size limitation + astra allows only < 1024 angles
|
||
split(4) = max(ceil(Nangles/1024), ceil( (cfg.iProjU*cfg.iProjV*min(1024,Nangles)*4) / gpu.TotalMemory/num_gpu)); % gpu memory limit
|
||
|
||
% RAM limits
|
||
if cfg.iVolX*cfg.iVolY*cfg.iVolZ > 2e6
|
||
freemem = 0.8 * utils.check_available_memory;
|
||
split(4) = max(split(4), num_gpu*(4*(cfg.iProjU*cfg.iProjV*cfg.iProjAngles...
|
||
+ cfg.iVolX*cfg.iVolY*cfg.iVolZ*(1+1/prod(split(1:3)))))...
|
||
/ (freemem * 1e6) ); % RAM memory limit
|
||
end
|
||
split(4) = ceil(split(4));
|
||
split(1:3) = 2.^nextpow2(split(1:3));
|
||
% if any(split ~= 1)
|
||
% fprintf('Automatically splitting to %ix%ix%ix(%i) cubes \n', split);
|
||
% end
|
||
|
||
end |