mirror of
https://github.com/c-sooyoung/bo-ptycho.git
synced 2026-09-17 20:29:07 +09:00
multi-gpu batched bo
This commit is contained in:
+14
-13
@@ -1,11 +1,11 @@
|
|||||||
job:
|
job:
|
||||||
type: "test"
|
type: "random+sobo"
|
||||||
random_iters: 5
|
random_iters: 8
|
||||||
sobo_iters: 10
|
sobo_iters: 128
|
||||||
|
|
||||||
io:
|
io:
|
||||||
input_data_path: '/home/swim/shared/Si_project/data/Si2V1_1.mat'
|
input_data_path: '/home/swim/shared/Si_project/data/Si2V1_2.mat'
|
||||||
result_dir: '/home/swim/bo-ptycho/results/test'
|
result_dir: '/home/swim/bo-ptycho/results/260812-Si/Si2V1_2'
|
||||||
verbosity: 1
|
verbosity: 1
|
||||||
|
|
||||||
ptycho:
|
ptycho:
|
||||||
@@ -16,7 +16,7 @@ ptycho:
|
|||||||
alpha_max: 30
|
alpha_max: 30
|
||||||
defocus: -200
|
defocus: -200
|
||||||
rot_ang: 0.3
|
rot_ang: 0.3
|
||||||
Nlayers: 5
|
Nlayers: 20
|
||||||
thickness: 250
|
thickness: 250
|
||||||
rbf: 37
|
rbf: 37
|
||||||
Nprobe: 1
|
Nprobe: 1
|
||||||
@@ -25,8 +25,10 @@ ptycho:
|
|||||||
tilt_x: 2
|
tilt_x: 2
|
||||||
tilt_y: 0
|
tilt_y: 0
|
||||||
scan_step_size: 0.36
|
scan_step_size: 0.36
|
||||||
Niter: 5
|
|
||||||
Niter_save_results: 5
|
Niter: 100
|
||||||
|
Niter_save_results: 100
|
||||||
|
|
||||||
CBED_size: 192
|
CBED_size: 192
|
||||||
ADU: 1
|
ADU: 1
|
||||||
extra_print_info: ''
|
extra_print_info: ''
|
||||||
@@ -36,11 +38,10 @@ ptycho:
|
|||||||
diff_pattern_blur: 1
|
diff_pattern_blur: 1
|
||||||
probe_change_start: 1
|
probe_change_start: 1
|
||||||
object_change_start: 1
|
object_change_start: 1
|
||||||
grouping: inf
|
grouping: 512
|
||||||
probe_position_search: 1
|
probe_position_search: 1
|
||||||
regularize_layers: 0.2
|
regularize_layers: 0.2
|
||||||
variable_probe: 'false'
|
variable_probe: 'false'
|
||||||
verbosity
|
|
||||||
|
|
||||||
bo:
|
bo:
|
||||||
batch: 4
|
batch: 4
|
||||||
@@ -50,12 +51,12 @@ bo:
|
|||||||
params:
|
params:
|
||||||
alpha_max:
|
alpha_max:
|
||||||
defocus:
|
defocus:
|
||||||
# radius: 100
|
radius: 100
|
||||||
rot_ang:
|
rot_ang:
|
||||||
Nlayers:
|
Nlayers:
|
||||||
radius: 2
|
radius: 5
|
||||||
type: int
|
type: int
|
||||||
thickness:
|
thickness:
|
||||||
# radius: 100
|
radius: 100
|
||||||
train_x:
|
train_x:
|
||||||
train_y:
|
train_y:
|
||||||
|
|||||||
@@ -3,7 +3,7 @@
|
|||||||
#SBATCH --nodes=1
|
#SBATCH --nodes=1
|
||||||
#SBATCH --ntasks=1
|
#SBATCH --ntasks=1
|
||||||
#SBATCH --cpus-per-task=1
|
#SBATCH --cpus-per-task=1
|
||||||
#SBATCH --gres=gpu:rtx-6000ada:1
|
#SBATCH --gres=gpu:rtx-6000ada:4
|
||||||
#SBATCH --time=100:00:00
|
#SBATCH --time=100:00:00
|
||||||
#SBATCH --output=/home/swim/slurm-logs/job_%j.log
|
#SBATCH --output=/home/swim/slurm-logs/job_%j.log
|
||||||
|
|
||||||
|
|||||||
@@ -1,5 +1,6 @@
|
|||||||
from .sobo import sobo_pipeline
|
# from .sobo import sobo_pipeline # depreacated
|
||||||
from .test import test_pipeline
|
from .test import test_pipeline
|
||||||
|
from .batched_sobo import sobo_pipeline
|
||||||
|
|
||||||
job_types = {
|
job_types = {
|
||||||
'random+sobo': sobo_pipeline,
|
'random+sobo': sobo_pipeline,
|
||||||
|
|||||||
@@ -0,0 +1,190 @@
|
|||||||
|
import os
|
||||||
|
import traceback
|
||||||
|
import multiprocessing as mp
|
||||||
|
import shutil
|
||||||
|
|
||||||
|
import bo
|
||||||
|
import ptycho
|
||||||
|
|
||||||
|
def run_ptycho_worker(
|
||||||
|
worker_id,
|
||||||
|
gpu_token,
|
||||||
|
job_config,
|
||||||
|
metric,
|
||||||
|
run_id,
|
||||||
|
result_queue,
|
||||||
|
):
|
||||||
|
# This process, and MATLAB launched from it,
|
||||||
|
# can see exactly one GPU.
|
||||||
|
os.environ["CUDA_VISIBLE_DEVICES"] = gpu_token
|
||||||
|
|
||||||
|
try:
|
||||||
|
PTYCHOENGINE = ptycho.engines[job_config["ptycho"]["engine"]]
|
||||||
|
|
||||||
|
ptycho_engine = PTYCHOENGINE(job_config)
|
||||||
|
ptycho_engine.run(run_id=run_id)
|
||||||
|
y_value = ptycho_engine.metric(metric)
|
||||||
|
|
||||||
|
result_queue.put(
|
||||||
|
(worker_id, y_value, None)
|
||||||
|
)
|
||||||
|
|
||||||
|
except Exception:
|
||||||
|
result_queue.put(
|
||||||
|
(worker_id, None, traceback.format_exc())
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def run_batch(
|
||||||
|
ctx,
|
||||||
|
gpu_tokens,
|
||||||
|
job_configs,
|
||||||
|
metric,
|
||||||
|
iteration,
|
||||||
|
):
|
||||||
|
result_queue = ctx.Queue()
|
||||||
|
processes = []
|
||||||
|
|
||||||
|
for i, job_config in enumerate(job_configs):
|
||||||
|
# Important: unique run_id for simultaneous jobs.
|
||||||
|
run_id = f"bo-{iteration:03d}-{i:02d}"
|
||||||
|
|
||||||
|
p = ctx.Process(
|
||||||
|
target=run_ptycho_worker,
|
||||||
|
args=(
|
||||||
|
i,
|
||||||
|
gpu_tokens[i],
|
||||||
|
job_config,
|
||||||
|
metric,
|
||||||
|
run_id,
|
||||||
|
result_queue,
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
p.start()
|
||||||
|
processes.append(p)
|
||||||
|
|
||||||
|
# Four jobs are now running concurrently.
|
||||||
|
|
||||||
|
results = [
|
||||||
|
result_queue.get()
|
||||||
|
for _ in processes
|
||||||
|
]
|
||||||
|
|
||||||
|
# Synchronization barrier.
|
||||||
|
for p in processes:
|
||||||
|
p.join()
|
||||||
|
|
||||||
|
# Completion order is arbitrary.
|
||||||
|
results.sort(key=lambda x: x[0])
|
||||||
|
|
||||||
|
for worker_id, _, error in results:
|
||||||
|
if error is not None:
|
||||||
|
raise RuntimeError(
|
||||||
|
f"Ptycho worker {worker_id} failed:\n{error}"
|
||||||
|
)
|
||||||
|
|
||||||
|
return [y_value for _, y_value, _ in results]
|
||||||
|
|
||||||
|
|
||||||
|
def sobo_pipeline(config):
|
||||||
|
|
||||||
|
result_dir = config["io"]["result_dir"]
|
||||||
|
if os.path.exists(result_dir):
|
||||||
|
shutil.rmtree(result_dir)
|
||||||
|
os.makedirs(result_dir, exist_ok=True)
|
||||||
|
|
||||||
|
RANDOM_ITERS = config["job"].get("random_iters", 0)
|
||||||
|
SOBO_ITERS = config["job"].get("sobo_iters", 0)
|
||||||
|
METRIC = config["bo"]["metric"]
|
||||||
|
BO_BATCH = config["bo"]["batch"]
|
||||||
|
|
||||||
|
# SLURM should expose the four GPUs allocated to this job.
|
||||||
|
visible = os.environ.get("CUDA_VISIBLE_DEVICES")
|
||||||
|
|
||||||
|
if visible is None:
|
||||||
|
raise RuntimeError(
|
||||||
|
"CUDA_VISIBLE_DEVICES is not set"
|
||||||
|
)
|
||||||
|
|
||||||
|
gpu_tokens = [
|
||||||
|
token.strip()
|
||||||
|
for token in visible.split(",")
|
||||||
|
if token.strip()
|
||||||
|
]
|
||||||
|
|
||||||
|
if len(gpu_tokens) < BO_BATCH:
|
||||||
|
raise RuntimeError(
|
||||||
|
f"Expected {BO_BATCH} allocated GPUs, got {len(gpu_tokens)}"
|
||||||
|
)
|
||||||
|
|
||||||
|
# Explicitly use spawn for CUDA / MATLAB isolation.
|
||||||
|
ctx = mp.get_context("spawn")
|
||||||
|
|
||||||
|
# ---------------------------------------------------------
|
||||||
|
# Random sampling
|
||||||
|
# ---------------------------------------------------------
|
||||||
|
|
||||||
|
randombo = bo.RandomBOEngine(config)
|
||||||
|
|
||||||
|
for j in range(RANDOM_ITERS):
|
||||||
|
print(
|
||||||
|
f"RANDOM sampling; iteration {j}",
|
||||||
|
flush=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
job_configs = randombo.ask(n=BO_BATCH)
|
||||||
|
|
||||||
|
y_values = run_batch(
|
||||||
|
ctx=ctx,
|
||||||
|
gpu_tokens=gpu_tokens,
|
||||||
|
job_configs=job_configs,
|
||||||
|
metric=METRIC,
|
||||||
|
iteration=j,
|
||||||
|
)
|
||||||
|
|
||||||
|
# Only the parent touches BO state / train_x / train_y.
|
||||||
|
for job_config, y_value in zip(
|
||||||
|
job_configs,
|
||||||
|
y_values,
|
||||||
|
):
|
||||||
|
randombo.tell(
|
||||||
|
job_config,
|
||||||
|
y_value,
|
||||||
|
)
|
||||||
|
|
||||||
|
# ---------------------------------------------------------
|
||||||
|
# SOBO
|
||||||
|
# ---------------------------------------------------------
|
||||||
|
|
||||||
|
sobo = bo.SingleObjectiveBOEngine(config)
|
||||||
|
|
||||||
|
sobo.train_x = randombo.train_x
|
||||||
|
sobo.train_y = randombo.train_y
|
||||||
|
|
||||||
|
for j in range(SOBO_ITERS):
|
||||||
|
iteration = RANDOM_ITERS + j
|
||||||
|
|
||||||
|
print(
|
||||||
|
f"SOBO sampling; iteration {iteration}",
|
||||||
|
flush=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
job_configs = sobo.ask(n=BO_BATCH)
|
||||||
|
|
||||||
|
y_values = run_batch(
|
||||||
|
ctx=ctx,
|
||||||
|
gpu_tokens=gpu_tokens,
|
||||||
|
job_configs=job_configs,
|
||||||
|
metric=METRIC,
|
||||||
|
iteration=iteration,
|
||||||
|
)
|
||||||
|
|
||||||
|
for job_config, y_value in zip(
|
||||||
|
job_configs,
|
||||||
|
y_values,
|
||||||
|
):
|
||||||
|
sobo.tell(
|
||||||
|
job_config,
|
||||||
|
y_value,
|
||||||
|
)
|
||||||
+39
-37
@@ -1,47 +1,49 @@
|
|||||||
import os
|
# depreacated, use pipelines.batched_sobo.sobo_pipeline()
|
||||||
import bo
|
|
||||||
import ptycho
|
|
||||||
|
|
||||||
def sobo_pipeline(config):
|
# import os
|
||||||
|
# import bo
|
||||||
|
# import ptycho
|
||||||
|
|
||||||
result_dir = config['io']['result_dir']
|
# def sobo_pipeline(config):
|
||||||
os.makedirs(result_dir, exist_ok=True)
|
|
||||||
|
|
||||||
RANDOM_ITERS = config['job'].get('random_iters', 0)
|
# result_dir = config['io']['result_dir']
|
||||||
SOBO_ITERS = config['job'].get('sobo_iters')
|
# os.makedirs(result_dir, exist_ok=True)
|
||||||
METRIC = config['bo']['metric']
|
|
||||||
PTYCHOENGINE = ptycho.engines[config['ptycho']['engine']]
|
# RANDOM_ITERS = config['job'].get('random_iters', 0)
|
||||||
|
# SOBO_ITERS = config['job'].get('sobo_iters')
|
||||||
|
# METRIC = config['bo']['metric']
|
||||||
|
# PTYCHOENGINE = ptycho.engines[config['ptycho']['engine']]
|
||||||
|
|
||||||
|
|
||||||
randombo = bo.RandomBOEngine(config)
|
# randombo = bo.RandomBOEngine(config)
|
||||||
|
|
||||||
bo_txt = os.path.join(result_dir, "bo.txt")
|
# bo_txt = os.path.join(result_dir, "bo.txt")
|
||||||
with open(bo_txt, "w") as f:
|
# with open(bo_txt, "w") as f:
|
||||||
f.write(f" iter\tmetric\t{"\t".join([p[:7] for p in randombo.params])}\n")
|
# f.write(f" iter\tmetric\t{"\t".join([p[:7] for p in randombo.params])}\n")
|
||||||
|
|
||||||
for j in range(RANDOM_ITERS):
|
# for j in range(RANDOM_ITERS):
|
||||||
print(f"RANDOM sampling; iteration {j}")
|
# print(f"RANDOM sampling; iteration {j}")
|
||||||
job_config = randombo.ask()
|
# job_config = randombo.ask()
|
||||||
ptycho_engine = PTYCHOENGINE(job_config)
|
# ptycho_engine = PTYCHOENGINE(job_config)
|
||||||
ptycho_engine.run(run_id=f"bo-{j:03d}")
|
# ptycho_engine.run(run_id=f"bo-{j:03d}")
|
||||||
y_value = ptycho_engine.metric(METRIC)
|
# y_value = ptycho_engine.metric(METRIC)
|
||||||
randombo.tell(job_config, y_value)
|
# randombo.tell(job_config, y_value)
|
||||||
with open(bo_txt, "a") as f:
|
# with open(bo_txt, "a") as f:
|
||||||
p = [f'{job_config['ptycho']['params'][key]:.2f}' for key in randombo.params]
|
# p = [f'{job_config['ptycho']['params'][key]:.2f}' for key in randombo.params]
|
||||||
f.write(f"{j: 8d}\t{y_value:.4f}\t{"\t".join(p)}\n")
|
# f.write(f"{j: 8d}\t{y_value:.4f}\t{"\t".join(p)}\n")
|
||||||
|
|
||||||
|
|
||||||
sobo = bo.SingleObjectiveBOEngine(config)
|
# sobo = bo.SingleObjectiveBOEngine(config)
|
||||||
sobo.train_x = randombo.train_x
|
# sobo.train_x = randombo.train_x
|
||||||
sobo.train_y = randombo.train_y
|
# sobo.train_y = randombo.train_y
|
||||||
|
|
||||||
for j in range(SOBO_ITERS):
|
# for j in range(SOBO_ITERS):
|
||||||
print(f"SOBO sampling; iteration {RANDOM_ITERS + j}")
|
# print(f"SOBO sampling; iteration {RANDOM_ITERS + j}")
|
||||||
job_config = sobo.ask()
|
# job_config = sobo.ask()
|
||||||
ptycho_engine = PTYCHOENGINE(job_config)
|
# ptycho_engine = PTYCHOENGINE(job_config)
|
||||||
ptycho_engine.run(run_id=f"bo-{RANDOM_ITERS + j:03d}")
|
# ptycho_engine.run(run_id=f"bo-{RANDOM_ITERS + j:03d}")
|
||||||
y_value = ptycho_engine.metric(METRIC)
|
# y_value = ptycho_engine.metric(METRIC)
|
||||||
sobo.tell(job_config, y_value)
|
# sobo.tell(job_config, y_value)
|
||||||
with open(bo_txt, "a") as f:
|
# with open(bo_txt, "a") as f:
|
||||||
p = [f'{job_config['ptycho']['params'][key]:.2f}' for key in sobo.params]
|
# p = [f'{job_config['ptycho']['params'][key]:.2f}' for key in sobo.params]
|
||||||
f.write(f"{RANDOM_ITERS + j: 8d}\t{y_value:.4f}\t{"\t".join(p)}\n")
|
# f.write(f"{RANDOM_ITERS + j: 8d}\t{y_value:.4f}\t{"\t".join(p)}\n")
|
||||||
|
|||||||
@@ -11,9 +11,7 @@ class FoldSlicePtychoEngine(PtychoEngine):
|
|||||||
|
|
||||||
def __init__(self, config):
|
def __init__(self, config):
|
||||||
super().__init__(config)
|
super().__init__(config)
|
||||||
self._output_dir = os.path.join(config['io']['result_dir'], 'fold_slice')
|
|
||||||
self._fold_slice_path = self.config['ptycho']['path']
|
self._fold_slice_path = self.config['ptycho']['path']
|
||||||
self._setup_txt_path = os.path.join(self._output_dir, 'setup.txt')
|
|
||||||
self.metric_methods = {
|
self.metric_methods = {
|
||||||
'log_fourier': self._log_fourier_metric,
|
'log_fourier': self._log_fourier_metric,
|
||||||
}
|
}
|
||||||
@@ -21,6 +19,9 @@ class FoldSlicePtychoEngine(PtychoEngine):
|
|||||||
|
|
||||||
def run(self, run_id="") -> None:
|
def run(self, run_id="") -> None:
|
||||||
|
|
||||||
|
self._output_dir = os.path.join(self.config['io']['result_dir'], f'fold_slice-{run_id}')
|
||||||
|
self._setup_txt_path = os.path.join(self._output_dir, 'setup.txt')
|
||||||
|
|
||||||
# generate setup.txt for fold_slice input
|
# generate setup.txt for fold_slice input
|
||||||
fold_slice_dict = {}
|
fold_slice_dict = {}
|
||||||
fold_slice_dict['raw_data'] = self.config['io']['input_data_path']
|
fold_slice_dict['raw_data'] = self.config['io']['input_data_path']
|
||||||
@@ -82,9 +83,11 @@ class FoldSlicePtychoEngine(PtychoEngine):
|
|||||||
|
|
||||||
log_fourier_error = self._log_fourier_metric()
|
log_fourier_error = self._log_fourier_metric()
|
||||||
# os.makedirs(os.path.join(self.config['io']['result_dir'], "mat"), exist_ok=True)
|
# os.makedirs(os.path.join(self.config['io']['result_dir'], "mat"), exist_ok=True)
|
||||||
os.makedirs(os.path.join(self.config['io']['result_dir'], "tiff"), exist_ok=True)
|
# os.makedirs(os.path.join(self.config['io']['result_dir'], "tiff"), exist_ok=True)
|
||||||
# shutil.copy(mat_path, os.path.join(self.config['io']['result_dir'], "mat", f"{log_fourier_error:.4f}_{run_id}.mat")) # saving .mat files takes a lot of space (expect 20+ GB for 64*64 scan size, 300 iterations)
|
# shutil.copy(mat_path, os.path.join(self.config['io']['result_dir'], "mat", f"{log_fourier_error:.4f}_{run_id}.mat")) # saving .mat files takes a lot of space (expect 20+ GB for 64*64 scan size, 300 iterations)
|
||||||
shutil.copy(image_path, os.path.join(self.config['io']['result_dir'], "tiff", f"{log_fourier_error:.4f}_{run_id}.tiff"))
|
# shutil.copy(image_path, os.path.join(self.config['io']['result_dir'], "tiff", f"{log_fourier_error:.4f}_{run_id}.tiff"))
|
||||||
|
|
||||||
|
shutil.rmtree(self._output_dir)
|
||||||
|
|
||||||
|
|
||||||
def metric(self, names):
|
def metric(self, names):
|
||||||
|
|||||||
Reference in New Issue
Block a user