# Makefile for FINUFFT (CPU code and its various interfaces, plus the GPU
# library via the 'cufinufft' target)

# For simplicity, this is the only makefile; there are no makefiles in
# subdirectories. This makefile is also useful to show humans how to compile
# FINUFFT and its various language interfaces and examples.
# Users should not need to edit this makefile (doing so would make it hard to
# stay up to date with the repo version). Rather, in order to change
# OS/platform-specific compilers and flags, create the file make.inc, which
# overrides the defaults below (which are for an ubuntu linux/GCC system).
# Read docs/install.rst, and make-platforms/make.inc.* for examples.

# Barnett 2017-2020. Malleo's expansion for guru interface, summer 2019.
# Barnett tidying Feb, May 2020. Libin Lu edits, 2020.
# Garrett Wright, Joakim Anden, Barnett: dual-prec lib build, Jun-Jul'20.
# Windows compatibility, jonas-kr, Sep '20.
# XSIMD dependency, Marco Barbone, June 2024.
# DUCC optional dependency to replace FFTW3. Barnett/Lu, 8/6/24.

# Compiler (CXX), and linking from C, fortran. We use GCC by default...
CXX = g++
CC = gcc
FC = gfortran
CLINK = -lstdc++
FLINK = $(CLINK)
PYTHON = python3
# baseline compile flags for GCC (no multithreading):
# Notes: 1) -Ofast breaks isfinite() & isnan(), so use -O3 which now is as fast
#        2) -fcx-limited-range for fortran-speed complex arith in C++
#        3) we use simply-expanded (:=) makefile variables, otherwise confusing
#        4) the extra math flags are for speed, but they do not impact accuracy;
#           they allow gcc to vectorize the code more effectively
CFLAGS := -O3 -funroll-loops -march=native -fcx-limited-range -ffp-contract=fast\
		  -fno-math-errno -fno-signed-zeros -fno-trapping-math -fassociative-math\
		  -freciprocal-math -fmerge-all-constants -ftree-vectorize $(CFLAGS) -Wfatal-errors
# do not allow semantic interposition. We do not want the users to override any of the internals
# it's default in clang but GCC is ELF compliant by default
CFLAGS += -fno-semantic-interposition
FFLAGS := $(CFLAGS) $(FFLAGS)
CXXFLAGS := $(CFLAGS) $(CXXFLAGS)

# fix for exported symbols in test
OBJFLAGS := -DFINUFFT_BUILD_TESTS -fvisibility=hidden -DFINUFFT_DLL -Ddll_EXPORTS

# FFTW base name, and math linking...
FFTWNAME = fftw3
# linux default is fftw3_omp, since 10% faster than fftw3_threads...
FFTWOMPSUFFIX = omp
LIBS := -lm
# multithreading for GCC: C++/C/Fortran, MATLAB, and octave (ICC differs)...
OMPFLAGS = -fopenmp
OMPLIBS = -lgomp
# we bundle any libs mex needs here with flags...
MOMPFLAGS = -D_OPENMP $(OMPLIBS)
OOMPFLAGS =
# MATLAB MEX compilation (also see below +=)...
MFLAGS := -DR2008OO -largeArrayDims
# location of MATLAB's mex compiler (could add flags to switch GCC, etc)...
MEX = mex
# octave, and its mkoctfile and flags (also see below +=)...
OCTAVE = octave
MKOCTFILE = mkoctfile
OFLAGS = -DR2008OO
# For experts only, location of MWrap executable (see docs/install.rst):
MWRAP = mwrap

# root directory for dependencies to be downloaded:
DEPS_ROOT := deps

# xsimd header-only dependency repo (VERSION can be a tag or commit)
XSIMD_URL := https://github.com/xtensor-stack/xsimd.git
XSIMD_VERSION := 14.3.0
XSIMD_DIR := $(DEPS_ROOT)/xsimd

# POET dispatcher dependency: each release ships one amalgamated header (with
# the generated poet/version.hpp inlined), so VERSION must be a release tag
POET_VERSION := v0.0.1
POET_DIR := $(DEPS_ROOT)/poet
POET_URL := https://github.com/flatironinstitute/poet/releases/download/$(POET_VERSION)/poet.hpp

# DUCC sources optional dependency repo
DUCC_URL := https://github.com/mreineck/ducc.git
DUCC_VERSION := ducc0_0_39_1
DUCC_DIR := $(DEPS_ROOT)/ducc
# this dummy file used as empty target by make...
DUCC_COOKIE := $(DUCC_DIR)/.finufft_has_ducc
# for internal DUCC compile...
DUCC_INCL := -I$(DUCC_DIR)/src
DUCC_SRC := $(DUCC_DIR)/src/ducc0
# for DUCC objects compile only (not our objects)...
DUCC_CXXFLAGS := -fPIC -std=c++17 -ffast-math $(CXXFLAGS)

# absolute path of this makefile, ie FINUFFT's top-level directory...
FINUFFT = $(dir $(realpath $(firstword $(MAKEFILE_LIST))))

# For your OS, override the above by setting make variables in make.inc ...
-include make.inc

# Now come flags that should be added, whatever user overrode in make.inc.
# -fPIC (position-indep code) needed to build dyn lib (.so)
# Also, we force return (via :=) to the land of simply-expanded variables...
INCL = -Iinclude -I$(XSIMD_DIR)/include -I$(POET_DIR)/include
# single-thread total list of math and FFT libs (now both precisions)...
# (Note: finufft tests use LIBSFFT; spread & util tests only need LIBS)
LIBSFFT := $(LIBS)
ifeq ($(FFT),DUCC)
  DUCC_SETUP := $(DUCC_COOKIE)
# so FINUFFT build can see DUCC headers...
  INCL += $(DUCC_INCL)
  DUCC_OBJS := $(DUCC_SRC)/infra/string_utils.o $(DUCC_SRC)/infra/threading.o $(DUCC_SRC)/infra/mav.o
  DUCC_SRCS := $(DUCC_OBJS:.o=.cc)
# FINUFFT's switchable FFT done via this compile directive...
  CXXFLAGS += -DFINUFFT_USE_DUCC0
# tell Python skbuild also to use DUCC (note wrapping in quotes)...
  PY_CMAKE_ARGS := "-DFINUFFT_USE_DUCC0=ON"
else
# link against FFTW3 single-threaded (leaves DUCC_OBJS and DUCC_SETUP undef)
  LIBSFFT += -l$(FFTWNAME) -l$(FFTWNAME)f
endif
CXXFLAGS := $(CXXFLAGS) $(INCL) -fPIC -std=c++17
CFLAGS := $(CFLAGS) $(INCL) -fPIC
# here /usr/include needed for fftw3.f "fortran header"... (JiriK: no longer)
FFLAGS := $(FFLAGS) $(INCL) -I/usr/include -fPIC
# Link time optimization (LTO):
# (works with GCC, Clang. Increases link time, reduces binary size, can speed up hot paths)
ifneq ($(LTO),OFF)
  LTOFLAGS := -flto=auto
  CFLAGS   += $(LTOFLAGS)
  CXXFLAGS += $(LTOFLAGS)
  FFLAGS   += $(LTOFLAGS)
  LDFLAGS  += $(LTOFLAGS)
endif

# multi-threaded libs & flags, and req'd flags (OO for new interface)...
ifneq ($(OMP),OFF)
  CXXFLAGS += $(OMPFLAGS)
  CFLAGS += $(OMPFLAGS)
  FFLAGS += $(OMPFLAGS)
  MFLAGS += $(MOMPFLAGS)
  OFLAGS += $(OOMPFLAGS)
  LIBS += $(OMPLIBS)
  LIBSFFT += $(OMPLIBS)
# fftw3 multithreaded libs...
  ifneq ($(FFT),DUCC)
    LIBSFFT += -l$(FFTWNAME)_$(FFTWOMPSUFFIX) -l$(FFTWNAME)f_$(FFTWOMPSUFFIX)
  endif
endif

# name & location of shared library we're building...
LIBNAME = libfinufft
ifeq ($(MINGW),ON)
  DYNLIB = lib/$(LIBNAME).dll
else
  DYNLIB = lib/$(LIBNAME).so
endif

STATICLIB = lib-static/$(LIBNAME).a
# absolute path to the .so, useful for linking so executables portable...
ABSDYNLIB = $(FINUFFT)$(DYNLIB)

# spreader objs
SOBJS = src/utils.o src/common/utils.o src/common/kernel.o src/common/pswf.o

# per-precision objs (each gets a _f.o single-precision variant via pattern rule)
PRECISION_OBJS = src/makeplan.o src/setpts.o src/execute.o \
                 src/spreadinterp.o \
                 src/spreadinterp_1d.o src/spreadinterp_2d.o src/spreadinterp_3d.o
# common objs compiled once for both precisions
COMMON_OBJS = src/fft.o src/c_interface.o fortran/finufftfort.o
# all lib dual-precision objs (note DUCC_OBJS empty if unused)
OBJS = $(SOBJS) $(PRECISION_OBJS) $(PRECISION_OBJS:%.o=%_f.o) $(COMMON_OBJS) $(DUCC_OBJS)

.PHONY: usage lib cufinufft checkgpu examples test perftest spreadtest spreadtestall fortran matlab octave all mex python clean objclean pyclean mexclean wheel docker-wheel gurutime docs web setup setupclean

default: usage

all: test lib examples fortran matlab octave python spreadtest spreadtestsweep perftest

usage:
	@echo "Makefile for FINUFFT. Please specify your task:"
	@echo " make lib - build the main CPU library (in lib/ and lib-static/)"
	@echo " make cufinufft - build the GPU library (needs the CUDA toolkit)"
	@echo " make checkgpu - compile and run quick GPU math validation tests"
	@echo " make examples - compile and run all codes in examples/"
	@echo " make test - compile and run quick math validation tests"
	@echo " make fortran - compile and run Fortran tests and examples"
	@echo " make matlab - compile MATLAB interfaces (no test)"
	@echo " make octave - compile then test octave interfaces"
	@echo " make python - compile then test python interfaces"
	@echo " make spreadtest - compile & run spreader-only perf tests"
	@echo " make spreadtestsweep - spreader-only perf tests, sweep all tols"
	@echo " make perftest - compile and run some performance tests (~1 min)"
	@echo " make all - do all the above (~3 min; assumes have MEX, etc)"
	@echo " make objclean - remove all object files, preserving libs & MEX"
	@echo " make clean - also remove all lib, MEX, py, and demo executables"
	@echo " make setup - check (and possibly download) dependencies"
	@echo " make setupclean - delete downloaded dependencies (try if errors)"
	@echo " make web - build HTML docs and serve at http://localhost:8042"
	@echo "For faster (multicore) compilation, append, for example, -j8"
	@echo ""
	@echo "Make options:"
	@echo " 'make [task] OMP=OFF' for single-threaded (no refs to OpenMP)"
	@echo " 'make [task] FFT=DUCC' for DUCC0 FFT (otherwise uses FFTW3)"
	@echo " 'make [task] LTO=OFF' disable link time optimization"
	@echo " You must at least 'make objclean' before changing such options!"
	@echo ""
	@echo "Also see docs/install.rst and docs/README"

# collect headers for implicit depends (we don't separate public from private here)
HEADERS = $(wildcard include/*.h include/finufft/*.h include/finufft/*.hpp include/finufft_common/*.h)

# implicit rules for objects (note -o ensures writes to correct dir)
%.o: %.cpp $(HEADERS)
	$(CXX) -c $(CXXFLAGS) $< -o $@
src/%.o: src/%.cpp $(HEADERS)
	$(CXX) $(OBJFLAGS) -c $(CXXFLAGS) $< -o $@
# single-precision variants: compile same source with -DFINUFFT_SINGLE
src/%_f.o: src/%.cpp $(HEADERS)
	$(CXX) $(OBJFLAGS) -DFINUFFT_SINGLE -c $(CXXFLAGS) $< -o $@
fortran/%.o: fortran/%.cpp $(HEADERS)
	$(CXX) $(OBJFLAGS) -c $(CXXFLAGS) $< -o $@
%.o: %.c $(HEADERS)
	$(CC) -c $(CFLAGS) $< -o $@
%.o: %.f
	$(FC) -c $(FFLAGS) $< -o $@

# rule for spreadinterp: includes auto-generated code, xsimd header-only dependency;
# if FFT=DUCC also setup ducc with fft.h dependency on $(DUCC_SETUP)...
# Note src/spreadinterp.cpp includes finufft/plan.hpp which pulls in FFT forward decls
# so fftw/ducc header needed for spreadinterp, though spreadinterp should not
# depend on fftw/ducc directly?
SHEAD = $(XSIMD_DIR)/include/xsimd/xsimd.hpp $(POET_DIR)/include/poet/poet.hpp
src/spreadinterp.o: src/spreadinterp.cpp include/finufft/spreadinterp.hpp include/finufft/utils.hpp include/finufft_common/kernel.h include/finufft_common/spread_opts.h $(SHEAD)

# we need xsimd functionality in plan.hpp, which is included by many other
# files, so make sure we install xsimd before we process any of those files.
include/finufft/plan.hpp: $(XSIMD_DIR)/include/xsimd/xsimd.hpp $(POET_DIR)/include/poet/poet.hpp

# lib -----------------------------------------------------------------------
# build library with double/single prec both bundled in...
lib: $(STATICLIB) $(DYNLIB)
$(STATICLIB): $(OBJS)
	ar rcs $(STATICLIB) $(OBJS)
ifeq ($(OMP),OFF)
	@echo "$(STATICLIB) built, single-thread version"
else
	@echo "$(STATICLIB) built, multithreaded version"
endif
$(DYNLIB): $(OBJS)
# using *absolute* path in the -o here is needed to make portable executables
# when compiled against it, in mac OSX, strangely...
	$(CXX) -shared ${LDFLAGS} $(OMPFLAGS) $(OBJS) -o $(ABSDYNLIB) $(LIBSFFT)
ifeq ($(OMP),OFF)
	@echo "$(DYNLIB) built, single-thread version"
else
	@echo "$(DYNLIB) built, multithreaded version"
endif

# here $(OMPFLAGS) and $(LIBSFFT) is even needed for linking under mac osx.
# see: http://www.cprogramming.com/tutorial/shared-libraries-linux-gcc.html
# Also note -l libs come after objects, as per modern GCC requirement.


# GPU library (cuFINUFFT) ----------------------------------------------------
# Needs the CUDA toolkit (nvcc + cuFFT); CMake remains the tested route, this
# mirrors src/cuda/CMakeLists.txt for sites that build with the makefile.
# Override NVCC/NVARCH (and CXX, used as nvcc's host compiler) in make.inc;
# see make-platforms/make.inc.{FI,CIMS,nersc_perlmutter} for site examples.
NVCC ?= nvcc
# fat binary by default: no GPU is needed at build time and the result runs on
# any device the toolkit supports. For one known GPU use eg NVARCH = -arch=sm_80
NVARCH ?= -arch=all-major
CUINCL = -Iinclude -Icontrib -I$(POET_DIR)/include -Isrc/cuda
NVCCFLAGS := -O3 -std=c++17 $(NVARCH) $(CUINCL) -ccbin=$(CXX) --extended-lambda \
	     --extra-device-vectorization -Xcompiler "-fPIC -fvisibility=hidden" $(NVCCFLAGS)
# pure-host TUs of the GPU lib (no kernels): -x c++ hands them straight to the
# host compiler, but still with nvcc's include paths (cuda_runtime.h, CCCL)
CUXXFLAGS := -x c++ -O3 -std=c++17 $(CUINCL) -ccbin=$(CXX) -Xcompiler "-fPIC -fvisibility=hidden" $(CUXXFLAGS)
CULIBS = -lcufft -lcudart
CULIBNAME = libcufinufft
CUDYNLIB = lib/$(CULIBNAME).so
CUSTATICLIB = lib-static/$(CULIBNAME).a
ABSCUDYNLIB = $(FINUFFT)$(CUDYNLIB)

# device TUs...
CUOBJS := $(addprefix src/cuda/,fft.o heuristics.o makeplan.o setpts.o execute.o spread_blockgather_inst.o)
# ...plus each *_inst.cu compiled once per dim (as configure_file does in CMake)
CUMETHODS = spread_nupts_driven spread_subprob spread_output_driven interp_nupts_driven interp_subprob
CUOBJS += $(foreach m,$(CUMETHODS),$(foreach d,1 2 3,src/cuda/$(m)_$(d)d.o))
# ...plus host-only TUs, and the precision-independent common objects
CUOBJS += src/cuda/c_interface.o src/cuda/spreadinterp.o $(SOBJS)
CUHEADERS = $(HEADERS) $(POET_DIR)/include/poet/poet.hpp $(wildcard include/cufinufft/*.hpp src/cuda/*.cuh)

src/cuda/%_1d.o: src/cuda/%_inst.cu $(CUHEADERS)
	$(NVCC) $(NVCCFLAGS) -DCUFINUFFT_DIM=1 -c $< -o $@
src/cuda/%_2d.o: src/cuda/%_inst.cu $(CUHEADERS)
	$(NVCC) $(NVCCFLAGS) -DCUFINUFFT_DIM=2 -c $< -o $@
src/cuda/%_3d.o: src/cuda/%_inst.cu $(CUHEADERS)
	$(NVCC) $(NVCCFLAGS) -DCUFINUFFT_DIM=3 -c $< -o $@
src/cuda/%.o: src/cuda/%.cu $(CUHEADERS)
	$(NVCC) $(NVCCFLAGS) -c $< -o $@
src/cuda/%.o: src/cuda/%.cpp $(CUHEADERS)
	$(NVCC) $(CUXXFLAGS) -c $< -o $@

cufinufft: $(CUSTATICLIB) $(CUDYNLIB)
$(CUSTATICLIB): $(CUOBJS)
	ar rcs $(CUSTATICLIB) $(CUOBJS)
	@echo "$(CUSTATICLIB) built, $(NVARCH)"
$(CUDYNLIB): $(CUOBJS)
	$(NVCC) -shared $(NVARCH) -ccbin=$(CXX) $(CUOBJS) -o $(ABSCUDYNLIB) $(CULIBS) $(OMPLIBS)
	@echo "$(CUDYNLIB) built, $(NVARCH)"

# GPU math validation: needs an actual GPU (submit to a compute node on a cluster)
# these use the internal C++ API, which the .so hides (-fvisibility=hidden), so
# they link against the archive
CUTESTS = test/cuda/cufinufft1d_test test/cuda/cufinufft2d_test test/cuda/cufinufft3d_test
test/cuda/%: test/cuda/%.cu $(CUSTATICLIB)
	$(NVCC) $(NVCCFLAGS) $< $(CUSTATICLIB) $(CULIBS) $(LIBS) -o $@
checkgpu: $(CUTESTS)
	test/cuda/cufinufft1d_test 0 1 2e3 4e3 1e-8 1e-7 d 2.0
	test/cuda/cufinufft2d_test 0 1 2e2 2e2 4e3 1e-8 1e-7 d 2.0
	test/cuda/cufinufft3d_test 0 1 20 40 30 4e3 1e-8 1e-7 d 2.0
	@echo "GPU math tests passed"


# examples (C++/C) -----------------------------------------------------------
# build all examples (single-prec codes separate, and not all have one)...
EXAMPLES := $(basename $(wildcard examples/*.c examples/*.cpp))
ifeq ($(OMP),OFF)
  EXAMPLES := $(filter-out $(basename $(wildcard examples/*thread*.cpp)),$(EXAMPLES))
endif
examples: $(EXAMPLES)
ifneq ($(MINGW),ON)
  # Windows-MSYS does not find the dynamic libraries, so we make a temporary copy
  # Windows-MSYS has same commands as Linux/OSX
  ifeq ($(MSYS),ON)
	cp $(DYNLIB) test
  endif
  # non-Windows-WSL: this task always runs them (note escaped $ to pass to bash)...
	for i in $(EXAMPLES); do echo $$i...; ./$$i; done
else
  # Windows-WSL does not find the dynamic libraries, so we make a temporary copy
	copy $(DYNLIB) examples
	for /f "delims= " %%i in ("$(subst /,\,$(EXAMPLES))") do (echo %%i & %%i.exe)
	del examples\$(LIBNAME).so
endif
	@echo "Done running: $(EXAMPLES)"
# fun fact: gnu make patterns match those with shortest "stem", so this works:
examples/%: examples/%.o $(DYNLIB)
	$(CXX) $(CXXFLAGS) ${LDFLAGS} $< $(ABSDYNLIB) $(LIBSFFT) -o $@
examples/%c: examples/%c.o $(DYNLIB)
	$(CC) $(CFLAGS) ${LDFLAGS} $< $(ABSDYNLIB) $(LIBSFFT) $(CLINK) -o $@
examples/%cf: examples/%cf.o $(DYNLIB)
	$(CC) $(CFLAGS) ${LDFLAGS} $< $(ABSDYNLIB) $(LIBSFFT) $(CLINK) -o $@


# test (library validation) --------------------------------------------------
# build (skipping .o) but don't run. Run with 'test' target
# Note: both precisions use same sources; single-prec executables get f suffix.
# generic tests link against our .so... (other libs needed for fftw_forget...)
test/%: test/%.cpp $(DYNLIB)
	$(CXX) $(CXXFLAGS) ${LDFLAGS} $< $(ABSDYNLIB) $(LIBSFFT) -o $@
test/%f: test/%.cpp $(DYNLIB)
	$(CXX) $(CXXFLAGS) ${LDFLAGS} -DSINGLE $< $(ABSDYNLIB) $(LIBSFFT) -o $@
# C test for error-path handling in the C interface.
test/error_handling: test/error_handling.c $(DYNLIB)
	$(CC) $(CFLAGS) ${LDFLAGS} $< $(ABSDYNLIB) $(LIBSFFT) $(CLINK) -o $@
# low-level tests that are cleaner if depend on only specific objects...
# testutils also unit-tests the upsampfac picker (finufft/heuristics.hpp), which pulls
# in kernel.o/pswf.o symbols; $(SOBJS) is exactly the precision-independent object set.
test/testutils: test/testutils.cpp $(SOBJS)
	$(CXX) $(CXXFLAGS) ${LDFLAGS} test/testutils.cpp $(SOBJS) $(LIBS) -o test/testutils
test/testutilsf: test/testutils.cpp $(SOBJS)
	$(CXX) $(CXXFLAGS) ${LDFLAGS} -DSINGLE test/testutils.cpp $(SOBJS) $(LIBS) -o test/testutilsf

# make sure all double-prec test executables ready for testing
CPPTESTS := $(basename $(wildcard test/*.cpp))
# kill off FFTW-specific tests if it's not the FFT we build with...
ifeq ($(FFT),DUCC)
  CPPTESTS := $(filter-out $(basename $(wildcard test/*fftw*.cpp)),$(CPPTESTS))
endif
# kill off tests demanding multithreading if single-thread build...
ifeq ($(OMP),OFF)
  CPPTESTS := $(filter-out $(basename $(wildcard test/fftw_lock_test.cpp)),$(CPPTESTS))
endif

# single-precision variants for C++ tests only, plus C-interface error handling test.
TESTS := $(CPPTESTS) $(CPPTESTS:%=%f) test/error_handling
test: $(TESTS)
ifneq ($(MINGW),ON)
  # non-Windows-WSL: it will fail if either of these return nonzero exit code...
  # Windows-MSYS does not find the dynamic libraries, so we make a temporary copy
  # Windows-MSYS has same commands as Linux/OSX
  ifeq ($(MSYS),ON)
	cp $(DYNLIB) test
  endif
	test/basicpassfail
	test/basicpassfailf
  # accuracy tests done in prec-switchable bash script... (small prob -> few thr)
	(cd test; export OMP_NUM_THREADS=4; ./check_finufft.sh; ./check_finufft.sh SINGLE)
else
  # Windows-WSL does not find the dynamic libraries, so we make a temporary copy...
	copy $(DYNLIB) test
	test/basicpassfail
	test/basicpassfailf
  # Windows does not feature a bash shell so we use WSL. Since most supplied gnu-make variants are 32bit executables and WSL runs only in 64bit environments, we have to refer to 64bit powershell explicitly on 32bit make...
  #	$(windir)\Sysnative\WindowsPowerShell\v1.0\powershell.exe "cd ./test; bash check_finufft.sh DOUBLE $(MINGW); bash check_finufft.sh SINGLE $(MINGW)"
  # with a recent version of gnu-make for Windows built for 64bit as it is part of the WinLibs standalone build of GCC and MinGW-w64 we can avoid these circumstances
	cd test
	bash -c "cd test; ./check_finufft.sh DOUBLE $(MINGW)"
	bash -c "cd test; ./check_finufft.sh SINGLE $(MINGW)"
	del test\$(LIBNAME).so
endif


# perftest (performance/developer tests) -------------------------------------
# generic perf test rules...
perftest/%: perftest/%.cpp $(STATICLIB)
	$(CXX) $(CXXFLAGS) ${LDFLAGS} $< $(STATICLIB) $(LIBSFFT) -o $@
perftest/%f: perftest/%.cpp $(DYNLIB)
	$(CXX) $(CXXFLAGS) ${LDFLAGS} -DSINGLE $< $(STATICLIB) $(LIBSFFT) -o $@

# spread/interp tests, double/single (used to be isolated from FINUFFT, now
# require FINUFFT lib to be built because they use its API)...
ST=perftest/spreadtestnd
STA=perftest/spreadtestndall
STF=$(ST)f
STAF=$(STA)f

$(ST): $(ST).cpp $(STATICLIB)
	$(CXX) $(CXXFLAGS) ${LDFLAGS} $< $(STATICLIB) $(LIBSFFT) -o $@
$(STF): $(ST).cpp $(DYNLIB)
	$(CXX) $(CXXFLAGS) ${LDFLAGS} -DSINGLE $< $(STATICLIB) $(LIBSFFT) -o $@
$(STA): $(STA).cpp $(DYNLIB)
	$(CXX) $(CXXFLAGS) ${LDFLAGS} $< $(STATICLIB) $(LIBSFFT) -o $@
$(STAF): $(STA).cpp $(DYNLIB)
	$(CXX) $(CXXFLAGS) ${LDFLAGS} -DSINGLE $< $(STATICLIB) $(LIBSFFT) -o $@

spreadtest: $(ST) $(STF)
# This set of executables is similar to ./spreadtestall.sh; consider unifying:
# Run one thread per core... (escape the $ to get single $ in bash; one big cmd)
	(export OMP_NUM_THREADS=$$(perftest/mynumcores.sh) ;\
	echo "\nRunning makefile double-precision spreader tests, $$OMP_NUM_THREADS threads..." ;\
	$(ST) 1 8e6 8e6 1e-6 ;\
	$(ST) 2 8e6 8e6 1e-6 ;\
	$(ST) 3 8e6 8e6 1e-6 ;\
	echo "\nRunning makefile single-precision spreader tests, $$OMP_NUM_THREADS threads..." ;\
	$(STF) 1 8e6 8e6 1e-3 ;\
	$(STF) 2 8e6 8e6 1e-3 ;\
	$(STF) 3 8e6 8e6 1e-3 )
# spread/interp speed test, sweep through all tolerances (for each prec, dim, dir):
spreadtestsweep: $(STA) $(STAF)
	(cd perftest; ./spreadtestndsweep.sh)
bigtest: perftest/big2d2f
	@echo "\nRunning >2^31 size example (takes 30 s and 30 GB RAM)..."
	perftest/big2d2f

PERFEXECS := $(basename $(wildcard test/finufft?d_test.cpp))
PERFEXECS += $(PERFEXECS:%=%f)
perftest: $(ST) $(STF) $(PERFEXECS) spreadtestsweep gurutime manysmallprobs bigtest
# here the tee cmd copies output to screen. 2>&1 grabs both stdout and stderr...
	(cd perftest ;\
	./spreadtestnd.sh 2>&1 | tee results/spreadtestnd_results.txt ;\
	./spreadtestnd.sh SINGLE 2>&1 | tee results/spreadtestndf_results.txt ;\
	./nuffttestnd.sh 2>&1 | tee results/nuffttestnd_results.txt ;\
	./nuffttestnd.sh SINGLE 2>&1 | tee results/nuffttestndf_results.txt )

# speed ratio of many-vector guru vs repeated single calls... (Andrea)
GTT=perftest/guru_timing_test
GTTF=$(GTT)f
gurutime: $(GTT) $(GTTF)
	for i in $(GTT) $(GTTF); do $$i 100 1 2 1e2 1e2 0 1e6 1e-3 1 0 2; done

# This was for a CCQ application... (zgemm was 10x faster! double-prec only)
manysmallprobs: perftest/manysmallprobs
	@echo "run manysmallprobs, double-prec, single-thread..."
	OMP_NUM_THREADS=1 perftest/manysmallprobs




# ======================= LANGUAGE INTERFACES ==============================

# fortran --------------------------------------------------------------------
FD = fortran/directft
# CMCL NUFFT fortran test codes (only needed by the nufft*_demo* codes)
CMCLOBJS = $(FD)/dirft1d.o $(FD)/dirft2d.o $(FD)/dirft3d.o $(FD)/dirft1df.o $(FD)/dirft2df.o $(FD)/dirft3df.o $(FD)/prini.o
# build examples list...
FE_DIR = fortran/examples
FE64 = $(FE_DIR)/simple1d1 $(FE_DIR)/simple1d1_f90 $(FE_DIR)/guru1d1 $(FE_DIR)/guru1d1_adjoint $(FE_DIR)/guru1d2_adjoint $(FE_DIR)/nufft1d_demo $(FE_DIR)/nufft2d_demo $(FE_DIR)/nufft3d_demo $(FE_DIR)/nufft2dmany_demo
# add the "f" single-prec suffix to all examples except double-prec only ones...
FE32 := $(filter-out %/simple1d1_f90f %/guru1d1_adjointf, $(FE64:%=%f))
# list of all fortran examples
FE = $(FE64) $(FE32)

# fortran target pattern match (no longer runs executables)
$(FE_DIR)/%: $(FE_DIR)/%.f $(CMCLOBJS) $(DYNLIB) include/finufft.fh
	$(FC) $(FFLAGS) ${LDFLAGS} $< $(CMCLOBJS) $(ABSDYNLIB) $(FLINK) -o $@
$(FE_DIR)/%f: $(FE_DIR)/%f.f $(CMCLOBJS) $(DYNLIB) include/finufft.fh
	$(FC) $(FFLAGS) ${LDFLAGS} $< $(CMCLOBJS) $(ABSDYNLIB) $(FLINK) -o $@
# fortran90 lone demo
$(FE_DIR)/simple1d1_f90: $(FE_DIR)/simple1d1.f90 include/finufft_mod.f90 $(CMCLOBJS) $(DYNLIB)
	$(FC) $(FFLAGS) ${LDFLAGS} include/finufft_mod.f90 $< $(CMCLOBJS) $(ABSDYNLIB) $(FLINK) -o $@

fortran: $(FE)
# this task runs them (note escaped $ to pass to bash)...
	for i in $(FE); do echo $$i...; ./$$i; done
	@echo "Done running: $(FE)"


# matlab ----------------------------------------------------------------------
# matlab .mex* executable... (matlab is so slow to start, and bad at batch
# scripting [can get stuck inside], not worth testing it here; user must test)
matlab: matlab/finufft.cpp $(STATICLIB)
	$(MEX) $< $(STATICLIB) $(INCL) $(MFLAGS) $(LIBSFFT) -output matlab/finufft

# octave .mex executable...
octave: matlab/finufft.cpp $(STATICLIB)
	(cd matlab; $(MKOCTFILE) --mex finufft.cpp -I../include ../$(STATICLIB) $(OFLAGS) $(LIBSFFT) -output finufft)
	@echo "Running octave interface tests; please wait 30 seconds..."
	(cd matlab ;\
	$(OCTAVE) test/fullmathtest.m ;\
	$(OCTAVE) test/check_finufft.m ;\
	$(OCTAVE) test/check_finufft_single.m ;\
	$(OCTAVE) test/check_opts.m ;\
	$(OCTAVE) examples/guru1d1.m ;\
	$(OCTAVE) examples/guru1d1_single.m ;\
	$(OCTAVE) examples/guru1d1_adjoint.m)

# for experts: force rebuilds fresh MEX (matlab/octave) gateway
# matlab/{cu}finufft.cpp via mwrap (needs recent version of mwrap >= 1.2)...
mex: matlab/finufft.mw matlab/cufinufft.mw
ifneq ($(MINGW),ON)
	(cd matlab ;\
	$(MWRAP) -mex finufft -c finufft.cpp -mb -cppcomplex finufft.mw ;\
	$(MWRAP) -mex cufinufft -c cufinufft.cu -mb -cppcomplex -gpu cufinufft.mw)
else
	(cd matlab & $(MWRAP) -mex finufft -c finufft.cpp -mb -cppcomplex finufft.mw)
	(cd matlab & $(MWRAP) -mex cufinufft -c cufinufft.cu -mb -cppcomplex -gpu cufinufft.mw)
endif
	(cd matlab; ./addmhelp.sh)
	(cd docs; ./genmatlabhelp.sh)



# python ---------------------------------------------------------------------
# this task uses pyproject.toml and cmake (as of v2.3), so no more lib/static dep
python:
# note use of CMAKE_ARGS which needs quotes; see scikit-build docs...
	FINUFFT_DIR=$(FINUFFT) CMAKE_ARGS=$(PY_CMAKE_ARGS) $(PYTHON) -m pip -v install python/finufft
# note to devs: if trouble w/ NumPy, use: pip install ./python --no-deps
	$(PYTHON) python/finufft/test/run_accuracy_tests.py
	$(PYTHON) python/finufft/examples/simple1d1.py
	$(PYTHON) python/finufft/examples/simpleopts1d1.py
	$(PYTHON) python/finufft/examples/guru1d1.py
	$(PYTHON) python/finufft/examples/guru1d1f.py
	$(PYTHON) python/finufft/examples/simple2d1.py
	$(PYTHON) python/finufft/examples/many2d1.py
	$(PYTHON) python/finufft/examples/guru2d1.py
	$(PYTHON) python/finufft/examples/guru2d1f.py

# general python packaging wheel for all OSs without wheel being fixed(required shared libs are not included in wheel)
python-dist: $(STATICLIB) $(DYNLIB)
	(export FINUFFT_DIR=$(shell pwd); cd python/finufft; $(PYTHON) -m pip wheel . -w wheelhouse)

# python packaging wheel for macosx with wheel being fixed(all required shared libs are included in wheel)
wheel: $(STATICLIB) $(DYNLIB)
	(export FINUFFT_DIR=$(shell pwd); cd python/finufft; $(PYTHON) -m pip wheel . -w wheelhouse; delocate-wheel -w fixed_wheel -v wheelhouse/finufft*.whl)

docker-wheel:
	docker run --rm -e package_name=finufft -v `pwd`:/io libinlu/manylinux2010_x86_64_fftw /io/python/ci/build-wheels.sh


# ================== SETUP/COMPILE OF EXTERNAL DEPENDENCIES ===============

# this utility can get a tag or commit (similar to CPMAddPackage):
define clone_repo
	@if [ ! -d "$(3)" ]; then \
		echo "Cloning repository $(1) at ref $(2) into directory $(3)"; \
		git clone --no-checkout $(1) $(3) && \
		cd $(3) && \
		git fetch origin --tags --force && \
		git fetch origin $(2) --force >/dev/null 2>&1 || true; \
		TARGET_COMMIT=$$(git rev-parse --verify $(2)^{commit} 2>/dev/null) || { echo "Error: Failed to resolve ref $(2) in $(3)."; exit 1; }; \
		git -c advice.detachedHead=false checkout $$TARGET_COMMIT; \
	else \
		cd $(3) && \
		git fetch origin --tags --force && \
		git fetch origin $(2) --force >/dev/null 2>&1 || true; \
		TARGET_COMMIT=$$(git rev-parse --verify $(2)^{commit} 2>/dev/null) || { echo "Error: Failed to resolve ref $(2) in $(3)."; exit 1; }; \
		CURRENT_COMMIT=$$(git rev-parse HEAD 2>/dev/null || echo ""); \
		if [ "$$CURRENT_COMMIT" = "$$TARGET_COMMIT" ]; then \
			echo "Directory $(3) already exists and is at the correct version $(2) ($$CURRENT_COMMIT)."; \
		else \
			echo "Directory $(3) exists but is at commit $$CURRENT_COMMIT. Checking out $(2) ($$TARGET_COMMIT)."; \
			git -c advice.detachedHead=false checkout $$TARGET_COMMIT || { echo "Error: Failed to checkout ref $(2) in $(3)."; exit 1; }; \
		fi; \
	fi
endef


# download: header-only, no compile needed...
$(XSIMD_DIR)/include/xsimd/xsimd.hpp:
	mkdir -p $(DEPS_ROOT)
	@echo "Checking XSIMD external dependency..."
	$(call clone_repo,$(XSIMD_URL),$(XSIMD_VERSION),$(XSIMD_DIR))
	@echo "xsimd installed in deps/xsimd"

# download: POET, one self-contained header, no compile needed...
$(POET_DIR)/include/poet/poet.hpp:
	mkdir -p $(dir $@)
	@echo "Downloading POET $(POET_VERSION)..."
	curl -sSfL -o $@ $(POET_URL)
	@echo "POET installed in $(POET_DIR)"

# download DUCC... (an empty target just used to track if installed)
$(DUCC_COOKIE):
	mkdir -p $(DEPS_ROOT)
	@echo "Checking DUCC external dependency..."
	$(call clone_repo,$(DUCC_URL),$(DUCC_VERSION),$(DUCC_DIR))
	touch $(DUCC_COOKIE)
	@echo "DUCC installed in deps/ducc"

# implicit rule for DUCC compile just needed objects, only used if FFT=DUCC.
# Needed since DUCC has no makefile (yet).
$(DUCC_SRCS): %.cc: $(DUCC_SETUP)
$(DUCC_OBJS): %.o: %.cc
	$(CXX) -c $(DUCC_CXXFLAGS) $(DUCC_INCL) $< -o $@

setup: $(XSIMD_DIR)/include/xsimd/xsimd.hpp $(POET_DIR)/include/poet/poet.hpp $(DUCC_SETUP)

setupclean:
	rm -rf $(DEPS_ROOT)


# =============================== DOCUMENTATION =============================

docs: docs/*.docsrc docs/matlabhelp.doc docs/makecdocs.sh
	(cd docs; ./makecdocs.sh)
# get the makefile help strings from make w/o args, stdout...
	make --no-print-directory 1> docs/makefile.doc
docs/matlabhelp.doc: docs/genmatlabhelp.sh matlab/*.sh matlab/*.docsrc matlab/*.docbit matlab/*.m
	(cd matlab; ./addmhelp.sh)
	(cd docs; ./genmatlabhelp.sh)

# build the sphinx HTML docs (needs the python pkgs in docs/requirements.txt)
# and serve them at http://localhost:8042 for local checking; Ctrl-C stops...
web:
	make -C docs html || { \
		echo ""; \
		echo "Doc build failed - missing sphinx deps? Set up an environment with uv:"; \
		echo "  uv venv ~/.venvs/finufft-docs   # skip if the venv already exists"; \
		echo "  uv pip install --python ~/.venvs/finufft-docs/bin/python sphinx -r docs/requirements.txt"; \
		echo "  source ~/.venvs/finufft-docs/bin/activate"; \
		echo "or with plain pip:"; \
		echo "  python3 -m venv ~/.venvs/finufft-docs   # skip if the venv already exists"; \
		echo "  ~/.venvs/finufft-docs/bin/pip install sphinx -r docs/requirements.txt"; \
		echo "  source ~/.venvs/finufft-docs/bin/activate"; \
		echo "then re-run 'make web' with the venv active."; \
		exit 1; }
	@echo "Serving docs at http://localhost:8042 (Ctrl-C to stop)"
	@$(PYTHON) -m http.server 8042 -d docs/_build/html



# =============================== CLEAN UP ==================================

clean: objclean pyclean
ifneq ($(MINGW),ON)
  # non-Windows-WSL clean up...
	rm -f $(STATICLIB) $(DYNLIB) $(CUSTATICLIB) $(CUDYNLIB)
	rm -f matlab/finufft.mex*
	rm -f $(TESTS) $(CUTESTS) test/results/*.out perftest/results/*.out
	rm -f $(EXAMPLES) $(FE) $(ST) $(STF) $(STA) $(STAF) $(GTT) $(GTTF)
	rm -f perftest/manysmallprobs perftest/big2d2f
	rm -f examples/core test/core perftest/core $(FE_DIR)/core
	rm -f fortran/examples/finufft_mod.mod
else
  # Windows-WSL clean up...
	del $(subst /,\,$(STATICLIB)), $(subst /,\,$(DYNLIB))
	del matlab\finufft.mex*
	for %%f in ($(subst /,\, $(TESTS))) do ((if exist %%f del %%f) & (if exist %%f.exe del %%f.exe))
	del test\results\*.out perftest\results\*.out
	for %%f in ($(subst /,\, $(EXAMPLES)), $(subst /,\,$(FE)), $(subst /,\,$(ST)), $(subst /,\,$(STF)), $(subst /,\,$(STA)), $(subst /,\,$(STAF)), $(subst /,\,$(GTT)), $(subst /,\,$(GTTF))) do ((if exist %%f del %%f) & (if exist %%f.exe del %%f.exe))
	del perftest\manysmallprobs, perftest\big2d2f
	del examples\core, test\core, perftest\core, $(subst /,\, $(FE_DIR))\core
	del fortran\examples\finufft_mod.mod
endif


# indiscriminate .o killer; needed before changing, eg, compiler, threading...
objclean:
ifneq ($(MINGW),ON)
  # some system other than Windows-WSL... (note: cleans DUCC objects regardless of FFT choice)
	rm -f src/*.o src/common/*.o src/cuda/*.o test/*.o examples/*.o matlab/*.o
	rm -f fortran/*.o $(FE_DIR)/*.o $(FD)/*.o finufft_mod.mod
	rm -f $(DUCC_SRC)/infra/*.o $(DUCC_SRC)/math/*.o
else
  # Windows-WSL...
	for /d %%d in (src,src\common,test,examples,matlab) do (for %%f in (%%d\*.o) do (del %%f))
	for /d %%d in (fortran,$(subst /,\, $(FE_DIR)),$(subst /,\, $(FD))) do (for %%f in (%%d\*.o) do (del %%f))
  # *** to del DUCC *.o
endif

pyclean:
ifneq ($(MINGW),ON)
  # some system other than Windows-WSL...
	rm -f python/finufft/*.pyc python/finufft/__pycache__/* python/test/*.pyc python/test/__pycache__/*
	rm -rf python/fixed_wheel python/wheelhouse
else
  # Windows-WSL...
	for /d %%d in (python\finufft,python\test) do (for %%f in (%%d\*.pyc) do (del %%f))
	for /d %%d in (python\finufft\__pycache__,python\test\__pycache__) do (for %%f in (%%d\*) do (del %%f))
	for /d %%d in (python\fixed_wheel,python\wheelhouse) do (if exist %%d (rmdir /s /q %%d))
endif

# for experts; only run this if you possess mwrap to rebuild the interfaces!
mexclean:
ifneq ($(MINGW),ON)
  # non-Windows-WSL...
	rm -f matlab/finufft_plan.m matlab/finufft.cpp matlab/finufft.mex*
else
  # Windows-WSL...
	del matlab\finufft_plan.m matlab\finufft.cpp matlab\finufft.mex*
endif
