From cfecbca4880d50d6f42a55aa2c86b1b5323f435e Mon Sep 17 00:00:00 2001 From: msilvestrec03 Date: Fri, 6 Mar 2026 12:18:39 +0100 Subject: [PATCH 01/27] prova 1 --- options.cfg | 47 ++++++++++++++++------------------------------- 1 file changed, 16 insertions(+), 31 deletions(-) diff --git a/options.cfg b/options.cfg index 429a2403..5177cae9 100644 --- a/options.cfg +++ b/options.cfg @@ -1,50 +1,35 @@ -# Compile PYLOM -# Compile with g++ or Intel C++ Compiler -# Compile with the most aggressive optimization setting (O3) -# Use the most pedantic compiler settings: must compile with no warnings at all -# -# The user may override any desired internal variable by redefining it via command-line: -# make CXX=g++ [...] -# make OPTL=-O2 [...] -# make FLAGS="-Wall -g" [...] -# -# Arnau Miro 2021 - ## Options # -PLATFORM = PC -VECTORIZATION = ON -OPENMP_PARALL = OFF -USE_MKL = ON -USE_FFTW = OFF -USE_GCC = OFF -USE_NVHPC = OFF -DEBUGGING = OFF -USE_GESVD = OFF -USE_COMPILED = ON +PLATFORM = MN5_GPP +VECTORIZATION = ON +OPENMP_PARALL = OFF +USE_MKL = ON +USE_FFTW = ON +USE_GCC = OFF +USE_NVHPC = OFF +DEBUGGING = OFF +USE_GESVD = OFF +USE_COMPILED = ON # Comma separated list of the modules to be compiled # if USE_COMPILED = ON -MODULES_COMPILED = MATH.MATHS,MATH.AVERAGING,MATH.QR,MATH.SVD,MATH.FFT,MATH.GEOMETRIC,MATH.TRUNCATION,MATH.STATS,MATH.REGRESSION,ROM.POD,ROM.DMD,ROM.SPOD - +MODULES_COMPILED = MATH.MATHS,MATH.AVERAGING,MATH.SVD,MATH.FFT,MATH.GEOMETRIC,MATH.TRUNCATION,MATH.STATS,MATH.REGRESSION,ROM.POD,ROM.DMD,ROM.SPOD ## Optimization, host and CPU type # OPTL = 3 HOST = Host -TUNE = skylake - +TUNE = sapphirerapids ## Python versions # PYTHON = python3 PIP = pip3 - ## Versions of the libraries # -ONEAPI_VERS = 2024.2.0.634 -OPENBLAS_VERS = 0.3.17 LAPACK_VERS = 3.9.0 -KISSFFT_VERS = 131.1.0 -FFTW_VERS = 3.3.8 +FFTW_VERS = 3.3.10 NFFT_VERS = 3.5.2 +KISSFFT_VERS = 131.1.0 +OPENBLAS_VERS = 0.3.17 +ONEAPI_VERS = 2023.2.0 From e1a9e6e6dff8f17fa45a289b6622b267fcedd82e Mon Sep 17 00:00:00 2001 From: msilvestrec03 Date: Fri, 6 Mar 2026 12:40:06 +0100 Subject: [PATCH 02/27] v1 RES + typo maths --- pyLOM/RES/__init__.py | 14 +++++++ pyLOM/RES/utils.py | 79 +++++++++++++++++++++++++++++++++++ pyLOM/RES/wrapper.py | 95 +++++++++++++++++++++++++++++++++++++++++++ pyLOM/__init__.py | 2 +- pyLOM/vmmath/maths.py | 4 +- 5 files changed, 191 insertions(+), 3 deletions(-) create mode 100644 pyLOM/RES/__init__.py create mode 100644 pyLOM/RES/utils.py create mode 100644 pyLOM/RES/wrapper.py diff --git a/pyLOM/RES/__init__.py b/pyLOM/RES/__init__.py new file mode 100644 index 00000000..78f57a32 --- /dev/null +++ b/pyLOM/RES/__init__.py @@ -0,0 +1,14 @@ +#!/usr/bin/env python +# +# pyLOM - Python Low Order Modeling. +# +# Resolvent Module +# +# Last rev: 20/02/2026 + +# Functions coming from Resolvent +from .wrapper import run, run_mine +from .utils import extract_modes, save, load + + +del wrapper diff --git a/pyLOM/RES/utils.py b/pyLOM/RES/utils.py new file mode 100644 index 00000000..75bd44a6 --- /dev/null +++ b/pyLOM/RES/utils.py @@ -0,0 +1,79 @@ +#!/usr/bin/env python +# +# pyLOM - Python Low Order Modeling. +# +# DMD general utilities. +# +# Last rev: 26/02/2026 +from __future__ import print_function, division + +import numpy as np + +from ..utils.gpu import cp +from .. import inp_out as io, PartitionTable +from ..utils import cr_nvtx as cr, gpu_to_cpu + +@cr('RES.extract_modes') +def extract_modes(U:np.ndarray,V:np.ndarray,ivar:int,npoints:int,modes:list=[],reshape:bool=True): + r''' + When performing POD of several variables simultaneously, this function separates the spatial modes from each of the variables. + + Args: + U (np.ndarray): RES response modes + V (np.ndarray): RES forcing modes + ivar (int): ID of the variable (i. e.) position in which it was concatenated to the rest of data (min=1, max=number of concatenated variables) + npoints (int): number of points in the domain per variable + modes (list, optional): list containing the id of the modes to separate (default ``[]``). + reshape (bool, optional): if true the output will be given as (len(modes)*npoints,) if not it the result will be (npoints, len(modes)) (default `` True ``) + + Returns: + np.ndarray: modes of the variable ivar + ''' + p = cp if type(U) is cp.ndarray else np + nvars = U.shape[0]//npoints + # Define modes to extract + if len(modes) == 0: modes = p.arange(1,U.shape[1]+1,dtype=p.int32) + # Allocate output array + out_U =p.zeros((npoints,len(modes)),U.dtype) + out_V =p.zeros((npoints,len(modes)),U.dtype) + for i,m in enumerate(modes): + out_U[:,i] = U[ivar-1:nvars*npoints:nvars,m-1] + out_V[:,i] = V[ivar-1:nvars*npoints:nvars,m-1] + # Return reshaped output + return out_U.reshape((len(modes)*npoints,),order='C') if reshape else out_U, out_V.reshape((len(modes)*npoints,),order='C') if reshape else out_V + +@cr('RES.save') +def save(fname:str,U:np.ndarray,S:np.ndarray,V:np.ndarray,ptable:PartitionTable,nvars:int=1,pointData:bool=True,mode:str='w'): + r''' + Store POD results in serial or parallel according to the partition used to compute the POD. It will be saved on a h5 file. + + Args: + fname (str): path to the .h5 file in which the POD will be saved + U (np.ndarray): response modes to save. To avoid saving the spatial modes, just give None as input + S (np.ndarray): singular values to save. To avoid saving the singular values, just give None as input + V (np.ndarray): forcing modes to save. To avoid saving the temporal coefficients, just give None as input + ptable (PartitionTable): partition table used to compute the POD + nvars (int, optional): number of concatenated variables when computing the POD (default ``1``) + pointData (bool, optional): bool to specify if the POD was performed either on point data or cell data (default ``True``) + mode (str, optional): mode in which the HDF5 file is opened, 'w' stands for write mode and 'a' stands for append mode. Write mode will overwrite the file and append mode will add the informaiton at the end of the current file, choose with great care what to do in your case (default ``w``). + + ''' + io.h5_save_POD(fname,gpu_to_cpu(U),gpu_to_cpu(S),gpu_to_cpu(V),ptable,nvars=nvars,pointData=pointData,mode=mode) + + +@cr('RES.load') +def load(fname:str,vars:list=['U','S','V'],nmod:int=-1,ptable:PartitionTable=None): + r''' + Load POD results from a .h5 file in serial or parallel according to the partition used to compute the POD. + + Args: + fname (str): path to the .h5 file in which the POD was saved + vars (list): list of variables to load. The following notation, consistent with the save function, is used, + 'U': spatial modes + 'S': singular values + 'V': temporal coefficients + the default option is to load them all, but it is not recommended to load the spatial modes if they are not going to be used during the rest of the script. + nmod (int, optional): number of modes to load. By default it will load all the saved modes (default, ``-1``) + ptable (PartitionTable, optional): partition table to use when loading the data (default ``None``). + ''' + return io.h5_load_POD(fname,vars,nmod,ptable) \ No newline at end of file diff --git a/pyLOM/RES/wrapper.py b/pyLOM/RES/wrapper.py new file mode 100644 index 00000000..000eb3ba --- /dev/null +++ b/pyLOM/RES/wrapper.py @@ -0,0 +1,95 @@ +#!/usr/bin/env python +# +# pyLOM - Python Low Order Modeling. +# +# Python interface for DMD. +# +# Last rev: 20/02/2026 +from __future__ import print_function + +import numpy as np + +from ..utils.gpu import cp +from ..vmmath import vecmat, matmul, temporal_mean, subtract_mean, svd, tsqr_svd, transpose, eigen, cholesky, diag, polar, vandermonde, conj, inv, flip, matmulp, vandermondeTime, vector_norm +from ..utils import cr_nvtx as cr, cr_start, cr_stop + +@cr('RES.run') +def run(Phi, muReal, muImag, bJov, dt, delta, freq, f): + + # Treatment of the parameters needed for future matrices + mu = muReal + 1j * muImag + Amplitude = bJov + # Frequency = np.angle(mu) / (dt * param) + # GrowthRate = np.log(np.abs(mu)) / dt + # Omega = GrowthRate + 1j * Frequency + Frequency = freq + GrowthRate = delta + Omega = GrowthRate + 1j * Frequency + + # Calculating Cholesky decomposition for Q = I + Matrix = transpose(vecmat(bJov, transpose(Phi))) + for ii in range(len(Matrix[0,:])): + Matrix[:,ii] = Matrix[:,ii] / vector_norm(Phi[:,ii]) + I = np.eye(len(mu)) + Lambda = diag(Omega) + Qhat = matmul(transpose(conj(Matrix)), Matrix) + Fhat = cholesky(Qhat) + + # Calculating modes for desired frequency + # H = matmul(matmul(Fhat, inv(-1j * f * I - Lambda)), np.linalg.pinv(Fhat)) + H = matmul(matmul(Fhat, inv(-1j * f * I - Lambda)), inv(Fhat)) + U, S, VT = svd(H) + V = transpose(conj(VT)) + + # URes = matmul(matmul(Matrix, np.linalg.pinv(Fhat)),U) + # VRes = matmul(matmul(Matrix, np.linalg.pinv(Fhat)),V) + + URes = matmul(matmul(Matrix, inv(Fhat)),U) + VRes = matmul(matmul(Matrix, inv(Fhat)),V) + + U_abs = np.abs(URes) + V_abs = np.abs(VRes) + # U_abs = np.abs(np.sqrt(URes**2)**2) + # V_abs = np.abs(np.sqrt(VRes**2)**2) + U_norm = np.zeros_like(U_abs) + V_norm = np.zeros_like(V_abs) + for jj in range(len(U_abs[0])): + U_norm[:,jj] = (U_abs[:,jj] - np.min(U_abs[:,jj])) / (np.max(U_abs[:,jj]) - np.min(U_abs[:,jj])) + V_norm[:,jj] = (V_abs[:,jj] - np.min(V_abs[:,jj])) / (np.max(V_abs[:,jj]) - np.min(V_abs[:,jj])) + # U_norm = (U_abs - np.min(U_abs)) / (np.max(U_abs) - np.min(U_abs)) + # V_norm = (V_abs - np.min(V_abs)) / (np.max(V_abs) - np.min(V_abs)) + + return U_norm, S, V_norm + +@cr('RES.run_mine') +def run_mine(Phi, muReal, muImag, delta, frequency, dt, f): + p = cp if type(Phi) is cp.ndarray else np + + # Treatment of the parameters needed for future matrices + GrowthRate, Frequency = delta, frequency + # Frequency = Frequency / (2 * p.pi) + Omega = GrowthRate + 1j * Frequency + + # Normalization of the modes + # for ii in range(len(Phi[0,:])): + # Phi[:,ii] = Phi[:,ii] / vector_norm(Phi[:,ii]) + I = p.eye(len(Omega)) + Lambda = diag(Omega) + + # Calculating modes for desired frequency + H = inv(-1j * f * I - Lambda) + U, S, VT = svd(H) + V = transpose(conj(VT)) + + U_res = matmul(Phi,U) + V_res = matmul(Phi,V) + + U_abs = p.abs(U_res) + V_abs = p.abs(V_res) + # U_norm = np.zeros_like(U_abs) + # V_norm = np.zeros_like(V_abs) + # for jj in range(len(U_abs[0])): + # U_norm[:,jj] = (U_abs[:,jj] - np.min(U_abs[:,jj])) / (np.max(U_abs[:,jj]) - np.min(U_abs[:,jj])) + # V_norm[:,jj] = (V_abs[:,jj] - np.min(V_abs[:,jj])) / (np.max(V_abs[:,jj]) - np.min(V_abs[:,jj])) + + return U_abs, S, V_abs diff --git a/pyLOM/__init__.py b/pyLOM/__init__.py index 2ceb83b8..6703cb63 100644 --- a/pyLOM/__init__.py +++ b/pyLOM/__init__.py @@ -20,7 +20,7 @@ from .utils.gpu import gpu_device # Import Low Order Models -from . import POD, PCA, DMD, SPOD, MANIFOLD, GPOD, LAMINE +from . import POD, PCA, DMD, SPOD, MANIFOLD, GPOD, RES # Import AI Models # The NN module overloads the memory when loaded diff --git a/pyLOM/vmmath/maths.py b/pyLOM/vmmath/maths.py index c049c4e7..e2a39d10 100644 --- a/pyLOM/vmmath/maths.py +++ b/pyLOM/vmmath/maths.py @@ -28,7 +28,7 @@ def transpose(A:np.ndarray) -> np.ndarray: p = cp if type(A) is cp.ndarray else np return p.transpose(A) -@cr('math.vector_norm') +@cr('math.vector_sum') def vector_sum(v:np.ndarray,start:int=0) -> float: r''' Sum of a vector @@ -308,4 +308,4 @@ def flip(A:np.ndarray) -> np.ndarray: np.ndarray: Flipped version of A (M,N) ''' p = cp if type(A) is cp.ndarray else np - return p.flip(A) \ No newline at end of file + return p.flip(A) From bc24f4dbf309cfc68f4cafa22420daebc8c1e4dd Mon Sep 17 00:00:00 2001 From: msilvestrec03 Date: Tue, 17 Mar 2026 12:37:48 +0100 Subject: [PATCH 03/27] resolvent utils and save_POD splited --- pyLOM/RES/utils.py | 34 +++++++++++++---- pyLOM/RES/wrapper.py | 68 ++++++++++++++++++--------------- pyLOM/inp_out/__init__.py | 2 +- pyLOM/inp_out/io_h5.py | 79 +++++++++++++++++++++++++++++++-------- 4 files changed, 129 insertions(+), 54 deletions(-) diff --git a/pyLOM/RES/utils.py b/pyLOM/RES/utils.py index 75bd44a6..41617c17 100644 --- a/pyLOM/RES/utils.py +++ b/pyLOM/RES/utils.py @@ -13,8 +13,9 @@ from .. import inp_out as io, PartitionTable from ..utils import cr_nvtx as cr, gpu_to_cpu + @cr('RES.extract_modes') -def extract_modes(U:np.ndarray,V:np.ndarray,ivar:int,npoints:int,modes:list=[],reshape:bool=True): +def extract_modes(U:np.ndarray,V:np.ndarray,ivar:int,npoints:int,modes:list=[],reshape:bool=True,kind:str="abs"): r''' When performing POD of several variables simultaneously, this function separates the spatial modes from each of the variables. @@ -25,6 +26,11 @@ def extract_modes(U:np.ndarray,V:np.ndarray,ivar:int,npoints:int,modes:list=[],r npoints (int): number of points in the domain per variable modes (list, optional): list containing the id of the modes to separate (default ``[]``). reshape (bool, optional): if true the output will be given as (len(modes)*npoints,) if not it the result will be (npoints, len(modes)) (default `` True ``) + kind (str): The aspect of the number to return. + Options are: + * 'real': Returns the real part. + * 'imag': Returns the imaginary part. + * 'abs': Returns the magnitude (absolute value). Returns: np.ndarray: modes of the variable ivar @@ -34,14 +40,26 @@ def extract_modes(U:np.ndarray,V:np.ndarray,ivar:int,npoints:int,modes:list=[],r # Define modes to extract if len(modes) == 0: modes = p.arange(1,U.shape[1]+1,dtype=p.int32) # Allocate output array - out_U =p.zeros((npoints,len(modes)),U.dtype) - out_V =p.zeros((npoints,len(modes)),U.dtype) - for i,m in enumerate(modes): - out_U[:,i] = U[ivar-1:nvars*npoints:nvars,m-1] - out_V[:,i] = V[ivar-1:nvars*npoints:nvars,m-1] + out_U =p.zeros((npoints,len(modes)),p.double if U.dtype == p.complex128 else p.float32) + out_V =p.zeros((npoints,len(modes)),p.double if V.dtype == p.complex128 else p.float32) + if kind == "real": + for i,m in enumerate(modes): + out_U[:,i] = (U[ivar-1:nvars*npoints:nvars,m-1].real) + out_V[:,i] = (V[ivar-1:nvars*npoints:nvars,m-1].real) + elif kind == "imag": + for i,m in enumerate(modes): + out_U[:,i] = (U[ivar-1:nvars*npoints:nvars,m-1].imag) + out_V[:,i] = (V[ivar-1:nvars*npoints:nvars,m-1].imag) + elif kind == "abs": + for i,m in enumerate(modes): + out_U[:,i] = (U[ivar-1:nvars*npoints:nvars,m-1].abs) + out_V[:,i] = (V[ivar-1:nvars*npoints:nvars,m-1].abs) + else: + raise ValueError("kind must be: real, imag, abs") # Return reshaped output return out_U.reshape((len(modes)*npoints,),order='C') if reshape else out_U, out_V.reshape((len(modes)*npoints,),order='C') if reshape else out_V + @cr('RES.save') def save(fname:str,U:np.ndarray,S:np.ndarray,V:np.ndarray,ptable:PartitionTable,nvars:int=1,pointData:bool=True,mode:str='w'): r''' @@ -58,7 +76,7 @@ def save(fname:str,U:np.ndarray,S:np.ndarray,V:np.ndarray,ptable:PartitionTable, mode (str, optional): mode in which the HDF5 file is opened, 'w' stands for write mode and 'a' stands for append mode. Write mode will overwrite the file and append mode will add the informaiton at the end of the current file, choose with great care what to do in your case (default ``w``). ''' - io.h5_save_POD(fname,gpu_to_cpu(U),gpu_to_cpu(S),gpu_to_cpu(V),ptable,nvars=nvars,pointData=pointData,mode=mode) + io.h5_save_RES(fname,gpu_to_cpu(U),gpu_to_cpu(S),gpu_to_cpu(V),ptable,nvars=nvars,pointData=pointData,mode=mode) @cr('RES.load') @@ -76,4 +94,4 @@ def load(fname:str,vars:list=['U','S','V'],nmod:int=-1,ptable:PartitionTable=Non nmod (int, optional): number of modes to load. By default it will load all the saved modes (default, ``-1``) ptable (PartitionTable, optional): partition table to use when loading the data (default ``None``). ''' - return io.h5_load_POD(fname,vars,nmod,ptable) \ No newline at end of file + return io.h5_load_RES(fname,vars,nmod,ptable) \ No newline at end of file diff --git a/pyLOM/RES/wrapper.py b/pyLOM/RES/wrapper.py index 000eb3ba..446edfb8 100644 --- a/pyLOM/RES/wrapper.py +++ b/pyLOM/RES/wrapper.py @@ -14,7 +14,8 @@ from ..utils import cr_nvtx as cr, cr_start, cr_stop @cr('RES.run') -def run(Phi, muReal, muImag, bJov, dt, delta, freq, f): +def run(Phi, muReal, muImag, bJov, dt, delta, freq, f, Q=None): + p = cp if type(Phi) is cp.ndarray else np # Treatment of the parameters needed for future matrices mu = muReal + 1j * muImag @@ -25,41 +26,48 @@ def run(Phi, muReal, muImag, bJov, dt, delta, freq, f): Frequency = freq GrowthRate = delta Omega = GrowthRate + 1j * Frequency - - # Calculating Cholesky decomposition for Q = I + # Normalization of modes (?) Matrix = transpose(vecmat(bJov, transpose(Phi))) for ii in range(len(Matrix[0,:])): Matrix[:,ii] = Matrix[:,ii] / vector_norm(Phi[:,ii]) - I = np.eye(len(mu)) + # Calculating Cholesky decomposition for Q + I = p.eye(len(mu)) Lambda = diag(Omega) - Qhat = matmul(transpose(conj(Matrix)), Matrix) - Fhat = cholesky(Qhat) - # Calculating modes for desired frequency - # H = matmul(matmul(Fhat, inv(-1j * f * I - Lambda)), np.linalg.pinv(Fhat)) - H = matmul(matmul(Fhat, inv(-1j * f * I - Lambda)), inv(Fhat)) - U, S, VT = svd(H) - V = transpose(conj(VT)) + if Q == None: + H = inv(-1j * f * I - Lambda) + U, S, VT = svd(H) + V = transpose(conj(VT)) + + U_res = matmul(Phi,U) + V_res = matmul(Phi,V) + + else: + Qhat = matmul(matmul(transpose(conj(Matrix)), Q), Matrix) + # Qhat = matmul(transpose(conj(Matrix)), Matrix) + Fhat = cholesky(Qhat) - # URes = matmul(matmul(Matrix, np.linalg.pinv(Fhat)),U) - # VRes = matmul(matmul(Matrix, np.linalg.pinv(Fhat)),V) + # Calculating modes for desired frequency + # H = matmul(matmul(Fhat, inv(-1j * f * I - Lambda)), np.linalg.pinv(Fhat)) + H = matmul(matmul(Fhat, inv(-1j * f * I - Lambda)), inv(Fhat)) + U, S, VT = svd(H) + V = transpose(conj(VT)) - URes = matmul(matmul(Matrix, inv(Fhat)),U) - VRes = matmul(matmul(Matrix, inv(Fhat)),V) + # URes = matmul(matmul(Matrix, np.linalg.pinv(Fhat)),U) + # VRes = matmul(matmul(Matrix, np.linalg.pinv(Fhat)),V) - U_abs = np.abs(URes) - V_abs = np.abs(VRes) - # U_abs = np.abs(np.sqrt(URes**2)**2) - # V_abs = np.abs(np.sqrt(VRes**2)**2) - U_norm = np.zeros_like(U_abs) - V_norm = np.zeros_like(V_abs) - for jj in range(len(U_abs[0])): - U_norm[:,jj] = (U_abs[:,jj] - np.min(U_abs[:,jj])) / (np.max(U_abs[:,jj]) - np.min(U_abs[:,jj])) - V_norm[:,jj] = (V_abs[:,jj] - np.min(V_abs[:,jj])) / (np.max(V_abs[:,jj]) - np.min(V_abs[:,jj])) - # U_norm = (U_abs - np.min(U_abs)) / (np.max(U_abs) - np.min(U_abs)) - # V_norm = (V_abs - np.min(V_abs)) / (np.max(V_abs) - np.min(V_abs)) + U_res = matmul(matmul(Matrix, inv(Fhat)),U) + V_res = matmul(matmul(Matrix, inv(Fhat)),V) + + # U_abs = np.abs(URes) + # V_abs = np.abs(VRes) + # U_norm = np.zeros_like(U_abs) + # V_norm = np.zeros_like(V_abs) + # for jj in range(len(U_abs[0])): + # U_norm[:,jj] = (U_abs[:,jj] - np.min(U_abs[:,jj])) / (np.max(U_abs[:,jj]) - np.min(U_abs[:,jj])) + # V_norm[:,jj] = (V_abs[:,jj] - np.min(V_abs[:,jj])) / (np.max(V_abs[:,jj]) - np.min(V_abs[:,jj])) - return U_norm, S, V_norm + return U_res, S, V_res @cr('RES.run_mine') def run_mine(Phi, muReal, muImag, delta, frequency, dt, f): @@ -84,12 +92,12 @@ def run_mine(Phi, muReal, muImag, delta, frequency, dt, f): U_res = matmul(Phi,U) V_res = matmul(Phi,V) - U_abs = p.abs(U_res) - V_abs = p.abs(V_res) + # U_abs = p.abs(U_res) + # V_abs = p.abs(V_res) # U_norm = np.zeros_like(U_abs) # V_norm = np.zeros_like(V_abs) # for jj in range(len(U_abs[0])): # U_norm[:,jj] = (U_abs[:,jj] - np.min(U_abs[:,jj])) / (np.max(U_abs[:,jj]) - np.min(U_abs[:,jj])) # V_norm[:,jj] = (V_abs[:,jj] - np.min(V_abs[:,jj])) / (np.max(V_abs[:,jj]) - np.min(V_abs[:,jj])) - return U_abs, S, V_abs + return U_res, S, V_res \ No newline at end of file diff --git a/pyLOM/inp_out/__init__.py b/pyLOM/inp_out/__init__.py index 266b14c6..957843c3 100644 --- a/pyLOM/inp_out/__init__.py +++ b/pyLOM/inp_out/__init__.py @@ -9,7 +9,7 @@ # Pickle and HDF5 exchange format from .io_pkl import pkl_load, pkl_save from .io_h5 import h5_load_dset, h5_save_dset, h5_append_dset, h5_save_mesh, h5_load_mesh -from .io_h5 import h5_save_QR, h5_load_QR, h5_save_POD, h5_load_POD, h5_save_DMD, h5_load_DMD, h5_save_SPOD, h5_load_SPOD +from .io_h5 import h5_save_QR, h5_load_QR, h5_save_POD, h5_load_POD, h5_save_DMD, h5_load_DMD, h5_save_SPOD, h5_load_SPOD, h5_save_RES, h5_load_RES from .io_h5 import h5_save_VAE, h5_create_compressed, h5_flush_compressed, h5_load_compressed, h5_save_graph_serial, h5_load_graph_serial # VTK HDF5 3D format diff --git a/pyLOM/inp_out/io_h5.py b/pyLOM/inp_out/io_h5.py index 864b4f72..169b8138 100644 --- a/pyLOM/inp_out/io_h5.py +++ b/pyLOM/inp_out/io_h5.py @@ -848,21 +848,7 @@ def h5_load_QR(fname,vars,ptable=None): file.close() return varList -@cr('h5IO.save_POD') -def h5_save_POD(fname,U,S,V,ptable,nvars=1,pointData=True,mode='w'): - ''' - Store POD variables into an HDF5 file. - Can be appended to another HDF by setting the - mode to 'a'. Then no partition table will be saved. - ''' - file = h5py.File(fname,mode,driver='mpio',comm=MPI_COMM) if not MPI_SIZE == 1 else h5py.File(fname,mode) - # Store attributes and partition table - if not mode == 'a': - file.attrs['Version'] = PYLOM_H5_VERSION - # Store partition table - h5_save_partition(file,ptable) - # Now create a POD group - group = file.create_group('POD') +def h5_save_usv(U,S,V,ptable,nvars,pointData,group): # Create the datasets for U, S and V group.create_dataset('pointData',(1,),dtype='u1',data=pointData) group.create_dataset('n_variables',(1,),dtype='u1',data=nvars) @@ -878,6 +864,24 @@ def h5_save_POD(fname,U,S,V,ptable,nvars=1,pointData=True,mode='w'): # Store U in parallel istart, iend = ptable.partition_bounds(MPI_RANK,ndim=nvars,points=pointData) if dsetU is not None: dsetU[istart:iend,:] = U + +@cr('h5IO.save_POD') +def h5_save_POD(fname,U,S,V,ptable,nvars=1,pointData=True,mode='w'): + ''' + Store POD variables into an HDF5 file. + Can be appended to another HDF by setting the + mode to 'a'. Then no partition table will be saved. + ''' + file = h5py.File(fname,mode,driver='mpio',comm=MPI_COMM) if not MPI_SIZE == 1 else h5py.File(fname,mode) + # Store attributes and partition table + if not mode == 'a': + file.attrs['Version'] = PYLOM_H5_VERSION + # Store partition table + h5_save_partition(file,ptable) + # Now create a POD group + group = file.create_group('POD') + # Create and store de datsets + h5_save_usv(U,S,V,ptable,nvars,pointData,group) file.close() @cr('h5IO.load_POD') @@ -1047,6 +1051,51 @@ def h5_load_SPOD(fname,vars,nmod,ptable=None): file.close() return varList +@cr('h5IO.save_RES') +def h5_save_RES(fname,U,S,V,ptable,nvars=1,pointData=True,mode='w'): + ''' + Store RES variables into an HDF5 file. + Can be appended to another HDF by setting the + mode to 'a'. Then no partition table will be saved. + ''' + file = h5py.File(fname,mode,driver='mpio',comm=MPI_COMM) if not MPI_SIZE == 1 else h5py.File(fname,mode) + # Store attributes and partition table + if not mode == 'a': + file.attrs['Version'] = PYLOM_H5_VERSION + # Store partition table + h5_save_partition(file,ptable) + # Now create a POD group + group = file.create_group('RES') + # Create and store de datsets + h5_save_usv(U,S,V,ptable,nvars,pointData,group) + file.close() + +@cr('h5IO.load_RES') +def h5_load_RES(fname,vars,nmod,ptable=None): + ''' + Load RES variables from an HDF5 file. + ''' + file = h5py.File(fname,'r',driver='mpio',comm=MPI_COMM) if not MPI_SIZE == 1 else h5py.File(fname,'r') + # Check the file version + version = tuple(file.attrs['Version']) + if not version == PYLOM_H5_VERSION: + raiseError('File version <%s> not matching the tool version <%s>!'%(str(file.attrs['Version']),str(PYLOM_H5_VERSION))) + # Read the requested variables S, V + varList = [] + if 'U' in vars: + # Check if we need to read the partition table + if ptable is None: ptable = h5_load_partition(file) + # Read + nvars = int(file['RES']['n_variables'][0]) + point = bool(file['RES']['pointData'][0]) + istart, iend = ptable.partition_bounds(MPI_RANK,ndim=nvars,points=point) + varList.append(np.array(file['RES']['U'][istart:iend,:]) if nmod < 0 else np.array(file['RES']['U'][istart:iend,:nmod])) + if 'S' in vars: varList.append( np.array(file['RES']['S'][:]) if nmod < 0 else np.array(file['RES']['S'][:nmod]) ) + if 'V' in vars: varList.append( np.array(file['RES']['V'][:,:]) if nmod < 0 else np.array(file['RES']['V'][:nmod,:]) ) + # Return + file.close() + return varList + @cr('io.create_compressed') def h5_create_compressed(fname:str,basedir:str,r:int,nmod:int,nvars:int,nlayers:int,conv_chan:int,kernel:int,nAEsG:int,nptxAE:int,dtype:np.dtype): r''' From ffe5283be30895ec23cbc7739fae8de264d230d9 Mon Sep 17 00:00:00 2001 From: msilvestrec03 Date: Fri, 27 Mar 2026 14:22:20 +0100 Subject: [PATCH 04/27] parallel resolvent --- pyLOM/RES/__init__.py | 2 +- pyLOM/RES/utils.py | 4 +- pyLOM/RES/wrapper.py | 119 +++++++++++++++++------------------------ pyLOM/inp_out/io_h5.py | 28 +++++++--- pyLOM/vmmath/maths.py | 18 +++++++ 5 files changed, 93 insertions(+), 78 deletions(-) diff --git a/pyLOM/RES/__init__.py b/pyLOM/RES/__init__.py index 78f57a32..d4c769dc 100644 --- a/pyLOM/RES/__init__.py +++ b/pyLOM/RES/__init__.py @@ -7,7 +7,7 @@ # Last rev: 20/02/2026 # Functions coming from Resolvent -from .wrapper import run, run_mine +from .wrapper import run from .utils import extract_modes, save, load diff --git a/pyLOM/RES/utils.py b/pyLOM/RES/utils.py index 41617c17..61773ef6 100644 --- a/pyLOM/RES/utils.py +++ b/pyLOM/RES/utils.py @@ -52,8 +52,8 @@ def extract_modes(U:np.ndarray,V:np.ndarray,ivar:int,npoints:int,modes:list=[],r out_V[:,i] = (V[ivar-1:nvars*npoints:nvars,m-1].imag) elif kind == "abs": for i,m in enumerate(modes): - out_U[:,i] = (U[ivar-1:nvars*npoints:nvars,m-1].abs) - out_V[:,i] = (V[ivar-1:nvars*npoints:nvars,m-1].abs) + out_U[:,i] = (abs(U[ivar-1:nvars*npoints:nvars,m-1])) + out_V[:,i] = (abs(V[ivar-1:nvars*npoints:nvars,m-1])) else: raise ValueError("kind must be: real, imag, abs") # Return reshaped output diff --git a/pyLOM/RES/wrapper.py b/pyLOM/RES/wrapper.py index 446edfb8..7c5bf944 100644 --- a/pyLOM/RES/wrapper.py +++ b/pyLOM/RES/wrapper.py @@ -4,7 +4,7 @@ # # Python interface for DMD. # -# Last rev: 20/02/2026 +# Last rev: 27/03/2026 from __future__ import print_function import numpy as np @@ -17,87 +17,68 @@ def run(Phi, muReal, muImag, bJov, dt, delta, freq, f, Q=None): p = cp if type(Phi) is cp.ndarray else np - # Treatment of the parameters needed for future matrices - mu = muReal + 1j * muImag - Amplitude = bJov - # Frequency = np.angle(mu) / (dt * param) - # GrowthRate = np.log(np.abs(mu)) / dt - # Omega = GrowthRate + 1j * Frequency - Frequency = freq - GrowthRate = delta - Omega = GrowthRate + 1j * Frequency # Normalization of modes (?) - Matrix = transpose(vecmat(bJov, transpose(Phi))) - for ii in range(len(Matrix[0,:])): - Matrix[:,ii] = Matrix[:,ii] / vector_norm(Phi[:,ii]) - # Calculating Cholesky decomposition for Q - I = p.eye(len(mu)) - Lambda = diag(Omega) - - if Q == None: - H = inv(-1j * f * I - Lambda) - U, S, VT = svd(H) + # Matrix = transpose(vecmat(bJov, transpose(Phi))) + # for ii in range(len(Matrix[0,:])): + # Matrix[:,ii] = Matrix[:,ii] / vector_norm(Phi[:,ii]) + + Omega = delta + 1j * freq + H = 1 / (-1j * f - Omega) + + if Q is None: + + # Proper treatment (?) + # Qhat = matmulp(transpose(conj(Phi)), Phi) + # Fhat = cholesky(Qhat) + # Fhat_inv = inv(Fhat) + # Hhat = matmul(Fhat, vecmat(H, Fhat_inv)) + # U, S, VT = svd(Hhat) + + U, S, VT = svd(diag(H)) V = transpose(conj(VT)) + # U_res = matmul(matmul(Phi, Fhat_inv),U) + # V_res = matmul(matmul(Phi, Fhat_inv),V) + U_res = matmul(Phi,U) V_res = matmul(Phi,V) + else: - Qhat = matmul(matmul(transpose(conj(Matrix)), Q), Matrix) - # Qhat = matmul(transpose(conj(Matrix)), Matrix) + + Qhat = matmulp(transpose(conj(Phi)), vecmat(Q, Phi)) Fhat = cholesky(Qhat) + Fhat_inv = inv(Fhat) - # Calculating modes for desired frequency - # H = matmul(matmul(Fhat, inv(-1j * f * I - Lambda)), np.linalg.pinv(Fhat)) - H = matmul(matmul(Fhat, inv(-1j * f * I - Lambda)), inv(Fhat)) - U, S, VT = svd(H) + Hhat = matmul(Fhat, vecmat(H, Fhat_inv)) + U, S, VT = svd(Hhat) V = transpose(conj(VT)) - # URes = matmul(matmul(Matrix, np.linalg.pinv(Fhat)),U) - # VRes = matmul(matmul(Matrix, np.linalg.pinv(Fhat)),V) + U_res = matmul(matmul(Phi, Fhat_inv),U) + V_res = matmul(matmul(Phi, Fhat_inv),V) + + return U_res, S, V_res - U_res = matmul(matmul(Matrix, inv(Fhat)),U) - V_res = matmul(matmul(Matrix, inv(Fhat)),V) - # U_abs = np.abs(URes) - # V_abs = np.abs(VRes) - # U_norm = np.zeros_like(U_abs) - # V_norm = np.zeros_like(V_abs) - # for jj in range(len(U_abs[0])): - # U_norm[:,jj] = (U_abs[:,jj] - np.min(U_abs[:,jj])) / (np.max(U_abs[:,jj]) - np.min(U_abs[:,jj])) - # V_norm[:,jj] = (V_abs[:,jj] - np.min(V_abs[:,jj])) / (np.max(V_abs[:,jj]) - np.min(V_abs[:,jj])) +@cr('RES.weighting') +def weighting(m, v_dims = 1, gamma=1.4, cp=1004.0, T=1, rho=None, compressible=False): # Not well parallelized + p = cp if type(rho) is cp.ndarray else np - return U_res, S, V_res + R = cp * (1 - gamma) + c = p.sqrt(gamma * R * T) # How do I find T? -@cr('RES.run_mine') -def run_mine(Phi, muReal, muImag, delta, frequency, dt, f): - p = cp if type(Phi) is cp.ndarray else np + V = np.array([]) + # Calcular els dV + dV = np.ones(len(m.xyz[:,0])) # How do I calculate the volumes? + + # Calcular Q + if compressible == False: + for ii in range(v_dims): + V = np.append(V, dV) + Q = 0.5 * rho * V - # Treatment of the parameters needed for future matrices - GrowthRate, Frequency = delta, frequency - # Frequency = Frequency / (2 * p.pi) - Omega = GrowthRate + 1j * Frequency - - # Normalization of the modes - # for ii in range(len(Phi[0,:])): - # Phi[:,ii] = Phi[:,ii] / vector_norm(Phi[:,ii]) - I = p.eye(len(Omega)) - Lambda = diag(Omega) - - # Calculating modes for desired frequency - H = inv(-1j * f * I - Lambda) - U, S, VT = svd(H) - V = transpose(conj(VT)) - - U_res = matmul(Phi,U) - V_res = matmul(Phi,V) - - # U_abs = p.abs(U_res) - # V_abs = p.abs(V_res) - # U_norm = np.zeros_like(U_abs) - # V_norm = np.zeros_like(V_abs) - # for jj in range(len(U_abs[0])): - # U_norm[:,jj] = (U_abs[:,jj] - np.min(U_abs[:,jj])) / (np.max(U_abs[:,jj]) - np.min(U_abs[:,jj])) - # V_norm[:,jj] = (V_abs[:,jj] - np.min(V_abs[:,jj])) / (np.max(V_abs[:,jj]) - np.min(V_abs[:,jj])) - - return U_res, S, V_res \ No newline at end of file + if compressible == True: + for ii in range(v_dims): + V = np.append(V, rho * dV) + V = np.append(V, c / (gamma * rho) * dV) + Q = 0.5 * V \ No newline at end of file diff --git a/pyLOM/inp_out/io_h5.py b/pyLOM/inp_out/io_h5.py index 169b8138..a89f9448 100644 --- a/pyLOM/inp_out/io_h5.py +++ b/pyLOM/inp_out/io_h5.py @@ -848,22 +848,31 @@ def h5_load_QR(fname,vars,ptable=None): file.close() return varList -def h5_save_usv(U,S,V,ptable,nvars,pointData,group): +def h5_save_usv(U,S,V,ptable,nvars,pointData,group,kind): # Create the datasets for U, S and V group.create_dataset('pointData',(1,),dtype='u1',data=pointData) group.create_dataset('n_variables',(1,),dtype='u1',data=nvars) Usize = (mpi_reduce(U.shape[0],op='sum',all=True),U.shape[1]) if U is not None else None dsetU = group.create_dataset('U',Usize,dtype=U.dtype) if U is not None else None dsetS = group.create_dataset('S',S.shape,dtype=S.dtype) if S is not None else None - dsetV = group.create_dataset('V',V.shape,dtype=V.dtype) if V is not None else None + if kind == 'POD': + dsetV = group.create_dataset('V',V.shape,dtype=V.dtype) if V is not None else None + elif kind == 'RES': + Vsize = (mpi_reduce(V.shape[0],op='sum',all=True),V.shape[1]) if V is not None else None + dsetV = group.create_dataset('V',Vsize,dtype=V.dtype) if V is not None else None + else: + raise ValueError("kind must be: POD, RES") # Store S and U that are repeated across the ranks # So it is enough that one rank stores them if is_rank_or_serial(0): if dsetS is not None: dsetS[:] = S - if dsetV is not None: dsetV[:] = V + if kind == 'POD': + if dsetV is not None: dsetV[:] = V # Store U in parallel istart, iend = ptable.partition_bounds(MPI_RANK,ndim=nvars,points=pointData) if dsetU is not None: dsetU[istart:iend,:] = U + if kind == 'RES': + if dsetV is not None: dsetV[istart:iend,:] = V @cr('h5IO.save_POD') def h5_save_POD(fname,U,S,V,ptable,nvars=1,pointData=True,mode='w'): @@ -881,7 +890,7 @@ def h5_save_POD(fname,U,S,V,ptable,nvars=1,pointData=True,mode='w'): # Now create a POD group group = file.create_group('POD') # Create and store de datsets - h5_save_usv(U,S,V,ptable,nvars,pointData,group) + h5_save_usv(U,S,V,ptable,nvars,pointData,group,kind='POD') file.close() @cr('h5IO.load_POD') @@ -1067,7 +1076,7 @@ def h5_save_RES(fname,U,S,V,ptable,nvars=1,pointData=True,mode='w'): # Now create a POD group group = file.create_group('RES') # Create and store de datsets - h5_save_usv(U,S,V,ptable,nvars,pointData,group) + h5_save_usv(U,S,V,ptable,nvars,pointData,group,kind='RES') file.close() @cr('h5IO.load_RES') @@ -1091,7 +1100,14 @@ def h5_load_RES(fname,vars,nmod,ptable=None): istart, iend = ptable.partition_bounds(MPI_RANK,ndim=nvars,points=point) varList.append(np.array(file['RES']['U'][istart:iend,:]) if nmod < 0 else np.array(file['RES']['U'][istart:iend,:nmod])) if 'S' in vars: varList.append( np.array(file['RES']['S'][:]) if nmod < 0 else np.array(file['RES']['S'][:nmod]) ) - if 'V' in vars: varList.append( np.array(file['RES']['V'][:,:]) if nmod < 0 else np.array(file['RES']['V'][:nmod,:]) ) + if 'V' in vars: + # Check if we need to read the partition table + if ptable is None: ptable = h5_load_partition(file) + # Read + nvars = int(file['RES']['n_variables'][0]) + point = bool(file['RES']['pointData'][0]) + istart, iend = ptable.partition_bounds(MPI_RANK,ndim=nvars,points=point) + varList.append(np.array(file['RES']['V'][istart:iend,:]) if nmod < 0 else np.array(file['RES']['V'][istart:iend,:nmod])) # Return file.close() return varList diff --git a/pyLOM/vmmath/maths.py b/pyLOM/vmmath/maths.py index e2a39d10..f9ad6cbb 100644 --- a/pyLOM/vmmath/maths.py +++ b/pyLOM/vmmath/maths.py @@ -309,3 +309,21 @@ def flip(A:np.ndarray) -> np.ndarray: ''' p = cp if type(A) is cp.ndarray else np return p.flip(A) + +@cr('math.abs') +def abs(A:np.ndarray) -> np.ndarray: + r''' + Returns the pointwise absolute value of complex A + + .. warning:: + This function is not implemented in the + compiled layer and will raise an error if used + + Args: + A (np.ndarray): Matrix A (M,N) + + Result: + np.ndarray: Absolute value version of A (M,N) + ''' + p = cp if type(A) is cp.ndarray else np + return p.abs(A) From 7de98e29d6da824054c440188e705de22e2b46f6 Mon Sep 17 00:00:00 2001 From: msilvestrec03 Date: Wed, 29 Apr 2026 15:38:26 +0200 Subject: [PATCH 05/27] plots RES --- pyLOM/RES/plots.py | 71 ++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 71 insertions(+) create mode 100644 pyLOM/RES/plots.py diff --git a/pyLOM/RES/plots.py b/pyLOM/RES/plots.py new file mode 100644 index 00000000..fe33594a --- /dev/null +++ b/pyLOM/RES/plots.py @@ -0,0 +1,71 @@ +#!/usr/bin/env python +# +# pyLOM - Python Low Order Modeling. +# +# Resolvent analysis plotting utilities. +# +# Last rev: 08/04/2026 +from __future__ import print_function, division + +import numpy as np +import matplotlib.pyplot as plt + +from ..vmmath import fft +from ..utils import gpu_to_cpu + + +def plotEnergy(S,fig=None,ax=None): + r''' + Plot the cummulative energy of a set of modes + + Args: + S (np.ndarray): array containing the singular values from the RES modes + fig (plt.figure, optional): figure object in which the plot will be done (default: ``[]``) + axs (plt.axes, optional): axes object in which the plot will be done (default: ``[]``) + + Returns: + [plt.figure, plt.axes]: figure and axes objects of the plot + ''' + S = gpu_to_cpu(S) + # Build figure and axes + if fig is None: + fig = plt.figure(figsize=(8,6),dpi=100) + if ax is None: + ax = fig.add_subplot(1,1,1) + # Plot accumulated residual + ax.plot(np.arange(1,S.shape[0]+1),np.cumsum(S**2)/np.sum(S**2),'bo') + # Set labels + ax.set_ylabel(r'Energy') + ax.set_xlabel(r'Modes') + # Return + return fig, ax + +def plotEvW(S, omega, modes, fig=None,ax=None): + r''' + Plot the cummulative energy of a set of modes + + Args: + S (np.ndarray): array containing the energy gains from the RES modes (columns) at diferent frequencies (rows) + omega (np.ndarray): array containing the frequencies from the respective energy gains + fig (plt.figure, optional): figure object in which the plot will be done (default: ``[]``) + axs (plt.axes, optional): axes object in which the plot will be done (default: ``[]``) + + Returns: + [plt.figure, plt.axes]: figure and axes objects of the plot + ''' + S = gpu_to_cpu(S) + omega = gpu_to_cpu(omega) + # Build figure and axes + if fig is None: + fig = plt.figure(figsize=(8,6),dpi=100) + if ax is None: + ax = fig.add_subplot(1,1,1) + # Plot energy gains + for ii in modes: + ax.plot(omega, S[:,ii], 'o-', label=f'mode {ii}') + # Set labels + ax.set_ylabel(r'Energy') + ax.set_xlabel(r'Frequencies') + ax.legend() + # Return + return fig, ax \ No newline at end of file From 0e673ff46b9d538cac68687ad3ec6ebe76732a9e Mon Sep 17 00:00:00 2001 From: msilvestrec03 Date: Wed, 29 Apr 2026 16:05:55 +0200 Subject: [PATCH 06/27] merge --- pyLOM/RES/__init__.py | 1 + pyLOM/RES/wrapper.py | 34 +++++----------------------------- pyLOM/vmmath/__init__.py | 2 +- 3 files changed, 7 insertions(+), 30 deletions(-) diff --git a/pyLOM/RES/__init__.py b/pyLOM/RES/__init__.py index d4c769dc..fbc53fb7 100644 --- a/pyLOM/RES/__init__.py +++ b/pyLOM/RES/__init__.py @@ -9,6 +9,7 @@ # Functions coming from Resolvent from .wrapper import run from .utils import extract_modes, save, load +from .plots import plotEnergy, plotEvW del wrapper diff --git a/pyLOM/RES/wrapper.py b/pyLOM/RES/wrapper.py index 7c5bf944..8b1ebed0 100644 --- a/pyLOM/RES/wrapper.py +++ b/pyLOM/RES/wrapper.py @@ -10,11 +10,11 @@ import numpy as np from ..utils.gpu import cp -from ..vmmath import vecmat, matmul, temporal_mean, subtract_mean, svd, tsqr_svd, transpose, eigen, cholesky, diag, polar, vandermonde, conj, inv, flip, matmulp, vandermondeTime, vector_norm +from ..vmmath import vecmat, matmul, temporal_mean, subtract_mean, svd, tsqr_svd, transpose, eigen, cholesky, diag, polar, vandermonde, conj, inv, flip, matmulp, vandermondeTime, vector_norm, dagger from ..utils import cr_nvtx as cr, cr_start, cr_stop @cr('RES.run') -def run(Phi, muReal, muImag, bJov, dt, delta, freq, f, Q=None): +def run(Phi, delta, freq, f, Q=None): p = cp if type(Phi) is cp.ndarray else np # Normalization of modes (?) @@ -35,7 +35,7 @@ def run(Phi, muReal, muImag, bJov, dt, delta, freq, f, Q=None): # U, S, VT = svd(Hhat) U, S, VT = svd(diag(H)) - V = transpose(conj(VT)) + V = dagger(VT) # U_res = matmul(matmul(Phi, Fhat_inv),U) # V_res = matmul(matmul(Phi, Fhat_inv),V) @@ -46,39 +46,15 @@ def run(Phi, muReal, muImag, bJov, dt, delta, freq, f, Q=None): else: - Qhat = matmulp(transpose(conj(Phi)), vecmat(Q, Phi)) + Qhat = matmulp(dagger(Phi), vecmat(Q, Phi)) Fhat = cholesky(Qhat) Fhat_inv = inv(Fhat) Hhat = matmul(Fhat, vecmat(H, Fhat_inv)) U, S, VT = svd(Hhat) - V = transpose(conj(VT)) + V = dagger(VT) U_res = matmul(matmul(Phi, Fhat_inv),U) V_res = matmul(matmul(Phi, Fhat_inv),V) return U_res, S, V_res - - -@cr('RES.weighting') -def weighting(m, v_dims = 1, gamma=1.4, cp=1004.0, T=1, rho=None, compressible=False): # Not well parallelized - p = cp if type(rho) is cp.ndarray else np - - R = cp * (1 - gamma) - c = p.sqrt(gamma * R * T) # How do I find T? - - V = np.array([]) - # Calcular els dV - dV = np.ones(len(m.xyz[:,0])) # How do I calculate the volumes? - - # Calcular Q - if compressible == False: - for ii in range(v_dims): - V = np.append(V, dV) - Q = 0.5 * rho * V - - if compressible == True: - for ii in range(v_dims): - V = np.append(V, rho * dV) - V = np.append(V, c / (gamma * rho) * dV) - Q = 0.5 * V \ No newline at end of file diff --git a/pyLOM/vmmath/__init__.py b/pyLOM/vmmath/__init__.py index d5f1e0e7..3a75d568 100644 --- a/pyLOM/vmmath/__init__.py +++ b/pyLOM/vmmath/__init__.py @@ -7,7 +7,7 @@ # Last rev: 27/10/2021 # Vector matrix routines -from .maths import transpose, vector_sum, vector_norm, vector_mean, matmul, matmulp, vecmat, argsort, eigen, polar, cholesky, vandermonde, conj, diag, inv, flip, vandermondeTime +from .maths import transpose, vector_sum, vector_norm, vector_mean, matmul, matmulp, vecmat, argsort, eigen, polar, cholesky, vandermonde, conj, diag, inv, flip, vandermondeTime, dagger # Averaging routines from .averaging import temporal_mean, subtract_mean, temporal_variance, norm_variance # Truncation routines From 3ff04ff6e7bf26fca253bdca6d3583b5f3779254 Mon Sep 17 00:00:00 2001 From: msilvestrec03 Date: Thu, 30 Apr 2026 14:31:19 +0200 Subject: [PATCH 07/27] dagger --- pyLOM/vmmath/maths.py | 13 +++++++++++++ pyLOM/vmmath/maths.pyx | 27 ++++++++++++++++++++++++++- 2 files changed, 39 insertions(+), 1 deletion(-) diff --git a/pyLOM/vmmath/maths.py b/pyLOM/vmmath/maths.py index f9ad6cbb..e10dfe10 100644 --- a/pyLOM/vmmath/maths.py +++ b/pyLOM/vmmath/maths.py @@ -327,3 +327,16 @@ def abs(A:np.ndarray) -> np.ndarray: ''' p = cp if type(A) is cp.ndarray else np return p.abs(A) + +@cr('math.dagger') +def dagger(A:np.ndarray) -> np.ndarray: + r''' + Returns the dagger (conjugated transpose) of complex A + + Args: + A (np.ndarray): Matrix A (M,N) + + Result: + np.ndarray: Dagger of A (N,M) + ''' + return conj(transpose(A)) diff --git a/pyLOM/vmmath/maths.pyx b/pyLOM/vmmath/maths.pyx index 604f4f70..11670243 100644 --- a/pyLOM/vmmath/maths.pyx +++ b/pyLOM/vmmath/maths.pyx @@ -1186,4 +1186,29 @@ def flip(real[:,:] A): Result: np.ndarray: Flipped version of A (M,N) ''' - raiseError('Function not implemented in Cython!') \ No newline at end of file + raiseError('Function not implemented in Cython!') + +@cr('math.dagger') +@cython.initializedcheck(False) +@cython.boundscheck(False) # turn off bounds-checking for entire function +@cython.wraparound(False) # turn off negative index wrapping for entire function +@cython.nonecheck(False) +@cython.cdivision(True) # turn off zero division check +def dagger(real_full[:,:] A): + r''' + Dagger of matrix A + + Args: + A (np.ndarray): Matrix to be daggerd + + Results + np.ndarray: dagger matrix + ''' + if real_full is np.complex128_t: + return _zconj(_ztranspose(A)) + elif real_full is np.complex64_t: + return _cconj(_ctranspose(A)) + elif real_full is double: + return _dtranspose(A) + else: + return _stranspose(A) \ No newline at end of file From 74c43445569d414fa0866237da55b6390df41c94 Mon Sep 17 00:00:00 2001 From: msilvestrec03 Date: Tue, 12 May 2026 10:26:44 +0200 Subject: [PATCH 08/27] wrapper not done --- pyLOM/RES/wrapper.pyx | 42 ++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 42 insertions(+) create mode 100644 pyLOM/RES/wrapper.pyx diff --git a/pyLOM/RES/wrapper.pyx b/pyLOM/RES/wrapper.pyx new file mode 100644 index 00000000..16f581d9 --- /dev/null +++ b/pyLOM/RES/wrapper.pyx @@ -0,0 +1,42 @@ +#!/usr/bin/env cpython +# +# pyLOM - Python Low Order Modeling. +# +# Python interface for RES. +# +# Last rev: 30/04/2026 +from __future__ import print_function, division + +cimport cython +cimport numpy as np + +import numpy as np + +#from libc.complex cimport creal, cimag +cdef extern from "" nogil: + float complex I + # Decomposing complex values + float cimagf(float complex z) + float crealf(float complex z) + double cimag(double complex z) + double creal(double complex z) +cdef double complex J = 1j +from libc.stdlib cimport malloc, free +from libc.string cimport memcpy, memset +from libc.math cimport sqrt, log, atan2 +from ..vmmath.cfuncs cimport real, real_complex +from ..vmmath.cfuncs cimport +from ..vmmath.cfuncs cimport +from ..vmmath.cfuncs cimport +from ..vmmath.cfuncs cimport + +from ..utils.cr import cr, cr_start, cr_stop +from ..utils.errors import raiseError + + +## RES run method +@cython.boundscheck(False) # turn off bounds-checking for entire function +@cython.wraparound(False) # turn off negative index wrapping for entire function +@cython.nonecheck(False) +@cython.cdivision(True) # turn off zero division check +def _srun(float[:,:] Phi, float delta, float freq, float f, ): \ No newline at end of file From 4f470aaae87696d296430902f278e06158bd4d64 Mon Sep 17 00:00:00 2001 From: msilvestrec03 Date: Thu, 14 May 2026 15:09:17 +0200 Subject: [PATCH 09/27] compiled resolvent and new functions compiled --- pyLOM/RES/wrapper.py | 48 ++++------- pyLOM/RES/wrapper.pyx | 110 +++++++++++++++++++++-- pyLOM/vmmath/cfuncs.pxd | 12 +++ pyLOM/vmmath/maths.pyx | 135 ++++++++++++++++++++++++----- pyLOM/vmmath/src/vector_matrix.c | 144 +++++++++++++++++++++++++++++++ pyLOM/vmmath/src/vector_matrix.h | 12 +++ 6 files changed, 407 insertions(+), 54 deletions(-) diff --git a/pyLOM/RES/wrapper.py b/pyLOM/RES/wrapper.py index 8b1ebed0..5a20735d 100644 --- a/pyLOM/RES/wrapper.py +++ b/pyLOM/RES/wrapper.py @@ -10,7 +10,7 @@ import numpy as np from ..utils.gpu import cp -from ..vmmath import vecmat, matmul, temporal_mean, subtract_mean, svd, tsqr_svd, transpose, eigen, cholesky, diag, polar, vandermonde, conj, inv, flip, matmulp, vandermondeTime, vector_norm, dagger +from ..vmmath import vecmat, matmul, svd, cholesky, inv, matmulp, dagger from ..utils import cr_nvtx as cr, cr_start, cr_stop @cr('RES.run') @@ -26,35 +26,25 @@ def run(Phi, delta, freq, f, Q=None): H = 1 / (-1j * f - Omega) if Q is None: - - # Proper treatment (?) - # Qhat = matmulp(transpose(conj(Phi)), Phi) - # Fhat = cholesky(Qhat) - # Fhat_inv = inv(Fhat) - # Hhat = matmul(Fhat, vecmat(H, Fhat_inv)) - # U, S, VT = svd(Hhat) - - U, S, VT = svd(diag(H)) - V = dagger(VT) - - # U_res = matmul(matmul(Phi, Fhat_inv),U) - # V_res = matmul(matmul(Phi, Fhat_inv),V) - - U_res = matmul(Phi,U) - V_res = matmul(Phi,V) - - + Qhat = matmulp(dagger(Phi), Phi) + else: - Qhat = matmulp(dagger(Phi), vecmat(Q, Phi)) - Fhat = cholesky(Qhat) - Fhat_inv = inv(Fhat) - - Hhat = matmul(Fhat, vecmat(H, Fhat_inv)) - U, S, VT = svd(Hhat) - V = dagger(VT) - - U_res = matmul(matmul(Phi, Fhat_inv),U) - V_res = matmul(matmul(Phi, Fhat_inv),V) + + Fhat = cholesky(Qhat) + Fhat_inv = inv(Fhat) + + Hhat = matmul(Fhat, vecmat(H, Fhat_inv)) + U, S, VT = svd(Hhat) + V = dagger(VT) + + U_res = matmul(Phi, matmul(Fhat_inv, U)) + V_res = matmul(Phi, matmul(Fhat_inv, V)) + + # # U, S, VT = svd(diag(H)) + # # V = dagger(VT) + + # # U_res = matmul(Phi,U) + # # V_res = matmul(Phi,V) return U_res, S, V_res diff --git a/pyLOM/RES/wrapper.pyx b/pyLOM/RES/wrapper.pyx index 16f581d9..d108b360 100644 --- a/pyLOM/RES/wrapper.pyx +++ b/pyLOM/RES/wrapper.pyx @@ -25,10 +25,10 @@ from libc.stdlib cimport malloc, free from libc.string cimport memcpy, memset from libc.math cimport sqrt, log, atan2 from ..vmmath.cfuncs cimport real, real_complex -from ..vmmath.cfuncs cimport -from ..vmmath.cfuncs cimport -from ..vmmath.cfuncs cimport -from ..vmmath.cfuncs cimport +# from ..vmmath.cfuncs cimport +# from ..vmmath.cfuncs cimport +from ..vmmath.cfuncs cimport c_csvd, c_cdagger, c_cmatmul, c_cmatmulp, c_cvecmat, c_ccholesky, c_cinverse +from ..vmmath.cfuncs cimport c_zsvd, c_zdagger, c_zmatmul, c_zmatmulp, c_cvecmat, c_zcholesky, c_cinverse from ..utils.cr import cr, cr_start, cr_stop from ..utils.errors import raiseError @@ -39,4 +39,104 @@ from ..utils.errors import raiseError @cython.wraparound(False) # turn off negative index wrapping for entire function @cython.nonecheck(False) @cython.cdivision(True) # turn off zero division check -def _srun(float[:,:] Phi, float delta, float freq, float f, ): \ No newline at end of file +def _srun(np.complex64_t[:,:] Phi, float[:] delta, float[:] freq, float f, float[:,:] Q=None): + + # Variables + cdef int m = Phi.shape[0], n = Phi.shape[1] + cdef int ii, retval + + # Compute the resolvent operator + cdef np.complex64_t *Omega + cdef np.complex64_t *H + Omega = malloc(n*sizeof(np.complex64_t)) + H = malloc(n*sizeof(np.complex64_t)) + cr_start('RES.resolvent_operator', 0) + for ii in range(n): + Omega[ii] = delta[ii] + J * freq[ii] + H[ii] = 1 / (-J * f - Omega[ii]) + free(Omega) + cr_stop('RES.resolvent_operator', 0) + + # Compute the Qhat (named Fhat for convenience) + cr_start('RES.Qhat', 0) + cdef np.complex64_t *Phi_dagger + cdef np.complex64_t *Fhat + Phi_dagger = malloc(m*n*sizeof(np.complex64_t)) + Fhat = malloc(n*n*sizeof(np.complex64_t)) + c_cdagger(Phi, Phi_dagger, m, n) + if Q is None: + c_cmatmulp(Fhat, Phi_dagger, Phi, n, n, m) + else: + cdef np.complex64_t *Phi_aux + Phi_aux = malloc(n*n*sizeof(np.complex64_t)) + memcpy(Phi_aux, &Phi[0,0], m*n*sizeof(np.complex64_t)) + c_cvecmat(Q, Phi_aux, m, n) # Phi_aux is overwritten + c_cmatmulp(Fhat, Phi_dagger, Phi_aux, n, n, m) + free(Phi_aux) + free(Phi_dagger) + cr_stop('RES.Qhat', 0) + + # Compute the Choleski decomposition + cr_start('RES.Choleski', 0) + retval = c_ccholesky(Fhat, n) + if not retval == 0: raiseError('Problems computing Cholesky factorization!') + cdef np.complex64_t *Fhat_inv + Fhat_inv = malloc(n*n*sizeof(np.complex64_t)) + memcpy(Fhat_inv, Fhat, m*n*sizeof(np.complex64_t)) + retval = c_cinv(Fhat_inv, n, 'L') + if not retval == 0: raiseError('Problems computing the Inverse!') + cr_stop('RES.Choleski', 0) + + # Compute Hhat + cr_start('RES.Hhat', 0) + cdef np.complex64_t *Hhat + cdef np.complex64_t *Fhat_aux + Hhat = malloc(n*n*sizeof(np.complex64_t)) + Fhat_aux = malloc(n*n*sizeof(np.complex64_t)) + memcpy(Fhat_aux, Fhat_inv, m*n*sizeof(np.complex64_t)) + c_cvecmat(H, Fhat_aux, n, n) + c_cmatmul(Hhat, Fhat, Fhat_aux, n, n, n) + free(H) + ree(H_aux) + free(Fhat) + free(Fhat_aux) + cr_stop('RES.Hhat', 0) + + # Compute the svd + cr_start('RES.svd', 0) + cdef np.complex64_t *U + cdef float *S + cdef np.complex64_t *V + cdef np.complex64_t *Vt + U = malloc(n*n*sizeof(np.complex64_t)) + S = malloc(n*sizeof(float)) + V = malloc(n*n*sizeof(np.complex64_t)) + Vt = malloc(n*n*sizeof(np.complex64_t)) + c_csvd(U, S, Vt, Hhat, n, n) + c_cdagger(Vt, V, n, n) + free(Hhat) + free(Vt) + cr_stop('RES.svd', 0) + + # Compute the projection + cr_start('RES.projection', 0) + cdef np.complex64_t *U_res + cdef np.complex64_t *V_res + cdef np.complex64_t *U_aux + cdef np.complex64_t *V_aux + U_res = malloc(m*n*sizeof(np.complex64_t)) + V_res = malloc(m*n*sizeof(np.complex64_t)) + U_aux = malloc(m*n*sizeof(np.complex64_t)) + V_aux = malloc(m*n*sizeof(np.complex64_t)) + c_cmatmul(U_aux, Fhat_inv, U, n, n, n) + c_cmatmul(V_aux, Fhat_inv, V, n, n, n) + c_cmatmul(U_res, Phi, U_aux, m, n, n) + c_cmatmul(V_res, Phi, V_aux, m, n, n) + free(Fhat_inv) + free(U) + free(U_aux) + free(V) + free(V_aux) + cr_stop('RES.projection', 0) + + return U_res, S, V_res \ No newline at end of file diff --git a/pyLOM/vmmath/cfuncs.pxd b/pyLOM/vmmath/cfuncs.pxd index c5ee4998..1e903eb5 100644 --- a/pyLOM/vmmath/cfuncs.pxd +++ b/pyLOM/vmmath/cfuncs.pxd @@ -24,6 +24,8 @@ cdef extern from "vector_matrix.h" nogil: cdef int c_sinv "sinv"(float *A, int m, int n) cdef int c_sinverse "sinverse"(float *A, int N, char *UoL) cdef void c_ssort "ssort"(float *v, int *index, int n) + cdef void c_sdiag "sdiag"(float *A, float *B, const int m) + cdef void c_sdiag2 "sdiag2"(float *A, float *B, const int m) # Double precision cdef void c_dtranspose "dtranspose"(double *A, double *B, const int m, const int n) cdef double c_dvector_sum "dvector_sum"(double *v, int start, int n) @@ -37,6 +39,8 @@ cdef extern from "vector_matrix.h" nogil: cdef int c_dinv "dinv"(double *A, int m, int n) cdef int c_dinverse "dinverse"(double *A, int N, char *UoL) cdef void c_dsort "dsort"(double *v, int *index, int n) + cdef void c_ddiag "ddiag"(double *A, double *B, const int m) + cdef void c_ddiag2 "ddiag2"(double *A, double *B, const int m) # Single complex precision cdef void c_ctranspose "ctranspose"(np.complex64_t *A, np.complex64_t *B, const int m, const int n) cdef void c_cmatmult "cmatmult"(np.complex64_t *C, np.complex64_t *A, np.complex64_t *B, const int m, const int n, const int k, const char *TA, const char *TB) @@ -51,6 +55,10 @@ cdef extern from "vector_matrix.h" nogil: cdef void c_cvandermonde "cvandermonde"(np.complex64_t *Vand, float *real, float *imag, int m, int n) cdef void c_cvandermonde_time "cvandermondeTime"(np.complex64_t *Vand, float *real, float *imag, int m, int n, float* t) cdef void c_csort "csort"(np.complex64_t *v, int *index, int n) + cdef void c_cconj "cconj"(np.complex64_t *A, np.complex64_t *B, const int m, const int n) + cdef void c_cdagger "cdagger"(np.complex64_t *A, np.complex64_t *B, const int m, const int n) + cdef void c_cdiag "cdiag"(np.complex64_t *A, np.complex64_t *B, const int m) + cdef void c_cdiag2 "cdiag2"(np.complex64_t *A, np.complex64_t *B, const int m) # Double complex precision cdef void c_ztranspose "ztranspose"(np.complex128_t *A, np.complex128_t *B, const int m, const int n) cdef void c_zmatmult "zmatmult"(np.complex128_t *C, np.complex128_t *A, np.complex128_t *B, const int m, const int n, const int k, const char *TA, const char *TB) @@ -65,6 +73,10 @@ cdef extern from "vector_matrix.h" nogil: cdef void c_zvandermonde "zvandermonde"(np.complex128_t *Vand, double *real, double *imag, int m, int n) cdef void c_zvandermonde_time "zvandermondeTime"(np.complex128_t *Vand, double *real, double *imag, int m, int n, double* t) cdef void c_zsort "zsort"(np.complex128_t *v, int *index, int n) + cdef void c_zconj "zconj"(np.complex128_t *A, np.complex128_t *B, const int m, const int n) + cdef void c_zdagger "zdagger"(np.complex128_t *A, np.complex128_t *B, const int m, const int n) + cdef void c_zdiag "zdiag"(np.complex128_t *A, np.complex128_t *B, const int m) + cdef void c_zdiag2 "zdiag2"(np.complex128_t *A, np.complex128_t *B, const int m) cdef extern from "averaging.h" nogil: # Single precision cdef void c_stemporal_mean "stemporal_mean"(float *out, float *X, const int m, const int n) diff --git a/pyLOM/vmmath/maths.pyx b/pyLOM/vmmath/maths.pyx index 11670243..c2b5ced1 100644 --- a/pyLOM/vmmath/maths.pyx +++ b/pyLOM/vmmath/maths.pyx @@ -24,10 +24,10 @@ from libc.stdlib cimport malloc, free from libc.string cimport memcpy, memset from libc.math cimport sqrt, atan2 from .cfuncs cimport real, real_complex, real_full -from .cfuncs cimport c_stranspose, c_svector_sum, c_svector_norm, c_svector_mean, c_smatmul, c_smatmulp, c_svecmat, c_svecmatT, c_sinv, c_ssort -from .cfuncs cimport c_dtranspose, c_dvector_sum, c_dvector_norm, c_dvector_mean, c_dmatmul, c_dmatmulp, c_dvecmat, c_dvecmatT, c_dinv, c_dsort -from .cfuncs cimport c_ctranspose, c_cmatmul, c_cmatmulp, c_cvecmat, c_cvecmatT, c_cinv, c_ceigen, c_ccholesky, c_cvandermonde, c_cvandermonde_time, c_csort -from .cfuncs cimport c_ztranspose, c_zmatmul, c_zmatmulp, c_zvecmat, c_zvecmatT,c_zinv, c_zeigen, c_zcholesky, c_zvandermonde, c_zvandermonde_time, c_zsort +from .cfuncs cimport c_stranspose, c_svector_sum, c_svector_norm, c_svector_mean, c_smatmul, c_smatmulp, c_svecmat, c_svecmatT, c_sinv, c_ssort, c_sdiag, c_sdiag2 +from .cfuncs cimport c_dtranspose, c_dvector_sum, c_dvector_norm, c_dvector_mean, c_dmatmul, c_dmatmulp, c_dvecmat, c_dvecmatT, c_dinv, c_dsort, c_ddiag, c_ddiag2 +from .cfuncs cimport c_ctranspose, c_cmatmul, c_cmatmulp, c_cvecmat, c_cvecmatT, c_cinv, c_ceigen, c_ccholesky, c_cvandermonde, c_cvandermonde_time, c_csort, c_cconj, c_cdagger, c_cdiag, c_cdiag2 +from .cfuncs cimport c_ztranspose, c_zmatmul, c_zmatmulp, c_zvecmat, c_zvecmatT,c_zinv, c_zeigen, c_zcholesky, c_zvandermonde, c_zvandermonde_time, c_zsort, c_zconj, c_zdagger, c_zdiag, c_zdiag2 from ..utils.cr import cr from ..utils.errors import raiseError @@ -954,8 +954,7 @@ cdef np.ndarray[np.float32_t,ndim=1] _sdiag(float[:,:] A): cdef int m = A.shape[0] cdef int ii cdef np.ndarray[np.float32_t,ndim=1] B = np.zeros((m,),dtype=np.float32) - for ii in range(m): - B[ii] = A[ii][ii] + c_sdiag(&A[0,0], &B[0], m) return B @cython.initializedcheck(False) @@ -970,8 +969,37 @@ cdef np.ndarray[np.double_t,ndim=1] _ddiag(double[:,:] A): cdef int m = A.shape[0] cdef int ii cdef np.ndarray[np.double_t,ndim=1] B = np.zeros((m,),dtype=np.double) - for ii in range(m): - B[ii] = A[ii][ii] + c_ddiag(&A[0,0], &B[0], m) + return B + +@cython.initializedcheck(False) +@cython.boundscheck(False) # turn off bounds-checking for entire function +@cython.wraparound(False) # turn off negative index wrapping for entire function +@cython.nonecheck(False) +@cython.cdivision(True) # turn off zero division check +cdef np.ndarray[np.complex64_t,ndim=1] _cdiag(np.complex64_t[:,:] A): + ''' + Returns the diagonal of A (A is a square matrix) + ''' + cdef int m = A.shape[0] + cdef int ii + cdef np.ndarray[np.complex64_t,ndim=1] B = np.zeros((m,),dtype=np.complex64) + c_cdiag(&A[0,0], &B[0], m) + return B + +@cython.initializedcheck(False) +@cython.boundscheck(False) # turn off bounds-checking for entire function +@cython.wraparound(False) # turn off negative index wrapping for entire function +@cython.nonecheck(False) +@cython.cdivision(True) # turn off zero division check +cdef np.ndarray[np.complex128_t,ndim=1] _zdiag(np.complex128_t[:,:] A): + ''' + Returns the diagonal of A (A is a square matrix) + ''' + cdef int m = A.shape[0] + cdef int ii + cdef np.ndarray[np.complex128_t,ndim=1] B = np.zeros((m,),dtype=np.complex128) + c_zdiag(&A[0,0], &B[0], m) return B @cython.initializedcheck(False) @@ -987,8 +1015,7 @@ cdef np.ndarray[np.float32_t,ndim=2] _sdiag2(float[:] A): cdef int ii cdef int jj cdef np.ndarray[np.float32_t,ndim=2] B = np.zeros((m,m),dtype=np.float32) - for ii in range(m): - B[ii,ii] = A[ii] + c_sdiag2(&A[0], &B[0,0], m) return B @cython.initializedcheck(False) @@ -1003,8 +1030,38 @@ cdef np.ndarray[np.double_t,ndim=2] _ddiag2(double[:] A): cdef int m = A.shape[0] cdef int ii cdef np.ndarray[np.double_t,ndim=2] B = np.zeros((m,m),dtype=np.double) - for ii in range(m): - B[ii,ii] = A[ii] + c_ddiag2(&A[0], &B[0,0], m) + return B + +@cython.initializedcheck(False) +@cython.boundscheck(False) # turn off bounds-checking for entire function +@cython.wraparound(False) # turn off negative index wrapping for entire function +@cython.nonecheck(False) +@cython.cdivision(True) # turn off zero division check +cdef np.ndarray[np.complex64_t,ndim=2] _cdiag2(np.complex64_t[:] A): + ''' + Returns a matrix with A in its diagonal + ''' + cdef int m = A.shape[0] + cdef int ii + cdef int jj + cdef np.ndarray[np.complex64_t,ndim=2] B = np.zeros((m,m),dtype=np.complex64) + c_cdiag2(&A[0], &B[0,0], m) + return B + +@cython.initializedcheck(False) +@cython.boundscheck(False) # turn off bounds-checking for entire function +@cython.wraparound(False) # turn off negative index wrapping for entire function +@cython.nonecheck(False) +@cython.cdivision(True) # turn off zero division check +cdef np.ndarray[np.complex128_t,ndim=2] _zdiag2(np.complex128_t[:] A): + ''' + Returns a matrix with A in its diagonal + ''' + cdef int m = A.shape[0] + cdef int ii + cdef np.ndarray[np.complex128_t,ndim=2] B = np.zeros((m,m),dtype=np.complex128) + c_zdiag2(&A[0], &B[0,0], m) return B @cr('math.diag') @@ -1026,11 +1083,19 @@ def diag(np.ndarray A): if A.ndim == 1: if A.dtype in [np.float64,np.double]: return _ddiag2(A) + elif A.dtype == np.complex64: + return _cdiag2(A) + elif A.dtype == np.complex128: + return _zdiag2(A) else: return _sdiag2(A) else: if A.dtype in [np.float64,np.double]: return _ddiag(A) + elif A.dtype == np.complex64: + return _cdiag(A) + elif A.dtype == np.complex128: + return _zdiag(A) else: return _sdiag(A) @@ -1048,9 +1113,7 @@ cdef np.ndarray[np.complex64_t,ndim=2] _cconj(np.complex64_t[:,:] A): cdef int ii cdef int jj cdef np.ndarray[np.complex64_t,ndim=2] B = np.zeros((m,n),dtype=np.complex64) - for ii in range(m): - for jj in range(n): - B[ii, jj] = crealf(A[ii][jj]) - cimagf(A[ii][jj])*I + c_cconj(&A[0,0], &B[0,0], m, n) return B @cython.initializedcheck(False) @@ -1067,9 +1130,7 @@ cdef np.ndarray[np.complex128_t,ndim=2] _zconj(np.complex128_t[:,:] A): cdef int ii cdef int jj cdef np.ndarray[np.complex128_t,ndim=2] B = np.zeros((m,n),dtype=np.complex128) - for ii in range(m): - for jj in range(n): - B[ii, jj] = creal(A[ii][jj]) - cimag(A[ii][jj])*J + c_zconj(&A[0,0], &B[0,0], m, n) return B @cr('math.conj') @@ -1188,6 +1249,40 @@ def flip(real[:,:] A): ''' raiseError('Function not implemented in Cython!') +@cython.initializedcheck(False) +@cython.boundscheck(False) # turn off bounds-checking for entire function +@cython.wraparound(False) # turn off negative index wrapping for entire function +@cython.nonecheck(False) +@cython.cdivision(True) # turn off zero division check +cdef np.ndarray[np.complex64_t,ndim=2] _cdagger(np.complex64_t[:,:] A): + ''' + Returns the dagger of A + ''' + cdef int m = A.shape[0] + cdef int n = A.shape[1] + cdef int ii + cdef int jj + cdef np.ndarray[np.complex64_t,ndim=2] B = np.zeros((n,m),dtype=np.complex64) + c_cdagger(&A[0,0], &B[0,0], m, n) + return B + +@cython.initializedcheck(False) +@cython.boundscheck(False) # turn off bounds-checking for entire function +@cython.wraparound(False) # turn off negative index wrapping for entire function +@cython.nonecheck(False) +@cython.cdivision(True) # turn off zero division check +cdef np.ndarray[np.complex128_t,ndim=2] _zdagger(np.complex128_t[:,:] A): + ''' + Returns the dagger of A + ''' + cdef int m = A.shape[0] + cdef int n = A.shape[1] + cdef int ii + cdef int jj + cdef np.ndarray[np.complex128_t,ndim=2] B = np.zeros((n,m),dtype=np.complex128) + c_zdagger(&A[0,0], &B[0,0], m, n) + return B + @cr('math.dagger') @cython.initializedcheck(False) @cython.boundscheck(False) # turn off bounds-checking for entire function @@ -1205,9 +1300,9 @@ def dagger(real_full[:,:] A): np.ndarray: dagger matrix ''' if real_full is np.complex128_t: - return _zconj(_ztranspose(A)) + return _zdagger(A) elif real_full is np.complex64_t: - return _cconj(_ctranspose(A)) + return _cdagger(A) elif real_full is double: return _dtranspose(A) else: diff --git a/pyLOM/vmmath/src/vector_matrix.c b/pyLOM/vmmath/src/vector_matrix.c index 752d3275..2c532ff6 100644 --- a/pyLOM/vmmath/src/vector_matrix.c +++ b/pyLOM/vmmath/src/vector_matrix.c @@ -1057,4 +1057,148 @@ void drandom_matrix(double *A, int m, int n, unsigned int seed){ AC_MAT(A,n,i,j) = (double)(rand()) / (double)(RAND_MAX); } } +} + +void cconj(scomplex_t *A, scomplex_t *B, const int m, const int n){ + /* + Returns the conjugated version of a matrix + */ + for (int ii = 0; ii < m; ii++){ + for (int jj = 0; jj < n; jj++){ + AC_MAT(B,n,ii,jj) = conjf(AC_MAT(A,n,ii,jj)); + } + } +} + +void zconj(dcomplex_t *A, dcomplex_t *B, const int m, const int n){ + /* + Returns the conjugated version of a matrix + */ + for (int ii = 0; ii < m; ii++){ + for (int jj = 0; jj < n; jj++){ + AC_MAT(B,n,ii,jj) = conj(AC_MAT(A,n,ii,jj)); + } + } +} + +void cdagger(scomplex_t *A, scomplex_t *B, const int m, const int n){ + /* + Returns the dagger version of a matrix + */ + for (int ii = 0; ii < m; ii++){ + for (int jj = 0; jj < n; jj++){ + AC_MAT(B,m,jj,ii) = conjf(AC_MAT(A,n,ii,jj)); + } + } +} + +void zdagger(dcomplex_t *A, dcomplex_t *B, const int m, const int n){ + /* + Returns the dagger version of a matrix + */ + for (int ii = 0; ii < m; ii++){ + for (int jj = 0; jj < n; jj++){ + AC_MAT(B,m,jj,ii) = conj(AC_MAT(A,n,ii,jj)); + } + } +} + +void sdiag(float *A, float *B, const int m){ + /* + Diagonal of a square matrix. + */ + for (int ii = 0; ii < m; ii++){ + B[ii] = AC_MAT(A,m,ii,ii); + } +} + +void ddiag(double *A, double *B, const int m){ + /* + Diagonal of a square matrix. + */ + for (int ii = 0; ii < m; ii++){ + B[ii] = AC_MAT(A,m,ii,ii); + } +} + +void cdiag(scomplex_t *A, scomplex_t *B, const int m){ + /* + Diagonal of a square matrix. + */ + for (int ii = 0; ii < m; ii++){ + B[ii] = AC_MAT(A,m,ii,ii); + } +} + +void zdiag(dcomplex_t *A, dcomplex_t *B, const int m){ + /* + Diagonal of a square matrix. + */ + for (int ii = 0; ii < m; ii++){ + B[ii] = AC_MAT(A,m,ii,ii); + } +} + +void sdiag2(float *A, float *B, const int m){ + /* + Diagonal matrix of an array. + */ + for (int ii = 0; ii < m; ii++){ + for (int jj = 0; jj < m; jj++){ + if (ii == jj){ + AC_MAT(B,m,ii,ii) = A[ii]; + } + else{ + AC_MAT(B,m,ii,jj) = 0; + } + } + } +} + +void ddiag2(double *A, double *B, const int m){ + /* + Diagonal matrix of an array. + */ + for (int ii = 0; ii < m; ii++){ + for (int jj = 0; jj < m; jj++){ + if (ii == jj){ + AC_MAT(B,m,ii,ii) = A[ii]; + } + else{ + AC_MAT(B,m,ii,jj) = 0; + } + } + } +} + +void cdiag2(scomplex_t *A, scomplex_t *B, const int m){ + /* + Diagonal matrix of an array. + */ + for (int ii = 0; ii < m; ii++){ + for (int jj = 0; jj < m; jj++){ + if (ii == jj){ + AC_MAT(B,m,ii,ii) = A[ii]; + } + else{ + AC_MAT(B,m,ii,jj) = 0; + } + } + } +} + +void zdiag2(dcomplex_t *A, dcomplex_t *B, const int m){ + /* + Diagonal matrix of an array. + */ + for (int ii = 0; ii < m; ii++){ + for (int jj = 0; jj < m; jj++){ + if (ii == jj){ + AC_MAT(B,m,ii,ii) = A[ii]; + } + else{ + AC_MAT(B,m,ii,jj) = 0; + } + } + } } \ No newline at end of file diff --git a/pyLOM/vmmath/src/vector_matrix.h b/pyLOM/vmmath/src/vector_matrix.h index 16f60033..a48e6ec3 100644 --- a/pyLOM/vmmath/src/vector_matrix.h +++ b/pyLOM/vmmath/src/vector_matrix.h @@ -25,6 +25,8 @@ int sinverse(float *A, int N, char *UoL); void ssort(float *v, int *index, int n); void srandom_matrix(float *A, int m, int n, unsigned int seed); void seuclidean_d(float *D, float *X, const int m, const int n); +void sdiag(float *A, float *B, const int m); +void sdiag2(float *A, float *B, const int m); // Double version void dtranspose(double *A, double *B, const int m, const int n); double dvector_sum(double *v, int start, int n); @@ -41,6 +43,8 @@ int dinverse(double *A, int N, char *UoL); void dsort(double *v, int *index, int n); void drandom_matrix(double *A, int m, int n, unsigned int seed); void deuclidean_d(double *D, double *X, const int m, const int n); +void ddiag(float *A, float *B, const int m); +void ddiag2(float *A, float *B, const int m); // Float complex version void ctranspose(scomplex_t *A, scomplex_t *B, const int m, const int n); void cmatmult(scomplex_t *C, scomplex_t *A, scomplex_t *B, const int m, const int n, const int k, const char *TA, const char *TB); @@ -55,6 +59,10 @@ int ccholesky(scomplex_t *A, int N); void cvandermonde(scomplex_t *Vand, float *real, float *imag, int m, int n); void cvandermondeTime(scomplex_t *Vand, float *real, float *imag, int m, int n, float *t); void csort(scomplex_t *v, int *index, int n); +void cconj(scomplex_t *A, scomplex_t *B, int m, int n); +void cdagger(scomplex_t *A, scomplex_t *B, int m, int n); +void cdiag(float *A, float *B, const int m); +void cdiag2(float *A, float *B, const int m); // Double complex version void ztranspose(dcomplex_t *A, dcomplex_t *B, const int m, const int n); void zmatmult(dcomplex_t *C, dcomplex_t *A, dcomplex_t *B, const int m, const int n, const int k, const char *TA, const char *TB); @@ -69,3 +77,7 @@ int zcholesky(dcomplex_t *A, int N); void zvandermonde(dcomplex_t *Vand, double *real, double *imag, int m, int n); void zvandermondeTime(dcomplex_t *Vand, double *real, double *imag, int m, int n, double *t); void zsort(dcomplex_t *v, int *index, int n); +void zconj(scomplex_t *A, scomplex_t *B, int m, int n); +void zdagger(scomplex_t *A, scomplex_t *B, int m, int n); +void zdiag(float *A, float *B, const int m); +void zdiag2(float *A, float *B, const int m); \ No newline at end of file From 88365d69f5179c695c0e853435d6bb6007585a43 Mon Sep 17 00:00:00 2001 From: msilvestrec03 Date: Fri, 15 May 2026 15:46:26 +0200 Subject: [PATCH 10/27] =?UTF-8?q?=C2=B7compiled=20version=20of=20RES?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- options.cfg | 2 +- pyLOM/RES/wrapper.pyx | 164 ++++++++++++++++++++++++++----- pyLOM/vmmath/src/vector_matrix.h | 16 +-- setup.py | 14 ++- 4 files changed, 163 insertions(+), 33 deletions(-) diff --git a/options.cfg b/options.cfg index 5177cae9..ec42117c 100644 --- a/options.cfg +++ b/options.cfg @@ -12,7 +12,7 @@ USE_GESVD = OFF USE_COMPILED = ON # Comma separated list of the modules to be compiled # if USE_COMPILED = ON -MODULES_COMPILED = MATH.MATHS,MATH.AVERAGING,MATH.SVD,MATH.FFT,MATH.GEOMETRIC,MATH.TRUNCATION,MATH.STATS,MATH.REGRESSION,ROM.POD,ROM.DMD,ROM.SPOD +MODULES_COMPILED = MATH.MATHS,MATH.AVERAGING,MATH.SVD,MATH.FFT,MATH.GEOMETRIC,MATH.TRUNCATION,MATH.STATS,MATH.REGRESSION,ROM.POD,ROM.DMD,ROM.SPOD,ROM.RES ## Optimization, host and CPU type # diff --git a/pyLOM/RES/wrapper.pyx b/pyLOM/RES/wrapper.pyx index d108b360..3d2b13d6 100644 --- a/pyLOM/RES/wrapper.pyx +++ b/pyLOM/RES/wrapper.pyx @@ -28,7 +28,7 @@ from ..vmmath.cfuncs cimport real, real_complex # from ..vmmath.cfuncs cimport # from ..vmmath.cfuncs cimport from ..vmmath.cfuncs cimport c_csvd, c_cdagger, c_cmatmul, c_cmatmulp, c_cvecmat, c_ccholesky, c_cinverse -from ..vmmath.cfuncs cimport c_zsvd, c_zdagger, c_zmatmul, c_zmatmulp, c_cvecmat, c_zcholesky, c_cinverse +from ..vmmath.cfuncs cimport c_zsvd, c_zdagger, c_zmatmul, c_zmatmulp, c_zvecmat, c_zcholesky, c_zinverse from ..utils.cr import cr, cr_start, cr_stop from ..utils.errors import raiseError @@ -39,8 +39,8 @@ from ..utils.errors import raiseError @cython.wraparound(False) # turn off negative index wrapping for entire function @cython.nonecheck(False) @cython.cdivision(True) # turn off zero division check -def _srun(np.complex64_t[:,:] Phi, float[:] delta, float[:] freq, float f, float[:,:] Q=None): - +def _crun(np.complex64_t[:,:] Phi, float[:] delta, float[:] freq, float f, float[:] Q=None): + # Variables cdef int m = Phi.shape[0], n = Phi.shape[1] cdef int ii, retval @@ -60,17 +60,19 @@ def _srun(np.complex64_t[:,:] Phi, float[:] delta, float[:] freq, float f, float # Compute the Qhat (named Fhat for convenience) cr_start('RES.Qhat', 0) cdef np.complex64_t *Phi_dagger + cdef np.complex64_t *Phi_aux cdef np.complex64_t *Fhat - Phi_dagger = malloc(m*n*sizeof(np.complex64_t)) + Phi_dagger = malloc(n*m*sizeof(np.complex64_t)) Fhat = malloc(n*n*sizeof(np.complex64_t)) - c_cdagger(Phi, Phi_dagger, m, n) + Phi_aux = malloc(n*n*sizeof(np.complex64_t)) + cdef np.ndarray[np.complex64_t,ndim=1] Q_aux = np.zeros((m),dtype=np.complex64) + c_cdagger(&Phi[0,0], Phi_dagger, m, n) if Q is None: - c_cmatmulp(Fhat, Phi_dagger, Phi, n, n, m) + c_cmatmulp(Fhat, Phi_dagger, &Phi[0,0], n, n, m) else: - cdef np.complex64_t *Phi_aux - Phi_aux = malloc(n*n*sizeof(np.complex64_t)) memcpy(Phi_aux, &Phi[0,0], m*n*sizeof(np.complex64_t)) - c_cvecmat(Q, Phi_aux, m, n) # Phi_aux is overwritten + Q_aux = np.array(Q, dtype=np.complex64) + c_cvecmat(&Q_aux[0], Phi_aux, m, n) # Phi_aux is overwritten c_cmatmulp(Fhat, Phi_dagger, Phi_aux, n, n, m) free(Phi_aux) free(Phi_dagger) @@ -82,8 +84,8 @@ def _srun(np.complex64_t[:,:] Phi, float[:] delta, float[:] freq, float f, float if not retval == 0: raiseError('Problems computing Cholesky factorization!') cdef np.complex64_t *Fhat_inv Fhat_inv = malloc(n*n*sizeof(np.complex64_t)) - memcpy(Fhat_inv, Fhat, m*n*sizeof(np.complex64_t)) - retval = c_cinv(Fhat_inv, n, 'L') + memcpy(Fhat_inv, Fhat, n*n*sizeof(np.complex64_t)) + retval = c_cinverse(Fhat_inv, n, 'L') if not retval == 0: raiseError('Problems computing the Inverse!') cr_stop('RES.Choleski', 0) @@ -93,11 +95,10 @@ def _srun(np.complex64_t[:,:] Phi, float[:] delta, float[:] freq, float f, float cdef np.complex64_t *Fhat_aux Hhat = malloc(n*n*sizeof(np.complex64_t)) Fhat_aux = malloc(n*n*sizeof(np.complex64_t)) - memcpy(Fhat_aux, Fhat_inv, m*n*sizeof(np.complex64_t)) + memcpy(Fhat_aux, Fhat_inv, n*n*sizeof(np.complex64_t)) c_cvecmat(H, Fhat_aux, n, n) c_cmatmul(Hhat, Fhat, Fhat_aux, n, n, n) free(H) - ree(H_aux) free(Fhat) free(Fhat_aux) cr_stop('RES.Hhat', 0) @@ -105,14 +106,13 @@ def _srun(np.complex64_t[:,:] Phi, float[:] delta, float[:] freq, float f, float # Compute the svd cr_start('RES.svd', 0) cdef np.complex64_t *U - cdef float *S + cdef np.ndarray[np.float32_t,ndim=1] S = np.zeros((n),dtype=np.float32) cdef np.complex64_t *V cdef np.complex64_t *Vt U = malloc(n*n*sizeof(np.complex64_t)) - S = malloc(n*sizeof(float)) V = malloc(n*n*sizeof(np.complex64_t)) Vt = malloc(n*n*sizeof(np.complex64_t)) - c_csvd(U, S, Vt, Hhat, n, n) + c_csvd(U, &S[0], Vt, Hhat, n, n) c_cdagger(Vt, V, n, n) free(Hhat) free(Vt) @@ -120,18 +120,128 @@ def _srun(np.complex64_t[:,:] Phi, float[:] delta, float[:] freq, float f, float # Compute the projection cr_start('RES.projection', 0) - cdef np.complex64_t *U_res - cdef np.complex64_t *V_res + # cdef np.complex64_t *U_res + # cdef np.complex64_t *V_res cdef np.complex64_t *U_aux cdef np.complex64_t *V_aux - U_res = malloc(m*n*sizeof(np.complex64_t)) - V_res = malloc(m*n*sizeof(np.complex64_t)) + # U_res = malloc(m*n*sizeof(np.complex64_t)) + # V_res = malloc(m*n*sizeof(np.complex64_t)) U_aux = malloc(m*n*sizeof(np.complex64_t)) V_aux = malloc(m*n*sizeof(np.complex64_t)) + cdef np.ndarray[np.complex64_t,ndim=2] U_res = np.zeros((m,n),dtype=np.complex64) + cdef np.ndarray[np.complex64_t,ndim=2] V_res = np.zeros((m,n),dtype=np.complex64) c_cmatmul(U_aux, Fhat_inv, U, n, n, n) c_cmatmul(V_aux, Fhat_inv, V, n, n, n) - c_cmatmul(U_res, Phi, U_aux, m, n, n) - c_cmatmul(V_res, Phi, V_aux, m, n, n) + c_cmatmul(&U_res[0,0], &Phi[0,0], U_aux, m, n, n) + c_cmatmul(&V_res[0,0], &Phi[0,0], V_aux, m, n, n) + free(Fhat_inv) + free(U) + free(U_aux) + free(V) + free(V_aux) + cr_stop('RES.projection', 0) + + return U_res, S, V_res + +@cython.boundscheck(False) # turn off bounds-checking for entire function +@cython.wraparound(False) # turn off negative index wrapping for entire function +@cython.nonecheck(False) +@cython.cdivision(True) # turn off zero division check +def _zrun(np.complex128_t[:,:] Phi, double[:] delta, double[:] freq, double f, double[:] Q=None): + + # Variables + cdef int m = Phi.shape[0], n = Phi.shape[1] + cdef int ii, retval + + # Compute the resolvent operator + cdef np.complex128_t *Omega + cdef np.complex128_t *H + Omega = malloc(n*sizeof(np.complex128_t)) + H = malloc(n*sizeof(np.complex128_t)) + cr_start('RES.resolvent_operator', 0) + for ii in range(n): + Omega[ii] = delta[ii] + J * freq[ii] + H[ii] = 1 / (-J * f - Omega[ii]) + free(Omega) + cr_stop('RES.resolvent_operator', 0) + + # Compute the Qhat (named Fhat for convenience) + cr_start('RES.Qhat', 0) + cdef np.complex128_t *Phi_dagger + cdef np.complex128_t *Phi_aux + cdef np.complex128_t *Fhat + Phi_dagger = malloc(n*m*sizeof(np.complex128_t)) + Fhat = malloc(n*n*sizeof(np.complex128_t)) + Phi_aux = malloc(n*n*sizeof(np.complex128_t)) + cdef np.ndarray[np.complex128_t,ndim=1] Q_aux = np.zeros((m),dtype=np.complex128) + c_zdagger(&Phi[0,0], Phi_dagger, m, n) + if Q is None: + c_zmatmulp(Fhat, Phi_dagger, &Phi[0,0], n, n, m) + else: + memcpy(Phi_aux, &Phi[0,0], m*n*sizeof(np.complex128_t)) + Q_aux = np.array(Q, dtype=np.complex128) + c_zvecmat(&Q_aux[0], Phi_aux, m, n) # Phi_aux is overwritten + c_zmatmulp(Fhat, Phi_dagger, Phi_aux, n, n, m) + free(Phi_aux) + free(Phi_dagger) + cr_stop('RES.Qhat', 0) + + # Compute the Choleski decomposition + cr_start('RES.Choleski', 0) + retval = c_zcholesky(Fhat, n) + if not retval == 0: raiseError('Problems computing Cholesky factorization!') + cdef np.complex128_t *Fhat_inv + Fhat_inv = malloc(n*n*sizeof(np.complex128_t)) + memcpy(Fhat_inv, Fhat, n*n*sizeof(np.complex128_t)) + retval = c_zinverse(Fhat_inv, n, 'L') + if not retval == 0: raiseError('Problems computing the Inverse!') + cr_stop('RES.Choleski', 0) + + # Compute Hhat + cr_start('RES.Hhat', 0) + cdef np.complex128_t *Hhat + cdef np.complex128_t *Fhat_aux + Hhat = malloc(n*n*sizeof(np.complex128_t)) + Fhat_aux = malloc(n*n*sizeof(np.complex128_t)) + memcpy(Fhat_aux, Fhat_inv, n*n*sizeof(np.complex128_t)) + c_zvecmat(H, Fhat_aux, n, n) + c_zmatmul(Hhat, Fhat, Fhat_aux, n, n, n) + free(H) + free(Fhat) + free(Fhat_aux) + cr_stop('RES.Hhat', 0) + + # Compute the svd + cr_start('RES.svd', 0) + cdef np.complex128_t *U + cdef np.ndarray[np.float64_t,ndim=1] S = np.zeros((n),dtype=np.float64) + cdef np.complex128_t *V + cdef np.complex128_t *Vt + U = malloc(n*n*sizeof(np.complex128_t)) + V = malloc(n*n*sizeof(np.complex128_t)) + Vt = malloc(n*n*sizeof(np.complex128_t)) + c_zsvd(U, &S[0], Vt, Hhat, n, n) + c_zdagger(Vt, V, n, n) + free(Hhat) + free(Vt) + cr_stop('RES.svd', 0) + + # Compute the projection + cr_start('RES.projection', 0) + # cdef np.complex128_t *U_res + # cdef np.complex128_t *V_res + cdef np.complex128_t *U_aux + cdef np.complex128_t *V_aux + # U_res = malloc(m*n*sizeof(np.complex128_t)) + # V_res = malloc(m*n*sizeof(np.complex128_t)) + U_aux = malloc(m*n*sizeof(np.complex128_t)) + V_aux = malloc(m*n*sizeof(np.complex128_t)) + cdef np.ndarray[np.complex128_t,ndim=2] U_res = np.zeros((m,n),dtype=np.complex128) + cdef np.ndarray[np.complex128_t,ndim=2] V_res = np.zeros((m,n),dtype=np.complex128) + c_zmatmul(U_aux, Fhat_inv, U, n, n, n) + c_zmatmul(V_aux, Fhat_inv, V, n, n, n) + c_zmatmul(&U_res[0,0], &Phi[0,0], U_aux, m, n, n) + c_zmatmul(&V_res[0,0], &Phi[0,0], V_aux, m, n, n) free(Fhat_inv) free(U) free(U_aux) @@ -139,4 +249,12 @@ def _srun(np.complex64_t[:,:] Phi, float[:] delta, float[:] freq, float f, float free(V_aux) cr_stop('RES.projection', 0) - return U_res, S, V_res \ No newline at end of file + return U_res, S, V_res + +def run(real_complex[:,:] Phi, real[:] delta, real[:] freq, real f, real[:] Q=None): + + if real_complex is np.complex128_t: + return _zrun(Phi, delta, freq, f, Q) + else: + return _crun(Phi, delta, freq, f, Q) + # return _crun(Phi, delta, freq, f, Q) \ No newline at end of file diff --git a/pyLOM/vmmath/src/vector_matrix.h b/pyLOM/vmmath/src/vector_matrix.h index a48e6ec3..759d2b34 100644 --- a/pyLOM/vmmath/src/vector_matrix.h +++ b/pyLOM/vmmath/src/vector_matrix.h @@ -43,8 +43,8 @@ int dinverse(double *A, int N, char *UoL); void dsort(double *v, int *index, int n); void drandom_matrix(double *A, int m, int n, unsigned int seed); void deuclidean_d(double *D, double *X, const int m, const int n); -void ddiag(float *A, float *B, const int m); -void ddiag2(float *A, float *B, const int m); +void ddiag(double *A, double *B, const int m); +void ddiag2(double *A, double *B, const int m); // Float complex version void ctranspose(scomplex_t *A, scomplex_t *B, const int m, const int n); void cmatmult(scomplex_t *C, scomplex_t *A, scomplex_t *B, const int m, const int n, const int k, const char *TA, const char *TB); @@ -61,8 +61,8 @@ void cvandermondeTime(scomplex_t *Vand, float *real, float *imag, int m, int n void csort(scomplex_t *v, int *index, int n); void cconj(scomplex_t *A, scomplex_t *B, int m, int n); void cdagger(scomplex_t *A, scomplex_t *B, int m, int n); -void cdiag(float *A, float *B, const int m); -void cdiag2(float *A, float *B, const int m); +void cdiag(scomplex_t *A, scomplex_t *B, const int m); +void cdiag2(scomplex_t *A, scomplex_t *B, const int m); // Double complex version void ztranspose(dcomplex_t *A, dcomplex_t *B, const int m, const int n); void zmatmult(dcomplex_t *C, dcomplex_t *A, dcomplex_t *B, const int m, const int n, const int k, const char *TA, const char *TB); @@ -77,7 +77,7 @@ int zcholesky(dcomplex_t *A, int N); void zvandermonde(dcomplex_t *Vand, double *real, double *imag, int m, int n); void zvandermondeTime(dcomplex_t *Vand, double *real, double *imag, int m, int n, double *t); void zsort(dcomplex_t *v, int *index, int n); -void zconj(scomplex_t *A, scomplex_t *B, int m, int n); -void zdagger(scomplex_t *A, scomplex_t *B, int m, int n); -void zdiag(float *A, float *B, const int m); -void zdiag2(float *A, float *B, const int m); \ No newline at end of file +void zconj(dcomplex_t *A, dcomplex_t *B, int m, int n); +void zdagger(dcomplex_t *A, dcomplex_t *B, int m, int n); +void zdiag(dcomplex_t *A, dcomplex_t *B, const int m); +void zdiag2(dcomplex_t *A, dcomplex_t *B, const int m); \ No newline at end of file diff --git a/setup.py b/setup.py index ce7c80de..5cd8777c 100644 --- a/setup.py +++ b/setup.py @@ -310,6 +310,18 @@ libraries = libraries, ) +Module_RES = Extension('pyLOM.RES.wrapper', + sources = ['pyLOM/RES/wrapper.pyx', + 'pyLOM/vmmath/src/vector_matrix.c', + 'pyLOM/vmmath/src/qr.c', + 'pyLOM/vmmath/src/svd.c', + ], + language = 'c', + include_dirs = include_dirs + ['pyLOM/vmmath/src',np.get_include(),mpi4py.get_include()], + extra_objects = extra_objects, + libraries = libraries, + ) + ## Build modules # Math module @@ -327,7 +339,7 @@ Module_ROM = [Module_POD] if 'rom.pod' in options['MODULES_COMPILED'] else [] Module_ROM += [Module_DMD] if 'rom.dmd' in options['MODULES_COMPILED'] else [] Module_ROM += [Module_SPOD] if 'rom.spod' in options['MODULES_COMPILED'] else [] - +Module_ROM += [Module_RES] if 'rom.res' in options['MODULES_COMPILED'] else [] ## Decide which modules to compile modules_list = Module_Math + Module_ROM if options['USE_COMPILED'] else [] From 21a078a9338e04b6a2a70766f2108aef27b6841b Mon Sep 17 00:00:00 2001 From: msilvestrec03 Date: Tue, 26 May 2026 16:13:02 +0200 Subject: [PATCH 11/27] documentation --- pyLOM/RES/utils.py | 28 ++++++++++++++-------------- pyLOM/RES/wrapper.py | 13 +++++++++++++ pyLOM/RES/wrapper.pyx | 42 +++++++++++++++++++++++++++++++++++++++--- 3 files changed, 66 insertions(+), 17 deletions(-) diff --git a/pyLOM/RES/utils.py b/pyLOM/RES/utils.py index 61773ef6..9a7d40d5 100644 --- a/pyLOM/RES/utils.py +++ b/pyLOM/RES/utils.py @@ -17,7 +17,7 @@ @cr('RES.extract_modes') def extract_modes(U:np.ndarray,V:np.ndarray,ivar:int,npoints:int,modes:list=[],reshape:bool=True,kind:str="abs"): r''' - When performing POD of several variables simultaneously, this function separates the spatial modes from each of the variables. + When performing RES of several variables simultaneously, this function separates the modes from each of the variables. Args: U (np.ndarray): RES response modes @@ -63,16 +63,16 @@ def extract_modes(U:np.ndarray,V:np.ndarray,ivar:int,npoints:int,modes:list=[],r @cr('RES.save') def save(fname:str,U:np.ndarray,S:np.ndarray,V:np.ndarray,ptable:PartitionTable,nvars:int=1,pointData:bool=True,mode:str='w'): r''' - Store POD results in serial or parallel according to the partition used to compute the POD. It will be saved on a h5 file. + Store RES results in serial or parallel according to the partition used to compute the RES. It will be saved on a h5 file. Args: - fname (str): path to the .h5 file in which the POD will be saved - U (np.ndarray): response modes to save. To avoid saving the spatial modes, just give None as input - S (np.ndarray): singular values to save. To avoid saving the singular values, just give None as input - V (np.ndarray): forcing modes to save. To avoid saving the temporal coefficients, just give None as input - ptable (PartitionTable): partition table used to compute the POD - nvars (int, optional): number of concatenated variables when computing the POD (default ``1``) - pointData (bool, optional): bool to specify if the POD was performed either on point data or cell data (default ``True``) + fname (str): path to the .h5 file in which the RES will be saved + U (np.ndarray): response modes to save. To avoid saving the response modes, just give None as input + S (np.ndarray): energy gains to save. To avoid saving the energy gains, just give None as input + V (np.ndarray): forcing modes to save. To avoid saving the forcing modes, just give None as input + ptable (PartitionTable): partition table used to compute the RES + nvars (int, optional): number of concatenated variables when computing the RES (default ``1``) + pointData (bool, optional): bool to specify if the RES was performed either on point data or cell data (default ``True``) mode (str, optional): mode in which the HDF5 file is opened, 'w' stands for write mode and 'a' stands for append mode. Write mode will overwrite the file and append mode will add the informaiton at the end of the current file, choose with great care what to do in your case (default ``w``). ''' @@ -82,14 +82,14 @@ def save(fname:str,U:np.ndarray,S:np.ndarray,V:np.ndarray,ptable:PartitionTable, @cr('RES.load') def load(fname:str,vars:list=['U','S','V'],nmod:int=-1,ptable:PartitionTable=None): r''' - Load POD results from a .h5 file in serial or parallel according to the partition used to compute the POD. + Load RES results from a .h5 file in serial or parallel according to the partition used to compute the RES. Args: - fname (str): path to the .h5 file in which the POD was saved + fname (str): path to the .h5 file in which the RES was saved vars (list): list of variables to load. The following notation, consistent with the save function, is used, - 'U': spatial modes - 'S': singular values - 'V': temporal coefficients + 'U': response modes + 'S': energy gains + 'V': forcing modes the default option is to load them all, but it is not recommended to load the spatial modes if they are not going to be used during the rest of the script. nmod (int, optional): number of modes to load. By default it will load all the saved modes (default, ``-1``) ptable (PartitionTable, optional): partition table to use when loading the data (default ``None``). diff --git a/pyLOM/RES/wrapper.py b/pyLOM/RES/wrapper.py index 5a20735d..ce8ac88e 100644 --- a/pyLOM/RES/wrapper.py +++ b/pyLOM/RES/wrapper.py @@ -15,6 +15,19 @@ @cr('RES.run') def run(Phi, delta, freq, f, Q=None): + ''' + Resolvent Analysis of snapshot matrix X + Inputs: + - X[ndims*nmesh,n_temp_snapshots]: data matrix + - delta: damping ratio of each mode + - freq: frequency of each mode + - f: target frequency + - Q: weighting matrix + Returns: + - U_res: response modes + - S: emergy gains + - V_res: forcing modes + ''' p = cp if type(Phi) is cp.ndarray else np # Normalization of modes (?) diff --git a/pyLOM/RES/wrapper.pyx b/pyLOM/RES/wrapper.pyx index 3d2b13d6..86408e44 100644 --- a/pyLOM/RES/wrapper.pyx +++ b/pyLOM/RES/wrapper.pyx @@ -40,7 +40,19 @@ from ..utils.errors import raiseError @cython.nonecheck(False) @cython.cdivision(True) # turn off zero division check def _crun(np.complex64_t[:,:] Phi, float[:] delta, float[:] freq, float f, float[:] Q=None): - + ''' + Resolvent Analysis of snapshot matrix X + Inputs: + - X[ndims*nmesh,n_temp_snapshots]: data matrix + - delta: damping ratio of each mode + - freq: frequency of each mode + - f: target frequency + - Q: weighting matrix + Returns: + - U_res: response modes + - S: emergy gains + - V_res: forcing modes + ''' # Variables cdef int m = Phi.shape[0], n = Phi.shape[1] cdef int ii, retval @@ -148,7 +160,19 @@ def _crun(np.complex64_t[:,:] Phi, float[:] delta, float[:] freq, float f, float @cython.nonecheck(False) @cython.cdivision(True) # turn off zero division check def _zrun(np.complex128_t[:,:] Phi, double[:] delta, double[:] freq, double f, double[:] Q=None): - + ''' + Resolvent Analysis of snapshot matrix X + Inputs: + - X[ndims*nmesh,n_temp_snapshots]: data matrix + - delta: damping ratio of each mode + - freq: frequency of each mode + - f: target frequency + - Q: weighting matrix + Returns: + - U_res: response modes + - S: emergy gains + - V_res: forcing modes + ''' # Variables cdef int m = Phi.shape[0], n = Phi.shape[1] cdef int ii, retval @@ -252,7 +276,19 @@ def _zrun(np.complex128_t[:,:] Phi, double[:] delta, double[:] freq, double f, d return U_res, S, V_res def run(real_complex[:,:] Phi, real[:] delta, real[:] freq, real f, real[:] Q=None): - + ''' + Resolvent Analysis of snapshot matrix X + Inputs: + - X[ndims*nmesh,n_temp_snapshots]: data matrix + - delta: damping ratio of each mode + - freq: frequency of each mode + - f: target frequency + - Q: weighting matrix + Returns: + - U_res: response modes + - S: emergy gains + - V_res: forcing modes + ''' if real_complex is np.complex128_t: return _zrun(Phi, delta, freq, f, Q) else: From c360aac19b61296764286354cb4550a078adbd9f Mon Sep 17 00:00:00 2001 From: msilvestrec03 Date: Fri, 29 May 2026 12:50:13 +0200 Subject: [PATCH 12/27] documentation plots --- pyLOM/RES/plots.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/pyLOM/RES/plots.py b/pyLOM/RES/plots.py index fe33594a..a7492406 100644 --- a/pyLOM/RES/plots.py +++ b/pyLOM/RES/plots.py @@ -42,11 +42,12 @@ def plotEnergy(S,fig=None,ax=None): def plotEvW(S, omega, modes, fig=None,ax=None): r''' - Plot the cummulative energy of a set of modes + Plot the energy gain for different frequencies Args: S (np.ndarray): array containing the energy gains from the RES modes (columns) at diferent frequencies (rows) omega (np.ndarray): array containing the frequencies from the respective energy gains + modes (np.ndarray): modes to plot fig (plt.figure, optional): figure object in which the plot will be done (default: ``[]``) axs (plt.axes, optional): axes object in which the plot will be done (default: ``[]``) From ca2e75c8b42a624072d9577bfde5723cf667caab Mon Sep 17 00:00:00 2001 From: msilvestrec03 Date: Fri, 29 May 2026 15:18:18 +0200 Subject: [PATCH 13/27] options --- options.cfg | 47 +++++++++++++++++++++++++++++++---------------- 1 file changed, 31 insertions(+), 16 deletions(-) diff --git a/options.cfg b/options.cfg index ec42117c..0c4bd26b 100644 --- a/options.cfg +++ b/options.cfg @@ -1,35 +1,50 @@ +# Compile PYLOM +# Compile with g++ or Intel C++ Compiler +# Compile with the most aggressive optimization setting (O3) +# Use the most pedantic compiler settings: must compile with no warnings at all +# +# The user may override any desired internal variable by redefining it via command-line: +# make CXX=g++ [...] +# make OPTL=-O2 [...] +# make FLAGS="-Wall -g" [...] +# +# Arnau Miro 2021 + ## Options # -PLATFORM = MN5_GPP -VECTORIZATION = ON -OPENMP_PARALL = OFF -USE_MKL = ON -USE_FFTW = ON -USE_GCC = OFF -USE_NVHPC = OFF -DEBUGGING = OFF -USE_GESVD = OFF -USE_COMPILED = ON +PLATFORM = PC +VECTORIZATION = ON +OPENMP_PARALL = OFF +USE_MKL = ON +USE_FFTW = OFF +USE_GCC = OFF +USE_NVHPC = OFF +DEBUGGING = OFF +USE_GESVD = OFF +USE_COMPILED = ON # Comma separated list of the modules to be compiled # if USE_COMPILED = ON -MODULES_COMPILED = MATH.MATHS,MATH.AVERAGING,MATH.SVD,MATH.FFT,MATH.GEOMETRIC,MATH.TRUNCATION,MATH.STATS,MATH.REGRESSION,ROM.POD,ROM.DMD,ROM.SPOD,ROM.RES +MODULES_COMPILED = MATH.MATHS,MATH.AVERAGING,MATH.QR,MATH.SVD,MATH.FFT,MATH.GEOMETRIC,MATH.TRUNCATION,MATH.STATS,MATH.REGRESSION,ROM.POD,ROM.DMD,ROM.SPOD,ROM.RES + ## Optimization, host and CPU type # OPTL = 3 HOST = Host -TUNE = sapphirerapids +TUNE = skylake + ## Python versions # PYTHON = python3 PIP = pip3 + ## Versions of the libraries # +ONEAPI_VERS = 2024.2.0.634 +OPENBLAS_VERS = 0.3.17 LAPACK_VERS = 3.9.0 -FFTW_VERS = 3.3.10 -NFFT_VERS = 3.5.2 KISSFFT_VERS = 131.1.0 -OPENBLAS_VERS = 0.3.17 -ONEAPI_VERS = 2023.2.0 +FFTW_VERS = 3.3.8 +NFFT_VERS = 3.5.2 From 79ea8a57d345a82ae2cbbe25d7e7c8f56dc7040c Mon Sep 17 00:00:00 2001 From: msilvestrec03 Date: Tue, 2 Jun 2026 14:29:10 +0200 Subject: [PATCH 14/27] =?UTF-8?q?=C3=82example=20jet?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- Examples/RES/example_RES_jet.py | 53 ++++++++++++++++++++ Examples/RES/example_RES_plots_jet.py | 70 +++++++++++++++++++++++++++ 2 files changed, 123 insertions(+) create mode 100644 Examples/RES/example_RES_jet.py create mode 100644 Examples/RES/example_RES_plots_jet.py diff --git a/Examples/RES/example_RES_jet.py b/Examples/RES/example_RES_jet.py new file mode 100644 index 00000000..e7eec6f8 --- /dev/null +++ b/Examples/RES/example_RES_jet.py @@ -0,0 +1,53 @@ +from __future__ import print_function, division + +import mpi4py +mpi4py.rc.recv_mprobe = False + +import os, numpy as np +import pyLOM + + +# Parameters +DATAFILE = './DATA/jetLES.h5' +VARIABLES = 'PRESS' + +param = 2 * np.pi # Define the normalization parameter for the frequency (in St) +f = 0.8 # Desired frequency (in St) +n_modes = 5 # Desired number of modes to save +modes = np.arange(1,n_modes+1,dtype=np.int32) + + +# Load the mesh +m = pyLOM.Mesh.load(DATAFILE) +pyLOM.pprint(0,'mesh loaded', flush=True) + + +# Load the dataset +d = pyLOM.Dataset.load(DATAFILE,ptable=m.partition_table) +X = d[VARIABLES] +t = d.get_variable('time') +dt = t[1] - t[0] +pyLOM.pprint(0,'dataset loaded', flush=True) + + +# Compute the DMD of the case +muReal, muImag, Phi, bJov = pyLOM.DMD.run(X, r=2*10**-1, remove_mean=True) +delta, omega = pyLOM.DMD.frequency_damping(muReal,muImag,dt) +freq = omega / param + + +# Compute the Resolvent Analysis +U, S, V = pyLOM.RES.run(Phi, delta, freq, f=f, Q=None) + + +# Extract the desired modes +U2, V2 = pyLOM.RES.extract_modes(U,V,1,len(d),modes=modes,kind='real') # kind can be 'real', 'imag' or 'abs' + + +# Write the modes to be visualized with paraview +d.add_field(f'forcing_modes',len(modes),V2) +d.add_field(f'response_modes',len(modes),U2) +pyLOM.io.pv_writer(m,d,f'modes_{f}',basedir='./',instants=[0],times=[0.],vars=[f'forcing_modes',f'response_modes'],fmt='vtkh5') + + +pyLOM.cr_info() diff --git a/Examples/RES/example_RES_plots_jet.py b/Examples/RES/example_RES_plots_jet.py new file mode 100644 index 00000000..9ec32404 --- /dev/null +++ b/Examples/RES/example_RES_plots_jet.py @@ -0,0 +1,70 @@ +from __future__ import print_function, division + +import mpi4py +mpi4py.rc.recv_mprobe = False + +import os, numpy as np +import matplotlib.pyplot as plt +import pyLOM + +pyLOM.gpu_device(gpu_per_node=4) + + +# Parameters +DATAFILE = './DATA/jetLES.h5' +VARIABLE = 'PRESS' + +param = 2 * np.pi # Define the normalization parameter for the frequency (in St) +f_list = [0.2, 0.4, 0.6, 0.8, 1.0] # Desired frequencies (in St) +n_modes = 5 # Desired number of modes to save +modes = np.arange(1,n_modes+1,dtype=np.int32) + + +# Load the mesh +m = pyLOM.Mesh.load(DATAFILE) +pyLOM.pprint(0,'mesh loaded', flush=True) + + +# Load the dataset +d = pyLOM.Dataset.load(DATAFILE,ptable=m.partition_table).to_gpu([VARIABLE]) +X = d[VARIABLE] +t = d.get_variable('time') +dt = t[1] - t[0] +pyLOM.pprint(0,'dataset loaded', flush=True) + + +# Compute the DMD of the case +muReal, muImag, Phi, bJov = pyLOM.DMD.run(X, r=2*10**-1, remove_mean=True) +delta, omega = pyLOM.DMD.frequency_damping(muReal,muImag,dt) +freq = omega / param + + +S_list = np.empty((0,Phi.shape[1])) +# Compute the Resolvent Analysis for every frequency +for f in f_list: + U, S, V = pyLOM.RES.run(Phi, delta, freq, f=f, Q=None) + S_list = np.vstack([S_list, S]) + + # Extract the desired modes + U2, V2 = pyLOM.RES.extract_modes(U,V,1,len(d),modes=modes,kind='real') # kind can be 'real', 'imag' or 'abs' + + + # Write the modes to be visualized with paraview + d.add_field(f'forcing_modes',len(modes),V2) + d.add_field(f'response_modes',len(modes),U2) + pyLOM.io.pv_writer(m,d.to_cpu(['forcing_modes','response_modes']),f'modes_{f}',basedir='./',instants=[0],times=[0.],vars=['forcing_modes','response_modes'],fmt='vtkh5') + + +if pyLOM.utils.is_rank_or_serial(0): + # Plot the cumulative energy gains + pyLOM.RES.plotEnergy(S_list[0,:]) + plt.savefig('energy.png', dpi=300) + + # Plot the energy gains vs frequency for the desired modes + pyLOM.RES.plotEvW(S_list, f_list, modes) + plt.savefig('EvW.png', dpi=300) + + + +pyLOM.cr_info() +pyLOM.show_plots() From 44d34b074f14fa9a9a89ea89f2264591bdfc5871 Mon Sep 17 00:00:00 2001 From: msilvestrec03 Date: Mon, 8 Jun 2026 12:16:27 +0200 Subject: [PATCH 15/27] =?UTF-8?q?=C2=B7bug=20in=20Phi=5Faux?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- Makefile | 3 ++- pyLOM/RES/wrapper.pyx | 15 +++------------ 2 files changed, 5 insertions(+), 13 deletions(-) diff --git a/Makefile b/Makefile index bc75fd85..cdd5f710 100644 --- a/Makefile +++ b/Makefile @@ -262,13 +262,14 @@ clean: -@cd pyLOM; rm -rf POD/__pycache__ POD/*.c POD/*.cpp POD/*.html -@cd pyLOM; rm -rf DMD/__pycache__ DMD/*.c DMD/*.cpp DMD/*.html -@cd pyLOM; rm -rf SPOD/__pycache__ SPOD/*.c SPOD/*.cpp SPOD/*.html + -@cd pyLOM; rm -rf RES/__pycache__ RES/*.c RES/*.cpp RES/*.html -@cd pyLOM; rm -rf vmmath/__pycache__ vmmath/*.c vmmath/*.cpp vmmath/*.html -@cd pyLOM; rm -rf inp_out/__pycache__ inp_out/*.c inp_out/*.cpp inp_out/*.html -@cd pyLOM; rm -rf NN/__pycache__ NN/architectures/__pycache__ cleanall: clean -@rm -rf build - -@cd pyLOM; rm vmmath/*.so POD/*.so DMD/*.so SPOD/*.so + -@cd pyLOM; rm vmmath/*.so POD/*.so DMD/*.so SPOD/*.so RES/*.so ifeq ($(USE_MKL),ON) uninstall_vector_matrix: uninstall_mkl diff --git a/pyLOM/RES/wrapper.pyx b/pyLOM/RES/wrapper.pyx index 86408e44..5cb81b2d 100644 --- a/pyLOM/RES/wrapper.pyx +++ b/pyLOM/RES/wrapper.pyx @@ -76,8 +76,8 @@ def _crun(np.complex64_t[:,:] Phi, float[:] delta, float[:] freq, float f, float cdef np.complex64_t *Fhat Phi_dagger = malloc(n*m*sizeof(np.complex64_t)) Fhat = malloc(n*n*sizeof(np.complex64_t)) - Phi_aux = malloc(n*n*sizeof(np.complex64_t)) - cdef np.ndarray[np.complex64_t,ndim=1] Q_aux = np.zeros((m),dtype=np.complex64) + Phi_aux = malloc(m*n*sizeof(np.complex64_t)) + cdef np.ndarray[np.complex64_t, ndim=1] Q_aux = np.zeros((m),dtype=np.complex64) c_cdagger(&Phi[0,0], Phi_dagger, m, n) if Q is None: c_cmatmulp(Fhat, Phi_dagger, &Phi[0,0], n, n, m) @@ -132,12 +132,8 @@ def _crun(np.complex64_t[:,:] Phi, float[:] delta, float[:] freq, float f, float # Compute the projection cr_start('RES.projection', 0) - # cdef np.complex64_t *U_res - # cdef np.complex64_t *V_res cdef np.complex64_t *U_aux cdef np.complex64_t *V_aux - # U_res = malloc(m*n*sizeof(np.complex64_t)) - # V_res = malloc(m*n*sizeof(np.complex64_t)) U_aux = malloc(m*n*sizeof(np.complex64_t)) V_aux = malloc(m*n*sizeof(np.complex64_t)) cdef np.ndarray[np.complex64_t,ndim=2] U_res = np.zeros((m,n),dtype=np.complex64) @@ -196,7 +192,7 @@ def _zrun(np.complex128_t[:,:] Phi, double[:] delta, double[:] freq, double f, d cdef np.complex128_t *Fhat Phi_dagger = malloc(n*m*sizeof(np.complex128_t)) Fhat = malloc(n*n*sizeof(np.complex128_t)) - Phi_aux = malloc(n*n*sizeof(np.complex128_t)) + Phi_aux = malloc(m*n*sizeof(np.complex128_t)) cdef np.ndarray[np.complex128_t,ndim=1] Q_aux = np.zeros((m),dtype=np.complex128) c_zdagger(&Phi[0,0], Phi_dagger, m, n) if Q is None: @@ -252,12 +248,8 @@ def _zrun(np.complex128_t[:,:] Phi, double[:] delta, double[:] freq, double f, d # Compute the projection cr_start('RES.projection', 0) - # cdef np.complex128_t *U_res - # cdef np.complex128_t *V_res cdef np.complex128_t *U_aux cdef np.complex128_t *V_aux - # U_res = malloc(m*n*sizeof(np.complex128_t)) - # V_res = malloc(m*n*sizeof(np.complex128_t)) U_aux = malloc(m*n*sizeof(np.complex128_t)) V_aux = malloc(m*n*sizeof(np.complex128_t)) cdef np.ndarray[np.complex128_t,ndim=2] U_res = np.zeros((m,n),dtype=np.complex128) @@ -293,4 +285,3 @@ def run(real_complex[:,:] Phi, real[:] delta, real[:] freq, real f, real[:] Q=No return _zrun(Phi, delta, freq, f, Q) else: return _crun(Phi, delta, freq, f, Q) - # return _crun(Phi, delta, freq, f, Q) \ No newline at end of file From a92e0bd9f68eba293cd9afbc42c0756ab158356c Mon Sep 17 00:00:00 2001 From: msilvestrec03 Date: Thu, 18 Jun 2026 12:12:16 +0200 Subject: [PATCH 16/27] cholesky update --- pyLOM/RES/wrapper.py | 2 +- pyLOM/RES/wrapper.pyx | 34 +++++++++++++++++++++------------- 2 files changed, 22 insertions(+), 14 deletions(-) diff --git a/pyLOM/RES/wrapper.py b/pyLOM/RES/wrapper.py index ce8ac88e..d7362d7a 100644 --- a/pyLOM/RES/wrapper.py +++ b/pyLOM/RES/wrapper.py @@ -44,7 +44,7 @@ def run(Phi, delta, freq, f, Q=None): else: Qhat = matmulp(dagger(Phi), vecmat(Q, Phi)) - Fhat = cholesky(Qhat) + Fhat = dagger(cholesky(Qhat)) Fhat_inv = inv(Fhat) Hhat = matmul(Fhat, vecmat(H, Fhat_inv)) diff --git a/pyLOM/RES/wrapper.pyx b/pyLOM/RES/wrapper.pyx index 5cb81b2d..556b6a98 100644 --- a/pyLOM/RES/wrapper.pyx +++ b/pyLOM/RES/wrapper.pyx @@ -69,36 +69,40 @@ def _crun(np.complex64_t[:,:] Phi, float[:] delta, float[:] freq, float f, float free(Omega) cr_stop('RES.resolvent_operator', 0) - # Compute the Qhat (named Fhat for convenience) + # Compute the Qhat (named Fhat_dagger for convenience) cr_start('RES.Qhat', 0) cdef np.complex64_t *Phi_dagger cdef np.complex64_t *Phi_aux - cdef np.complex64_t *Fhat + cdef np.complex64_t *Fhat_dagger Phi_dagger = malloc(n*m*sizeof(np.complex64_t)) - Fhat = malloc(n*n*sizeof(np.complex64_t)) Phi_aux = malloc(m*n*sizeof(np.complex64_t)) + Fhat_dagger = malloc(n*n*sizeof(np.complex64_t)) cdef np.ndarray[np.complex64_t, ndim=1] Q_aux = np.zeros((m),dtype=np.complex64) c_cdagger(&Phi[0,0], Phi_dagger, m, n) if Q is None: - c_cmatmulp(Fhat, Phi_dagger, &Phi[0,0], n, n, m) + c_cmatmulp(Fhat_dagger, Phi_dagger, &Phi[0,0], n, n, m) else: memcpy(Phi_aux, &Phi[0,0], m*n*sizeof(np.complex64_t)) Q_aux = np.array(Q, dtype=np.complex64) c_cvecmat(&Q_aux[0], Phi_aux, m, n) # Phi_aux is overwritten - c_cmatmulp(Fhat, Phi_dagger, Phi_aux, n, n, m) + c_cmatmulp(Fhat_dagger, Phi_dagger, Phi_aux, n, n, m) free(Phi_aux) free(Phi_dagger) cr_stop('RES.Qhat', 0) # Compute the Choleski decomposition cr_start('RES.Choleski', 0) - retval = c_ccholesky(Fhat, n) + cdef np.complex64_t *Fhat + Fhat = malloc(n*n*sizeof(np.complex64_t)) + retval = c_ccholesky(Fhat_dagger, n) if not retval == 0: raiseError('Problems computing Cholesky factorization!') + c_cdagger(Fhat_dagger, Fhat, n, n) cdef np.complex64_t *Fhat_inv Fhat_inv = malloc(n*n*sizeof(np.complex64_t)) memcpy(Fhat_inv, Fhat, n*n*sizeof(np.complex64_t)) - retval = c_cinverse(Fhat_inv, n, 'L') + retval = c_cinverse(Fhat_inv, n, 'U') if not retval == 0: raiseError('Problems computing the Inverse!') + free(Fhat_dagger) cr_stop('RES.Choleski', 0) # Compute Hhat @@ -189,31 +193,35 @@ def _zrun(np.complex128_t[:,:] Phi, double[:] delta, double[:] freq, double f, d cr_start('RES.Qhat', 0) cdef np.complex128_t *Phi_dagger cdef np.complex128_t *Phi_aux - cdef np.complex128_t *Fhat + cdef np.complex128_t *Fhat_dagger Phi_dagger = malloc(n*m*sizeof(np.complex128_t)) - Fhat = malloc(n*n*sizeof(np.complex128_t)) Phi_aux = malloc(m*n*sizeof(np.complex128_t)) + Fhat_dagger = malloc(n*n*sizeof(np.complex128_t)) cdef np.ndarray[np.complex128_t,ndim=1] Q_aux = np.zeros((m),dtype=np.complex128) c_zdagger(&Phi[0,0], Phi_dagger, m, n) if Q is None: - c_zmatmulp(Fhat, Phi_dagger, &Phi[0,0], n, n, m) + c_zmatmulp(Fhat_dagger, Phi_dagger, &Phi[0,0], n, n, m) else: memcpy(Phi_aux, &Phi[0,0], m*n*sizeof(np.complex128_t)) Q_aux = np.array(Q, dtype=np.complex128) c_zvecmat(&Q_aux[0], Phi_aux, m, n) # Phi_aux is overwritten - c_zmatmulp(Fhat, Phi_dagger, Phi_aux, n, n, m) + c_zmatmulp(Fhat_dagger, Phi_dagger, Phi_aux, n, n, m) free(Phi_aux) free(Phi_dagger) + free(Fhat_dagger) cr_stop('RES.Qhat', 0) # Compute the Choleski decomposition cr_start('RES.Choleski', 0) - retval = c_zcholesky(Fhat, n) + cdef np.complex128_t *Fhat + Fhat = malloc(n*n*sizeof(np.complex128_t)) + retval = c_zcholesky(Fhat_dagger, n) if not retval == 0: raiseError('Problems computing Cholesky factorization!') + c_zdagger(Fhat_dagger, Fhat, n, n) cdef np.complex128_t *Fhat_inv Fhat_inv = malloc(n*n*sizeof(np.complex128_t)) memcpy(Fhat_inv, Fhat, n*n*sizeof(np.complex128_t)) - retval = c_zinverse(Fhat_inv, n, 'L') + retval = c_zinverse(Fhat_inv, n, 'U') if not retval == 0: raiseError('Problems computing the Inverse!') cr_stop('RES.Choleski', 0) From ef9d0620ad9870faeb8b3313df642db32034d7ea Mon Sep 17 00:00:00 2001 From: msilvestrec03 Date: Sat, 20 Jun 2026 12:32:40 +0200 Subject: [PATCH 17/27] time RES --- pyLOM/RES/wrapper.pyx | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/pyLOM/RES/wrapper.pyx b/pyLOM/RES/wrapper.pyx index 556b6a98..33c8cffb 100644 --- a/pyLOM/RES/wrapper.pyx +++ b/pyLOM/RES/wrapper.pyx @@ -208,7 +208,6 @@ def _zrun(np.complex128_t[:,:] Phi, double[:] delta, double[:] freq, double f, d c_zmatmulp(Fhat_dagger, Phi_dagger, Phi_aux, n, n, m) free(Phi_aux) free(Phi_dagger) - free(Fhat_dagger) cr_stop('RES.Qhat', 0) # Compute the Choleski decomposition @@ -223,6 +222,7 @@ def _zrun(np.complex128_t[:,:] Phi, double[:] delta, double[:] freq, double f, d memcpy(Fhat_inv, Fhat, n*n*sizeof(np.complex128_t)) retval = c_zinverse(Fhat_inv, n, 'U') if not retval == 0: raiseError('Problems computing the Inverse!') + free(Fhat_dagger) cr_stop('RES.Choleski', 0) # Compute Hhat @@ -275,6 +275,12 @@ def _zrun(np.complex128_t[:,:] Phi, double[:] delta, double[:] freq, double f, d return U_res, S, V_res +@cr('RES.run') +@cython.initializedcheck(False) +@cython.boundscheck(False) # turn off bounds-checking for entire function +@cython.wraparound(False) # turn off negative index wrapping for entire function +@cython.nonecheck(False) +@cython.cdivision(True) # turn off zero division check def run(real_complex[:,:] Phi, real[:] delta, real[:] freq, real f, real[:] Q=None): ''' Resolvent Analysis of snapshot matrix X From 975ff33812cfb8bb059b67884103e0928f5297ba Mon Sep 17 00:00:00 2001 From: msilvestrec03 Date: Wed, 19 Aug 2026 08:37:36 +0200 Subject: [PATCH 18/27] commented changehs --- Examples/RES/example_RES_jet.py | 2 +- Examples/RES/example_RES_plots_jet.py | 2 +- pyLOM/RES/wrapper.py | 25 +++---- pyLOM/RES/wrapper.pyx | 22 +++--- pyLOM/__init__.py | 2 +- pyLOM/inp_out/io_h5.py | 99 ++++++++++++--------------- pyLOM/vmmath/src/vector_matrix.c | 44 ++++-------- 7 files changed, 77 insertions(+), 119 deletions(-) diff --git a/Examples/RES/example_RES_jet.py b/Examples/RES/example_RES_jet.py index e7eec6f8..89ab9826 100644 --- a/Examples/RES/example_RES_jet.py +++ b/Examples/RES/example_RES_jet.py @@ -31,7 +31,7 @@ # Compute the DMD of the case -muReal, muImag, Phi, bJov = pyLOM.DMD.run(X, r=2*10**-1, remove_mean=True) +muReal, muImag, Phi, bJov = pyLOM.DMD.run(X, r=2e-1, remove_mean=True) delta, omega = pyLOM.DMD.frequency_damping(muReal,muImag,dt) freq = omega / param diff --git a/Examples/RES/example_RES_plots_jet.py b/Examples/RES/example_RES_plots_jet.py index 9ec32404..cc95850e 100644 --- a/Examples/RES/example_RES_plots_jet.py +++ b/Examples/RES/example_RES_plots_jet.py @@ -34,7 +34,7 @@ # Compute the DMD of the case -muReal, muImag, Phi, bJov = pyLOM.DMD.run(X, r=2*10**-1, remove_mean=True) +muReal, muImag, Phi, bJov = pyLOM.DMD.run(X, r=2e-1, remove_mean=True) delta, omega = pyLOM.DMD.frequency_damping(muReal,muImag,dt) freq = omega / param diff --git a/pyLOM/RES/wrapper.py b/pyLOM/RES/wrapper.py index d7362d7a..9f8b77fc 100644 --- a/pyLOM/RES/wrapper.py +++ b/pyLOM/RES/wrapper.py @@ -14,13 +14,13 @@ from ..utils import cr_nvtx as cr, cr_start, cr_stop @cr('RES.run') -def run(Phi, delta, freq, f, Q=None): +def run(Phi, delta, omega, f, Q=None): ''' Resolvent Analysis of snapshot matrix X Inputs: - X[ndims*nmesh,n_temp_snapshots]: data matrix - delta: damping ratio of each mode - - freq: frequency of each mode + - omega: frequency of each mode - f: target frequency - Q: weighting matrix Returns: @@ -30,14 +30,11 @@ def run(Phi, delta, freq, f, Q=None): ''' p = cp if type(Phi) is cp.ndarray else np - # Normalization of modes (?) - # Matrix = transpose(vecmat(bJov, transpose(Phi))) - # for ii in range(len(Matrix[0,:])): - # Matrix[:,ii] = Matrix[:,ii] / vector_norm(Phi[:,ii]) - - Omega = delta + 1j * freq + # Compute the resolvent operator + Omega = delta + 1j * omega H = 1 / (-1j * f - Omega) - + + # Compute the metric if Q is None: Qhat = matmulp(dagger(Phi), Phi) @@ -47,17 +44,13 @@ def run(Phi, delta, freq, f, Q=None): Fhat = dagger(cholesky(Qhat)) Fhat_inv = inv(Fhat) + # Solve the optimization problem Hhat = matmul(Fhat, vecmat(H, Fhat_inv)) U, S, VT = svd(Hhat) V = dagger(VT) + # Project the solution U_res = matmul(Phi, matmul(Fhat_inv, U)) V_res = matmul(Phi, matmul(Fhat_inv, V)) - - # # U, S, VT = svd(diag(H)) - # # V = dagger(VT) - - # # U_res = matmul(Phi,U) - # # V_res = matmul(Phi,V) - return U_res, S, V_res + return U_res, S, V_res \ No newline at end of file diff --git a/pyLOM/RES/wrapper.pyx b/pyLOM/RES/wrapper.pyx index 33c8cffb..08325c50 100644 --- a/pyLOM/RES/wrapper.pyx +++ b/pyLOM/RES/wrapper.pyx @@ -25,8 +25,6 @@ from libc.stdlib cimport malloc, free from libc.string cimport memcpy, memset from libc.math cimport sqrt, log, atan2 from ..vmmath.cfuncs cimport real, real_complex -# from ..vmmath.cfuncs cimport -# from ..vmmath.cfuncs cimport from ..vmmath.cfuncs cimport c_csvd, c_cdagger, c_cmatmul, c_cmatmulp, c_cvecmat, c_ccholesky, c_cinverse from ..vmmath.cfuncs cimport c_zsvd, c_zdagger, c_zmatmul, c_zmatmulp, c_zvecmat, c_zcholesky, c_zinverse @@ -39,13 +37,13 @@ from ..utils.errors import raiseError @cython.wraparound(False) # turn off negative index wrapping for entire function @cython.nonecheck(False) @cython.cdivision(True) # turn off zero division check -def _crun(np.complex64_t[:,:] Phi, float[:] delta, float[:] freq, float f, float[:] Q=None): +def _crun(np.complex64_t[:,:] Phi, float[:] delta, float[:] omega, float f, float[:] Q=None): ''' Resolvent Analysis of snapshot matrix X Inputs: - X[ndims*nmesh,n_temp_snapshots]: data matrix - delta: damping ratio of each mode - - freq: frequency of each mode + - omega: frequency of each mode - f: target frequency - Q: weighting matrix Returns: @@ -64,7 +62,7 @@ def _crun(np.complex64_t[:,:] Phi, float[:] delta, float[:] freq, float f, float H = malloc(n*sizeof(np.complex64_t)) cr_start('RES.resolvent_operator', 0) for ii in range(n): - Omega[ii] = delta[ii] + J * freq[ii] + Omega[ii] = delta[ii] + J * omega[ii] H[ii] = 1 / (-J * f - Omega[ii]) free(Omega) cr_stop('RES.resolvent_operator', 0) @@ -159,13 +157,13 @@ def _crun(np.complex64_t[:,:] Phi, float[:] delta, float[:] freq, float f, float @cython.wraparound(False) # turn off negative index wrapping for entire function @cython.nonecheck(False) @cython.cdivision(True) # turn off zero division check -def _zrun(np.complex128_t[:,:] Phi, double[:] delta, double[:] freq, double f, double[:] Q=None): +def _zrun(np.complex128_t[:,:] Phi, double[:] delta, double[:] omega, double f, double[:] Q=None): ''' Resolvent Analysis of snapshot matrix X Inputs: - X[ndims*nmesh,n_temp_snapshots]: data matrix - delta: damping ratio of each mode - - freq: frequency of each mode + - omega: frequency of each mode - f: target frequency - Q: weighting matrix Returns: @@ -184,7 +182,7 @@ def _zrun(np.complex128_t[:,:] Phi, double[:] delta, double[:] freq, double f, d H = malloc(n*sizeof(np.complex128_t)) cr_start('RES.resolvent_operator', 0) for ii in range(n): - Omega[ii] = delta[ii] + J * freq[ii] + Omega[ii] = delta[ii] + J * omega[ii] H[ii] = 1 / (-J * f - Omega[ii]) free(Omega) cr_stop('RES.resolvent_operator', 0) @@ -281,13 +279,13 @@ def _zrun(np.complex128_t[:,:] Phi, double[:] delta, double[:] freq, double f, d @cython.wraparound(False) # turn off negative index wrapping for entire function @cython.nonecheck(False) @cython.cdivision(True) # turn off zero division check -def run(real_complex[:,:] Phi, real[:] delta, real[:] freq, real f, real[:] Q=None): +def run(real_complex[:,:] Phi, real[:] delta, real[:] omega, real f, real[:] Q=None): ''' Resolvent Analysis of snapshot matrix X Inputs: - X[ndims*nmesh,n_temp_snapshots]: data matrix - delta: damping ratio of each mode - - freq: frequency of each mode + - omega: frequency of each mode - f: target frequency - Q: weighting matrix Returns: @@ -296,6 +294,6 @@ def run(real_complex[:,:] Phi, real[:] delta, real[:] freq, real f, real[:] Q=No - V_res: forcing modes ''' if real_complex is np.complex128_t: - return _zrun(Phi, delta, freq, f, Q) + return _zrun(Phi, delta, omega, f, Q) else: - return _crun(Phi, delta, freq, f, Q) + return _crun(Phi, delta, omega, f, Q) diff --git a/pyLOM/__init__.py b/pyLOM/__init__.py index 6703cb63..88faec11 100644 --- a/pyLOM/__init__.py +++ b/pyLOM/__init__.py @@ -20,7 +20,7 @@ from .utils.gpu import gpu_device # Import Low Order Models -from . import POD, PCA, DMD, SPOD, MANIFOLD, GPOD, RES +from . import POD, PCA, DMD, SPOD, MANIFOLD, GPOD, RES, LAMINE # Import AI Models # The NN module overloads the memory when loaded diff --git a/pyLOM/inp_out/io_h5.py b/pyLOM/inp_out/io_h5.py index a89f9448..b790f211 100644 --- a/pyLOM/inp_out/io_h5.py +++ b/pyLOM/inp_out/io_h5.py @@ -848,32 +848,64 @@ def h5_load_QR(fname,vars,ptable=None): file.close() return varList -def h5_save_usv(U,S,V,ptable,nvars,pointData,group,kind): +def h5_save_USV(U,S,V,ptable,nvars,pointData,group,projected=False): # Create the datasets for U, S and V group.create_dataset('pointData',(1,),dtype='u1',data=pointData) group.create_dataset('n_variables',(1,),dtype='u1',data=nvars) Usize = (mpi_reduce(U.shape[0],op='sum',all=True),U.shape[1]) if U is not None else None dsetU = group.create_dataset('U',Usize,dtype=U.dtype) if U is not None else None dsetS = group.create_dataset('S',S.shape,dtype=S.dtype) if S is not None else None - if kind == 'POD': + if not projected: dsetV = group.create_dataset('V',V.shape,dtype=V.dtype) if V is not None else None - elif kind == 'RES': + else: Vsize = (mpi_reduce(V.shape[0],op='sum',all=True),V.shape[1]) if V is not None else None dsetV = group.create_dataset('V',Vsize,dtype=V.dtype) if V is not None else None - else: - raise ValueError("kind must be: POD, RES") + # Store S and U that are repeated across the ranks # So it is enough that one rank stores them if is_rank_or_serial(0): if dsetS is not None: dsetS[:] = S - if kind == 'POD': + if not projected: if dsetV is not None: dsetV[:] = V # Store U in parallel istart, iend = ptable.partition_bounds(MPI_RANK,ndim=nvars,points=pointData) if dsetU is not None: dsetU[istart:iend,:] = U - if kind == 'RES': + if projected: if dsetV is not None: dsetV[istart:iend,:] = V +def h5_load_USV(fname,vars,nmod,ptable=None,group='POD',projected=False): + # Load the dataset for U, S, V + file = h5py.File(fname,'r',driver='mpio',comm=MPI_COMM) if not MPI_SIZE == 1 else h5py.File(fname,'r') + # Check the file version + version = tuple(file.attrs['Version']) + if not version == PYLOM_H5_VERSION: + raiseError('File version <%s> not matching the tool version <%s>!'%(str(file.attrs['Version']),str(PYLOM_H5_VERSION))) + # Read the requested variables S, V + varList = [] + if 'U' in vars: + # Check if we need to read the partition table + if ptable is None: ptable = h5_load_partition(file) + # Read + nvars = int(file[group]['n_variables'][0]) + point = bool(file[group]['pointData'][0]) + istart, iend = ptable.partition_bounds(MPI_RANK,ndim=nvars,points=point) + varList.append(np.array(file[group]['U'][istart:iend,:]) if nmod < 0 else np.array(file[group]['U'][istart:iend,:nmod])) + if 'S' in vars: varList.append( np.array(file[group]['S'][:]) if nmod < 0 else np.array(file[group]['S'][:nmod]) ) + if 'V' in vars: + if not projected: + varList.append( np.array(file[group]['V'][:,:]) if nmod < 0 else np.array(file[group]['V'][:nmod,:]) ) + else: + # Check if we need to read the partition table + if ptable is None: ptable = h5_load_partition(file) + # Read + nvars = int(file['RES']['n_variables'][0]) + point = bool(file['RES']['pointData'][0]) + istart, iend = ptable.partition_bounds(MPI_RANK,ndim=nvars,points=point) + varList.append(np.array(file[group]['V'][istart:iend,:]) if nmod < 0 else np.array(file[group]['V'][istart:iend,:nmod])) + # Return + file.close() + return varList + @cr('h5IO.save_POD') def h5_save_POD(fname,U,S,V,ptable,nvars=1,pointData=True,mode='w'): ''' @@ -890,7 +922,7 @@ def h5_save_POD(fname,U,S,V,ptable,nvars=1,pointData=True,mode='w'): # Now create a POD group group = file.create_group('POD') # Create and store de datsets - h5_save_usv(U,S,V,ptable,nvars,pointData,group,kind='POD') + h5_save_USV(U,S,V,ptable,nvars,pointData,group,projected=False) file.close() @cr('h5IO.load_POD') @@ -898,26 +930,7 @@ def h5_load_POD(fname,vars,nmod,ptable=None): ''' Load POD variables from an HDF5 file. ''' - file = h5py.File(fname,'r',driver='mpio',comm=MPI_COMM) if not MPI_SIZE == 1 else h5py.File(fname,'r') - # Check the file version - version = tuple(file.attrs['Version']) - if not version == PYLOM_H5_VERSION: - raiseError('File version <%s> not matching the tool version <%s>!'%(str(file.attrs['Version']),str(PYLOM_H5_VERSION))) - # Read the requested variables S, V - varList = [] - if 'U' in vars: - # Check if we need to read the partition table - if ptable is None: ptable = h5_load_partition(file) - # Read - nvars = int(file['POD']['n_variables'][0]) - point = bool(file['POD']['pointData'][0]) - istart, iend = ptable.partition_bounds(MPI_RANK,ndim=nvars,points=point) - varList.append(np.array(file['POD']['U'][istart:iend,:]) if nmod < 0 else np.array(file['POD']['U'][istart:iend,:nmod])) - if 'S' in vars: varList.append( np.array(file['POD']['S'][:]) if nmod < 0 else np.array(file['POD']['S'][:nmod]) ) - if 'V' in vars: varList.append( np.array(file['POD']['V'][:,:]) if nmod < 0 else np.array(file['POD']['V'][:nmod,:]) ) - # Return - file.close() - return varList + return h5_load_USV(fname,vars,nmod,ptable,group='POD',projected=False) @cr('h5IO.save_DMD') @@ -1076,7 +1089,7 @@ def h5_save_RES(fname,U,S,V,ptable,nvars=1,pointData=True,mode='w'): # Now create a POD group group = file.create_group('RES') # Create and store de datsets - h5_save_usv(U,S,V,ptable,nvars,pointData,group,kind='RES') + h5_save_USV(U,S,V,ptable,nvars,pointData,group,projected=True) file.close() @cr('h5IO.load_RES') @@ -1084,33 +1097,7 @@ def h5_load_RES(fname,vars,nmod,ptable=None): ''' Load RES variables from an HDF5 file. ''' - file = h5py.File(fname,'r',driver='mpio',comm=MPI_COMM) if not MPI_SIZE == 1 else h5py.File(fname,'r') - # Check the file version - version = tuple(file.attrs['Version']) - if not version == PYLOM_H5_VERSION: - raiseError('File version <%s> not matching the tool version <%s>!'%(str(file.attrs['Version']),str(PYLOM_H5_VERSION))) - # Read the requested variables S, V - varList = [] - if 'U' in vars: - # Check if we need to read the partition table - if ptable is None: ptable = h5_load_partition(file) - # Read - nvars = int(file['RES']['n_variables'][0]) - point = bool(file['RES']['pointData'][0]) - istart, iend = ptable.partition_bounds(MPI_RANK,ndim=nvars,points=point) - varList.append(np.array(file['RES']['U'][istart:iend,:]) if nmod < 0 else np.array(file['RES']['U'][istart:iend,:nmod])) - if 'S' in vars: varList.append( np.array(file['RES']['S'][:]) if nmod < 0 else np.array(file['RES']['S'][:nmod]) ) - if 'V' in vars: - # Check if we need to read the partition table - if ptable is None: ptable = h5_load_partition(file) - # Read - nvars = int(file['RES']['n_variables'][0]) - point = bool(file['RES']['pointData'][0]) - istart, iend = ptable.partition_bounds(MPI_RANK,ndim=nvars,points=point) - varList.append(np.array(file['RES']['V'][istart:iend,:]) if nmod < 0 else np.array(file['RES']['V'][istart:iend,:nmod])) - # Return - file.close() - return varList + return h5_load_USV(fname,vars,nmod,ptable,group='RES',projected=True) @cr('io.create_compressed') def h5_create_compressed(fname:str,basedir:str,r:int,nmod:int,nvars:int,nlayers:int,conv_chan:int,kernel:int,nAEsG:int,nptxAE:int,dtype:np.dtype): diff --git a/pyLOM/vmmath/src/vector_matrix.c b/pyLOM/vmmath/src/vector_matrix.c index 2c532ff6..92a98c0b 100644 --- a/pyLOM/vmmath/src/vector_matrix.c +++ b/pyLOM/vmmath/src/vector_matrix.c @@ -1143,15 +1143,10 @@ void sdiag2(float *A, float *B, const int m){ /* Diagonal matrix of an array. */ + memset(B, 0, m*m*sizeof(float)); + for (int ii = 0; ii < m; ii++){ - for (int jj = 0; jj < m; jj++){ - if (ii == jj){ - AC_MAT(B,m,ii,ii) = A[ii]; - } - else{ - AC_MAT(B,m,ii,jj) = 0; - } - } + AC_MAT(B,m,ii,ii) = A[ii]; } } @@ -1159,15 +1154,10 @@ void ddiag2(double *A, double *B, const int m){ /* Diagonal matrix of an array. */ + memset(B, 0, m*m*sizeof(double)); + for (int ii = 0; ii < m; ii++){ - for (int jj = 0; jj < m; jj++){ - if (ii == jj){ - AC_MAT(B,m,ii,ii) = A[ii]; - } - else{ - AC_MAT(B,m,ii,jj) = 0; - } - } + AC_MAT(B,m,ii,ii) = A[ii]; } } @@ -1175,15 +1165,10 @@ void cdiag2(scomplex_t *A, scomplex_t *B, const int m){ /* Diagonal matrix of an array. */ + memset(B, 0, m*m*sizeof(scomplex_t)); + for (int ii = 0; ii < m; ii++){ - for (int jj = 0; jj < m; jj++){ - if (ii == jj){ - AC_MAT(B,m,ii,ii) = A[ii]; - } - else{ - AC_MAT(B,m,ii,jj) = 0; - } - } + AC_MAT(B,m,ii,ii) = A[ii]; } } @@ -1191,14 +1176,9 @@ void zdiag2(dcomplex_t *A, dcomplex_t *B, const int m){ /* Diagonal matrix of an array. */ + memset(B, 0, m*m*sizeof(dcomplex_t)); + for (int ii = 0; ii < m; ii++){ - for (int jj = 0; jj < m; jj++){ - if (ii == jj){ - AC_MAT(B,m,ii,ii) = A[ii]; - } - else{ - AC_MAT(B,m,ii,jj) = 0; - } - } + AC_MAT(B,m,ii,ii) = A[ii]; } } \ No newline at end of file From ba4480586c3369d3c1b55649737a0b2dd3366bc3 Mon Sep 17 00:00:00 2001 From: msilvestrec03 Date: Mon, 14 Sep 2026 16:42:18 +0200 Subject: [PATCH 19/27] resolvent rafel --- .gitignore | 2 + options.cfg | 17 +- pyLOM/DMD/__init__.py | 2 +- pyLOM/DMD/wrapper.py | 41 +- pyLOM/DMD/wrapper.pyx | 905 +++++++++++++++++++++++++++++++++++++- pyLOM/RES/__init__.py | 2 +- pyLOM/RES/wrapper.py | 25 +- pyLOM/RES/wrapper.pyx | 183 +++++++- pyLOM/vmmath/__init__.py | 6 +- pyLOM/vmmath/linear.pxd | 25 ++ pyLOM/vmmath/linear.py | 106 +++++ pyLOM/vmmath/linear.pyx | 471 ++++++++++++++++++++ pyLOM/vmmath/src/linear.c | 114 +++++ pyLOM/vmmath/src/linear.h | 12 + setup.py | 35 +- 15 files changed, 1891 insertions(+), 55 deletions(-) create mode 100644 pyLOM/vmmath/linear.pxd create mode 100644 pyLOM/vmmath/linear.py create mode 100644 pyLOM/vmmath/linear.pyx create mode 100644 pyLOM/vmmath/src/linear.c create mode 100644 pyLOM/vmmath/src/linear.h diff --git a/.gitignore b/.gitignore index 02557f1a..8e9aa4f0 100644 --- a/.gitignore +++ b/.gitignore @@ -40,6 +40,8 @@ pyLOM/DMD/wrapper.c pyLOM/DMD/wrapper.html pyLOM/SPOD/wrapper.c pyLOM/SPOD/wrapper.html +pyLOM/RES/wrapper.c +pyLOM/RES/wrapper.html pyLOM/vmmath/*.c pyLOM/vmmath/*.html pyLOM/inp_out/*.c diff --git a/options.cfg b/options.cfg index 0c4bd26b..66baa273 100644 --- a/options.cfg +++ b/options.cfg @@ -12,11 +12,11 @@ ## Options # -PLATFORM = PC +PLATFORM = MN5_GPP VECTORIZATION = ON OPENMP_PARALL = OFF USE_MKL = ON -USE_FFTW = OFF +USE_FFTW = ON USE_GCC = OFF USE_NVHPC = OFF DEBUGGING = OFF @@ -24,14 +24,14 @@ USE_GESVD = OFF USE_COMPILED = ON # Comma separated list of the modules to be compiled # if USE_COMPILED = ON -MODULES_COMPILED = MATH.MATHS,MATH.AVERAGING,MATH.QR,MATH.SVD,MATH.FFT,MATH.GEOMETRIC,MATH.TRUNCATION,MATH.STATS,MATH.REGRESSION,ROM.POD,ROM.DMD,ROM.SPOD,ROM.RES +MODULES_COMPILED = MATH.MATHS,MATH.AVERAGING,MATH.QR,MATH.SVD,MATH.FFT,MATH.GEOMETRIC,MATH.TRUNCATION,MATH.STATS,MATH.REGRESSION,MATH.LINEAR,MATH.DATAPROCESSING,ROM.POD,ROM.DMD,ROM.SPOD,ROM.RES ## Optimization, host and CPU type # OPTL = 3 HOST = Host -TUNE = skylake +TUNE = sapphirerapids ## Python versions @@ -39,12 +39,11 @@ TUNE = skylake PYTHON = python3 PIP = pip3 - ## Versions of the libraries # -ONEAPI_VERS = 2024.2.0.634 -OPENBLAS_VERS = 0.3.17 LAPACK_VERS = 3.9.0 -KISSFFT_VERS = 131.1.0 -FFTW_VERS = 3.3.8 +FFTW_VERS = 3.3.10 NFFT_VERS = 3.5.2 +KISSFFT_VERS = 131.1.0 +OPENBLAS_VERS = 0.3.17 +ONEAPI_VERS = 2023.2.0 diff --git a/pyLOM/DMD/__init__.py b/pyLOM/DMD/__init__.py index 75f724bc..a1d06676 100644 --- a/pyLOM/DMD/__init__.py +++ b/pyLOM/DMD/__init__.py @@ -7,7 +7,7 @@ # Last rev: 30/09/2021 # Functions coming from DMD -from .wrapper import run, frequency_damping, reconstruction_jovanovic +from .wrapper import run, frequency_damping, reconstruction_jovanovic, run_new from .utils import extract_modes, save, load from .plots import plotMode, ritzSpectrum, amplitudeFrequency, dampingFrequency, plotResidual, plotSnapshot diff --git a/pyLOM/DMD/wrapper.py b/pyLOM/DMD/wrapper.py index c082e067..17b00202 100644 --- a/pyLOM/DMD/wrapper.py +++ b/pyLOM/DMD/wrapper.py @@ -10,8 +10,7 @@ import numpy as np from ..utils.gpu import cp -from ..vmmath import vecmat, matmul, temporal_mean, subtract_mean, tsqr_svd, transpose, eigen, cholesky, diag, polar, vandermonde, conj, inv, flip, matmulp, vandermondeTime -from ..POD import truncate +from ..vmmath import matmul, temporal_mean, subtract_mean, transpose, eigen, cholesky, diag, polar, vandermonde, conj, inv, flip, vandermondeTime, linear_operator, concatenate, separate from ..utils import cr_nvtx as cr, cr_start, cr_stop @@ -60,32 +59,16 @@ def run(X, r, remove_mean = True): - b: Amplitude of the DMD modes - X_DMD: Reconstructed flow ''' - # Remove temporal mean or not, depending on the user choice - if remove_mean: - cr_start('DMD.temporal_mean',0) - #Compute temporal mean - X_mean = temporal_mean(X) - #Subtract temporal mean - Y = subtract_mean(X, X_mean) - cr_stop('DMD.temporal_mean',0) + # Prepare matrices and calculate the linear operator + if (type(X) is list): + Y, Z = concatenate(X, remove_mean=remove_mean) + U, S, VT, Atilde = linear_operator(Y, Z, r) + else: - Y = X.copy() - - # Compute SVD - cr_start('DMD.SVD',0) - U, S, VT = tsqr_svd(Y[:, :-1]) - cr_stop('DMD.SVD',0) - # Truncate according to residual - cr_start('DMD.truncate', 0) - U, S, VT = truncate(U, S, VT, r) - cr_stop('DMD.truncate', 0) - - # Project A (Jacobian of the snapshots) into POD basis - cr_start('DMD.linear_mapping',0) - aux1 = matmulp(transpose(U), Y[:, 1:]) - aux2 = transpose(vecmat(1./S, VT)) - Atilde = matmul(aux1, aux2) - cr_stop('DMD.linear_mapping',0) + Y, Z = separate(X, remove_mean=remove_mean) + U, S, VT, Atilde = linear_operator(Y, Z, r) + + del U # Eigendecomposition of Atilde: Eigenvectors given as complex matrix # NOTE: there is no implementation of eig in cupy yet @@ -96,12 +79,12 @@ def run(X, r, remove_mean = True): w = cp.asarray(w) if type(Atilde) is cp.ndarray else w # Mode computation - Phi = matmul(matmul(matmul(Y[:, 1:], transpose(VT)), diag(1/S)), w)/(muReal + muImag*1J) + Phi = matmul(matmul(matmul(Z, transpose(VT)), diag(1/S)), w)/(muReal + muImag*1J) cr_stop('DMD.modes',0) # Amplitudes according to: Jovanovic et. al. 2014 DOI: 10.1063 cr_start('DMD.amplitudes',0) - Vand = vandermonde(muReal, muImag, muReal.shape[0], Y.shape[1]-1) + Vand = vandermonde(muReal, muImag, muReal.shape[0], Y.shape[1]) P = matmul(transpose(conj(w)), w)*conj(matmul(Vand, transpose(conj(Vand)))) Pl = cholesky(P) G = matmul(diag(S), VT) diff --git a/pyLOM/DMD/wrapper.pyx b/pyLOM/DMD/wrapper.pyx index 14c611c1..af43b7b4 100644 --- a/pyLOM/DMD/wrapper.pyx +++ b/pyLOM/DMD/wrapper.pyx @@ -29,6 +29,9 @@ from ..vmmath.cfuncs cimport c_stranspose, c_smatmul, c_smatmulp, c_svecmat, c_s from ..vmmath.cfuncs cimport c_dtranspose, c_dmatmul, c_dmatmulp, c_dvecmat, c_dtemporal_mean, c_dsubtract_mean, c_dtsqr_svd, c_dsvd, c_dcompute_truncation_residual, c_dcompute_truncation from ..vmmath.cfuncs cimport c_cmatmult, c_cvecmat, c_cinverse, c_ccholesky, c_ceigen, c_cvandermonde, c_cvandermonde_time, c_csort from ..vmmath.cfuncs cimport c_zmatmult, c_zvecmat, c_zinverse, c_zcholesky, c_zeigen, c_zvandermonde, c_zvandermonde_time, c_zsort +from ..vmmath.linear cimport _sconcatenate, _slinear_operator, _sseparate +from ..vmmath.linear cimport _dconcatenate, _dlinear_operator, _dseparate + from ..utils.cr import cr, cr_start, cr_stop from ..utils.errors import raiseError @@ -313,6 +316,7 @@ def _srun(float[:,:] X, float r, int remove_mean): # Return return muReal, muImag, Phi, bJov + @cython.boundscheck(False) # turn off bounds-checking for entire function @cython.wraparound(False) # turn off negative index wrapping for entire function @@ -737,4 +741,903 @@ def reconstruction_jovanovic(real_complex[:,:] Phi, real[:] muReal, real[:] muIm if real_complex is np.complex128_t: return _dreconstruction_jovanovic(Phi,muReal,muImag,t,bJov) else: - return _sreconstruction_jovanovic(Phi,muReal,muImag,t,bJov) \ No newline at end of file + return _sreconstruction_jovanovic(Phi,muReal,muImag,t,bJov) + + +@cython.boundscheck(False) # turn off bounds-checking for entire function +@cython.wraparound(False) # turn off negative index wrapping for entire function +@cython.nonecheck(False) +@cython.cdivision(True) # turn off zero division check +def _srun_old(float[:,:] X, float r, int remove_mean): + ''' + Run DMD analysis of a matrix X. + + Inputs: + - X[ndims*nmesh,n_temp_snapshots]: data matrix + - remove_mean: whether or not to remove the mean flow + - r: maximum truncation residual + + Returns: + - Phi: DMD Modes + - muReal: Real part of the eigenvalues + - muImag: Imaginary part of the eigenvalues + - b: Amplitude of the DMD modes + - Variables needed to reconstruct flow + ''' + # Variables + cdef int m, n, irow, icol, iaux + cdef float[:, ::1] Y + cdef float[:, ::1] Z + cdef float[:, ::1] U + cdef float[::1] S + cdef float[:, ::1] VT + cdef float[:, ::1] Atilde + + # Create the snapshot of matrices and separate it into Y, Z + Y, Z = _sseparate(X, remove_mean) + m = Z.shape[0] + n = Z.shape[1] + 1 + # Compute the linear operator in lower dimension + U, S, VT, Atilde = _slinear_operator(Y, Z, r) + + # Create auxiliar matrix + del Y + cdef int nr + nr = Atilde.shape[1] + cdef float *aux2 + aux2 = malloc(nr*(n-1)*sizeof(float)) + for icol in range(n-1): + for irow in range(nr): + aux2[icol*nr + irow] = VT[irow, icol]/S[irow] + + # Compute eigenmodes + cdef float *auxmuReal + cdef float *auxmuImag + cdef np.complex64_t *w + auxmuReal = malloc(nr*sizeof(float)) + auxmuImag = malloc(nr*sizeof(float)) + w = malloc(nr*nr*sizeof(np.complex64_t)) + cr_start('DMD.eigendecomposition',0) + retval = c_ceigen(auxmuReal,auxmuImag,w,Ã[0,0],nr,nr) + cr_stop('DMD.eigendecomposition',0) + del Atilde + + # Computation of DMD modes + cr_start('DMD.modes',0) + cdef np.complex64_t *auxPhi + cdef np.complex64_t *aux1C + cdef np.complex64_t *aux2C + auxPhi = malloc(m*nr*sizeof(np.complex64_t)) + aux1C = malloc(nr*sizeof(np.complex64_t)) + aux2C = malloc(nr*sizeof(np.complex64_t)) + for iaux in range(m): + for icol in range(nr): + aux1C[icol] = 0 + 0*I + for irow in range(n-1): + aux1C[icol] += Z[iaux, irow]*aux2[irow*nr + icol] + c_cmatmult(aux2C, aux1C, w, 1, nr, nr, 'N', 'N') + memcpy(&auxPhi[iaux*nr], aux2C, nr*sizeof(np.complex64_t)) + free(aux2) + del Z + cdef float a + cdef float b + cdef float c + cdef float d + cdef float div + for icol in range(nr): + c = auxmuReal[icol] + d = auxmuImag[icol] + div = c*c + d*d + for iaux in range(m): + a = crealf(auxPhi[iaux*nr + icol]) + b = cimagf(auxPhi[iaux*nr + icol]) + auxPhi[iaux*nr + icol] = (a*c + b*d)/div + (b*c - a*d)/div*I + cr_stop('DMD.modes',0) + + # Amplitudes according to: Jovanovic et. al. 2014 DOI: 10.1063 + cdef np.complex64_t *auxbJov + cdef np.complex64_t *aux3C + cdef np.complex64_t *Vand + cdef np.complex64_t *P + cdef np.complex64_t *Pinv + cdef np.complex64_t *q + + auxbJov = malloc(nr*sizeof(np.complex64_t)) + aux3C = malloc(nr*nr*sizeof(np.complex64_t)) + aux4C = malloc(nr*nr*sizeof(np.complex64_t)) + Vand = malloc((nr*(n-1))*sizeof(np.complex64_t)) + P = malloc(nr*nr*sizeof(np.complex64_t)) + Pinv = malloc(nr*nr*sizeof(np.complex64_t)) + q = malloc(nr*sizeof(np.complex64_t)) + + cr_start('DMD.amplitudes', 0) + c_cvandermonde(Vand, auxmuReal, auxmuImag, nr, n-1) + c_cmatmult(aux3C, w, w, nr, nr, nr, 'C', 'N') + c_cmatmult(aux4C, Vand, Vand, nr, nr, n-1, 'N', 'C') + + for irow in range(nr): + for icol in range(nr): # Loop on the columns of the Vandermonde matrix + P[irow*nr + icol] = crealf(aux3C[irow*nr + icol])*crealf(aux4C[irow*nr + icol]) + P[irow*nr + icol] += -crealf(aux3C[irow*nr + icol])*cimagf(aux4C[irow*nr + icol])*I + P[irow*nr + icol] += cimagf(aux3C[irow*nr + icol])*crealf(aux4C[irow*nr + icol])*I + P[irow*nr + icol] += cimagf(aux3C[irow*nr + icol])*cimagf(aux4C[irow*nr + icol]) + retval = c_ccholesky(P, nr) + if not retval == 0: raiseError('Problems computing Cholesky factorization!') + + for iaux in range(nr): + for irow in range(nr): + aux1C[irow] = 0 + 0*I + for icol in range(n-1):# casting Vr to a complex, at the same time, it is multipilied per S and Vand + aux1C[irow] += S[irow]*VT[irow, icol]*(crealf(Vand[iaux*(n-1) + icol])+cimagf(Vand[iaux*(n-1) + icol])*I) + aux2C[irow] = w[irow*nr + iaux] + c_cmatmult(&q[iaux], aux1C, aux2C, 1, 1, nr, 'N', 'N') + + memcpy(Pinv, P, nr*nr*sizeof(np.complex64_t)) + cdef int ii + cdef int jj + for ii in range(nr): + q[ii] = crealf(q[ii]) - cimagf(q[ii])*I + for jj in range(nr - ii): + P[ii*nr + ii+jj] = crealf(P[(ii+jj)*nr + ii]) - cimagf(P[(ii+jj)*nr + ii])*I + P[(ii+jj)*nr + ii] = crealf(Pinv[ii*nr + ii+jj]) - cimagf(Pinv[ii*nr + ii+jj])*I + + retval = c_cinverse(Pinv, nr, 'L') + if not retval == 0: raiseError('Problems computing the Inverse!') + + c_cmatmult(aux1C, Pinv, q, nr, 1, nr, 'N', 'N') + + retval = c_cinverse(P, nr, 'U') + if not retval == 0: raiseError('Problems computing the Inverse!') + + c_cmatmult(auxbJov, P, aux1C, nr, 1, nr, 'N', 'N') + cr_stop('DMD.amplitudes',0) + + # Free allocated arrays before reordering + del U + del S + del VT + free(aux1C) + free(aux2C) + free(aux3C) + free(aux4C) + free(w) + free(Vand) + free(q) + free(P) + free(Pinv) + + # Order modes and eigenvalues according to its amplitude + cdef int *auxOrd + auxOrd = malloc(nr*sizeof(int)) + cdef np.ndarray[np.float32_t,ndim=1] muReal = np.zeros((nr),dtype=np.float32) + cdef np.ndarray[np.float32_t,ndim=1] muImag = np.zeros((nr),dtype=np.float32) + cdef np.ndarray[np.complex64_t,ndim=2] Phi = np.zeros((m,nr),order='C',dtype=np.complex64) + cdef np.ndarray[np.complex64_t,ndim=1] bJov = np.zeros((nr,),dtype=np.complex64) + + cr_start('DMD.qsort', 0) + c_csort(auxbJov, auxOrd, nr) + cr_stop('DMD.qsort', 0) + cr_start('DMD.sort', 0) + for ii in range(nr): + muReal[nr-(auxOrd[ii]+1)] = auxmuReal[ii] + muImag[nr-(auxOrd[ii]+1)] = auxmuImag[ii] + bJov[nr-(auxOrd[ii]+1)] = auxbJov[ii] + for jj in range(m): + Phi[jj,nr-(auxOrd[ii]+1)] = auxPhi[jj*nr + ii] + cr_stop('DMD.sort', 0) + + # Free the variables that had to be ordered + free(auxmuReal) + free(auxmuImag) + free(auxbJov) + free(auxPhi) + free(auxOrd) + + # Ensure that all conjugate modes are in the same order + cr_start('DMD.conjugate', 0) + cdef bint p = 0 + cdef float iimag + for ii in range(nr): + if p == 1: + p = 0 + continue + iimag = muImag[ii] + if iimag < 0: + muImag[ii] = muImag[ii+1] + muImag[ii+1] = -muImag[ii] + bJov[ii] = crealf(bJov[ii]) + cimagf(bJov[ii+1])*I + bJov[ii+1] = crealf(bJov[ii+1]) - cimagf(bJov[ii])*I + for jj in range(m): + Phi[jj,ii] = crealf(Phi[jj,ii]) + cimagf(Phi[jj,ii+1])*I + Phi[jj,ii+1] = crealf(Phi[jj,ii+1]) - cimagf(Phi[jj,ii+1])*I + p = 1 + continue + if iimag > 0: + p = 1 + continue + cr_stop('DMD.conjugate', 0) + + # Return + return muReal, muImag, Phi, bJov + +@cython.boundscheck(False) # turn off bounds-checking for entire function +@cython.wraparound(False) # turn off negative index wrapping for entire function +@cython.nonecheck(False) +@cython.cdivision(True) # turn off zero division check +def _drun_old(double[:,:] X, double r, int remove_mean): + ''' + Run DMD analysis of a matrix X. + + Inputs: + - X[ndims*nmesh,n_temp_snapshots]: data matrix + - remove_mean: whether or not to remove the mean flow + - r: maximum truncation residual + + Returns: + - Phi: DMD Modes + - muReal: Real part of the eigenvalues + - muImag: Imaginary part of the eigenvalues + - b: Amplitude of the DMD modes + - Variables needed to reconstruct flow + ''' + # Variables + cdef int m, n, irow, icol, iaux + cdef double[:, ::1] Y + cdef double[:, ::1] Z + cdef double[:, ::1] U + cdef double[::1] S + cdef double[:, ::1] VT + cdef double[:, ::1] Atilde + + # Create the snapshot of matrices and separate it into Y, Z + Y, Z = _dseparate(X, remove_mean) + m = Z.shape[0] + n = Z.shape[1] + 1 + # Compute the linear operator in lower dimension + U, S, VT, Atilde = _dlinear_operator(Y, Z, r) + # Create auxiliar matrix + del Y + cdef int nr + nr = Atilde.shape[1] + cdef double *aux2 + aux2 = malloc(nr*(n-1)*sizeof(double)) + for icol in range(n-1): + for irow in range(nr): + aux2[icol*nr + irow] = VT[irow, icol]/S[irow] + + # Compute eigenmodes + cdef double *auxmuReal + cdef double *auxmuImag + cdef np.complex128_t *w + auxmuReal = malloc(nr*sizeof(double)) + auxmuImag = malloc(nr*sizeof(double)) + w = malloc(nr*nr*sizeof(np.complex128_t)) + cr_start('DMD.eigendecomposition',0) + retval = c_zeigen(auxmuReal,auxmuImag,w,Ã[0,0],nr,nr) + cr_stop('DMD.eigendecomposition',0) + del Atilde + + # Computation of DMD modes + cr_start('DMD.modes',0) + cdef np.complex128_t *auxPhi + cdef np.complex128_t *aux1C + cdef np.complex128_t *aux2C + auxPhi = malloc(m*nr*sizeof(np.complex128_t)) + aux1C = malloc(nr*sizeof(np.complex128_t)) + aux2C = malloc(nr*sizeof(np.complex128_t)) + for iaux in range(m): + for icol in range(nr): + aux1C[icol] = 0 + 0*J + for irow in range(n-1): + aux1C[icol] += Z[iaux, irow]*aux2[irow*nr + icol] + c_zmatmult(aux2C, aux1C, w, 1, nr, nr, 'N', 'N') + memcpy(&auxPhi[iaux*nr], aux2C, nr*sizeof(np.complex128_t)) + free(aux2) + del Z + cdef double a + cdef double b + cdef double c + cdef double d + cdef double div + for icol in range(nr): + c = auxmuReal[icol] + d = auxmuImag[icol] + div = c*c + d*d + for iaux in range(m): + a = creal(auxPhi[iaux*nr + icol]) + b = cimag(auxPhi[iaux*nr + icol]) + auxPhi[iaux*nr + icol] = (a*c + b*d)/div + (b*c - a*d)/div*J + cr_stop('DMD.modes',0) + + # Amplitudes according to: Jovanovic et. al. 2014 DOI: 10.1063 + cdef np.complex128_t *auxbJov + cdef np.complex128_t *aux3C + cdef np.complex128_t *Vand + cdef np.complex128_t *P + cdef np.complex128_t *Pinv + cdef np.complex128_t *q + + auxbJov = malloc(nr*sizeof(np.complex128_t)) + aux3C = malloc(nr*nr*sizeof(np.complex128_t)) + aux4C = malloc(nr*nr*sizeof(np.complex128_t)) + Vand = malloc((nr*(n-1))*sizeof(np.complex128_t)) + P = malloc(nr*nr*sizeof(np.complex128_t)) + Pinv = malloc(nr*nr*sizeof(np.complex128_t)) + q = malloc(nr*sizeof(np.complex128_t)) + + cr_start('DMD.amplitudes', 0) + c_zvandermonde(Vand, auxmuReal, auxmuImag, nr, n-1) + c_zmatmult(aux3C, w, w, nr, nr, nr, 'C', 'N') + c_zmatmult(aux4C, Vand, Vand, nr, nr, n-1, 'N', 'C') + + for irow in range(nr): + for icol in range(nr): #Loop on the columns of the Vandermonde matrix + P[irow*nr + icol] = creal(aux3C[irow*nr + icol])*creal(aux4C[irow*nr + icol]) + P[irow*nr + icol] += -creal(aux3C[irow*nr + icol])*cimag(aux4C[irow*nr + icol])*J + P[irow*nr + icol] += cimag(aux3C[irow*nr + icol])*creal(aux4C[irow*nr + icol])*J + P[irow*nr + icol] += cimag(aux3C[irow*nr + icol])*cimag(aux4C[irow*nr + icol]) + retval = c_zcholesky(P, nr) + if not retval == 0: raiseError('Problems computing Cholesky factorization!') + + for iaux in range(nr): + for irow in range(nr): + aux1C[irow] = 0 + 0*J + for icol in range(n-1):#casting Vr to a complex, at the same time, it is multipilied per S and Vand + aux1C[irow] += S[irow]*VT[irow, icol]*(creal(Vand[iaux*(n-1) + icol])+cimag(Vand[iaux*(n-1) + icol])*J) + aux2C[irow] = w[irow*nr + iaux] + c_zmatmult(&q[iaux], aux1C, aux2C, 1, 1, nr, 'N', 'N') + + memcpy(Pinv, P, nr*nr*sizeof(np.complex128_t)) + cdef int ii + cdef int jj + for ii in range(nr): + q[ii] = creal(q[ii]) - cimag(q[ii])*J + for jj in range(nr - ii): + P[ii*nr + ii+jj] = creal(P[(ii+jj)*nr + ii]) - cimag(P[(ii+jj)*nr + ii])*J + P[(ii+jj)*nr + ii] = creal(Pinv[ii*nr + ii+jj]) - cimag(Pinv[ii*nr + ii+jj])*J + + retval = c_zinverse(Pinv, nr, 'L') + if not retval == 0: raiseError('Problems computing the Inverse!') + + c_zmatmult(aux1C, Pinv, q, nr, 1, nr, 'N', 'N') + + retval = c_zinverse(P, nr, 'U') + if not retval == 0: raiseError('Problems computing the Inverse!') + + c_zmatmult(auxbJov, P, aux1C, nr, 1, nr, 'N', 'N') + cr_stop('DMD.amplitudes',0) + + # Free allocated arrays before reordering + del U + del S + del VT + free(aux1C) + free(aux2C) + free(aux3C) + free(aux4C) + free(w) + free(Vand) + free(q) + free(P) + free(Pinv) + + # Order modes and eigenvalues according to its amplitude + cdef int *auxOrd + auxOrd = malloc(nr*sizeof(int)) + cdef np.ndarray[np.double_t,ndim=1] muReal = np.zeros((nr),dtype=np.double) + cdef np.ndarray[np.double_t,ndim=1] muImag = np.zeros((nr),dtype=np.double) + cdef np.ndarray[np.complex128_t,ndim=2] Phi = np.zeros((m,nr),order='C',dtype=np.complex128) + cdef np.ndarray[np.complex128_t,ndim=1] bJov = np.zeros((nr,),dtype=np.complex128) + + cr_start('DMD.qsort', 0) + c_zsort(auxbJov, auxOrd, nr) + cr_stop('DMD.qsort', 0) + cr_start('DMD.sort', 0) + for ii in range(nr): + muReal[nr-(auxOrd[ii]+1)] = auxmuReal[ii] + muImag[nr-(auxOrd[ii]+1)] = auxmuImag[ii] + bJov[nr-(auxOrd[ii]+1)] = auxbJov[ii] + for jj in range(m): + Phi[jj,nr-(auxOrd[ii]+1)] = auxPhi[jj*nr + ii] + cr_stop('DMD.sort', 0) + + # Free the variables that had to be ordered + free(auxmuReal) + free(auxmuImag) + free(auxbJov) + free(auxPhi) + free(auxOrd) + + # Ensure that all conjugate modes are in the same order + cr_start('DMD.conjugate', 0) + cdef bint p = 0 + cdef double iimag + for ii in range(nr): + if p == 1: + p = 0 + continue + iimag = muImag[ii] + if iimag < 0: + muImag[ii] = muImag[ii+1] + muImag[ii+1] = -muImag[ii] + bJov[ii] = creal(bJov[ii]) + cimag(bJov[ii+1])*J + bJov[ii+1] = creal(bJov[ii+1]) - cimag(bJov[ii])*J + for jj in range(m): + Phi[jj,ii] = creal(Phi[jj,ii]) + cimag(Phi[jj,ii+1])*J + Phi[jj,ii+1] = creal(Phi[jj,ii+1]) - cimag(Phi[jj,ii+1])*J + p = 1 + continue + if iimag > 0: + p = 1 + continue + cr_stop('DMD.conjugate', 0) + + # Return + return muReal, muImag, Phi, bJov + + +@cython.boundscheck(False) # turn off bounds-checking for entire function +@cython.wraparound(False) # turn off negative index wrapping for entire function +@cython.nonecheck(False) +@cython.cdivision(True) # turn off zero division check +def _srun_new(list X, float r, int remove_mean): + ''' + Run DMD analysis of a matrix X. + + Inputs: + - X[ndims*nmesh,n_temp_snapshots]: data matrix + - remove_mean: whether or not to remove the mean flow + - r: maximum truncation residual + + Returns: + - Phi: DMD Modes + - muReal: Real part of the eigenvalues + - muImag: Imaginary part of the eigenvalues + - b: Amplitude of the DMD modes + - Variables needed to reconstruct flow + ''' + # Variables + cdef int m, n, irow, icol, iaux + cdef float[:, ::1] Y + cdef float[:, ::1] Z + cdef float[:, ::1] U + cdef float[::1] S + cdef float[:, ::1] VT + cdef float[:, ::1] Atilde + + # Create the snapshot of matrices and separate it into Y, Z + Y, Z = _sconcatenate(X, remove_mean) + m = Z.shape[0] + n = Z.shape[1] + 1 + # Compute the linear operator in lower dimension + U, S, VT, Atilde = _slinear_operator(Y, Z, r) + # Create auxiliar matrix + del Y + cdef int nr + nr = Atilde.shape[1] + cdef float *aux2 + aux2 = malloc(nr*(n-1)*sizeof(float)) + for icol in range(n-1): + for irow in range(nr): + aux2[icol*nr + irow] = VT[irow, icol]/S[irow] + + # Compute eigenmodes + cdef float *auxmuReal + cdef float *auxmuImag + cdef np.complex64_t *w + auxmuReal = malloc(nr*sizeof(float)) + auxmuImag = malloc(nr*sizeof(float)) + w = malloc(nr*nr*sizeof(np.complex64_t)) + cr_start('DMD.eigendecomposition',0) + retval = c_ceigen(auxmuReal,auxmuImag,w,Ã[0,0],nr,nr) + cr_stop('DMD.eigendecomposition',0) + del Atilde + + # Computation of DMD modes + cr_start('DMD.modes',0) + cdef np.complex64_t *auxPhi + cdef np.complex64_t *aux1C + cdef np.complex64_t *aux2C + auxPhi = malloc(m*nr*sizeof(np.complex64_t)) + aux1C = malloc(nr*sizeof(np.complex64_t)) + aux2C = malloc(nr*sizeof(np.complex64_t)) + for iaux in range(m): + for icol in range(nr): + aux1C[icol] = 0 + 0*I + for irow in range(n-1): + aux1C[icol] += Z[iaux, irow]*aux2[irow*nr + icol] + c_cmatmult(aux2C, aux1C, w, 1, nr, nr, 'N', 'N') + memcpy(&auxPhi[iaux*nr], aux2C, nr*sizeof(np.complex64_t)) + free(aux2) + del Z + cdef float a + cdef float b + cdef float c + cdef float d + cdef float div + for icol in range(nr): + c = auxmuReal[icol] + d = auxmuImag[icol] + div = c*c + d*d + for iaux in range(m): + a = crealf(auxPhi[iaux*nr + icol]) + b = cimagf(auxPhi[iaux*nr + icol]) + auxPhi[iaux*nr + icol] = (a*c + b*d)/div + (b*c - a*d)/div*I + cr_stop('DMD.modes',0) + + # Amplitudes according to: Jovanovic et. al. 2014 DOI: 10.1063 + cdef np.complex64_t *auxbJov + cdef np.complex64_t *aux3C + cdef np.complex64_t *Vand + cdef np.complex64_t *P + cdef np.complex64_t *Pinv + cdef np.complex64_t *q + + auxbJov = malloc(nr*sizeof(np.complex64_t)) + aux3C = malloc(nr*nr*sizeof(np.complex64_t)) + aux4C = malloc(nr*nr*sizeof(np.complex64_t)) + Vand = malloc((nr*(n-1))*sizeof(np.complex64_t)) + P = malloc(nr*nr*sizeof(np.complex64_t)) + Pinv = malloc(nr*nr*sizeof(np.complex64_t)) + q = malloc(nr*sizeof(np.complex64_t)) + + cr_start('DMD.amplitudes', 0) + c_cvandermonde(Vand, auxmuReal, auxmuImag, nr, n-1) + c_cmatmult(aux3C, w, w, nr, nr, nr, 'C', 'N') + c_cmatmult(aux4C, Vand, Vand, nr, nr, n-1, 'N', 'C') + + for irow in range(nr): + for icol in range(nr): # Loop on the columns of the Vandermonde matrix + P[irow*nr + icol] = crealf(aux3C[irow*nr + icol])*crealf(aux4C[irow*nr + icol]) + P[irow*nr + icol] += -crealf(aux3C[irow*nr + icol])*cimagf(aux4C[irow*nr + icol])*I + P[irow*nr + icol] += cimagf(aux3C[irow*nr + icol])*crealf(aux4C[irow*nr + icol])*I + P[irow*nr + icol] += cimagf(aux3C[irow*nr + icol])*cimagf(aux4C[irow*nr + icol]) + retval = c_ccholesky(P, nr) + if not retval == 0: raiseError('Problems computing Cholesky factorization!') + + for iaux in range(nr): + for irow in range(nr): + aux1C[irow] = 0 + 0*I + for icol in range(n-1):# casting Vr to a complex, at the same time, it is multipilied per S and Vand + aux1C[irow] += S[irow]*VT[irow, icol]*(crealf(Vand[iaux*(n-1) + icol])+cimagf(Vand[iaux*(n-1) + icol])*I) + aux2C[irow] = w[irow*nr + iaux] + c_cmatmult(&q[iaux], aux1C, aux2C, 1, 1, nr, 'N', 'N') + + memcpy(Pinv, P, nr*nr*sizeof(np.complex64_t)) + cdef int ii + cdef int jj + for ii in range(nr): + q[ii] = crealf(q[ii]) - cimagf(q[ii])*I + for jj in range(nr - ii): + P[ii*nr + ii+jj] = crealf(P[(ii+jj)*nr + ii]) - cimagf(P[(ii+jj)*nr + ii])*I + P[(ii+jj)*nr + ii] = crealf(Pinv[ii*nr + ii+jj]) - cimagf(Pinv[ii*nr + ii+jj])*I + + retval = c_cinverse(Pinv, nr, 'L') + if not retval == 0: raiseError('Problems computing the Inverse!') + + c_cmatmult(aux1C, Pinv, q, nr, 1, nr, 'N', 'N') + + retval = c_cinverse(P, nr, 'U') + if not retval == 0: raiseError('Problems computing the Inverse!') + + c_cmatmult(auxbJov, P, aux1C, nr, 1, nr, 'N', 'N') + cr_stop('DMD.amplitudes',0) + + # Free allocated arrays before reordering + del U + del S + del VT + free(aux1C) + free(aux2C) + free(aux3C) + free(aux4C) + free(w) + free(Vand) + free(q) + free(P) + free(Pinv) + + # Order modes and eigenvalues according to its amplitude + cdef int *auxOrd + auxOrd = malloc(nr*sizeof(int)) + cdef np.ndarray[np.float32_t,ndim=1] muReal = np.zeros((nr),dtype=np.float32) + cdef np.ndarray[np.float32_t,ndim=1] muImag = np.zeros((nr),dtype=np.float32) + cdef np.ndarray[np.complex64_t,ndim=2] Phi = np.zeros((m,nr),order='C',dtype=np.complex64) + cdef np.ndarray[np.complex64_t,ndim=1] bJov = np.zeros((nr,),dtype=np.complex64) + + cr_start('DMD.qsort', 0) + c_csort(auxbJov, auxOrd, nr) + cr_stop('DMD.qsort', 0) + cr_start('DMD.sort', 0) + for ii in range(nr): + muReal[nr-(auxOrd[ii]+1)] = auxmuReal[ii] + muImag[nr-(auxOrd[ii]+1)] = auxmuImag[ii] + bJov[nr-(auxOrd[ii]+1)] = auxbJov[ii] + for jj in range(m): + Phi[jj,nr-(auxOrd[ii]+1)] = auxPhi[jj*nr + ii] + cr_stop('DMD.sort', 0) + + # Free the variables that had to be ordered + free(auxmuReal) + free(auxmuImag) + free(auxbJov) + free(auxPhi) + free(auxOrd) + + # Ensure that all conjugate modes are in the same order + cr_start('DMD.conjugate', 0) + cdef bint p = 0 + cdef float iimag + for ii in range(nr): + if p == 1: + p = 0 + continue + iimag = muImag[ii] + if iimag < 0: + muImag[ii] = muImag[ii+1] + muImag[ii+1] = -muImag[ii] + bJov[ii] = crealf(bJov[ii]) + cimagf(bJov[ii+1])*I + bJov[ii+1] = crealf(bJov[ii+1]) - cimagf(bJov[ii])*I + for jj in range(m): + Phi[jj,ii] = crealf(Phi[jj,ii]) + cimagf(Phi[jj,ii+1])*I + Phi[jj,ii+1] = crealf(Phi[jj,ii+1]) - cimagf(Phi[jj,ii+1])*I + p = 1 + continue + if iimag > 0: + p = 1 + continue + cr_stop('DMD.conjugate', 0) + + # Return + return muReal, muImag, Phi, bJov + +@cython.boundscheck(False) # turn off bounds-checking for entire function +@cython.wraparound(False) # turn off negative index wrapping for entire function +@cython.nonecheck(False) +@cython.cdivision(True) # turn off zero division check +def _drun_new(list X, double r, int remove_mean): + ''' + Run DMD analysis of a matrix X. + + Inputs: + - X[ndims*nmesh,n_temp_snapshots]: data matrix + - remove_mean: whether or not to remove the mean flow + - r: maximum truncation residual + + Returns: + - Phi: DMD Modes + - muReal: Real part of the eigenvalues + - muImag: Imaginary part of the eigenvalues + - b: Amplitude of the DMD modes + - Variables needed to reconstruct flow + ''' + # Variables + cdef int m, n, irow, icol, iaux + cdef double[:, ::1] Y + cdef double[:, ::1] Z + cdef double[:, ::1] U + cdef double[::1] S + cdef double[:, ::1] VT + cdef double[:, ::1] Atilde + + # Create the snapshot of matrices and separate it into Y, Z + cr_start('DMD.concatenate', 0) + Y, Z = _dconcatenate(X, remove_mean) + cr_stop('DMD.concatenate', 0) + m = Z.shape[0] + n = Z.shape[1] + 1 + # Compute the linear operator in lower dimension + cr_start('DMD.linear_operator', 0) + U, S, VT, Atilde = _dlinear_operator(Y, Z, r) + cr_stop('DMD.linear_operator', 0) + # Create auxiliar matrix + del Y + cdef int nr + nr = Atilde.shape[1] + cdef double *aux2 + aux2 = malloc(nr*(n-1)*sizeof(double)) + for icol in range(n-1): + for irow in range(nr): + aux2[icol*nr + irow] = VT[irow, icol]/S[irow] + + # Compute eigenmodes + cdef double *auxmuReal + cdef double *auxmuImag + cdef np.complex128_t *w + auxmuReal = malloc(nr*sizeof(double)) + auxmuImag = malloc(nr*sizeof(double)) + w = malloc(nr*nr*sizeof(np.complex128_t)) + cr_start('DMD.eigendecomposition',0) + retval = c_zeigen(auxmuReal,auxmuImag,w,Ã[0,0],nr,nr) + cr_stop('DMD.eigendecomposition',0) + del Atilde + + # Computation of DMD modes + cr_start('DMD.modes',0) + cdef np.complex128_t *auxPhi + cdef np.complex128_t *aux1C + cdef np.complex128_t *aux2C + auxPhi = malloc(m*nr*sizeof(np.complex128_t)) + aux1C = malloc(nr*sizeof(np.complex128_t)) + aux2C = malloc(nr*sizeof(np.complex128_t)) + for iaux in range(m): + for icol in range(nr): + aux1C[icol] = 0 + 0*I + for irow in range(n-1): + aux1C[icol] += Z[iaux, irow]*aux2[irow*nr + icol] + c_zmatmult(aux2C, aux1C, w, 1, nr, nr, 'N', 'N') + memcpy(&auxPhi[iaux*nr], aux2C, nr*sizeof(np.complex128_t)) + free(aux2) + del Z + cdef double a + cdef double b + cdef double c + cdef double d + cdef double div + for icol in range(nr): + c = auxmuReal[icol] + d = auxmuImag[icol] + div = c*c + d*d + for iaux in range(m): + a = crealf(auxPhi[iaux*nr + icol]) + b = cimagf(auxPhi[iaux*nr + icol]) + auxPhi[iaux*nr + icol] = (a*c + b*d)/div + (b*c - a*d)/div*I + cr_stop('DMD.modes',0) + + # Amplitudes according to: Jovanovic et. al. 2014 DOI: 10.1063 + cdef np.complex128_t *auxbJov + cdef np.complex128_t *aux3C + cdef np.complex128_t *Vand + cdef np.complex128_t *P + cdef np.complex128_t *Pinv + cdef np.complex128_t *q + + auxbJov = malloc(nr*sizeof(np.complex128_t)) + aux3C = malloc(nr*nr*sizeof(np.complex128_t)) + aux4C = malloc(nr*nr*sizeof(np.complex128_t)) + Vand = malloc((nr*(n-1))*sizeof(np.complex128_t)) + P = malloc(nr*nr*sizeof(np.complex128_t)) + Pinv = malloc(nr*nr*sizeof(np.complex128_t)) + q = malloc(nr*sizeof(np.complex128_t)) + + cr_start('DMD.amplitudes', 0) + c_zvandermonde(Vand, auxmuReal, auxmuImag, nr, n-1) + c_zmatmult(aux3C, w, w, nr, nr, nr, 'C', 'N') + c_zmatmult(aux4C, Vand, Vand, nr, nr, n-1, 'N', 'C') + + for irow in range(nr): + for icol in range(nr): # Loop on the columns of the Vandermonde matrix + P[irow*nr + icol] = crealf(aux3C[irow*nr + icol])*crealf(aux4C[irow*nr + icol]) + P[irow*nr + icol] += -crealf(aux3C[irow*nr + icol])*cimagf(aux4C[irow*nr + icol])*I + P[irow*nr + icol] += cimagf(aux3C[irow*nr + icol])*crealf(aux4C[irow*nr + icol])*I + P[irow*nr + icol] += cimagf(aux3C[irow*nr + icol])*cimagf(aux4C[irow*nr + icol]) + retval = c_zcholesky(P, nr) + if not retval == 0: raiseError('Problems computing Cholesky factorization!') + + for iaux in range(nr): + for irow in range(nr): + aux1C[irow] = 0 + 0*I + for icol in range(n-1):# casting Vr to a complex, at the same time, it is multipilied per S and Vand + aux1C[irow] += S[irow]*VT[irow, icol]*(crealf(Vand[iaux*(n-1) + icol])+cimagf(Vand[iaux*(n-1) + icol])*I) + aux2C[irow] = w[irow*nr + iaux] + c_zmatmult(&q[iaux], aux1C, aux2C, 1, 1, nr, 'N', 'N') + + memcpy(Pinv, P, nr*nr*sizeof(np.complex128_t)) + cdef int ii + cdef int jj + for ii in range(nr): + q[ii] = crealf(q[ii]) - cimagf(q[ii])*I + for jj in range(nr - ii): + P[ii*nr + ii+jj] = crealf(P[(ii+jj)*nr + ii]) - cimagf(P[(ii+jj)*nr + ii])*I + P[(ii+jj)*nr + ii] = crealf(Pinv[ii*nr + ii+jj]) - cimagf(Pinv[ii*nr + ii+jj])*I + + retval = c_zinverse(Pinv, nr, 'L') + if not retval == 0: raiseError('Problems computing the Inverse!') + + c_zmatmult(aux1C, Pinv, q, nr, 1, nr, 'N', 'N') + + retval = c_zinverse(P, nr, 'U') + if not retval == 0: raiseError('Problems computing the Inverse!') + + c_zmatmult(auxbJov, P, aux1C, nr, 1, nr, 'N', 'N') + cr_stop('DMD.amplitudes',0) + + # Free allocated arrays before reordering + del U + del S + del VT + free(aux1C) + free(aux2C) + free(aux3C) + free(aux4C) + free(w) + free(Vand) + free(q) + free(P) + free(Pinv) + + # Order modes and eigenvalues according to its amplitude + cdef int *auxOrd + auxOrd = malloc(nr*sizeof(int)) + cdef np.ndarray[np.double_t,ndim=1] muReal = np.zeros((nr),dtype=np.double) + cdef np.ndarray[np.double_t,ndim=1] muImag = np.zeros((nr),dtype=np.double) + cdef np.ndarray[np.complex128_t,ndim=2] Phi = np.zeros((m,nr),order='C',dtype=np.complex128) + cdef np.ndarray[np.complex128_t,ndim=1] bJov = np.zeros((nr,),dtype=np.complex128) + + cr_start('DMD.qsort', 0) + c_zsort(auxbJov, auxOrd, nr) + cr_stop('DMD.qsort', 0) + cr_start('DMD.sort', 0) + for ii in range(nr): + muReal[nr-(auxOrd[ii]+1)] = auxmuReal[ii] + muImag[nr-(auxOrd[ii]+1)] = auxmuImag[ii] + bJov[nr-(auxOrd[ii]+1)] = auxbJov[ii] + for jj in range(m): + Phi[jj,nr-(auxOrd[ii]+1)] = auxPhi[jj*nr + ii] + cr_stop('DMD.sort', 0) + + # Free the variables that had to be ordered + free(auxmuReal) + free(auxmuImag) + free(auxbJov) + free(auxPhi) + free(auxOrd) + + # Ensure that all conjugate modes are in the same order + cr_start('DMD.conjugate', 0) + cdef bint p = 0 + cdef double iimag + for ii in range(nr): + if p == 1: + p = 0 + continue + iimag = muImag[ii] + if iimag < 0: + muImag[ii] = muImag[ii+1] + muImag[ii+1] = -muImag[ii] + bJov[ii] = crealf(bJov[ii]) + cimagf(bJov[ii+1])*I + bJov[ii+1] = crealf(bJov[ii+1]) - cimagf(bJov[ii])*I + for jj in range(m): + Phi[jj,ii] = crealf(Phi[jj,ii]) + cimagf(Phi[jj,ii+1])*I + Phi[jj,ii+1] = crealf(Phi[jj,ii+1]) - cimagf(Phi[jj,ii+1])*I + p = 1 + continue + if iimag > 0: + p = 1 + continue + cr_stop('DMD.conjugate', 0) + + # Return + return muReal, muImag, Phi, bJov + +@cr('DMD.run_new') +@cython.boundscheck(False) # turn off bounds-checking for entire function +@cython.wraparound(False) # turn off negative index wrapping for entire function +@cython.nonecheck(False) +@cython.cdivision(True) # turn off zero division check +def run_new(object X, object r, int remove_mean=True): + ''' + Run DMD analysis of a matrix X. + + Inputs: + - X[ndims*nmesh,n_temp_snapshots]: data matrix + - remove_mean: whether or not to remove the mean flow + - r: maximum truncation residual + + Returns: + - Phi: DMD Modes + - muReal: Real part of the eigenvalues + - muImag: Imaginary part of the eigenvalues + - b: Amplitude of the DMD modes + - Variables needed to reconstruct flow + ''' + if isinstance(X, list): + if X[0].dtype == np.double: + return _drun_new(X,r,remove_mean) + else: + return _srun_new(X,r,remove_mean) + else: + if X[0].dtype == np.double: + return _drun_old(X,r,remove_mean) + else: + return _srun_old(X,r,remove_mean) \ No newline at end of file diff --git a/pyLOM/RES/__init__.py b/pyLOM/RES/__init__.py index fbc53fb7..22f0cc00 100644 --- a/pyLOM/RES/__init__.py +++ b/pyLOM/RES/__init__.py @@ -7,7 +7,7 @@ # Last rev: 20/02/2026 # Functions coming from Resolvent -from .wrapper import run +from .wrapper import run, run_new from .utils import extract_modes, save, load from .plots import plotEnergy, plotEvW diff --git a/pyLOM/RES/wrapper.py b/pyLOM/RES/wrapper.py index 9f8b77fc..ef88f309 100644 --- a/pyLOM/RES/wrapper.py +++ b/pyLOM/RES/wrapper.py @@ -10,7 +10,7 @@ import numpy as np from ..utils.gpu import cp -from ..vmmath import vecmat, matmul, svd, cholesky, inv, matmulp, dagger +from ..vmmath import vecmat, matmul, svd, cholesky, inv, matmulp, dagger, concatenate, linear_operator, separate, resolvent from ..utils import cr_nvtx as cr, cr_start, cr_stop @cr('RES.run') @@ -53,4 +53,25 @@ def run(Phi, delta, omega, f, Q=None): U_res = matmul(Phi, matmul(Fhat_inv, U)) V_res = matmul(Phi, matmul(Fhat_inv, V)) - return U_res, S, V_res \ No newline at end of file + return U_res, S, V_res + +def run_new(X, w, r, remove_mean = True): + + # Prepare matrices and calculate the linear operator + if (type(X) is list): + Y, Z = concatenate(X, remove_mean=remove_mean) + U1, S1, VT1, Atilde = linear_operator(Y, Z, r) + + else: + Y, Z = separate(X, remove_mean=remove_mean) + U1, S1, VT1, Atilde = linear_operator(Y, Z, r) + + del S1, VT1 + + # Calculate the resolvent of the linear operator + U2, S, V = resolvent(Atilde, w) + + # Project the solution + U = matmul(U1, U2) + + return U, S, V \ No newline at end of file diff --git a/pyLOM/RES/wrapper.pyx b/pyLOM/RES/wrapper.pyx index 08325c50..126f99eb 100644 --- a/pyLOM/RES/wrapper.pyx +++ b/pyLOM/RES/wrapper.pyx @@ -24,9 +24,11 @@ cdef double complex J = 1j from libc.stdlib cimport malloc, free from libc.string cimport memcpy, memset from libc.math cimport sqrt, log, atan2 -from ..vmmath.cfuncs cimport real, real_complex +from ..vmmath.cfuncs cimport real, real_complex, real_float, real_double from ..vmmath.cfuncs cimport c_csvd, c_cdagger, c_cmatmul, c_cmatmulp, c_cvecmat, c_ccholesky, c_cinverse from ..vmmath.cfuncs cimport c_zsvd, c_zdagger, c_zmatmul, c_zmatmulp, c_zvecmat, c_zcholesky, c_zinverse +from ..vmmath.linear cimport _sconcatenate, _slinear_operator, _sseparate, _cresolvent +from ..vmmath.linear cimport _dconcatenate, _dlinear_operator, _dseparate, _zresolvent from ..utils.cr import cr, cr_start, cr_stop from ..utils.errors import raiseError @@ -297,3 +299,182 @@ def run(real_complex[:,:] Phi, real[:] delta, real[:] omega, real f, real[:] Q=N return _zrun(Phi, delta, omega, f, Q) else: return _crun(Phi, delta, omega, f, Q) + +def _crun_old(float[:,:] X, np.complex64_t w, float r, int remove_mean): + + # Variables + cdef int m, n + cdef float[:, ::1] Y + cdef float[:, ::1] Z + cdef float[:, ::1] U1 + cdef float[::1] S1 + cdef float[:, ::1] VT1 + cdef float[:, ::1] Atilde + + # Create the snapshot of matrices and separate it into Y, Z + Y, Z = _sseparate(X, remove_mean) + m = Z.shape[0] + n = Z.shape[1] + 1 + # Compute the linear operator in lower dimension + U1, S1, VT1, Atilde = _slinear_operator(Y, Z, r) + cdef int nr + nr = Atilde.shape[1] + + del S1, VT1 + + cdef np.complex64_t[:, ::1] U2 + cdef np.ndarray[np.float32_t,ndim=1] S = np.zeros((nr),dtype=np.float32) + cdef np.complex64_t[:, ::1] V2 + + U2, S, V2 = _cresolvent(Atilde, w) + + del Atilde + + cdef np.ndarray[np.complex64_t,ndim=2] U = np.zeros((m,nr),dtype=np.complex64) + cdef np.ndarray[np.complex64_t,ndim=2] V = np.zeros((m,nr),dtype=np.complex64) + c_cmatmul(&U[0,0], &U1[0,0], &U2[0,0], m, nr, nr) + c_cmatmul(&V[0,0], &U1[0,0], &V2[0,0], m, nr, nr) + + del U1, U2, V2 + + return U, S, V + +def _zrun_old(double[:,:] X, np.complex128_t w, double r, int remove_mean): + print('variables', flush=True) + # Variables + cdef int m, n + cdef double[:, ::1] Y + cdef double[:, ::1] Z + cdef double[:, ::1] U1 + cdef double[::1] S1 + cdef double[:, ::1] VT1 + cdef double[:, ::1] Atilde + print('snapshots', flush=True) + # Create the snapshot of matrices and separate it into Y, Z + Y, Z = _dseparate(X, remove_mean) + m = Z.shape[0] + n = Z.shape[1] + 1 + print('linear operator', flush=True) + # Compute the linear operator in lower dimension + U1, S1, VT1, Atilde = _dlinear_operator(Y, Z, r) + cdef int nr + nr = Atilde.shape[1] + + del S1, VT1 + print('resolvent', flush=True) + cdef np.complex128_t[:, ::1] U2 + cdef np.ndarray[np.double_t,ndim=1] S = np.zeros((nr),dtype=np.double) + cdef np.complex128_t[:, ::1] V2 + + U2, S, V2 = _zresolvent(Atilde, w) + + del Atilde + print('project', flush=True) + cdef np.ndarray[np.complex128_t,ndim=2] U = np.zeros((m,nr),dtype=np.complex128) + cdef np.ndarray[np.complex128_t,ndim=2] V = np.zeros((m,nr),dtype=np.complex128) + c_zmatmul(&U[0,0], &U1[0,0], &U2[0,0], m, nr, nr) + c_zmatmul(&V[0,0], &U1[0,0], &V2[0,0], m, nr, nr) + + del U1, U2, V2 + + return U, S, V + +def _crun_new(list X, np.complex64_t w, float r, int remove_mean): + + # Variables + cdef int m, n + cdef float[:, ::1] Y + cdef float[:, ::1] Z + cdef float[:, ::1] U1 + cdef float[::1] S1 + cdef float[:, ::1] VT1 + cdef float[:, ::1] Atilde + + # Create the snapshot of matrices and separate it into Y, Z + Y, Z = _sconcatenate(X, remove_mean) + m = Z.shape[0] + n = Z.shape[1] + 1 + # Compute the linear operator in lower dimension + U1, S1, VT1, Atilde = _slinear_operator(Y, Z, r) + cdef int nr + nr = Atilde.shape[1] + + del S1, VT1 + + cdef np.complex64_t[:, ::1] U2 + cdef np.ndarray[np.float32_t,ndim=1] S = np.zeros((nr),dtype=np.float32) + cdef np.complex64_t[:, ::1] V2 + + U2, S, V2 = _cresolvent(Atilde, w) + + del Atilde + + cdef np.ndarray[np.complex64_t,ndim=2] U = np.zeros((m,nr),dtype=np.complex64) + cdef np.ndarray[np.complex64_t,ndim=2] V = np.zeros((m,nr),dtype=np.complex64) + c_cmatmul(&U[0,0], &U1[0,0], &U2[0,0], m, nr, nr) + c_cmatmul(&V[0,0], &U1[0,0], &V2[0,0], m, nr, nr) + + del U1, U2 + + return U, S, V + +def _zrun_new(list X, np.complex128_t w, double r, int remove_mean): + + # Variables + cdef int m, n + cdef double[:, ::1] Y + cdef double[:, ::1] Z + cdef double[:, ::1] U1 + cdef double[::1] S1 + cdef double[:, ::1] VT1 + cdef double[:, ::1] Atilde + + # Create the snapshot of matrices and separate it into Y, Z + Y, Z = _dconcatenate(X, remove_mean) + m = Z.shape[0] + n = Z.shape[1] + 1 + # Compute the linear operator in lower dimension + U1, S1, VT1, Atilde = _dlinear_operator(Y, Z, r) + cdef int nr + nr = Atilde.shape[1] + + del S1, VT1 + + cdef np.complex128_t[:, ::1] U2 + cdef np.ndarray[np.double_t,ndim=1] S = np.zeros((nr),dtype=np.double) + cdef np.complex128_t[:, ::1] V2 + + U2, S, V2 = _zresolvent(Atilde, w) + + del Atilde + + cdef np.ndarray[np.complex128_t,ndim=2] U = np.zeros((m,nr),dtype=np.complex128) + cdef np.ndarray[np.complex128_t,ndim=2] V = np.zeros((m,nr),dtype=np.complex128) + c_zmatmul(&U[0,0], &U1[0,0], &U2[0,0], m, nr, nr) + c_zmatmul(&V[0,0], &U1[0,0], &V2[0,0], m, nr, nr) + + del U1, U2, V2 + + return U, S, V + +@cr('RES.run_new') +@cython.boundscheck(False) # turn off bounds-checking for entire function +@cython.wraparound(False) # turn off negative index wrapping for entire function +@cython.nonecheck(False) +@cython.cdivision(True) # turn off zero division check +def run_new(object X, object w, object r, int remove_mean=True): + + if isinstance(X, list): + if X[0].dtype == np.double: + print('function1', flush=True) + return _zrun_new(X,w,r,remove_mean) + else: + print('function2', flush=True) + return _crun_new(X,w,r,remove_mean) + else: + if X[0].dtype == np.double: + print('function3', flush=True) + return _zrun_old(X,w,r,remove_mean) + else: + print('function4', flush=True) + return _crun_old(X,w,r,remove_mean) \ No newline at end of file diff --git a/pyLOM/vmmath/__init__.py b/pyLOM/vmmath/__init__.py index 3a75d568..8cbaf574 100644 --- a/pyLOM/vmmath/__init__.py +++ b/pyLOM/vmmath/__init__.py @@ -11,7 +11,7 @@ # Averaging routines from .averaging import temporal_mean, subtract_mean, temporal_variance, norm_variance # Truncation routines -from .truncation import compute_truncation_residual, energy, local_energy +from .truncation import compute_truncation_residual, energy, local_energy, remove_rows # Statistics routines from .stats import RMSE, MAE, r2, MRE_array # QR routines @@ -26,6 +26,8 @@ from .regression import least_squares, ridge_regresion # Data processing module from .dataprocessing import data_splitting, time_delay_embedding, find_random_sensors +# Linear matrix module +from .linear import linear_operator, concatenate, separate, resolvent -del maths, averaging, truncation, stats, geometric, regression +del maths, averaging, truncation, stats, geometric, regression, dataprocessing, linear diff --git a/pyLOM/vmmath/linear.pxd b/pyLOM/vmmath/linear.pxd new file mode 100644 index 00000000..bf8075c7 --- /dev/null +++ b/pyLOM/vmmath/linear.pxd @@ -0,0 +1,25 @@ +#!/usr/bin/env cpython +# +# pyLOM - Python Low Order Modeling. +# +# Linear operator module - exporting of Cython functions. +# +# Last rev: 07/09/2026 + +cimport numpy as np + +# Float precision +cdef tuple _slinear_operator(float[:,:] Y, float[:,:] Z, float r) +cdef tuple _sconcatenate(list X, int remove_mean) +cdef tuple _sseparate(float[:,:] X, int remove_mean) + +# Double precision +cdef tuple _dlinear_operator(double[:,:] Y, double[:,:] Z, double r) +cdef tuple _dconcatenate(list X, int remove_mean) +cdef tuple _dseparate(double[:,:] X, int remove_mean) + +# Complex single precision +cdef tuple _cresolvent(float[:,:] A, np.complex64_t f) + +# Complex double precision +cdef tuple _zresolvent(double[:,:] A, np.complex128_t f) \ No newline at end of file diff --git a/pyLOM/vmmath/linear.py b/pyLOM/vmmath/linear.py new file mode 100644 index 00000000..a1987c1a --- /dev/null +++ b/pyLOM/vmmath/linear.py @@ -0,0 +1,106 @@ +#!/usr/bin/env cpython +# +# pyLOM - Python Low Order Modeling. +# +# Linear operator module +# +# Last rev: 31/08/2026 +from __future__ import print_function, division + +import numpy as np + +from ..utils.gpu import cp +from ..vmmath import vecmat, matmul, tsqr_svd, transpose, matmulp, remove_rows, temporal_mean, subtract_mean, svd, dagger +from ..POD import truncate +from ..utils import cr_nvtx as cr, cr_start, cr_stop + +def linear_operator(Y, Z, r): + + cr_start('DMD.SVD',0) + U_aux, S, VT = tsqr_svd(Y) + cr_stop('DMD.SVD',0) + if Y.shape[0] > Z.shape[0]: + U = remove_rows(U_aux, Z.shape[0]) + else: + U = U_aux.copy() + + # Truncate according to residual + cr_start('DMD.truncate', 0) + U, S, VT = truncate(U, S, VT, r) + cr_stop('DMD.truncate', 0) + + # Project A (Jacobian of the snapshots) into POD basis + cr_start('DMD.linear_mapping',0) + aux1 = matmulp(transpose(U), Z) + aux2 = transpose(vecmat(1./S, VT)) + Atilde = matmul(aux1, aux2) + cr_stop('DMD.linear_mapping',0) + + return U, S, VT, Atilde + +def concatenate(X=[], remove_mean=False): + + # Create the matrices Y, Z + dtype = X[0].dtype + m = X[0].shape[0] + n = sum([M.shape[1] - 1 for M in X]) + if m > n: + Y = np.zeros((m, n), dtype=dtype) + else: + Y = np.zeros((n, n), dtype=dtype) + Z = np.zeros((m, n), dtype=dtype) + + # Fill the matrices + offset = 0 + for M in X: + ni = M.shape[1] + if remove_mean: + cr_start('DMD.temporal_mean',0) + mean = temporal_mean(M) + + Y[:m,offset:offset + ni-1] = subtract_mean(M[:,:-1], mean) + Z[:,offset:offset + ni-1] = subtract_mean(M[:,1:], mean) + offset += (ni-1) + cr_stop('DMD.temporal_mean',0) + else: + Y[:m,offset:offset + ni-1] = M[:,:-1] + Z[:,offset:offset + ni-1] = M[:,1:] + offset += (ni-1) + + return Y, Z + +def separate(X, remove_mean=False): + + # Create the matrices Y, Z + dtype = X[0].dtype + m = X.shape[0] + n = X.shape[1] + if m > n: + Y = np.zeros((m, n), dtype=dtype) + else: + Y = np.zeros((n, n), dtype=dtype) + Z = np.zeros((m, n), dtype=dtype) + + # Fill the matrices + if remove_mean: + cr_start('DMD.temporal_mean',0) + mean = temporal_mean(X) + + Y[:m,:] = subtract_mean(X[:,:-1], mean) + Z[:,:] = subtract_mean(X[:,1:], mean) + cr_stop('DMD.temporal_mean',0) + else: + Y[:m,:] = X[:,:-1].copy() + Z[:,:] = X[:,1:].copy() + + return Y, Z + +def resolvent(A, f): + + # H_inv = (-1j * f - A) + H_inv = (f - A) # COMPTE ELS SGINES! + V, S_inv, UT = svd(H_inv) + U = dagger(UT) + S = 1 / S_inv + + return U, S, V \ No newline at end of file diff --git a/pyLOM/vmmath/linear.pyx b/pyLOM/vmmath/linear.pyx new file mode 100644 index 00000000..00927e2c --- /dev/null +++ b/pyLOM/vmmath/linear.pyx @@ -0,0 +1,471 @@ +#!/usr/bin/env cpython +# +# pyLOM - Python Low Order Modeling. +# +# Linear operator module +# +# Last rev: 31/08/2026 +cimport cython +cimport numpy as np + +import numpy as np + +cdef extern from "" nogil: + float complex I + # Decomposing complex values + float cimagf(float complex z) + float crealf(float complex z) + double cimag(double complex z) + double creal(double complex z) +cdef double complex J = 1j +from libc.stdlib cimport malloc, free +from libc.string cimport memcpy, memset +from ..vmmath.cfuncs cimport real, real_complex, real_float, real_double, real_full +from ..vmmath.cfuncs cimport c_stranspose, c_smatmul, c_smatmulp, c_stsqr_svd, c_scompute_truncation_residual, c_scompute_truncation, c_stemporal_mean, c_ssubtract_mean +from ..vmmath.cfuncs cimport c_dtranspose, c_dmatmul, c_dmatmulp, c_dtsqr_svd, c_dcompute_truncation_residual, c_dcompute_truncation, c_dtemporal_mean, c_dsubtract_mean +from ..vmmath.cfuncs cimport c_csvd, c_cdagger +from ..vmmath.cfuncs cimport c_zsvd, c_zdagger + +from ..utils.cr import cr, cr_start, cr_stop +from ..utils.errors import raiseError + +## Cython functions +@cython.initializedcheck(False) +@cython.boundscheck(False) # turn off bounds-checking for entire function +@cython.wraparound(False) # turn off negative index wrapping for entire function +@cython.nonecheck(False) +@cython.cdivision(True) # turn off zero division check +cdef tuple _slinear_operator(float[:,:] Y, float[:,:] Z, float r): + ''' + I dont understand mn here (old version). I think it is to avoid a mmalloc(my*nn*sizeof(float)) + S = malloc(nn*sizeof(float)) + V = malloc(nn*nn*sizeof(float)) + retval = c_stsqr_svd(U_aux, S, V, &Y[0,0], my, nn) + cr_stop('DMD.SVD',0) + if not retval == 0: raiseError('Problems computing SVD!') + + # Remove artificial 0 rows + cdef float *U + U = malloc(mz*nn*sizeof(float)) + memcpy(U, U_aux, mz*nn*sizeof(float)) + free(U_aux) + + # Truncate + cr_start('DMD.truncate',0) + cdef int nr + + nr = int(r) if r > 1 else c_scompute_truncation_residual(S,r,nn) + cdef np.ndarray[np.float32_t,ndim=2] Ur = np.zeros((mz, nr),dtype=np.float32) + cdef np.ndarray[np.float32_t,ndim=1] Sr = np.zeros((nr),dtype=np.float32) + cdef np.ndarray[np.float32_t,ndim=2] Vr = np.zeros((nr, nn),dtype=np.float32) + c_scompute_truncation(&Ur[0,0],&Sr[0],&Vr[0,0],U,S,V,mz,nn,nn,nr) + + free(U) + free(V) + free(S) + cr_stop('DMD.truncate',0) + + # Project Jacobian of the snapshots into the POD basis + cr_start('DMD.linear_mapping',0) + cdef float *aux1 + cdef float *aux2 + cdef float *aux3 + cdef float *Urt + aux1 = malloc(nr*nn*sizeof(float)) + aux2 = malloc(nr*nn*sizeof(float)) + aux3 = malloc(nr*sizeof(float)) + Urt = malloc(nr*mz*sizeof(float)) + cdef np.ndarray[np.float32_t,ndim=2] Atilde = np.zeros((nr, nr),dtype=np.float32) + c_stranspose(&Ur[0,0], Urt, mz, nr) + c_smatmulp(aux1, Urt, &Z[0,0], nr, nn, mz) + for icol in range(nn): + for irow in range(nr): + aux2[icol*nr + irow] = Vr[irow, icol]/Sr[irow] + c_smatmul(Ã[0,0], aux1, aux2, nr, nr, nn) + free(aux1) + free(aux2) + free(aux3) + free(Urt) + cr_stop('DMD.linear_mapping',0) + + return Ur, Sr, Vr, Atilde + + +@cython.initializedcheck(False) +@cython.boundscheck(False) # turn off bounds-checking for entire function +@cython.wraparound(False) # turn off negative index wrapping for entire function +@cython.nonecheck(False) +@cython.cdivision(True) # turn off zero division check +cdef tuple _dlinear_operator(double[:,:] Y, double[:,:] Z, double r): + ''' + I dont understand mn here (old version). I think it is to avoid a mmalloc(my*nn*sizeof(double)) + S = malloc(nn*sizeof(double)) + V = malloc(nn*nn*sizeof(double)) + retval = c_dtsqr_svd(U_aux, S, V, &Y[0,0], my, nn) + cr_stop('DMD.SVD',0) + if not retval == 0: raiseError('Problems computing SVD!') + + # Remove artificial 0 rows + cdef double *U + U = malloc(mz*nn*sizeof(double)) + memcpy(U, U_aux, mz*nn*sizeof(double)) + free(U_aux) + + # Truncate + cr_start('DMD.truncate',0) + cdef int nr + + nr = int(r) if r > 1 else c_dcompute_truncation_residual(S,r,n-1) + cdef np.ndarray[np.double_t,ndim=2] Ur = np.zeros((mz, nr),dtype=np.double) + cdef np.ndarray[np.double_t,ndim=1] Sr = np.zeros((nr),dtype=np.double) + cdef np.ndarray[np.double_t,ndim=2] Vr = np.zeros((nr, nn),dtype=np.double) + c_dcompute_truncation(&Ur[0,0],&Sr[0],&Vr[0,0],U,S,V,mz,nn,nn,nr) + + free(U) + free(V) + free(S) + cr_stop('DMD.truncate',0) + + # Project Jacobian of the snapshots into the POD basis + cr_start('DMD.linear_mapping',0) + cdef double *aux1 + cdef double *aux2 + cdef double *aux3 + cdef double *Urt + aux1 = malloc(nr*nn*sizeof(double)) + aux2 = malloc(nr*nn*sizeof(double)) + aux3 = malloc(nr*sizeof(double)) + Urt = malloc(nr*mz*sizeof(double)) + cdef np.ndarray[np.double_t,ndim=2] Atilde = np.zeros((nr, nr),dtype=np.double) + c_dtranspose(&Ur[0,0], Urt, mz, nr) + c_dmatmulp(aux1, Urt, &Z[0,0], nr, nn, mz) + for icol in range(nn): + for irow in range(nr): + aux2[icol*nr + irow] = Vr[irow, icol]/Sr[irow] + c_dmatmul(Ã[0,0], aux1, aux2, nr, nr, nn) + free(aux1) + free(aux2) + free(aux3) + free(Urt) + cr_stop('DMD.linear_mapping',0) + + return Ur, Sr, Vr, Atilde + +@cython.boundscheck(False) # turn off bounds-checking for entire function +@cython.wraparound(False) # turn off negative index wrapping for entire function +@cython.nonecheck(False) +@cython.cdivision(True) # turn off zero division check +def linear_operator(real[:,:] Y, real[:,:] Z, real r): + ''' + + ''' + if real is double: + return _dlinear_operator(Y,Z,r) + else: + return _slinear_operator(Y,Z,r) + +@cython.initializedcheck(False) +@cython.boundscheck(False) # turn off bounds-checking for entire function +@cython.wraparound(False) # turn off negative index wrapping for entire function +@cython.nonecheck(False) +@cython.cdivision(True) # turn off zero division check +cdef tuple _sconcatenate(list X, int remove_mean): + cdef int ii, jj, m, m_aux, n, ni, offset + + m = X[0].shape[0] + n = sum([M.shape[1] - 1 for M in X]) + + # Create the matrices Y, Z + if m > n: + m_aux = m + else: + m_aux = n + cdef np.ndarray[np.float32_t,ndim=2] Y = np.zeros((m_aux, n),dtype=np.float32) + cdef np.ndarray[np.float32_t,ndim=2] Z = np.zeros((m, n),dtype=np.float32) + + # Fill the matrices + offset = 0 + cdef float *X_mean + cdef float *X_meanless + cdef float[:, ::1] M + X_mean = malloc(m*sizeof(float)) + for ii in range(len(X)): + M = X[ii] + ni = M.shape[1] + X_meanless = malloc(m*ni*sizeof(float)) + if remove_mean: + cr_start('DMD.temporal_mean',0) + c_stemporal_mean(X_mean,&M[0,0],m,ni) + c_ssubtract_mean(X_meanless,&M[0,0],X_mean,m,ni) + for jj in range(m): + memcpy(&Y[jj, offset], &X_meanless[jj*ni], (ni-1)*sizeof(float)) + memcpy(&Z[jj, offset], &X_meanless[jj*ni+1], (ni-1)*sizeof(float)) + offset += (ni-1) + cr_stop('DMD.temporal_mean',0) + else: + Y[:m,offset:offset + ni-1] = M[:,:-1] + Z[:,offset:offset + ni-1] = M[:,1:] + offset += (ni-1) + free(X_meanless) + + free(X_mean) + + return Y, Z + +@cython.initializedcheck(False) +@cython.boundscheck(False) # turn off bounds-checking for entire function +@cython.wraparound(False) # turn off negative index wrapping for entire function +@cython.nonecheck(False) +@cython.cdivision(True) # turn off zero division check +cdef tuple _dconcatenate(list X, int remove_mean): + cdef int ii, jj, m, m_aux, n, ni, offset + + m = X[0].shape[0] + n = sum([M.shape[1] - 1 for M in X]) + + # Create the matrices Y, Z + if m > n: + m_aux = m + else: + m_aux = n + cdef np.ndarray[np.double_t,ndim=2] Y = np.zeros((m_aux, n),dtype=np.double) + cdef np.ndarray[np.double_t,ndim=2] Z = np.zeros((m, n),dtype=np.double) + + # Fill the matrices + offset = 0 + cdef double *X_mean + cdef double *X_meanless + cdef double[:, ::1] M + X_mean = malloc(m*sizeof(double)) + for ii in range(len(X)): + M = X[ii] + ni = M.shape[1] + X_meanless = malloc(m*ni*sizeof(double)) + if remove_mean: + cr_start('DMD.temporal_mean',0) + c_dtemporal_mean(X_mean,&M[0,0],m,ni) + c_dsubtract_mean(X_meanless,&M[0,0],X_mean,m,ni) + for jj in range(m): + memcpy(&Y[jj, offset], &X_meanless[jj*ni], (ni-1)*sizeof(double)) + memcpy(&Z[jj, offset], &X_meanless[jj*ni+1], (ni-1)*sizeof(double)) + offset += (ni-1) + cr_stop('DMD.temporal_mean',0) + else: + Y[:m,offset:offset + ni-1] = M[:,:-1] + Z[:,offset:offset + ni-1] = M[:,1:] + offset += (ni-1) + free(X_meanless) + + free(X_mean) + + return Y, Z + +@cython.boundscheck(False) # turn off bounds-checking for entire function +@cython.wraparound(False) # turn off negative index wrapping for entire function +@cython.nonecheck(False) +@cython.cdivision(True) # turn off zero division check +def concatenate(list X, remove_mean=False): + + if X[0].dtype == np.double: + return _dconcatenate(X,remove_mean) + else: + return _sconcatenate(X,remove_mean) + + +@cython.initializedcheck(False) +@cython.boundscheck(False) # turn off bounds-checking for entire function +@cython.wraparound(False) # turn off negative index wrapping for entire function +@cython.nonecheck(False) +@cython.cdivision(True) # turn off zero division check +cdef tuple _sseparate(float[:,:] X, int remove_mean): + cdef int jj, m, m_aux, n + + m = X.shape[0] + n = X.shape[1] + + # Create the matrices Y, Z + if m > n: + m_aux = m + else: + m_aux = n + cdef np.ndarray[np.float32_t,ndim=2] Y = np.zeros((m_aux, n),dtype=np.float32) + cdef np.ndarray[np.float32_t,ndim=2] Z = np.zeros((m, n),dtype=np.float32) + + # Fill the matrices + cdef float *X_mean + cdef float *X_meanless + X_mean = malloc(m*sizeof(float)) + X_meanless = malloc(m*n*sizeof(float)) + if remove_mean: + cr_start('DMD.temporal_mean',0) + c_stemporal_mean(X_mean,&X[0,0],m,n) + c_ssubtract_mean(X_meanless,&X[0,0],X_mean,m,n) + for jj in range(m): + memcpy(&Y[jj, 0], &X_meanless[jj*n], (n-1)*sizeof(float)) + memcpy(&Z[jj, 0], &X_meanless[jj*n+1], (n-1)*sizeof(float)) + cr_stop('DMD.temporal_mean',0) + else: + Y[:m,:] = X[:,:-1] + Z[:,:] = X[:,1:] + + free(X_meanless) + free(X_mean) + + return Y, Z + +@cython.initializedcheck(False) +@cython.boundscheck(False) # turn off bounds-checking for entire function +@cython.wraparound(False) # turn off negative index wrapping for entire function +@cython.nonecheck(False) +@cython.cdivision(True) # turn off zero division check +cdef tuple _dseparate(double[:,:] X, int remove_mean): + cdef int jj, m, m_aux, n + + m = X.shape[0] + n = X.shape[1] + + # Create the matrices Y, Z + if m > n: + m_aux = m + else: + m_aux = n + cdef np.ndarray[np.double_t,ndim=2] Y = np.zeros((m_aux, n),dtype=np.double) + cdef np.ndarray[np.double_t,ndim=2] Z = np.zeros((m, n),dtype=np.double) + + # Fill the matrices + cdef double *X_mean + cdef double *X_meanless + X_mean = malloc(m*sizeof(double)) + X_meanless = malloc(m*n*sizeof(double)) + if remove_mean: + cr_start('DMD.temporal_mean',0) + c_dtemporal_mean(X_mean,&X[0,0],m,n) + c_dsubtract_mean(X_meanless,&X[0,0],X_mean,m,n) + for jj in range(m): + memcpy(&Y[jj, 0], &X_meanless[jj*n], (n-1)*sizeof(double)) + memcpy(&Z[jj, 0], &X_meanless[jj*n+1], (n-1)*sizeof(double)) + cr_stop('DMD.temporal_mean',0) + else: + Y[:m,:] = X[:,:-1] + Z[:,:] = X[:,1:] + + free(X_meanless) + free(X_mean) + + return Y, Z + + +@cython.boundscheck(False) # turn off bounds-checking for entire function +@cython.wraparound(False) # turn off negative index wrapping for entire function +@cython.nonecheck(False) +@cython.cdivision(True) # turn off zero division check +def separate(real[:,:] X, remove_mean=False): + + if real is double: + return _dseparate(X,remove_mean) + else: + return _sseparate(X,remove_mean) + + +@cython.initializedcheck(False) +@cython.boundscheck(False) # turn off bounds-checking for entire function +@cython.wraparound(False) # turn off negative index wrapping for entire function +@cython.nonecheck(False) +@cython.cdivision(True) # turn off zero division check +cdef tuple _cresolvent(float[:,:] A, np.complex64_t f): + + cdef int ii, jj, n + n = A.shape[0] + + cdef np.complex64_t *H_inv + H_inv = malloc(n*n*sizeof(np.complex64_t)) + + # COMPTE ELS SIGNES! + for ii in range(n): + for jj in range(n): + if ii == jj: + H_inv[ii*n + jj] = f - A[ii, jj] + else: + H_inv[ii*n + jj] = -A[ii, jj] + + cdef np.complex64_t *UT + UT = malloc(n*n*sizeof(np.complex64_t)) + cdef np.ndarray[np.float32_t,ndim=1] S_inv = np.zeros((n),dtype=np.float32) + cdef np.ndarray[np.float32_t,ndim=1] S = np.zeros((n),dtype=np.float32) + cdef np.ndarray[np.complex64_t,ndim=2] V = np.zeros((n, n),dtype=np.complex64) + cdef np.ndarray[np.complex64_t,ndim=2] U = np.zeros((n, n),dtype=np.complex64) + c_csvd(&V[0,0], &S_inv[0], UT, H_inv, n, n) + + for ii in range(n): + S[ii] = 1 / S_inv[ii] + + c_cdagger(UT, &U[0,0], n, n) + + return U, S, V + +@cython.initializedcheck(False) +@cython.boundscheck(False) # turn off bounds-checking for entire function +@cython.wraparound(False) # turn off negative index wrapping for entire function +@cython.nonecheck(False) +@cython.cdivision(True) # turn off zero division check +cdef tuple _zresolvent(double[:,:] A, np.complex128_t f): + + cdef int ii, jj, n + n = A.shape[0] + + cdef np.complex128_t *H_inv + H_inv = malloc(n*n*sizeof(np.complex128_t)) + + # COMPTE ELS SIGNES! + for ii in range(n): + for jj in range(n): + if ii == jj: + H_inv[ii*n + jj] = f - A[ii, jj] + else: + H_inv[ii*n + jj] = -A[ii, jj] + + cdef np.complex128_t *UT + UT = malloc(n*n*sizeof(np.complex128_t)) + cdef np.ndarray[np.double_t,ndim=1] S_inv = np.zeros((n),dtype=np.double) + cdef np.ndarray[np.double_t,ndim=1] S = np.zeros((n),dtype=np.double) + cdef np.ndarray[np.complex128_t,ndim=2] V = np.zeros((n, n),dtype=np.complex128) + cdef np.ndarray[np.complex128_t,ndim=2] U = np.zeros((n, n),dtype=np.complex128) + c_zsvd(&V[0,0], &S[0], UT, H_inv, n, n) + + for ii in range(n): + S[ii] = 1 / S_inv[ii] + + c_zdagger(UT, &U[0,0], n, n) + + return U, S, V + +@cython.boundscheck(False) # turn off bounds-checking for entire function +@cython.wraparound(False) # turn off negative index wrapping for entire function +@cython.nonecheck(False) +@cython.cdivision(True) # turn off zero division check +def resolvent(real[:,:] X, object f): + + if real is double: + return _zresolvent(X,f) + else: + return _cresolvent(X,f) \ No newline at end of file diff --git a/pyLOM/vmmath/src/linear.c b/pyLOM/vmmath/src/linear.c new file mode 100644 index 00000000..20815e18 --- /dev/null +++ b/pyLOM/vmmath/src/linear.c @@ -0,0 +1,114 @@ +/* + Linear operator +*/ + +#include +#include +#include "mpi.h" +#include "averaging.h" +#include "vector_matrix.h" +#include "truncation.h" +#include "svd.h" + + +int slinear_operator(float *U, float *S, float *VT, float *Atilde, float *Y, float *Z, const float r, const int my, const int mz, const int nn) { + int icol, irow, retval; + + // Compute SVD + float *U_aux; + float *S_all; + float *VT_all; + U_aux = (float*)malloc(my*nn*sizeof(float)); + S_all = (float*)malloc(nn*sizeof(float)); + VT_all = (float*)malloc(nn*nn*sizeof(float)); + retval = stsqr_svd(U_aux, S_all, VT_all, Y, my, nn); + free(Y); + + float *U_all; + U_all = (float*)malloc(mz*nn*sizeof(float)); + sremove_rows(U_aux, U_all, mz, nn); + free(U_aux); + + // Truncate + int nr; + if (r > 1) nr = r; + else scompute_truncation_residual(S_all, r, nn); + scompute_truncation(U, S, VT, U_all, S_all, VT_all, mz, nn, nn, nr); + + free(U_all); + free(S_all); + free(VT_all); + + // Project Jacobian of the snapshots into the POD basis + float *aux1, *aux2, *aux3, *Urt; + aux1 = (float*)malloc(nr*nn*sizeof(float)); + aux2 = (float*)malloc(nr*nn*sizeof(float)); + aux3 = (float*)malloc(nr*sizeof(float)); + Urt = (float*)malloc(nr*mz*sizeof(float)); + stranspose(U, Urt, mz, nr); + smatmulp(aux1, Urt, Z, nr, nn, mz); + free(Z); + + for (icol=0; icol 1) nr = r; + else dcompute_truncation_residual(S_all, r, nn); + dcompute_truncation(U, S, VT, U_all, S_all, VT_all, mz, nn, nn, nr); + + free(U_all); + free(S_all); + free(VT_all); + + // Project Jacobian of the snapshots into the POD basis + double *aux1, *aux2, *aux3, *Urt; + aux1 = (double*)malloc(nr*nn*sizeof(double)); + aux2 = (double*)malloc(nr*nn*sizeof(double)); + aux3 = (double*)malloc(nr*sizeof(double)); + Urt = (double*)malloc(nr*mz*sizeof(double)); + dtranspose(U, Urt, mz, nr); + dmatmulp(aux1, Urt, Z, nr, nn, mz); + free(Z); + + for (icol=0; icol Date: Mon, 21 Sep 2026 09:29:43 +0200 Subject: [PATCH 20/27] functional resolvent --- pyLOM/RES/wrapper.py | 5 +- pyLOM/RES/wrapper.pyx | 54 +++++++------- pyLOM/vmmath/__init__.py | 2 +- pyLOM/vmmath/cfuncs.pxd | 18 ++++- pyLOM/vmmath/linear.py | 8 +- pyLOM/vmmath/linear.pyx | 152 ++++++++++++++++++++++++++++++++------ pyLOM/vmmath/src/linear.c | 62 +++++++++++++++- pyLOM/vmmath/src/linear.h | 15 +++- setup.py | 3 + 9 files changed, 261 insertions(+), 58 deletions(-) diff --git a/pyLOM/RES/wrapper.py b/pyLOM/RES/wrapper.py index ef88f309..49e7a190 100644 --- a/pyLOM/RES/wrapper.py +++ b/pyLOM/RES/wrapper.py @@ -55,7 +55,7 @@ def run(Phi, delta, omega, f, Q=None): return U_res, S, V_res -def run_new(X, w, r, remove_mean = True): +def run_new(X, w, r, remove_mean=True): # Prepare matrices and calculate the linear operator if (type(X) is list): @@ -69,9 +69,10 @@ def run_new(X, w, r, remove_mean = True): del S1, VT1 # Calculate the resolvent of the linear operator - U2, S, V = resolvent(Atilde, w) + U2, S, V2 = resolvent(Atilde, w) # Project the solution U = matmul(U1, U2) + V = matmul(U1, V2) return U, S, V \ No newline at end of file diff --git a/pyLOM/RES/wrapper.pyx b/pyLOM/RES/wrapper.pyx index 126f99eb..e2d90e79 100644 --- a/pyLOM/RES/wrapper.pyx +++ b/pyLOM/RES/wrapper.pyx @@ -330,17 +330,19 @@ def _crun_old(float[:,:] X, np.complex64_t w, float r, int remove_mean): del Atilde + cdef np.ndarray[np.complex64_t, ndim=2] U1_c = np.ascontiguousarray(U1, dtype=np.complex64) + cdef np.ndarray[np.complex64_t,ndim=2] U = np.zeros((m,nr),dtype=np.complex64) cdef np.ndarray[np.complex64_t,ndim=2] V = np.zeros((m,nr),dtype=np.complex64) - c_cmatmul(&U[0,0], &U1[0,0], &U2[0,0], m, nr, nr) - c_cmatmul(&V[0,0], &U1[0,0], &V2[0,0], m, nr, nr) + c_cmatmul(&U[0,0], &U1_c[0,0], &U2[0,0], m, nr, nr) + c_cmatmul(&V[0,0], &U1_c[0,0], &V2[0,0], m, nr, nr) - del U1, U2, V2 + del U1, U2, V2, U1_c return U, S, V def _zrun_old(double[:,:] X, np.complex128_t w, double r, int remove_mean): - print('variables', flush=True) + # Variables cdef int m, n cdef double[:, ::1] Y @@ -349,19 +351,19 @@ def _zrun_old(double[:,:] X, np.complex128_t w, double r, int remove_mean): cdef double[::1] S1 cdef double[:, ::1] VT1 cdef double[:, ::1] Atilde - print('snapshots', flush=True) + # Create the snapshot of matrices and separate it into Y, Z Y, Z = _dseparate(X, remove_mean) m = Z.shape[0] n = Z.shape[1] + 1 - print('linear operator', flush=True) + # Compute the linear operator in lower dimension U1, S1, VT1, Atilde = _dlinear_operator(Y, Z, r) cdef int nr nr = Atilde.shape[1] del S1, VT1 - print('resolvent', flush=True) + cdef np.complex128_t[:, ::1] U2 cdef np.ndarray[np.double_t,ndim=1] S = np.zeros((nr),dtype=np.double) cdef np.complex128_t[:, ::1] V2 @@ -369,13 +371,15 @@ def _zrun_old(double[:,:] X, np.complex128_t w, double r, int remove_mean): U2, S, V2 = _zresolvent(Atilde, w) del Atilde - print('project', flush=True) + + cdef np.ndarray[np.complex128_t, ndim=2] U1_c = np.ascontiguousarray(U1, dtype=np.complex128) + cdef np.ndarray[np.complex128_t,ndim=2] U = np.zeros((m,nr),dtype=np.complex128) cdef np.ndarray[np.complex128_t,ndim=2] V = np.zeros((m,nr),dtype=np.complex128) - c_zmatmul(&U[0,0], &U1[0,0], &U2[0,0], m, nr, nr) - c_zmatmul(&V[0,0], &U1[0,0], &V2[0,0], m, nr, nr) + c_zmatmul(&U[0,0], &U1_c[0,0], &U2[0,0], m, nr, nr) + c_zmatmul(&V[0,0], &U1_c[0,0], &V2[0,0], m, nr, nr) - del U1, U2, V2 + del U1, U2, V2, U1_c return U, S, V @@ -409,12 +413,14 @@ def _crun_new(list X, np.complex64_t w, float r, int remove_mean): del Atilde + cdef np.ndarray[np.complex64_t, ndim=2] U1_c = np.ascontiguousarray(U1, dtype=np.complex64) + cdef np.ndarray[np.complex64_t,ndim=2] U = np.zeros((m,nr),dtype=np.complex64) cdef np.ndarray[np.complex64_t,ndim=2] V = np.zeros((m,nr),dtype=np.complex64) - c_cmatmul(&U[0,0], &U1[0,0], &U2[0,0], m, nr, nr) - c_cmatmul(&V[0,0], &U1[0,0], &V2[0,0], m, nr, nr) + c_cmatmul(&U[0,0], &U1_c[0,0], &U2[0,0], m, nr, nr) + c_cmatmul(&V[0,0], &U1_c[0,0], &V2[0,0], m, nr, nr) - del U1, U2 + del U1, U2, V2, U1_c return U, S, V @@ -448,12 +454,14 @@ def _zrun_new(list X, np.complex128_t w, double r, int remove_mean): del Atilde + cdef np.ndarray[np.complex128_t, ndim=2] U1_c = np.ascontiguousarray(U1, dtype=np.complex128) + cdef np.ndarray[np.complex128_t,ndim=2] U = np.zeros((m,nr),dtype=np.complex128) cdef np.ndarray[np.complex128_t,ndim=2] V = np.zeros((m,nr),dtype=np.complex128) - c_zmatmul(&U[0,0], &U1[0,0], &U2[0,0], m, nr, nr) - c_zmatmul(&V[0,0], &U1[0,0], &V2[0,0], m, nr, nr) + c_zmatmul(&U[0,0], &U1_c[0,0], &U2[0,0], m, nr, nr) + c_zmatmul(&V[0,0], &U1_c[0,0], &V2[0,0], m, nr, nr) - del U1, U2, V2 + del U1, U2, V2, U1_c return U, S, V @@ -466,15 +474,11 @@ def run_new(object X, object w, object r, int remove_mean=True): if isinstance(X, list): if X[0].dtype == np.double: - print('function1', flush=True) - return _zrun_new(X,w,r,remove_mean) + return _zrun_new(X,np.complex128(w),np.double(r),remove_mean) else: - print('function2', flush=True) - return _crun_new(X,w,r,remove_mean) + return _crun_new(X,np.complex64(w),np.float32(r),remove_mean) else: if X[0].dtype == np.double: - print('function3', flush=True) - return _zrun_old(X,w,r,remove_mean) + return _zrun_old(X,np.complex128(w),np.double(r),remove_mean) else: - print('function4', flush=True) - return _crun_old(X,w,r,remove_mean) \ No newline at end of file + return _crun_old(X,np.complex64(w),np.float32(r),remove_mean) \ No newline at end of file diff --git a/pyLOM/vmmath/__init__.py b/pyLOM/vmmath/__init__.py index 8cbaf574..474802b9 100644 --- a/pyLOM/vmmath/__init__.py +++ b/pyLOM/vmmath/__init__.py @@ -27,7 +27,7 @@ # Data processing module from .dataprocessing import data_splitting, time_delay_embedding, find_random_sensors # Linear matrix module -from .linear import linear_operator, concatenate, separate, resolvent +from .linear import linear_operator, concatenate, separate, resolvent, flip_columns del maths, averaging, truncation, stats, geometric, regression, dataprocessing, linear diff --git a/pyLOM/vmmath/cfuncs.pxd b/pyLOM/vmmath/cfuncs.pxd index 1e903eb5..cd237d44 100644 --- a/pyLOM/vmmath/cfuncs.pxd +++ b/pyLOM/vmmath/cfuncs.pxd @@ -176,7 +176,17 @@ cdef extern from "regression.h" nogil: # Double version cdef void c_dleast_squares "dleast_squares"(double *out, double *A, double *b, const int m, const int n) cdef void c_dridge_regression "dridge_regression"(double *out, double *A, double *b, double lam, const int m, const int n) - +cdef extern from "linear.h" nogil: + # Single precision + cdef int c_slinear_operator "slinear_operator"(float *U, float *S, float *VT, float *Atilde, float *Y, float *Z, const float r, const int m, const int n) + cdef void c_sflip_columns "sflip_columns"(float *A, float *B, int m, int n) + # Double precision + cdef int c_dlinear_operator "dlinear_operator"(float *U, float *S, float *VT, float *Atilde, float *Y, float *Z, const float r, const int m, const int n) + cdef void c_dflip_columns "dflip_columns"(double *A, double *B, int m, int n) + # Single complex precision + cdef void c_cflip_columns "cflip_columns"(np.complex64_t *A, np.complex64_t *B, int m, int n) + # Double complex precision + cdef void c_zflip_columns "zflip_columns"(np.complex128_t *A, np.complex128_t *B, int m, int n) ## Fused type between double and complex ctypedef fused real: @@ -189,4 +199,10 @@ ctypedef fused real_full: float double np.complex64_t + np.complex128_t +ctypedef fused real_float: + float + np.complex64_t +ctypedef fused real_double: + double np.complex128_t \ No newline at end of file diff --git a/pyLOM/vmmath/linear.py b/pyLOM/vmmath/linear.py index a1987c1a..93e01d9a 100644 --- a/pyLOM/vmmath/linear.py +++ b/pyLOM/vmmath/linear.py @@ -103,4 +103,10 @@ def resolvent(A, f): U = dagger(UT) S = 1 / S_inv - return U, S, V \ No newline at end of file + return U, S, V + +def flip_columns(A): + + B = np.fliplr(A) + + return B \ No newline at end of file diff --git a/pyLOM/vmmath/linear.pyx b/pyLOM/vmmath/linear.pyx index 00927e2c..17017b7b 100644 --- a/pyLOM/vmmath/linear.pyx +++ b/pyLOM/vmmath/linear.pyx @@ -5,11 +5,13 @@ # Linear operator module # # Last rev: 31/08/2026 + cimport cython cimport numpy as np import numpy as np +#from libc.complex cimport creal, cimag cdef extern from "" nogil: float complex I # Decomposing complex values @@ -21,10 +23,10 @@ cdef double complex J = 1j from libc.stdlib cimport malloc, free from libc.string cimport memcpy, memset from ..vmmath.cfuncs cimport real, real_complex, real_float, real_double, real_full -from ..vmmath.cfuncs cimport c_stranspose, c_smatmul, c_smatmulp, c_stsqr_svd, c_scompute_truncation_residual, c_scompute_truncation, c_stemporal_mean, c_ssubtract_mean -from ..vmmath.cfuncs cimport c_dtranspose, c_dmatmul, c_dmatmulp, c_dtsqr_svd, c_dcompute_truncation_residual, c_dcompute_truncation, c_dtemporal_mean, c_dsubtract_mean -from ..vmmath.cfuncs cimport c_csvd, c_cdagger -from ..vmmath.cfuncs cimport c_zsvd, c_zdagger +from ..vmmath.cfuncs cimport c_stranspose, c_smatmul, c_smatmulp, c_stsqr_svd, c_scompute_truncation_residual, c_scompute_truncation, c_stemporal_mean, c_ssubtract_mean, c_sflip_columns +from ..vmmath.cfuncs cimport c_dtranspose, c_dmatmul, c_dmatmulp, c_dtsqr_svd, c_dcompute_truncation_residual, c_dcompute_truncation, c_dtemporal_mean, c_dsubtract_mean, c_dflip_columns +from ..vmmath.cfuncs cimport c_csvd, c_cdagger, c_cflip_columns +from ..vmmath.cfuncs cimport c_zsvd, c_zdagger, c_zflip_columns from ..utils.cr import cr, cr_start, cr_stop from ..utils.errors import raiseError @@ -408,18 +410,31 @@ cdef tuple _cresolvent(float[:,:] A, np.complex64_t f): else: H_inv[ii*n + jj] = -A[ii, jj] - cdef np.complex64_t *UT - UT = malloc(n*n*sizeof(np.complex64_t)) - cdef np.ndarray[np.float32_t,ndim=1] S_inv = np.zeros((n),dtype=np.float32) + cdef np.complex64_t *UT_flip + cdef np.complex64_t *V_flip + cdef np.complex64_t *U_flip + cdef float *S_inv + UT_flip = malloc(n*n*sizeof(np.complex64_t)) + V_flip = malloc(n*n*sizeof(np.complex64_t)) + U_flip = malloc(n*n*sizeof(np.complex64_t)) + S_inv = malloc(n*sizeof(np.complex64_t)) cdef np.ndarray[np.float32_t,ndim=1] S = np.zeros((n),dtype=np.float32) - cdef np.ndarray[np.complex64_t,ndim=2] V = np.zeros((n, n),dtype=np.complex64) - cdef np.ndarray[np.complex64_t,ndim=2] U = np.zeros((n, n),dtype=np.complex64) - c_csvd(&V[0,0], &S_inv[0], UT, H_inv, n, n) + c_csvd(V_flip, S_inv, UT_flip, H_inv, n, n) + free(H_inv) for ii in range(n): - S[ii] = 1 / S_inv[ii] - - c_cdagger(UT, &U[0,0], n, n) + S[ii] = 1 / S_inv[n-1-ii] + free(S_inv) + + c_cdagger(UT_flip, U_flip, n, n) + free(UT_flip) + + cdef np.ndarray[np.complex64_t,ndim=2] V = np.zeros((n, n),dtype=np.complex64) + cdef np.ndarray[np.complex64_t,ndim=2] U = np.zeros((n, n),dtype=np.complex64) + c_cflip_columns(U_flip, &U[0,0], n, n) + c_cflip_columns(V_flip, &V[0,0], n, n) + free(U_flip) + free(V_flip) return U, S, V @@ -444,19 +459,32 @@ cdef tuple _zresolvent(double[:,:] A, np.complex128_t f): else: H_inv[ii*n + jj] = -A[ii, jj] - cdef np.complex128_t *UT - UT = malloc(n*n*sizeof(np.complex128_t)) - cdef np.ndarray[np.double_t,ndim=1] S_inv = np.zeros((n),dtype=np.double) + cdef np.complex128_t *UT_flip + cdef np.complex128_t *V_flip + cdef np.complex128_t *U_flip + cdef double *S_inv + UT_flip = malloc(n*n*sizeof(np.complex128_t)) + V_flip = malloc(n*n*sizeof(np.complex128_t)) + U_flip = malloc(n*n*sizeof(np.complex128_t)) + S_inv = malloc(n*sizeof(np.complex128_t)) cdef np.ndarray[np.double_t,ndim=1] S = np.zeros((n),dtype=np.double) - cdef np.ndarray[np.complex128_t,ndim=2] V = np.zeros((n, n),dtype=np.complex128) - cdef np.ndarray[np.complex128_t,ndim=2] U = np.zeros((n, n),dtype=np.complex128) - c_zsvd(&V[0,0], &S[0], UT, H_inv, n, n) + c_zsvd(V_flip, S_inv, UT_flip, H_inv, n, n) + free(H_inv) for ii in range(n): - S[ii] = 1 / S_inv[ii] + S[ii] = 1 / S_inv[n-1-ii] + free(S_inv) + + c_zdagger(UT_flip, U_flip, n, n) + free(UT_flip) + + cdef np.ndarray[np.complex128_t,ndim=2] V = np.zeros((n, n),dtype=np.complex128) + cdef np.ndarray[np.complex128_t,ndim=2] U = np.zeros((n, n),dtype=np.complex128) + c_zflip_columns(U_flip, &U[0,0], n, n) + c_zflip_columns(V_flip, &V[0,0], n, n) + free(U_flip) + free(V_flip) - c_zdagger(UT, &U[0,0], n, n) - return U, S, V @cython.boundscheck(False) # turn off bounds-checking for entire function @@ -466,6 +494,82 @@ cdef tuple _zresolvent(double[:,:] A, np.complex128_t f): def resolvent(real[:,:] X, object f): if real is double: - return _zresolvent(X,f) + return _zresolvent(X,np.complex128(f)) + else: + return _cresolvent(X,np.complex64(f)) + + +@cython.initializedcheck(False) +@cython.boundscheck(False) # turn off bounds-checking for entire function +@cython.wraparound(False) # turn off negative index wrapping for entire function +@cython.nonecheck(False) +@cython.cdivision(True) # turn off zero division check +cdef np.ndarray[np.float32_t,ndim=2] _sflip_columns(float[:,:] A): + ''' + + ''' + cdef int m = A.shape[0], n = A.shape[1] + cdef np.ndarray[np.float32_t,ndim=2] B = np.zeros((m,n),dtype=np.float32) + c_sflip_columns(&A[0,0], &B[0,0], m,n) + return B + +@cython.initializedcheck(False) +@cython.boundscheck(False) # turn off bounds-checking for entire function +@cython.wraparound(False) # turn off negative index wrapping for entire function +@cython.nonecheck(False) +@cython.cdivision(True) # turn off zero division check +cdef np.ndarray[np.double_t,ndim=2] _dflip_columns(double[:,:] A): + ''' + + ''' + cdef int m = A.shape[0], n = A.shape[1] + cdef np.ndarray[np.double_t,ndim=2] B = np.zeros((m,n),dtype=np.double) + c_dflip_columns(&A[0,0], &B[0,0], m,n) + return B + +@cython.initializedcheck(False) +@cython.boundscheck(False) # turn off bounds-checking for entire function +@cython.wraparound(False) # turn off negative index wrapping for entire function +@cython.nonecheck(False) +@cython.cdivision(True) # turn off zero division check +cdef np.ndarray[np.complex64_t,ndim=2] _cflip_columns(np.complex64_t[:,:] A): + ''' + + ''' + cdef int m = A.shape[0], n = A.shape[1] + cdef np.ndarray[np.complex64_t,ndim=2] B = np.zeros((m,n),dtype=np.complex64) + c_cflip_columns(&A[0,0], &B[0,0], m,n) + return B + +@cython.initializedcheck(False) +@cython.boundscheck(False) # turn off bounds-checking for entire function +@cython.wraparound(False) # turn off negative index wrapping for entire function +@cython.nonecheck(False) +@cython.cdivision(True) # turn off zero division check +cdef np.ndarray[np.complex128_t,ndim=2] _zflip_columns(np.complex128_t[:,:] A): + ''' + + ''' + cdef int m = A.shape[0], n = A.shape[1] + cdef np.ndarray[np.complex128_t,ndim=2] B = np.zeros((m,n),dtype=np.complex128) + c_zflip_columns(&A[0,0], &B[0,0], m,n) + return B + +@cr('math.flip_columns') +@cython.initializedcheck(False) +@cython.boundscheck(False) # turn off bounds-checking for entire function +@cython.wraparound(False) # turn off negative index wrapping for entire function +@cython.nonecheck(False) +@cython.cdivision(True) # turn off zero division check +def flip_columns(real_full[:,:] A): + r''' + + ''' + if real_full is np.complex128_t: + return _zflip_columns(A) + elif real_full is np.complex64_t: + return _cflip_columns(A) + elif real_full is double: + return _dflip_columns(A) else: - return _cresolvent(X,f) \ No newline at end of file + return _sflip_columns(A) \ No newline at end of file diff --git a/pyLOM/vmmath/src/linear.c b/pyLOM/vmmath/src/linear.c index 20815e18..06072250 100644 --- a/pyLOM/vmmath/src/linear.c +++ b/pyLOM/vmmath/src/linear.c @@ -1,15 +1,31 @@ /* Linear operator */ - #include +#include #include +#include #include "mpi.h" +typedef float _Complex scomplex_t; +typedef double _Complex dcomplex_t; + +#ifdef USE_MKL +#define MKL_Complex8 scomplex_t +#define MKL_Complex16 dcomplex_t +#include "mkl.h" +#include "mkl_lapacke.h" +#else +#include "cblas.h" +#include "lapacke.h" +#endif + #include "averaging.h" #include "vector_matrix.h" #include "truncation.h" #include "svd.h" +#include "linear.h" +#define AC_MAT(A,n,i,j) *((A)+(n)*(i)+(j)) int slinear_operator(float *U, float *S, float *VT, float *Atilde, float *Y, float *Z, const float r, const int my, const int mz, const int nn) { int icol, irow, retval; @@ -111,4 +127,48 @@ int dlinear_operator(double *U, double *S, double *VT, double *Atilde, double *Y free(Urt); return retval; +} + +void sflip_columns(float *A, float *B, int m, int n) { + + int ii, jj; + + for (ii=0; ii +typedef float _Complex scomplex_t; +typedef double _Complex dcomplex_t; #ifdef USE_MKL #define MKL_Complex8 scomplex_t #define MKL_Complex16 dcomplex_t #include "mkl.h" #endif -// Float version +// Single precision int slinear_operator(float *U, float *S, float *VT, float *Atilde, float *Y, float *Z, const float r, const int m, const int n); -// DOuble version -int dlinear_operator(double *U, double *S, double *VT, double *Atilde, double *Y, double *Z, const double r, const int m, const int n); \ No newline at end of file +void sflip_columns(float *A, float *B, int m, int n); +// Double precision +int dlinear_operator(double *U, double *S, double *VT, double *Atilde, double *Y, double *Z, const double r, const int m, const int n); +void dflip_columns(double *A, double *B, int m, int n); +// Single complex precision +void cflip_columns(scomplex_t *A, scomplex_t *B, int m, int n); +// Double complex precision +void zflip_columns(dcomplex_t *A, dcomplex_t *B, int m, int n); \ No newline at end of file diff --git a/setup.py b/setup.py index 6973c99d..4218d0db 100644 --- a/setup.py +++ b/setup.py @@ -177,6 +177,7 @@ 'pyLOM/vmmath/src/truncation.c', 'pyLOM/vmmath/src/stats.c', 'pyLOM/vmmath/src/regression.c', + 'pyLOM/vmmath/src/linear.c', ], language = 'c', include_dirs = include_dirs + ['pyLOM/vmmath/src',np.get_include(),mpi4py.get_include()], @@ -276,6 +277,7 @@ 'pyLOM/vmmath/src/svd.c', 'pyLOM/vmmath/src/truncation.c', 'pyLOM/vmmath/src/averaging.c', + 'pyLOM/vmmath/src/linear.c', ], language = 'c', include_dirs = include_dirs + ['pyLOM/vmmath/src',np.get_include(),mpi4py.get_include()], @@ -330,6 +332,7 @@ 'pyLOM/vmmath/src/qr.c', 'pyLOM/vmmath/src/svd.c', 'pyLOM/vmmath/src/truncation.c', + 'pyLOM/vmmath/src/linear.c', ], language = 'c', include_dirs = include_dirs + ['pyLOM/vmmath/src',np.get_include(),mpi4py.get_include()], From ac95ac0cb3110947adf31ce0d3ab5b3156e3d74c Mon Sep 17 00:00:00 2001 From: msilvestrec03 Date: Mon, 21 Sep 2026 12:16:06 +0200 Subject: [PATCH 21/27] remove rows added --- pyLOM/vmmath/truncation.py | 7 +++++- pyLOM/vmmath/truncation.pyx | 45 ++++++++++++++++++++++++++++++++++++- 2 files changed, 50 insertions(+), 2 deletions(-) diff --git a/pyLOM/vmmath/truncation.py b/pyLOM/vmmath/truncation.py index 0815acde..7e52e396 100644 --- a/pyLOM/vmmath/truncation.py +++ b/pyLOM/vmmath/truncation.py @@ -74,4 +74,9 @@ def energy(original, rec): global_den = mpi_reduce(local_den,op='sum',all=True) # Compute Ek (this will be identical on all ranks) - return 1 - global_num / global_den \ No newline at end of file + return 1 - global_num / global_den + +def remove_rows(A, rows): + B = np.zeros((rows, A.shape[1]), dtype=A.dtype) + B[:,:] = A[:rows,:] + return B \ No newline at end of file diff --git a/pyLOM/vmmath/truncation.pyx b/pyLOM/vmmath/truncation.pyx index 26dd7fe7..981ed242 100644 --- a/pyLOM/vmmath/truncation.pyx +++ b/pyLOM/vmmath/truncation.pyx @@ -11,6 +11,7 @@ cimport numpy as np import numpy as np +from libc.string cimport memcpy from libc.math cimport fabs from .cfuncs cimport real, c_svector_norm, c_dvector_norm, c_senergy, c_denergy, c_slocal_energy, c_dlocal_energy from ..utils.cr import cr @@ -177,4 +178,46 @@ def local_energy(real[:,:] A, real[:,:] B): if real is double: return _dlocal_energy(A,B) else: - return _slocal_energy(A,B) \ No newline at end of file + return _slocal_energy(A,B) + +@cython.initializedcheck(False) +@cython.boundscheck(False) # turn off bounds-checking for entire function +@cython.wraparound(False) # turn off negative index wrapping for entire function +@cython.nonecheck(False) +@cython.cdivision(True) # turn off zero division check +cdef float _sremove_rows(float[:,:] A, int rows): + ''' + Get A with the disired number of rows + ''' + cdef int n = A.shape[1] + cdef np.ndarray[np.float32_t,ndim=2] B = np.zeros((rows,n),dtype=np.float32) + memcpy(&B[0,0],&A[0,0],rows*n*sizeof(float)) + return B + +@cython.initializedcheck(False) +@cython.boundscheck(False) # turn off bounds-checking for entire function +@cython.wraparound(False) # turn off negative index wrapping for entire function +@cython.nonecheck(False) +@cython.cdivision(True) # turn off zero division check +cdef float _dremove_rows(double[:,:] A, int rows): + ''' + Get A with the disired number of rows + ''' + cdef int n = A.shape[1] + cdef np.ndarray[np.double_t,ndim=2] B = np.zeros((rows,n),dtype=np.double) + memcpy(&B[0,0],&A[0,0],rows*n*sizeof(double)) + return B + +@cython.initializedcheck(False) +@cython.boundscheck(False) # turn off bounds-checking for entire function +@cython.wraparound(False) # turn off negative index wrapping for entire function +@cython.nonecheck(False) +@cython.cdivision(True) # turn off zero division check +def remove_rows(real[:,:] A, int rows): + ''' + Get A with the disired number of rows + ''' + if real is double: + return _dremove_rows(A, rows) + else: + return _sremove_rows(A, rows) \ No newline at end of file From 3f52812eacdf7060d7b6cf15d14b16249a4273cc Mon Sep 17 00:00:00 2001 From: msilvestrec03 Date: Mon, 21 Sep 2026 12:56:16 +0200 Subject: [PATCH 22/27] completeness of the module --- pyLOM/vmmath/dataprocessing.py | 3 ++- pyLOM/vmmath/src/linear.c | 2 +- pyLOM/vmmath/src/truncation.c | 8 ++++++++ pyLOM/vmmath/src/truncation.h | 2 ++ 4 files changed, 13 insertions(+), 2 deletions(-) diff --git a/pyLOM/vmmath/dataprocessing.py b/pyLOM/vmmath/dataprocessing.py index 618ed6b8..ef921735 100644 --- a/pyLOM/vmmath/dataprocessing.py +++ b/pyLOM/vmmath/dataprocessing.py @@ -1,7 +1,8 @@ import numpy as np from ..utils.mpi import MPI_RANK, mpi_bcast, mpi_reduce -from ..utils import raiseError, is_rank_or_serial +from ..utils import raiseError, is_rank_or_serial, cr_nvtx as cr, cr_start, cr_stop + def data_splitting(Nt:int, mode:str, seed:int=-1): diff --git a/pyLOM/vmmath/src/linear.c b/pyLOM/vmmath/src/linear.c index 06072250..47a1ffad 100644 --- a/pyLOM/vmmath/src/linear.c +++ b/pyLOM/vmmath/src/linear.c @@ -132,7 +132,7 @@ int dlinear_operator(double *U, double *S, double *VT, double *Atilde, double *Y void sflip_columns(float *A, float *B, int m, int n) { int ii, jj; - + // openmp?? for (ii=0; ii Date: Mon, 21 Sep 2026 15:58:24 +0200 Subject: [PATCH 23/27] fix in separate --- pyLOM/vmmath/linear.pyx | 12 ++++++------ 1 file changed, 6 insertions(+), 6 deletions(-) diff --git a/pyLOM/vmmath/linear.pyx b/pyLOM/vmmath/linear.pyx index 17017b7b..ea3d6ec7 100644 --- a/pyLOM/vmmath/linear.pyx +++ b/pyLOM/vmmath/linear.pyx @@ -309,9 +309,9 @@ cdef tuple _sseparate(float[:,:] X, int remove_mean): if m > n: m_aux = m else: - m_aux = n - cdef np.ndarray[np.float32_t,ndim=2] Y = np.zeros((m_aux, n),dtype=np.float32) - cdef np.ndarray[np.float32_t,ndim=2] Z = np.zeros((m, n),dtype=np.float32) + m_aux = n-1 + cdef np.ndarray[np.float32_t,ndim=2] Y = np.zeros((m_aux, n-1),dtype=np.float32) + cdef np.ndarray[np.float32_t,ndim=2] Z = np.zeros((m, n-1),dtype=np.float32) # Fill the matrices cdef float *X_mean @@ -350,9 +350,9 @@ cdef tuple _dseparate(double[:,:] X, int remove_mean): if m > n: m_aux = m else: - m_aux = n - cdef np.ndarray[np.double_t,ndim=2] Y = np.zeros((m_aux, n),dtype=np.double) - cdef np.ndarray[np.double_t,ndim=2] Z = np.zeros((m, n),dtype=np.double) + m_aux = n-1 + cdef np.ndarray[np.double_t,ndim=2] Y = np.zeros((m_aux, n-1),dtype=np.double) + cdef np.ndarray[np.double_t,ndim=2] Z = np.zeros((m, n-1),dtype=np.double) # Fill the matrices cdef double *X_mean From 0d6352b3295cfc00e656b95368176c839a4c6b50 Mon Sep 17 00:00:00 2001 From: msilvestrec03 Date: Tue, 22 Sep 2026 10:55:27 +0200 Subject: [PATCH 24/27] Modify options in options.cfg for platform and versions Updated platform, FFTW, and optimization settings. --- options.cfg | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/options.cfg b/options.cfg index 66baa273..16c55233 100644 --- a/options.cfg +++ b/options.cfg @@ -12,11 +12,11 @@ ## Options # -PLATFORM = MN5_GPP +PLATFORM = PC VECTORIZATION = ON OPENMP_PARALL = OFF USE_MKL = ON -USE_FFTW = ON +USE_FFTW = OFF USE_GCC = OFF USE_NVHPC = OFF DEBUGGING = OFF @@ -31,7 +31,7 @@ MODULES_COMPILED = MATH.MATHS,MATH.AVERAGING,MATH.QR,MATH.SVD,MATH.FFT,MATH.GEOM # OPTL = 3 HOST = Host -TUNE = sapphirerapids +TUNE = skylake ## Python versions @@ -41,9 +41,9 @@ PIP = pip3 ## Versions of the libraries # +ONEAPI_VERS = 2024.2.0.634 +OPENBLAS_VERS = 0.3.34 LAPACK_VERS = 3.9.0 -FFTW_VERS = 3.3.10 -NFFT_VERS = 3.5.2 KISSFFT_VERS = 131.1.0 -OPENBLAS_VERS = 0.3.17 -ONEAPI_VERS = 2023.2.0 +FFTW_VERS = 3.3.8 +NFFT_VERS = 3.5.2 From 8572d6e126baa50f3bd859c9311aaeea2679e90f Mon Sep 17 00:00:00 2001 From: msilvestrec03 Date: Wed, 23 Sep 2026 09:32:47 +0200 Subject: [PATCH 25/27] function definition --- pyLOM/vmmath/cfuncs.pxd | 4 ++-- pyLOM/vmmath/linear.pyx | 4 +++- pyLOM/vmmath/src/linear.c | 16 ++++++++++++---- pyLOM/vmmath/src/linear.h | 4 ++-- 4 files changed, 19 insertions(+), 9 deletions(-) diff --git a/pyLOM/vmmath/cfuncs.pxd b/pyLOM/vmmath/cfuncs.pxd index cd237d44..b1fb0cf1 100644 --- a/pyLOM/vmmath/cfuncs.pxd +++ b/pyLOM/vmmath/cfuncs.pxd @@ -178,10 +178,10 @@ cdef extern from "regression.h" nogil: cdef void c_dridge_regression "dridge_regression"(double *out, double *A, double *b, double lam, const int m, const int n) cdef extern from "linear.h" nogil: # Single precision - cdef int c_slinear_operator "slinear_operator"(float *U, float *S, float *VT, float *Atilde, float *Y, float *Z, const float r, const int m, const int n) + cdef int c_slinear_operator "slinear_operator"(float *U, float *S, float *VT, float *Atilde, float *Y, float *Z, const float r, const int my, const int mz, const int nn) cdef void c_sflip_columns "sflip_columns"(float *A, float *B, int m, int n) # Double precision - cdef int c_dlinear_operator "dlinear_operator"(float *U, float *S, float *VT, float *Atilde, float *Y, float *Z, const float r, const int m, const int n) + cdef int c_dlinear_operator "dlinear_operator"(double *U, double *S, double *VT, double *Atilde, double *Y, double *Z, const double r, const int my, const int mz, const int nn); cdef void c_dflip_columns "dflip_columns"(double *A, double *B, int m, int n) # Single complex precision cdef void c_cflip_columns "cflip_columns"(np.complex64_t *A, np.complex64_t *B, int m, int n) diff --git a/pyLOM/vmmath/linear.pyx b/pyLOM/vmmath/linear.pyx index ea3d6ec7..b4d568e8 100644 --- a/pyLOM/vmmath/linear.pyx +++ b/pyLOM/vmmath/linear.pyx @@ -565,6 +565,7 @@ def flip_columns(real_full[:,:] A): r''' ''' + cr_start('flip', 0) if real_full is np.complex128_t: return _zflip_columns(A) elif real_full is np.complex64_t: @@ -572,4 +573,5 @@ def flip_columns(real_full[:,:] A): elif real_full is double: return _dflip_columns(A) else: - return _sflip_columns(A) \ No newline at end of file + return _sflip_columns(A) + cr_stop('flip', 0) \ No newline at end of file diff --git a/pyLOM/vmmath/src/linear.c b/pyLOM/vmmath/src/linear.c index 47a1ffad..c87f37eb 100644 --- a/pyLOM/vmmath/src/linear.c +++ b/pyLOM/vmmath/src/linear.c @@ -132,7 +132,9 @@ int dlinear_operator(double *U, double *S, double *VT, double *Atilde, double *Y void sflip_columns(float *A, float *B, int m, int n) { int ii, jj; - // openmp?? + // #ifdef USE_OMP + // #pragma omp parallel for collapse(2) private(ii,jj) shared(A,B) firstprivate(m,n) + // #endif for (ii=0; ii Date: Wed, 30 Sep 2026 09:20:27 +0200 Subject: [PATCH 26/27] c functions --- pyLOM/RES/wrapper.pyx | 271 ++++++++++++++++++++++++++++++-------- pyLOM/vmmath/cfuncs.pxd | 8 +- pyLOM/vmmath/linear.py | 2 +- pyLOM/vmmath/src/linear.c | 131 ++++++++++++++++++ pyLOM/vmmath/src/linear.h | 7 +- 5 files changed, 361 insertions(+), 58 deletions(-) diff --git a/pyLOM/RES/wrapper.pyx b/pyLOM/RES/wrapper.pyx index e2d90e79..7133b5f8 100644 --- a/pyLOM/RES/wrapper.pyx +++ b/pyLOM/RES/wrapper.pyx @@ -25,8 +25,10 @@ from libc.stdlib cimport malloc, free from libc.string cimport memcpy, memset from libc.math cimport sqrt, log, atan2 from ..vmmath.cfuncs cimport real, real_complex, real_float, real_double -from ..vmmath.cfuncs cimport c_csvd, c_cdagger, c_cmatmul, c_cmatmulp, c_cvecmat, c_ccholesky, c_cinverse -from ..vmmath.cfuncs cimport c_zsvd, c_zdagger, c_zmatmul, c_zmatmulp, c_zvecmat, c_zcholesky, c_zinverse +from ..vmmath.cfuncs cimport c_svecmat, c_sseparate, c_stsqr_svd, c_sremove_rows, c_scompute_truncation_residual, c_scompute_truncation, c_stranspose, c_smatmul, c_smatmulp +from ..vmmath.cfuncs cimport c_dvecmat, c_dseparate, c_dtsqr_svd, c_dremove_rows, c_dcompute_truncation_residual, c_dcompute_truncation, c_dtranspose, c_dmatmul, c_dmatmulp +from ..vmmath.cfuncs cimport c_csvd, c_cdagger, c_cmatmul, c_cmatmulp, c_cvecmat, c_ccholesky, c_cinverse, c_cresolvent +from ..vmmath.cfuncs cimport c_zsvd, c_zdagger, c_zmatmul, c_zmatmulp, c_zvecmat, c_zcholesky, c_zinverse, c_zresolvent from ..vmmath.linear cimport _sconcatenate, _slinear_operator, _sseparate, _cresolvent from ..vmmath.linear cimport _dconcatenate, _dlinear_operator, _dseparate, _zresolvent @@ -300,86 +302,245 @@ def run(real_complex[:,:] Phi, real[:] delta, real[:] omega, real f, real[:] Q=N else: return _crun(Phi, delta, omega, f, Q) -def _crun_old(float[:,:] X, np.complex64_t w, float r, int remove_mean): +def _crun_old(float[:,:] X, np.complex64_t w, float r, int remove_mean, float[:] Q): # Variables - cdef int m, n - cdef float[:, ::1] Y - cdef float[:, ::1] Z - cdef float[:, ::1] U1 - cdef float[::1] S1 - cdef float[:, ::1] VT1 - cdef float[:, ::1] Atilde - + cdef int m, n, ii, m_aux + m = X.shape[0] + n = X.shape[1] + + if m > n: + m_aux = m + else: + m_aux = n-1 + + cdef float *Y + cdef float *Z + Y = malloc(m_aux*(n-1)*sizeof(float)) + Z = malloc(m_aux*(n-1)*sizeof(float)) + if Q is not None: + c_svecmat(&Q[0], &X[0,0], m, n) + # Create the snapshot of matrices and separate it into Y, Z - Y, Z = _sseparate(X, remove_mean) - m = Z.shape[0] - n = Z.shape[1] + 1 + c_sseparate(Y, Z, &X[0,0], m, n, remove_mean) + # Compute the linear operator in lower dimension - U1, S1, VT1, Atilde = _slinear_operator(Y, Z, r) - cdef int nr - nr = Atilde.shape[1] + cdef int icol, irow, retval + + ## Compute SVD + cdef float *U_aux + cdef float *S_all + cdef float *VT_all + U_aux = malloc(m_aux*(n-1)*sizeof(float)) + S_all = malloc((n-1)*sizeof(float)) + VT_all = malloc((n-1)*(n-1)*sizeof(float)) + + retval = c_stsqr_svd(U_aux, S_all, VT_all, Y, m_aux, (n-1)) + free(Y) + + cdef float *U_all + U_all = malloc(m_aux*(n-1)*sizeof(float)) + c_sremove_rows(U_aux, U_all, m, (n-1)) + free(U_aux) - del S1, VT1 + ## Truncate + cdef int nr + if r > 1: + nr = int(r) + else: + nr = c_scompute_truncation_residual(S_all, r, (n-1)) + + cdef float *Ur + cdef float *Sr + cdef float *VTr + Ur = malloc(m*nr*sizeof(float)) + Sr = malloc(nr*sizeof(float)) + VTr = malloc(nr*(n-1)*sizeof(float)) + + c_scompute_truncation(Ur, Sr, VTr, U_all, S_all, VT_all, m, (n-1), (n-1), nr) + free(U_all) + free(S_all) + free(VT_all) + + # Project Jacobian of the snapshots into the POD basis + cdef float *aux1 + cdef float *aux2 + cdef float *aux3 + cdef float *Atilde + cdef float *Urt + aux1 = malloc(nr*(n-1)*sizeof(float)) + aux2 = malloc(nr*(n-1)*sizeof(float)) + aux3 = malloc(nr*sizeof(float)) + Atilde = malloc(nr*nr*sizeof(float)) + Urt = malloc(nr*m*sizeof(float)) + c_stranspose(Ur, Urt, m, nr) + c_smatmulp(aux1, Urt, Z, nr, n-1, m) + free(Z) + + for icol in range(n-1): + for irow in range(nr): + aux2[icol*nr + irow] = VTr[irow*(n-1) + icol]/Sr[irow] + c_smatmul(Atilde, aux1, aux2, nr, nr, n-1) + free(aux1) + free(aux2) + free(aux3) + free(Urt) + free(VTr) + free(Sr) - cdef np.complex64_t[:, ::1] U2 cdef np.ndarray[np.float32_t,ndim=1] S = np.zeros((nr),dtype=np.float32) - cdef np.complex64_t[:, ::1] V2 + cdef np.complex64_t *U2 + cdef np.complex64_t *V2 + U2 = malloc(nr*nr*sizeof(np.complex64_t)) + V2 = malloc(nr*nr*sizeof(np.complex64_t)) - U2, S, V2 = _cresolvent(Atilde, w) - - del Atilde + c_cresolvent(U2, &S[0], V2, Atilde, w, nr) + free(Atilde) - cdef np.ndarray[np.complex64_t, ndim=2] U1_c = np.ascontiguousarray(U1, dtype=np.complex64) + cdef np.complex64_t *U_c + U_c = malloc(m*nr*sizeof(np.complex64_t)) + for ii in range(m*nr): + U_c[ii] = Ur[ii] + free(Ur) + cdef np.ndarray[np.complex64_t,ndim=2] U = np.zeros((m,nr),dtype=np.complex64) cdef np.ndarray[np.complex64_t,ndim=2] V = np.zeros((m,nr),dtype=np.complex64) - c_cmatmul(&U[0,0], &U1_c[0,0], &U2[0,0], m, nr, nr) - c_cmatmul(&V[0,0], &U1_c[0,0], &V2[0,0], m, nr, nr) - - del U1, U2, V2, U1_c + c_cmatmul(&U[0,0], U_c, U2, m, nr, nr) + c_cmatmul(&V[0,0], U_c, V2, m, nr, nr) + free(U2) + free(V2) + free(U_c) + + cdef np.complex64_t *Q_inv + Q_inv = malloc(m*sizeof(np.complex64_t)) + if Q is not None: + for ii in range(m): + Q_inv[ii] = 1.0 / Q[ii] + c_cvecmat(Q_inv, &U[0,0], m, nr) + c_cvecmat(Q_inv, &V[0,0], m, nr) + free(Q_inv) return U, S, V -def _zrun_old(double[:,:] X, np.complex128_t w, double r, int remove_mean): +def _zrun_old(double[:,:] X, np.complex128_t w, double r, int remove_mean, double[:] Q): # Variables - cdef int m, n - cdef double[:, ::1] Y - cdef double[:, ::1] Z - cdef double[:, ::1] U1 - cdef double[::1] S1 - cdef double[:, ::1] VT1 - cdef double[:, ::1] Atilde + cdef int m, n, ii, m_aux + m = X.shape[0] + n = X.shape[1] + + if m > n: + m_aux = m + else: + m_aux = n-1 + + cdef double *Y + cdef double *Z + Y = malloc(m_aux*(n-1)*sizeof(double)) + Z = malloc(m_aux*(n-1)*sizeof(double)) + if Q is not None: + c_dvecmat(&Q[0], &X[0,0], m, n) # Create the snapshot of matrices and separate it into Y, Z - Y, Z = _dseparate(X, remove_mean) - m = Z.shape[0] - n = Z.shape[1] + 1 + c_dseparate(Y, Z, &X[0,0], m, n, remove_mean) # Compute the linear operator in lower dimension - U1, S1, VT1, Atilde = _dlinear_operator(Y, Z, r) - cdef int nr - nr = Atilde.shape[1] + cdef int icol, irow, retval + + ## Compute SVD + cdef double *U_aux + cdef double *S_all + cdef double *VT_all + U_aux = malloc(m_aux*(n-1)*sizeof(double)) + S_all = malloc((n-1)*sizeof(double)) + VT_all = malloc((n-1)*(n-1)*sizeof(double)) + + retval = c_dtsqr_svd(U_aux, S_all, VT_all, Y, m_aux, (n-1)) + free(Y) + + cdef double *U_all + U_all = malloc(m_aux*(n-1)*sizeof(double)) + c_dremove_rows(U_aux, U_all, m, (n-1)) + free(U_aux) - del S1, VT1 + ## Truncate + cdef int nr + if r > 1: + nr = int(r) + else: + nr = c_dcompute_truncation_residual(S_all, r, (n-1)) + + cdef double *Ur + cdef double *Sr + cdef double *VTr + Ur = malloc(m*nr*sizeof(double)) + Sr = malloc(nr*sizeof(double)) + VTr = malloc(nr*(n-1)*sizeof(double)) + + c_dcompute_truncation(Ur, Sr, VTr, U_all, S_all, VT_all, m, (n-1), (n-1), nr) + free(U_all) + free(S_all) + free(VT_all) + + # Project Jacobian of the snapshots into the POD basis + cdef double *aux1 + cdef double *aux2 + cdef double *aux3 + cdef double *Atilde + cdef double *Urt + aux1 = malloc(nr*(n-1)*sizeof(double)) + aux2 = malloc(nr*(n-1)*sizeof(double)) + aux3 = malloc(nr*sizeof(double)) + Atilde = malloc(nr*nr*sizeof(double)) + Urt = malloc(nr*m*sizeof(double)) + c_dtranspose(Ur, Urt, m, nr) + c_dmatmulp(aux1, Urt, Z, nr, n-1, m) + free(Z) + + for icol in range(n-1): + for irow in range(nr): + aux2[icol*nr + irow] = VTr[irow*(n-1) + icol]/Sr[irow] + c_dmatmul(Atilde, aux1, aux2, nr, nr, n-1) + free(aux1) + free(aux2) + free(aux3) + free(Urt) + free(VTr) + free(Sr) - cdef np.complex128_t[:, ::1] U2 cdef np.ndarray[np.double_t,ndim=1] S = np.zeros((nr),dtype=np.double) - cdef np.complex128_t[:, ::1] V2 + cdef np.complex128_t *U2 + cdef np.complex128_t *V2 + U2 = malloc(nr*nr*sizeof(np.complex128_t)) + V2 = malloc(nr*nr*sizeof(np.complex128_t)) - U2, S, V2 = _zresolvent(Atilde, w) + c_zresolvent(U2, &S[0], V2, Atilde, w, nr) + free(Atilde) - del Atilde - - cdef np.ndarray[np.complex128_t, ndim=2] U1_c = np.ascontiguousarray(U1, dtype=np.complex128) + cdef np.complex128_t *U_c + U_c = malloc(m*nr*sizeof(np.complex128_t)) + + for ii in range(m*nr): + U_c[ii] = Ur[ii] + free(Ur) cdef np.ndarray[np.complex128_t,ndim=2] U = np.zeros((m,nr),dtype=np.complex128) cdef np.ndarray[np.complex128_t,ndim=2] V = np.zeros((m,nr),dtype=np.complex128) - c_zmatmul(&U[0,0], &U1_c[0,0], &U2[0,0], m, nr, nr) - c_zmatmul(&V[0,0], &U1_c[0,0], &V2[0,0], m, nr, nr) - - del U1, U2, V2, U1_c + c_zmatmul(&U[0,0], U_c, U2, m, nr, nr) + c_zmatmul(&V[0,0], U_c, V2, m, nr, nr) + free(U2) + free(V2) + free(U_c) + + cdef np.complex128_t *Q_inv + Q_inv = malloc(m*sizeof(np.complex128_t)) + if Q is not None: + for ii in range(m): + Q_inv[ii] = 1.0 / Q[ii] + c_zvecmat(Q_inv, &U[0,0], m, nr) + c_zvecmat(Q_inv, &V[0,0], m, nr) + free(Q_inv) return U, S, V @@ -470,7 +631,7 @@ def _zrun_new(list X, np.complex128_t w, double r, int remove_mean): @cython.wraparound(False) # turn off negative index wrapping for entire function @cython.nonecheck(False) @cython.cdivision(True) # turn off zero division check -def run_new(object X, object w, object r, int remove_mean=True): +def run_new(object X, object w, object r, int remove_mean=True, real[:] Q=None): if isinstance(X, list): if X[0].dtype == np.double: @@ -479,6 +640,6 @@ def run_new(object X, object w, object r, int remove_mean=True): return _crun_new(X,np.complex64(w),np.float32(r),remove_mean) else: if X[0].dtype == np.double: - return _zrun_old(X,np.complex128(w),np.double(r),remove_mean) + return _zrun_old(X,np.complex128(w),np.double(r),remove_mean, Q) else: return _crun_old(X,np.complex64(w),np.float32(r),remove_mean) \ No newline at end of file diff --git a/pyLOM/vmmath/cfuncs.pxd b/pyLOM/vmmath/cfuncs.pxd index b1fb0cf1..fd3faa98 100644 --- a/pyLOM/vmmath/cfuncs.pxd +++ b/pyLOM/vmmath/cfuncs.pxd @@ -164,11 +164,13 @@ cdef extern from "truncation.h" nogil: cdef void c_scompute_truncation "scompute_truncation"(float *Ur, float *Sr, float *VTr, float *U, float *S, float *VT, const int m, const int n, const int nmod, const int N) cdef float c_senergy "senergy"(float *A, float *B, const int m, const int n) cdef float c_slocal_energy "slocal_energy"(float *A, float *B, const int m, const int n) + cdef void c_sremove_rows "sremove_rows"(float *A, float *B, int rows, int n) # Double precision cdef int c_dcompute_truncation_residual "dcompute_truncation_residual"(double *S, double res, const int n) cdef void c_dcompute_truncation "dcompute_truncation"(double *Ur, double *Sr, double *VTr, double *U, double *S, double *VT, const int m, const int n, const int nmod, const int N) cdef double c_denergy "denergy"(double *A, double *B, const int m, const int n) cdef double c_dlocal_energy "dlocal_energy"(double *A, double *B, const int m, const int n) + cdef void c_dremove_rows "dremove_rows"(double *A, double *B, int rows, int n) cdef extern from "regression.h" nogil: # Float version cdef void c_sleast_squares "sleast_squares"(float *out, float *A, float *b, const int m, const int n) @@ -180,13 +182,17 @@ cdef extern from "linear.h" nogil: # Single precision cdef int c_slinear_operator "slinear_operator"(float *U, float *S, float *VT, float *Atilde, float *Y, float *Z, const float r, const int my, const int mz, const int nn) cdef void c_sflip_columns "sflip_columns"(float *A, float *B, int m, int n) + cdef void c_sseparate "sseparate"(float *Y, float *Z, float *X, const int m, const int n, int remove_mean) # Double precision - cdef int c_dlinear_operator "dlinear_operator"(double *U, double *S, double *VT, double *Atilde, double *Y, double *Z, const double r, const int my, const int mz, const int nn); + cdef int c_dlinear_operator "dlinear_operator"(double *U, double *S, double *VT, double *Atilde, double *Y, double *Z, const double r, const int my, const int mz, const int nn) cdef void c_dflip_columns "dflip_columns"(double *A, double *B, int m, int n) + cdef void c_dseparate "dseparate"(double *Y, double *Z, double *X, const int m, const int n, int remove_mean) # Single complex precision cdef void c_cflip_columns "cflip_columns"(np.complex64_t *A, np.complex64_t *B, int m, int n) + cdef void c_cresolvent "cresolvent"(np.complex64_t *U, float *S, np.complex64_t *V, float *A, np.complex64_t w, const int n) # Double complex precision cdef void c_zflip_columns "zflip_columns"(np.complex128_t *A, np.complex128_t *B, int m, int n) + cdef void c_zresolvent "zresolvent"(np.complex128_t *U, double *S, np.complex128_t *V, double *A, np.complex128_t w, const int n) ## Fused type between double and complex ctypedef fused real: diff --git a/pyLOM/vmmath/linear.py b/pyLOM/vmmath/linear.py index 93e01d9a..63bf209d 100644 --- a/pyLOM/vmmath/linear.py +++ b/pyLOM/vmmath/linear.py @@ -43,7 +43,7 @@ def concatenate(X=[], remove_mean=False): # Create the matrices Y, Z dtype = X[0].dtype m = X[0].shape[0] - n = sum([M.shape[1] - 1 for M in X]) + n = np.sum([M.shape[1] - 1 for M in X]) if m > n: Y = np.zeros((m, n), dtype=dtype) else: diff --git a/pyLOM/vmmath/src/linear.c b/pyLOM/vmmath/src/linear.c index c87f37eb..9f0118a4 100644 --- a/pyLOM/vmmath/src/linear.c +++ b/pyLOM/vmmath/src/linear.c @@ -2,6 +2,7 @@ Linear operator */ #include +#include #include #include #include @@ -179,4 +180,134 @@ void zflip_columns(dcomplex_t *A, dcomplex_t *B, int m, int n) { AC_MAT(B,n,ii,jj) = AC_MAT(A,n,ii,n-jj-1); } } +} + +void sseparate(float *Y, float *Z, float *X, const int m, const int n, int remove_mean) { + + int m_aux, ii; + + float *X_mean; + float *X_meanless; + X_mean = (float*)malloc(m*sizeof(float)); + X_meanless = (float*)malloc(m*n*sizeof(float)); + + if (remove_mean){ + stemporal_mean(X_mean, X, m, n); + ssubtract_mean(X_meanless, X, X_mean, m, n); + + for (ii=0; ii +#include typedef float _Complex scomplex_t; typedef double _Complex dcomplex_t; #ifdef USE_MKL @@ -12,10 +13,14 @@ typedef double _Complex dcomplex_t; // Single precision int slinear_operator(float *U, float *S, float *VT, float *Atilde, float *Y, float *Z, const float r, const int my, const int mz, const int nn); void sflip_columns(float *A, float *B, int m, int n); +void sseparate(float *Y, float *Z, float *X, const int m, const int n, int remove_mean); // Double precision int dlinear_operator(double *U, double *S, double *VT, double *Atilde, double *Y, double *Z, const double r, const int my, const int mz, const int nn); void dflip_columns(double *A, double *B, int m, int n); +void dseparate(double *Y, double *Z, double *X, const int m, const int n, int remove_mean); // Single complex precision void cflip_columns(scomplex_t *A, scomplex_t *B, int m, int n); +void cresolvent(scomplex_t *U, float *S, scomplex_t *V, float *A, scomplex_t w, const int n); // Double complex precision -void zflip_columns(dcomplex_t *A, dcomplex_t *B, int m, int n); \ No newline at end of file +void zflip_columns(dcomplex_t *A, dcomplex_t *B, int m, int n); +void zresolvent(dcomplex_t *U, double *S, dcomplex_t *V, double *A, dcomplex_t w, const int n); \ No newline at end of file From a1c9f1a3e61fd20713fc377d40e18f383aec937a Mon Sep 17 00:00:00 2001 From: msilvestrec03 Date: Wed, 30 Sep 2026 16:20:53 +0200 Subject: [PATCH 27/27] resolvent for list of datasets at c --- pyLOM/RES/wrapper.pyx | 308 ++++++++++++++++++++++++++++++-------- pyLOM/vmmath/cfuncs.pxd | 2 + pyLOM/vmmath/src/linear.c | 78 +++++++++- pyLOM/vmmath/src/linear.h | 2 + 4 files changed, 323 insertions(+), 67 deletions(-) diff --git a/pyLOM/RES/wrapper.pyx b/pyLOM/RES/wrapper.pyx index 7133b5f8..3b55d343 100644 --- a/pyLOM/RES/wrapper.pyx +++ b/pyLOM/RES/wrapper.pyx @@ -21,12 +21,12 @@ cdef extern from "" nogil: double cimag(double complex z) double creal(double complex z) cdef double complex J = 1j -from libc.stdlib cimport malloc, free +from libc.stdlib cimport malloc, calloc, free from libc.string cimport memcpy, memset from libc.math cimport sqrt, log, atan2 from ..vmmath.cfuncs cimport real, real_complex, real_float, real_double -from ..vmmath.cfuncs cimport c_svecmat, c_sseparate, c_stsqr_svd, c_sremove_rows, c_scompute_truncation_residual, c_scompute_truncation, c_stranspose, c_smatmul, c_smatmulp -from ..vmmath.cfuncs cimport c_dvecmat, c_dseparate, c_dtsqr_svd, c_dremove_rows, c_dcompute_truncation_residual, c_dcompute_truncation, c_dtranspose, c_dmatmul, c_dmatmulp +from ..vmmath.cfuncs cimport c_svecmat, c_sconcatenate, c_sseparate, c_stsqr_svd, c_sremove_rows, c_scompute_truncation_residual, c_scompute_truncation, c_stranspose, c_smatmul, c_smatmulp +from ..vmmath.cfuncs cimport c_dvecmat, c_dconcatenate, c_dseparate, c_dtsqr_svd, c_dremove_rows, c_dcompute_truncation_residual, c_dcompute_truncation, c_dtranspose, c_dmatmul, c_dmatmulp from ..vmmath.cfuncs cimport c_csvd, c_cdagger, c_cmatmul, c_cmatmulp, c_cvecmat, c_ccholesky, c_cinverse, c_cresolvent from ..vmmath.cfuncs cimport c_zsvd, c_zdagger, c_zmatmul, c_zmatmulp, c_zvecmat, c_zcholesky, c_zinverse, c_zresolvent from ..vmmath.linear cimport _sconcatenate, _slinear_operator, _sseparate, _cresolvent @@ -309,15 +309,15 @@ def _crun_old(float[:,:] X, np.complex64_t w, float r, int remove_mean, float[:] m = X.shape[0] n = X.shape[1] - if m > n: + if m > (n - 1): m_aux = m else: m_aux = n-1 cdef float *Y cdef float *Z - Y = malloc(m_aux*(n-1)*sizeof(float)) - Z = malloc(m_aux*(n-1)*sizeof(float)) + Y = calloc(m_aux*(n-1), sizeof(float)) + Z = malloc(m*(n-1)*sizeof(float)) if Q is not None: c_svecmat(&Q[0], &X[0,0], m, n) @@ -430,15 +430,15 @@ def _zrun_old(double[:,:] X, np.complex128_t w, double r, int remove_mean, doubl m = X.shape[0] n = X.shape[1] - if m > n: + if m > (n - 1): m_aux = m else: m_aux = n-1 cdef double *Y cdef double *Z - Y = malloc(m_aux*(n-1)*sizeof(double)) - Z = malloc(m_aux*(n-1)*sizeof(double)) + Y = calloc(m_aux*(n-1), sizeof(double)) + Z = malloc(m*(n-1)*sizeof(double)) if Q is not None: c_dvecmat(&Q[0], &X[0,0], m, n) @@ -544,85 +544,273 @@ def _zrun_old(double[:,:] X, np.complex128_t w, double r, int remove_mean, doubl return U, S, V -def _crun_new(list X, np.complex64_t w, float r, int remove_mean): +def _crun_new(list X, np.complex64_t w, float r, int remove_mean, float[:] Q): # Variables - cdef int m, n - cdef float[:, ::1] Y - cdef float[:, ::1] Z - cdef float[:, ::1] U1 - cdef float[::1] S1 - cdef float[:, ::1] VT1 - cdef float[:, ::1] Atilde - + cdef int m, ii, m_aux, n_total, n_matrix, n_local + m = X[0].shape[0] + n_matrix = len(X) + + cdef float **X_list + cdef int *n_list + X_pointer = malloc(n_matrix*sizeof(float*)) + n_list = malloc(n_matrix*sizeof(int)) + ### LOOP X + n_total = 0 + cdef float[:, ::1] M + for ii in range(len(X)): + M = X[ii] + n_local = M.shape[1] + n_total += n_local - 1 + if Q is not None: + c_svecmat(&Q[0], &M[0,0], m, n_local) + n_list[ii] = n_local + X_pointer[ii] = &M[0,0] + + if m > n_total: + m_aux = m + else: + m_aux = n_total + + cdef float *Y + cdef float *Z + Y = calloc(m_aux*n_total, sizeof(float)) + Z = malloc(m*n_total*sizeof(float)) + # Create the snapshot of matrices and separate it into Y, Z - Y, Z = _sconcatenate(X, remove_mean) - m = Z.shape[0] - n = Z.shape[1] + 1 + c_sconcatenate(Y, Z, X_pointer, n_list, n_matrix, m, n_total, remove_mean) + ### MODIFICAR A SOBRE D'AIXO!!! # Compute the linear operator in lower dimension - U1, S1, VT1, Atilde = _slinear_operator(Y, Z, r) + cdef int icol, irow, retval + + ## Compute SVD + cdef float *U_aux + cdef float *S_all + cdef float *VT_all + U_aux = malloc(m_aux*n_total*sizeof(float)) + S_all = malloc(n_total*sizeof(float)) + VT_all = malloc(n_total*n_total*sizeof(float)) + + retval = c_stsqr_svd(U_aux, S_all, VT_all, Y, m_aux, n_total) + free(Y) + + cdef float *U_all + U_all = malloc(m_aux*n_total*sizeof(float)) + c_sremove_rows(U_aux, U_all, m, n_total) + free(U_aux) + + ## Truncate cdef int nr - nr = Atilde.shape[1] + if r > 1: + nr = int(r) + else: + nr = c_scompute_truncation_residual(S_all, r, n_total) - del S1, VT1 + cdef float *Ur + cdef float *Sr + cdef float *VTr + Ur = malloc(m*nr*sizeof(float)) + Sr = malloc(nr*sizeof(float)) + VTr = malloc(nr*n_total*sizeof(float)) - cdef np.complex64_t[:, ::1] U2 - cdef np.ndarray[np.float32_t,ndim=1] S = np.zeros((nr),dtype=np.float32) - cdef np.complex64_t[:, ::1] V2 + c_scompute_truncation(Ur, Sr, VTr, U_all, S_all, VT_all, m, n_total, n_total, nr) + free(U_all) + free(S_all) + free(VT_all) + + # Project Jacobian of the snapshots into the POD basis + cdef float *aux1 + cdef float *aux2 + cdef float *aux3 + cdef float *Atilde + cdef float *Urt + aux1 = malloc(nr*n_total*sizeof(float)) + aux2 = malloc(nr*n_total*sizeof(float)) + aux3 = malloc(nr*sizeof(float)) + Atilde = malloc(nr*nr*sizeof(float)) + Urt = malloc(nr*m*sizeof(float)) + c_stranspose(Ur, Urt, m, nr) + c_smatmulp(aux1, Urt, Z, nr, n_total, m) + free(Z) - U2, S, V2 = _cresolvent(Atilde, w) + for icol in range(n_total): + for irow in range(nr): + aux2[icol*nr + irow] = VTr[irow*n_total + icol]/Sr[irow] + c_smatmul(Atilde, aux1, aux2, nr, nr, n_total) + free(aux1) + free(aux2) + free(aux3) + free(Urt) + free(VTr) + free(Sr) - del Atilde + cdef np.ndarray[np.float32_t,ndim=1] S = np.zeros((nr),dtype=np.float32) + cdef np.complex64_t *U2 + cdef np.complex64_t *V2 + U2 = malloc(nr*nr*sizeof(np.complex64_t)) + V2 = malloc(nr*nr*sizeof(np.complex64_t)) - cdef np.ndarray[np.complex64_t, ndim=2] U1_c = np.ascontiguousarray(U1, dtype=np.complex64) + c_cresolvent(U2, &S[0], V2, Atilde, w, nr) + free(Atilde) + cdef np.complex64_t *U_c + U_c = malloc(m*nr*sizeof(np.complex64_t)) + + for ii in range(m*nr): + U_c[ii] = Ur[ii] + free(Ur) + cdef np.ndarray[np.complex64_t,ndim=2] U = np.zeros((m,nr),dtype=np.complex64) cdef np.ndarray[np.complex64_t,ndim=2] V = np.zeros((m,nr),dtype=np.complex64) - c_cmatmul(&U[0,0], &U1_c[0,0], &U2[0,0], m, nr, nr) - c_cmatmul(&V[0,0], &U1_c[0,0], &V2[0,0], m, nr, nr) + c_cmatmul(&U[0,0], U_c, U2, m, nr, nr) + c_cmatmul(&V[0,0], U_c, V2, m, nr, nr) + free(U2) + free(V2) + free(U_c) - del U1, U2, V2, U1_c + cdef np.complex64_t *Q_inv + Q_inv = malloc(m*sizeof(np.complex64_t)) + if Q is not None: + for ii in range(m): + Q_inv[ii] = 1.0 / Q[ii] + c_cvecmat(Q_inv, &U[0,0], m, nr) + c_cvecmat(Q_inv, &V[0,0], m, nr) + free(Q_inv) return U, S, V -def _zrun_new(list X, np.complex128_t w, double r, int remove_mean): +def _zrun_new(list X, np.complex128_t w, double r, int remove_mean, double[:] Q): # Variables - cdef int m, n - cdef double[:, ::1] Y - cdef double[:, ::1] Z - cdef double[:, ::1] U1 - cdef double[::1] S1 - cdef double[:, ::1] VT1 - cdef double[:, ::1] Atilde - + cdef int m, ii, m_aux, n_total, n_matrix, n_local + m = X[0].shape[0] + n_matrix = len(X) + + cdef double **X_list + cdef int *n_list + X_pointer = malloc(n_matrix*sizeof(double*)) + n_list = malloc(n_matrix*sizeof(int)) + ### LOOP X + n_total = 0 + cdef double[:, ::1] M + for ii in range(len(X)): + M = X[ii] + n_local = M.shape[1] + n_total += n_local - 1 + if Q is not None: + c_dvecmat(&Q[0], &M[0,0], m, n_local) + n_list[ii] = n_local + X_pointer[ii] = &M[0,0] + + if m > n_total: + m_aux = m + else: + m_aux = n_total + + cdef double *Y + cdef double *Z + Y = calloc(m_aux*n_total, sizeof(double)) + Z = malloc(m*n_total*sizeof(double)) + # Create the snapshot of matrices and separate it into Y, Z - Y, Z = _dconcatenate(X, remove_mean) - m = Z.shape[0] - n = Z.shape[1] + 1 + c_dconcatenate(Y, Z, X_pointer, n_list, n_matrix, m, n_total, remove_mean) + ### MODIFICAR A SOBRE D'AIXO!!! # Compute the linear operator in lower dimension - U1, S1, VT1, Atilde = _dlinear_operator(Y, Z, r) + cdef int icol, irow, retval + + ## Compute SVD + cdef double *U_aux + cdef double *S_all + cdef double *VT_all + U_aux = malloc(m_aux*n_total*sizeof(double)) + S_all = malloc(n_total*sizeof(double)) + VT_all = malloc(n_total*n_total*sizeof(double)) + + retval = c_dtsqr_svd(U_aux, S_all, VT_all, Y, m_aux, n_total) + free(Y) + + cdef double *U_all + U_all = malloc(m_aux*n_total*sizeof(double)) + c_dremove_rows(U_aux, U_all, m, n_total) + free(U_aux) + + ## Truncate cdef int nr - nr = Atilde.shape[1] + if r > 1: + nr = int(r) + else: + nr = c_dcompute_truncation_residual(S_all, r, n_total) - del S1, VT1 + cdef double *Ur + cdef double *Sr + cdef double *VTr + Ur = malloc(m*nr*sizeof(double)) + Sr = malloc(nr*sizeof(double)) + VTr = malloc(nr*n_total*sizeof(double)) - cdef np.complex128_t[:, ::1] U2 - cdef np.ndarray[np.double_t,ndim=1] S = np.zeros((nr),dtype=np.double) - cdef np.complex128_t[:, ::1] V2 + c_dcompute_truncation(Ur, Sr, VTr, U_all, S_all, VT_all, m, n_total, n_total, nr) + free(U_all) + free(S_all) + free(VT_all) + + # Project Jacobian of the snapshots into the POD basis + cdef double *aux1 + cdef double *aux2 + cdef double *aux3 + cdef double *Atilde + cdef double *Urt + aux1 = malloc(nr*n_total*sizeof(double)) + aux2 = malloc(nr*n_total*sizeof(double)) + aux3 = malloc(nr*sizeof(double)) + Atilde = malloc(nr*nr*sizeof(double)) + Urt = malloc(nr*m*sizeof(double)) + c_dtranspose(Ur, Urt, m, nr) + c_dmatmulp(aux1, Urt, Z, nr, n_total, m) + free(Z) - U2, S, V2 = _zresolvent(Atilde, w) + for icol in range(n_total): + for irow in range(nr): + aux2[icol*nr + irow] = VTr[irow*n_total + icol]/Sr[irow] + c_dmatmul(Atilde, aux1, aux2, nr, nr, n_total) + free(aux1) + free(aux2) + free(aux3) + free(Urt) + free(VTr) + free(Sr) - del Atilde + cdef np.ndarray[np.double_t,ndim=1] S = np.zeros((nr),dtype=np.double) + cdef np.complex128_t *U2 + cdef np.complex128_t *V2 + U2 = malloc(nr*nr*sizeof(np.complex128_t)) + V2 = malloc(nr*nr*sizeof(np.complex128_t)) - cdef np.ndarray[np.complex128_t, ndim=2] U1_c = np.ascontiguousarray(U1, dtype=np.complex128) + c_zresolvent(U2, &S[0], V2, Atilde, w, nr) + free(Atilde) + cdef np.complex128_t *U_c + U_c = malloc(m*nr*sizeof(np.complex128_t)) + + for ii in range(m*nr): + U_c[ii] = Ur[ii] + free(Ur) + cdef np.ndarray[np.complex128_t,ndim=2] U = np.zeros((m,nr),dtype=np.complex128) cdef np.ndarray[np.complex128_t,ndim=2] V = np.zeros((m,nr),dtype=np.complex128) - c_zmatmul(&U[0,0], &U1_c[0,0], &U2[0,0], m, nr, nr) - c_zmatmul(&V[0,0], &U1_c[0,0], &V2[0,0], m, nr, nr) + c_zmatmul(&U[0,0], U_c, U2, m, nr, nr) + c_zmatmul(&V[0,0], U_c, V2, m, nr, nr) + free(U2) + free(V2) + free(U_c) - del U1, U2, V2, U1_c + cdef np.complex128_t *Q_inv + Q_inv = malloc(m*sizeof(np.complex128_t)) + if Q is not None: + for ii in range(m): + Q_inv[ii] = 1.0 / Q[ii] + c_zvecmat(Q_inv, &U[0,0], m, nr) + c_zvecmat(Q_inv, &V[0,0], m, nr) + free(Q_inv) return U, S, V @@ -635,11 +823,11 @@ def run_new(object X, object w, object r, int remove_mean=True, real[:] Q=None): if isinstance(X, list): if X[0].dtype == np.double: - return _zrun_new(X,np.complex128(w),np.double(r),remove_mean) + return _zrun_new(X,np.complex128(w),np.double(r),remove_mean, Q) else: - return _crun_new(X,np.complex64(w),np.float32(r),remove_mean) + return _crun_new(X,np.complex64(w),np.float32(r),remove_mean, Q) else: if X[0].dtype == np.double: return _zrun_old(X,np.complex128(w),np.double(r),remove_mean, Q) else: - return _crun_old(X,np.complex64(w),np.float32(r),remove_mean) \ No newline at end of file + return _crun_old(X,np.complex64(w),np.float32(r),remove_mean, Q) \ No newline at end of file diff --git a/pyLOM/vmmath/cfuncs.pxd b/pyLOM/vmmath/cfuncs.pxd index fd3faa98..2d79b597 100644 --- a/pyLOM/vmmath/cfuncs.pxd +++ b/pyLOM/vmmath/cfuncs.pxd @@ -182,10 +182,12 @@ cdef extern from "linear.h" nogil: # Single precision cdef int c_slinear_operator "slinear_operator"(float *U, float *S, float *VT, float *Atilde, float *Y, float *Z, const float r, const int my, const int mz, const int nn) cdef void c_sflip_columns "sflip_columns"(float *A, float *B, int m, int n) + cdef void c_sconcatenate "sconcatenate"(float *Y, float *Z, float **X, int *n_list, const int n_matrix, const int m, const int n_total, int remove_mean) cdef void c_sseparate "sseparate"(float *Y, float *Z, float *X, const int m, const int n, int remove_mean) # Double precision cdef int c_dlinear_operator "dlinear_operator"(double *U, double *S, double *VT, double *Atilde, double *Y, double *Z, const double r, const int my, const int mz, const int nn) cdef void c_dflip_columns "dflip_columns"(double *A, double *B, int m, int n) + cdef void c_dconcatenate "dconcatenate"(double *Y, double *Z, double **X, int *n_list, const int n_matrix, const int m, const int n_total, int remove_mean) cdef void c_dseparate "dseparate"(double *Y, double *Z, double *X, const int m, const int n, int remove_mean) # Single complex precision cdef void c_cflip_columns "cflip_columns"(np.complex64_t *A, np.complex64_t *B, int m, int n) diff --git a/pyLOM/vmmath/src/linear.c b/pyLOM/vmmath/src/linear.c index 9f0118a4..91fdf296 100644 --- a/pyLOM/vmmath/src/linear.c +++ b/pyLOM/vmmath/src/linear.c @@ -182,9 +182,73 @@ void zflip_columns(dcomplex_t *A, dcomplex_t *B, int m, int n) { } } +void sconcatenate(float *Y, float *Z, float **X, int *n_list, const int n_matrix, const int m, const int n_total, int remove_mean) { + + int ii, jj, n_jj; + + float *X_mean; + float *X_meanless; + int n_sum = 0; + X_mean = (float*)malloc(m*sizeof(float)); + + for (jj=0; jj