benchmarking: moved compilation of kernel to evaluate function, as it required too much memory

2025-05-19 11:58:24 +02:00
parent f33551e25f
commit e29199d865
2 changed files with 7 additions and 6 deletions
--- a/package/src/Transpiler.jl
+++ b/package/src/Transpiler.jl
@ -35,7 +35,7 @@ end
 "
 A simplified version of the evaluate function. It takes a list of already compiled kernels to be executed. This should yield better performance, where the same expressions should be evaluated multiple times i.e. for parameter optimisation.
 "
-function evaluate(kernels::Vector{CuFunction}, cudaVars::CuArray{Float32}, nrOfVariableSets::Integer, parameters::Vector{Vector{Float32}})::Matrix{Float32}
+function evaluate(kernels::Vector{String}, cudaVars::CuArray{Float32}, nrOfVariableSets::Integer, parameters::Vector{Vector{Float32}}, kernelName::String)::Matrix{Float32}

 	cudaParams = Utils.create_cuda_array(parameters, NaN32) # maybe make constant (see PerformanceTests.jl for more info)

@ -46,7 +46,8 @@ function evaluate(kernels::Vector{CuFunction}, cudaVars::CuArray{Float32}, nrOfV
 	blocks = cld(nrOfVariableSets, threads)
 	
 	@inbounds Threads.@threads for i in eachindex(kernels)
-		cudacall(kernels[i], (CuPtr{Float32},CuPtr{Float32},CuPtr{Float32}), cudaVars, cudaParams, cudaResults; threads=threads, blocks=blocks)
+		compiledKernel = CompileKernel(kernel[i], kernelName)
+		cudacall(compiledKernel, (CuPtr{Float32},CuPtr{Float32},CuPtr{Float32}), cudaVars, cudaParams, cudaResults; threads=threads, blocks=blocks)
 	end

 	return cudaResults