diff --git a/Project.toml b/Project.toml
index 94a4fe2..c34e178 100644
--- a/Project.toml
+++ b/Project.toml
@@ -37,6 +37,7 @@ NativeFileDialog = "e1fe445b-aa65-4df4-81c1-2041507f0fd4"
NaturalSort = "c020b1a1-e9b0-503a-9c33-f039bfc54a85"
Parameters = "d96e819e-fc66-5662-9728-84c9c7592b0a"
PlotlyBase = "a03496cd-edff-5a9b-9e67-9cda94a718b5"
+Polyester = "f401cd3a-86c5-430c-a3c3-63b7194639e4"
Printf = "de0858da-6303-5e67-8744-51eddeeeb8d7"
ProgressMeter = "92933f4c-e287-5a05-a399-4b506db050ca"
SavitzkyGolay = "c4bf5708-b6a6-4fbe-bcd0-6850ed671584"
@@ -54,6 +55,8 @@ CSV = "0.10"
CairoMakie = "0.13"
ColorSchemes = "3.30"
Colors = "0.12"
+CodecBase = "0.3"
+ContinuousWavelets = "1.1"
DataFrames = "1.7"
Dates = "1.11"
FileIO = "1.17"
@@ -61,14 +64,16 @@ GLMakie = "0.11"
Genie = "5.31"
GenieFramework = "2.8"
HistogramThresholding = "0.3"
+Interpolations = "0.15"
ImageBinarization = "0.3"
ImageComponentAnalysis = "0.2"
ImageContrastAdjustment = "0.3"
-ImageCore = "0.10.5"
+ImageCore = "0.10"
ImageFiltering = "0.7"
ImageMorphology = "0.4"
ImageSegmentation = "1.9"
Images = "0.26"
+JLD2 = "0.6"
JSON = "0.21"
Libz = "1.0"
LinearAlgebra = "1.11"
@@ -76,13 +81,15 @@ Loess = "0.6"
Mmap = "1.11"
NativeFileDialog = "0.2"
NaturalSort = "1.0"
+Parameters = "0.12"
PlotlyBase = "0.8"
Printf = "1.11"
ProgressMeter = "1.11"
SavitzkyGolay = "0.9"
Serialization = "1.11"
+Setfield = "1.1"
Statistics = "1.11"
StatsBase = "0.34"
StipplePlotly = "0.13"
UUIDs = "1.11"
-julia = "1.11"
+julia = "1.11, 1.12"
diff --git a/app.jl b/app.jl
index 224aeb4..a99a9b4 100644
--- a/app.jl
+++ b/app.jl
@@ -352,6 +352,7 @@ end
# Preprocessing results
@in selected_spectrum_id_for_plot = 1
@in last_plot_type = "single"
+ @in last_plot_mode = "lines"
@in feature_matrix_result::Union{Nothing, Matrix{Float64}} = nothing
@in bin_info_result::Union{Nothing, Vector} = nothing
@@ -625,6 +626,17 @@ end
return
end
+ # --- Close previous dataset if one is open ---
+ if msi_data !== nothing
+ println("DEBUG: Closing previously loaded dataset before opening new one: $(basename(full_route))")
+ close(msi_data)
+ msi_data = nothing
+ GC.gc()
+ if Sys.islinux()
+ ccall(:malloc_trim, Int32, (Int32,), 0)
+ end
+ end
+
msg = "Opening file: $(basename(picked_route))..."
try
@@ -679,7 +691,7 @@ end
loaded_data = OpenMSIData(local_full_route)
is_imzML = loaded_data.source isa ImzMLSource
- if isempty(get(existing_entry, "metadata", Dict()))
+ if existing_entry == nothing
msg = "Performing first-time metadata analysis for: $(basename(picked_route))..."
precompute_analytics(loaded_data)
end
@@ -889,6 +901,18 @@ end
selected_folder_main = dataset_name
msi_data = loaded_data
+
+ # Determine plot mode from loaded data
+ df = msi_data.spectrum_stats_df
+ if df !== nothing && "Mode" in names(df)
+ profile_count = count(==(MSI_src.PROFILE), df.Mode)
+ total_count = length(df.Mode)
+ last_plot_mode = profile_count > total_count / 2 ? "lines" : "stem"
+ println("DEBUG: Auto-detected plot mode: $(last_plot_mode)")
+ else
+ last_plot_mode = "lines" # Default
+ end
+
log_memory_usage("Full Load", msi_data)
eTime = round(time() - sTime, digits=3)
@@ -1093,7 +1117,7 @@ end
try
# --- 1. Initial Checks and Data Loading ---
println("DEBUG: Performing initial checks and data loading...")
- if msi_data === nothing || isempty(selected_folder_main)
+ if isempty(selected_folder_main)
msg = "No dataset loaded. Please load a file using 'Select an imzMl / mzML file'."
warning_msg = true
println("DEBUG: $msg")
@@ -1111,13 +1135,24 @@ end
end
target_path = entry["source_path"]
- # Ensure msi_data is for the currently selected file
- if full_route != target_path
- println("DEBUG: Active file path changed. Reloading MSI data: $(basename(target_path))")
+ # Ensure msi_data is for the currently selected file and load if needed
+ if msi_data === nothing || full_route != target_path
+ println("DEBUG: Active file path changed or data not in memory. Reloading MSI data: $(basename(target_path))")
if msi_data !== nothing; close(msi_data); end
msg = "Reloading $(basename(target_path)) for analysis..."
full_route = target_path
msi_data = OpenMSIData(target_path)
+
+ # Determine plot mode from loaded data
+ df = msi_data.spectrum_stats_df
+ if df !== nothing && "Mode" in names(df)
+ profile_count = count(==(MSI_src.PROFILE), df.Mode)
+ total_count = length(df.Mode)
+ last_plot_mode = profile_count > total_count / 2 ? "lines" : "stem"
+ println("DEBUG: Auto-detected plot mode for pipeline: $(last_plot_mode)")
+ else
+ last_plot_mode = "lines" # Default
+ end
else
println("DEBUG: Using already loaded MSI data for $(basename(target_path)).")
end
@@ -1422,7 +1457,7 @@ end
# 4. Update Results Display
current_pipeline_step = "Updating results..."
subset_label = enable_subset_processing ? " (from subset of $(length(current_spectra)) spectra)" : ""
- println("DEBUG: Updating results display after pipeline completion for plot type: $(last_plot_type)")
+ println("DEBUG: Updating results display after pipeline completion for plot type: $(last_plot_type), mode: $(last_plot_mode)")
if last_plot_type == "single"
display_spectrum_idx = findfirst(s -> s.id == selected_spectrum_id_for_plot, current_spectra)
@@ -1430,12 +1465,35 @@ end
processed_spectrum = current_spectra[display_spectrum_idx]
println("DEBUG: Displaying spectrum $(selected_spectrum_id_for_plot) after processing.")
- after_trace = PlotlyBase.scatter(
- x=processed_spectrum.mz,
- y=processed_spectrum.intensity,
- mode="lines",
- name="Processed Spectrum"
- )
+ # Determine plot mode for this specific spectrum
+ spectrum_mode_for_plot = "lines" # Default to lines
+ if msi_data.spectrum_stats_df !== nothing && "Mode" in names(msi_data.spectrum_stats_df)
+ if selected_spectrum_id_for_plot > 0 && selected_spectrum_id_for_plot <= length(msi_data.spectrum_stats_df.Mode)
+ mode = msi_data.spectrum_stats_df.Mode[selected_spectrum_id_for_plot]
+ if mode == MSI_src.CENTROID
+ spectrum_mode_for_plot = "stem"
+ end
+ end
+ end
+
+ mz_down, int_down = downsample_spectrum(processed_spectrum.mz, processed_spectrum.intensity)
+
+ local after_trace
+ if spectrum_mode_for_plot == "stem"
+ after_trace = PlotlyBase.stem(
+ x=mz_down,
+ y=int_down,
+ name="Processed Spectrum",
+ marker=attr(size=1, color="blue", opacity=0)
+ )
+ else # lines
+ after_trace = PlotlyBase.scatter(
+ x=mz_down,
+ y=int_down,
+ mode="lines",
+ name="Processed Spectrum"
+ )
+ end
traces_after = [after_trace]
@@ -1454,9 +1512,25 @@ end
plotdata_after = traces_after
plotlayout_after = PlotlyBase.Layout(
- title="After Preprocessing (Spectrum $(selected_spectrum_id_for_plot))$(subset_label)",
- xaxis_title="m/z",
- yaxis_title="Intensity",
+ title=PlotlyBase.attr(
+ text="After Preprocessing (Spectrum $(selected_spectrum_id_for_plot))$(subset_label)",
+ font=PlotlyBase.attr(
+ family="Roboto, Lato, sans-serif",
+ size=18,
+ color="black"
+ )
+ ),
+ hovermode="closest",
+ xaxis=PlotlyBase.attr(
+ title="m/z",
+ showgrid=true
+ ),
+ yaxis=PlotlyBase.attr(
+ title="Intensity",
+ showgrid=true,
+ tickformat=".3g"
+ ),
+ margin=attr(l=0, r=0, t=120, b=0, pad=0),
legend=attr(x=0.98, y=0.98, xanchor="right", yanchor="top")
)
else
@@ -1464,21 +1538,97 @@ end
end
elseif last_plot_type == "mean"
mz, intensity = get_processed_mean_spectrum(current_spectra)
- plotdata_after = [PlotlyBase.scatter(x=mz, y=intensity, mode="lines", name="Processed Mean Spectrum")]
+ mz_down, int_down = downsample_spectrum(mz, intensity)
+ local trace
+ if last_plot_mode == "stem"
+ trace = PlotlyBase.stem(
+ x=mz_down,
+ y=int_down,
+ name="Processed Mean Spectrum",
+ marker=attr(size=1, color="blue", opacity=0.5),
+ hoverinfo="x",
+ hovertemplate="m/z: %{x:.4f}"
+ )
+ else
+ trace = PlotlyBase.scatter(
+ x=mz_down,
+ y=int_down,
+ mode="lines",
+ name="Processed Mean Spectrum",
+ marker=attr(size=1, color="blue", opacity=0.5),
+ hoverinfo="x",
+ hovertemplate="m/z: %{x:.4f}"
+ )
+ end
+ plotdata_after = [trace]
plotlayout_after = PlotlyBase.Layout(
- title="After Preprocessing (Mean Spectrum)$(subset_label)",
- xaxis_title="m/z",
- yaxis_title="Average Intensity",
- legend=attr(x=0.98, y=0.98, xanchor="right", yanchor="top")
+ title=PlotlyBase.attr(
+ text="After Preprocessing (Mean Spectrum)$(subset_label)",
+ font=PlotlyBase.attr(
+ family="Roboto, Lato, sans-serif",
+ size=18,
+ color="black"
+ )
+ ),
+ hovermode="closest",
+ xaxis=PlotlyBase.attr(
+ title="m/z",
+ showgrid=true
+ ),
+ yaxis=PlotlyBase.attr(
+ title="Average Intensity",
+ showgrid=true,
+ tickformat=".3g"
+ ),
+ margin=attr(l=0, r=0, t=120, b=0, pad=0),
+ legend=attr(x=1.0, y=1.0, xanchor="right", yanchor="top")
)
elseif last_plot_type == "sum"
mz, intensity = get_processed_sum_spectrum(current_spectra)
- plotdata_after = [PlotlyBase.scatter(x=mz, y=intensity, mode="lines", name="Processed Sum Spectrum")]
+ mz_down, int_down = downsample_spectrum(mz, intensity)
+ local trace
+ if last_plot_mode == "stem"
+ trace = PlotlyBase.stem(
+ x=mz_down,
+ y=int_down,
+ name="Processed Sum Spectrum",
+ marker=attr(size=1, color="blue", opacity=0.5),
+ hoverinfo="x",
+ hovertemplate="m/z: %{x:.4f}"
+ )
+ else
+ trace = PlotlyBase.scatter(
+ x=mz_down,
+ y=int_down,
+ mode="lines",
+ name="Processed Sum Spectrum",
+ marker=attr(size=1, color="blue", opacity=0.5),
+ hoverinfo="x",
+ hovertemplate="m/z: %{x:.4f}"
+ )
+ end
+ plotdata_after = [trace]
plotlayout_after = PlotlyBase.Layout(
- title="After Preprocessing (Sum Spectrum)$(subset_label)",
- xaxis_title="m/z",
- yaxis_title="Total Intensity",
- legend=attr(x=0.98, y=0.98, xanchor="right", yanchor="top")
+ title=PlotlyBase.attr(
+ text="After Preprocessing (Sum Spectrum)$(subset_label)",
+ font=PlotlyBase.attr(
+ family="Roboto, Lato, sans-serif",
+ size=18,
+ color="black"
+ )
+ ),
+ hovermode="closest",
+ xaxis=PlotlyBase.attr(
+ title="m/z",
+ showgrid=true
+ ),
+ yaxis=PlotlyBase.attr(
+ title="Total Intensity",
+ showgrid=true,
+ tickformat=".3g"
+ ),
+ margin=attr(l=0, r=0, t=120, b=0, pad=0),
+ legend=attr(x=1.0, y=1.0, xanchor="right", yanchor="top")
)
end
@@ -2670,15 +2820,8 @@ end
# This handler will now correctly load the first image from the newly selected folder.
@onchange selected_folder_main begin
- if msi_data !== nothing
- close(msi_data)
- end
- msi_data = nothing
- log_memory_usage("Folder Changed (msi_data cleared)", msi_data)
- GC.gc() # Trigger garbage collection
- if Sys.islinux()
- ccall(:malloc_trim, Int32, (Int32,), 0) # Ensure Julia returns the freed memory to OS
- end
+ # The msi_data object lifecycle is managed by the btnSearch handler.
+ # This handler is now only for updating the UI images when the folder changes.
if !isempty(selected_folder_main)
folder_path = joinpath("public", selected_folder_main)
@@ -2914,7 +3057,7 @@ end
warning_msg = true
@error "3D plot generation failed" exception=(e, catch_backtrace())
finally
- is_processing = true
+ is_processing = false
SpectraEnabled=true
GC.gc() # Trigger garbage collection
if Sys.islinux()
@@ -2971,7 +3114,7 @@ end
warning_msg = true
@error "3D TrIQ plot generation failed" exception=(e, catch_backtrace())
finally
- is_processing = true
+ is_processing = false
SpectraEnabled=true
GC.gc() # Trigger garbage collection
if Sys.islinux()
@@ -3010,7 +3153,7 @@ end
msg="Failed to load and process image: $e"
warning_msg=true
finally
- is_processing = true
+ is_processing = false
SpectraEnabled=true
GC.gc()
if Sys.islinux()
@@ -3048,7 +3191,7 @@ end
msg="Failed to load and process image: $e"
warning_msg=true
finally
- is_processing = true
+ is_processing = false
SpectraEnabled=true
GC.gc()
if Sys.islinux()
@@ -3065,16 +3208,21 @@ end
@onchange Nmass begin
if !isempty(xSpectraMz)
df = msi_data.spectrum_stats_df
- profile_count = 0
- if df !== nothing && hasproperty(df, :Mode)
+ plot_as_lines = false # Default to stem
+ if df !== nothing && hasproperty(df, :Mode) && !isempty(df.Mode)
profile_count = count(==(MSI_src.PROFILE), df.Mode)
+ plot_as_lines = profile_count > length(df.Mode) / 2
end
- if profile_count > 0
+ # Downsample for plotting performance
+ mz_down, int_down = MSI_src.downsample_spectrum(xSpectraMz, ySpectraMz)
+
+ local traceSpectra
+ if plot_as_lines
# Main spectrum trace
traceSpectra = PlotlyBase.scatter(
- x=xSpectraMz,
- y=ySpectraMz,
+ x=mz_down,
+ y=int_down,
marker=attr(size=1, color="blue", opacity=0.5),
name="Spectrum",
hoverinfo="x",
@@ -3084,8 +3232,8 @@ end
else
# Main spectrum trace
traceSpectra = PlotlyBase.stem(
- x=xSpectraMz,
- y=ySpectraMz,
+ x=mz_down,
+ y=int_down,
marker=attr(size=1, color="blue", opacity=0.5),
name="Spectrum",
hoverinfo="x",
@@ -3252,14 +3400,15 @@ end
@mounted watchplots()
- @onchange isready @time begin
+ @onchange isready begin
# is_processing = true
if isready && !registry_init_done
- initialization_message = "Pre-compiling functions at startup..."
+ sTime=time()
+ msg = "Pre-compiling functions at startup..."
warmup_init()
- initialization_message = "Pre-compilation finished."
+ msg = "Pre-compilation finished."
try
- initialization_message = "Synchronizing registry with filesystem on backend init..."
+ msg = "Synchronizing registry with filesystem on backend init..."
reg_path = abspath(joinpath(@__DIR__, "public", "registry.json"))
registry = isfile(reg_path) ? load_registry(reg_path) : Dict{String, Any}()
@@ -3288,7 +3437,7 @@ end
end
if !isempty(new_folders) || !isempty(removed_folders)
- initialization_message = "Registry changed, saving..."
+ msg = "Registry changed, saving..."
save_registry(reg_path, registry)
end
@@ -3309,8 +3458,10 @@ end
is_initializing = false # Hide loading screen when initialization is complete
end
end
+ fTime=time()
+ eTime=round(fTime-sTime,digits=3)
is_initializing = false # Hide loading screen when initialization is complete (current code is hidden due to incompatibility)
- initialization_message = "Done."
+ msg = "The app took $(eTime) seconds to get ready."
log_memory_usage("App Ready", msi_data)
end
# is_processing = false
diff --git a/app.jl.html b/app.jl.html
index 38fcd8a..736c96c 100644
--- a/app.jl.html
+++ b/app.jl.html
@@ -708,7 +708,7 @@
Image visualizer
+ class="q-ma-sm" style="min-width: 200px;" v-on:focus="refetch_folders = true">
@@ -731,7 +731,7 @@
TrIQ visualizer
+ class="q-ma-sm" style="min-width: 200px;" v-on:focus="refetch_folders = true">
@@ -752,7 +752,7 @@
+ class="q-ma-sm" style="min-width: 200px;" v-on:focus="refetch_folders = true">
@@ -1002,7 +1002,7 @@
Dataset Summary
+ style="min-width: 250px;" standout="custom-standout" v-on:focus="refetch_folders = true">
diff --git a/config/env/global.jl b/config/env/global.jl
index 876c895..94f0b9d 100644
--- a/config/env/global.jl
+++ b/config/env/global.jl
@@ -1,2 +1,2 @@
-#ENV["GENIE_ENV"] = "dev"
-ENV["GENIE_ENV"] = "prod"
+ENV["GENIE_ENV"] = "dev"
+#ENV["GENIE_ENV"] = "prod"
diff --git a/julia_imzML_visual.jl b/julia_imzML_visual.jl
index f4bef93..37325e4 100644
--- a/julia_imzML_visual.jl
+++ b/julia_imzML_visual.jl
@@ -550,12 +550,13 @@ function meanSpectrumPlot(data::MSIData, dataset_name::String=""; mask_path::Uni
layout.title.text = "Empty " * layout.title.text
else
df = data.spectrum_stats_df
- profile_count = 0
- if df !== nothing && hasproperty(df, :Mode)
+ plot_as_lines = false # Default to stem for safety if no mode info
+ if df !== nothing && hasproperty(df, :Mode) && !isempty(df.Mode)
profile_count = count(==(MSI_src.PROFILE), df.Mode)
+ plot_as_lines = profile_count > length(df.Mode) / 2
end
- if profile_count > 0
+ if plot_as_lines
trace = PlotlyBase.scatter(x=xSpectraMz, y=ySpectraMz, mode="lines", marker=attr(size=1, color="blue", opacity=0.5), name="Average", hoverinfo="x", hovertemplate="m/z: %{x:.4f}")
else
trace = PlotlyBase.stem(x=xSpectraMz, y=ySpectraMz, marker=attr(size=1, color="blue", opacity=0.5), name="Average", hoverinfo="x", hovertemplate="m/z: %{x:.4f}")
@@ -715,7 +716,7 @@ Generates a plot for a spectrum specified by its linear `id` for any type of MSI
function nSpectrumPlot(data::MSIData, id::Int, dataset_name::String=""; mask_path::Union{String, Nothing}=nothing)
local mz::AbstractVector, intensity::AbstractVector
local plot_title::String
- local spectrum_mode = MSI_src.CENTROID # Default to centroid
+ local spectrum_mode = MSI_src.PROFILE # Default to profile
local spectrum_id::Int = id # Spectrum ID is the input id
# Validate ID
@@ -848,12 +849,13 @@ function sumSpectrumPlot(data::MSIData, dataset_name::String=""; mask_path::Unio
layout.title.text = "Empty " * layout.title.text
else
df = data.spectrum_stats_df
- profile_count = 0
- if df !== nothing && hasproperty(df, :Mode)
+ plot_as_lines = false # Default to stem for safety
+ if df !== nothing && hasproperty(df, :Mode) && !isempty(df.Mode)
profile_count = count(==(MSI_src.PROFILE), df.Mode)
+ plot_as_lines = profile_count > length(df.Mode) / 2
end
- if profile_count > 0
+ if plot_as_lines
trace = PlotlyBase.scatter(x=xSpectraMz, y=ySpectraMz, mode="lines", marker=attr(size=1, color="blue", opacity=0.5), name="Total", hoverinfo="x", hovertemplate="m/z: %{x:.4f}")
else
trace = PlotlyBase.stem(x=xSpectraMz, y=ySpectraMz, marker=attr(size=1, color="blue", opacity=0.5), name="Total", hoverinfo="x", hovertemplate="m/z: %{x:.4f}")
diff --git a/src/Common.jl b/src/Common.jl
index 8de5929..850a466 100644
--- a/src/Common.jl
+++ b/src/Common.jl
@@ -129,7 +129,7 @@ end
# --- Shared Constants ---
const DEFAULT_CACHE_SIZE = 100
const DEFAULT_NUM_BINS = 2000
-const DEFAULT_BLOOM_FILTER_SIZE = 1000 # Default size for empty spectra
+const DEFAULT_BLOOM_FILTER_SIZE = 10000 # Default size for empty spectra
const DEFAULT_FALSE_POSITIVE_RATE = 0.01
"""
diff --git a/src/MSIData.jl b/src/MSIData.jl
index 7789a89..584adaa 100644
--- a/src/MSIData.jl
+++ b/src/MSIData.jl
@@ -6,10 +6,44 @@ including caching and iteration logic, for handling large mzML and imzML dataset
efficiently.
"""
-using Base64, Libz, Serialization, Printf, DataFrames, Base.Threads, StatsBase # For reading binary data
+using Base64, Libz, Serialization, Printf, DataFrames, Base.Threads, StatsBase
+using .Platform # Import the new Platform module
const FILE_HANDLE_LOCK = ReentrantLock()
+"""
+ calculate_optimal_analytics_chunk_size(total_spectra::Int, system_caps::NamedTuple) -> Int
+
+Calculates an optimal chunk size for analytics processing based on total number of spectra
+and detected system capabilities (total memory, number of CPU threads).
+
+The heuristic aims to balance memory usage and parallelism:
+- Smaller chunks for less memory or fewer threads.
+- Larger chunks for more memory or more threads, but with a practical upper limit.
+"""
+function calculate_optimal_analytics_chunk_size(total_spectra::Int, system_caps::NamedTuple)::Int
+ # Base chunk size - can be refined with more profiling
+ base_chunk_size = 5000
+
+ # Adjust based on available RAM (heuristic: more RAM, larger chunks)
+ # Assume 1GB RAM can handle a certain chunk size comfortably
+ # E.g., for 32GB RAM, allow 2x more than for 16GB
+ memory_factor = max(1.0, system_caps.total_memory_gb / 16.0) # Scale based on 16GB baseline
+
+ # Adjust based on CPU threads (heuristic: more threads, potentially more, smaller chunks to keep threads busy, or fewer large chunks)
+ # A simple approach for now: scale with sqrt of threads to avoid overly large chunks with many threads
+ thread_factor = max(1.0, sqrt(system_caps.num_cpu_threads / 4.0)) # Scale based on 4-thread baseline
+
+ # Combine factors, with a practical minimum and maximum
+ dynamic_chunk_size = round(Int, base_chunk_size * memory_factor * thread_factor)
+
+ # Ensure chunk size is within reasonable bounds
+ min_chunk_size = 500
+ max_chunk_size = 20000
+
+ return clamp(dynamic_chunk_size, min_chunk_size, max_chunk_size)
+end
+
"""
ThreadSafeFileHandle
@@ -737,7 +771,7 @@ function precompute_analytics(msi_data::MSIData)
bloom_filters = nothing
# Process in chunks to reduce memory pressure
- chunk_size = min(1000, num_spectra)
+ chunk_size = min(10000, num_spectra)
for chunk_start in 1:chunk_size:num_spectra
chunk_end = min(chunk_start + chunk_size - 1, num_spectra)
diff --git a/src/Platform.jl b/src/Platform.jl
new file mode 100644
index 0000000..9aaa60a
--- /dev/null
+++ b/src/Platform.jl
@@ -0,0 +1,29 @@
+# src/Platform.jl
+
+module Platform
+
+using Base.Threads # For Threads.nthreads()
+
+# Export functions for system capability detection
+export detect_system_capabilities
+
+"""
+ detect_system_capabilities()
+
+Detects and returns key system capabilities such as total memory and number of CPU threads.
+
+# Returns
+- A `NamedTuple` with fields:
+ - `total_memory_gb::Float64`: Total physical memory in gigabytes.
+ - `num_cpu_threads::Int`: Number of available CPU threads.
+"""
+function detect_system_capabilities()::NamedTuple
+ total_memory_bytes = Sys.total_memory()
+ total_memory_gb = total_memory_bytes / (1024^3) # Convert bytes to gigabytes
+
+ num_cpu_threads = Sys.CPU_THREADS # Total logical CPU cores
+
+ return (total_memory_gb=total_memory_gb, num_cpu_threads=num_cpu_threads)
+end
+
+end # module Platform
diff --git a/start_MSI_GUI.jl b/start_MSI_GUI.jl
index e8b7fab..ffc8250 100644
--- a/start_MSI_GUI.jl
+++ b/start_MSI_GUI.jl
@@ -1,8 +1,8 @@
using Pkg
sTime=time()
Pkg.activate(".")
-#ENV["GENIE_ENV"] = "dev"
-ENV["GENIE_ENV"] = "prod"
+ENV["GENIE_ENV"] = "dev"
+#ENV["GENIE_ENV"] = "prod"
manifest_path = joinpath(@__DIR__, "Manifest.toml")
diff --git a/test/benchmark.jl b/test/benchmark.jl
new file mode 100644
index 0000000..8622a8d
--- /dev/null
+++ b/test/benchmark.jl
@@ -0,0 +1,932 @@
+# test/benchmark.jl
+
+# ===================================================================
+# Performance Benchmark Suite for JuliaMSI
+# ===================================================================
+# This script compares the performance of the new `JuliaMSI` pipeline
+# against the old `julia_mzML_imzML` library.
+#
+# It measures the time and memory allocation for loading an `.imzML`
+# file and generating an image slice multiple times.
+#
+# Instructions:
+# 1. Fill in the placeholder paths in the "CONFIG" section below.
+# 2. Run the script from the project's root directory:
+# julia --project=. test/benchmark.jl
+# 3. Check `test/results/` for `benchmark_results.csv` and plots.
+# ===================================================================
+
+using Pkg
+Pkg.activate(joinpath(@__DIR__, "..")) # Activate main project environment
+
+using DataFrames
+using CSV
+using CairoMakie
+using ColorSchemes
+using Statistics
+using Printf
+using JSON
+using MSI_src # Import the new library
+using .MSI_src: get_mz_slice, quantize_intensity, save_bitmap, get_multiple_mz_slices
+using julia_mzML_imzML
+
+# Remember, to run this, you must enter pkg mode in julia and add this library:
+# add github.com/CINVESTAV-LABI/julia_mzML_imzML
+
+# ===================================================================
+# CONFIG: PLEASE FILL IN YOUR FILE PATHS AND PARAMETERS
+# ===================================================================
+
+struct BenchmarkCase
+ filepath::String
+ mz_value::Float64
+ mz_tolerance::Float64
+ name::String # New field for a descriptive name
+end
+
+const BENCHMARK_CASES = [
+ # Add your benchmark cases here. Example:
+ # BenchmarkCase("/path/to/your/small_file.imzML", 309.06, 0.1),
+ # BenchmarkCase("/path/to/your/medium_file.imzML", 896.0, 1.0),
+ # BenchmarkCase("/path/to/your/large_file.imzML", 100.0, 0.1),
+ BenchmarkCase("/home/pixel/Documents/Cinvestav_2025/Analisis/imzML_LA-ESI/180817_NEG_Thaliana_Leaf_bottom_1_0841.imzML",116.07,0.1, "Thaliana Leaf"),
+ BenchmarkCase("/home/pixel/Documents/Cinvestav_2025/Analisis/imzML_LTP/ltpmsi-chilli.imzML",420,0.1, "Chilli Pepper"), # Chilli
+ BenchmarkCase("/home/pixel/Documents/Cinvestav_2025/Analisis/imzML_DESI/ColAd_Individual/40TopL,10TopR,30BottomL,20BottomR/40TopL,10TopR,30BottomL,20BottomR-centroid.imzML",885.5,0.1, "Colon Cancer Human"), # Human Cancer
+ BenchmarkCase("/home/pixel/Documents/Cinvestav_2025/Analisis/imzML_AP_SMALDI/HR2MSImouseurinarybladderS096.imzML", 716.053,0.1, "Mouse Urinary Bladder"), # Mouse bladder
+ BenchmarkCase("/home/pixel/Documents/Cinvestav_2025/Analisis/Liv2_imzML_TIMSConvert-selected/Liv2.imzML",796.18,0.1, "Liver Cut"), #Lib2
+ BenchmarkCase("/home/pixel/Documents/Cinvestav_2025/Analisis/salida/Stomach_DHB_uncompressed.imzML",804.3,0.1, "Mouse Stomach"), # Mouse Stomach
+]
+
+const NUM_REPETITIONS = 50 # Number of times to generate the image for averaging
+const BULK_MZ_VARIATIONS = 50 # Number of m/z variations to generate for the bulk benchmark (e.g., mz, mz+0.1, ..., mz+(N-1)*0.1)
+const RESULTS_DIR = joinpath(@__DIR__, "results")
+
+# Helper function to generate m/z variations for bulk processing
+function get_mz_variation_vector(base_mz::Float64, num_variations::Int)
+ return [base_mz + i * 0.1 for i in 0:(num_variations - 1)]
+end
+
+# Helper function to get the combined size of .imzML and .ibd files
+function get_total_file_size_mb(filepath::String)
+ imzml_size = isfile(filepath) ? filesize(filepath) : 0
+ ibd_path = replace(filepath, r"\.(imzML|imzml)$" => ".ibd")
+ ibd_size = isfile(ibd_path) ? filesize(ibd_path) : 0
+ return round((imzml_size + ibd_size) / 1024^2, digits=2)
+end
+
+# ===================================================================
+# BENCHMARK: JuliaMSI (New Library)
+# ===================================================================
+
+"""
+ benchmark_julianew(case::BenchmarkCase)
+
+Runs the benchmark for the modern JuliaMSI library.
+"""
+function benchmark_julianew(case::BenchmarkCase)
+ println("--- Benchmarking JuliaMSI (New) on $(basename(case.filepath)) ---")
+
+ # 1. Load the data
+ load_stats = @timed OpenMSIData(case.filepath)
+ msi_data = load_stats.value
+ load_time = load_stats.time
+ load_mem = load_stats.bytes / 1024^2 # Convert to MB
+
+ image_times = []
+ image_mems = []
+ local final_image # Declare final_image outside the loop
+
+ # 2. Generate image multiple times
+ for i in 1:NUM_REPETITIONS
+ print("\rRepetition $i/$NUM_REPETITIONS...")
+ image_stats = @timed get_mz_slice(msi_data, case.mz_value, case.mz_tolerance)
+ final_image = image_stats.value
+ push!(image_times, image_stats.time)
+ push!(image_mems, image_stats.bytes / 1024^2)
+ end
+ println("\nDone.")
+
+ # 3. Save one resulting image for validation
+ output_path_bitmap = joinpath(RESULTS_DIR, "benchmark_JuliaMSI_$(basename(case.filepath)).bmp")
+ # Quantize and save the image
+ quantized_image, _ = quantize_intensity(final_image, 20) # 256 levels
+ save_bitmap(output_path_bitmap, quantized_image, MSI_src.ViridisPalette)
+ println("Saved validation image to $output_path_bitmap")
+
+ # 4. Calculate statistics
+ println("DEBUG (New): NUM_REPETITIONS = $NUM_REPETITIONS")
+ println("DEBUG (New): load_time = $load_time")
+ println("DEBUG (New): image_times = $(image_times)")
+ println("DEBUG (New): length(image_times) = $(length(image_times))")
+ avg_image_time = mean(image_times)
+ avg_image_mem = mean(image_mems) # Added missing calculation
+ println("DEBUG (New): avg_image_time = $avg_image_time")
+ println("DEBUG (New): sum(image_times) = $(sum(image_times))")
+ total_time = load_time + sum(image_times)
+ println("DEBUG (New): calculated total_time = $total_time")
+
+ return (
+ library="JuliaMSI",
+ filepath=basename(case.filepath),
+ file_size_mb=get_total_file_size_mb(case.filepath), # Use helper function
+ load_time_s=load_time,
+ avg_image_time_s=avg_image_time,
+ total_time_s=total_time,
+ load_memory_mb=load_mem,
+ avg_image_memory_mb=avg_image_mem,
+ name=case.name # Added case name to results
+ )
+end
+
+"""
+ benchmark_julianew_bulk(case::BenchmarkCase)
+
+Runs the benchmark for the modern JuliaMSI library's bulk slice generation.
+"""
+function benchmark_julianew_bulk(case::BenchmarkCase)
+ println("--- Benchmarking JuliaMSI (Bulk) on $(basename(case.filepath)) ---")
+
+ # Generate multiple m/z values for bulk processing
+ mz_values_for_bulk = get_mz_variation_vector(case.mz_value, BULK_MZ_VARIATIONS)
+ println(" Generating $(length(mz_values_for_bulk)) m/z variations: $(mz_values_for_bulk)")
+
+ # 1. Load the data
+ load_stats = @timed OpenMSIData(case.filepath)
+ msi_data = load_stats.value
+ load_time = load_stats.time
+ load_mem = load_stats.bytes / 1024^2 # Convert to MB
+
+ image_times = []
+ image_mems = []
+ local final_slices_dict # Declare outside the loop
+
+ # 2. Generate multiple images in bulk (repeated for averaging)
+ for i in 1:1
+ print("\rRepetition $i/$NUM_REPETITIONS for bulk...")
+ # Assume MSI_src.get_multiple_mz_slices exists and returns Dict{Real, Matrix{Float64}}
+ image_stats = @timed MSI_src.get_multiple_mz_slices(msi_data, mz_values_for_bulk, case.mz_tolerance)
+ final_slices_dict = image_stats.value
+ push!(image_times, image_stats.time)
+
+ # Calculate memory for all slices combined
+ total_slices_mem = sum(Base.summarysize(slice) for slice in values(final_slices_dict)) / 1024^2
+ push!(image_mems, total_slices_mem)
+ end
+ println("\nDone.")
+
+ # No validation image saving for bulk due to multiple slices
+
+ # 3. Calculate statistics
+ avg_image_time = mean(image_times)
+ avg_image_mem = mean(image_mems)
+ total_time = load_time + sum(image_times) # Total time includes all repetitions
+
+ return (
+ library="JuliaMSIBulk",
+ filepath=basename(case.filepath),
+ file_size_mb=get_total_file_size_mb(case.filepath), # Use helper function
+ load_time_s=load_time,
+ avg_image_time_s=avg_image_time,
+ total_time_s=total_time,
+ load_memory_mb=load_mem,
+ avg_image_memory_mb=avg_image_mem,
+ name=case.name # Added case name to results
+ )
+end
+
+
+"""
+ benchmark_juliaold(case::BenchmarkCase)
+
+Runs the benchmark for the old library using dynamic environment switching.
+"""
+function benchmark_juliaold(case::BenchmarkCase)
+ println("--- Benchmarking julia_mzML_imzML (Old) on $(basename(case.filepath)) ---")
+ try
+ # --- 1. Load Data ---
+ load_stats = @timed LoadImzml(case.filepath)
+ msi_data = load_stats.value
+ load_time = load_stats.time
+ load_mem = load_stats.bytes / 1024^2 # Convert to MB
+
+ image_times = []
+ image_mems = []
+ local last_image_stats_value # Declare outside the loop
+
+ # --- 2. Generate Image Repeatedly ---
+ for i in 1:NUM_REPETITIONS
+ print("\rRepetition $i/$NUM_REPETITIONS...")
+ image_stats = @timed GetSlice(msi_data, case.mz_value, case.mz_tolerance)
+ last_image_stats_value = image_stats.value # Store the value
+ push!(image_times, image_stats.time)
+ push!(image_mems, image_stats.bytes / 1024^2)
+ end
+ println("\nDone.")
+
+ # --- 3. Save Validation Image ---
+ final_image = last_image_stats_value
+ output_path_bitmap = joinpath(RESULTS_DIR, "benchmark_old_lib_$(basename(case.filepath)).bmp")
+ SaveBitmap(output_path_bitmap, IntQuant(final_image), ViridisPalette)
+ println("Saved validation image to $output_path_bitmap")
+
+ # --- 4. Calculate Averages and Totals ---
+ println("DEBUG (Old): NUM_REPETITIONS = $NUM_REPETITIONS")
+ println("DEBUG (Old): load_time = $load_time")
+ println("DEBUG (Old): image_times = $(image_times)")
+ println("DEBUG (Old): length(image_times) = $(length(image_times))")
+ avg_image_time = mean(image_times)
+ println("DEBUG (Old): avg_image_time = $avg_image_time")
+ println("DEBUG (Old): sum(image_times) = $(sum(image_times))")
+ total_time = load_time + sum(image_times)
+ println("DEBUG (Old): calculated total_time = $total_time")
+ avg_image_mem = mean(image_mems) # Added missing calculation
+ return (
+ library="julia_mzML_imzML",
+ filepath=basename(case.filepath),
+ file_size_mb=get_total_file_size_mb(case.filepath), # Use helper function
+ load_time_s=load_time,
+ avg_image_time_s=avg_image_time,
+ total_time_s=total_time,
+ load_memory_mb=load_mem,
+ avg_image_memory_mb=avg_image_mem,
+ name=case.name # Added case name to results
+ )
+
+ catch e
+ println("ERROR benchmarking old library: $e")
+ println("Make sure julia_mzML_imzML is installed and OLD_LIB_ENV_PATH is set correctly.")
+ return nothing
+ end
+end
+
+function benchmark_multi_file_walkthrough(cases::Vector{BenchmarkCase})
+ println("--- Multi-File Walkthrough Benchmark ---")
+ println("Simulating workflow: load file -> generate 1 image -> close -> next file")
+
+ results = []
+
+ for case in cases
+ if !isfile(case.filepath)
+ @warn "File not found: $(case.filepath). Skipping."
+ continue
+ end
+
+ println("\nProcessing: $(case.name)")
+
+ # --- NEW LIBRARY (JuliaMSI) ---
+ println(" [JuliaMSI]")
+ # Time includes metadata loading + single image generation
+ new_total_stats = @timed begin
+ data = OpenMSIData(case.filepath)
+ img = get_mz_slice(data, case.mz_value, case.mz_tolerance)
+ close(data) # Explicitly release resources
+ end
+
+ # --- OLD LIBRARY (julia_mzML_imzML) ---
+ println(" [julia_mzML_imzML]")
+ old_total_stats = try
+ @timed begin
+ data = LoadImzml(case.filepath) # Loads ENTIRE dataset into RAM
+ img = GetSlice(data, case.mz_value, case.mz_tolerance)
+ # Note: No explicit close function in old library
+ end
+ catch e
+ @warn "Old library failed on $(basename(case.filepath)): $e"
+ (time=Inf, bytes=Inf, value=nothing)
+ end
+
+ push!(results, (
+ name=case.name,
+ file_size_mb=get_total_file_size_mb(case.filepath),
+ julianew_time_s=new_total_stats.time,
+ julianew_memory_mb=new_total_stats.bytes/1024^2,
+ juliaold_time_s=old_total_stats.time,
+ juliaold_memory_mb=old_total_stats.bytes/1024^2,
+ speedup=old_total_stats.time / new_total_stats.time
+ ))
+
+ # Force garbage collection between files to level the playing field
+ GC.gc()
+ if Sys.islinux()
+ ccall(:malloc_trim, Int32, (Int32,), 0)
+ end
+ end
+
+ return results
+end
+
+function benchmark_real_world_pipeline(cases::Vector{BenchmarkCase})
+ println("--- REAL-WORLD PIPELINE BENCHMARK ---")
+ println("Simulating: Load → Multiple Slices → Process → Save → Next File")
+
+ # Realistic m/z values (similar to your UI)
+ mz_values = [116.07, 420.0, 885.5, 716.053, 796.18, 804.3]
+ tolerance = 0.1
+ results = []
+
+ for case in cases
+ if !isfile(case.filepath)
+ @warn "File not found: $(case.filepath). Skipping."
+ continue
+ end
+
+ println("\nProcessing: $(case.name)")
+
+ # --- NEW LIBRARY (JuliaMSI) with bulk processing ---
+ println(" [JuliaMSI - Bulk Processing]")
+ new_stats = try
+ @timed begin
+ # 1. Load (lazy, metadata only)
+ data = OpenMSIData(case.filepath)
+
+ # 2. Generate ALL slices in one bulk operation
+ slice_dict = get_multiple_mz_slices(data, mz_values, tolerance)
+
+ # 3. Process each slice (simulate quantization, saving, etc.)
+ for (mass, slice) in slice_dict
+ # Simulate minimal processing for fair comparison
+ # In real pipeline: TrIQ, median filter, save bitmap
+ processed = quantize_intensity(slice, 256)[1]
+ end
+
+ # 4. Close and cleanup
+ close(data)
+ end
+ catch e
+ @warn "New library failed: $e"
+ (time=Inf, bytes=Inf, value=nothing)
+ end
+
+ # --- OLD LIBRARY (sequential processing) ---
+ println(" [Old Library - Sequential]")
+ old_stats = try
+ @timed begin
+ # 1. Load (everything into RAM)
+ data = LoadImzml(case.filepath)
+
+ # 2. Generate slices one by one
+ for mass in mz_values
+ slice = GetSlice(data, mass, tolerance)
+ # Same processing simulation
+ processed = IntQuant(slice)
+ end
+
+ # 3. Old library doesn't have explicit close
+ # Data stays in RAM until GC
+ end
+ catch e
+ @warn "Old library failed: $e"
+ (time=Inf, bytes=Inf, value=nothing)
+ end
+
+ # Memory cleanup between files (mirroring your pipeline)
+ GC.gc(true)
+ if Sys.islinux()
+ ccall(:malloc_trim, Int32, (Int32,), 0)
+ end
+
+ push!(results, (
+ name=case.name,
+ file_size_mb=get_total_file_size_mb(case.filepath),
+ n_slices=length(mz_values),
+ julianew_time_s=new_stats.time,
+ julianew_memory_mb=new_stats.bytes/1024^2,
+ juliaold_time_s=old_stats.time,
+ juliaold_memory_mb=old_stats.bytes/1024^2,
+ speedup=old_stats.time / new_stats.time,
+ # NEW METRICS for your pipeline:
+ memory_reduction_pct=100 * (old_stats.bytes - new_stats.bytes) / old_stats.bytes,
+ slices_per_second_new=length(mz_values) / new_stats.time,
+ slices_per_second_old=length(mz_values) / old_stats.time
+ ))
+ end
+
+ return results
+end
+
+# ===================================================================
+# Main Runner
+function run_all_benchmarks(skip_benchmark::Bool=false)
+ results_csv_path = joinpath(RESULTS_DIR, "benchmark_results.csv")
+ multi_file_walkthrough_csv_path = joinpath(RESULTS_DIR, "multi_file_walkthrough.csv")
+ real_world_pipeline_csv_path = joinpath(RESULTS_DIR, "real_world_pipeline.csv")
+
+ local all_results::DataFrame
+ local walkthrough_df::DataFrame = DataFrame() # Initialize as empty
+ local pipeline_df::DataFrame = DataFrame() # Initialize as empty
+
+ if skip_benchmark && isfile(results_csv_path) && isfile(multi_file_walkthrough_csv_path) && isfile(real_world_pipeline_csv_path)
+ @info "Skipping benchmark run. Loading results from $results_csv_path"
+ all_results = CSV.read(results_csv_path, DataFrame)
+ walkthrough_df = CSV.read(multi_file_walkthrough_csv_path, DataFrame)
+ pipeline_df = CSV.read(real_world_pipeline_csv_path, DataFrame)
+ else
+ if isempty(BENCHMARK_CASES)
+ @warn "No benchmark cases found. Please fill in the `BENCHMARK_CASES` constant in `test/benchmark.jl`."
+ return
+ end
+
+ all_results = DataFrame()
+
+ for case in BENCHMARK_CASES
+ if !isfile(case.filepath)
+ @warn "File not found: $(case.filepath). Skipping."
+ continue
+ end
+
+ # Run for new library
+ new_results = benchmark_julianew(case)
+ push!(all_results, new_results, cols=:union)
+
+ GC.gc() # Trigger garbage collection
+ if Sys.islinux()
+ ccall(:malloc_trim, Int32, (Int32,), 0) # Ensure julia returns the freed memory to OS
+ end
+
+ # Run for new library (Bulk)
+ new_bulk_results = benchmark_julianew_bulk(case)
+ push!(all_results, new_bulk_results, cols=:union)
+
+ GC.gc() # Trigger garbage collection
+ if Sys.islinux()
+ ccall(:malloc_trim, Int32, (Int32,), 0) # Ensure julia returns the freed memory to OS
+ end
+
+ # Run for old library
+ old_results = benchmark_juliaold(case)
+ if old_results !== nothing
+ push!(all_results, old_results, cols=:union)
+ end
+
+ GC.gc() # Trigger garbage collection
+ if Sys.islinux()
+ ccall(:malloc_trim, Int32, (Int32,), 0) # Ensure julia returns the freed memory to OS
+ end
+ end
+
+ # Save results to CSV (only if benchmark was actually run)
+ CSV.write(results_csv_path, all_results)
+ println("\nBenchmark results saved to $results_csv_path")
+
+ # --- Multi-File Walkthrough Benchmark ---
+ println("\n" * "="^60)
+ println("MULTI-FILE WALKTHROUGH BENCHMARK")
+ println("="^60)
+
+ walkthrough_results = benchmark_multi_file_walkthrough(BENCHMARK_CASES)
+
+ # Create a summary DataFrame
+ walkthrough_df = DataFrame(
+ name=[r.name for r in walkthrough_results],
+ file_size_mb=[r.file_size_mb for r in walkthrough_results],
+ JuliaMSI_time_s=[r.julianew_time_s for r in walkthrough_results],
+ Old_time_s=[r.juliaold_time_s for r in walkthrough_results],
+ speedup=[r.speedup for r in walkthrough_results],
+ JuliaMSI_mem_mb=[r.julianew_memory_mb for r in walkthrough_results],
+ Old_mem_mb=[r.juliaold_memory_mb for r in walkthrough_results]
+ )
+
+ println("\nMulti-file walkthrough results:")
+ show(walkthrough_df, allrows=true)
+
+ # Save to CSV
+ CSV.write(multi_file_walkthrough_csv_path, walkthrough_df)
+
+ # --- REAL-WORLD PIPELINE BENCHMARK ---
+ println("\n" * "="^60)
+ println("REAL-WORLD PIPELINE BENCHMARK")
+ println("="^60)
+
+ pipeline_results = benchmark_real_world_pipeline(BENCHMARK_CASES)
+
+ # Create a summary DataFrame
+ pipeline_df = DataFrame(
+ name=[r.name for r in pipeline_results],
+ file_size_mb=[r.file_size_mb for r in pipeline_results],
+ n_slices=[r.n_slices for r in pipeline_results],
+ JuliaMSI_time_s=[r.julianew_time_s for r in pipeline_results],
+ Old_time_s=[r.juliaold_time_s for r in pipeline_results],
+ speedup=[r.speedup for r in pipeline_results],
+ JuliaMSI_mem_mb=[r.julianew_memory_mb for r in pipeline_results],
+ Old_mem_mb=[r.juliaold_memory_mb for r in pipeline_results],
+ memory_reduction_pct=[r.memory_reduction_pct for r in pipeline_results],
+ slices_per_second_new=[r.slices_per_second_new for r in pipeline_results],
+ slices_per_second_old=[r.slices_per_second_old for r in pipeline_results]
+ )
+
+ println("\nReal-world pipeline results:")
+ show(pipeline_df, allrows=true)
+
+ # Save to CSV
+ CSV.write(real_world_pipeline_csv_path, pipeline_df)
+ end
+
+ # Generate and save plots (always happens)
+ generate_plots(all_results)
+ plot_walkthrough_results(walkthrough_df)
+ plot_real_world_pipeline_results(pipeline_df)
+end
+
+# ===================================================================
+# PLOTTING - Professional comparative visualizations
+# ===================================================================
+function generate_plots(df::DataFrame)
+ if isempty(df)
+ println("Cannot generate plots from empty dataframe.")
+ return
+ end
+
+ unique_names = unique(df.name) # Use the new name field
+ num_files = length(unique_names)
+
+ # Calculate positions for dodged bars
+ positions = 1:num_files # Define positions here
+
+ # Generate A-Z labels for X-axis
+ alphabet_labels = [string(Char('A' + i - 1)) for i in 1:num_files]
+
+ # Use a scientific color palette
+ colors = Dict(
+ "JuliaMSI" => "#2E86AB", # Professional blue
+ "julia_mzML_imzML" => "#A23B72", # Professional magenta
+ "JuliaMSIBulk" => "#2CA02C" # Professional green for bulk
+ )
+
+ # Shorter labels for display
+ library_labels = Dict(
+ "JuliaMSI" => "New",
+ "julia_mzML_imzML" => "Old",
+ "JuliaMSIBulk" => "New Bulk"
+ )
+
+ # Define specific colors for the speedup metrics for each library
+ speedup_colors_new = ["#1E88E5", "#43A047", "#FFB300", "#E53935"] # Blue, Green, Yellow, Red for JuliaMSI
+ speedup_colors_bulk = ["#0056B3", "#008000", "#FFA500", "#CC0000"] # Slightly darker/different hues for JuliaMSIBulk
+
+ # Set modern theme
+ set_theme!(Theme(
+ fontsize = 72, # General font size (1.5x of 48)
+ font = "Helvetica", # Changed to Helvetica
+ Axis = (
+ backgroundcolor = :white,
+ spinewidth = 1.5,
+ xgridvisible = false,
+ ygridvisible = true,
+ ygridcolor = (:gray, 0.2),
+ ygridwidth = 0.5,
+ xticklabelrotation = π/8,
+ xticklabalign = (:right, :center),
+ yticklabalign = (:right, :center),
+ xlabelsize = 72, # Axis label size (1.5x of 48)
+ ylabelsize = 72, # Axis label size (1.5x of 48)
+ titlesize = 72, # Axis title size (1.5x of 48)
+ xticklabelsize = 60, # Tick label size (1.5x of 40)
+ yticklabelsize = 60, # Tick label size (1.5x of 40)
+ titlefont = :bold, # Ensured bold
+ ),
+ Legend = (
+ backgroundcolor = (:white, 0.95),
+ framecolor = :gray,
+ framewidth = 1,
+ padding = (10, 10, 5, 5),
+ labelsize = 72, # Legend label size (1.5x of 48)
+ titlesize = 72, # Legend title size (1.5x of 48)
+ )
+ ))
+
+ # ===================================================================
+ # Plot 1: Main Performance Dashboard
+ # ===================================================================
+ fig1 = Figure(size = (3600, 2400)) # Explicitly set figure size
+
+ # Define grid layout with specific areas
+ g1 = fig1[2, 1:2] = GridLayout()
+
+ # Total Time
+ ax1 = Axis(g1[1, 1],
+ title = "Total Time",
+ ylabel = "Seconds", # Changed to linear scale, removed "(log scale)"
+ xticks = (positions, alphabet_labels) # Use alphabetical labels for x-axis
+ )
+
+ # Average Image Time
+ ax2 = Axis(g1[1, 2],
+ title = "Image Generation Time",
+ ylabel = "Seconds",
+ yticks = LinearTicks(5),
+ xticks = (positions, alphabet_labels) # Use alphabetical labels for x-axis
+ )
+
+ # Loading Memory
+ ax3 = Axis(g1[2, 1],
+ title = "Loading Memory",
+ ylabel = "MB",
+ yticks = LinearTicks(5),
+ xticks = (positions, alphabet_labels) # Use alphabetical labels for x-axis
+ )
+
+ # Image Generation Memory
+ ax4 = Axis(g1[2, 2],
+ title = "Image Generation Memory",
+ ylabel = "MB",
+ yticks = LinearTicks(5),
+ xticks = (positions, alphabet_labels) # Use alphabetical labels for x-axis
+ )
+
+ bar_width = 0.25 # Reduced bar width
+ x_offsets = [-0.3, 0, 0.3] # Offsets for 3 bars to avoid overlap
+
+ # Arrays to store speedup factors
+ total_time_speedup_new = zeros(num_files)
+ image_time_speedup_new = zeros(num_files)
+ load_mem_speedup_new = zeros(num_files)
+ img_mem_speedup_new = zeros(num_files)
+
+ total_time_speedup_bulk = zeros(num_files)
+ image_time_speedup_bulk = zeros(num_files)
+ load_mem_speedup_bulk = zeros(num_files)
+ img_mem_speedup_bulk = zeros(num_files)
+
+ # Initialize a list to hold rows for the summary DataFrame
+ summary_df_rows = []
+
+ for (i, name_for_df_filter) in enumerate(unique_names) # Iterate through unique names
+ # Get data for this file
+ new_data = df[(df.name .== name_for_df_filter) .& (df.library .== "JuliaMSI"), :]
+ bulk_data = df[(df.name .== name_for_df_filter) .& (df.library .== "JuliaMSIBulk"), :]
+ old_data = df[(df.name .== name_for_df_filter) .& (df.library .== "julia_mzML_imzML"), :]
+
+ # Plot total time
+ barplot!(ax1, [i + x_offsets[1], i + x_offsets[2], i + x_offsets[3]],
+ [new_data.total_time_s[1], bulk_data.total_time_s[1], old_data.total_time_s[1]],
+ color = [colors["JuliaMSI"], colors["JuliaMSIBulk"], colors["julia_mzML_imzML"]],
+ width = bar_width
+ )
+
+ # Plot average image time
+ barplot!(ax2, [i + x_offsets[1], i + x_offsets[2], i + x_offsets[3]],
+ [new_data.avg_image_time_s[1], bulk_data.avg_image_time_s[1], old_data.avg_image_time_s[1]],
+ color = [colors["JuliaMSI"], colors["JuliaMSIBulk"], colors["julia_mzML_imzML"]],
+ width = bar_width
+ )
+
+ # Plot loading memory
+ barplot!(ax3, [i + x_offsets[1], i + x_offsets[2], i + x_offsets[3]],
+ [new_data.load_memory_mb[1], bulk_data.load_memory_mb[1], old_data.load_memory_mb[1]],
+ color = [colors["JuliaMSI"], colors["JuliaMSIBulk"], colors["julia_mzML_imzML"]],
+ width = bar_width
+ )
+
+ # Plot image generation memory
+ barplot!(ax4, [i + x_offsets[1], i + x_offsets[2], i + x_offsets[3]],
+ [new_data.avg_image_memory_mb[1], bulk_data.avg_image_memory_mb[1], old_data.avg_image_memory_mb[1]],
+ color = [colors["JuliaMSI"], colors["JuliaMSIBulk"], colors["julia_mzML_imzML"]],
+ width = bar_width
+ )
+
+ # Calculate speedup factors (higher is better for new library)
+ if nrow(new_data) == 1 && nrow(old_data) == 1
+ total_time_speedup_new[i] = old_data.total_time_s[1] / new_data.total_time_s[1]
+ image_time_speedup_new[i] = old_data.avg_image_time_s[1] / new_data.avg_image_time_s[1]
+ load_mem_speedup_new[i] = old_data.load_memory_mb[1] / new_data.load_memory_mb[1]
+ img_mem_speedup_new[i] = old_data.avg_image_memory_mb[1] / new_data.avg_image_memory_mb[1]
+ end
+ if nrow(bulk_data) == 1 && nrow(old_data) == 1
+ total_time_speedup_bulk[i] = old_data.total_time_s[1] / bulk_data.total_time_s[1]
+ image_time_speedup_bulk[i] = old_data.avg_image_time_s[1] / bulk_data.avg_image_time_s[1]
+ load_mem_speedup_bulk[i] = old_data.load_memory_mb[1] / bulk_data.load_memory_mb[1]
+ img_mem_speedup_bulk[i] = old_data.avg_image_memory_mb[1] / bulk_data.avg_image_memory_mb[1]
+ end
+
+ # Collect data for the summary table CSV
+ current_case_name = unique_names[i]
+ case_row = df[(df.name .== current_case_name) .& (df.library .== "JuliaMSI"), :] # Any library row will have file_size_mb
+ file_size = nrow(case_row) == 1 ? case_row.file_size_mb[1] : NaN
+
+ push!(summary_df_rows, Dict(
+ "DS" => unique_names[i],
+ "File Size (MB)" => file_size,
+ "JMSI-TT Speedup" => round(total_time_speedup_new[i], digits=2),
+ "JMSI-IT Speedup" => round(image_time_speedup_new[i], digits=2),
+ "JMSI-LM Speedup" => round(load_mem_speedup_new[i], digits=2),
+ "JMSI-IM Speedup" => round(img_mem_speedup_new[i], digits=2),
+ "JBulk-TT Speedup" => round(total_time_speedup_bulk[i], digits=2),
+ "JBulk-IT Speedup" => round(image_time_speedup_bulk[i], digits=2),
+ "JBulk-LM Speedup" => round(load_mem_speedup_bulk[i], digits=2),
+ "JBulk-IM Speedup" => round(img_mem_speedup_bulk[i], digits=2)
+ ))
+ end
+
+ # Create DataFrame from collected rows
+ summary_df = DataFrame(summary_df_rows)
+
+ # Save the summary table to CSV
+ CSV.write(joinpath(RESULTS_DIR, "benchmark_summary_table.csv"), summary_df)
+ println("\nBenchmark summary table saved to $(joinpath(RESULTS_DIR, "benchmark_summary_table.csv"))")
+
+ # Add overall title
+ Label(fig1[1, :], "MSI Library Performance Benchmark", fontsize = 72, font = :bold)
+
+ # Add a legend
+ Legend(fig1[3, :], # Moved to row 3, spanning all columns
+ [PolyElement(color = colors["JuliaMSI"]), PolyElement(color = colors["JuliaMSIBulk"]), PolyElement(color = colors["julia_mzML_imzML"])],
+ [library_labels["JuliaMSI"], library_labels["JuliaMSIBulk"], library_labels["julia_mzML_imzML"]],
+ "Library",
+ orientation = :horizontal,
+ tellwidth = false, tellheight = true,
+ halign = :center, valign = :top, # Centered horizontally, aligned to top
+ margin = (10, 10, 10, 10)
+ )
+
+ # Adjust layout spacing
+ colgap!(g1, 1, 30)
+ rowgap!(g1, 1, 20)
+
+ # Ensure all rows and columns in g1 expand to fill available space
+ rowsize!(g1, 1, Auto())
+ rowsize!(g1, 2, Auto())
+ colsize!(g1, 1, Auto())
+ colsize!(g1, 2, Auto())
+
+ # Ensure the row containing the legend expands in the top-level figure layout
+ rowsize!(fig1.layout, 3, Auto())
+
+
+
+ # Save plots
+
+ save(joinpath(RESULTS_DIR, "performance_dashboard.png"), fig1, px_per_unit = 1)
+ println("\nEnhanced visualizations saved to $RESULTS_DIR")
+end
+
+function plot_walkthrough_results(walkthrough_df)
+ fig = Figure(size=(3600, 2400))
+
+ # Plot 1: Total time per file
+ ax1 = Axis(fig[1, 1],
+ title="Time per File (Load + Generate 1 Image)",
+ ylabel="Time (seconds)",
+ xticks=(collect(1:nrow(walkthrough_df)), walkthrough_df.name),
+ xticklabelrotation=π/8)
+
+ # Grouped bars
+ barplot!(ax1, collect(1:nrow(walkthrough_df)) .- 0.2, walkthrough_df.JuliaMSI_time_s,
+ width=0.4, color="#2E86AB", label="JuliaMSI (New)")
+ barplot!(ax1, collect(1:nrow(walkthrough_df)) .+ 0.2, walkthrough_df.Old_time_s,
+ width=0.4, color="#A23B72", label="julia_mzML_imzML (Old)")
+
+ # Plot 2: Speedup factor
+ ax2 = Axis(fig[1, 2],
+ title="Speedup Factor (Old/New Time)",
+ ylabel="Speedup (Higher = Better)",
+ xticks=(collect(1:nrow(walkthrough_df)), walkthrough_df.name),
+ xticklabelrotation=π/8)
+
+ # Bars above 1 = new library is faster
+ colors = [x >= 1 ? "#2CA02C" : "#E53935" for x in walkthrough_df.speedup]
+ barplot!(ax2, collect(1:nrow(walkthrough_df)), walkthrough_df.speedup, color=colors)
+ hlines!(ax2, [1.0], color=:black, linestyle=:dash, linewidth=2)
+
+ # Plot 3: Memory comparison
+ ax3 = Axis(fig[2, 1],
+ title="Memory Usage per File",
+ ylabel="Memory (MB)",
+ xticks=(collect(1:nrow(walkthrough_df)), walkthrough_df.name),
+ xticklabelrotation=π/8)
+
+ barplot!(ax3, collect(1:nrow(walkthrough_df)) .- 0.2, walkthrough_df.JuliaMSI_mem_mb,
+ width=0.4, color="#2E86AB", label="JuliaMSI")
+ barplot!(ax3, collect(1:nrow(walkthrough_df)) .+ 0.2, walkthrough_df.Old_mem_mb,
+ width=0.4, color="#A23B72", label="Old Library")
+
+ # Plot 4: Time vs File Size (scatter)
+ ax4 = Axis(fig[2, 2],
+ title="Processing Time vs File Size",
+ xlabel="File Size (MB)",
+ ylabel="Time (seconds)")
+
+ scatter!(ax4, walkthrough_df.file_size_mb, walkthrough_df.JuliaMSI_time_s,
+ color="#2E86AB", markersize=20, label="JuliaMSI")
+ scatter!(ax4, walkthrough_df.file_size_mb, walkthrough_df.Old_time_s,
+ color="#A23B72", markersize=20, label="Old Library")
+
+ # Add linear trend lines
+ if nrow(walkthrough_df) > 1
+ linreg_new = hcat(fill(1.0, nrow(walkthrough_df)), walkthrough_df.file_size_mb) \ walkthrough_df.JuliaMSI_time_s
+ linreg_old = hcat(fill(1.0, nrow(walkthrough_df)), walkthrough_df.file_size_mb) \ walkthrough_df.Old_time_s
+
+ x_vals = [minimum(walkthrough_df.file_size_mb), maximum(walkthrough_df.file_size_mb)]
+ lines!(ax4, x_vals, linreg_new[1] .+ linreg_new[2] .* x_vals, color="#2E86AB", linewidth=2)
+ lines!(ax4, x_vals, linreg_old[1] .+ linreg_old[2] .* x_vals, color="#A23B72", linewidth=2)
+ end
+
+ # Add legends
+ Legend(fig[0, :], [PolyElement(color="#2E86AB"), PolyElement(color="#A23B72")],
+ ["JuliaMSI (New)", "julia_mzML_imzML (Old)"],
+ orientation=:horizontal, tellwidth=false, framevisible=true)
+
+ # Adjust layout
+ rowgap!(fig.layout, 1)
+ colgap!(fig.layout, 1)
+
+ save(joinpath(RESULTS_DIR, "multi_file_walkthrough.png"), fig, px_per_unit=1)
+ println("\nMulti-file walkthrough plot saved.")
+end
+
+function plot_real_world_pipeline_results(pipeline_df::DataFrame)
+ fig = Figure(size=(3600, 2400)) # Larger figure for more detailed plots
+
+ # Filter out rows with Inf or NaN values in critical plotting columns
+ # Create a copy to avoid modifying the original DataFrame
+ filtered_df = deepcopy(pipeline_df)
+
+ # Filter for finite values in key metrics for plotting
+ filter!(row -> isfinite(row.JuliaMSI_time_s) && isfinite(row.Old_time_s) &&
+ isfinite(row.speedup) && isfinite(row.JuliaMSI_mem_mb) &&
+ isfinite(row.Old_mem_mb) && isfinite(row.memory_reduction_pct) &&
+ isfinite(row.slices_per_second_new) && isfinite(row.slices_per_second_old),
+ filtered_df)
+
+ if isempty(filtered_df)
+ println("No finite data points for real-world pipeline plotting after filtering. Skipping plot generation.")
+ save(joinpath(RESULTS_DIR, "real_world_pipeline.png"), fig, px_per_unit=1)
+ return
+ end
+
+ num_rows_filtered = nrow(filtered_df)
+ x_positions = collect(1:num_rows_filtered)
+
+ # Plot 1: Total Time Comparison
+ ax1 = Axis(fig[1, 1],
+ title="Real-World Pipeline: Total Time",
+ ylabel="Time (seconds)",
+ xticks=(x_positions, filtered_df.name),
+ xticklabelrotation=π/8)
+
+ barplot!(ax1, x_positions .- 0.2, filtered_df.JuliaMSI_time_s,
+ width=0.4, color="#2E86AB", label="JuliaMSI (New)")
+ barplot!(ax1, x_positions .+ 0.2, filtered_df.Old_time_s,
+ width=0.4, color="#A23B72", label="Old Library")
+
+ # Plot 2: Speedup Factor
+ ax2 = Axis(fig[1, 2],
+ title="Real-World Pipeline: Speedup Factor (Old/New Time)",
+ ylabel="Speedup (Higher = Better)",
+ xticks=(x_positions, filtered_df.name),
+ xticklabelrotation=π/8)
+
+ colors_speedup = [x >= 1 ? "#2CA02C" : "#E53935" for x in filtered_df.speedup]
+ barplot!(ax2, x_positions, filtered_df.speedup, color=colors_speedup)
+ hlines!(ax2, [1.0], color=:black, linestyle=:dash, linewidth=2)
+
+ # Plot 3: Memory Usage Comparison
+ ax3 = Axis(fig[2, 1],
+ title="Real-World Pipeline: Memory Usage",
+ ylabel="Memory (MB)",
+ xticks=(x_positions, filtered_df.name),
+ xticklabelrotation=π/8)
+
+ barplot!(ax3, x_positions .- 0.2, filtered_df.JuliaMSI_mem_mb,
+ width=0.4, color="#2E86AB", label="JuliaMSI")
+ barplot!(ax3, x_positions .+ 0.2, filtered_df.Old_mem_mb,
+ width=0.4, color="#A23B72", label="Old Library")
+
+ # Plot 4: Memory Reduction Percentage
+ ax4 = Axis(fig[2, 2],
+ title="Real-World Pipeline: Memory Reduction vs Old (%)",
+ ylabel="Memory Reduction (%)",
+ xticks=(x_positions, filtered_df.name),
+ xticklabelrotation=π/8)
+
+ colors_mem_red = [x > 0 ? "#2CA02C" : "#E53935" for x in filtered_df.memory_reduction_pct]
+ barplot!(ax4, x_positions, filtered_df.memory_reduction_pct, color=colors_mem_red)
+ hlines!(ax4, [0.0], color=:black, linestyle=:dash, linewidth=2)
+
+ # Plot 5: Slices Per Second
+ ax5 = Axis(fig[3, 1],
+ title="Real-World Pipeline: Slices per Second",
+ ylabel="Slices/sec (Higher = Better)",
+ xticks=(x_positions, filtered_df.name),
+ xticklabelrotation=π/8)
+
+ barplot!(ax5, x_positions .- 0.2, filtered_df.slices_per_second_new,
+ width=0.4, color="#2E86AB", label="JuliaMSI")
+ barplot!(ax5, x_positions .+ 0.2, filtered_df.slices_per_second_old,
+ width=0.4, color="#A23B72", label="Old Library")
+
+ # Overall Legend
+ Legend(fig[0, :], [PolyElement(color="#2E86AB"), PolyElement(color="#A23B72")],
+ ["JuliaMSI (New)", "julia_mzML_imzML (Old)"],
+ orientation=:horizontal, tellwidth=false, framevisible=true)
+
+ # Adjust layout
+ rowgap!(fig.layout, 1)
+ colgap!(fig.layout, 1)
+
+ save(joinpath(RESULTS_DIR, "real_world_pipeline.png"), fig, px_per_unit=1)
+ println("\nReal-world pipeline plot saved.")
+end
+
+# --- Execute ---
+mkpath(RESULTS_DIR)
+run_all_benchmarks(true) # true: skip processing files # false: create new matrix