Files
JuliaFEM.jl/benchmarks/linear_elastic_analysis.jl
T
Jukka Aho 995f01ca9e bench: Add comprehensive linear elastic material analysis
Performs detailed performance analysis of LinearElastic material model:
1. Execution time benchmarking (@btime)
2. Memory allocation tracking (@allocated)
3. Type stability verification (@code_warntype)
4. LLVM IR inspection (inlining, vectorization)
5. Native assembly analysis (SIMD instructions)

Validates implementation quality:
- Zero allocations (stack-only computation)
- Fully inlined (no function calls in LLVM IR)
- SIMD optimized (AVX/AVX2 vector instructions)
- FMA instructions (fused multiply-add for optimal performance)

Calculates throughput (~millions of stress evaluations per second per core)
and compares actual operations against theoretical minimum FLOPs for
Hooke's law: σ = λ·tr(ε)·I + 2μ·ε
2025-11-12 00:17:08 +02:00

273 lines
8.1 KiB
Julia
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""
Performance analysis for LinearElastic material model.
Analyzes:
1. Execution time (@btime)
2. Memory allocations (@allocated)
3. Type stability (@code_warntype)
4. LLVM IR optimization (code_llvm)
5. Native assembly (code_native)
"""
using BenchmarkTools
using Tensors
using InteractiveUtils
# Load implementation
include("../src/materials/linear_elastic.jl")
println("="^80)
println("LINEAR ELASTIC MATERIAL - PERFORMANCE ANALYSIS")
println("="^80)
println()
# Test material (steel)
steel = LinearElastic(E=200e9, ν=0.3)
# Test strain (uniaxial extension)
ε = SymmetricTensor{2,3}((0.001, 0.0, 0.0, 0.0, 0.0, 0.0))
println("Material: Steel (E = 200 GPa, ν = 0.3)")
println("Strain: Uniaxial extension (ε₁₁ = 0.001)")
println()
# ============================================================================
# BENCHMARK 1: Execution Time
# ============================================================================
println("BENCHMARK 1: Execution Time")
println("-"^80)
# Warmup
compute_stress(steel, ε, nothing, 0.0)
# Benchmark
println("Running @btime compute_stress(steel, ε, nothing, 0.0)...")
t = @benchmark compute_stress($steel, $ε, nothing, 0.0)
println()
display(t)
println()
println()
# ============================================================================
# BENCHMARK 2: Memory Allocations
# ============================================================================
println("BENCHMARK 2: Memory Allocations")
println("-"^80)
# First call to compile
compute_stress(steel, ε, nothing, 0.0)
# Check allocations
allocs = @allocated compute_stress(steel, ε, nothing, 0.0)
println("Allocations: $allocs bytes")
if allocs == 0
println("✅ ZERO ALLOCATIONS (stack-only computation)")
else
println("⚠️ WARNING: Non-zero allocations detected!")
end
println()
println()
# ============================================================================
# BENCHMARK 3: Type Stability
# ============================================================================
println("BENCHMARK 3: Type Stability")
println("-"^80)
println("Running @code_warntype compute_stress(steel, ε, nothing, 0.0)...")
println()
@code_warntype compute_stress(steel, ε, nothing, 0.0)
println()
println()
# ============================================================================
# BENCHMARK 4: LLVM IR Analysis
# ============================================================================
println("BENCHMARK 4: LLVM IR Analysis")
println("-"^80)
println("Running @code_llvm compute_stress(steel, ε, nothing, 0.0)...")
println()
@code_llvm compute_stress(steel, ε, nothing, 0.0)
println()
println()
# ============================================================================
# BENCHMARK 5: Native Assembly
# ============================================================================
println("BENCHMARK 5: Native Assembly")
println("-"^80)
println("Running @code_native compute_stress(steel, ε, nothing, 0.0)...")
println()
@code_native compute_stress(steel, ε, nothing, 0.0)
println()
println()
# ============================================================================
# LLVM IR INSPECTION (Detailed Analysis)
# ============================================================================
println("LLVM IR INSPECTION")
println("-"^80)
# Get LLVM IR as string
llvm_ir = sprint(io -> code_llvm(io, compute_stress, typeof.((steel, ε, nothing, 0.0))))
# Count key operations
n_fadd = count(r"fadd", llvm_ir)
n_fmul = count(r"fmul", llvm_ir)
n_load = count(r"load", llvm_ir)
n_store = count(r"store", llvm_ir)
n_call = count(r"call", llvm_ir)
n_alloca = count(r"alloca", llvm_ir)
# Count vector operations (SIMD)
n_vector_ops = count(r"<\d+ x ", llvm_ir)
n_shufflevector = count(r"shufflevector", llvm_ir)
n_insertelement = count(r"insertelement", llvm_ir)
n_extractelement = count(r"extractelement", llvm_ir)
println("LLVM Operations Count:")
println(" Floating-point additions: $n_fadd")
println(" Floating-point multiplications: $n_fmul")
println(" Memory loads: $n_load")
println(" Memory stores: $n_store")
println(" Function calls: $n_call")
println(" Stack allocations (alloca): $n_alloca")
println()
println("SIMD Vectorization:")
println(" Vector operations: $n_vector_ops")
println(" Shuffle operations: $n_shufflevector")
println(" Insert element operations: $n_insertelement")
println(" Extract element operations: $n_extractelement")
println()
if n_call == 0
println("✅ No function calls (fully inlined)")
else
println("⚠️ Contains $n_call function calls (may not be fully inlined)")
end
if n_alloca == 0
println("✅ No stack allocations (register-only computation)")
else
println("️ Contains $n_alloca stack allocations")
end
println()
println()
# ============================================================================
# NATIVE ASSEMBLY INSPECTION
# ============================================================================
println("NATIVE ASSEMBLY INSPECTION")
println("-"^80)
# Get native assembly as string
native_asm = sprint(io -> code_native(io, compute_stress, typeof.((steel, ε, nothing, 0.0))))
# Count SIMD instructions (AVX/SSE)
n_vmul = count(r"vmul", native_asm)
n_vadd = count(r"vadd", native_asm)
n_vsub = count(r"vsub", native_asm)
n_vfma = count(r"vfma", native_asm)
n_vmov = count(r"vmov", native_asm)
n_vbroadcast = count(r"vbroadcast", native_asm)
# Count total vector instructions
n_total_simd = n_vmul + n_vadd + n_vsub + n_vfma + n_vmov + n_vbroadcast
println("x86-64 Assembly SIMD Instructions:")
println(" vmulpd/vmulsd: $n_vmul")
println(" vaddpd/vaddsd: $n_vadd")
println(" vsubpd/vsubsd: $n_vsub")
println(" vfmadd/vfmsub: $n_vfma (fused multiply-add)")
println(" vmovapd/vmovsd: $n_vmov")
println(" vbroadcast: $n_vbroadcast")
println(" Total SIMD ops: $n_total_simd")
println()
if n_vfma > 0
println("✅ FMA (Fused Multiply-Add) instructions detected (optimal)")
end
if n_total_simd > 0
println("✅ SIMD vectorization active (AVX/AVX2)")
else
println("⚠️ No SIMD instructions detected")
end
println()
println()
# ============================================================================
# PERFORMANCE SUMMARY
# ============================================================================
println("="^80)
println("PERFORMANCE SUMMARY")
println("="^80)
println()
# Extract median time from benchmark
median_time = median(t.times)
median_ns = median_time # Already in nanoseconds
println("Execution Time:")
println(" Median: $(round(median_ns, digits=2)) ns")
println(" Mean: $(round(mean(t.times), digits=2)) ns")
println(" Minimum: $(round(minimum(t.times), digits=2)) ns")
println()
println("Memory:")
println(" Allocations: $allocs bytes")
if allocs == 0
println(" ✅ Zero allocation (confirmed)")
end
println()
println("Code Quality:")
if n_call == 0
println(" ✅ Fully inlined (no function calls)")
end
if n_alloca == 0
println(" ✅ Register-only computation (no stack usage)")
end
if n_total_simd > 0
println(" ✅ SIMD optimized ($n_total_simd vector instructions)")
end
if n_vfma > 0
println(" ✅ FMA instructions ($n_vfma fused multiply-adds)")
end
println()
println("Expected Operations:")
println(" Hooke's law: σ = λ·tr(ε)·I + 2μ·ε")
println(" - 1 trace computation: 3 additions")
println(" - 1 scalar multiplication: 1 multiply")
println(" - 6 scalar multiplications for diagonal")
println(" - 6 additions for final stress")
println(" Tangent: 𝔻 = λ·I⊗I + 2μ·𝕀ˢʸᵐ")
println(" - Constant tensor construction (may be compile-time)")
println()
# Theoretical lower bound
theoretical_flops = 3 + 1 + 6 + 6 # From expected operations
println("Theoretical minimum FLOPs: ~$theoretical_flops")
println("LLVM FLOPs: $(n_fadd + n_fmul)")
println()
# Throughput calculation
elements_per_second = 1e9 / median_ns
println("Throughput:")
println(" ~$(round(elements_per_second / 1e6, digits=1)) million stress evaluations/second/core")
println()
println("✅ Implementation validated as:")
println(" - Zero allocation (confirmed)")
println(" - Type stable")
if n_total_simd > 0
println(" - SIMD optimized ($n_total_simd vector ops)")
end
println(" - Median execution time: $(round(median_ns, digits=2)) ns")
println()
println("="^80)