From e6e6c27bd9efe864036078394eda8b873af44be9 Mon Sep 17 00:00:00 2001 From: Lars Pastewka Date: Wed, 16 Sep 2026 18:16:49 +0200 Subject: [PATCH 1/2] DOC: Changelog and version for a first release (v1.0.0) muTopOpt had no changelog, so this one opens with a summary of what the project does rather than a diff against a previous release. __version__ goes 0.0.1 -> 1.0.0; flit reads it from the module, so that is the only place it lives. Two things must happen before this is tagged, both noted in the files: - the muGrid pin in pyproject.toml is still >=0.112.0, but ConsistentDoubleWell now calls muGrid's NodalMomentOperator, which no released muGrid has (1.2.0 does not). The pin has to name whichever muGrid release ships it; - this branch is cut from main, so it does not itself contain the two feature branches whose behaviour the changelog describes (the fused double well, and the version banner written into the output file). Merge those first. Co-Authored-By: Claude Opus 5 (1M context) --- CHANGELOG.md | 50 ++++++++++++++++++++++++++++++++++++++++++++ muTopOpt/__init__.py | 2 +- pyproject.toml | 5 +++++ 3 files changed, 56 insertions(+), 1 deletion(-) create mode 100644 CHANGELOG.md diff --git a/CHANGELOG.md b/CHANGELOG.md new file mode 100644 index 0000000..1dac55e --- /dev/null +++ b/CHANGELOG.md @@ -0,0 +1,50 @@ +Change log for muTopOpt +======================= + +v1.0.0 (16Sep26) +---------------- + +First release. muTopOpt does FFT-accelerated finite-element topology +optimization of periodic metamaterials: it designs a unit cell whose +homogenized stiffness (or conductivity) matches a prescribed target, by the +method of Jödicke et al., *Topology optimization of metamaterials with +FFT-accelerated micromechanical solvers*. + +What it does + +- Stress-matching objective against a target effective stiffness, given either + as bulk and shear moduli or as Young's modulus and Poisson's ratio, with + phase-field regularization and no explicit volume constraint. A conductivity + analogue (`FluxTargetProblem`) shares the same machinery +- Exact sensitivities by the discrete adjoint method. The finite-difference + gradient check in `test/test_gradient.py` is the correctness gate for the + whole pipeline and runs serially and under MPI +- Dimension-agnostic: the same code paths run 2D (3 load cases) and 3D (6) +- Two density discretizations: element-wise (per-pixel, FD-Laplacian penalty) + and nodal finite-element (element-consistent H¹ seminorm), the latter acting + as an implicit sensitivity filter so the optimizer can merge or dissolve + features instead of locking in the initial topology +- Two outer optimizers, both MPI-distributed through NuMPI: a bound-constrained + L-BFGS and a trust-region Newton-CG with exact Hessian-vector products from + the second-order adjoint. The trust region is the default where available, + since its acceptance test compares against a computable predicted reduction + and so cannot drown in inner-solve noise the way a line search does +- Adaptive inner CG tolerance coupled to the outer optimizer, and + precision-aware tolerance defaults +- Restart from a previous run's output, Fourier-resampled if the grids differ +- NetCDF output, flushed per frame, carrying the full invocation and the + versions that produced it + +Scale + +No stiffness tensor and no strain or stress field is ever stored: the operator, +the preconditioner and the sensitivity are all matrix-free and fused, and all +fields share one ghosted, MPI-decomposed, optionally device-resident layout. +A 512³ design in single precision fits in about 55 GiB on one GPU and runs +entirely on device. + +Requirements + +`muGrid` provides the FFT engine, the domain decomposition, the fused operators +and the preconditioners. This release needs a muGrid that provides +`NodalMomentOperator` (see the pin in `pyproject.toml`). diff --git a/muTopOpt/__init__.py b/muTopOpt/__init__.py index 4da0db0..abf45d3 100644 --- a/muTopOpt/__init__.py +++ b/muTopOpt/__init__.py @@ -41,7 +41,7 @@ rho, info = optimize_bounded_lbfgs(problem, initial_density(homog.nb_pixels)) """ -__version__ = "0.0.1" +__version__ = "1.0.0" from .conduction import HomogenizationConductivity, SimpConductivity from .conduction_problem import FluxLoadCase, FluxTargetProblem diff --git a/pyproject.toml b/pyproject.toml index 71a2532..c7dd79b 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -17,6 +17,11 @@ dependencies = [ "numpy", # >= 0.112.0 for the dtype-threaded preconditioner factories, which let a # float32 Homogenization run the whole solve in single precision. + # + # NOTE: the consistent nodal double well (muTopOpt.nodal.ConsistentDoubleWell) + # calls muGrid's NodalMomentOperator, which is newer than this pin. Raise the + # pin to the first muGrid release that ships it before tagging v1.0.0 -- + # muGrid 1.2.0 does not have it. "muGrid>=0.112.0", "NuMPI", ] From c5cecd800a517ba5fc3ff5d5e5c3d3ee332f89dd Mon Sep 17 00:00:00 2001 From: Lars Pastewka Date: Wed, 16 Sep 2026 18:34:56 +0200 Subject: [PATCH 2/2] DOC: Correct the 512^3 memory figure to what was measured The "about 55 GiB" figure was extrapolated from single objective/gradient evaluations at 128^3 and 192^3, which never exercise the outer optimizer's own state. A real 512^3 L-BFGS run sits at 54 GiB through the early solves and then steps to 61.5 GiB and beyond as the later load cases and the L-BFGS history allocate. Replaced with a claim that does not turn on a number still in motion: it fits on a single 128 GB unified-memory accelerator, with the measured figure given as a floor rather than an estimate. Co-Authored-By: Claude Opus 5 (1M context) --- CHANGELOG.md | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 1dac55e..935b26f 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -40,8 +40,9 @@ Scale No stiffness tensor and no strain or stress field is ever stored: the operator, the preconditioner and the sensitivity are all matrix-free and fused, and all fields share one ghosted, MPI-decomposed, optionally device-resident layout. -A 512³ design in single precision fits in about 55 GiB on one GPU and runs -entirely on device. +A 512³ design in single precision fits on a single 128 GB unified-memory +accelerator -- measured above 60 GiB resident during the first L-BFGS +iterations of an MI300A run -- and runs entirely on device. Requirements