diff --git a/CMakeLists.txt b/CMakeLists.txt index b57f1f00..608ff1bf 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -390,7 +390,7 @@ if(CVC_BUILD_PYCVC) endif() # ************* Install documentation & resources ************* -install(FILES README.md USAGE.md LICENSE +install(FILES README.md USAGE.md LICENSE THIRD_PARTY_NOTICES.md DESTINATION ${CMAKE_INSTALL_DOCDIR} COMPONENT libcvc) diff --git a/README.md b/README.md index bafa5e9c..13efd8a8 100644 --- a/README.md +++ b/README.md @@ -705,6 +705,15 @@ libcvc is licensed under the **GNU Lesser General Public License, version 2.1** The bundled XmlRpc++ sources under `inc/xmlrpc/` and `src/xmlrpc/` are Copyright (c) 2002-2003 Chris Morley and are used under LGPL-2.1-or-later. +Portions of `src/cvcGL/LowMemoryPolyDataMapper.cpp` are derived from VTK 9.5.0, +Copyright (c) Ken Martin, Will Schroeder, Bill Lorensen, and are used under the +BSD-3-Clause license. That and the other third-party code in, or compiled into, +libcvc and cvcGL -- stb_image, PocketFFT, the contour library, kazlib, sdf, +mtxlib, the DTU `.vtk` reader, VTK's marching-cubes table, the ABYSSAL ocean +port in the cvcGL examples, and the vendored CMake modules -- are listed with +their notices in [THIRD_PARTY_NOTICES.md](THIRD_PARTY_NOTICES.md), which is +installed with the documentation. + ## Credits Based on research and software developed at the **Computational Visualization Center, University of Texas at Austin**. diff --git a/THIRD_PARTY_NOTICES.md b/THIRD_PARTY_NOTICES.md new file mode 100644 index 00000000..f83f4be2 --- /dev/null +++ b/THIRD_PARTY_NOTICES.md @@ -0,0 +1,451 @@ +# Third-party notices + +libcvc is licensed under the GNU Lesser General Public License, version 2.1 +(see [LICENSE](LICENSE)). This file records the code in this repository -- and +the code compiled into its libraries from a dependency -- that comes from, or +is derived from, authors other than The University of Texas at Austin, each +with its notice exactly as the source carries it. It is installed with the +documentation by the libcvc and cvcGL installs and ships in the cvcgl-examples +package. + +Compiled into libcvc: + +- [stb_image / stb_image_write](#stb_image-230-and-stb_image_write-116) +- [PocketFFT](#pocketfft) +- [XmlRpc++](#xmlrpc) +- [readvtk / writevtk](#readvtk--writevtk-technical-university-of-denmark) +- [The contour library](#the-contour-library) and + [kazlib's dictionary](#kazlib-dictionary) +- [sdf](#sdf-signed-distance-function) and [mtxlib](#mtxlib) +- [VTK](#vtk-the-visualization-toolkit) (also compiled into libcvcGL) + +Compiled into the cvcGL examples: [ABYSSAL](#abyssal). + +In the build scripts only: [CMake modules](#cmake-modules-build-scripts-only). + +## stb_image 2.30 and stb_image_write 1.16 + +`src/cvc/image/third_party/stb_image.h` and `stb_image_write.h` (Sean Barrett +and contributors), compiled into libcvc by `src/cvc/image/stb_io.cpp`. Both +files end with this notice: + +``` +This software is available under 2 licenses -- choose whichever you prefer. +------------------------------------------------------------------------------ +ALTERNATIVE A - MIT License +Copyright (c) 2017 Sean Barrett +Permission is hereby granted, free of charge, to any person obtaining a copy of +this software and associated documentation files (the "Software"), to deal in +the Software without restriction, including without limitation the rights to +use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +of the Software, and to permit persons to whom the Software is furnished to do +so, subject to the following conditions: +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. +------------------------------------------------------------------------------ +ALTERNATIVE B - Public Domain (www.unlicense.org) +This is free and unencumbered software released into the public domain. +Anyone is free to copy, modify, publish, use, compile, sell, or distribute this +software, either in source code form or as a compiled binary, for any purpose, +commercial or non-commercial, and by any means. +In jurisdictions that recognize copyright laws, the author or authors of this +software dedicate any and all copyright interest in the software to the public +domain. We make this dedication for the benefit of the public at large and to +the detriment of our heirs and successors. We intend this dedication to be an +overt act of relinquishment in perpetuity of all present and future rights to +this software under copyright law. +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN +ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION +WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. +``` + +## PocketFFT + +Not vendored: `pocketfft_hdronly.h` comes from the cvcpkg `pocketfft` package +(github.com/mreineck/pocketfft, `cpp` branch, commit +`c90e55b3d529f8efa40ed01a20de22405f45fc65`). Being header-only, its code is +compiled into libcvc (`src/cvc/volume/volume_ops.cpp`) whenever +`CVC_FFT_PROVIDER=pocketfft`, the default. The header's notice: + +``` +This file is part of pocketfft. + +Copyright (C) 2010-2024 Max-Planck-Society +Copyright (C) 2019-2020 Peter Bell + +For the odd-sized DCT-IV transforms: + Copyright (C) 2003, 2007-14 Matteo Frigo + Copyright (C) 2003, 2007-14 Massachusetts Institute of Technology + +For the prev_good_size search: + Copyright (C) 2024 Tan Ping Liang, Peter Bell + +For the safeguards against integer overflow in good_size search: + Copyright (C) 2024 Cris Luengo + +Authors: Martin Reinecke, Peter Bell + +All rights reserved. + +Redistribution and use in source and binary forms, with or without modification, +are permitted provided that the following conditions are met: + +* Redistributions of source code must retain the above copyright notice, this + list of conditions and the following disclaimer. +* Redistributions in binary form must reproduce the above copyright notice, this + list of conditions and the following disclaimer in the documentation and/or + other materials provided with the distribution. +* Neither the name of the copyright holder nor the names of its contributors may + be used to endorse or promote products derived from this software without + specific prior written permission. + +THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND +ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED +WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE +DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE FOR +ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES +(INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; +LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON +ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT +(INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS +SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. +``` + +## XmlRpc++ + +The bundled XmlRpc++ sources under `inc/xmlrpc/` and `src/xmlrpc/`, built only +with `CVC_USING_XMLRPC=ON` (default OFF), are used under the GNU Lesser General +Public License, version 2.1 or later. The headers in `inc/xmlrpc/` carry this +notice (the `.cpp` files carry none of their own): + +``` +// XmlRpc++ Copyright (c) 2002-2003 by Chris Morley +// This library is free software; you can redistribute it and/or +// modify it under the terms of the GNU Lesser General Public +// License as published by the Free Software Foundation; either +// version 2.1 of the License, or (at your option) any later version. +// +// This library is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU +// Lesser General Public License for more details. +// +// You should have received a copy of the GNU Lesser General Public +// License along with this library; if not, write to the Free Software +// Foundation, Inc., 59 Temple Place, Suite 330, Boston, MA 02111-1307 +``` + +Its `inc/xmlrpc/base64.h` carries only: + +``` +// base64.hpp +// Autor Konstantin Pilipchuk +// mailto:lostd@ukr.net +``` + +## readvtk / writevtk (Technical University of Denmark) + +The legacy `.vtk` reader and writer in `src/cvc/volume/vtk_io.cpp`, compiled +into libcvc, carry this notice and no license terms of their own: + +``` +/* readvtk/writevtk: + * Copyright (c) The Technical University of Denmark + * Author: Ojaswa Sharma + * E-mail: os@imm.dtu.dk + * File: vtkIO.h + * .vtk I/O + */ +``` + +## The contour library + +`src/cvc/geometry/cvc-mesher/contour/` (with copies of its `cubes.h` in +`cvc-mesher/LBIE/` and `SDF/SignDistanceFunction_v2/`), compiled into libcvc +with `CVC_ENABLE_MESHER` (default ON). Its files carry these lines and no +license terms of their own; files without a header belong to the same library: + +``` +Copyright (c) 1997 Dan Schikore +Copyright (c) 1997 Dan Schikore - modified by Emilio Camahort, 1999 +Copyright (c) 1997 Dan Schikore - updated by Emilio Camahort, 1998 +Copyright (c) 1997 Dan Schikore - updated by Emilio Camahort, 1999 +Copyright (c) 1998 Emilio Camahort +Copyright (c) 1998 Emilio Camahort, Dan Schikore +Copyright (c) 1998 Emilio Camahort - ecamahor@cs.utexas.edu +Copyright (c) 1999 Emilio Camahort +``` + +and, in the priority-queue headers (`ipqueue.h` and its siblings): + +``` + * Author: Dan Schikore + * + * Adapted from pqueue.h + * Ported to C++ by Raymund Merkert - June 1995 + * Changes by Fausto Bernardini - Sept 1995 +``` + +## kazlib dictionary + +`src/cvc/geometry/cvc-mesher/contour/dict.c` and `dict.h`, compiled with the +contour library: + +``` +/* + * Dictionary Abstract Data Type + * Copyright (C) 1997 Kaz Kylheku + * + * Free Software License: + * + * All rights are reserved by the author, with the following exceptions: + * Permission is granted to freely reproduce and distribute this software, + * possibly in exchange for a fee, provided that this copyright notice appears + * intact. Permission is also granted to adapt this software to produce + * derivative works, as long as the modified versions carry this copyright + * notice and additional notices stating that the work has been modified. + * This source code may be translated into executable form and incorporated + * into proprietary software; there is no requirement for such software to + * contain a copyright notice related to this source. +``` + +## sdf (signed distance function) + +`src/cvc/geometry/SDF/SignDistanceFunction_v2/`, compiled into libcvc with +`CVC_ENABLE_SDF` (default ON), is used under the GNU Lesser General Public +License, version 2.1. Its files that carry a header carry: + +``` + Copyright (c): Xiaoyu Zhang (xiaoyu@csusm.edu) + + This file is part of sdf (signed distance function). + + sdf is free software; you can redistribute it and/or + modify it under the terms of the GNU Lesser General Public + License version 2.1 as published by the Free Software Foundation. + + sdf is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU + Lesser General Public License for more details. + + You should have received a copy of the GNU Lesser General Public + License along with this library; if not, write to the Free Software + Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA +``` + +and, in `diskio.h` and `geom.h`, `(c) 2000 Xiaoyu Zhang`. + +## mtxlib + +`src/cvc/geometry/SDF/SignDistanceFunction_v2/mtxlib.h` and `mtxlib.cpp` (C++ +Matrix Library 2.6), compiled with sdf. Portions Copyright (C) Dante Treglia II +and Mark A. DeLoura, 2000. The notice, as `mtxlib.h` carries it (`mtxlib.cpp` +has the same text): + +``` +/* Copyright (C) Dante Treglia II and Mark A. DeLoura, 2000. + * All rights reserved worldwide. + * + * This software is provided "as is" without express or implied + * warranties. You may freely copy and compile this source into + * applications you distribute provided that the copyright text + * below is included in the resulting source code, for example: + * "Portions Copyright (C) Dante Treglia II and Mark A. DeLoura, 2000" + */ +``` + +## VTK (The Visualization Toolkit) + +- Portions of `src/cvcGL/LowMemoryPolyDataMapper.cpp` (libcvcGL) -- the + regions between `BEGIN VTK 9.5.0-DERIVED` and `END VTK 9.5.0-DERIVED` -- are + derived from VTK 9.5.0 (`Rendering/OpenGL2`: `vtkGLSLModCoincidentTopology.cxx`, + `vtkOpenGLLowMemoryPolyDataMapper.cxx`, `vtkOpenGLLowMemoryCellTypeAgent.cxx`, + `vtkOpenGLLowMemoryVerticesAgent.cxx`, `vtkOpenGLLowMemoryLinesAgent.cxx`, + `vtkOpenGLLowMemoryPolygonsAgent.cxx`). The same notice is reproduced at the + top of that source file. +- The marching-cubes triangulation table in `inc/cvc/volren/detail/mc_tables.h` + (libcvc's volume renderer) is VTK's `triCases` from `vtkMarchingCubesCases.h`, + by way of volrover's libiso. + +Both are used under VTK's license: + +``` +SPDX-FileCopyrightText: Copyright (c) Ken Martin, Will Schroeder, Bill Lorensen +SPDX-License-Identifier: BSD-3-Clause + +Copyright (c) 1993-2015 Ken Martin, Will Schroeder, Bill Lorensen +All rights reserved. + +Redistribution and use in source and binary forms, with or without +modification, are permitted provided that the following conditions are met: + + * Redistributions of source code must retain the above copyright notice, + this list of conditions and the following disclaimer. + + * Redistributions in binary form must reproduce the above copyright notice, + this list of conditions and the following disclaimer in the documentation + and/or other materials provided with the distribution. + + * Neither name of Ken Martin, Will Schroeder, or Bill Lorensen nor the names + of any contributors may be used to endorse or promote products derived + from this software without specific prior written permission. + +THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS ``AS IS'' +AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE +IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE +ARE DISCLAIMED. IN NO EVENT SHALL THE AUTHORS OR CONTRIBUTORS BE LIABLE FOR +ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL +DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR +SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER +CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, +OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE +OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. +``` + +## ABYSSAL + +The GPU FFT ocean in `src/cvcGL/examples/OceanFFT.h` and `OceanFFT.cpp` -- its +spectrum, time evolution, butterfly IFFT and assembly -- is a C++/VTK port of +ABYSSAL (github.com/Token-Gremlin/natural-disasters). It is built into the +`lsystem_coast` example (the cvcgl-examples package) and the `cvcgl_ocean_fft` +test. Upstream's `LICENSE`: + +``` +MIT License + +Copyright (c) 2026 Davi (Token-Gremlin) + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. +``` + +## CMake modules (build scripts only) + +These run at configure time; nothing from them is compiled or installed. + +`CMake/BundleUtilities.cmake`: + +``` +#============================================================================= +# CMake - Cross Platform Makefile Generator +# Copyright 2000-2009 Kitware, Inc., Insight Software Consortium +# All rights reserved. +# +# Redistribution and use in source and binary forms, with or without +# modification, are permitted provided that the following conditions +# are met: +# +# * Redistributions of source code must retain the above copyright +# notice, this list of conditions and the following disclaimer. +# +# * Redistributions in binary form must reproduce the above copyright +# notice, this list of conditions and the following disclaimer in the +# documentation and/or other materials provided with the distribution. +# +# * Neither the names of Kitware, Inc., the Insight Software Consortium, +# nor the names of their contributors may be used to endorse or promote +# products derived from this software without specific prior written +# permission. +# +# THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS +# "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT +# LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR +# A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT +# HOLDER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, +# SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT +# LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, +# DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY +# THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT +# (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE +# OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. +#============================================================================= +``` + +`CMake/GetPrerequisites.cmake` (the `Copyright.txt` it refers to is not in this +repository): + +``` +#============================================================================= +# Copyright 2008-2009 Kitware, Inc. +# +# Distributed under the OSI-approved BSD License (the "License"); +# see accompanying file Copyright.txt for details. +# +# This software is distributed WITHOUT ANY WARRANTY; without even the +# implied warranty of MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. +# See the License for more information. +#============================================================================= +# (To distribute this file outside of CMake, substitute the full +# License text for the above reference.) +``` + +`CMake/FindPETSc.cmake` (the `COPYING-CMAKE-SCRIPTS` it refers to is not in this +repository): + +``` +# Redistribution and use is allowed according to the terms of the BSD license. +# For details see the accompanying COPYING-CMAKE-SCRIPTS file. +``` + +`CMake/cuda/FindCUDA.cmake` and `CMake/cuda/FindCUDA/{make2cmake,parse_cubin,run_nvcc}.cmake` +(`run_nvcc.cmake` names NVIDIA only): + +``` +# James Bigler, NVIDIA Corp (nvidia.com - jbigler) +# Abe Stephens, SCI Institute -- http://www.sci.utah.edu/~abe/FindCuda.html +# +# Copyright (c) 2008 - 2009 NVIDIA Corporation. All rights reserved. +# +# Copyright (c) 2007-2009 +# Scientific Computing and Imaging Institute, University of Utah +# +# This code is licensed under the MIT License. See the FindCUDA.cmake script +# for the text of the license. + +# The MIT License +# +# License for the specific language governing rights and limitations under +# Permission is hereby granted, free of charge, to any person obtaining a +# copy of this software and associated documentation files (the "Software"), +# to deal in the Software without restriction, including without limitation +# the rights to use, copy, modify, merge, publish, distribute, sublicense, +# and/or sell copies of the Software, and to permit persons to whom the +# Software is furnished to do so, subject to the following conditions: +# +# The above copyright notice and this permission notice shall be included +# in all copies or substantial portions of the Software. +# +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS +# OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL +# THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING +# FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER +# DEALINGS IN THE SOFTWARE. +``` diff --git a/cvcpkg/recipes/cvcgl-cuda/recipe.yaml b/cvcpkg/recipes/cvcgl-cuda/recipe.yaml index 3e2375b3..af170f47 100644 --- a/cvcpkg/recipes/cvcgl-cuda/recipe.yaml +++ b/cvcpkg/recipes/cvcgl-cuda/recipe.yaml @@ -5,13 +5,18 @@ recipe: # Reset to 1 for the 3.3.0 upstream version: cvc_revision counts # revisions OF an upstream version, so a version bump restarts it. # Bumped 1 -> 3 (2 is #522's): ships share/cvcGL/ (the wasm state shim), as cvcgl 3.4.0+cvc.18. - cvc_revision: 3 + # Bumped 3 -> 4 (3 is master's wasm app contract above, and the 2 set aside for this change + # sits below it; none published yet): carries cvcGL's fast-draw low-memory mapper and the + # BSD-3-Clause license metadata for its VTK 9.5.0-derived code (same as cvcgl 3.4.0+cvc.19). + cvc_revision: 4 maintainer: "cvcpkg group" maintainer_email: "info@cvcpkg.org" maintainer_url: "https://cvcpkg.org" description: "cvcGL (CUDA/GPU build) — the Qt-free VTK scene-graph library built against the CUDA libcvc-cuda SDK, so its GPU-only code paths (GL <-> CUDA interop: mapping a volume's CUDA unified/device buffer into a GL 3D texture for on-device volume rendering) compile in. Same cmake package (cvc::cvcGL) and same API as the CPU `cvcgl` package." homepage: https://github.com/transfix/libcvc - license: LGPL-2.1-only + # BSD-3-Clause: libcvcGL carries code derived from VTK 9.5.0 + # (LowMemoryPolyDataMapper.cpp); THIRD_PARTY_NOTICES.md ships in share/doc/cvcGL. + license: "LGPL-2.1-only AND BSD-3-Clause" tags: [scientific, volumetric, scene, vtk, graphics, cuda, gpu] # The GPU build of cvcGL — identical sources to the CPU `cvcgl` recipe, built diff --git a/cvcpkg/recipes/cvcgl-examples/build-wasm.sh b/cvcpkg/recipes/cvcgl-examples/build-wasm.sh index dfce801d..c143dea2 100755 --- a/cvcpkg/recipes/cvcgl-examples/build-wasm.sh +++ b/cvcpkg/recipes/cvcgl-examples/build-wasm.sh @@ -75,3 +75,8 @@ python3 "$CVC_SOURCE_DIR/src/cvcGL/examples/wasm/build-pages.py" \ --bin "$CVC_BUILD_DIR/bin" \ --out "$WEB" \ --assets "$CVC_SOURCE_DIR/src/cvcGL/examples/wasm/gallery-assets" +# The .wasm files embed third-party code (VTK-derived code in cvcGL, BSD-3-Clause; +# stb, sdf / mtxlib and the DTU .vtk reader in the cvc core) whose notices a binary +# redistribution must reproduce: ship them, and the LICENSE they link to, beside +# the gallery. +install -m 644 "$CVC_SOURCE_DIR/THIRD_PARTY_NOTICES.md" "$CVC_SOURCE_DIR/LICENSE" "$WEB/" diff --git a/cvcpkg/recipes/cvcgl-examples/recipe.yaml b/cvcpkg/recipes/cvcgl-examples/recipe.yaml index ab2484e8..20bbaec1 100644 --- a/cvcpkg/recipes/cvcgl-examples/recipe.yaml +++ b/cvcpkg/recipes/cvcgl-examples/recipe.yaml @@ -18,13 +18,22 @@ recipe: # CVC_STATE_EXEC option is gone; state_exec is always built), so the wasm # gallery carries the program lanes that Ariadne apps run on. # Bumped 3 -> 17 (15 published, 16 is #522's): every wasm demo is a cvcgl_wasm_app; ships glsync. - cvc_revision: 17 + # Bumped 17 -> 18 (15 went out as a native publish, 17 is master's wasm app contract above, and + # the 16 set aside for this change sits below it; revisions stay monotonic across platforms): + # ships THIRD_PARTY_NOTICES.md + LICENSE (native share/cvcgl-examples/ and the wasm web/ + # payload), declares LGPL-2.1-only AND BSD-3-Clause AND MIT, and rebuilds the demos against + # cvcgl 3.4.0+cvc.19 (fast-draw low-memory mapper). + cvc_revision: 18 maintainer: "cvcpkg group" maintainer_email: "info@cvcpkg.org" maintainer_url: "https://cvcpkg.org" description: "cvcGL example programs as ready-to-run commands: the GRL-SNAM navigation demos (nav_city_swarm / nav_city_drive / nav_fog_ghost / nav_finale — the torch-free cvc::nav reactive swarm rendered natively) plus terrain_lab, as native binaries in bin/, AND the full navigation gallery as a threaded (wasm-mt) WebAssembly browser build (share/cvcgl-examples/web/) with a cvcgl-examples-web launcher that serves it cross-origin-isolated (COOP/COEP) and opens a browser." homepage: https://github.com/transfix/libcvc - license: LGPL-2.1-only + # BSD-3-Clause: the demos link cvcGL statically, which carries code derived + # from VTK 9.5.0. MIT: lsystem_coast's OceanFFT is a port of ABYSSAL. Both + # notices are in THIRD_PARTY_NOTICES.md, shipped in share/cvcgl-examples/ (and + # share/cvcgl-examples/web/ for the browser build). + license: "LGPL-2.1-only AND BSD-3-Clause AND MIT" tags: [scientific, graphics, examples, vtk, wasm] # The examples are demos of cvcGL-as-a-toolkit, packaged separately from the diff --git a/cvcpkg/recipes/cvcgl/recipe.yaml b/cvcpkg/recipes/cvcgl/recipe.yaml index 585eacb7..7b5363d7 100644 --- a/cvcpkg/recipes/cvcgl/recipe.yaml +++ b/cvcpkg/recipes/cvcgl/recipe.yaml @@ -70,13 +70,22 @@ recipe: # scene-bounds walk deferred under hidden chrome. ABI-stale: SceneGraph/GraphicsNode/ # NullGraphicNode layouts changed, so the published +cvc.14 must not be mixed with this build. # Bumped 15 -> 18 (16 published, 17 is #522's): cvcgl_wasm_app(), the WebGL state shim, FrameYield. - cvc_revision: 18 + # Bumped 18 -> 19 (16 went out as a native publish of master without this, 18 is master's wasm + # app contract above, and the 17 set aside for this change sits below it): the fast-draw + # low-memory mapper (LowMemoryPolyDataMapper: skips empty cell-type agents and re-binds the + # cached shader program; ~2/3 fewer GL calls per frame on WebGL2 with byte-identical pixels; + # CVCGL_LOWMEM_DRAW / setDrawPath for A/B), now the base of StreamingLowMemoryPolyDataMapper; + # late shader replacements reach the GPU on the low-memory path. License metadata now + # LGPL-2.1-only AND BSD-3-Clause (VTK 9.5.0-derived code). + cvc_revision: 19 maintainer: "cvcpkg group" maintainer_email: "info@cvcpkg.org" maintainer_url: "https://cvcpkg.org" description: "cvcGL — the Qt-free VTK scene-graph library extracted from volrover3: a reactive SceneGraph of GraphicsNode/GeometryNode/VolumeNode over cvc::state that renders cvc::geometry / cvc::volume via VTK. Ships libcvcGL.a + headers + a cmake package config; provides the cvcGL cmake package (cvc::cvcGL target). Consumed by pycvc-gl and other non-Qt scene consumers." homepage: https://github.com/transfix/libcvc - license: LGPL-2.1-only + # BSD-3-Clause: libcvcGL carries code derived from VTK 9.5.0 + # (LowMemoryPolyDataMapper.cpp); THIRD_PARTY_NOTICES.md ships in share/doc/cvcGL. + license: "LGPL-2.1-only AND BSD-3-Clause" tags: [scientific, volumetric, scene, vtk, graphics] # cvcGL is its OWN package, separate from libcvc: a plain libcvc bundle is a diff --git a/cvcpkg/recipes/libcvc/recipe.yaml b/cvcpkg/recipes/libcvc/recipe.yaml index c580d2e9..5fc171b3 100644 --- a/cvcpkg/recipes/libcvc/recipe.yaml +++ b/cvcpkg/recipes/libcvc/recipe.yaml @@ -134,7 +134,12 @@ recipe: # caster-aware shadow baking, the cheaper bounds walk). ABI: SceneGraph, GraphicsNode and # NullGraphicNode layouts changed (#512 and the frame-cost fixes) with the SOVERSION # unchanged, so dependents must rebuild. cvcgl bumps in lockstep (14 -> 15). - cvc_revision: 15 + # Bumped 15 -> 17 (16 went out as a native publish of master without this): + # lockstep with cvcgl's fast-draw low-memory mapper + # (LowMemoryPolyDataMapper, ~2/3 fewer GL calls per frame on WebGL2) and the expanded + # THIRD_PARTY_NOTICES.md. libcvc's own sources are unchanged by it; the bump keeps the pair's + # revisions aligned for consumers that pin both. + cvc_revision: 17 maintainer: "cvcpkg group" maintainer_email: "info@cvcpkg.org" maintainer_url: "https://cvcpkg.org" diff --git a/inc/cvc/gl/LowMemoryPolyDataMapper.h b/inc/cvc/gl/LowMemoryPolyDataMapper.h new file mode 100644 index 00000000..899c7a4e --- /dev/null +++ b/inc/cvc/gl/LowMemoryPolyDataMapper.h @@ -0,0 +1,220 @@ +/* + Copyright 2026 The University of Texas at Austin + + This file is part of libcvc. + + libcvc is free software; you can redistribute it and/or + modify it under the terms of the GNU Lesser General Public + License version 2.1 as published by the Free Software Foundation. + + libcvc is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU + Lesser General Public License for more details. + + You should have received a copy of the GNU Lesser General Public + License along with this library; if not, write to the Free Software + Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA +*/ + +// LowMemoryPolyDataMapper -- VTK 9.5.0's low-memory (GLES3/WebGL2) poly-data +// mapper with two pixel-identical cuts to its per-draw cost. +// +// On GLES3/WebGL2 VTK's object factory hands every vtkPolyDataMapper::New() a +// vtkOpenGLLowMemoryPolyDataMapper. Its draw runs four cell-type "agents" in a +// row (verts, lines, polys, strips), and each one does the full shader setup +// whether or not it has anything to draw: every array texture re-bound, every +// camera, material, shadow and agent uniform re-sent, the VAO bound and +// unbound. A typical mesh has one cell type, so three quarters of that is +// thrown away -- about 110 of the ~160 GL calls a lit mesh costs per draw. On +// top of that, every draw re-resolves its shader program through the shader +// cache: five source copies, re-substitution and an MD5 of ~16 KB, to find the +// program it already has. +// +// This subclass: +// 1. skips the cell types that draw nothing (a cell type draws iff its first +// cell group can render, exactly the condition VTK's agent tests), and +// 2. re-binds the program it resolved last time while the shader has not been +// rebuilt since (same ShaderBuildTimeStamp, same render window shader cache) +// instead of looking it up again. +// Nothing else changes: the drawn cell type gets the same uniforms, textures and +// draw call in the same order, so frames are byte-identical (the cvcgl_lowmem_ +// fastdraw test pins that, and pins the replica against VTK call-for-call). +// +// The cell-type agents are VTK private classes (headers not installed, symbols +// hidden), so the drawn agent's PreDraw/Draw/PostDraw is replicated here from +// VTK 9.5.0 (BSD-3-Clause; the notice is in LowMemoryPolyDataMapper.cpp and +// THIRD_PARTY_NOTICES.md). Picking, vertex visibility and a coincident-offset +// layout where skipping a cell type would change the depth offset a drawn one +// sees all take the stock (all four agents) draw. +// +// The replica is active (replicaActive()) only against an UNPATCHED VTK 9.5.0. +// The version macros alone cannot tell a patched 9.5.0 from a pristine one, so: +// * Contract for the libcvc-deps VTK recipe: any patch that touches the +// Rendering/OpenGL2 low-memory mapper, its cell-type agents, +// vtkDrawTexturedElements or the GLSL mods must also add +// #define VTK_CVC_LOWMEM_PATCHLEVEL // n >= 1 +// to the installed vtkOpenGLLowMemoryPolyDataMapper.h. The replica turns +// itself off whenever that macro is defined (every DrawPath is then the +// stock draw) until it is re-validated against the patched code and taught +// that patch level. +// * At configure time cvcGL also hashes the installed +// vtkOpenGLLowMemoryPolyDataMapper.h, vtkDrawTexturedElements.h and +// vtkGLSLModCoincidentTopology.h against pristine 9.5.0 and builds the +// replica off (CVC_GL_LOWMEM_REPLICA_DISABLED) if any of them differs. +// cvcgl_lowmem_fastdraw pins the replica against the stock draw call for call, +// so it is the gate for any VTK recipe change. +// +// VTK v9.7.0 (tagged 2026-08-14) is NOT a drop-in replacement. It carries +// upstream f59c2d0b297 (vtkDrawTexturedElements re-binds its cached program +// unless GetShader() was called -- this class's cut 2) and 15ed6edc478 +// (per-program uniform-location and last-value caches in vtkShaderProgram, the +// GLSL mods and the low-memory mapper), but it still runs all four cell-type +// agents per draw. A VTK upgrade switches the replica off (version check), so +// the empty-agent skip (cut 1) must be re-ported against 9.7's agents or that +// saving is lost. 9.7.0 also contains d1cde2af1a8, which makes every +// vtkOpenGLState::vtkglBlitFramebuffer issue two glCheckFramebufferStatus calls +// first -- at a cvcGL frame's two blits, 4 synchronous round trips per frame on +// WebGL; revert or avoid it in the wasm VTK build. +// +// The shader-cache skip (cut 2) relies on one invariant: this->Shaders (the +// vtkShader sources vtkDrawTexturedElements resolves its program from) changes +// only inside UpdateShaders, and RenderPieceStart always follows UpdateShaders +// with ShaderBuildTimeStamp.Modified(). vtkDrawTexturedElements::GetShader() is +// public and hands out a mutable vtkShader, so this class hides it with a +// version that drops the cached program first (as upstream f59c2d0b297 does); +// anything that edits shader sources by another route must call +// invalidateProgramCache(). Nothing in cvcGL edits them outside +// UpdateShaders today. +// +// For every VTK version, RenderPieceStart also rebuilds the shader program when +// the actor's shader property (replacements, custom uniform declarations) +// changed after the program was built: VTK's low-memory mapper (9.5 through +// 9.7) ignores the shader property once its program exists, so a +// GeometryNode::add*ShaderReplacement made after the first draw would never +// reach the GPU on WebGL. +// +// It also zero-initialises the shift/scale members VTK 9.5.0 leaves +// uninitialised (with DISABLE_SHIFT_SCALE nothing ever assigns them, so a +// garbage "in use" byte made the mapper upload a shift/scale copy of the +// positions). +#ifndef CVC_GL_LOW_MEMORY_POLY_DATA_MAPPER_H +#define CVC_GL_LOW_MEMORY_POLY_DATA_MAPPER_H + +#include +#include +#include +#include +#include +#include + +namespace cvc { +namespace gl { + +class LowMemoryPolyDataMapper : public vtkOpenGLLowMemoryPolyDataMapper { +public: + static LowMemoryPolyDataMapper *New(); + vtkTypeMacro(LowMemoryPolyDataMapper, vtkOpenGLLowMemoryPolyDataMapper); + + // How every LowMemoryPolyDataMapper in the process draws. The initial value + // comes from CVCGL_LOWMEM_DRAW (stock | all | fast, default fast); the other + // two paths exist for A/B measurement and for the tests: + // Stock VTK's own RenderPieceDraw (all four agents, full lookup). + // AllCellTypes the replica with nothing skipped -- issues the same GL calls + // as Stock, call for call. + // Fast the replica, empty cell types skipped, program re-bound. + enum class DrawPath : int { Stock = 0, AllCellTypes = 1, Fast = 2 }; + static void setDrawPath(DrawPath path); + static DrawPath drawPath(); + + // True when this build carries the VTK 9.5.0 replica (an unpatched 9.5.0, see + // above). False means every path is the stock draw. + static bool replicaActive(); + + // Per-mapper counters, read and reset on the render thread. + struct Stats { + std::uint64_t draws = 0; // RenderPieceDraw calls + std::uint64_t stockDraws = 0; // of which drew through VTK's own path + std::uint64_t cellTypesSkipped = 0; // agents not run because they draw nothing + std::uint64_t coincidentFallbacks = 0; // draws that ran all four for the depth offset + std::uint64_t programLookups = 0; // full shader-cache lookups (copy + MD5) + std::uint64_t programLookupsSkipped = 0; // cached program re-bound instead + }; + const Stats &stats() const { return m_stats; } + void resetStats() { m_stats = Stats{}; } + + // Test instrumentation: when set, the replica reports where each stage of a + // draw begins and ends (the cvcgl_lowmem_fastdraw GL trace interleaves these + // with the GL calls to cut a trace into draws and cell-type blocks). Null -- + // the default -- costs one relaxed load per stage. Stock draws report nothing. + enum class DrawStage : int { + DrawBegin, // arg: 1 if this draw skips its empty cell types (Fast only) + LookupBegin, // full shader-cache lookup + LookupEnd, + RebindBegin, // cached program re-bound instead + RebindEnd, + AgentBegin, // arg: cell type 0-3 + AgentEnd, // arg: cell type 0-3 + AgentSkipped, + DrawEnd + }; + using StageObserver = void (*)(DrawStage stage, int arg); + static void setStageObserver(StageObserver observer); + + // Drop the program the last lookup resolved, so the next draw looks it up in + // the shader cache again. For anything that edits this->Shaders outside + // UpdateShaders (see the invariant above). + void invalidateProgramCache(); + // vtkDrawTexturedElements::GetShader (public, non-virtual) hands out a + // mutable source; this hides it so a caller holding this class also drops + // the cached program. + vtkShader *GetShader(vtkShader::Type shaderType); + + void RenderPieceStart(vtkRenderer *ren, vtkActor *act) override; + void RenderPieceDraw(vtkRenderer *ren, vtkActor *act) override; + void ReleaseGraphicsResources(vtkWindow *win) override; + +protected: + LowMemoryPolyDataMapper(); + ~LowMemoryPolyDataMapper() override = default; + + // Bind this draw's shader program: the full lookup, or -- when allowSkip and + // nothing it depends on changed -- the program the last lookup returned. + void readyProgram(vtkRenderer *ren, bool allowSkip); + // Whether cell type t (0 verts, 1 lines, 2 polys, 3 strips) draws anything. + bool renderable(int t) const; + // Whether skipping the non-drawing cell types leaves every drawn one, and the + // program afterwards, with the depth offset the stock draw would. + bool coincidentSkipSafe(vtkActor *act) const; + // VTK 9.5.0's vtkOpenGLLowMemoryCellTypeAgent::PreDraw / Draw for cell type t. + void agentPreDraw(int t, vtkRenderer *ren, vtkActor *act); + void agentDraw(int t, vtkRenderer *ren, vtkActor *act); + + vtkWeakPointer m_cache; // cache the program was last resolved from + vtkMTimeType m_builtStamp = 0; // ShaderBuildTimeStamp at that lookup + Stats m_stats; + +private: + LowMemoryPolyDataMapper(const LowMemoryPolyDataMapper &) = delete; + void operator=(const LowMemoryPolyDataMapper &) = delete; +}; + +// Which mapper newPolyDataMapper() returns. The initial value comes from the +// CVCGL_LOWMEM_MAPPER environment variable (auto | force | off, default auto): +// Auto LowMemoryPolyDataMapper wherever VTK's factory would hand out a +// vtkOpenGLLowMemoryPolyDataMapper (GLES3/WebGL2), the factory's mapper +// everywhere else (vtkOpenGLPolyDataMapper on desktop GL). +// Force LowMemoryPolyDataMapper everywhere, desktop GL included -- for native +// tests and GL-call measurement of the WebGL path. +// Off the factory's mapper everywhere. +enum class LowMemoryMapperPolicy : int { Auto = 0, Force = 1, Off = 2 }; +void setLowMemoryMapperPolicy(LowMemoryMapperPolicy policy); +LowMemoryMapperPolicy lowMemoryMapperPolicy(); + +// The poly-data mapper for a cvcGL node, chosen by the policy above. +vtkSmartPointer newPolyDataMapper(); + +} // namespace gl +} // namespace cvc + +#endif // CVC_GL_LOW_MEMORY_POLY_DATA_MAPPER_H diff --git a/inc/cvc/gl/StreamingMappers.h b/inc/cvc/gl/StreamingMappers.h index f26273a4..e600bab5 100644 --- a/inc/cvc/gl/StreamingMappers.h +++ b/inc/cvc/gl/StreamingMappers.h @@ -60,9 +60,9 @@ #include #include +#include #include // StreamingMapperKind #include -#include #include #include @@ -139,10 +139,14 @@ class StreamingPolyDataMapper : public vtkOpenGLPolyDataMapper { void operator=(const StreamingPolyDataMapper &) = delete; }; -class StreamingLowMemoryPolyDataMapper : public vtkOpenGLLowMemoryPolyDataMapper { +// On top of cvc::gl::LowMemoryPolyDataMapper, so a streaming overlay also gets +// its per-draw cuts (empty cell types skipped, cached program re-bound) and its +// shift/scale initialisation; the draw range below narrows cell group 0 before +// that draw runs. +class StreamingLowMemoryPolyDataMapper : public LowMemoryPolyDataMapper { public: static StreamingLowMemoryPolyDataMapper *New(); - vtkTypeMacro(StreamingLowMemoryPolyDataMapper, vtkOpenGLLowMemoryPolyDataMapper); + vtkTypeMacro(StreamingLowMemoryPolyDataMapper, LowMemoryPolyDataMapper); StreamingMapperCore &core() { return Core; } const StreamingMapperCore &core() const { return Core; } @@ -151,7 +155,7 @@ class StreamingLowMemoryPolyDataMapper : public vtkOpenGLLowMemoryPolyDataMapper void RenderPieceDraw(vtkRenderer *ren, vtkActor *act) override; protected: - StreamingLowMemoryPolyDataMapper(); + StreamingLowMemoryPolyDataMapper() = default; ~StreamingLowMemoryPolyDataMapper() override = default; void ComputeBounds() override; diff --git a/src/cvcGL/BBoxNode.cpp b/src/cvcGL/BBoxNode.cpp index fbdccbd9..222043ce 100644 --- a/src/cvcGL/BBoxNode.cpp +++ b/src/cvcGL/BBoxNode.cpp @@ -1,5 +1,6 @@ #include #include +#include #include #include #include @@ -21,10 +22,9 @@ namespace cvc { namespace gl { BBoxNode::BBoxNode() - : m_actor(vtkSmartPointer::New()), - m_mapper(vtkSmartPointer::New()), m_bbox(-1.0, -1.0, -1.0, 1.0, 1.0, 1.0), - m_transform(vtkSmartPointer::New()), m_coordinatesVisible(true), - m_coordinateLabelFontSize(12), m_renderer(nullptr) { + : m_actor(vtkSmartPointer::New()), m_mapper(newPolyDataMapper()), + m_bbox(-1.0, -1.0, -1.0, 1.0, 1.0, 1.0), m_transform(vtkSmartPointer::New()), + m_coordinatesVisible(true), m_coordinateLabelFontSize(12), m_renderer(nullptr) { m_transform->Identity(); m_actor->SetMapper(m_mapper); diff --git a/src/cvcGL/CMakeLists.txt b/src/cvcGL/CMakeLists.txt index b408fa8d..6bba0b54 100644 --- a/src/cvcGL/CMakeLists.txt +++ b/src/cvcGL/CMakeLists.txt @@ -63,6 +63,53 @@ target_include_directories(cvcGL PUBLIC ) target_link_libraries(cvcGL PUBLIC cvc::cvc ${VTK_LIBRARIES}) +# LowMemoryPolyDataMapper carries a replica of VTK 9.5.0's private low-memory +# cell-type agents, valid only against an UNPATCHED 9.5.0 (the contract is in +# inc/cvc/gl/LowMemoryPolyDataMapper.h). The version number cannot tell a +# libcvc-deps recipe patch from pristine 9.5.0, so: a patch to that code must +# define VTK_CVC_LOWMEM_PATCHLEVEL in the installed +# vtkOpenGLLowMemoryPolyDataMapper.h (checked in the source), and here the +# installed headers the replica builds on are hashed against v9.5.0's (line +# endings normalised) -- any difference builds the replica off. +if(VTK_VERSION VERSION_EQUAL "9.5.0") + set(_cvc_lowmem_pristine + "vtkOpenGLLowMemoryPolyDataMapper.h=a4100d6d90f72861ce4eacd6e29506bbab727fd28fe60d6fdb319d2695d61c1a" + "vtkDrawTexturedElements.h=07df517fba54da88b2ac1b1ad4ca847aea7b2eda72422bbeeafe0f8d20a520af" + "vtkGLSLModCoincidentTopology.h=708637ca8dc0695a60c1bb71269220b726cfe7f37af54c214430d4f8f2b42033") + set(_cvc_lowmem_inc "${VTK_PREFIX_PATH}/include/vtk-${VTK_VERSION_MAJOR}.${VTK_VERSION_MINOR}") + set(_cvc_lowmem_changed "") + set(_cvc_lowmem_missing "") + foreach(_entry IN LISTS _cvc_lowmem_pristine) + string(REPLACE "=" ";" _pair "${_entry}") + list(GET _pair 0 _hdr) + list(GET _pair 1 _want) + if(EXISTS "${_cvc_lowmem_inc}/${_hdr}") + # A VTK re-installed into the same prefix (a patched recipe) re-runs the + # configure, and with it this check, on the next build. + set_property(DIRECTORY APPEND PROPERTY CMAKE_CONFIGURE_DEPENDS "${_cvc_lowmem_inc}/${_hdr}") + file(READ "${_cvc_lowmem_inc}/${_hdr}" _text) + string(REPLACE "\r\n" "\n" _text "${_text}") + string(SHA256 _have "${_text}") + if(NOT _have STREQUAL _want) + list(APPEND _cvc_lowmem_changed "${_hdr}") + endif() + else() + list(APPEND _cvc_lowmem_missing "${_hdr}") + endif() + endforeach() + if(_cvc_lowmem_changed) + target_compile_definitions(cvcGL PRIVATE CVC_GL_LOWMEM_REPLICA_DISABLED=1) + message(WARNING "cvcGL: VTK 9.5.0 headers differ from upstream v9.5.0 (${_cvc_lowmem_changed}); " + "LowMemoryPolyDataMapper's replica is OFF (stock draw). Re-validate it with " + "cvcgl_lowmem_fastdraw against the patched VTK before re-enabling.") + elseif(_cvc_lowmem_missing) + message(STATUS "cvcGL: could not find ${_cvc_lowmem_missing} under ${_cvc_lowmem_inc}; " + "LowMemoryPolyDataMapper relies on VTK_CVC_LOWMEM_PATCHLEVEL alone") + else() + message(STATUS "cvcGL: VTK 9.5.0 low-memory mapper headers are pristine; replica enabled") + endif() +endif() + # --------------------------------------------------------------------------- # Dear ImGui overlay (optional): real menus/panels/widgets drawn inside the VTK # render, native AND WebAssembly. The cvcpkg imgui bundle ships no CMake config, @@ -299,6 +346,11 @@ install(FILES # outside Emscripten): the state shim cvcgl_wasm_app() links. install(FILES "${CMAKE_CURRENT_SOURCE_DIR}/wasm/webgl_state_shadow.js" DESTINATION ${CVCGL_WASM_DATADIR}) +# libcvcGL carries code derived from VTK 9.5.0 (BSD-3-Clause, in +# LowMemoryPolyDataMapper.cpp), whose notice a binary redistribution must +# reproduce in its documentation: the standalone cvcGL package ships it too. +install(FILES "${CMAKE_CURRENT_SOURCE_DIR}/../../THIRD_PARTY_NOTICES.md" + DESTINATION ${CMAKE_INSTALL_DOCDIR}) # ── functional smoke test ── enable_testing() @@ -703,6 +755,36 @@ if(NOT EMSCRIPTEN) endif() endif() +# LowMemoryPolyDataMapper (the GLES3/WebGL2 mapper with empty cell types and the +# per-draw shader-cache lookup taken out of its draw), forced on natively. Records +# an ordered trace of every GL call at the driver -- entry point, bound program / +# texture unit, argument bytes, uniform values (test/gl_call_counter.cpp swaps +# VTK's glad function pointers) -- and asserts the VTK 9.5.0 replica's trace +# equals the stock draw's exactly and Fast's equals it with the empty cell-type +# blocks removed wherever removing them leaves every draw call's uniform state +# unchanged (derived by replaying the trace's glUniform* calls, not from the +# mapper's own decision) and that Fast's draw calls see that state; compares +# RGBA frames byte for byte -- shadows on and off, bake and steady frames, every +# representation / scalar / edge / line kind, the hardware selector, released +# resources, a re-opened window, the offset / picking / vertex-visibility guards, +# and shader replacements made after the first draw. This is the gate for any +# libcvc-deps VTK recipe change. Renders for real; skips where nothing +# rasterises unless CVC_REQUIRE_RENDER=1. CVCGL_LOWMEM_REPORT=1 prints per-draw +# breakdowns; CVCGL_LOWMEM_DUMP= writes the traces of a failing comparison. +# Desktop GL only: the wasm VTK calls GLES directly, not through glad. +# +# cvcgl_lowmem_bench (not a test) prints GL calls (counted pass) and draw-stage / +# frame CPU (timed pass, no GL wrappers) per frame and per draw, Stock vs Fast, +# for a city of single-colour box nodes on a textured ground with overlays, +# shadows on and off. +if(NOT EMSCRIPTEN) + add_executable(cvcgl_lowmem_fastdraw test/cvcgl_lowmem_fastdraw.cpp test/gl_call_counter.cpp) + target_link_libraries(cvcgl_lowmem_fastdraw PRIVATE cvcGL) + add_test(NAME cvcgl_lowmem_fastdraw COMMAND cvcgl_lowmem_fastdraw) + add_executable(cvcgl_lowmem_bench test/cvcgl_lowmem_bench.cpp test/gl_call_counter.cpp) + target_link_libraries(cvcgl_lowmem_bench PRIVATE cvcGL) +endif() + # SdlInput: SDL3 event -> normalized cvc::gl::InputEvent translation (the §4.6 input seam's SDL # source). Pushes synthetic SDL key/mouse events onto the queue and verifies poll() returns them. # Headless (the SDL EVENTS subsystem needs no window). SKIPs (rc 0) in a build without SDL. diff --git a/src/cvcGL/GeometryNode.cpp b/src/cvcGL/GeometryNode.cpp index a4abde8b..050beb6d 100644 --- a/src/cvcGL/GeometryNode.cpp +++ b/src/cvcGL/GeometryNode.cpp @@ -5,6 +5,7 @@ #include #include #include +#include #include #include #include @@ -48,14 +49,13 @@ namespace cvc { namespace gl { GeometryNode::GeometryNode(cvc::app &ctx, const std::string &statePath, const std::string &name) - : GeometryNode(ctx, statePath, name, vtkSmartPointer::New()) {} + : GeometryNode(ctx, statePath, name, newPolyDataMapper()) {} GeometryNode::GeometryNode(cvc::app &ctx, const std::string &statePath, const std::string &name, vtkSmartPointer mapper) : GraphicsNode(ctx, statePath, name), m_hasGeometry(false), m_renderMode(GeometryRenderMode::TRIS), m_useSingleColor(false), - m_actor(vtkSmartPointer::New()), - m_mapper(mapper ? mapper : vtkSmartPointer::New()), + m_actor(vtkSmartPointer::New()), m_mapper(mapper ? mapper : newPolyDataMapper()), m_polyData(vtkSmartPointer::New()), m_textureFlipV(false) { m_mapper->SetInputData(m_polyData); m_actor->SetMapper(m_mapper); diff --git a/src/cvcGL/GridNode.cpp b/src/cvcGL/GridNode.cpp index b51d7536..1ea4cb14 100644 --- a/src/cvcGL/GridNode.cpp +++ b/src/cvcGL/GridNode.cpp @@ -1,6 +1,7 @@ #include #include #include +#include #include #include #include @@ -24,12 +25,10 @@ namespace gl { GridNode::GridNode(cvc::app &ctx, const std::string &statePath, const std::string &name) : GraphicsNode(ctx, statePath, name), m_yzActor(vtkSmartPointer::New()), m_xzActor(vtkSmartPointer::New()), m_xyActor(vtkSmartPointer::New()), - m_yzMapper(vtkSmartPointer::New()), - m_xzMapper(vtkSmartPointer::New()), - m_xyMapper(vtkSmartPointer::New()), - m_bounds(-10.0, -10.0, -10.0, 10.0, 10.0, 10.0), m_divisionsX(64), m_divisionsY(64), - m_divisionsZ(64), m_tickIntervalX(8), m_tickIntervalY(8), m_tickIntervalZ(8), - m_tickLabelFontSize(12), m_yzPlaneVisible(true), m_xzPlaneVisible(true), + m_yzMapper(newPolyDataMapper()), m_xzMapper(newPolyDataMapper()), + m_xyMapper(newPolyDataMapper()), m_bounds(-10.0, -10.0, -10.0, 10.0, 10.0, 10.0), + m_divisionsX(64), m_divisionsY(64), m_divisionsZ(64), m_tickIntervalX(8), m_tickIntervalY(8), + m_tickIntervalZ(8), m_tickLabelFontSize(12), m_yzPlaneVisible(true), m_xzPlaneVisible(true), m_xyPlaneVisible(true), m_renderer(nullptr) { // Initialize default colors m_yzPlaneColor[0] = m_yzPlaneColor[1] = m_yzPlaneColor[2] = 0.5; diff --git a/src/cvcGL/LowMemoryPolyDataMapper.cpp b/src/cvcGL/LowMemoryPolyDataMapper.cpp new file mode 100644 index 00000000..ba676227 --- /dev/null +++ b/src/cvcGL/LowMemoryPolyDataMapper.cpp @@ -0,0 +1,489 @@ +/* + Copyright 2026 The University of Texas at Austin + + This file is part of libcvc. + + libcvc is free software; you can redistribute it and/or + modify it under the terms of the GNU Lesser General Public + License version 2.1 as published by the Free Software Foundation. + + libcvc is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU + Lesser General Public License for more details. + + You should have received a copy of the GNU Lesser General Public + License along with this library; if not, write to the Free Software + Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA +*/ + +/* + Portions of this file -- the regions between "BEGIN VTK 9.5.0-DERIVED" and + "END VTK 9.5.0-DERIVED" below -- are derived from VTK 9.5.0 + (Rendering/OpenGL2: vtkGLSLModCoincidentTopology.cxx, + vtkOpenGLLowMemoryPolyDataMapper.cxx, vtkOpenGLLowMemoryCellTypeAgent.cxx, + vtkOpenGLLowMemoryVerticesAgent.cxx, vtkOpenGLLowMemoryLinesAgent.cxx, + vtkOpenGLLowMemoryPolygonsAgent.cxx) and are used under VTK's license: + + SPDX-FileCopyrightText: Copyright (c) Ken Martin, Will Schroeder, Bill Lorensen + SPDX-License-Identifier: BSD-3-Clause + + Copyright (c) 1993-2015 Ken Martin, Will Schroeder, Bill Lorensen + All rights reserved. + + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions are met: + + * Redistributions of source code must retain the above copyright notice, + this list of conditions and the following disclaimer. + + * Redistributions in binary form must reproduce the above copyright notice, + this list of conditions and the following disclaimer in the documentation + and/or other materials provided with the distribution. + + * Neither name of Ken Martin, Will Schroeder, or Bill Lorensen nor the names + of any contributors may be used to endorse or promote products derived + from this software without specific prior written permission. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS ``AS IS'' + AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE + IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE + ARE DISCLAIMED. IN NO EVENT SHALL THE AUTHORS OR CONTRIBUTORS BE LIABLE FOR + ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL + DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR + SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER + CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, + OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE + OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + + The rest of the file is libcvc's own (LGPL-2.1, above). The notice is also + recorded in THIRD_PARTY_NOTICES.md. +*/ + +// No vtkOpenGL*ErrorMacro anywhere in this file: those macros are inline in the +// including translation unit, so a cvcGL build without NDEBUG would issue a +// glGetError per draw -- a synchronous round trip on WebGL. +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +// The replica is VTK 9.5.0's own agent code, so it is on only for an UNPATCHED +// 9.5.0 (the contract is in the class header): +// VTK_CVC_LOWMEM_PATCHLEVEL defined by any libcvc-deps VTK recipe patch +// to the low-memory / vtkDrawTexturedElements +// code, in the installed +// vtkOpenGLLowMemoryPolyDataMapper.h; +// CVC_GL_LOWMEM_REPLICA_DISABLED set by cvcGL's configure-time check when an +// installed header differs from 9.5.0's. +#if VTK_MAJOR_VERSION == 9 && VTK_MINOR_VERSION == 5 && VTK_BUILD_VERSION == 0 && \ + !defined(VTK_CVC_LOWMEM_PATCHLEVEL) && !defined(CVC_GL_LOWMEM_REPLICA_DISABLED) +#define CVC_GL_LOWMEM_REPLICA 1 +#else +#define CVC_GL_LOWMEM_REPLICA 0 +#endif + +namespace cvc { +namespace gl { + +namespace { + +int drawPathFromEnvironment() { + const char *v = std::getenv("CVCGL_LOWMEM_DRAW"); + if (v && std::strcmp(v, "stock") == 0) + return static_cast(LowMemoryPolyDataMapper::DrawPath::Stock); + if (v && std::strcmp(v, "all") == 0) + return static_cast(LowMemoryPolyDataMapper::DrawPath::AllCellTypes); + return static_cast(LowMemoryPolyDataMapper::DrawPath::Fast); +} + +std::atomic &drawPathStore() { + static std::atomic path{drawPathFromEnvironment()}; + return path; +} + +int policyFromEnvironment() { + const char *v = std::getenv("CVCGL_LOWMEM_MAPPER"); + if (v && std::strcmp(v, "force") == 0) + return static_cast(LowMemoryMapperPolicy::Force); + if (v && std::strcmp(v, "off") == 0) + return static_cast(LowMemoryMapperPolicy::Off); + return static_cast(LowMemoryMapperPolicy::Auto); +} + +std::atomic &policyStore() { + static std::atomic policy{policyFromEnvironment()}; + return policy; +} + +// Constant-initialised: no guard on the per-stage load. +std::atomic g_stageObserver{nullptr}; + +// Does VTK's object factory hand out the low-memory mapper on this build? Asked +// once, on the first node construction (the OpenGL2 factory is registered by +// then: cvcGL's sources carry the module auto-init). +bool factoryIsLowMemory() { + static const bool lowMemory = + vtkSmartPointer::New()->IsA("vtkOpenGLLowMemoryPolyDataMapper") != 0; + return lowMemory; +} + +#if CVC_GL_LOWMEM_REPLICA +// ---- BEGIN VTK 9.5.0-DERIVED ------------------------------------------------ +// Portions derived from VTK 9.5.0, Copyright (c) Ken Martin, Will Schroeder, +// Bill Lorensen; All rights reserved. SPDX-License-Identifier: BSD-3-Clause +// (full notice at the top of this file). coincidentValue() below is +// vtkGLSLModCoincidentTopology::GetCoincidentParameters. +// +// The depth offset vtkGLSLModCoincidentTopology::GetCoincidentParameters (VTK +// 9.5.0) computes for one primitive class, as the floats it would upload. +// slot: 0 points, 1 lines, 2 polygons. `set` false: the mod uploads nothing. +struct CoincidentValue { + bool set = false; + float factor = 0.0f, offset = 0.0f; + bool operator==(const CoincidentValue &o) const { + return set == o.set && (!set || (factor == o.factor && offset == o.offset)); + } + bool operator!=(const CoincidentValue &o) const { return !(*this == o); } +}; + +CoincidentValue coincidentValue(vtkMapper *mapper, vtkProperty *prop, int slot) { + float factor = 0.0f, offset = 0.0f; + if (vtkMapper::GetResolveCoincidentTopology() == VTK_RESOLVE_SHIFT_ZBUFFER) + offset = static_cast(vtkMapper::GetResolveCoincidentTopologyZShift() * 4.0); + if (vtkMapper::GetResolveCoincidentTopology() == VTK_RESOLVE_POLYGON_OFFSET || + (prop->GetEdgeVisibility() && prop->GetRepresentation() == VTK_SURFACE)) { + double f = 0.0, u = 0.0; + if (slot == 0) + mapper->GetCoincidentTopologyPointOffsetParameter(u); + else if (slot == 1) + mapper->GetCoincidentTopologyLineOffsetParameters(f, u); + else + mapper->GetCoincidentTopologyPolygonOffsetParameters(f, u); + factor = static_cast(f); + offset = static_cast(u); + } + CoincidentValue v; + v.set = factor != 0.0f || offset != 0.0f; + v.factor = factor; + v.offset = offset; + return v; +} +// ---- END VTK 9.5.0-DERIVED -------------------------------------------------- +#endif + +} // namespace + +vtkStandardNewMacro(LowMemoryPolyDataMapper); + +LowMemoryPolyDataMapper::LowMemoryPolyDataMapper() { +#if VTK_MAJOR_VERSION == 9 + // VTK 9.5.0 declares these without initialisers (9.5.1 initialises the bool + // only). With DISABLE_SHIFT_SCALE nothing ever assigns them, so whether the + // mapper uploaded a (garbage) shift/scale copy of the positions depended on a + // heap byte. + this->CoordinateShiftAndScaleInUse = false; + this->ShiftValues.fill(0.0); + this->ScaleValues.fill(1.0); +#endif +} + +void LowMemoryPolyDataMapper::setDrawPath(DrawPath path) { + drawPathStore().store(static_cast(path), std::memory_order_relaxed); +} + +LowMemoryPolyDataMapper::DrawPath LowMemoryPolyDataMapper::drawPath() { + return static_cast(drawPathStore().load(std::memory_order_relaxed)); +} + +bool LowMemoryPolyDataMapper::replicaActive() { return CVC_GL_LOWMEM_REPLICA != 0; } + +void LowMemoryPolyDataMapper::setStageObserver(StageObserver observer) { + g_stageObserver.store(observer, std::memory_order_relaxed); +} + +void LowMemoryPolyDataMapper::invalidateProgramCache() { + m_cache = nullptr; + m_builtStamp = 0; +} + +vtkShader *LowMemoryPolyDataMapper::GetShader(vtkShader::Type shaderType) { + // The caller may edit the source: the next draw must resolve it again. + invalidateProgramCache(); + return this->vtkDrawTexturedElements::GetShader(shaderType); +} + +void LowMemoryPolyDataMapper::RenderPieceStart(vtkRenderer *ren, vtkActor *act) { + // VTK's low-memory mapper (9.5 through 9.7) never looks at the actor's shader + // property once its program is built: IsShaderUpToDate ignores it, where the + // classic mapper compares GetShaderMTime(). A replacement added or cleared + // after the first draw (GeometryNode::add*ShaderReplacement / + // clearShaderReplacements), or a custom uniform declared after it, would + // never reach the GPU. Dropping the program makes the base rebuild it + // (UpdateShaders + a new ShaderBuildTimeStamp, so the draw that follows also + // takes the full shader-cache lookup). Uniform VALUE changes do not move + // GetShaderMTime, so they never cost a rebuild. + if (vtkShaderProperty *sp = act->GetShaderProperty()) + if (sp->GetShaderMTime() > this->ShaderBuildTimeStamp.GetMTime()) + this->ShaderProgram = nullptr; + Superclass::RenderPieceStart(ren, act); +} + +void LowMemoryPolyDataMapper::ReleaseGraphicsResources(vtkWindow *win) { + // The cached program belongs to that window's context; look it up afresh. + invalidateProgramCache(); + Superclass::ReleaseGraphicsResources(win); +} + +#if !CVC_GL_LOWMEM_REPLICA + +void LowMemoryPolyDataMapper::RenderPieceDraw(vtkRenderer *ren, vtkActor *act) { + ++m_stats.draws; + ++m_stats.stockDraws; + Superclass::RenderPieceDraw(ren, act); +} +void LowMemoryPolyDataMapper::readyProgram(vtkRenderer *ren, bool) { + this->vtkDrawTexturedElements::ReadyShaderProgram(ren); +} +bool LowMemoryPolyDataMapper::renderable(int) const { return true; } +bool LowMemoryPolyDataMapper::coincidentSkipSafe(vtkActor *) const { return false; } +void LowMemoryPolyDataMapper::agentPreDraw(int, vtkRenderer *, vtkActor *) {} +void LowMemoryPolyDataMapper::agentDraw(int, vtkRenderer *, vtkActor *) {} + +#else + +namespace { +inline void observeStage(LowMemoryPolyDataMapper::StageObserver observe, + LowMemoryPolyDataMapper::DrawStage stage, int arg = 0) { + if (observe) + observe(stage, arg); +} +} // namespace + +// ---- BEGIN VTK 9.5.0-DERIVED ------------------------------------------------ +// Portions derived from VTK 9.5.0, Copyright (c) Ken Martin, Will Schroeder, +// Bill Lorensen; All rights reserved. SPDX-License-Identifier: BSD-3-Clause +// (full notice at the top of this file). RenderPieceDraw's agent loop is +// vtkOpenGLLowMemoryPolyDataMapper::RenderPieceDraw; renderable(), agentPreDraw() +// and agentDraw() are vtkOpenGLLowMemoryCellTypeAgent::PreDraw / Draw and the +// vtkOpenGLLowMemory{Vertices,Lines,Polygons}Agent::PreDrawInternal they call. +// readyProgram() and coincidentSkipSafe() are libcvc's own. +void LowMemoryPolyDataMapper::RenderPieceDraw(vtkRenderer *ren, vtkActor *act) { + ++m_stats.draws; + const DrawPath path = drawPath(); + // Selection (picking state, point-picking sizes) and the vertex-visibility + // pass stay on VTK's own path; the replica covers the plain draw only. + if (path == DrawPath::Stock || ren->GetSelector() || act->GetProperty()->GetVertexVisibility()) { + ++m_stats.stockDraws; + Superclass::RenderPieceDraw(ren, act); + return; + } + const bool fast = path == DrawPath::Fast; + const StageObserver observe = g_stageObserver.load(std::memory_order_relaxed); + const bool skip = fast && coincidentSkipSafe(act); // CPU only (no GL) + observeStage(observe, DrawStage::DrawBegin, skip ? 1 : 0); + readyProgram(ren, fast); + // Uniforms common to every cell type; also fires UpdateShaderEvent, which + // GeometryNode's custom shader textures hang off. + this->SetShaderParameters(ren, act); + if (!this->ShaderProgram) { + observeStage(observe, DrawStage::DrawEnd); + return; // compile/link failure: VTK's agents would dereference null here + } + if (fast && !skip) + ++m_stats.coincidentFallbacks; + for (int t = 0; t < 4; ++t) { // verts, lines, polys, strips: VTK's order + if (skip && !renderable(t)) { + ++m_stats.cellTypesSkipped; + observeStage(observe, DrawStage::AgentSkipped, t); + continue; + } + observeStage(observe, DrawStage::AgentBegin, t); + agentPreDraw(t, ren, act); + agentDraw(t, ren, act); + // vtkOpenGLLowMemoryCellTypeAgent::PostDraw; every 9.5.0 PostDrawInternal is empty. + this->vtkDrawTexturedElements::PostDraw(ren, act, this); + observeStage(observe, DrawStage::AgentEnd, t); + } + observeStage(observe, DrawStage::DrawEnd); +} + +void LowMemoryPolyDataMapper::readyProgram(vtkRenderer *ren, bool allowSkip) { + const StageObserver observe = g_stageObserver.load(std::memory_order_relaxed); + auto *win = vtkOpenGLRenderWindow::SafeDownCast(ren->GetRenderWindow()); + vtkOpenGLShaderCache *cache = win ? win->GetShaderCache() : nullptr; + const vtkMTimeType stamp = this->ShaderBuildTimeStamp.GetMTime(); + // this->Shaders only changes in UpdateShaders, which RenderPieceStart follows + // with ShaderBuildTimeStamp.Modified() (GetShader() and + // invalidateProgramCache() clear m_cache for any other route); while the + // stamp and the cache are the ones the last lookup saw, that lookup would + // find the same program again. ReadyShaderProgram(program) still compiles it + // if its context was released and binds it exactly as the lookup's tail + // would, so the two issue the same GL calls. + if (allowSkip && cache && this->ShaderProgram && cache == m_cache.GetPointer() && + stamp == m_builtStamp && this->ElementType != AbstractPatches) { + observeStage(observe, DrawStage::RebindBegin); + this->ShaderProgram = cache->ReadyShaderProgram(this->ShaderProgram.GetPointer()); + observeStage(observe, DrawStage::RebindEnd); + ++m_stats.programLookupsSkipped; + return; + } + observeStage(observe, DrawStage::LookupBegin); + this->vtkDrawTexturedElements::ReadyShaderProgram(ren); // copy, substitute, MD5, find, bind + observeStage(observe, DrawStage::LookupEnd); + ++m_stats.programLookups; + m_cache = cache; + m_builtStamp = stamp; +} + +bool LowMemoryPolyDataMapper::renderable(int t) const { + // vtkOpenGLLowMemoryCellTypeAgent::Draw draws cell group 0 iff it CanRender. + const auto &groups = this->Primitives[t].CellGroups; + return !groups.empty() && groups[0].CanRender; +} + +bool LowMemoryPolyDataMapper::coincidentSkipSafe(vtkActor *act) const { + // vtkGLSLModCoincidentTopology sets cOffset/cFactor in every PreDraw, per + // primitive class, and only when the value is non-zero; otherwise the program + // keeps whatever was set last. So a cell type whose own offset is zero sees + // what an earlier cell type -- possibly one with nothing to draw -- left + // behind, and the last one leaves a value for the next draw of that program. + // Skipping is safe when every drawn type, and the program afterwards, ends up + // with the same value either way. Conservative: the mod declares the offset + // uniforms only if the shader was BUILT under a non-zero offset, which this + // does not ask -- with the global mode switched on after the build (an offset + // VTK then never applies) it falls back for nothing. + vtkProperty *prop = act->GetProperty(); + const int mode = vtkMapper::GetResolveCoincidentTopology(); + if (mode != VTK_RESOLVE_POLYGON_OFFSET && mode != VTK_RESOLVE_SHIFT_ZBUFFER && + !(prop->GetEdgeVisibility() && prop->GetRepresentation() == VTK_SURFACE)) + return true; // no offset uniform is ever set (VTK's default) + auto *self = const_cast(this); + const bool points = prop->GetRepresentation() == VTK_POINTS; + CoincidentValue stock, skipped; // unset: unchanged since before this draw + for (int t = 0; t < 4; ++t) { + const int slot = points ? 0 : (t == 0 ? 0 : (t == 1 ? 1 : 2)); + const CoincidentValue v = coincidentValue(self, prop, slot); + if (v.set) + stock = v; + if (renderable(t)) { + if (v.set) + skipped = v; + if (stock != skipped) + return false; + } + } + return stock == skipped; +} + +void LowMemoryPolyDataMapper::agentPreDraw(int t, vtkRenderer *ren, vtkActor *act) { + // vtkOpenGLLowMemory{Vertices,Lines,Polygons}Agent::PreDrawInternal (the + // strips agent is the polygons agent) ... + vtkProperty *prop = act->GetProperty(); + int pointsPerPrimitive = 3; + if (t == 0) { + pointsPerPrimitive = 1; + this->ElementType = vtkDrawTexturedElements::ElementShape::Point; + this->NumberOfInstances = 1; + this->ShaderProgram->SetUniformi("cellType", VTK_VERTEX); + } else if (t == 1) { + pointsPerPrimitive = 2; + this->ElementType = vtkDrawTexturedElements::ElementShape::Line; + if (prop->GetLineWidth() > 1) + this->NumberOfInstances = 2 * vtkMath::Ceil(prop->GetLineWidth()); + else + this->NumberOfInstances = 1; + this->ShaderProgram->SetUniformi("cellType", VTK_LINE); + } else { + this->ElementType = vtkDrawTexturedElements::ElementShape::Triangle; + this->NumberOfInstances = 1; + this->ShaderProgram->SetUniformi("cellType", VTK_TRIANGLE); + } + // ... then vtkOpenGLLowMemoryCellTypeAgent::PreDraw (never in the vertex + // visibility pass, which takes the stock draw). + if (prop->GetRepresentation() == VTK_POINTS) + this->ElementType = vtkDrawTexturedElements::ElementShape::Point; + bool needLighting = false; + if (prop->GetRepresentation() == VTK_POINTS) { + needLighting = prop->GetInterpolation() != VTK_FLAT && this->HasPointNormals; + } else { + const bool isTrisOrStrips = pointsPerPrimitive >= 3; + needLighting = isTrisOrStrips || (!isTrisOrStrips && prop->GetInterpolation() != VTK_FLAT && + this->HasPointNormals); + } + this->ShaderProgram->SetUniformi("enable_lights", needLighting); + this->ShaderProgram->SetUniformi("vertex_pass", false); + switch (this->ElementType) { + case vtkDrawTexturedElements::ElementShape::Point: + this->ShaderProgram->SetUniformi("primitiveSize", 1); + break; + case vtkDrawTexturedElements::ElementShape::Line: + this->ShaderProgram->SetUniformi("primitiveSize", 2); + break; + case vtkDrawTexturedElements::ElementShape::Triangle: + default: + this->ShaderProgram->SetUniformi("primitiveSize", 3); + break; + } + // PointPicking is only ever set with a selector, which takes the stock draw. + this->ShaderProgram->SetUniformf("pointSize", prop->GetPointSize()); + this->vtkDrawTexturedElements::PreDraw(ren, act, this); +} + +void LowMemoryPolyDataMapper::agentDraw(int t, vtkRenderer *ren, vtkActor *act) { + // vtkOpenGLLowMemoryCellTypeAgent::Draw, cell group 0. + const auto &groups = this->Primitives[t].CellGroups; + if (groups.empty() || !groups[0].CanRender) + return; + const auto &group = groups[0]; + const auto &offsets = group.Offsets; + this->FirstVertexId = offsets.VertexIdOffset; + this->NumberOfElements = group.NumberOfElements; + // when rendering vertices, increase number of elements and draw 1 instance. + if (act->GetProperty()->GetRepresentation() == VTK_POINTS) { + this->NumberOfElements *= (t == 0 ? 1 : (t == 1 ? 2 : 3)); + this->NumberOfInstances = 1; + } + vtkShaderProgram *program = this->ShaderProgram; + program->SetUniformi("cellIdOffset", offsets.CellIdOffset); + program->SetUniformi("vertexIdOffset", offsets.VertexIdOffset); + program->SetUniformi("edgeValueBufferOffset", offsets.EdgeValueBufferOffset); + program->SetUniformi("pointIdOffset", offsets.PointIdOffset); + program->SetUniformi("primitiveIdOffset", offsets.PrimitiveIdOffset); + program->SetUniformi("usesCellMap", group.UsesCellMapBuffer); + program->SetUniformi("usesEdgeValues", group.UsesEdgeValueBuffer); + this->vtkDrawTexturedElements::DrawInstancedElementsImpl(ren, act, this); +} +// ---- END VTK 9.5.0-DERIVED -------------------------------------------------- + +#endif // CVC_GL_LOWMEM_REPLICA + +void setLowMemoryMapperPolicy(LowMemoryMapperPolicy policy) { + policyStore().store(static_cast(policy), std::memory_order_relaxed); +} + +LowMemoryMapperPolicy lowMemoryMapperPolicy() { + return static_cast(policyStore().load(std::memory_order_relaxed)); +} + +vtkSmartPointer newPolyDataMapper() { + const LowMemoryMapperPolicy policy = lowMemoryMapperPolicy(); + if (policy == LowMemoryMapperPolicy::Force || + (policy == LowMemoryMapperPolicy::Auto && factoryIsLowMemory())) + return vtkSmartPointer::New(); + return vtkSmartPointer::New(); +} + +} // namespace gl +} // namespace cvc diff --git a/src/cvcGL/StreamingMappers.cpp b/src/cvcGL/StreamingMappers.cpp index dfcefe4c..dce607fd 100644 --- a/src/cvcGL/StreamingMappers.cpp +++ b/src/cvcGL/StreamingMappers.cpp @@ -36,7 +36,6 @@ #include #include #include -#include #include #include #include @@ -190,19 +189,12 @@ void StreamingPolyDataMapper::RenderPieceDraw(vtkRenderer *ren, vtkActor *act) { } // ─────────────────────────────── low-memory mapper ────────────────────────── +// The base, LowMemoryPolyDataMapper, initialises the shift/scale members VTK +// 9.5.0 leaves uninitialised: with garbage there the mapper uploaded a +// shift/scale COPY of the positions, which also defeats streaming (the uploaded +// array must be the input's own). vtkStandardNewMacro(StreamingLowMemoryPolyDataMapper); -StreamingLowMemoryPolyDataMapper::StreamingLowMemoryPolyDataMapper() { - // VTK 9.5.0 never initialises these (9.5.2 fixes the bool only). With - // DISABLE_SHIFT_SCALE nothing ever assigns them, so the mapper applied a - // GARBAGE shift/scale copy of the positions whenever the heap byte under the - // bool happened to be non-zero -- nondeterministic, and it also defeats the - // streaming path (the uploaded array is then a copy, not the input's). - this->CoordinateShiftAndScaleInUse = false; - this->ShiftValues.fill(0.0); - this->ScaleValues.fill(1.0); -} - void StreamingLowMemoryPolyDataMapper::ComputeBounds() { if (Core.beforeBounds) Core.beforeBounds(); @@ -216,14 +208,9 @@ void StreamingLowMemoryPolyDataMapper::ComputeBounds() { void StreamingLowMemoryPolyDataMapper::RenderPieceStart(vtkRenderer *ren, vtkActor *act) { if (Core.beforeDraw) Core.beforeDraw(ren); - // VTK 9.5's low-memory mapper never looks at the actor's shader property once - // its program is built: IsShaderUpToDate ignores it, where the classic mapper - // compares GetShaderMTime(). A replacement added or cleared after the first - // draw (GeometryNode::add*ShaderReplacement / clearShaderReplacements) would - // never reach the GPU. Dropping the program makes the base rebuild it. - if (vtkShaderProperty *sp = act->GetShaderProperty()) - if (sp->GetShaderMTime() > this->ShaderBuildTimeStamp.GetMTime()) - this->ShaderProgram = nullptr; + // (The base, LowMemoryPolyDataMapper::RenderPieceStart, rebuilds the program + // when the actor's shader property changed after it was built -- VTK's + // low-memory mapper ignores that property once its program exists.) // IsUpToDate() false => the base deletes EVERY array texture and re-binds them // all (the per-frame churn this mapper exists to avoid). That only happens on // a real input/topology/shader change; streaming writes never touch the diff --git a/src/cvcGL/examples/CMakeLists.txt b/src/cvcGL/examples/CMakeLists.txt index 6fdac1a4..d7a7f13e 100644 --- a/src/cvcGL/examples/CMakeLists.txt +++ b/src/cvcGL/examples/CMakeLists.txt @@ -205,6 +205,14 @@ set_target_properties(${_cvcgl_example_bins} PROPERTIES INSTALL_RPATH_USE_LINK_PATH ON) install(TARGETS ${_cvcgl_example_bins} RUNTIME DESTINATION bin COMPONENT cvcgl-examples) +# The bins carry third-party code whose notices a binary redistribution must +# reproduce -- VTK-derived code in the statically linked cvcGL (BSD-3-Clause) and +# lsystem_coast's OceanFFT, a port of ABYSSAL (MIT) -- so the component ships +# THIRD_PARTY_NOTICES.md, with the LICENSE it links to. (The browser build gets +# the same two files from cvcpkg/recipes/cvcgl-examples/build-wasm.sh.) +install(FILES "${CMAKE_CURRENT_SOURCE_DIR}/../../../THIRD_PARTY_NOTICES.md" + "${CMAKE_CURRENT_SOURCE_DIR}/../../../LICENSE" + DESTINATION share/cvcgl-examples COMPONENT cvcgl-examples) # Unit tests for nav_common's bounds-safe compositing (the minimap-overflow crash # guard) + the occupancy rasterizer. Only when GoogleTest is available (CVC_BUILD_TESTS diff --git a/src/cvcGL/nodes.cpp b/src/cvcGL/nodes.cpp index 502a2e94..ff25ec9b 100644 --- a/src/cvcGL/nodes.cpp +++ b/src/cvcGL/nodes.cpp @@ -11,6 +11,7 @@ #include #include #include +#include #include #include #include @@ -294,7 +295,7 @@ void GeometryShape::setGeometry(const cvc::geometry &geom) { } vtkSmartPointer GeometryShape::createProp(RenderView &) { - auto mapper = vtkSmartPointer::New(); + auto mapper = newPolyDataMapper(); if (m_polyData) mapper->SetInputData(m_polyData); auto actor = vtkSmartPointer::New(); diff --git a/src/cvcGL/test/cvcgl_lowmem_bench.cpp b/src/cvcGL/test/cvcgl_lowmem_bench.cpp new file mode 100644 index 00000000..e15c1beb --- /dev/null +++ b/src/cvcGL/test/cvcgl_lowmem_bench.cpp @@ -0,0 +1,484 @@ +/* + Copyright 2026 The University of Texas at Austin + + This file is part of libcvc. + + libcvc is free software; you can redistribute it and/or + modify it under the terms of the GNU Lesser General Public + License version 2.1 as published by the Free Software Foundation. +*/ + +// Not a test: prints what cvc::gl::LowMemoryPolyDataMapper saves per frame and +// per draw, DrawPath::Stock against DrawPath::Fast, on desktop GL with the +// low-memory mapper forced (the mapper WebGL2 gets). +// +// Scene (a demo-style frame): a textured ground, a merged city mesh of N boxes, +// V single-colour vehicle nodes, translucent route ribbons with a depth offset, +// line trails, and optionally many extra small single-colour nodes; shadows on +// (camera -> strided shadow baker -> shadow map -> translucent -> volumetric -> +// overlay) and off. Each node's mapper is swapped for one that tallies the GL +// calls or the CPU time of its own draws, so "per draw" is exactly +// RenderPieceDraw. Frames are rendered without glFinish; times are CPU. +// +// Every configuration runs twice per path: a TIMED pass with VTK's own GL +// function pointers (no counting wrapper on any call; each mapper draw only +// reads the clock) and a COUNTED pass with the wrappers installed, which also +// track uniform redundancy per (program, location) -- that costs CPU in +// proportion to the calls, so its times are not reported. Times come from the +// timed pass, call counts from the counted one. +// +// cvcgl_lowmem_bench [--frames=40] [--boxes=20000] [--vehicles=6] [--extra=0] +// [--size=960x540] +#include "gl_call_counter.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +using cvc::gl::GeometryNode; +using cvc::gl::GeometryRenderMode; +using cvc::gl::LowMemoryPolyDataMapper; +using cvc::gl::SceneGraph; +using cvc::gl::SceneRenderer; +using cvcgl_test::GLCalls; +using cvcgl_test::glcallsRead; +using DrawPath = LowMemoryPolyDataMapper::DrawPath; +using clk = std::chrono::steady_clock; + +namespace { + +double msSince(clk::time_point t0) { + return std::chrono::duration(clk::now() - t0).count(); +} + +// Tallies the GL calls (counted pass) or the CPU time (timed pass) of every +// draw of every TallyMapper. +bool g_counting = false; +GLCalls g_drawCalls; +double g_drawMs = 0, g_draws = 0; + +class TallyMapper : public LowMemoryPolyDataMapper { +public: + static TallyMapper *New(); + vtkTypeMacro(TallyMapper, LowMemoryPolyDataMapper); + void RenderPieceDraw(vtkRenderer *ren, vtkActor *act) override { + g_draws += 1; + if (g_counting) { + const GLCalls before = glcallsRead(); + Superclass::RenderPieceDraw(ren, act); + g_drawCalls += glcallsRead() - before; + return; + } + const auto t0 = clk::now(); + Superclass::RenderPieceDraw(ren, act); + g_drawMs += msSince(t0); + } + +protected: + TallyMapper() = default; +}; +vtkStandardNewMacro(TallyMapper); + +class PeekNode : public GeometryNode { +public: + using GeometryNode::GeometryNode; + vtkActor *actor() { return vtkActor::SafeDownCast(getProp()); } +}; + +// Replace a node's mapper with a TallyMapper carrying the same settings +// (GeometryNode configures nothing else on it that these nodes use). +void swapInTally(PeekNode &n) { + vtkActor *a = n.actor(); + auto *old = LowMemoryPolyDataMapper::SafeDownCast(a->GetMapper()); + if (!old) + return; + vtkNew m; + m->SetInputData(vtkPolyData::SafeDownCast(old->GetInput())); + m->SetScalarVisibility(old->GetScalarVisibility()); + m->SetScalarMode(old->GetScalarMode()); + m->SetColorMode(old->GetColorMode()); + m->SetVBOShiftScaleMethod(old->GetVBOShiftScaleMethod()); + double f = 0, u = 0; + old->GetRelativeCoincidentTopologyPolygonOffsetParameters(f, u); + m->SetRelativeCoincidentTopologyPolygonOffsetParameters(f, u); + old->GetRelativeCoincidentTopologyLineOffsetParameters(f, u); + m->SetRelativeCoincidentTopologyLineOffsetParameters(f, u); + old->GetRelativeCoincidentTopologyPointOffsetParameter(u); + m->SetRelativeCoincidentTopologyPointOffsetParameter(u); + a->SetMapper(m); +} + +void addBox(cvc::geometry &g, double cx, double cy, double cz, double sx, double sy, double sz) { + const double x0 = cx - sx, x1 = cx + sx, y0 = cy - sy, y1 = cy + sy, z0 = cz - sz, z1 = cz + sz; + const double faces[5][4][3] = {{{x0, y0, z1}, {x1, y0, z1}, {x1, y1, z1}, {x0, y1, z1}}, + {{x0, y0, z0}, {x1, y0, z0}, {x1, y0, z1}, {x0, y0, z1}}, + {{x1, y1, z0}, {x0, y1, z0}, {x0, y1, z1}, {x1, y1, z1}}, + {{x0, y1, z0}, {x0, y0, z0}, {x0, y0, z1}, {x0, y1, z1}}, + {{x1, y0, z0}, {x1, y1, z0}, {x1, y1, z1}, {x1, y0, z1}}}; + for (const auto &f : faces) { + const unsigned b = static_cast(g.points().size()); + for (const auto &p : f) + g.points().push_back({p[0], p[1], p[2]}); + g.tris().push_back({b, b + 1, b + 2}); + g.tris().push_back({b, b + 2, b + 3}); + } +} + +struct Options { + int frames = 40, boxes = 20000, vehicles = 6, extra = 0, w = 960, h = 540; +}; + +struct Bench { + explicit Bench(cvc::app &app) : sg(app, "lowmem_bench") { sg.setDiagnosticChromeVisible(false); } + SceneGraph sg; + std::unique_ptr sr; + std::vector> nodes, vehicles; + + std::shared_ptr node(const std::string &name) { + auto n = sg.getGraphicsRoot()->addGraphicsChild(name); + nodes.push_back(n); + return n; + } +}; + +void build(Bench &b, const Options &o) { + const double half = 600.0; + { + auto ground = b.node("ground"); + cvc::geometry g; + const int n = 64; + for (int j = 0; j < n; ++j) + for (int i = 0; i < n; ++i) { + const double x = -half + 2 * half * i / (n - 1), y = -half + 2 * half * j / (n - 1); + g.points().push_back({x, y, 3.0 * std::sin(x / 90) * std::cos(y / 70)}); + g.uvs().push_back({double(i) / (n - 1), double(j) / (n - 1)}); + } + for (int j = 0; j + 1 < n; ++j) + for (int i = 0; i + 1 < n; ++i) { + const unsigned a = j * n + i, c = a + n; + g.tris().push_back({a, a + 1, c + 1}); + g.tris().push_back({a, c + 1, c}); + } + ground->setGeometry(g); + ground->setUseSingleColor(true); + ground->setColor(1, 1, 1); + ground->setAmbient(0.5); + ground->setDiffuse(0.6); + cvc::image img(512, 512, cvc::image::pixel_format::RGBA, cvc::image::data_type::u8); + unsigned char *p = img.data(); + for (int j = 0; j < 512; ++j) + for (int i = 0; i < 512; ++i) { + unsigned char *q = p + 4 * (j * 512 + i); + q[0] = static_cast(i / 2); + q[1] = static_cast(j / 2); + q[2] = 128; + q[3] = 255; + } + ground->setTexture(img); + } + if (o.boxes > 0) { + auto city = b.node("city"); + cvc::geometry g; + const int side = static_cast(std::ceil(std::sqrt(double(o.boxes)))); + for (int k = 0; k < o.boxes; ++k) { + const double x = -half * 0.9 + 1.8 * half * (k % side) / side; + const double y = -half * 0.9 + 1.8 * half * (k / side) / side; + const double hgt = 2.0 + 10.0 * (0.5 + 0.5 * std::sin(k * 0.37)); + addBox(g, x, y, hgt, 2.5, 2.5, hgt); + } + city->setGeometry(g); + city->setUseSingleColor(true); + city->setColor(0.72, 0.70, 0.66); + } + for (int v = 0; v < o.vehicles; ++v) { + auto veh = b.node("veh" + std::to_string(v)); + cvc::geometry g; + addBox(g, 0, 0, 2.0, 4.0, 2.0, 1.5); + veh->setGeometry(g); + veh->setUseSingleColor(true); + veh->setColor(0.2 + 0.1 * v, 0.5, 0.9 - 0.1 * v); + b.vehicles.push_back(veh); + } + for (int r = 0; r < 4; ++r) { + auto rib = b.node("route" + std::to_string(r)); + cvc::geometry g; + for (int k = 0; k < 120; ++k) { + const double x = -400 + 7.0 * k, y = -200 + 120.0 * r + 25 * std::sin(k * 0.1); + g.points().push_back({x, y - 2.0, 4.0}); + g.points().push_back({x, y + 2.0, 4.0}); + } + for (unsigned k = 0; k + 1 < 120; ++k) { + g.tris().push_back({2 * k, 2 * k + 1, 2 * k + 3}); + g.tris().push_back({2 * k, 2 * k + 3, 2 * k + 2}); + } + rib->setGeometry(g); + rib->setUseSingleColor(true); + rib->setColor(0.98, 0.8, 0.25); + rib->setOpacity(0.55); + rib->setDepthOffset(2.0); + + auto trail = b.node("trail" + std::to_string(r)); + cvc::geometry t; + for (int k = 0; k < 200; ++k) { + t.points().push_back({-450 + 4.0 * k, -150 + 110.0 * r + 15 * std::cos(k * 0.07), 5.0}); + if (k) + t.lines().push_back({static_cast(k - 1), static_cast(k)}); + } + trail->setGeometry(t); + trail->setRenderMode(GeometryRenderMode::LINES); // after setGeometry, which picks a mode + trail->setUseSingleColor(true); + trail->setColor(0.3, 1.0, 0.4); + } + for (int e = 0; e < o.extra; ++e) { + auto n = b.node("extra" + std::to_string(e)); + cvc::geometry g; + addBox(g, -500 + 1000.0 * (e % 20) / 20, -500 + 1000.0 * (e / 20) / 20, 3, 3, 3, 3); + n->setGeometry(g); + n->setUseSingleColor(true); + n->setColor(0.9, 0.4, 0.3); + } + b.sg.addSpotLight(-500, -700, 900, 0, 0, 0, 45.0, 1.0, 0.97, 0.9, 1.1); + b.sg.addFillLight(600, 500, 600, 0, 0, 0, 0.8, 0.85, 1.0, 0.4); +} + +void poseVehicles(Bench &b, int frame) { + for (size_t v = 0; v < b.vehicles.size(); ++v) { + const double t = 0.02 * frame + 1.1 * v, c = std::cos(t), s = std::sin(t); + const double m[16] = { + c, -s, 0, 300 * std::cos(0.3 * t + v), s, c, 0, 250 * std::sin(0.4 * t), 0, 0, 1, 0, 0, + 0, 0, 1}; + b.vehicles[v]->setPoseMatrix(m); + } +} + +// One pass: timed (frameMs, drawMs) or counted (frame, draw). +struct Run { + double frames = 0, frameMs = 0, drawMs = 0, draws = 0; + GLCalls frame, draw; + LowMemoryPolyDataMapper::Stats stats; + Run &operator+=(const Run &o) { + frames += o.frames; + frameMs += o.frameMs; + drawMs += o.drawMs; + draws += o.draws; + frame += o.frame; + draw += o.draw; + stats.programLookups += o.stats.programLookups; + stats.programLookupsSkipped += o.stats.programLookupsSkipped; + stats.stockDraws += o.stats.stockDraws; + return *this; + } +}; + +LowMemoryPolyDataMapper::Stats sumStats(Bench &b) { + LowMemoryPolyDataMapper::Stats s; + for (auto &n : b.nodes) + if (auto *m = LowMemoryPolyDataMapper::SafeDownCast(n->actor()->GetMapper())) { + const auto &t = m->stats(); + s.draws += t.draws; + s.cellTypesSkipped += t.cellTypesSkipped; + s.programLookups += t.programLookups; + s.programLookupsSkipped += t.programLookupsSkipped; + s.stockDraws += t.stockDraws; + } + return s; +} + +Run measure(Bench &b, DrawPath path, bool bake, int frames, int &frameNo, bool counted) { + LowMemoryPolyDataMapper::setDrawPath(path); + if (counted) + cvcgl_test::glcallsInstall(); + else + cvcgl_test::glcallsUninstall(); // VTK's own pointers: no wrapper cost in the timing + g_counting = counted; + if (counted) { + // One frame on this path first: the first frame after a path switch differs + // by a call or two of vtkOpenGLState-cached state (glPointSize). + poseVehicles(b, frameNo++); + if (bake) + b.sg.invalidateShadowBake(); + b.sr->render(); + } + Run r; + for (auto &n : b.nodes) + if (auto *m = LowMemoryPolyDataMapper::SafeDownCast(n->actor()->GetMapper())) + m->resetStats(); + for (int f = 0; f < frames; ++f) { + poseVehicles(b, frameNo++); + if (bake) + b.sg.invalidateShadowBake(); + g_drawCalls = GLCalls(); + g_drawMs = 0; + g_draws = 0; + if (counted) { + const GLCalls before = glcallsRead(); + b.sr->render(); + r.frame += glcallsRead() - before; + r.draw += g_drawCalls; + } else { + const auto t0 = clk::now(); + b.sr->render(); + r.frameMs += msSince(t0); + r.drawMs += g_drawMs; + } + r.draws += g_draws; + r.frames += 1; + } + r.stats = sumStats(b); + g_counting = false; + return r; +} + +std::vector rgba(Bench &b) { + vtkNew px; + const int *sz = b.sr->renderWindow()->GetSize(); + b.sr->renderWindow()->GetRGBACharPixelData(0, 0, sz[0] - 1, sz[1] - 1, 0, px); + const unsigned char *p = px->GetPointer(0); + return std::vector(p, p + px->GetNumberOfValues()); +} + +// st/ft: the timed passes; sc/fc: the counted passes (Stock / Fast). +void report(const char *label, const Run &st, const Run &ft, const Run &sc, const Run &fc) { + const double n = st.frames, nc = sc.frames; + std::printf("\n== %s (%g timed + %g counted frames per path)\n", label, n, nc); + std::printf(" Stock Fast\n"); + std::printf(" draws/frame %8.1f %8.1f\n", st.draws / n, ft.draws / n); + std::printf(" GL calls/frame %8.1f %8.1f (%+.1f, %+.0f%%)\n", sc.frame.total() / nc, + fc.frame.total() / nc, (fc.frame.total() - sc.frame.total()) / nc, + 100.0 * (fc.frame.total() - sc.frame.total()) / sc.frame.total()); + std::printf(" in mapper draws %8.1f %8.1f\n", sc.draw.total() / nc, + fc.draw.total() / nc); + std::printf(" uniforms %8.1f %8.1f (redundant %.1f -> %.1f)\n", + sc.frame.uniforms() / nc, fc.frame.uniforms() / nc, sc.frame.uniformRedundant / nc, + fc.frame.uniformRedundant / nc); + std::printf(" desktop round-trips (upper bound) %8.1f %8.1f\n", sc.frame.roundTrips() / nc, + fc.frame.roundTrips() / nc); + std::printf(" GL calls/draw %8.1f %8.1f\n", sc.draw.total() / sc.draws, + fc.draw.total() / fc.draws); + std::printf(" timed, no GL wrappers:\n"); + std::printf(" draw-stage CPU ms/frame %6.3f %8.3f\n", st.drawMs / n, ft.drawMs / n); + std::printf(" draw-stage CPU ms/draw %6.4f %8.4f\n", st.drawMs / st.draws, + ft.drawMs / ft.draws); + std::printf(" frame CPU ms (no finish)%6.2f %8.2f\n", st.frameMs / n, ft.frameMs / n); + std::printf(" lookups/frame %8.1f %8.1f (skipped %.1f)\n", + double(st.stats.programLookups + st.stats.stockDraws) / n, + double(ft.stats.programLookups) / n, double(ft.stats.programLookupsSkipped) / n); + std::printf(" Stock per frame: %s\n", sc.frame.str(nc).c_str()); + std::printf(" Fast per frame: %s\n", fc.frame.str(nc).c_str()); + std::printf(" Stock per draw : %s\n", sc.draw.str(sc.draws).c_str()); + std::printf(" Fast per draw : %s\n", fc.draw.str(fc.draws).c_str()); +} + +} // namespace + +int main(int argc, char **argv) { +#ifdef _WIN32 + _putenv_s("CVCGL_LOWMEM_MAPPER", "force"); +#else + setenv("CVCGL_LOWMEM_MAPPER", "force", 1); + setenv("__GL_SYNC_TO_VBLANK", "0", 0); // NVIDIA throttles offscreen swaps otherwise + setenv("vblank_mode", "0", 0); +#endif + Options o; + for (int i = 1; i < argc; ++i) { + const std::string a = argv[i]; + auto val = [&](const char *k) { + return a.rfind(k, 0) == 0 ? a.c_str() + std::strlen(k) : nullptr; + }; + if (const char *v = val("--frames=")) + o.frames = std::atoi(v); + else if (const char *v = val("--boxes=")) + o.boxes = std::atoi(v); + else if (const char *v = val("--vehicles=")) + o.vehicles = std::atoi(v); + else if (const char *v = val("--extra=")) + o.extra = std::atoi(v); + else if (const char *v = val("--size=")) + std::sscanf(v, "%dx%d", &o.w, &o.h); + } + if (!LowMemoryPolyDataMapper::replicaActive()) { + std::printf("this VTK is not 9.5.0: LowMemoryPolyDataMapper draws the stock way\n"); + return 0; + } + cvc::app app; + Bench b(app); + build(b, o); + b.sr = std::make_unique(b.sg, o.w, o.h, /*offscreen=*/true, "bench"); + // No MSAA: with VTK's default 8x, NVIDIA's resolve varies by 1 LSB at a few edge + // pixels between identical frames, which would hide the Stock/Fast comparison. + b.sr->renderWindow()->SetMultiSamples(0); + b.sr->setBackground(0.2, 0.26, 0.36); + b.sr->setCamera(-700, -900, 650, 0, 0, 0, 0, 0, 1, 42.0, 10.0, 6000.0); + for (auto &n : b.nodes) + swapInTally(*n); + b.sg.setShadowsEnabled(true); + b.sg.setShadowResolution(2048); + b.sg.setShadowUpdateInterval(100000); // bake only when invalidated + b.sr->render(); + b.sr->render(); + cvcgl_test::glcallsInstall(); + std::printf("lowmem bench: %zu nodes, %d boxes merged, %d vehicles, %d extra nodes, %dx%d\n", + b.nodes.size(), o.boxes, o.vehicles, o.extra, o.w, o.h); + + int frameNo = 0; + for (int shadows = 1; shadows >= 0; --shadows) { + if (!shadows) + b.sg.setShadowsEnabled(false); + for (DrawPath p : {DrawPath::Stock, DrawPath::Fast}) // warm both paths + measure(b, p, shadows != 0, 2, frameNo, false); + const auto kinds = shadows ? std::vector{true, false} : std::vector{false}; + for (bool bake : kinds) { + // Timed: interleave the two paths in halves so drift hits both alike. + Run st, ft, sc, fc; + for (int half = 0; half < 2; ++half) { + st += measure(b, DrawPath::Stock, bake, o.frames / 2, frameNo, false); + ft += measure(b, DrawPath::Fast, bake, o.frames / 2, frameNo, false); + } + // Counted: GL call counts are the same every frame of a kind; a few do. + const int countedFrames = std::max(2, o.frames / 8); + sc = measure(b, DrawPath::Stock, bake, countedFrames, frameNo, true); + fc = measure(b, DrawPath::Fast, bake, countedFrames, frameNo, true); + cvcgl_test::glcallsUninstall(); + report(shadows ? (bake ? "shadows ON, bake every frame" : "shadows ON, steady (no bake)") + : "shadows OFF", + st, ft, sc, fc); + // Same pose on both paths, then compare the frames. + LowMemoryPolyDataMapper::setDrawPath(DrawPath::Stock); + poseVehicles(b, 7); + if (bake) + b.sg.invalidateShadowBake(); + b.sr->render(); + const auto a = rgba(b); + LowMemoryPolyDataMapper::setDrawPath(DrawPath::Fast); + if (bake) + b.sg.invalidateShadowBake(); + b.sr->render(); + const auto c = rgba(b); + long d = 0; + for (size_t i = 0; i < a.size() && i < c.size(); ++i) + d += a[i] != c[i]; + std::printf(" pixels Stock vs Fast: %s (%ld differing bytes of %zu)\n", + d == 0 && a.size() == c.size() ? "IDENTICAL" : "DIFFER", d, a.size()); + } + } + return 0; +} diff --git a/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp b/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp new file mode 100644 index 00000000..e1b9ae16 --- /dev/null +++ b/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp @@ -0,0 +1,2091 @@ +/* + Copyright 2026 The University of Texas at Austin + + This file is part of libcvc. + + libcvc is free software; you can redistribute it and/or + modify it under the terms of the GNU Lesser General Public + License version 2.1 as published by the Free Software Foundation. +*/ + +// cvc::gl::LowMemoryPolyDataMapper -- the WebGL2/GLES mapper with the empty +// cell types and the per-draw shader-cache lookup taken out of its draw. +// +// Runs that mapper natively (policy Force) on cvcGL's real pipeline -- a +// SceneGraph with shadows on (camera -> strided shadow baker -> shadow map -> +// translucent -> volumetric -> overlay) and off -- over GeometryNodes of every +// kind a scene uses (lit single colour, textured ground, translucent with a +// depth offset, wireframe lines, points, per-vertex colour, quads) plus raw VTK +// actors whose mapper counts the GL calls of its own draws: surface, wireframe +// and points representations of a triangle mesh, edge visibility (with and +// without its own offset), cell colours on quads, point colours, texture +// coordinates, translucency, verts, strips, and lines flat / lit / wide / as +// points. Checks: +// 1. the policy: CVCGL_LOWMEM_MAPPER is read, Force/Off/Auto pick the mapper; +// 2. the replica: the ordered GL trace of a DrawPath::AllCellTypes frame -- +// every call, its bound program or texture unit, its argument bytes and +// the uniform values it sends -- equals the Stock frame's exactly (bake +// and plain frames, shadows on and off, and the hardware selector's cell +// and point picking passes); +// 3. Fast's trace equals that same trace with what Fast leaves out removed, +// DERIVED from the AllCellTypes trace alone (the replica marks where each +// draw's stages begin and end): a draw's cell-type blocks that issued no +// draw call go iff replaying the trace's glUniform* calls without them +// leaves every draw call's uniform state unchanged -- never from the +// mapper's own skip decision, whose markers are only counted. And Fast's +// own trace, replayed the same way, shows every draw call the uniform +// state (every location of its program) it sees on AllCellTypes, and +// leaves each program what the next frame's draws read before writing. A +// skipped shader-cache lookup removes no GL call -- the lookup's GL calls +// are the bind tail the re-bind issues too; what it saves is CPU, which +// the stats check; +// 4. pixels: AllCellTypes and Fast frames (RGBA) match Stock's -- as do a +// Stock frame right after Fast, and frames across a recompile and across a +// new window -- byte for byte, and so do pairs of Stock frames of one GL +// trace. Only on Apple's renderer (GL version or renderer string naming +// Apple), which rounds a few bytes 1 LSB apart now and then, in one window +// and across windows, is every pixel check -- the pairs of Stock frames +// too -- held to a fixed envelope instead: identical, or no byte off by +// more than 1 and at most 64 bytes; see checkPixels(); +// 5. the saving: Fast draws a single-cell-type mesh in <= 2 VAO binds and a +// fraction of the stock calls, with no desktop round-trip in any draw +// beyond Stock's (none, but VTK's own warning on a line width the driver +// cannot draw), and a steady frame does no shader-cache lookup; +// 6. lifecycle: a shader rebuild (the shadow bake's pass change), a mapper's +// released resources, GetShader(), programs released in the shader cache +// and the scene re-opened in a new window all go back through the full +// lookup (or a recompile) and still match; +// 7. guards: an offset layout where skipping would change a drawn type's +// depth offset, picking and vertex visibility all draw the stock way (and +// pick the same; on Apple's renderer point picking may differ by 1 point +// per prop, see comparePicking()); +// 8. a plain GeometryNode's shader replacement added (and cleared) after its +// first draw reaches the GPU, on every draw path. +// Frames are rendered without MSAA: with VTK's default 8x, NVIDIA's resolve +// varies by 1 LSB at a few edge pixels between identical GL streams. +// Renders for real; skips (rc 0) where nothing rasterises unless +// CVC_REQUIRE_RENDER=1. Set CVCGL_LOWMEM_REPORT=1 for the per-draw breakdown. +// +// NOT assert(): built Release, where NDEBUG makes assert() a no-op. +#include "gl_call_counter.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include // GL_POINTS ... (desktop-only test) + +using cvc::gl::GeometryNode; +using cvc::gl::GeometryRenderMode; +using cvc::gl::LowMemoryMapperPolicy; +using cvc::gl::LowMemoryPolyDataMapper; +using cvc::gl::SceneGraph; +using cvc::gl::SceneRenderer; +using cvcgl_test::GLCalls; +using cvcgl_test::glcallsInstall; +using cvcgl_test::glcallsRead; +using cvcgl_test::kTraceMarkBase; +using cvcgl_test::Trace; +using cvcgl_test::TraceRec; +using DrawPath = LowMemoryPolyDataMapper::DrawPath; +using DrawStage = LowMemoryPolyDataMapper::DrawStage; + +namespace { + +int g_failures = 0, g_checks = 0; +bool g_report = false; + +bool check(bool ok, const std::string &what, const std::string &detail = "") { + ++g_checks; + if (!ok) + ++g_failures; + std::printf(" [%s] %s%s%s\n", ok ? "PASS" : "FAIL", what.c_str(), detail.empty() ? "" : " -- ", + detail.c_str()); + std::fflush(stdout); + return ok; +} + +std::string fmt(const char *f, double a, double b = 0, double c = 0) { + char buf[256]; + std::snprintf(buf, sizeof buf, f, a, b, c); + return buf; +} + +const char *pathName(DrawPath p) { + return p == DrawPath::Stock ? "Stock" : p == DrawPath::AllCellTypes ? "AllCellTypes" : "Fast"; +} + +// Counts VTK errors (shader compile/link failures arrive here, not as exceptions). +class ErrorCounter : public vtkOutputWindow { +public: + static ErrorCounter *New(); + vtkTypeMacro(ErrorCounter, vtkOutputWindow); + static int &errors() { + static int n = 0; + return n; + } + void DisplayErrorText(const char *t) override { + ++errors(); + std::fprintf(stderr, "VTK ERROR: %s\n", t); + } + void DisplayWarningText(const char *t) override { std::fprintf(stderr, "VTK WARNING: %s\n", t); } + void DisplayGenericWarningText(const char *t) override { + std::fprintf(stderr, "VTK WARNING: %s\n", t); + } + void DisplayText(const char *t) override { std::fprintf(stderr, "%s", t); } +}; +vtkStandardNewMacro(ErrorCounter); + +// ── GL traces ──────────────────────────────────────────────────────────────── +// The replica's stage observer: each stage boundary becomes a trace marker. +void onStage(DrawStage stage, int arg) { cvcgl_test::glcallsMark(static_cast(stage), arg); } + +bool isStage(const TraceRec &r, DrawStage stage) { + return r.marker() && r.entry - kTraceMarkBase == static_cast(stage); +} + +int countStage(const Trace &t, DrawStage stage) { + int n = 0; + for (const auto &r : t) + n += isStage(r, stage); + return n; +} + +// The GL calls alone. +Trace glOnly(const Trace &t) { + Trace out; + out.reserve(t.size()); + for (const auto &r : t) + if (!r.marker()) + out.push_back(r); + return out; +} + +// Equal GL traces (every call, in order, with its context and bytes)? If not, +// where they first part. +bool sameTrace(const Trace &a, const Trace &b, std::string *why) { + const size_t n = std::min(a.size(), b.size()); + size_t i = 0; + while (i < n && a[i] == b[i]) + ++i; + if (i == n && a.size() == b.size()) + return true; + if (why) { + *why = fmt("%.0f vs %.0f calls, first difference at #%.0f: ", double(a.size()), + double(b.size()), double(i)); + *why += i < a.size() ? cvcgl_test::glcallsDescribe(a[i]) : std::string("(end)"); + *why += " vs "; + *why += i < b.size() ? cvcgl_test::glcallsDescribe(b[i]) : std::string("(end)"); + } + return false; +} + +// vtkOpenGLState caches glPointSize and glLineWidth (desktop GL only: GLES 3 / +// WebGL have neither call). Whether a kept block's vtkglPointSize reaches GL +// therefore depends on what the blocks before it set, so leaving out an empty +// block can add or drop one of these calls without changing what any draw +// sees. Compare them as that: drop the calls, and append to every GL_POINTS +// draw the point size it draws with and to every line draw the line width +// (tracked from the trace's initial-state markers and its calls). +Trace asDrawState(const Trace &t) { + static const int pointSize = cvcgl_test::glcallsEntry("PointSize"); + static const int lineWidth = cvcgl_test::glcallsEntry("LineWidth"); + static const int dispatch = cvcgl_test::glcallsEntry("DispatchCompute"); + std::string ps(4, '\0'), lw(4, '\0'); + Trace out; + out.reserve(t.size()); + for (const auto &r : t) { + if (r.marker()) { + if (r.entry == kTraceMarkBase + cvcgl_test::kTraceInitPointSize) + ps = r.args; + else if (r.entry == kTraceMarkBase + cvcgl_test::kTraceInitLineWidth) + lw = r.args; + else + out.push_back(r); + continue; + } + if (r.entry == pointSize) { + ps = r.args; + continue; + } + if (r.entry == lineWidth) { + lw = r.args; + continue; + } + out.push_back(r); + if (cvcgl_test::glcallsIsDraw(r.entry) && r.entry != dispatch && r.args.size() >= 4) { + GLenum mode = 0; + std::memcpy(&mode, r.args.data(), sizeof mode); + if (mode == GL_POINTS) + out.back().args += ps; + else if (mode == GL_LINES || mode == GL_LINE_STRIP || mode == GL_LINE_LOOP) + out.back().args += lw; + } + } + return out; +} + +// ── the uniforms each draw call sees ───────────────────────────────────────── +// A trace's glUniform* calls replayed into (program, location) -> value: at +// every draw call, the bound program's locations the trace has written so far +// (a location it has not written holds what the program had before the trace). +struct UniformView { + struct DrawCall { + size_t at = 0; // index in the trace + unsigned program = 0; + std::map written; + }; + std::vector calls; + std::map> end; // each program, at the trace's end +}; + +// resetDraw >= 0: forget what the program of the trace's resetDraw-th replica +// draw (DrawBegin) holds when that draw first touches it -- that draw then sees +// only what it writes itself, whatever an earlier draw left. +UniformView replayUniforms(const Trace &t, int resetDraw = -1) { + static const int link = cvcgl_test::glcallsEntry("LinkProgram"); + static const int del = cvcgl_test::glcallsEntry("DeleteProgram"); + UniformView v; + auto &state = v.end; + int draw = -1; + bool resetPending = false; + for (size_t i = 0; i < t.size(); ++i) { + const TraceRec &r = t[i]; + if (r.marker()) { + if (isStage(r, DrawStage::DrawBegin) && ++draw == resetDraw) + resetPending = true; + continue; + } + const bool uniform = cvcgl_test::glcallsIsUniform(r.entry); + const bool drawCall = cvcgl_test::glcallsIsDraw(r.entry); + if (resetPending && (uniform || drawCall)) { + state.erase(r.ctx); + resetPending = false; + } + if ((r.entry == link || r.entry == del) && r.args.size() >= sizeof(unsigned)) { + unsigned program = 0; // a (re)link resets every uniform; a deleted name is reused + std::memcpy(&program, r.args.data(), sizeof program); + state.erase(program); + } else if (uniform) { + for (auto &w : cvcgl_test::glcallsUniformWrites(r)) + state[r.ctx][w.location] = std::move(w.value); + } else if (drawCall) { + v.calls.push_back({i, r.ctx, state[r.ctx]}); + } + } + return v; +} + +// A UniformWrite value ("3f:") as its type and numbers. +std::string uniformValue(const std::string &v) { + const size_t colon = v.find(':'); + if (colon == std::string::npos) + return v; + const std::string type = v.substr(0, colon); + const bool f = type.find('f') != std::string::npos; // "3f", "Matrix4f", "Matrix4fT" + const bool u = type.find("ui") != std::string::npos; + std::string out = type + " ("; + char buf[32]; + for (size_t at = colon + 1, k = 0; at + 4 <= v.size() && k < 16; at += 4, ++k) { + if (f) { + float x; + std::memcpy(&x, v.data() + at, sizeof x); + std::snprintf(buf, sizeof buf, k ? " %g" : "%g", double(x)); + } else if (u) { + unsigned x; + std::memcpy(&x, v.data() + at, sizeof x); + std::snprintf(buf, sizeof buf, k ? " %u" : "%u", x); + } else { + int x; + std::memcpy(&x, v.data() + at, sizeof x); + std::snprintf(buf, sizeof buf, k ? " %d" : "%d", x); + } + out += buf; + } + return out + ")"; +} + +// Do two traces' draw calls see the same uniforms? The traces issue the same +// draw calls in the same order (the trace checks pin that), and at each one the +// bound program must hold the same value at every location either trace wrote: +// written to the same value in both, or written in neither so far (both see +// the value from before the trace). Where a draw call sees such an inherited +// value it reads what the previous frame left -- for a repeated frame, what this +// frame leaves -- so for those (program, location)s the two frames' end values +// must agree too. End values no draw call reads before writing are not +// compared: on AllCellTypes a lines-only draw leaves the empty polys/strips +// blocks' cellType and primitiveSize in its program, Fast leaves the lines +// values, and every cell-type block writes both before it draws. +bool sameUniformsSeen(const UniformView &a, const UniformView &b, const Trace &ta, + std::string *why) { + const auto describe = [](const std::map &m, int loc) { + const auto it = m.find(loc); + return it == m.end() ? std::string("(not written yet)") : uniformValue(it->second); + }; + if (a.calls.size() != b.calls.size()) { + if (why) + *why = fmt("%.0f vs %.0f draw calls", double(a.calls.size()), double(b.calls.size())); + return false; + } + static const std::map kNone; + const auto endOf = [](const UniformView &v, + unsigned program) -> const std::map & { + const auto it = v.end.find(program); + return it == v.end.end() ? kNone : it->second; + }; + std::set> inherited; + for (size_t k = 0; k < a.calls.size(); ++k) { + const auto &x = a.calls[k], &y = b.calls[k]; + if (x.program != y.program) { + if (why) + *why = fmt("draw call %.0f: program %.0f vs %.0f", double(k), double(x.program), + double(y.program)); + return false; + } + std::set locations; + for (const auto *m : {&x.written, &y.written, &endOf(a, x.program), &endOf(b, x.program)}) + for (const auto &kv : *m) + locations.insert(kv.first); + for (int loc : locations) { + const auto i = x.written.find(loc), j = y.written.find(loc); + if (i == x.written.end() && j == y.written.end()) { + inherited.insert({x.program, loc}); + continue; + } + if (i != x.written.end() && j != y.written.end() && i->second == j->second) + continue; + if (why) + *why = fmt("draw call %.0f (trace #%.0f, ", double(k), double(x.at)) + + cvcgl_test::glcallsDescribe(ta[x.at]) + fmt("): location %.0f: ", double(loc)) + + describe(x.written, loc) + " vs " + describe(y.written, loc); + return false; + } + } + for (const auto &[program, loc] : inherited) { + const auto &ea = endOf(a, program), &eb = endOf(b, program); + const auto i = ea.find(loc), j = eb.find(loc); + const bool same = (i == ea.end() && j == eb.end()) || + (i != ea.end() && j != eb.end() && i->second == j->second); + if (!same) { + if (why) + *why = fmt("program %.0f location %.0f, read before the frame writes it: the frame " + "leaves ", + double(program), double(loc)) + + describe(ea, loc) + " vs " + describe(eb, loc); + return false; + } + } + return true; +} + +// For each replica draw (DrawBegin) of a trace: its cell-type blocks that issued +// no draw call. +std::vector> emptyBlocks(const Trace &t) { + std::vector> out; + bool inBlock = false, drew = false; + int type = 0; + for (const auto &r : t) { + if (!r.marker()) { + drew = drew || (inBlock && cvcgl_test::glcallsIsDraw(r.entry)); + continue; + } + if (isStage(r, DrawStage::DrawBegin)) { + out.emplace_back(); + } else if (isStage(r, DrawStage::AgentBegin)) { + inBlock = true; + drew = false; + type = static_cast(r.ctx); + } else if (isStage(r, DrawStage::AgentEnd)) { + inBlock = false; + if (!drew && !out.empty()) + out.back().push_back(type); + } + } + return out; +} + +// `t` without the cell-type blocks (AgentBegin..AgentEnd, markers included) +// that drop(draw, type) selects; draw counts the trace's replica draws. +template Trace without(const Trace &t, Drop drop) { + Trace out; + out.reserve(t.size()); + int draw = -1; + bool dropping = false; + for (const auto &r : t) { + if (isStage(r, DrawStage::DrawBegin)) + ++draw; + if (isStage(r, DrawStage::AgentBegin) && drop(draw, static_cast(r.ctx))) + dropping = true; + if (!dropping) + out.push_back(r); + if (isStage(r, DrawStage::AgentEnd)) + dropping = false; + } + return out; +} + +// What Fast's GL trace must be, derived from an AllCellTypes trace alone -- +// never from the mapper's own skip decision (coincidentSkipSafe). A replica +// draw's empty cell-type blocks may go iff removing them (that draw's alone) +// changes no uniform any draw call sees, with the draw's program state on +// entry treated as unknown (what earlier draws leave in a shared program is not +// the draw's to rely on): the blocks' writes either reach no draw call or are +// written again, to the same value, before one reads them. Everything else +// stays -- including each full lookup's GL calls, which are the +// ReadyShaderProgram(program) tail (compile if released, bind) that the re-bind +// replacing it issues too. +struct Expected { + Trace trace; // with the markers of what stays + int blocksRemoved = 0, blocksKept = 0, lookups = 0, draws = 0, fallbacks = 0; + std::string firstFallback; // why the first draw that keeps its empty blocks must +}; + +Expected expectFast(const Trace &all) { + Expected e; + const auto empty = emptyBlocks(all); + const auto isEmpty = [&](int d, int type) { + const auto &v = empty[static_cast(d)]; + return std::find(v.begin(), v.end(), type) != v.end(); + }; + std::vector removable(empty.size(), false); + for (size_t d = 0; d < empty.size(); ++d) { + if (empty[d].empty()) + continue; + const int draw = static_cast(d); + const Trace removed = + without(all, [&](int dd, int type) { return dd == draw && isEmpty(dd, type); }); + std::string why; + removable[d] = + sameUniformsSeen(replayUniforms(all, draw), replayUniforms(removed, draw), all, &why); + if (!removable[d]) { + ++e.fallbacks; + if (e.firstFallback.empty()) + e.firstFallback = fmt("draw %.0f: ", double(d)) + why; + } + } + e.trace = without(all, [&](int d, int type) { + return d >= 0 && removable[static_cast(d)] && isEmpty(d, type); + }); + e.draws = static_cast(empty.size()); + for (size_t d = 0; d < empty.size(); ++d) + (removable[d] ? e.blocksRemoved : e.blocksKept) += static_cast(empty[d].size()); + e.lookups = countStage(all, DrawStage::LookupBegin); + return e; +} + +// The production mapper plus a GL-call tally of its own RenderPieceDraw calls. +class ProbeMapper : public LowMemoryPolyDataMapper { +public: + static ProbeMapper *New(); + vtkTypeMacro(ProbeMapper, LowMemoryPolyDataMapper); + GLCalls calls; + double draws = 0; + void clearTally() { + calls = GLCalls(); + draws = 0; + } + void RenderPieceDraw(vtkRenderer *ren, vtkActor *act) override { + const GLCalls before = glcallsRead(); + Superclass::RenderPieceDraw(ren, act); + calls += glcallsRead() - before; + draws += 1; + } + +protected: + ProbeMapper() = default; +}; +vtkStandardNewMacro(ProbeMapper); + +// A GeometryNode whose actor the test can reach. +class PeekNode : public GeometryNode { +public: + using GeometryNode::GeometryNode; + vtkActor *actor() { return vtkActor::SafeDownCast(getProp()); } +}; + +// ── geometry ───────────────────────────────────────────────────────────────── +// A closed, lit box (separate faces so the normals are flat). +cvc::geometry boxGeometry(double cx, double cy, double cz, double sx, double sy, double sz) { + cvc::geometry g; + const double x0 = cx - sx, x1 = cx + sx, y0 = cy - sy, y1 = cy + sy, z0 = cz - sz, z1 = cz + sz; + const double faces[6][4][3] = {{{x0, y0, z1}, {x1, y0, z1}, {x1, y1, z1}, {x0, y1, z1}}, // top + {{x0, y0, z0}, {x0, y1, z0}, {x1, y1, z0}, {x1, y0, z0}}, // bottom + {{x0, y0, z0}, {x1, y0, z0}, {x1, y0, z1}, {x0, y0, z1}}, // -y + {{x1, y1, z0}, {x0, y1, z0}, {x0, y1, z1}, {x1, y1, z1}}, // +y + {{x0, y1, z0}, {x0, y0, z0}, {x0, y0, z1}, {x0, y1, z1}}, // -x + {{x1, y0, z0}, {x1, y1, z0}, {x1, y1, z1}, {x1, y0, z1}}}; // +x + for (const auto &f : faces) { + const unsigned base = static_cast(g.points().size()); + for (const auto &p : f) + g.points().push_back({p[0], p[1], p[2]}); + g.tris().push_back({base, base + 1, base + 2}); + g.tris().push_back({base, base + 2, base + 3}); + } + return g; +} + +// An n x n height-field grid over [-s, s]^2 with UVs. +cvc::geometry groundGeometry(int n, double s) { + cvc::geometry g; + for (int j = 0; j < n; ++j) + for (int i = 0; i < n; ++i) { + const double x = -s + 2 * s * i / (n - 1), y = -s + 2 * s * j / (n - 1); + g.points().push_back({x, y, 0.6 * std::sin(x * 0.35) * std::cos(y * 0.3)}); + g.uvs().push_back({double(i) / (n - 1), double(j) / (n - 1)}); + } + for (int j = 0; j + 1 < n; ++j) + for (int i = 0; i + 1 < n; ++i) { + const unsigned a = j * n + i, b = a + 1, c = a + n, d = c + 1; + g.tris().push_back({a, b, d}); + g.tris().push_back({a, d, c}); + } + return g; +} + +cvc::geometry ribbonGeometry(int n, double x0, double y0, double z) { + cvc::geometry g; + for (int k = 0; k < n; ++k) { + g.points().push_back({x0 + 0.8 * k, y0 - 0.6, z}); + g.points().push_back({x0 + 0.8 * k, y0 + 0.6, z}); + } + for (int k = 0; k + 1 < n; ++k) { + const unsigned l0 = 2 * k, r0 = l0 + 1, l1 = l0 + 2, r1 = l0 + 3; + g.tris().push_back({l0, r0, r1}); + g.tris().push_back({l0, r1, l1}); + } + return g; +} + +cvc::geometry polylineGeometry(int n, double x0, double y0, double z) { + cvc::geometry g; + for (int k = 0; k < n; ++k) { + g.points().push_back({x0 + 0.5 * k, y0 + std::sin(k * 0.4), z}); + if (k) + g.lines().push_back({static_cast(k - 1), static_cast(k)}); + } + return g; +} + +cvc::geometry cloudGeometry(int n, double x0, double y0, double z) { + cvc::geometry g; + for (int k = 0; k < n; ++k) + g.points().push_back({x0 + 0.4 * (k % 10), y0 + 0.4 * (k / 10), z + 0.1 * (k % 3)}); + return g; +} + +cvc::geometry colouredGeometry(double cx, double cy) { + cvc::geometry g = boxGeometry(cx, cy, 1.0, 1.0, 1.0, 1.0); + for (size_t i = 0; i < g.points().size(); ++i) + g.colors().push_back({(i % 3) / 2.0, ((i / 3) % 3) / 2.0, ((i / 9) % 3) / 2.0}); + return g; +} + +cvc::image checkerImage(int n) { + cvc::image img(n, n, cvc::image::pixel_format::RGBA, cvc::image::data_type::u8); + unsigned char *p = img.data(); + for (int j = 0; j < n; ++j) + for (int i = 0; i < n; ++i) { + unsigned char *q = p + 4 * (j * n + i); + const bool on = ((i / 8) + (j / 8)) % 2 == 0; + q[0] = on ? 200 : 60; + q[1] = static_cast(i * 255 / n); + q[2] = static_cast(j * 255 / n); + q[3] = 255; + } + return img; +} + +// Raw VTK poly data for the probe actors. +vtkSmartPointer withNormals(vtkPolyData *pd) { + vtkNew f; + f->SetInputData(pd); + f->SplittingOff(); + f->ComputePointNormalsOn(); + f->ComputeCellNormalsOff(); + f->Update(); + auto out = vtkSmartPointer::New(); + out->ShallowCopy(pd); + out->GetPointData()->SetNormals(f->GetOutput()->GetPointData()->GetNormals()); + return out; +} + +vtkSmartPointer spherePoly(double r) { + vtkNew s; + s->SetRadius(r); + s->SetThetaResolution(24); + s->SetPhiResolution(18); + s->Update(); + auto pd = vtkSmartPointer::New(); + pd->DeepCopy(s->GetOutput()); + return pd; +} + +vtkSmartPointer terrainPoly(int n, double x0, double y0, double s) { + auto pts = vtkSmartPointer::New(); + pts->SetDataTypeToFloat(); + auto tc = vtkSmartPointer::New(); + tc->SetNumberOfComponents(2); + for (int j = 0; j < n; ++j) + for (int i = 0; i < n; ++i) { + pts->InsertNextPoint(x0 + s * i / (n - 1), y0 + s * j / (n - 1), 0.05 * ((i + j) % 3)); + tc->InsertNextTuple2(double(i) / (n - 1), double(j) / (n - 1)); + } + auto polys = vtkSmartPointer::New(); + for (int j = 0; j + 1 < n; ++j) + for (int i = 0; i + 1 < n; ++i) { + const vtkIdType a = j * n + i, b = a + 1, c = a + n, d = c + 1; + const vtkIdType t1[3] = {a, b, d}, t2[3] = {a, d, c}; + polys->InsertNextCell(3, t1); + polys->InsertNextCell(3, t2); + } + auto pd = vtkSmartPointer::New(); + pd->SetPoints(pts); + pd->SetPolys(polys); + pd->GetPointData()->SetTCoords(tc); + return withNormals(pd); +} + +vtkSmartPointer ribbonPoly(int n, double x0, double y0, double z) { + auto pts = vtkSmartPointer::New(); + pts->SetDataTypeToFloat(); + auto polys = vtkSmartPointer::New(); + for (int k = 0; k < n; ++k) { + pts->InsertNextPoint(x0 + 0.7 * k, y0 - 0.5, z); + pts->InsertNextPoint(x0 + 0.7 * k, y0 + 0.5, z); + } + for (int k = 0; k + 1 < n; ++k) { + const vtkIdType l0 = 2 * k, r0 = l0 + 1, l1 = l0 + 2, r1 = l0 + 3; + const vtkIdType t1[3] = {l0, r0, r1}, t2[3] = {l0, r1, l1}; + polys->InsertNextCell(3, t1); + polys->InsertNextCell(3, t2); + } + auto pd = vtkSmartPointer::New(); + pd->SetPoints(pts); + pd->SetPolys(polys); + return withNormals(pd); +} + +vtkSmartPointer linesPoly(int n, double x0, double y0, double z) { + auto pts = vtkSmartPointer::New(); + pts->SetDataTypeToFloat(); + auto lines = vtkSmartPointer::New(); + for (int k = 0; k < n; ++k) { + pts->InsertNextPoint(x0 + 0.6 * k, y0 + 0.8 * std::cos(k * 0.5), z); + if (k) { + const vtkIdType ids[2] = {k - 1, k}; + lines->InsertNextCell(2, ids); + } + } + auto pd = vtkSmartPointer::New(); + pd->SetPoints(pts); + pd->SetLines(lines); + return pd; +} + +// One cell type of a 3 x 4 point patch: 'v' verts, 's' a triangle strip. (One +// type per data set: VTK 9.5.0's low-memory mapper crashes on its first upload +// of a data set with two or more cell types -- vtkDrawTexturedElements:: +// AppendArrayToTexture flags a not-yet-created buffer dirty; fixed upstream in +// 9.5.1 by 26ef893b5d9. cvcGL nodes always produce a single cell type.) +vtkSmartPointer patchPoly(char kind, double x0, double y0, double z) { + auto pts = vtkSmartPointer::New(); + pts->SetDataTypeToFloat(); + for (int k = 0; k < 12; ++k) + pts->InsertNextPoint(x0 + (k % 4), y0 + (k / 4), z + 0.2 * (k % 2)); + auto cells = vtkSmartPointer::New(); + auto pd = vtkSmartPointer::New(); + pd->SetPoints(pts); + if (kind == 'v') { + for (vtkIdType k = 0; k < 12; ++k) + cells->InsertNextCell(1, &k); + pd->SetVerts(cells); + } else { + const vtkIdType a[8] = {0, 4, 1, 5, 2, 6, 3, 7}, b[8] = {4, 8, 5, 9, 6, 10, 7, 11}; + cells->InsertNextCell(8, a); + cells->InsertNextCell(8, b); + pd->SetStrips(cells); + } + return pd; +} + +// An n x n patch of quads (4-point polygons: VTK triangulates them, so the +// draw needs the cell map) with one RGB colour per quad. +vtkSmartPointer quadsPoly(int n, double x0, double y0, double z) { + auto pts = vtkSmartPointer::New(); + pts->SetDataTypeToFloat(); + for (int j = 0; j <= n; ++j) + for (int i = 0; i <= n; ++i) + pts->InsertNextPoint(x0 + 0.8 * i, y0 + 0.8 * j, z + 0.15 * ((i + j) % 2)); + auto quads = vtkSmartPointer::New(); + auto colours = vtkSmartPointer::New(); + colours->SetNumberOfComponents(3); + colours->SetName("quadColours"); + for (int j = 0; j < n; ++j) + for (int i = 0; i < n; ++i) { + const vtkIdType a = j * (n + 1) + i, ids[4] = {a, a + 1, a + n + 2, a + n + 1}; + quads->InsertNextCell(4, ids); + const unsigned char c[3] = {static_cast(40 + 200 * i / n), + static_cast(40 + 200 * j / n), + static_cast((i + j) % 2 ? 220 : 60)}; + colours->InsertNextTypedTuple(c); + } + auto pd = vtkSmartPointer::New(); + pd->SetPoints(pts); + pd->SetPolys(quads); + pd->GetCellData()->SetScalars(colours); + return withNormals(pd); +} + +// One RGB colour per point. +vtkSmartPointer withPointColours(vtkSmartPointer pd) { + auto colours = vtkSmartPointer::New(); + colours->SetNumberOfComponents(3); + colours->SetName("pointColours"); + for (vtkIdType k = 0; k < pd->GetNumberOfPoints(); ++k) { + const unsigned char c[3] = {static_cast(37 * k % 256), + static_cast(91 * k % 256), + static_cast(151 * k % 256)}; + colours->InsertNextTypedTuple(c); + } + pd->GetPointData()->SetScalars(colours); + return pd; +} + +// Lines with a point normal each (VTK lights lines only with normals and +// non-flat interpolation). +vtkSmartPointer withLineNormals(vtkSmartPointer pd) { + auto normals = vtkSmartPointer::New(); + normals->SetNumberOfComponents(3); + normals->SetName("Normals"); + for (vtkIdType k = 0; k < pd->GetNumberOfPoints(); ++k) { + const double a = 0.3 * k; + normals->InsertNextTuple3(0.3 * std::cos(a), 0.3 * std::sin(a), 0.9); + } + pd->GetPointData()->SetNormals(normals); + return pd; +} + +vtkSmartPointer checkerTexture(int n) { + auto img = vtkSmartPointer::New(); + img->SetDimensions(n, n, 1); + img->AllocateScalars(VTK_UNSIGNED_CHAR, 4); + auto *px = static_cast(img->GetScalarPointer()); + for (int j = 0; j < n; ++j) + for (int i = 0; i < n; ++i) { + unsigned char *q = px + 4 * (j * n + i); + q[0] = static_cast(i * 255 / n); + q[1] = ((i / 4 + j / 4) % 2) ? 220 : 40; + q[2] = 128; + q[3] = 255; + } + auto tex = vtkSmartPointer::New(); + tex->SetInputData(img); + tex->InterpolateOn(); + return tex; +} + +// ── the scene ──────────────────────────────────────────────────────────────── +struct Probe { + std::string name; + vtkSmartPointer mapper; + vtkSmartPointer actor; + // Expected to take the coincident-offset fallback (all four cell types) on + // the Fast path. + bool fallback = false; +}; + +struct Scene { + explicit Scene(cvc::app &app) : sg(app, "lowmem") { sg.setDiagnosticChromeVisible(false); } + SceneGraph sg; + std::unique_ptr sr; + std::vector> nodes; + std::deque probes; // stable references across probe() + int w = 320, h = 240; + + void open() { + sr = std::make_unique(sg, w, h, /*offscreen=*/true, "lowmem"); + // No MSAA: with VTK's default 8x, NVIDIA's resolve differs by 1 LSB at a few + // edge pixels from one frame to the next for an identical GL stream, so two + // stock frames would not compare equal either. + sr->renderWindow()->SetMultiSamples(0); + sr->setBackground(0.18, 0.22, 0.30); + sr->setCamera(-14, -20, 16, 0, 0, 0, 0, 0, 1, 40.0, 1.0, 200.0); + for (auto &p : probes) + sr->renderer()->AddActor(p.actor); + } + + std::shared_ptr node(const std::string &name) { + auto n = sg.getGraphicsRoot()->addGraphicsChild(name); + nodes.push_back(n); + return n; + } + + Probe &probe(const std::string &name, vtkSmartPointer pd) { + Probe p; + p.name = name; + p.mapper = vtkSmartPointer::New(); + p.mapper->SetInputData(pd); + p.mapper->ScalarVisibilityOff(); + p.actor = vtkSmartPointer::New(); + p.actor->SetMapper(p.mapper); + p.actor->GetProperty()->SetSpecular(0.3); + p.actor->GetProperty()->SetSpecularPower(20); + probes.push_back(p); + return probes.back(); + } + + // Every LowMemoryPolyDataMapper drawing in this scene. + std::vector mappers() { + std::vector out; + vtkActorCollection *actors = sr->renderer()->GetActors(); + actors->InitTraversal(); + while (vtkActor *a = actors->GetNextActor()) + if (auto *m = LowMemoryPolyDataMapper::SafeDownCast(a->GetMapper())) + out.push_back(m); + return out; + } + LowMemoryPolyDataMapper::Stats stats() { + LowMemoryPolyDataMapper::Stats s; + for (auto *m : mappers()) { + const auto &t = m->stats(); + s.draws += t.draws; + s.stockDraws += t.stockDraws; + s.cellTypesSkipped += t.cellTypesSkipped; + s.coincidentFallbacks += t.coincidentFallbacks; + s.programLookups += t.programLookups; + s.programLookupsSkipped += t.programLookupsSkipped; + } + return s; + } + void resetCounters() { + for (auto *m : mappers()) + m->resetStats(); + for (auto &p : probes) + p.mapper->clearTally(); + } + + // Render one frame; the GL calls it issued. + GLCalls frame() { + const GLCalls before = glcallsRead(); + sr->render(); + return glcallsRead() - before; + } + std::vector rgba() { + vtkNew px; + sr->renderWindow()->GetRGBACharPixelData(0, 0, w - 1, h - 1, /*front=*/0, px); + const unsigned char *p = px->GetPointer(0); + return std::vector(p, p + px->GetNumberOfValues()); + } +}; + +void buildCity(Scene &s) { + // Nodes: every kind of GeometryNode a scene uses. + auto veh = s.node("veh"); + veh->setGeometry(boxGeometry(0, 0, 1.2, 1.6, 0.9, 0.7)); + veh->setUseSingleColor(true); + veh->setColor(0.3, 0.5, 0.9); + const double pose[16] = {0.8, -0.6, 0, 2.5, 0.6, 0.8, 0, -1.0, 0, 0, 1, 0.1, 0, 0, 0, 1}; + veh->setPoseMatrix(pose); + + auto ground = s.node("ground"); + ground->setGeometry(groundGeometry(40, 14.0)); + ground->setUseSingleColor(true); + ground->setColor(1, 1, 1); + ground->setAmbient(0.4); + ground->setDiffuse(0.7); + ground->setTexture(checkerImage(64)); + + auto ribbon = s.node("ribbon"); + ribbon->setGeometry(ribbonGeometry(30, -9, 4, 0.9)); + ribbon->setUseSingleColor(true); + ribbon->setColor(0.98, 0.85, 0.3); + ribbon->setOpacity(0.6); + ribbon->setDepthOffset(2.0); + + auto trail = s.node("trail"); + trail->setGeometry(polylineGeometry(40, -10, -6, 0.8)); + trail->setRenderMode(GeometryRenderMode::LINES); // after setGeometry, which picks a mode + trail->setUseSingleColor(true); + trail->setColor(0.2, 0.95, 0.3); + trail->setLineWidth(2.0); + + auto cloud = s.node("cloud"); + cloud->setGeometry(cloudGeometry(60, 5, -9, 1.5)); + cloud->setRenderMode(GeometryRenderMode::POINTS); + cloud->setUseSingleColor(true); + cloud->setColor(0.95, 0.3, 0.9); + cloud->setPointSize(4.0); + + auto vcol = s.node("vcol"); + vcol->setUseSingleColor(false); + vcol->setGeometry(colouredGeometry(-4, -3)); + + auto quads = s.node("quads"); + { + cvc::geometry g; + for (int j = 0; j < 4; ++j) + for (int i = 0; i < 4; ++i) + g.points().push_back({1.0 + 0.9 * i, 3.0 + 0.9 * j, 0.4 + 0.1 * ((i + j) % 2)}); + for (unsigned j = 0; j + 1 < 4; ++j) + for (unsigned i = 0; i + 1 < 4; ++i) { + const unsigned a = j * 4 + i; + g.quads().push_back({a, a + 1, a + 5, a + 4}); + } + quads->setGeometry(g); + } + quads->setRenderMode(GeometryRenderMode::QUADS); + quads->setUseSingleColor(true); + quads->setColor(0.6, 0.4, 0.95); + + // Probes: the same kinds as raw actors whose draws are tallied. + { + Probe &p = s.probe("lit", spherePoly(1.4)); + p.actor->GetProperty()->SetColor(0.85, 0.35, 0.25); + vtkNew m; + m->SetElement(0, 3, 6.0); + m->SetElement(1, 3, 3.0); + m->SetElement(2, 3, 1.6); + p.actor->SetUserMatrix(m); + } + { + Probe &p = s.probe("textured", terrainPoly(24, -13, 6, 7)); + p.actor->GetProperty()->SetColor(1, 1, 1); + p.actor->SetTexture(checkerTexture(32)); + } + { + Probe &p = s.probe("translucent", ribbonPoly(24, -6, 8.5, 0.7)); + p.actor->GetProperty()->SetColor(0.3, 0.9, 0.95); + p.actor->GetProperty()->SetOpacity(0.5); + p.mapper->SetRelativeCoincidentTopologyPolygonOffsetParameters(0.0, -2.0); + p.mapper->SetRelativeCoincidentTopologyLineOffsetParameters(0.0, -2.0); + p.mapper->SetRelativeCoincidentTopologyPointOffsetParameter(-2.0); + } + { + Probe &p = s.probe("lines", linesPoly(30, -12, -10, 0.5)); + p.actor->GetProperty()->SetColor(0.95, 0.95, 0.2); + } + { + Probe &p = s.probe("verts", patchPoly('v', 8, -12, 0.8)); + p.actor->GetProperty()->SetColor(0.7, 0.7, 0.9); + p.actor->GetProperty()->SetPointSize(5.0); + } + { + Probe &p = s.probe("strips", patchPoly('s', 2, 9, 0.6)); + p.actor->GetProperty()->SetColor(0.5, 0.9, 0.6); + } + // Representations of a triangle mesh other than the surface. + { + Probe &p = s.probe("wireframe", spherePoly(1.3)); + p.actor->GetProperty()->SetRepresentationToWireframe(); + p.actor->GetProperty()->SetColor(0.95, 0.6, 0.2); + p.actor->SetPosition(-8, 0, 2); + } + { + Probe &p = s.probe("pointsRep", spherePoly(1.3)); // lit: point normals, Gouraud + p.actor->GetProperty()->SetRepresentationToPoints(); + p.actor->GetProperty()->SetPointSize(3.0); + p.actor->GetProperty()->SetColor(0.3, 0.95, 0.95); + p.actor->SetPosition(8, -4, 2); + } + // Edge visibility on a surface puts every draw under the coincident-offset + // rules: with VTK's defaults the polys draw sees the -4 units the empty lines + // block left in the program, so Fast must fall back to all four ... + { + Probe &p = s.probe("edges", spherePoly(1.2)); + p.actor->GetProperty()->EdgeVisibilityOn(); + p.actor->GetProperty()->SetEdgeColor(0.1, 0.1, 0.1); + p.actor->GetProperty()->SetColor(0.8, 0.8, 0.5); + p.actor->SetPosition(0, 6, 2); + p.fallback = true; + } + // ... while one whose own polygon offset is non-zero skips them. + { + Probe &p = s.probe("edgesOffset", ribbonPoly(10, -7, -12, 0.5)); + p.actor->GetProperty()->EdgeVisibilityOn(); + p.actor->GetProperty()->SetEdgeColor(0.9, 0.1, 0.1); + p.actor->GetProperty()->SetColor(0.4, 0.5, 0.9); + p.mapper->SetRelativeCoincidentTopologyPolygonOffsetParameters(0.0, -2.0); + p.mapper->SetRelativeCoincidentTopologyLineOffsetParameters(0.0, -2.0); + p.mapper->SetRelativeCoincidentTopologyPointOffsetParameter(-2.0); + } + // Scalars: cell colours on quads (cell map), point colours on a sphere. + { + Probe &p = s.probe("cellColours", quadsPoly(5, -3, -9, 0.3)); + p.mapper->ScalarVisibilityOn(); + p.mapper->SetScalarModeToUseCellData(); + p.mapper->SetColorModeToDirectScalars(); + } + { + Probe &p = s.probe("pointColours", withPointColours(spherePoly(1.1))); + p.mapper->ScalarVisibilityOn(); + p.mapper->SetScalarModeToUsePointData(); + p.mapper->SetColorModeToDirectScalars(); + p.actor->SetPosition(-3, 3, 3); + } + // Lines: flat with normals (unlit), Gouraud with normals and wide (lit, + // instanced), and drawn as points. + { + Probe &p = s.probe("flatLines", withLineNormals(linesPoly(20, -9, 8, 1.0))); + p.actor->GetProperty()->SetInterpolationToFlat(); + p.actor->GetProperty()->SetColor(0.9, 0.3, 0.3); + } + { + Probe &p = s.probe("litLines", withLineNormals(linesPoly(20, -9, 10, 1.2))); + p.actor->GetProperty()->SetInterpolationToGouraud(); + p.actor->GetProperty()->SetLineWidth(3.0); + p.actor->GetProperty()->SetColor(0.3, 0.3, 0.9); + } + { + Probe &p = s.probe("linePoints", linesPoly(20, 0, -6, 1.4)); + p.actor->GetProperty()->SetRepresentationToPoints(); + p.actor->GetProperty()->SetPointSize(4.0); + p.actor->GetProperty()->SetColor(0.95, 0.95, 0.95); + } +} + +void addLights(SceneGraph &sg) { + sg.addSpotLight(-12, -16, 26, 0, 0, 0, 40.0, 1.0, 0.97, 0.9, 1.0); + sg.addFillLight(14, 10, 20, 0, 0, 0, 0.8, 0.85, 1.0, 0.4); +} + +struct PathRun { + GLCalls first, steady; // whole frame + Trace firstTrace, steadyTrace; // whole frame, with stage markers + std::vector firstPx, steadyPx; // RGBA + LowMemoryPolyDataMapper::Stats firstStats, steadyStats; + std::vector probeCalls; // steady frame, per probe + std::vector probeDraws; // steady frame, per probe + std::vector probeFirst; // first frame, per probe + std::vector probeFirstDraws; +}; + +// Two frames on one draw path: `first` (a real shadow bake when shadows are on: +// every opaque mapper draws twice, rebuilding its shader for each pass) and +// `steady` (re-uses the bake; nothing rebuilt). +PathRun runPath(Scene &s, DrawPath path) { + LowMemoryPolyDataMapper::setDrawPath(path); + // One frame on this path first, so the counted frames start from the GL + // state this path leaves (on desktop the empty verts agent's glPointSize is + // cached state, so the first frame after a switch would differ by one call). + s.sr->render(); + PathRun r; + // vtkShadowMapBakerPass only re-bakes when a light or prop changed since the + // last bake; touch one so the first frame really bakes. + s.sg.invalidateShadowBake(); + if (vtkActor *a = s.sr->renderer()->GetActors()->GetLastActor()) + a->Modified(); + s.resetCounters(); + cvcgl_test::glcallsTraceStart(); + r.first = s.frame(); + r.firstTrace = cvcgl_test::glcallsTraceStop(); + r.firstStats = s.stats(); + for (auto &p : s.probes) { + r.probeFirst.push_back(p.mapper->calls); + r.probeFirstDraws.push_back(p.mapper->draws); + } + r.firstPx = s.rgba(); + s.resetCounters(); + cvcgl_test::glcallsTraceStart(); + r.steady = s.frame(); + r.steadyTrace = cvcgl_test::glcallsTraceStop(); + r.steadyStats = s.stats(); + for (auto &p : s.probes) { + r.probeCalls.push_back(p.mapper->calls); + r.probeDraws.push_back(p.mapper->draws); + } + r.steadyPx = s.rgba(); + return r; +} + +// ── pixels, against the renderer's own self-variance ───────────────────────── +// Two RGBA frames: how many bytes differ, by how much at most, and where first. +struct PixelDiff { + long bytes = 0; // -1: frames of different sizes, or empty + int maxDelta = 0; + long first = -1; // the pixel (index) of the first differing byte +}; + +PixelDiff pixelDiff(const std::vector &a, const std::vector &b) { + PixelDiff d; + if (a.size() != b.size() || a.empty()) { + d.bytes = -1; + return d; + } + for (size_t i = 0; i < a.size(); ++i) { + const int delta = std::abs(int(a[i]) - int(b[i])); + if (!delta) + continue; + if (d.first < 0) + d.first = static_cast(i / 4); + ++d.bytes; + d.maxDelta = std::max(d.maxDelta, delta); + } + return d; +} + +std::string describe(const PixelDiff &d, int w) { + if (d.bytes < 0) + return "frames of different sizes (or none)"; + if (d.bytes == 0) + return "byte-identical"; + return fmt("%.0f differing bytes, max delta %.0f", double(d.bytes), double(d.maxDelta)) + + fmt(", first at pixel (%.0f, %.0f)", double(d.first % w), double(d.first / w)); +} + +// Pixels are judged against what the renderer itself does with one GL stream. +// Two Stock frames whose GL traces are identical -- VTK's own draws, every +// call with its arguments and uniform values, in order, in one window -- can +// differ only by what the rasteriser does with that stream. NVIDIA and Mesa +// llvmpipe give the same bytes every time. Apple's software renderer (macOS +// CI, "Apple Software Renderer", GL 4.1 APPLE-23.1.1) does not, and not only +// across windows. Measured on the first attempts of macOS CI runs, every +// difference by a single LSB (max delta 1), in some runs and not in others: +// - in one window, shadows on (never off): AllCellTypes, Fast and a Stock +// frame right after Fast against Stock 1 to 7 bytes apart, Fast after a +// mapper's resources were released 1 to 8 -- in runs where every pair of +// Stock frames sampled in that window was byte-identical, so no pair a run +// samples can bound what its other frames do; +// - across a program recompile (the shader cache's programs released) or a +// new window (new context, new cache): Fast, and Stock against Stock -- +// the fast path not involved -- 1 to 10 bytes apart. +// +// Exact mode (every renderer but Apple's): every pixel check is byte for byte, +// and a sampled pair -- two Stock frames of one GL trace, in one window, +// neither right after Fast -- that differs fails the run on its own. +// +// Apple-tolerance mode (the GL version or renderer string names Apple): every +// pixel check -- in one window, across a recompile, across a new window -- +// passes iff its two frames are byte-identical or differ by LSB rounding +// alone: no byte off by more than 1 (kAppleLsbMaxDelta), at most 64 bytes +// (kAppleLsbMaxBytes, ~6x the most measured). The envelope is the +// renderer's, fixed, not the run's: no frame widens it, so no frame can +// excuse itself, and a shading change (a colour off by 2 or more) or a +// changed region (more than 64 bytes) fails. The pairs of Stock frames are +// held to the same envelope, and those that differ are printed as well, a +// diagnostic of the renderer's self-variance in that run; they excuse +// nothing. Byte counts that are no multiple of 3 (1, 2, 4, 5, ...) mean +// pixels that changed in some colour channels and not the others: rounding, +// not a shading change. Hence the relaxation is Apple's alone, where it is +// needed; the GL traces stay the exact oracle on every renderer, and point +// picking has its own rule (comparePicking()). +constexpr long kAppleLsbMaxBytes = 64; +constexpr int kAppleLsbMaxDelta = 1; + +// Apple-tolerance mode: identical, or LSB rounding at a few pixels. +bool withinAppleLsbEnvelope(const PixelDiff &d) { + return d.bytes >= 0 && d.bytes <= kAppleLsbMaxBytes && d.maxDelta <= kAppleLsbMaxDelta; +} + +std::string appleLsbEnvelope() { + return fmt("Apple LSB envelope (delta<=%.0f, bytes<=%.0f)", double(kAppleLsbMaxDelta), + double(kAppleLsbMaxBytes)); +} + +// The renderer the frames are drawn by, as renderAvailable() reads it, and +// whether its pixels are judged in Apple-tolerance mode (exact otherwise). +std::string g_glRenderer, g_glVersion; +bool g_appleTolerance = false; + +// Every pixel check's verdict, and its detail: byte for byte, or in +// Apple-tolerance mode the fixed LSB envelope -- the same rule wherever the +// two frames come from. +bool pixelsMatch(const PixelDiff &d) { + return g_appleTolerance ? withinAppleLsbEnvelope(d) : d.bytes == 0; +} + +std::string pixelVerdict(const PixelDiff &d, int w) { + std::string detail = describe(d, w); + if (g_appleTolerance && d.bytes > 0) + detail += (pixelsMatch(d) ? ": within the " : ": BEYOND the ") + appleLsbEnvelope(); + return detail; +} + +struct SelfVarianceSample { + std::string scope, what; + PixelDiff d; + int w; +}; +std::vector g_noisy; // the sampled pairs that differed +std::map g_pairs; // pairs sampled, per scope +std::map g_pairsSkipped; // not sampled (traces differ), per scope + +struct PixelCheck { + std::string what; + PixelDiff d; + int w; +}; +std::vector g_pixelChecks; + +void sampleSelfVariance(const std::string &scope, const std::string &what, const Trace &ta, + const Trace &tb, const std::vector &a, + const std::vector &b, int w) { + std::string why; + if (!sameTrace(ta, tb, &why)) { + ++g_pairsSkipped[scope]; // not one GL stream: what the frames differ by is not noise + std::printf(" self-variance: %s not sampled, GL traces differ -- %s\n", what.c_str(), + why.c_str()); + return; + } + const PixelDiff d = pixelDiff(a, b); + if (d.bytes < 0) { + ++g_pairsSkipped[scope]; + return; + } + ++g_pairs[scope]; + if (!g_appleTolerance) { // exact mode: the pair is itself a check + check(d.bytes == 0, what + " is byte-identical on a deterministic renderer", + d.bytes ? describe(d, w) + ", identical GL traces" : std::string()); + return; + } + // Apple-tolerance mode: the pair is held to the same fixed envelope as every + // other pixel check, and printed when it differs at all. + check(withinAppleLsbEnvelope(d), + what + " is byte-identical or within the " + appleLsbEnvelope() + " on Apple's renderer", + d.bytes ? describe(d, w) + ", identical GL traces" : std::string()); + if (d.bytes == 0) + return; + g_noisy.push_back({scope, what, d, w}); // a diagnostic: it excuses nothing + std::printf(" self-variance (diagnostic): %s -- %s, identical GL traces\n", what.c_str(), + pixelVerdict(d, w).c_str()); +} + +// Two Stock runs of one draw path: the first frames (a shadow bake when +// shadows are on) and the steady frames, which sample that bake -- so the +// steady pair counts only if the first frames' traces match too. +void sampleSelfVariance(const std::string &scope, const std::string &what, const PathRun &a, + const PathRun &b, int w) { + const bool firstSame = sameTrace(a.firstTrace, b.firstTrace, nullptr); + sampleSelfVariance(scope, what + ", first frame", a.firstTrace, b.firstTrace, a.firstPx, + b.firstPx, w); + if (firstSame) + sampleSelfVariance(scope, what + ", steady frame", a.steadyTrace, b.steadyTrace, a.steadyPx, + b.steadyPx, w); + else + ++g_pairsSkipped[scope]; +} + +// Deferred: judged by checkPixels(), after the run's samples are printed. +void samePixels(const std::string &what, const std::vector &a, + const std::vector &b, int w) { + g_pixelChecks.push_back({what, pixelDiff(a, b), w}); +} + +void checkPixels() { + if (g_pixelChecks.empty()) + return; + if (!g_appleTolerance) { + std::printf("pixels (exact mode: byte for byte)\n"); + } else { + std::printf("pixels (apple-tolerance mode: every check, in one window or across a recompile " + "or a new window, byte-identical or within the %s; the pairs of Stock frames " + "sampled are held to the same envelope and excuse nothing)\n", + appleLsbEnvelope().c_str()); + std::set scopes; + for (const auto &p : g_pairs) + scopes.insert(p.first); + for (const auto &p : g_pairsSkipped) + scopes.insert(p.first); + for (const auto &scope : scopes) { + int noisy = 0; + for (const auto &n : g_noisy) + noisy += n.scope == scope; + std::string line = " self-variance (diagnostic): " + scope + + fmt(": %.0f pairs of Stock frames, ", double(g_pairs[scope])); + line += noisy ? fmt("%.0f differ (listed above)", double(noisy)) : std::string("none differ"); + if (g_pairsSkipped[scope]) + line += fmt("; %.0f not sampled", double(g_pairsSkipped[scope])); + std::printf("%s\n", line.c_str()); + } + } + for (const auto &c : g_pixelChecks) + check(pixelsMatch(c.d), c.what, pixelVerdict(c.d, c.w)); +} + +// Every pixel the background colour? (a frame that drew nothing) +bool drewSomething(const std::vector &px) { + if (px.size() < 8) + return false; + for (size_t i = 4; i < px.size(); i += 4) + if (px[i] != px[0] || px[i + 1] != px[1] || px[i + 2] != px[2]) + return true; + return false; +} + +// ── 1. policy ──────────────────────────────────────────────────────────────── +void testPolicy() { + std::printf("policy\n"); + check(cvc::gl::lowMemoryMapperPolicy() == LowMemoryMapperPolicy::Force, + "CVCGL_LOWMEM_MAPPER=force is read on first use"); + check(LowMemoryPolyDataMapper::SafeDownCast(cvc::gl::newPolyDataMapper()) != nullptr, + "Force: newPolyDataMapper() is a LowMemoryPolyDataMapper"); + const bool factoryLowMem = + vtkSmartPointer::New()->IsA("vtkOpenGLLowMemoryPolyDataMapper") != 0; + cvc::gl::setLowMemoryMapperPolicy(LowMemoryMapperPolicy::Off); + check(LowMemoryPolyDataMapper::SafeDownCast(cvc::gl::newPolyDataMapper()) == nullptr, + "Off: newPolyDataMapper() is the factory's mapper"); + cvc::gl::setLowMemoryMapperPolicy(LowMemoryMapperPolicy::Auto); + check((LowMemoryPolyDataMapper::SafeDownCast(cvc::gl::newPolyDataMapper()) != nullptr) == + factoryLowMem, + "Auto: LowMemoryPolyDataMapper exactly where the factory hands out the low-memory one", + factoryLowMem ? "factory: low-memory" : "factory: classic"); + cvc::gl::setLowMemoryMapperPolicy(LowMemoryMapperPolicy::Force); + if (!std::getenv("CVCGL_LOWMEM_DRAW")) + check(LowMemoryPolyDataMapper::drawPath() == DrawPath::Fast, "the default draw path is Fast"); +} + +// Can this build rasterise at all? Asked of a plain node in a scene of its own. +bool renderAvailable(cvc::app &app) { + bool drew = false; + { + Scene s(app); + auto n = s.node("control"); + n->setGeometry(boxGeometry(0, 0, 0, 5, 5, 0.5)); + n->setColor(1, 1, 1); + s.w = 48; + s.h = 48; + s.open(); + s.sr->setCamera(0, 0, 50, 0, 0, 0, 0, 1, 0, 30.0, 1.0, 200.0); + s.sr->render(); + drew = drewSomething(s.rgba()); + // Which renderer the pixel and round-trip checks below ran on, and so how + // its pixels are judged (see checkPixels()). + if (auto *win = vtkOpenGLRenderWindow::SafeDownCast(s.sr->renderWindow())) { + const char *report = win->ReportCapabilities(); + const std::string caps = report ? report : ""; + const auto value = [&caps](const std::string &key) { + const size_t at = caps.find(key); + if (at == std::string::npos) + return std::string(); + const size_t end = std::min(caps.find('\n', at), caps.size()); + const size_t from = std::min(caps.find_first_not_of(' ', at + key.size()), end); + return caps.substr(from, end - from); + }; + g_glRenderer = value("OpenGL renderer string:"); + g_glVersion = value("OpenGL version string:"); + const auto apple = [](const std::string &v) { + return v.find("APPLE") != std::string::npos || v.find("Apple") != std::string::npos; + }; + g_appleTolerance = apple(g_glVersion) || apple(g_glRenderer); + std::printf(" maximum hardware line width: %g\n", + double(win->GetMaximumHardwareLineWidth())); + } + const char *mode = g_appleTolerance ? "apple-tolerance" : "exact"; + std::string rule = "every pair of Stock frames and every check byte for byte"; + if (g_appleTolerance) + rule = "every check, in one window or across a recompile or a new window, byte-identical or " + "within the " + + appleLsbEnvelope() + + ", pairs of Stock frames held to the same envelope (they excuse nothing); point " + "picking by <= 1 point per prop, cell picking exact"; + std::printf(" pixel checks: %s (OpenGL renderer \"%s\", version \"%s\") -- %s\n", mode, + g_glRenderer.c_str(), g_glVersion.c_str(), rule.c_str()); + } + if (drew) + return true; + const char *require = std::getenv("CVC_REQUIRE_RENDER"); + if (require && *require && std::string(require) != "0") + check(false, "CVC_REQUIRE_RENDER is set, but this build did not rasterise"); + else + std::printf(" skipped: this build did not rasterise\n"); + return false; +} + +// ── 2-5. replica, derived Fast trace, pixels, savings ──────────────────────── +// The replica (AllCellTypes) against Stock, call for call, and Fast against the +// trace derived from it, for one frame of each kind. +// CVCGL_LOWMEM_DUMP=: write the traces of each failing comparison there +// (_.txt, one call per line), for a diff. +void dumpTraces(const std::string &what, const std::vector> &ts) { + static int dumped = 0; + const char *dir = std::getenv("CVCGL_LOWMEM_DUMP"); + if (!dir || !*dir) + return; + ++dumped; + for (const auto &[name, t] : ts) { + const std::string path = std::string(dir) + "/" + std::to_string(dumped) + "_" + name + ".txt"; + if (FILE *f = std::fopen(path.c_str(), "w")) { + std::fprintf(f, "# %s\n", what.c_str()); + for (const auto &r : t) + std::fprintf(f, "%s\n", cvcgl_test::glcallsDescribe(r).c_str()); + std::fclose(f); + } + } +} + +// mayDraw: whether the frame has replica draws at all (a cell-picking trace is +// selection passes only, every one of them the stock draw). +void checkTraces(const std::string &what, const Trace &stock, const Trace &all, const Trace &fast, + const LowMemoryPolyDataMapper::Stats &fastStats, bool mayDraw = true) { + std::string why; + check(countStage(stock, DrawStage::DrawBegin) == 0 && + countStage(all, DrawStage::RebindBegin) == 0 && + countStage(all, DrawStage::AgentSkipped) == 0 && + (countStage(all, DrawStage::DrawBegin) > 0) == mayDraw, + what + ": Stock reports no stages; AllCellTypes looks every program up and runs every " + "cell type"); + // (Each comparison runs before its message is built: argument order is unspecified.) + const bool replica = sameTrace(glOnly(all), glOnly(stock), &why); + if (!replica) + dumpTraces(what, {{"stock", stock}, {"all", all}}); + check(replica, + what + ": AllCellTypes trace == Stock trace (every call, program/unit, bytes, values)", + replica ? fmt("%.0f calls", double(stock.size())) : why); + why.clear(); + // The draw state is read off the full AllCellTypes trace, before the blocks + // go: what each draw saw there is what it must see on Fast. + const Trace allState = asDrawState(all), fastState = asDrawState(fast); + const Expected e = expectFast(allState); + const Trace got = glOnly(fastState); + const bool derived = sameTrace(got, glOnly(e.trace), &why); + if (!derived) + dumpTraces(what, {{"all", all}, {"fast", fast}, {"expected", e.trace}, {"fast_state", got}}); + check(derived, + what + ": Fast trace == AllCellTypes trace minus the empty cell-type blocks whose " + "removal no draw call's uniforms see (cached point size / line width compared " + "at the draws)", + derived ? fmt("%.0f calls; %.0f blocks removed, %.0f kept", double(got.size()), + double(e.blocksRemoved), double(e.blocksKept)) + + (e.firstFallback.empty() ? "" : "; kept e.g. " + e.firstFallback) + : why); + why.clear(); + // Fast's own trace, replayed: every draw call sees the uniforms it sees on + // AllCellTypes, and the frame leaves what the next frame's draws read. + const UniformView allUniforms = replayUniforms(allState), + fastUniforms = replayUniforms(fastState); + const bool uniforms = sameUniformsSeen(allUniforms, fastUniforms, allState, &why); + if (!uniforms) + dumpTraces(what + " (uniforms)", {{"all", allState}, {"fast", fastState}}); + check(uniforms, + what + ": every Fast draw call sees AllCellTypes' uniform state (every location of " + "its program), and the frame leaves the values the next frame reads", + uniforms ? fmt("%.0f draw calls", double(fastUniforms.calls.size())) : why); + // The mapper's own markers and stats: counts only. + const int skipped = countStage(fast, DrawStage::AgentSkipped); + check(e.blocksRemoved == skipped && (skipped > 0) == mayDraw && + static_cast(skipped) == double(fastStats.cellTypesSkipped), + what + ": Fast skipped as many blocks as derived", + fmt("derived %.0f, Fast skipped %.0f (stats %.0f)", double(e.blocksRemoved), + double(skipped), double(fastStats.cellTypesSkipped))); + const int looked = countStage(fast, DrawStage::LookupBegin); + const int rebound = countStage(fast, DrawStage::RebindBegin); + int fastFallbacks = 0; + for (const auto &r : fast) + fastFallbacks += isStage(r, DrawStage::DrawBegin) && r.ctx == 0; + check(looked + rebound == e.lookups && countStage(fast, DrawStage::DrawBegin) == e.draws && + e.fallbacks == fastFallbacks && + static_cast(e.fallbacks) == double(fastStats.coincidentFallbacks), + what + ": every AllCellTypes lookup is a Fast lookup or re-bind; same draws, same " + "fallbacks", + fmt("lookups %.0f = %.0f + %.0f re-bound", double(e.lookups), double(looked), + double(rebound)) + + fmt(", fallbacks derived %.0f, Fast %.0f (stats %.0f)", double(e.fallbacks), + double(fastFallbacks), double(fastStats.coincidentFallbacks))); +} + +// The hardware selector's passes on one draw path: what it selected, the GL +// trace of the selection, the stats. +struct Pick { + std::string signature; + std::vector> hits; // (prop id, items), in prop-id order + Trace trace; + LowMemoryPolyDataMapper::Stats stats; +}; + +// Apple-tolerance mode, point picking: the same props, each picked within +// `tolerance` items of the reference. Where the counts differ, *diff lists them. +bool pickCountsWithin(const Pick &ref, const Pick &p, double tolerance, const char *name, + std::string *diff) { + if (p.hits.size() != ref.hits.size()) + return false; + for (size_t i = 0; i < ref.hits.size(); ++i) { + if (p.hits[i].first != ref.hits[i].first || + std::abs(p.hits[i].second - ref.hits[i].second) > tolerance) + return false; + if (p.hits[i].second != ref.hits[i].second) + *diff += fmt("; prop %.0f: Stock %.0f, ", ref.hits[i].first, ref.hits[i].second) + name + + fmt(" %.0f", p.hits[i].second); + } + return true; +} + +Pick pick(Scene &s, DrawPath p, int association) { + LowMemoryPolyDataMapper::setDrawPath(p); + s.sr->render(); // the same state before each path's selection + s.resetCounters(); + vtkNew sel; + sel->SetRenderer(s.sr->renderer()); + sel->SetArea(0, 0, s.w - 1, s.h - 1); + sel->SetFieldAssociation(association); + cvcgl_test::glcallsTraceStart(); + vtkSmartPointer res = vtk::TakeSmartPointer(sel->Select()); + Pick out; + out.trace = cvcgl_test::glcallsTraceStop(); + auto &hits = out.hits; + for (unsigned i = 0; res && i < res->GetNumberOfNodes(); ++i) { + vtkSelectionNode *n = res->GetNode(i); + hits.emplace_back(n->GetProperties()->Get(vtkSelectionNode::PROP_ID()), + n->GetSelectionList() ? double(n->GetSelectionList()->GetNumberOfTuples()) + : 0.0); + } + std::sort(hits.begin(), hits.end()); + for (const auto &h : hits) + out.signature += fmt("[%.0f:%.0f]", h.first, h.second); + out.stats = s.stats(); + return out; +} + +// Picking takes the stock draw on every path: identical selections and GL. +// (Point picking first renders one ordinary frame for the depth buffer -- +// vtkOpenGLHardwareSelector::BeginSelection -- with no selector set, so those +// draws take the path under test; the trace check covers them like a frame.) +void comparePicking(Scene &s, const std::string &label) { + // Replica draws in one ordinary frame (vertex visibility is stock anyway). + LowMemoryPolyDataMapper::setDrawPath(DrawPath::Fast); + s.sr->render(); + s.resetCounters(); + s.sr->render(); + const double frameDraws = double(s.stats().draws - s.stats().stockDraws); + for (int assoc : + {vtkDataObject::FIELD_ASSOCIATION_CELLS, vtkDataObject::FIELD_ASSOCIATION_POINTS}) { + const bool points = assoc == vtkDataObject::FIELD_ASSOCIATION_POINTS; + const std::string what = label + (points ? ": point picking" : ": cell picking"); + // Warm: the selection passes' shader variants are compiled by the first + // selection on any path; compare the ones after. + for (DrawPath p : {DrawPath::Stock, DrawPath::Fast}) + pick(s, p, assoc); + const Pick stock = pick(s, DrawPath::Stock, assoc); + const Pick all = pick(s, DrawPath::AllCellTypes, assoc); + const Pick fast = pick(s, DrawPath::Fast, assoc); + const Pick stock2 = pick(s, DrawPath::Stock, assoc); + { + std::string why; + const bool reproducible = sameTrace(glOnly(stock2.trace), glOnly(stock.trace), &why); + if (!reproducible) + dumpTraces(what + " (Stock twice)", {{"stock", stock.trace}, {"stock2", stock2.trace}}); + check(reproducible, what + ": the selection trace is reproducible (Stock twice)", + reproducible ? fmt("%.0f calls", double(stock.trace.size())) : why); + } + const double ordinary = points ? frameDraws : 0.0; + check(double(fast.stats.draws - fast.stats.stockDraws) == ordinary && + double(all.stats.draws - all.stats.stockDraws) == ordinary && + fast.stats.stockDraws > 0, + what + ": every selection-pass draw takes the stock path, on every path", + fmt("stock %.0f of %.0f draws", double(fast.stats.stockDraws), double(fast.stats.draws)) + + fmt(" (+%.0f in the ordinary depth frame)", ordinary)); + const bool sameSelection = !stock.signature.empty() && fast.signature == stock.signature && + all.signature == stock.signature; + // Apple's renderer, point picking only: the selector's point pass picked + // 1117 points of a prop on one path and 1118 on another, in runs where + // AllCellTypes' GL trace equalled Stock's -- the rasteriser's rounding at + // a point's edge, like its pixels. There a path may pick 1 point more or + // fewer per prop (the same props); cell picking, and every other renderer, + // stay exact. + std::string within; + const bool nearSelection = !sameSelection && g_appleTolerance && points && + !stock.signature.empty() && + pickCountsWithin(stock, all, 1.0, "AllCellTypes", &within) && + pickCountsWithin(stock, fast, 1.0, "Fast", &within); + std::string detail = sameSelection ? stock.signature + : "Stock " + stock.signature + " AllCellTypes " + + all.signature + " Fast " + fast.signature; + if (nearSelection) + detail += ": within 1 point per prop on Apple's renderer" + within; + else if (g_appleTolerance && points) + detail += sameSelection ? " (Apple's renderer: <= 1 point per prop)" + : ": BEYOND 1 point per prop on Apple's renderer"; + check(sameSelection || nearSelection, what + " selects the same on every path", detail); + checkTraces(what, stock.trace, all.trace, fast.trace, fast.stats, points); + } + LowMemoryPolyDataMapper::setDrawPath(DrawPath::Fast); +} + +void comparePaths(Scene &s, const std::string &label, bool shadows) { + std::printf("%s\n", label.c_str()); + // Settle: build every shader both ways first so no path pays a first compile + // (Stock last: the first counted Stock run must not follow a Fast one). + for (DrawPath p : {DrawPath::Fast, DrawPath::Stock}) + runPath(s, p); + // Pairs of the Stock runs before Fast's (one after AllCellTypes) are checked + // byte-identical -- on Apple's renderer, within its LSB envelope, and printed + // as its self-variance; the Stock run right after Fast is judged, like the + // AllCellTypes and Fast frames. + const PathRun stock = runPath(s, DrawPath::Stock); + const PathRun stock2 = runPath(s, DrawPath::Stock); + const PathRun all = runPath(s, DrawPath::AllCellTypes); + const PathRun stock3 = runPath(s, DrawPath::Stock); + const PathRun fast = runPath(s, DrawPath::Fast); + const PathRun afterFast = runPath(s, DrawPath::Stock); + + check(drewSomething(stock.steadyPx), label + ": the scene draws"); + std::string why; + bool reproducible = true; + for (const PathRun *again : {&stock2, &stock3, &afterFast}) + reproducible = reproducible && sameTrace(again->firstTrace, stock.firstTrace, &why) && + sameTrace(again->steadyTrace, stock.steadyTrace, &why); + check(reproducible, label + ": the GL trace is reproducible (Stock four times, same trace)", + reproducible ? fmt("%.0f calls", double(stock.steadyTrace.size())) : why); + sampleSelfVariance(label, label + ": Stock twice", stock, stock2, s.w); + sampleSelfVariance(label, label + ": Stock twice", stock, stock3, s.w); + sampleSelfVariance(label, label + ": Stock twice", stock2, stock3, s.w); + checkTraces(label + ", first frame" + (shadows ? " (shadow bake)" : ""), stock.firstTrace, + all.firstTrace, fast.firstTrace, fast.firstStats); + checkTraces(label + ", steady frame", stock.steadyTrace, all.steadyTrace, fast.steadyTrace, + fast.steadyStats); + samePixels(label + ": AllCellTypes pixels == Stock, first frame", stock.firstPx, all.firstPx, + s.w); + samePixels(label + ": AllCellTypes pixels == Stock, steady frame", stock.steadyPx, all.steadyPx, + s.w); + samePixels(label + ": Fast pixels == Stock, first frame", stock.firstPx, fast.firstPx, s.w); + samePixels(label + ": Fast pixels == Stock, steady frame", stock.steadyPx, fast.steadyPx, s.w); + samePixels(label + ": a Stock frame right after Fast == Stock, first frame", stock.firstPx, + afterFast.firstPx, s.w); + samePixels(label + ": a Stock frame right after Fast == Stock, steady frame", stock.steadyPx, + afterFast.steadyPx, s.w); + + std::printf(" frame GL calls, first : Stock %s\n", stock.first.str().c_str()); + std::printf(" Fast %s\n", fast.first.str().c_str()); + std::printf(" frame GL calls, steady : Stock %s\n", stock.steady.str().c_str()); + std::printf(" Fast %s\n", fast.steady.str().c_str()); + check(fast.steady.total() < 0.75 * stock.steady.total(), + label + ": Fast frame issues well under Stock's GL calls", + fmt("steady %.0f -> %.0f, first %.0f", stock.steady.total(), fast.steady.total(), + stock.first.total()) + + fmt(" -> %.0f", fast.first.total())); + check(fast.steady.roundTrips() == stock.steady.roundTrips(), + label + ": Fast adds no desktop round-trip to the frame", + fmt("%.0f vs %.0f", fast.steady.roundTrips(), stock.steady.roundTrips())); + check(fast.steadyStats.programLookups == 0 && fast.steadyStats.programLookupsSkipped > 0, + label + ": Fast steady frame re-binds every program without a lookup", + fmt("lookups %.0f, skipped %.0f", double(fast.steadyStats.programLookups), + double(fast.steadyStats.programLookupsSkipped))); + // The only fallbacks are the probes built to need one. + double expectFallbacks = 0; + for (size_t i = 0; i < s.probes.size(); ++i) + expectFallbacks += s.probes[i].fallback ? fast.probeDraws[i] : 0.0; + check(fast.steadyStats.cellTypesSkipped > 0 && + double(fast.steadyStats.coincidentFallbacks) == expectFallbacks && + fast.steadyStats.stockDraws == 0, + label + ": Fast skipped empty cell types; falls back only where an offset needs it", + fmt("skipped %.0f of %.0f draws x 4, fallbacks %.0f", + double(fast.steadyStats.cellTypesSkipped), double(fast.steadyStats.draws), + double(fast.steadyStats.coincidentFallbacks))); + check(stock.steadyStats.stockDraws == stock.steadyStats.draws && stock.steadyStats.draws > 0, + label + ": Stock draws through VTK's own path"); + if (shadows) + check(fast.firstStats.programLookups > 0, + label + ": a shadow bake (render-pass change) takes the full lookup", + fmt("lookups %.0f", double(fast.firstStats.programLookups))); + + // Per draw, on the probes (steady frame: one draw per probe and pass). + // Without the window's limit no width counts as one the driver lacks. + auto *glWin = vtkOpenGLRenderWindow::SafeDownCast(s.sr->renderWindow()); + const double maxLineWidth = + glWin ? glWin->GetMaximumHardwareLineWidth() : std::numeric_limits::infinity(); + for (size_t i = 0; i < s.probes.size(); ++i) { + const Probe &p = s.probes[i]; + const double nStock = stock.probeDraws[i], nFast = fast.probeDraws[i]; + const double nAll = all.probeDraws[i]; + if (nStock <= 0 || nFast <= 0 || nAll <= 0) { + check(false, label + ": probe " + p.name + " drew"); + continue; + } + const GLCalls &cs = stock.probeCalls[i]; + const GLCalls &cf = fast.probeCalls[i]; + const GLCalls &ca = all.probeCalls[i]; + std::printf(" per draw %-12s Stock %s\n", p.name.c_str(), cs.str(nStock).c_str()); + std::printf(" %-21s Fast %s\n", "", cf.str(nFast).c_str()); + // Desktop round-trips: Stock's draw has none but VTK's own warning about a + // line width the driver cannot draw -- vtkDrawTexturedElements::PreDraw + // (9.5.0) logs it with one glGetString(GL_VERSION) per draw of lines. A + // core profile without wide lines (macOS: GetMaximumHardwareLineWidth() 1) + // does that for litLines (width 3) on every path; where the driver draws + // the width (NVIDIA, llvmpipe) Stock has none. Fast and AllCellTypes add + // none. + const double lineWidth = p.actor->GetProperty()->GetLineWidth(); + vtkPolyData *input = p.mapper->GetInput(); + const bool drawsLines = (input && input->GetNumberOfLines() > 0) || + p.actor->GetProperty()->GetRepresentation() == VTK_WIREFRAME; + const bool vtkWarning = drawsLines && lineWidth > maxLineWidth && + cs.roundTrips() == cs.count("GetString") && + cs.count("GetString") == nStock; + const bool stockExplained = cs.roundTrips() == 0 || vtkWarning; + const double tripsStock = cs.roundTrips() / nStock; + check(stockExplained && cf.roundTrips() / nFast <= tripsStock && + ca.roundTrips() / nAll <= tripsStock, + label + ": " + p.name + + " draws issue no desktop round-trip beyond Stock's (none, but VTK's warning on a " + "line width the driver lacks)", + fmt("per draw: Stock %.1f, AllCellTypes %.1f", tripsStock, ca.roundTrips() / nAll) + + fmt(", Fast %.1f", cf.roundTrips() / nFast) + + (cs.roundTrips() == 0 + ? std::string() + : fmt("; Stock's glGetString %.1f, line width %g, driver max %g", + cs.count("GetString") / nStock, lineWidth, maxLineWidth))); + check(cf.count("DrawArraysInstanced") == cs.count("DrawArraysInstanced") && + cf.count("DrawArraysInstanced") == nFast, + label + ": " + p.name + " one draw call per draw, as Stock"); + if (p.fallback) { + check(cf.total() / nFast == cs.total() / nStock, + label + ": " + p.name + " falls back to all four cell types (Stock's calls)", + fmt("%.1f vs %.1f per draw", cf.total() / nFast, cs.total() / nStock)); + continue; + } + check(cf.count("UseProgram") <= 2 * nFast, label + ": " + p.name + " <= 2 glUseProgram/draw"); + check(cf.count("BindVertexArray") == 2 * nFast, + label + ": " + p.name + " binds the VAO once per draw (+ unbind)", + fmt("%.1f per draw", cf.count("BindVertexArray") / nFast)); + const double perFast = cf.total() / nFast, perStock = cs.total() / nStock; + check(perFast <= 0.45 * perStock, label + ": " + p.name + " Fast <= 45% of Stock calls/draw", + fmt("%.1f -> %.1f", perStock, perFast)); + } + if (g_report) { + LowMemoryPolyDataMapper::setDrawPath(DrawPath::Fast); + s.resetCounters(); + s.sr->render(); + for (auto *m : s.mappers()) { + const auto &t = m->stats(); + vtkPolyData *in = m->GetInput(); + std::printf( + " mapper %-24s points %5lld v/l/p/s %lld/%lld/%lld/%lld: draws %llu skipped %llu " + "fallbacks %llu lookups %llu/%llu\n", + m->GetClassName(), static_cast(in->GetNumberOfPoints()), + static_cast(in->GetNumberOfVerts()), + static_cast(in->GetNumberOfLines()), + static_cast(in->GetNumberOfPolys()), + static_cast(in->GetNumberOfStrips()), static_cast(t.draws), + static_cast(t.cellTypesSkipped), + static_cast(t.coincidentFallbacks), + static_cast(t.programLookups), + static_cast(t.programLookupsSkipped)); + } + for (size_t i = 0; i < s.probes.size(); ++i) + std::printf(" first frame per draw %-12s Stock %s\n %-33s Fast %s\n", + s.probes[i].name.c_str(), + stock.probeFirst[i].str(std::max(1.0, stock.probeFirstDraws[i])).c_str(), "", + fast.probeFirst[i].str(std::max(1.0, fast.probeFirstDraws[i])).c_str()); + } + // The hardware selector over the whole probe set. Not with shadows on: + // cvcGL's shadow pass chain renders props outside the selector's + // BeginRenderProp/EndRenderProp, which VTK rejects ("Too many props") on + // every path alike. + if (!shadows) + comparePicking(s, label); +} + +// ── 6. lifecycle ───────────────────────────────────────────────────────────── +// scope: the scene's label, under which this lifecycle's pairs of Stock frames are counted +// and printed. +void testLifecycle(Scene &s, const std::string &scope) { + std::printf("lifecycle\n"); + LowMemoryPolyDataMapper::setDrawPath(DrawPath::Fast); + s.sr->render(); + + // GetShader() hands out a mutable shader source: that mapper's next draw + // resolves its program in the cache again; nobody else's does. + { + s.sr->render(); + s.resetCounters(); + LowMemoryPolyDataMapper *m = s.probes.front().mapper; + m->GetShader(vtkShader::Fragment); + s.sr->render(); + const auto all = s.stats(); + check(m->stats().programLookups == 1 && all.programLookups == 1, + "GetShader(): that mapper's next draw looks its program up (and only that one)", + fmt("lookups: mapper %.0f, scene %.0f", double(m->stats().programLookups), + double(all.programLookups))); + s.resetCounters(); + m->invalidateProgramCache(); + s.sr->render(); + check(m->stats().programLookups == 1 && s.stats().programLookups == 1, + "invalidateProgramCache(): the same"); + } + + // A mapper's graphics resources released (what removing a prop from its + // renderer does): its next draw looks the program up afresh. + for (auto *m : s.mappers()) + m->ReleaseGraphicsResources(s.sr->renderWindow()); + s.resetCounters(); + s.sr->render(); + const auto rel = s.stats(); + check(rel.programLookups == rel.draws && rel.programLookupsSkipped == 0 && rel.draws > 0, + "after a mapper's ReleaseGraphicsResources its next draw looks the program up", + fmt("lookups %.0f of %.0f draws", double(rel.programLookups), double(rel.draws))); + const PathRun released = runPath(s, DrawPath::Fast); + // Stock to compare against: two runs, a self-variance pair (checked + // byte-identical, or on Apple's renderer within its LSB envelope), after one + // that follows Fast (judged, not sampled). + const PathRun afterFast = runPath(s, DrawPath::Stock); + const PathRun before = runPath(s, DrawPath::Stock); + const PathRun before2 = runPath(s, DrawPath::Stock); + sampleSelfVariance(scope, "lifecycle: Stock twice", before, before2, s.w); + samePixels("after mapper release: Fast pixels == Stock", before.steadyPx, released.steadyPx, s.w); + samePixels("lifecycle: a Stock frame right after Fast == Stock", before.steadyPx, + afterFast.steadyPx, s.w); + + // The shader cache's programs released and recompiled in place (the cache + // keeps the objects): the re-bound program is compiled again before use. + auto *win = vtkOpenGLRenderWindow::SafeDownCast(s.sr->renderWindow()); + win->GetShaderCache()->ReleaseGraphicsResources(win); + LowMemoryPolyDataMapper::setDrawPath(DrawPath::Fast); + s.sr->render(); + const PathRun after = runPath(s, DrawPath::Fast); + // Across the recompile: byte for byte, or on Apple's renderer the LSB + // envelope, like every pixel check (checkPixels()). + samePixels("programs released in the cache: Fast recompiles and matches Stock", before.steadyPx, + after.steadyPx, s.w); + + // The same scene -- the same nodes and mappers -- closed and re-opened in a + // new window (new context, new shader cache): the first draws resolve fully + // and the frames match the old window's. + s.sr.reset(); + s.open(); + s.sg.setShadowsEnabled(true); // the pass chain belongs to the renderer, not the scene + s.sg.setShadowUpdateInterval(1000); + s.resetCounters(); + s.sr->render(); + glcallsInstall(); // the new context re-loaded glad + const auto st = s.stats(); + check(st.programLookups > 0 && st.programLookupsSkipped == 0, + "re-opened in a new window: the first draws look their programs up", + fmt("lookups %.0f, skipped %.0f", double(st.programLookups), + double(st.programLookupsSkipped))); + // The window's first Stock run follows that Fast frame: neither sampled nor + // compared. The two after it are a self-variance pair -- in this window, + // never with the old window's, which the cross-window checks compare. + runPath(s, DrawPath::Stock); + const PathRun a = runPath(s, DrawPath::Stock); + const PathRun a2 = runPath(s, DrawPath::Stock); + const PathRun b = runPath(s, DrawPath::Fast); + sampleSelfVariance(scope, "new window: Stock twice", a, a2, s.w); + samePixels("new window: Fast pixels == Stock", a.steadyPx, b.steadyPx, s.w); + // Across the windows: byte for byte, or on Apple's renderer the LSB + // envelope, like every pixel check (checkPixels()). + samePixels("new window: the Stock frame == the old window's", before.steadyPx, a.steadyPx, s.w); + samePixels("new window: the Fast frame == the old window's Stock frame", before.steadyPx, + b.steadyPx, s.w); +} + +// ── 7. guards ──────────────────────────────────────────────────────────────── +void testGuards(cvc::app &app) { + std::printf("guards\n"); + Scene s(app); + // Coincident offsets under the global POLYGON_OFFSET mode, where VTK's + // defaults give points -8 and lines -4 units but polygons 0: + // zeroPoly polys only, polygon offset 0 -> stock's empty verts/lines agents + // leave -4 in the program for the polys draw; skipping them would + // not. Must fall back to all four. + // shifted polys only, the same relative offset on every class -> the polys + // draw sets its own non-zero value. Safe to skip. + Probe &zero = s.probe("zeroPoly", spherePoly(1.5)); + zero.actor->GetProperty()->SetColor(0.9, 0.4, 0.3); + Probe &shifted = s.probe("shifted", ribbonPoly(12, -4, 3, 0.2)); + shifted.mapper->SetRelativeCoincidentTopologyPolygonOffsetParameters(0.0, -3.0); + shifted.mapper->SetRelativeCoincidentTopologyLineOffsetParameters(0.0, -3.0); + shifted.mapper->SetRelativeCoincidentTopologyPointOffsetParameter(-3.0); + Probe &vv = s.probe("vertexVis", terrainPoly(6, -5, -5, 4)); + vv.actor->GetProperty()->VertexVisibilityOn(); + vv.actor->GetProperty()->SetVertexColor(1, 0, 0); + // The mode is set before the first render: vtkGLSLModCoincidentTopology + // declares the offset uniforms only if the shader is BUILT under an offset, + // and nothing rebuilds it when the global mode changes later -- switched on + // after the build, the offsets never reach the GPU on any path. + const int savedMode = vtkMapper::GetResolveCoincidentTopology(); + vtkMapper::SetResolveCoincidentTopologyToPolygonOffset(); + s.open(); + s.sr->setCamera(0, -14, 10, 0, 0, 0, 0, 0, 1, 40.0, 1.0, 100.0); + s.sr->render(); + glcallsInstall(); + for (const Probe *p : {&zero, &shifted}) { + vtkShaderProgram *prog = p->mapper->vtkDrawTexturedElements::GetShaderProgram(); + check(prog && prog->IsUniformUsed("cOffset") && prog->IsUniformUsed("cFactor"), + "POLYGON_OFFSET mode: " + p->name + "'s program carries the depth offset uniforms"); + } + + // Stock last in the settle, so that neither sampled Stock run follows Fast. + for (DrawPath p : {DrawPath::Fast, DrawPath::Stock}) + runPath(s, p); + const PathRun stock = runPath(s, DrawPath::Stock); + const PathRun all = runPath(s, DrawPath::AllCellTypes); + const PathRun stock2 = runPath(s, DrawPath::Stock); + const PathRun fast = runPath(s, DrawPath::Fast); // last: the mapper stats below are Fast's + sampleSelfVariance("guards", "POLYGON_OFFSET mode: Stock twice", stock, stock2, s.w); + samePixels("POLYGON_OFFSET mode: Fast pixels == Stock", stock.steadyPx, fast.steadyPx, s.w); + checkTraces("POLYGON_OFFSET mode", stock.steadyTrace, all.steadyTrace, fast.steadyTrace, + fast.steadyStats); + const auto &zs = zero.mapper->stats(); + const auto &ss = shifted.mapper->stats(); + check(zs.coincidentFallbacks == zs.draws && zs.cellTypesSkipped == 0 && zs.draws > 0, + "zero polygon offset under defaults that offset points/lines: all four agents run", + fmt("fallbacks %.0f of %.0f draws", double(zs.coincidentFallbacks), double(zs.draws))); + check(ss.coincidentFallbacks == 0 && ss.cellTypesSkipped == 3 * ss.draws && ss.draws > 0, + "an offset the drawn type sets itself: empty types skipped", + fmt("skipped %.0f over %.0f draws", double(ss.cellTypesSkipped), double(ss.draws))); + vtkMapper::SetResolveCoincidentTopology(savedMode); + + const auto &vs = vv.mapper->stats(); + check(vs.stockDraws == vs.draws && vs.draws > 0, "vertex visibility draws the stock way"); + + // Picking: the selector's passes draw the stock way, and select the same. + comparePicking(s, "guards scene"); +} + +// ── 8. shader replacements after the first draw ───────────────────────────── +// VTK's low-memory mapper ignores its actor's shader property once the program +// is built; LowMemoryPolyDataMapper::RenderPieceStart rebuilds it. A plain +// GeometryNode (no streaming), every draw path. +void testLateShaderReplacement(cvc::app &app) { + std::printf("shader replacement after the first draw\n"); + for (DrawPath p : {DrawPath::Stock, DrawPath::AllCellTypes, DrawPath::Fast}) { + LowMemoryPolyDataMapper::setDrawPath(p); + Scene s(app); + auto n = s.node("plain"); + n->setGeometry(boxGeometry(0, 0, 0, 5, 5, 0.5)); + n->setUseSingleColor(true); + n->setColor(1, 1, 1); + s.w = 48; + s.h = 48; + s.open(); + s.sr->setCamera(0, 0, 50, 0, 0, 0, 0, 1, 0, 30.0, 1.0, 200.0); + s.sr->render(); + s.sr->render(); + const auto centre = [&](const std::vector &px) { + const size_t i = 4 * (static_cast(s.h / 2) * s.w + s.w / 2); + return std::vector(px.begin() + i, px.begin() + i + 3); + }; + const auto corner = [](const std::vector &px) { + return std::vector(px.begin(), px.begin() + 3); + }; + const auto drawn = s.rgba(); + const bool visible = centre(drawn) != corner(drawn); + + n->addFragmentShaderReplacement("//VTK::UniformFlow::Impl", + "//VTK::UniformFlow::Impl\n discard;\n"); + s.resetCounters(); + s.sr->render(); + const auto discarded = s.rgba(); + const auto st = s.stats(); + check(visible && centre(discarded) == corner(discarded), + std::string(pathName(p)) + ": a replacement added after the first draw reaches the GPU", + fmt("lookups %.0f", double(st.programLookups + st.stockDraws))); + + // Byte for byte, or on Apple's renderer the LSB envelope, like every pixel + // check (no renderer has been seen to vary on this unshadowed 48x48 scene). + n->clearShaderReplacements(); + s.sr->render(); + const PixelDiff back = pixelDiff(drawn, s.rgba()); + check(pixelsMatch(back), + std::string(pathName(p)) + ": cleared again, the original frame comes back", + pixelVerdict(back, s.w)); + } + LowMemoryPolyDataMapper::setDrawPath(DrawPath::Fast); +} + +} // namespace + +int main() { +#ifdef _WIN32 + _putenv_s("CVCGL_LOWMEM_MAPPER", "force"); + if (!std::getenv("__GL_SYNC_TO_VBLANK")) + _putenv_s("__GL_SYNC_TO_VBLANK", "0"); + if (!std::getenv("vblank_mode")) + _putenv_s("vblank_mode", "0"); +#else + setenv("CVCGL_LOWMEM_MAPPER", "force", 1); + setenv("__GL_SYNC_TO_VBLANK", "0", 0); // NVIDIA throttles offscreen swaps otherwise + setenv("vblank_mode", "0", 0); +#endif + const char *rep = std::getenv("CVCGL_LOWMEM_REPORT"); + g_report = rep && *rep && std::string(rep) != "0"; + vtkOutputWindow::SetInstance(vtkSmartPointer::New()); + + testPolicy(); + cvc::app app; + if (!LowMemoryPolyDataMapper::replicaActive()) { + // The shader-property rebuild is not part of the replica: every VTK. + std::printf(" replica off (VTK is not an unpatched 9.5.0): every path draws the stock way\n"); + if (renderAvailable(app)) + testLateShaderReplacement(app); + std::printf("%s: cvcgl_lowmem_fastdraw (%d checks, %d failed)\n", + g_failures == 0 ? "PASS" : "FAIL", g_checks, g_failures); + return g_failures == 0 ? 0 : 1; + } + + LowMemoryPolyDataMapper::setStageObserver(&onStage); + if (renderAvailable(app)) { + // Shadows on and off in separate scenes: SceneGraph::setShadowsEnabled(false) + // drops the shadow passes without releasing their GL resources, which VTK + // reports as an error when they are destroyed. + for (bool shadows : {true, false}) { + Scene s(app); + buildCity(s); + addLights(s.sg); + s.open(); + if (shadows) { + check(s.sg.setShadowsEnabled(true), "shadows enabled"); + s.sg.setShadowUpdateInterval(1000); // bake only when invalidated + } + s.sr->render(); + glcallsInstall(); + check(s.mappers().size() == s.nodes.size() + s.probes.size(), + "every node and probe draws through LowMemoryPolyDataMapper", + fmt("%.0f of %.0f", double(s.mappers().size()), + double(s.nodes.size() + s.probes.size()))); + const std::string label = shadows ? "shadows on" : "shadows off"; + comparePaths(s, label, shadows); + if (shadows) + testLifecycle(s, label); + } + testGuards(app); + testLateShaderReplacement(app); + } + LowMemoryPolyDataMapper::setStageObserver(nullptr); + checkPixels(); + check(ErrorCounter::errors() == 0, "no VTK errors", + fmt("%.0f errors", double(ErrorCounter::errors()))); + std::printf("%s: cvcgl_lowmem_fastdraw (%d checks, %d failed)\n", + g_failures == 0 ? "PASS" : "FAIL", g_checks, g_failures); + return g_failures == 0 ? 0 : 1; +} diff --git a/src/cvcGL/test/gl_call_counter.cpp b/src/cvcGL/test/gl_call_counter.cpp new file mode 100644 index 00000000..66098e38 --- /dev/null +++ b/src/cvcGL/test/gl_call_counter.cpp @@ -0,0 +1,728 @@ +/* + Copyright 2026 The University of Texas at Austin + + This file is part of libcvc. + + libcvc is free software; you can redistribute it and/or + modify it under the terms of the GNU Lesser General Public + License version 2.1 as published by the Free Software Foundation. +*/ + +#include "gl_call_counter.h" + +#include +#include +#include +#include +#include +#include +#include + +namespace cvcgl_test { + +namespace { + +// ── the wrapped entry points ──────────────────────────────────────────────── +// Every GL entry point VTK 9.5.0's Rendering/OpenGL2, Rendering/VolumeOpenGL2 +// and Rendering/Core call (a grep of their sources against glad), cvcGL's own, +// and the GLES3/GL4 core families those may grow into. + +// Counted and traced with no context. +#define CVC_GL_PLAIN(X) \ + X(AttachShader) \ + X(BeginQuery) \ + X(BeginQueryIndexed) \ + X(BeginTransformFeedback) \ + X(BindBuffer) \ + X(BindBufferBase) \ + X(BindBufferRange) \ + X(BindFragDataLocation) \ + X(BindFramebuffer) \ + X(BindRenderbuffer) \ + X(BindVertexArray) \ + X(BlendEquation) \ + X(BlendEquationSeparate) \ + X(BlendFunc) \ + X(BlendFuncSeparate) \ + X(BlitFramebuffer) \ + X(BufferData) \ + X(BufferSubData) \ + X(ClampColor) \ + X(Clear) \ + X(ClearBufferfi) \ + X(ClearBufferfv) \ + X(ClearBufferiv) \ + X(ClearBufferuiv) \ + X(ClearColor) \ + X(ClearDepth) \ + X(ClearDepthf) \ + X(ClearStencil) \ + X(ColorMask) \ + X(ColorMaski) \ + X(CompileShader) \ + X(CopyBufferSubData) \ + X(CreateProgram) \ + X(CreateShader) \ + X(CullFace) \ + X(DebugMessageCallback) \ + X(DebugMessageControl) \ + X(DebugMessageInsert) \ + X(DeleteBuffers) \ + X(DeleteFramebuffers) \ + X(DeleteQueries) \ + X(DeleteRenderbuffers) \ + X(DeleteShader) \ + X(DeleteSync) \ + X(DeleteTextures) \ + X(DeleteVertexArrays) \ + X(DepthFunc) \ + X(DepthMask) \ + X(DetachShader) \ + X(Disable) \ + X(DisableVertexAttribArray) \ + X(Disablei) \ + X(DrawBuffer) \ + X(DrawBuffers) \ + X(Enable) \ + X(EnableVertexAttribArray) \ + X(Enablei) \ + X(EndQuery) \ + X(EndQueryIndexed) \ + X(EndTransformFeedback) \ + X(FenceSync) \ + X(Flush) \ + X(FramebufferRenderbuffer) \ + X(FramebufferTexture2D) \ + X(FramebufferTexture3D) \ + X(FramebufferTextureLayer) \ + X(GenBuffers) \ + X(GenFramebuffers) \ + X(GenQueries) \ + X(GenRenderbuffers) \ + X(GenTextures) \ + X(GenVertexArrays) \ + X(InvalidateFramebuffer) \ + X(LineWidth) \ + X(MemoryBarrier) \ + X(PatchParameteri) \ + X(PixelStorei) \ + X(PointSize) \ + X(PolygonOffset) \ + X(QueryCounter) \ + X(ReadBuffer) \ + X(RenderbufferStorage) \ + X(RenderbufferStorageMultisample) \ + X(Scissor) \ + X(ShaderSource) \ + X(StencilFunc) \ + X(StencilFuncSeparate) \ + X(StencilMask) \ + X(StencilMaskSeparate) \ + X(StencilOp) \ + X(StencilOpSeparate) \ + X(TransformFeedbackVaryings) \ + X(UniformBlockBinding) \ + X(UnmapBuffer) \ + X(VertexAttribDivisor) \ + X(VertexAttribDivisorARB) \ + X(VertexAttribIPointer) \ + X(VertexAttribPointer) \ + X(Viewport) + +// Texture-state calls: act on the active unit's binding (traced with the unit). +#define CVC_GL_UNIT(X) \ + X(BindTexture) \ + X(CopyTexImage2D) \ + X(CopyTexSubImage2D) \ + X(GenerateMipmap) \ + X(TexBuffer) \ + X(TexImage1D) \ + X(TexImage2D) \ + X(TexImage2DMultisample) \ + X(TexImage3D) \ + X(TexParameterf) \ + X(TexParameterfv) \ + X(TexParameteri) \ + X(TexStorage2D) \ + X(TexStorage2DMultisample) \ + X(TexStorage3D) \ + X(TexSubImage2D) \ + X(TexSubImage3D) + +// Draw / dispatch calls (traced with the bound program). +#define CVC_GL_DRAW(X) \ + X(DispatchCompute) \ + X(DrawArrays) \ + X(DrawArraysInstanced) \ + X(DrawArraysInstancedARB) \ + X(DrawElements) \ + X(DrawElementsInstanced) \ + X(DrawElementsInstancedARB) \ + X(DrawRangeElements) + +// Desktop round-trips (upper bound for WebGL): the caller waits on the driver. +#define CVC_GL_SYNC(X) \ + X(CheckFramebufferStatus) \ + X(ClientWaitSync) \ + X(Finish) \ + X(GetAttribLocation) \ + X(GetBooleanv) \ + X(GetBufferPointerv) \ + X(GetDoublev) \ + X(GetError) \ + X(GetFloatv) \ + X(GetFramebufferAttachmentParameteriv) \ + X(GetIntegerv) \ + X(GetNamedRenderbufferParameteriv) \ + X(GetProgramInfoLog) \ + X(GetProgramiv) \ + X(GetQueryObjectiv) \ + X(GetQueryObjectui64v) \ + X(GetQueryObjectuiv) \ + X(GetRenderbufferParameteriv) \ + X(GetShaderInfoLog) \ + X(GetShaderiv) \ + X(GetString) \ + X(GetStringi) \ + X(GetTexImage) \ + X(GetTexLevelParameteriv) \ + X(GetTextureLevelParameteriv) \ + X(GetUniformBlockIndex) \ + X(GetUniformLocation) \ + X(IsEnabled) \ + X(IsProgram) \ + X(MapBuffer) \ + X(MapBufferRange) \ + X(ReadPixels) + +// glUniform{1,2,3,4}{f,i,ui}: every argument is a scalar. +#define CVC_GL_USCALAR(X) \ + X(Uniform1f) \ + X(Uniform2f) \ + X(Uniform3f) \ + X(Uniform4f) \ + X(Uniform1i) \ + X(Uniform2i) \ + X(Uniform3i) \ + X(Uniform4i) \ + X(Uniform1ui) \ + X(Uniform2ui) \ + X(Uniform3ui) \ + X(Uniform4ui) + +// glUniform{1,2,3,4}{f,i,ui}v(location, count, const T *v): (name, T, components). +#define CVC_GL_UVEC(X) \ + X(Uniform1fv, GLfloat, 1) \ + X(Uniform2fv, GLfloat, 2) \ + X(Uniform3fv, GLfloat, 3) \ + X(Uniform4fv, GLfloat, 4) \ + X(Uniform1iv, GLint, 1) \ + X(Uniform2iv, GLint, 2) \ + X(Uniform3iv, GLint, 3) \ + X(Uniform4iv, GLint, 4) \ + X(Uniform1uiv, GLuint, 1) \ + X(Uniform2uiv, GLuint, 2) \ + X(Uniform3uiv, GLuint, 3) \ + X(Uniform4uiv, GLuint, 4) + +// glUniformMatrix*fv(location, count, transpose, const GLfloat *v): (name, floats). +#define CVC_GL_UMAT(X) \ + X(UniformMatrix2fv, 4) \ + X(UniformMatrix3fv, 9) \ + X(UniformMatrix4fv, 16) \ + X(UniformMatrix2x3fv, 6) \ + X(UniformMatrix3x2fv, 6) \ + X(UniformMatrix2x4fv, 8) \ + X(UniformMatrix4x2fv, 8) \ + X(UniformMatrix3x4fv, 12) \ + X(UniformMatrix4x3fv, 12) + +// Tracked state: the bound program and the active texture unit. +#define CVC_GL_STATE(X) \ + X(UseProgram) \ + X(LinkProgram) \ + X(DeleteProgram) \ + X(ActiveTexture) + +enum Cls : unsigned char { kPlain, kUnit, kDraw, kSync, kUniform, kState }; + +enum Entry : int { +#define CVC_E(name, ...) E_##name, + CVC_GL_PLAIN(CVC_E) CVC_GL_UNIT(CVC_E) CVC_GL_DRAW(CVC_E) CVC_GL_SYNC(CVC_E) CVC_GL_USCALAR(CVC_E) + CVC_GL_UVEC(CVC_E) CVC_GL_UMAT(CVC_E) CVC_GL_STATE(CVC_E) +#undef CVC_E + E_COUNT +}; + +const char *const g_names[E_COUNT] = { +#define CVC_N(name, ...) #name, + CVC_GL_PLAIN(CVC_N) CVC_GL_UNIT(CVC_N) CVC_GL_DRAW(CVC_N) CVC_GL_SYNC(CVC_N) + CVC_GL_USCALAR(CVC_N) CVC_GL_UVEC(CVC_N) CVC_GL_UMAT(CVC_N) CVC_GL_STATE(CVC_N) +#undef CVC_N +}; + +constexpr Cls g_cls[E_COUNT] = { +#define CVC_PLAIN(...) kPlain, +#define CVC_UNIT(...) kUnit, +#define CVC_DRAW(...) kDraw, +#define CVC_SYNC(...) kSync, +#define CVC_UNI(...) kUniform, +#define CVC_STATE(...) kState, + CVC_GL_PLAIN(CVC_PLAIN) CVC_GL_UNIT(CVC_UNIT) CVC_GL_DRAW(CVC_DRAW) CVC_GL_SYNC(CVC_SYNC) + CVC_GL_USCALAR(CVC_UNI) CVC_GL_UVEC(CVC_UNI) CVC_GL_UMAT(CVC_UNI) CVC_GL_STATE(CVC_STATE) +#undef CVC_PLAIN +#undef CVC_UNIT +#undef CVC_DRAW +#undef CVC_SYNC +#undef CVC_UNI +#undef CVC_STATE +}; + +double g_n[E_COUNT]; +double g_uniformRedundant = 0; + +// Tracked GL state, and the open trace. +GLuint g_program = 0; +unsigned g_unit = 0; +bool g_tracing = false; +Trace g_trace; + +// Uniform values are per program: (program, location) -> last bytes sent. +std::map, std::string> g_last; + +// argBytes starts with the GLint location; the rest is the value as sent. +void noteUniform(const std::string &argBytes) { + GLint loc = 0; + std::memcpy(&loc, argBytes.data(), sizeof loc); + std::string v = argBytes.substr(sizeof loc); + auto key = std::make_pair(g_program, loc); + auto it = g_last.find(key); + if (it != g_last.end() && it->second == v) + g_uniformRedundant += 1; + else + g_last[key] = std::move(v); +} +void forgetProgram(GLuint p) { + for (auto it = g_last.begin(); it != g_last.end();) + it = it->first.first == p ? g_last.erase(it) : std::next(it); +} + +void record(int id, std::string &&args) { + TraceRec r; + r.entry = id; + switch (g_cls[id]) { + case kUnit: + r.ctx = g_unit; + break; + case kDraw: + case kUniform: + r.ctx = g_program; + break; + default: + r.ctx = 0; + } + r.args = std::move(args); + g_trace.push_back(std::move(r)); +} + +// Floats are recorded as GL computes with them: -0.0 as +0.0. (VTK's camera +// and actor key-matrix caches hand out a zero normal-matrix entry with either +// sign depending on which earlier render last recomputed them -- history, not +// the draw path: two Stock hardware selections in a row differ that way.) +template F canonicalZero(F v) { return v == F(0) ? F(0) : v; } + +template void appendArg(std::string &s, T v) { + if constexpr (std::is_floating_point_v) { + const T c = canonicalZero(v); + char b[sizeof(T)]; + std::memcpy(b, &c, sizeof(T)); + s.append(b, sizeof(T)); + } else if constexpr (std::is_same_v) { + // Location / attribute names: the string, so the lookup is identified. + if (v) + s.append(v, std::strlen(v) + 1); + else + s.push_back('\0'); + } else if constexpr (std::is_pointer_v) { + s.push_back(v ? '\1' : '\0'); // bulk data / out-params: only whether it is there + } else { + char b[sizeof(T)]; + std::memcpy(b, &v, sizeof(T)); + s.append(b, sizeof(T)); + } +} + +// The generic wrapper: count, note uniform values, trace. +template struct Wrap; +template struct Wrap { + static constexpr bool kIsUniform = g_cls[Id] == kUniform; + static inline R(GLAD_API_PTR *orig)(A...) = nullptr; + static R GLAD_API_PTR fn(A... a) { + g_n[Id] += 1; + if (kIsUniform || g_tracing) { + std::string bytes; + (appendArg(bytes, a), ...); + if constexpr (kIsUniform) + noteUniform(bytes); + if (g_tracing) + record(Id, std::move(bytes)); + } + return orig(a...); + } + static void hook(R(GLAD_API_PTR *&ptr)(A...)) { + if (ptr && ptr != &fn) { + orig = ptr; + ptr = &fn; + } + } + static void unhook(R(GLAD_API_PTR *&ptr)(A...)) { + if (ptr == &fn) + ptr = orig; + } +}; + +// glUniform*v / glUniformMatrix*: the values behind the pointer. +#define CVC_O(name, ...) decltype(glad_gl##name) o_##name = nullptr; +CVC_GL_UVEC(CVC_O) +CVC_GL_UMAT(CVC_O) +CVC_GL_STATE(CVC_O) +#undef CVC_O + +void uniformArray(int id, GLint l, GLsizei c, const std::string &prefix, const void *v, + size_t bytesPerElement, bool isFloat) { + g_n[id] += 1; + std::string b; + appendArg(b, l); + appendArg(b, c); + b += prefix; + if (v && c > 0) { + const size_t n = bytesPerElement * static_cast(c); + if (isFloat) { + for (size_t k = 0; k < n; k += sizeof(GLfloat)) { + GLfloat f; + std::memcpy(&f, static_cast(v) + k, sizeof f); + appendArg(b, f); + } + } else { + b.append(static_cast(v), n); + } + } + noteUniform(b); + if (g_tracing) + record(id, std::move(b)); +} + +#define CVC_UV_W(name, T, k) \ + void GLAD_API_PTR w_##name(GLint l, GLsizei c, const T *v) { \ + uniformArray(E_##name, l, c, std::string(), v, sizeof(T) * (k), std::is_same_v); \ + o_##name(l, c, v); \ + } +CVC_GL_UVEC(CVC_UV_W) +#undef CVC_UV_W + +#define CVC_UM_W(name, k) \ + void GLAD_API_PTR w_##name(GLint l, GLsizei c, GLboolean t, const GLfloat *v) { \ + uniformArray(E_##name, l, c, std::string(1, static_cast(t)), v, sizeof(GLfloat) * (k), \ + true); \ + o_##name(l, c, t, v); \ + } +CVC_GL_UMAT(CVC_UM_W) +#undef CVC_UM_W + +void GLAD_API_PTR w_UseProgram(GLuint p) { + g_n[E_UseProgram] += 1; + g_program = p; + if (g_tracing) { + std::string b; + appendArg(b, p); + record(E_UseProgram, std::move(b)); + } + o_UseProgram(p); +} +void GLAD_API_PTR w_LinkProgram(GLuint p) { + g_n[E_LinkProgram] += 1; + forgetProgram(p); + if (g_tracing) { + std::string b; + appendArg(b, p); + record(E_LinkProgram, std::move(b)); + } + o_LinkProgram(p); +} +void GLAD_API_PTR w_DeleteProgram(GLuint p) { + g_n[E_DeleteProgram] += 1; + forgetProgram(p); + if (g_tracing) { + std::string b; + appendArg(b, p); + record(E_DeleteProgram, std::move(b)); + } + o_DeleteProgram(p); +} +void GLAD_API_PTR w_ActiveTexture(GLenum t) { + g_n[E_ActiveTexture] += 1; + g_unit = t - GL_TEXTURE0; + if (g_tracing) { + std::string b; + appendArg(b, t); + record(E_ActiveTexture, std::move(b)); + } + o_ActiveTexture(t); +} + +// The real glGetIntegerv / glGetFloatv, whether or not they are wrapped now. +void rawGetIntegerv(GLenum what, GLint *v) { + using W = Wrap; + auto fn = glad_glGetIntegerv == &W::fn ? W::orig : glad_glGetIntegerv; + if (fn) + fn(what, v); +} +void rawGetFloatv(GLenum what, GLfloat *v) { + using W = Wrap; + auto fn = glad_glGetFloatv == &W::fn ? W::orig : glad_glGetFloatv; + if (fn) + fn(what, v); +} + +} // namespace + +bool TraceRec::marker() const { return entry >= kTraceMarkBase; } + +void glcallsInstall() { + // Start from the context's real state (calls made while uninstalled, or in + // another context, were not tracked). + GLint program = 0, unit = GL_TEXTURE0; + rawGetIntegerv(GL_CURRENT_PROGRAM, &program); + rawGetIntegerv(GL_ACTIVE_TEXTURE, &unit); + g_program = static_cast(program); + g_unit = static_cast(unit - GL_TEXTURE0); +#define CVC_H(name, ...) Wrap::hook(glad_gl##name); + CVC_GL_PLAIN(CVC_H) + CVC_GL_UNIT(CVC_H) + CVC_GL_DRAW(CVC_H) + CVC_GL_SYNC(CVC_H) + CVC_GL_USCALAR(CVC_H) +#undef CVC_H +#define CVC_W(name, ...) \ + if (glad_gl##name && glad_gl##name != w_##name) { \ + o_##name = glad_gl##name; \ + glad_gl##name = w_##name; \ + } + CVC_GL_UVEC(CVC_W) + CVC_GL_UMAT(CVC_W) + CVC_GL_STATE(CVC_W) +#undef CVC_W +} + +void glcallsUninstall() { +#define CVC_U(name, ...) Wrap::unhook(glad_gl##name); + CVC_GL_PLAIN(CVC_U) + CVC_GL_UNIT(CVC_U) + CVC_GL_DRAW(CVC_U) + CVC_GL_SYNC(CVC_U) + CVC_GL_USCALAR(CVC_U) +#undef CVC_U +#define CVC_R(name, ...) \ + if (glad_gl##name == w_##name) \ + glad_gl##name = o_##name; + CVC_GL_UVEC(CVC_R) + CVC_GL_UMAT(CVC_R) + CVC_GL_STATE(CVC_R) +#undef CVC_R +} + +GLCalls glcallsRead() { + GLCalls r; + r.n.assign(g_n, g_n + E_COUNT); + r.uniformRedundant = g_uniformRedundant; + return r; +} + +int glcallsEntries() { return E_COUNT; } +const char *glcallsName(int i) { return i >= 0 && i < E_COUNT ? g_names[i] : "?"; } + +void glcallsTraceStart() { + g_trace.clear(); + g_tracing = true; + for (auto [kind, what] : {std::make_pair(kTraceInitPointSize, GLenum(GL_POINT_SIZE)), + std::make_pair(kTraceInitLineWidth, GLenum(GL_LINE_WIDTH))}) { + GLfloat v = 0; + rawGetFloatv(what, &v); + TraceRec r; + r.entry = kTraceMarkBase + kind; + appendArg(r.args, v); + g_trace.push_back(std::move(r)); + } +} + +Trace glcallsTraceStop() { + g_tracing = false; + Trace t; + t.swap(g_trace); + return t; +} + +void glcallsMark(int kind, int arg) { + if (!g_tracing) + return; + TraceRec r; + r.entry = kTraceMarkBase + kind; + r.ctx = static_cast(arg); + g_trace.push_back(std::move(r)); +} + +bool glcallsIsDraw(int entry) { return entry >= 0 && entry < E_COUNT && g_cls[entry] == kDraw; } + +bool glcallsIsUniform(int entry) { + return entry >= 0 && entry < E_COUNT && g_cls[entry] == kUniform; +} + +std::vector glcallsUniformWrites(const TraceRec &r) { + std::vector out; + if (r.marker() || !glcallsIsUniform(r.entry) || r.args.size() < sizeof(GLint)) + return out; + // The args as recorded: location, then count for the *v forms, then the + // transpose byte for the matrices, then the values. The type is the name less + // "Uniform" and the trailing "v": "3f", "1ui", "Matrix4x3f". + const std::string name = g_names[r.entry]; + const bool matrix = name.compare(0, 13, "UniformMatrix") == 0; + const bool array = name.back() == 'v'; + std::string type = name.substr(7, name.size() - 7 - (array ? 1 : 0)); + GLint location = 0; + std::memcpy(&location, r.args.data(), sizeof location); + size_t at = sizeof location; + GLsizei count = 1; + if (array) { + if (r.args.size() < at + sizeof count) + return out; + std::memcpy(&count, r.args.data() + at, sizeof count); + at += sizeof count; + } + if (matrix) { + if (r.args.size() < at + 1) + return out; + type += r.args[at] ? "T" : ""; + ++at; + } + if (location < 0 || count <= 0) + return out; + const size_t each = (r.args.size() - at) / static_cast(count); + for (GLsizei k = 0; k < count; ++k) + out.push_back({location + k, type + ':' + r.args.substr(at + k * each, each)}); + return out; +} + +int glcallsEntry(const std::string &name) { + for (int i = 0; i < E_COUNT; ++i) + if (name == g_names[i]) + return i; + return -1; +} + +std::string glcallsDescribe(const TraceRec &r) { + char buf[96]; + if (r.marker()) { + std::snprintf(buf, sizeof buf, "mark %d (%u)", r.entry - kTraceMarkBase, r.ctx); + return buf; + } + const Cls c = r.entry >= 0 && r.entry < E_COUNT ? g_cls[r.entry] : kPlain; + std::snprintf(buf, sizeof buf, "gl%s%s%u [", glcallsName(r.entry), + c == kUnit ? " unit=" + : (c == kDraw || c == kUniform) ? " prog=" + : " ", + r.ctx); + std::string out = buf; + for (size_t i = 0; i < r.args.size() && i < 48; ++i) { + std::snprintf(buf, sizeof buf, i ? " %02x" : "%02x", static_cast(r.args[i])); + out += buf; + } + out += r.args.size() > 48 ? " ...]" : "]"; + return out; +} + +GLCalls GLCalls::operator-(const GLCalls &o) const { + GLCalls r; + r.n.resize(n.size()); + for (size_t i = 0; i < n.size(); ++i) + r.n[i] = n[i] - (i < o.n.size() ? o.n[i] : 0.0); + r.uniformRedundant = uniformRedundant - o.uniformRedundant; + return r; +} + +GLCalls &GLCalls::operator+=(const GLCalls &o) { + if (n.size() < o.n.size()) + n.resize(o.n.size(), 0.0); + for (size_t i = 0; i < o.n.size(); ++i) + n[i] += o.n[i]; + uniformRedundant += o.uniformRedundant; + return *this; +} + +double GLCalls::total() const { + double s = 0; + for (double v : n) + s += v; + return s; +} + +double GLCalls::count(const std::string &entry) const { + for (int i = 0; i < E_COUNT && i < static_cast(n.size()); ++i) + if (entry == g_names[i]) + return n[i]; + return 0; +} + +double GLCalls::uniforms() const { + double s = 0; + for (int i = 0; i < static_cast(n.size()) && i < E_COUNT; ++i) + s += g_cls[i] == kUniform ? n[i] : 0.0; + return s; +} + +double GLCalls::roundTrips() const { + double s = 0; + for (int i = 0; i < static_cast(n.size()) && i < E_COUNT; ++i) + s += g_cls[i] == kSync ? n[i] : 0.0; + return s; +} + +bool GLCalls::sameAs(const GLCalls &o, std::string *firstDifference) const { + for (int i = 0; i < E_COUNT; ++i) { + const double a = i < static_cast(n.size()) ? n[i] : 0.0; + const double b = i < static_cast(o.n.size()) ? o.n[i] : 0.0; + if (a != b) { + if (firstDifference) { + char buf[128]; + std::snprintf(buf, sizeof buf, "gl%s %g vs %g", g_names[i], a, b); + *firstDifference = buf; + } + return false; + } + } + return true; +} + +std::string GLCalls::str(double per, int top) const { + if (per <= 0) + per = 1; + char buf[192]; + std::snprintf(buf, sizeof buf, + "total %.1f (uniforms %.1f, redundant %.1f, desktop round-trips (upper bound) " + "%.1f) |", + total() / per, uniforms() / per, uniformRedundant / per, roundTrips() / per); + std::string out = buf; + std::vector> v; + for (int i = 0; i < static_cast(n.size()); ++i) + if (n[i] != 0) + v.emplace_back(n[i], i); + std::sort(v.begin(), v.end(), [](const auto &a, const auto &b) { + return a.first != b.first ? a.first > b.first : a.second < b.second; + }); + for (int k = 0; k < static_cast(v.size()) && k < top; ++k) { + std::snprintf(buf, sizeof buf, " %s=%.2f", g_names[v[k].second], v[k].first / per); + out += buf; + } + return out; +} + +} // namespace cvcgl_test diff --git a/src/cvcGL/test/gl_call_counter.h b/src/cvcGL/test/gl_call_counter.h new file mode 100644 index 00000000..b93adff9 --- /dev/null +++ b/src/cvcGL/test/gl_call_counter.h @@ -0,0 +1,110 @@ +/* + Copyright 2026 The University of Texas at Austin + + This file is part of libcvc. + + libcvc is free software; you can redistribute it and/or + modify it under the terms of the GNU Lesser General Public + License version 2.1 as published by the Free Software Foundation. +*/ + +// Test-only GL call counter and tracer: swaps VTK's glad function pointers +// (glad_gl*) for wrappers, so a test can see every GL call the scene issues -- +// as counts per entry point, and (while a trace is open) as an ordered trace of +// (entry point, bound program / active texture unit, argument bytes, uniform +// values). Uniform uploads are also checked against the last value sent to the +// same (program, location): a re-send of an unchanged value counts as +// redundant. Every entry point VTK 9.5.0's Rendering/OpenGL2 and +// RenderingVolumeOpenGL2 and cvcGL call is wrapped, plus the GLES3-core +// families they may grow into (glUniform*ui*, glClearBuffer*, glTexStorage*, +// ...), so frame totals are complete. Desktop GL only (the wasm VTK calls GLES +// directly, not via glad). glcallsUninstall() puts VTK's pointers back, for +// timing without the wrappers' cost. +#ifndef CVCGL_TEST_GL_CALL_COUNTER_H +#define CVCGL_TEST_GL_CALL_COUNTER_H + +#include +#include + +namespace cvcgl_test { + +struct GLCalls { + std::vector n; // per entry point (index = glcallsName(i)) + double uniformRedundant = 0; // uniform uploads that repeated the current value + + GLCalls operator-(const GLCalls &o) const; + GLCalls &operator+=(const GLCalls &o); + double total() const; + double count(const std::string &entry) const; // entry without the "gl" prefix + double uniforms() const; // every glUniform* + // Desktop round-trips (upper bound): calls that wait on the driver on desktop + // GL -- glGet*, glIs*, glGetError, glGet*Location, glCheckFramebufferStatus, + // glReadPixels, glMapBuffer*, query results, glFinish. NOT a WebGL census: + // Firefox answers some of them (cached limits, glIsProgram, ...) in the + // content process, so on WebGL this is an upper bound. + double roundTrips() const; + // Same count for every entry point; otherwise the first difference. + bool sameAs(const GLCalls &o, std::string *firstDifference = nullptr) const; + // "total T (uniforms U, redundant R, desktop round-trips (upper bound) S) | + // Entry=k ..." with every number divided by `per`; the `top` largest entries. + std::string str(double per = 1.0, int top = 14) const; +}; + +// One traced GL call, or a marker the test inserted (glcallsMark). +struct TraceRec { + int entry = 0; // glcallsName(entry); >= kTraceMarkBase: marker kTraceMarkBase + kind + unsigned ctx = 0; // the bound program for glUniform* and draw calls, the active texture + // unit (0-based) for texture-state calls, 0 for the rest; a marker's arg + std::string args; // the arguments' bytes as passed (a pointer: whether it is null), the + // values a glUniform*v / glUniformMatrix* sends, the name a + // glGet*Location looks up + bool marker() const; + bool operator==(const TraceRec &o) const { + return entry == o.entry && ctx == o.ctx && args == o.args; + } + bool operator!=(const TraceRec &o) const { return !(*this == o); } +}; +using Trace = std::vector; +constexpr int kTraceMarkBase = 1 << 20; +// Marker kinds glcallsTraceStart() puts first: the GL state the trace starts +// from that vtkOpenGLState caches, args = the float value. +constexpr int kTraceInitPointSize = 1000, kTraceInitLineWidth = 1001; + +// (Re)install the wrappers. Call once a GL context exists (VTK has loaded glad), +// and again after anything that may reload it; idempotent. +void glcallsInstall(); +// Put VTK's own function pointers back (no wrapper on any call); idempotent. +void glcallsUninstall(); +GLCalls glcallsRead(); +int glcallsEntries(); +const char *glcallsName(int i); + +// Start recording every wrapped call in order (clears the previous trace; the +// trace opens with kTraceInitPointSize / kTraceInitLineWidth markers); stop and +// take it. +void glcallsTraceStart(); +Trace glcallsTraceStop(); +// Insert a marker at this point of the open trace (no-op when none is open). +void glcallsMark(int kind, int arg); +// glDraw* / glDispatchCompute. +bool glcallsIsDraw(int entry); +// glUniform* (scalar, vector or matrix). +bool glcallsIsUniform(int entry); +// A traced glUniform* call as the uniform locations it sets (in its bound +// program, the record's ctx): an upload of n array elements is n writes, at +// location, location + 1, ... Each value is the element's type and bytes, so +// glUniform1i(l, v) and glUniform1iv(l, 1, &v) set the same value. Empty for any +// other record, and for location -1 (which GL ignores). +struct UniformWrite { + int location = 0; + std::string value; +}; +std::vector glcallsUniformWrites(const TraceRec &r); +// The index of an entry point ("PointSize"), or -1. +int glcallsEntry(const std::string &name); +// "glUniform1i prog=3 [05 00 00 00 01 00 00 00]" / "mark 5 (2)". +std::string glcallsDescribe(const TraceRec &r); + +} // namespace cvcgl_test + +#endif