From 1177ec25878fccef325cbf34b178232a079d3d08 Mon Sep 17 00:00:00 2001 From: Joe Rivera Date: Thu, 1 Oct 2026 17:39:05 -0500 Subject: [PATCH 01/18] perf(cvcGL): LowMemoryPolyDataMapper -- skip empty cell types and the per-draw shader lookup On GLES3/WebGL2 every cvcGL mesh draws through VTK 9.5.0's vtkOpenGLLowMemoryPolyDataMapper, whose RenderPieceDraw runs four cell-type agents (verts, lines, polys, strips) whether or not they have anything to draw. Each empty one still re-binds every array texture, re-sends every camera/material/shadow/agent uniform and binds+unbinds the VAO: ~110 of the ~160 GL calls a lit mesh costs per draw. Every draw also re-resolves its program through the shader cache (5 source copies, re-substitution, an MD5 of ~16 KB) to find the program it already has. cvc::gl::LowMemoryPolyDataMapper replicates the drawn agent's PreDraw/Draw/PostDraw from VTK 9.5.0 (the agents are private, hidden VTK classes), skips the cell types that cannot render, and re-binds the cached program while ShaderBuildTimeStamp and the window's shader cache are unchanged. The drawn type gets the same uniforms, textures and draw call in the same order, so frames are byte-identical. The stock draw is kept for picking, vertex visibility, and coincident-offset layouts where skipping a type would change the depth offset a drawn type (or the next draw of the same program) sees. The replica compiles only against exactly 9.5.0; elsewhere the class forwards to VTK. It also zero-initialises CoordinateShiftAndScaleInUse/ShiftValues/ ScaleValues, which 9.5.0 leaves uninitialised. newPolyDataMapper() picks the mapper for every cvcGL node (GeometryNode, BBoxNode, GridNode, GeometryShape): CVCGL_LOWMEM_MAPPER=auto (default: LowMemoryPolyDataMapper exactly where VTK's factory hands out the low-memory mapper, i.e. wasm), force (also on desktop GL, for native tests and measurement) or off. DrawPath Stock / AllCellTypes / Fast (default; initial value from CVCGL_LOWMEM_DRAW) is a process-wide switch for A/B. --- inc/cvc/gl/LowMemoryPolyDataMapper.h | 146 +++++++++++ src/cvcGL/BBoxNode.cpp | 8 +- src/cvcGL/GeometryNode.cpp | 6 +- src/cvcGL/GridNode.cpp | 11 +- src/cvcGL/LowMemoryPolyDataMapper.cpp | 362 ++++++++++++++++++++++++++ src/cvcGL/nodes.cpp | 3 +- 6 files changed, 522 insertions(+), 14 deletions(-) create mode 100644 inc/cvc/gl/LowMemoryPolyDataMapper.h create mode 100644 src/cvcGL/LowMemoryPolyDataMapper.cpp diff --git a/inc/cvc/gl/LowMemoryPolyDataMapper.h b/inc/cvc/gl/LowMemoryPolyDataMapper.h new file mode 100644 index 000000000..1bf72f753 --- /dev/null +++ b/inc/cvc/gl/LowMemoryPolyDataMapper.h @@ -0,0 +1,146 @@ +/* + Copyright 2026 The University of Texas at Austin + + This file is part of libcvc. + + libcvc is free software; you can redistribute it and/or + modify it under the terms of the GNU Lesser General Public + License version 2.1 as published by the Free Software Foundation. + + libcvc is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU + Lesser General Public License for more details. + + You should have received a copy of the GNU Lesser General Public + License along with this library; if not, write to the Free Software + Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA +*/ + +// LowMemoryPolyDataMapper -- VTK 9.5.0's low-memory (GLES3/WebGL2) poly-data +// mapper with two pixel-identical cuts to its per-draw cost. +// +// On GLES3/WebGL2 VTK's object factory hands every vtkPolyDataMapper::New() a +// vtkOpenGLLowMemoryPolyDataMapper. Its draw runs four cell-type "agents" in a +// row (verts, lines, polys, strips), and each one does the full shader setup +// whether or not it has anything to draw: every array texture re-bound, every +// camera, material, shadow and agent uniform re-sent, the VAO bound and +// unbound. A typical mesh has one cell type, so three quarters of that is +// thrown away -- about 110 of the ~160 GL calls a lit mesh costs per draw. On +// top of that, every draw re-resolves its shader program through the shader +// cache: five source copies, re-substitution and an MD5 of ~16 KB, to find the +// program it already has. +// +// This subclass: +// 1. skips the cell types that draw nothing (a cell type draws iff its first +// cell group can render, exactly the condition VTK's agent tests), and +// 2. re-binds the program it resolved last time while the shader has not been +// rebuilt since (same ShaderBuildTimeStamp, same render window shader cache) +// instead of looking it up again. +// Nothing else changes: the drawn cell type gets the same uniforms, textures and +// draw call in the same order, so frames are byte-identical (the cvcgl_lowmem_ +// fastdraw test pins that, and pins the replica against VTK call-for-call). +// +// The cell-type agents are VTK private classes (headers not installed, symbols +// hidden), so the drawn agent's PreDraw/Draw/PostDraw is replicated here from +// VTK 9.5.0. That replica compiles only against exactly 9.5.0; on any other +// VTK this class forwards to the stock draw and only the factory below remains. +// Picking, vertex visibility and a coincident-offset layout where skipping a +// cell type would change the depth offset a drawn one sees all take the stock +// (all four agents) draw. +// +// It also zero-initialises the shift/scale members VTK 9.5.0 leaves +// uninitialised (with DISABLE_SHIFT_SCALE nothing ever assigns them, so a +// garbage "in use" byte made the mapper upload a shift/scale copy of the +// positions). +#ifndef CVC_GL_LOW_MEMORY_POLY_DATA_MAPPER_H +#define CVC_GL_LOW_MEMORY_POLY_DATA_MAPPER_H + +#include +#include +#include +#include +#include + +namespace cvc { +namespace gl { + +class LowMemoryPolyDataMapper : public vtkOpenGLLowMemoryPolyDataMapper { +public: + static LowMemoryPolyDataMapper *New(); + vtkTypeMacro(LowMemoryPolyDataMapper, vtkOpenGLLowMemoryPolyDataMapper); + + // How every LowMemoryPolyDataMapper in the process draws. The initial value + // comes from CVCGL_LOWMEM_DRAW (stock | all | fast, default fast); the other + // two paths exist for A/B measurement and for the tests: + // Stock VTK's own RenderPieceDraw (all four agents, full lookup). + // AllCellTypes the replica with nothing skipped -- issues the same GL calls + // as Stock, call for call. + // Fast the replica, empty cell types skipped, program re-bound. + enum class DrawPath : int { Stock = 0, AllCellTypes = 1, Fast = 2 }; + static void setDrawPath(DrawPath path); + static DrawPath drawPath(); + + // True when this build carries the VTK 9.5.0 replica. False means every path + // is the stock draw. + static bool replicaActive(); + + // Per-mapper counters, read and reset on the render thread. + struct Stats { + std::uint64_t draws = 0; // RenderPieceDraw calls + std::uint64_t stockDraws = 0; // of which drew through VTK's own path + std::uint64_t cellTypesSkipped = 0; // agents not run because they draw nothing + std::uint64_t coincidentFallbacks = 0; // draws that ran all four for the depth offset + std::uint64_t programLookups = 0; // full shader-cache lookups (copy + MD5) + std::uint64_t programLookupsSkipped = 0; // cached program re-bound instead + }; + const Stats &stats() const { return m_stats; } + void resetStats() { m_stats = Stats{}; } + + void RenderPieceDraw(vtkRenderer *ren, vtkActor *act) override; + void ReleaseGraphicsResources(vtkWindow *win) override; + +protected: + LowMemoryPolyDataMapper(); + ~LowMemoryPolyDataMapper() override = default; + + // Bind this draw's shader program: the full lookup, or -- when allowSkip and + // nothing it depends on changed -- the program the last lookup returned. + void readyProgram(vtkRenderer *ren, bool allowSkip); + // Whether cell type t (0 verts, 1 lines, 2 polys, 3 strips) draws anything. + bool renderable(int t) const; + // Whether skipping the non-drawing cell types leaves every drawn one, and the + // program afterwards, with the depth offset the stock draw would. + bool coincidentSkipSafe(vtkActor *act) const; + // VTK 9.5.0's vtkOpenGLLowMemoryCellTypeAgent::PreDraw / Draw for cell type t. + void agentPreDraw(int t, vtkRenderer *ren, vtkActor *act); + void agentDraw(int t, vtkRenderer *ren, vtkActor *act); + + vtkWeakPointer m_cache; // cache the program was last resolved from + vtkMTimeType m_builtStamp = 0; // ShaderBuildTimeStamp at that lookup + Stats m_stats; + +private: + LowMemoryPolyDataMapper(const LowMemoryPolyDataMapper &) = delete; + void operator=(const LowMemoryPolyDataMapper &) = delete; +}; + +// Which mapper newPolyDataMapper() returns. The initial value comes from the +// CVCGL_LOWMEM_MAPPER environment variable (auto | force | off, default auto): +// Auto LowMemoryPolyDataMapper wherever VTK's factory would hand out a +// vtkOpenGLLowMemoryPolyDataMapper (GLES3/WebGL2), the factory's mapper +// everywhere else (vtkOpenGLPolyDataMapper on desktop GL). +// Force LowMemoryPolyDataMapper everywhere, desktop GL included -- for native +// tests and GL-call measurement of the WebGL path. +// Off the factory's mapper everywhere. +enum class LowMemoryMapperPolicy : int { Auto = 0, Force = 1, Off = 2 }; +void setLowMemoryMapperPolicy(LowMemoryMapperPolicy policy); +LowMemoryMapperPolicy lowMemoryMapperPolicy(); + +// The poly-data mapper for a cvcGL node, chosen by the policy above. +vtkSmartPointer newPolyDataMapper(); + +} // namespace gl +} // namespace cvc + +#endif // CVC_GL_LOW_MEMORY_POLY_DATA_MAPPER_H diff --git a/src/cvcGL/BBoxNode.cpp b/src/cvcGL/BBoxNode.cpp index fbdccbd95..222043ce6 100644 --- a/src/cvcGL/BBoxNode.cpp +++ b/src/cvcGL/BBoxNode.cpp @@ -1,5 +1,6 @@ #include #include +#include #include #include #include @@ -21,10 +22,9 @@ namespace cvc { namespace gl { BBoxNode::BBoxNode() - : m_actor(vtkSmartPointer::New()), - m_mapper(vtkSmartPointer::New()), m_bbox(-1.0, -1.0, -1.0, 1.0, 1.0, 1.0), - m_transform(vtkSmartPointer::New()), m_coordinatesVisible(true), - m_coordinateLabelFontSize(12), m_renderer(nullptr) { + : m_actor(vtkSmartPointer::New()), m_mapper(newPolyDataMapper()), + m_bbox(-1.0, -1.0, -1.0, 1.0, 1.0, 1.0), m_transform(vtkSmartPointer::New()), + m_coordinatesVisible(true), m_coordinateLabelFontSize(12), m_renderer(nullptr) { m_transform->Identity(); m_actor->SetMapper(m_mapper); diff --git a/src/cvcGL/GeometryNode.cpp b/src/cvcGL/GeometryNode.cpp index a4abde8ba..050beb6d1 100644 --- a/src/cvcGL/GeometryNode.cpp +++ b/src/cvcGL/GeometryNode.cpp @@ -5,6 +5,7 @@ #include #include #include +#include #include #include #include @@ -48,14 +49,13 @@ namespace cvc { namespace gl { GeometryNode::GeometryNode(cvc::app &ctx, const std::string &statePath, const std::string &name) - : GeometryNode(ctx, statePath, name, vtkSmartPointer::New()) {} + : GeometryNode(ctx, statePath, name, newPolyDataMapper()) {} GeometryNode::GeometryNode(cvc::app &ctx, const std::string &statePath, const std::string &name, vtkSmartPointer mapper) : GraphicsNode(ctx, statePath, name), m_hasGeometry(false), m_renderMode(GeometryRenderMode::TRIS), m_useSingleColor(false), - m_actor(vtkSmartPointer::New()), - m_mapper(mapper ? mapper : vtkSmartPointer::New()), + m_actor(vtkSmartPointer::New()), m_mapper(mapper ? mapper : newPolyDataMapper()), m_polyData(vtkSmartPointer::New()), m_textureFlipV(false) { m_mapper->SetInputData(m_polyData); m_actor->SetMapper(m_mapper); diff --git a/src/cvcGL/GridNode.cpp b/src/cvcGL/GridNode.cpp index b51d75366..1ea4cb14c 100644 --- a/src/cvcGL/GridNode.cpp +++ b/src/cvcGL/GridNode.cpp @@ -1,6 +1,7 @@ #include #include #include +#include #include #include #include @@ -24,12 +25,10 @@ namespace gl { GridNode::GridNode(cvc::app &ctx, const std::string &statePath, const std::string &name) : GraphicsNode(ctx, statePath, name), m_yzActor(vtkSmartPointer::New()), m_xzActor(vtkSmartPointer::New()), m_xyActor(vtkSmartPointer::New()), - m_yzMapper(vtkSmartPointer::New()), - m_xzMapper(vtkSmartPointer::New()), - m_xyMapper(vtkSmartPointer::New()), - m_bounds(-10.0, -10.0, -10.0, 10.0, 10.0, 10.0), m_divisionsX(64), m_divisionsY(64), - m_divisionsZ(64), m_tickIntervalX(8), m_tickIntervalY(8), m_tickIntervalZ(8), - m_tickLabelFontSize(12), m_yzPlaneVisible(true), m_xzPlaneVisible(true), + m_yzMapper(newPolyDataMapper()), m_xzMapper(newPolyDataMapper()), + m_xyMapper(newPolyDataMapper()), m_bounds(-10.0, -10.0, -10.0, 10.0, 10.0, 10.0), + m_divisionsX(64), m_divisionsY(64), m_divisionsZ(64), m_tickIntervalX(8), m_tickIntervalY(8), + m_tickIntervalZ(8), m_tickLabelFontSize(12), m_yzPlaneVisible(true), m_xzPlaneVisible(true), m_xyPlaneVisible(true), m_renderer(nullptr) { // Initialize default colors m_yzPlaneColor[0] = m_yzPlaneColor[1] = m_yzPlaneColor[2] = 0.5; diff --git a/src/cvcGL/LowMemoryPolyDataMapper.cpp b/src/cvcGL/LowMemoryPolyDataMapper.cpp new file mode 100644 index 000000000..cb4fde5e2 --- /dev/null +++ b/src/cvcGL/LowMemoryPolyDataMapper.cpp @@ -0,0 +1,362 @@ +/* + Copyright 2026 The University of Texas at Austin + + This file is part of libcvc. + + libcvc is free software; you can redistribute it and/or + modify it under the terms of the GNU Lesser General Public + License version 2.1 as published by the Free Software Foundation. + + libcvc is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU + Lesser General Public License for more details. + + You should have received a copy of the GNU Lesser General Public + License along with this library; if not, write to the Free Software + Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA +*/ + +// No vtkOpenGL*ErrorMacro anywhere in this file: those macros are inline in the +// including translation unit, so a cvcGL build without NDEBUG would issue a +// glGetError per draw -- a synchronous round trip on WebGL. +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#if VTK_MAJOR_VERSION == 9 && VTK_MINOR_VERSION == 5 && VTK_BUILD_VERSION == 0 +#define CVC_GL_LOWMEM_REPLICA 1 +#else +#define CVC_GL_LOWMEM_REPLICA 0 +#endif + +namespace cvc { +namespace gl { + +namespace { + +int drawPathFromEnvironment() { + const char *v = std::getenv("CVCGL_LOWMEM_DRAW"); + if (v && std::strcmp(v, "stock") == 0) + return static_cast(LowMemoryPolyDataMapper::DrawPath::Stock); + if (v && std::strcmp(v, "all") == 0) + return static_cast(LowMemoryPolyDataMapper::DrawPath::AllCellTypes); + return static_cast(LowMemoryPolyDataMapper::DrawPath::Fast); +} + +std::atomic &drawPathStore() { + static std::atomic path{drawPathFromEnvironment()}; + return path; +} + +int policyFromEnvironment() { + const char *v = std::getenv("CVCGL_LOWMEM_MAPPER"); + if (v && std::strcmp(v, "force") == 0) + return static_cast(LowMemoryMapperPolicy::Force); + if (v && std::strcmp(v, "off") == 0) + return static_cast(LowMemoryMapperPolicy::Off); + return static_cast(LowMemoryMapperPolicy::Auto); +} + +std::atomic &policyStore() { + static std::atomic policy{policyFromEnvironment()}; + return policy; +} + +// Does VTK's object factory hand out the low-memory mapper on this build? Asked +// once, on the first node construction (the OpenGL2 factory is registered by +// then: cvcGL's sources carry the module auto-init). +bool factoryIsLowMemory() { + static const bool lowMemory = + vtkSmartPointer::New()->IsA("vtkOpenGLLowMemoryPolyDataMapper") != 0; + return lowMemory; +} + +#if CVC_GL_LOWMEM_REPLICA +// The depth offset vtkGLSLModCoincidentTopology::GetCoincidentParameters (VTK +// 9.5.0) computes for one primitive class, as the floats it would upload. +// slot: 0 points, 1 lines, 2 polygons. `set` false: the mod uploads nothing. +struct CoincidentValue { + bool set = false; + float factor = 0.0f, offset = 0.0f; + bool operator==(const CoincidentValue &o) const { + return set == o.set && (!set || (factor == o.factor && offset == o.offset)); + } + bool operator!=(const CoincidentValue &o) const { return !(*this == o); } +}; + +CoincidentValue coincidentValue(vtkMapper *mapper, vtkProperty *prop, int slot) { + float factor = 0.0f, offset = 0.0f; + if (vtkMapper::GetResolveCoincidentTopology() == VTK_RESOLVE_SHIFT_ZBUFFER) + offset = static_cast(vtkMapper::GetResolveCoincidentTopologyZShift() * 4.0); + if (vtkMapper::GetResolveCoincidentTopology() == VTK_RESOLVE_POLYGON_OFFSET || + (prop->GetEdgeVisibility() && prop->GetRepresentation() == VTK_SURFACE)) { + double f = 0.0, u = 0.0; + if (slot == 0) + mapper->GetCoincidentTopologyPointOffsetParameter(u); + else if (slot == 1) + mapper->GetCoincidentTopologyLineOffsetParameters(f, u); + else + mapper->GetCoincidentTopologyPolygonOffsetParameters(f, u); + factor = static_cast(f); + offset = static_cast(u); + } + CoincidentValue v; + v.set = factor != 0.0f || offset != 0.0f; + v.factor = factor; + v.offset = offset; + return v; +} +#endif + +} // namespace + +vtkStandardNewMacro(LowMemoryPolyDataMapper); + +LowMemoryPolyDataMapper::LowMemoryPolyDataMapper() { +#if CVC_GL_LOWMEM_REPLICA + // VTK 9.5.0 declares these without initialisers. With DISABLE_SHIFT_SCALE + // nothing ever assigns them, so whether the mapper uploaded a (garbage) + // shift/scale copy of the positions depended on a heap byte. + this->CoordinateShiftAndScaleInUse = false; + this->ShiftValues.fill(0.0); + this->ScaleValues.fill(1.0); +#endif +} + +void LowMemoryPolyDataMapper::setDrawPath(DrawPath path) { + drawPathStore().store(static_cast(path), std::memory_order_relaxed); +} + +LowMemoryPolyDataMapper::DrawPath LowMemoryPolyDataMapper::drawPath() { + return static_cast(drawPathStore().load(std::memory_order_relaxed)); +} + +bool LowMemoryPolyDataMapper::replicaActive() { return CVC_GL_LOWMEM_REPLICA != 0; } + +void LowMemoryPolyDataMapper::ReleaseGraphicsResources(vtkWindow *win) { + // The cached program belongs to that window's context; look it up afresh. + m_cache = nullptr; + m_builtStamp = 0; + Superclass::ReleaseGraphicsResources(win); +} + +#if !CVC_GL_LOWMEM_REPLICA + +void LowMemoryPolyDataMapper::RenderPieceDraw(vtkRenderer *ren, vtkActor *act) { + ++m_stats.draws; + ++m_stats.stockDraws; + Superclass::RenderPieceDraw(ren, act); +} +void LowMemoryPolyDataMapper::readyProgram(vtkRenderer *ren, bool) { + this->vtkDrawTexturedElements::ReadyShaderProgram(ren); +} +bool LowMemoryPolyDataMapper::renderable(int) const { return true; } +bool LowMemoryPolyDataMapper::coincidentSkipSafe(vtkActor *) const { return false; } +void LowMemoryPolyDataMapper::agentPreDraw(int, vtkRenderer *, vtkActor *) {} +void LowMemoryPolyDataMapper::agentDraw(int, vtkRenderer *, vtkActor *) {} + +#else + +void LowMemoryPolyDataMapper::RenderPieceDraw(vtkRenderer *ren, vtkActor *act) { + ++m_stats.draws; + const DrawPath path = drawPath(); + // Selection (picking state, point-picking sizes) and the vertex-visibility + // pass stay on VTK's own path; the replica covers the plain draw only. + if (path == DrawPath::Stock || ren->GetSelector() || act->GetProperty()->GetVertexVisibility()) { + ++m_stats.stockDraws; + Superclass::RenderPieceDraw(ren, act); + return; + } + const bool fast = path == DrawPath::Fast; + readyProgram(ren, fast); + // Uniforms common to every cell type; also fires UpdateShaderEvent, which + // GeometryNode's custom shader textures hang off. + this->SetShaderParameters(ren, act); + if (!this->ShaderProgram) + return; // compile/link failure: VTK's agents would dereference null here + bool skip = fast; + if (skip && !coincidentSkipSafe(act)) { + skip = false; + ++m_stats.coincidentFallbacks; + } + for (int t = 0; t < 4; ++t) { // verts, lines, polys, strips: VTK's order + if (skip && !renderable(t)) { + ++m_stats.cellTypesSkipped; + continue; + } + agentPreDraw(t, ren, act); + agentDraw(t, ren, act); + // vtkOpenGLLowMemoryCellTypeAgent::PostDraw; every 9.5.0 PostDrawInternal is empty. + this->vtkDrawTexturedElements::PostDraw(ren, act, this); + } +} + +void LowMemoryPolyDataMapper::readyProgram(vtkRenderer *ren, bool allowSkip) { + auto *win = vtkOpenGLRenderWindow::SafeDownCast(ren->GetRenderWindow()); + vtkOpenGLShaderCache *cache = win ? win->GetShaderCache() : nullptr; + const vtkMTimeType stamp = this->ShaderBuildTimeStamp.GetMTime(); + // this->Shaders only changes in UpdateShaders, which RenderPieceStart follows + // with ShaderBuildTimeStamp.Modified(); while the stamp and the cache are the + // ones the last lookup saw, that lookup would find the same program again. + // ReadyShaderProgram(program) still compiles it if its context was released + // and binds it exactly as the lookup's tail would. + if (allowSkip && cache && this->ShaderProgram && cache == m_cache.GetPointer() && + stamp == m_builtStamp && this->ElementType != AbstractPatches) { + this->ShaderProgram = cache->ReadyShaderProgram(this->ShaderProgram.GetPointer()); + ++m_stats.programLookupsSkipped; + return; + } + this->vtkDrawTexturedElements::ReadyShaderProgram(ren); // copy, substitute, MD5, find, bind + ++m_stats.programLookups; + m_cache = cache; + m_builtStamp = stamp; +} + +bool LowMemoryPolyDataMapper::renderable(int t) const { + // vtkOpenGLLowMemoryCellTypeAgent::Draw draws cell group 0 iff it CanRender. + const auto &groups = this->Primitives[t].CellGroups; + return !groups.empty() && groups[0].CanRender; +} + +bool LowMemoryPolyDataMapper::coincidentSkipSafe(vtkActor *act) const { + // vtkGLSLModCoincidentTopology sets cOffset/cFactor in every PreDraw, per + // primitive class, and only when the value is non-zero; otherwise the program + // keeps whatever was set last. So a cell type whose own offset is zero sees + // what an earlier cell type -- possibly one with nothing to draw -- left + // behind, and the last one leaves a value for the next draw of that program. + // Skipping is safe when every drawn type, and the program afterwards, ends up + // with the same value either way. + vtkProperty *prop = act->GetProperty(); + const int mode = vtkMapper::GetResolveCoincidentTopology(); + if (mode != VTK_RESOLVE_POLYGON_OFFSET && mode != VTK_RESOLVE_SHIFT_ZBUFFER && + !(prop->GetEdgeVisibility() && prop->GetRepresentation() == VTK_SURFACE)) + return true; // no offset uniform is ever set (VTK's default) + auto *self = const_cast(this); + const bool points = prop->GetRepresentation() == VTK_POINTS; + CoincidentValue stock, skipped; // unset: unchanged since before this draw + for (int t = 0; t < 4; ++t) { + const int slot = points ? 0 : (t == 0 ? 0 : (t == 1 ? 1 : 2)); + const CoincidentValue v = coincidentValue(self, prop, slot); + if (v.set) + stock = v; + if (renderable(t)) { + if (v.set) + skipped = v; + if (stock != skipped) + return false; + } + } + return stock == skipped; +} + +void LowMemoryPolyDataMapper::agentPreDraw(int t, vtkRenderer *ren, vtkActor *act) { + // vtkOpenGLLowMemory{Vertices,Lines,Polygons}Agent::PreDrawInternal (the + // strips agent is the polygons agent) ... + vtkProperty *prop = act->GetProperty(); + int pointsPerPrimitive = 3; + if (t == 0) { + pointsPerPrimitive = 1; + this->ElementType = vtkDrawTexturedElements::ElementShape::Point; + this->NumberOfInstances = 1; + this->ShaderProgram->SetUniformi("cellType", VTK_VERTEX); + } else if (t == 1) { + pointsPerPrimitive = 2; + this->ElementType = vtkDrawTexturedElements::ElementShape::Line; + if (prop->GetLineWidth() > 1) + this->NumberOfInstances = 2 * vtkMath::Ceil(prop->GetLineWidth()); + else + this->NumberOfInstances = 1; + this->ShaderProgram->SetUniformi("cellType", VTK_LINE); + } else { + this->ElementType = vtkDrawTexturedElements::ElementShape::Triangle; + this->NumberOfInstances = 1; + this->ShaderProgram->SetUniformi("cellType", VTK_TRIANGLE); + } + // ... then vtkOpenGLLowMemoryCellTypeAgent::PreDraw (never in the vertex + // visibility pass, which takes the stock draw). + if (prop->GetRepresentation() == VTK_POINTS) + this->ElementType = vtkDrawTexturedElements::ElementShape::Point; + bool needLighting = false; + if (prop->GetRepresentation() == VTK_POINTS) { + needLighting = prop->GetInterpolation() != VTK_FLAT && this->HasPointNormals; + } else { + const bool isTrisOrStrips = pointsPerPrimitive >= 3; + needLighting = isTrisOrStrips || (!isTrisOrStrips && prop->GetInterpolation() != VTK_FLAT && + this->HasPointNormals); + } + this->ShaderProgram->SetUniformi("enable_lights", needLighting); + this->ShaderProgram->SetUniformi("vertex_pass", false); + switch (this->ElementType) { + case vtkDrawTexturedElements::ElementShape::Point: + this->ShaderProgram->SetUniformi("primitiveSize", 1); + break; + case vtkDrawTexturedElements::ElementShape::Line: + this->ShaderProgram->SetUniformi("primitiveSize", 2); + break; + case vtkDrawTexturedElements::ElementShape::Triangle: + default: + this->ShaderProgram->SetUniformi("primitiveSize", 3); + break; + } + // PointPicking is only ever set with a selector, which takes the stock draw. + this->ShaderProgram->SetUniformf("pointSize", prop->GetPointSize()); + this->vtkDrawTexturedElements::PreDraw(ren, act, this); +} + +void LowMemoryPolyDataMapper::agentDraw(int t, vtkRenderer *ren, vtkActor *act) { + // vtkOpenGLLowMemoryCellTypeAgent::Draw, cell group 0. + const auto &groups = this->Primitives[t].CellGroups; + if (groups.empty() || !groups[0].CanRender) + return; + const auto &group = groups[0]; + const auto &offsets = group.Offsets; + this->FirstVertexId = offsets.VertexIdOffset; + this->NumberOfElements = group.NumberOfElements; + // when rendering vertices, increase number of elements and draw 1 instance. + if (act->GetProperty()->GetRepresentation() == VTK_POINTS) { + this->NumberOfElements *= (t == 0 ? 1 : (t == 1 ? 2 : 3)); + this->NumberOfInstances = 1; + } + vtkShaderProgram *program = this->ShaderProgram; + program->SetUniformi("cellIdOffset", offsets.CellIdOffset); + program->SetUniformi("vertexIdOffset", offsets.VertexIdOffset); + program->SetUniformi("edgeValueBufferOffset", offsets.EdgeValueBufferOffset); + program->SetUniformi("pointIdOffset", offsets.PointIdOffset); + program->SetUniformi("primitiveIdOffset", offsets.PrimitiveIdOffset); + program->SetUniformi("usesCellMap", group.UsesCellMapBuffer); + program->SetUniformi("usesEdgeValues", group.UsesEdgeValueBuffer); + this->vtkDrawTexturedElements::DrawInstancedElementsImpl(ren, act, this); +} + +#endif // CVC_GL_LOWMEM_REPLICA + +void setLowMemoryMapperPolicy(LowMemoryMapperPolicy policy) { + policyStore().store(static_cast(policy), std::memory_order_relaxed); +} + +LowMemoryMapperPolicy lowMemoryMapperPolicy() { + return static_cast(policyStore().load(std::memory_order_relaxed)); +} + +vtkSmartPointer newPolyDataMapper() { + const LowMemoryMapperPolicy policy = lowMemoryMapperPolicy(); + if (policy == LowMemoryMapperPolicy::Force || + (policy == LowMemoryMapperPolicy::Auto && factoryIsLowMemory())) + return vtkSmartPointer::New(); + return vtkSmartPointer::New(); +} + +} // namespace gl +} // namespace cvc diff --git a/src/cvcGL/nodes.cpp b/src/cvcGL/nodes.cpp index 502a2e94e..ff25ec9b0 100644 --- a/src/cvcGL/nodes.cpp +++ b/src/cvcGL/nodes.cpp @@ -11,6 +11,7 @@ #include #include #include +#include #include #include #include @@ -294,7 +295,7 @@ void GeometryShape::setGeometry(const cvc::geometry &geom) { } vtkSmartPointer GeometryShape::createProp(RenderView &) { - auto mapper = vtkSmartPointer::New(); + auto mapper = newPolyDataMapper(); if (m_polyData) mapper->SetInputData(m_polyData); auto actor = vtkSmartPointer::New(); From eda59f72a6a5b89e0457809536ae6a00180d780c Mon Sep 17 00:00:00 2001 From: Joe Rivera Date: Thu, 1 Oct 2026 17:39:13 -0500 Subject: [PATCH 02/18] test(cvcGL): pin LowMemoryPolyDataMapper call-for-call and pixel-exact; add a bench cvcgl_lowmem_fastdraw forces the low-memory mapper natively and runs cvcGL's real pipeline (shadow baker -> shadow map -> translucent -> volumetric -> overlay, and shadows off) over every GeometryNode kind plus raw actors whose mapper tallies the GL calls of its own draws. test/gl_call_counter swaps VTK's glad pointers for counting wrappers (with uniform-redundancy tracking). It checks the mapper policy; that DrawPath::AllCellTypes issues exactly VTK's GL calls per entry point on bake and steady frames; that Stock, AllCellTypes and Fast frames are byte-identical RGBA; the per-draw saving (one VAO bind/unbind, no sync query, no shader-cache lookup on a steady frame); lifecycle (shadow-bake rebuilds, a mapper's released resources, programs released in the shader cache, the scene re-opened in a new window); and the guards (offset layouts that need all four agents, picking, vertex visibility). Frames render without MSAA: VTK's default 8x is not frame-to-frame deterministic on NVIDIA (1 LSB at a few edge pixels for an identical GL stream). cvcgl_lowmem_bench (not a test) prints GL calls and draw-stage CPU per frame and per draw, Stock vs Fast, for a demo-style scene with shadows on and off. --- src/cvcGL/CMakeLists.txt | 22 + src/cvcGL/test/cvcgl_lowmem_bench.cpp | 440 ++++++++++ src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp | 969 +++++++++++++++++++++++ src/cvcGL/test/gl_call_counter.cpp | 383 +++++++++ src/cvcGL/test/gl_call_counter.h | 52 ++ 5 files changed, 1866 insertions(+) create mode 100644 src/cvcGL/test/cvcgl_lowmem_bench.cpp create mode 100644 src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp create mode 100644 src/cvcGL/test/gl_call_counter.cpp create mode 100644 src/cvcGL/test/gl_call_counter.h diff --git a/src/cvcGL/CMakeLists.txt b/src/cvcGL/CMakeLists.txt index 661388ca5..b95df6472 100644 --- a/src/cvcGL/CMakeLists.txt +++ b/src/cvcGL/CMakeLists.txt @@ -595,6 +595,28 @@ if(NOT EMSCRIPTEN) endif() endif() +# LowMemoryPolyDataMapper (the GLES3/WebGL2 mapper with empty cell types and the +# per-draw shader-cache lookup taken out of its draw), forced on natively. Counts +# every GL call at the driver (test/gl_call_counter.cpp swaps VTK's glad function +# pointers) to pin the VTK 9.5.0 replica call-for-call against the stock draw and +# measure the saving, and compares RGBA frames byte for byte -- shadows on and +# off, bake and steady frames, released resources, a re-opened window, the +# offset / picking / vertex-visibility guards. Renders for real; skips where +# nothing rasterises unless CVC_REQUIRE_RENDER=1. CVCGL_LOWMEM_REPORT=1 prints +# per-draw breakdowns. Desktop GL only: the wasm VTK calls GLES directly, not +# through glad. +# +# cvcgl_lowmem_bench (not a test) prints GL calls and draw-stage CPU per frame +# and per draw, Stock vs Fast, for a city of single-colour box nodes on a +# textured ground with overlays, shadows on and off. +if(NOT EMSCRIPTEN) + add_executable(cvcgl_lowmem_fastdraw test/cvcgl_lowmem_fastdraw.cpp test/gl_call_counter.cpp) + target_link_libraries(cvcgl_lowmem_fastdraw PRIVATE cvcGL) + add_test(NAME cvcgl_lowmem_fastdraw COMMAND cvcgl_lowmem_fastdraw) + add_executable(cvcgl_lowmem_bench test/cvcgl_lowmem_bench.cpp test/gl_call_counter.cpp) + target_link_libraries(cvcgl_lowmem_bench PRIVATE cvcGL) +endif() + # SdlInput: SDL3 event -> normalized cvc::gl::InputEvent translation (the §4.6 input seam's SDL # source). Pushes synthetic SDL key/mouse events onto the queue and verifies poll() returns them. # Headless (the SDL EVENTS subsystem needs no window). SKIPs (rc 0) in a build without SDL. diff --git a/src/cvcGL/test/cvcgl_lowmem_bench.cpp b/src/cvcGL/test/cvcgl_lowmem_bench.cpp new file mode 100644 index 000000000..a358a73a7 --- /dev/null +++ b/src/cvcGL/test/cvcgl_lowmem_bench.cpp @@ -0,0 +1,440 @@ +/* + Copyright 2026 The University of Texas at Austin + + This file is part of libcvc. + + libcvc is free software; you can redistribute it and/or + modify it under the terms of the GNU Lesser General Public + License version 2.1 as published by the Free Software Foundation. +*/ + +// Not a test: prints what cvc::gl::LowMemoryPolyDataMapper saves per frame and +// per draw, DrawPath::Stock against DrawPath::Fast, on desktop GL with the +// low-memory mapper forced (the mapper WebGL2 gets). +// +// Scene (a demo-style frame): a textured ground, a merged city mesh of N boxes, +// V single-colour vehicle nodes, translucent route ribbons with a depth offset, +// line trails, and optionally many extra small single-colour nodes; shadows on +// (camera -> strided shadow baker -> shadow map -> translucent -> volumetric -> +// overlay) and off. Each node's mapper is swapped for one that tallies the GL +// calls and CPU time of its own draws, so "per draw" is exactly +// RenderPieceDraw. Frames are rendered without glFinish; times are CPU. +// +// cvcgl_lowmem_bench [--frames=40] [--boxes=20000] [--vehicles=6] [--extra=0] +// [--size=960x540] +#include "gl_call_counter.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +using cvc::gl::GeometryNode; +using cvc::gl::GeometryRenderMode; +using cvc::gl::LowMemoryPolyDataMapper; +using cvc::gl::SceneGraph; +using cvc::gl::SceneRenderer; +using cvcgl_test::GLCalls; +using cvcgl_test::glcallsRead; +using DrawPath = LowMemoryPolyDataMapper::DrawPath; +using clk = std::chrono::steady_clock; + +namespace { + +double msSince(clk::time_point t0) { + return std::chrono::duration(clk::now() - t0).count(); +} + +// Tallies the GL calls and CPU time of every draw of every TallyMapper. +GLCalls g_drawCalls; +double g_drawMs = 0, g_draws = 0; + +class TallyMapper : public LowMemoryPolyDataMapper { +public: + static TallyMapper *New(); + vtkTypeMacro(TallyMapper, LowMemoryPolyDataMapper); + void RenderPieceDraw(vtkRenderer *ren, vtkActor *act) override { + const GLCalls before = glcallsRead(); + const auto t0 = clk::now(); + Superclass::RenderPieceDraw(ren, act); + g_drawMs += msSince(t0); + g_drawCalls += glcallsRead() - before; + g_draws += 1; + } + +protected: + TallyMapper() = default; +}; +vtkStandardNewMacro(TallyMapper); + +class PeekNode : public GeometryNode { +public: + using GeometryNode::GeometryNode; + vtkActor *actor() { return vtkActor::SafeDownCast(getProp()); } +}; + +// Replace a node's mapper with a TallyMapper carrying the same settings +// (GeometryNode configures nothing else on it that these nodes use). +void swapInTally(PeekNode &n) { + vtkActor *a = n.actor(); + auto *old = LowMemoryPolyDataMapper::SafeDownCast(a->GetMapper()); + if (!old) + return; + vtkNew m; + m->SetInputData(vtkPolyData::SafeDownCast(old->GetInput())); + m->SetScalarVisibility(old->GetScalarVisibility()); + m->SetScalarMode(old->GetScalarMode()); + m->SetColorMode(old->GetColorMode()); + m->SetVBOShiftScaleMethod(old->GetVBOShiftScaleMethod()); + double f = 0, u = 0; + old->GetRelativeCoincidentTopologyPolygonOffsetParameters(f, u); + m->SetRelativeCoincidentTopologyPolygonOffsetParameters(f, u); + old->GetRelativeCoincidentTopologyLineOffsetParameters(f, u); + m->SetRelativeCoincidentTopologyLineOffsetParameters(f, u); + old->GetRelativeCoincidentTopologyPointOffsetParameter(u); + m->SetRelativeCoincidentTopologyPointOffsetParameter(u); + a->SetMapper(m); +} + +void addBox(cvc::geometry &g, double cx, double cy, double cz, double sx, double sy, double sz) { + const double x0 = cx - sx, x1 = cx + sx, y0 = cy - sy, y1 = cy + sy, z0 = cz - sz, z1 = cz + sz; + const double faces[5][4][3] = {{{x0, y0, z1}, {x1, y0, z1}, {x1, y1, z1}, {x0, y1, z1}}, + {{x0, y0, z0}, {x1, y0, z0}, {x1, y0, z1}, {x0, y0, z1}}, + {{x1, y1, z0}, {x0, y1, z0}, {x0, y1, z1}, {x1, y1, z1}}, + {{x0, y1, z0}, {x0, y0, z0}, {x0, y0, z1}, {x0, y1, z1}}, + {{x1, y0, z0}, {x1, y1, z0}, {x1, y1, z1}, {x1, y0, z1}}}; + for (const auto &f : faces) { + const unsigned b = static_cast(g.points().size()); + for (const auto &p : f) + g.points().push_back({p[0], p[1], p[2]}); + g.tris().push_back({b, b + 1, b + 2}); + g.tris().push_back({b, b + 2, b + 3}); + } +} + +struct Options { + int frames = 40, boxes = 20000, vehicles = 6, extra = 0, w = 960, h = 540; +}; + +struct Bench { + explicit Bench(cvc::app &app) : sg(app, "lowmem_bench") { sg.setDiagnosticChromeVisible(false); } + SceneGraph sg; + std::unique_ptr sr; + std::vector> nodes, vehicles; + + std::shared_ptr node(const std::string &name) { + auto n = sg.getGraphicsRoot()->addGraphicsChild(name); + nodes.push_back(n); + return n; + } +}; + +void build(Bench &b, const Options &o) { + const double half = 600.0; + { + auto ground = b.node("ground"); + cvc::geometry g; + const int n = 64; + for (int j = 0; j < n; ++j) + for (int i = 0; i < n; ++i) { + const double x = -half + 2 * half * i / (n - 1), y = -half + 2 * half * j / (n - 1); + g.points().push_back({x, y, 3.0 * std::sin(x / 90) * std::cos(y / 70)}); + g.uvs().push_back({double(i) / (n - 1), double(j) / (n - 1)}); + } + for (int j = 0; j + 1 < n; ++j) + for (int i = 0; i + 1 < n; ++i) { + const unsigned a = j * n + i, c = a + n; + g.tris().push_back({a, a + 1, c + 1}); + g.tris().push_back({a, c + 1, c}); + } + ground->setGeometry(g); + ground->setUseSingleColor(true); + ground->setColor(1, 1, 1); + ground->setAmbient(0.5); + ground->setDiffuse(0.6); + cvc::image img(512, 512, cvc::image::pixel_format::RGBA, cvc::image::data_type::u8); + unsigned char *p = img.data(); + for (int j = 0; j < 512; ++j) + for (int i = 0; i < 512; ++i) { + unsigned char *q = p + 4 * (j * 512 + i); + q[0] = static_cast(i / 2); + q[1] = static_cast(j / 2); + q[2] = 128; + q[3] = 255; + } + ground->setTexture(img); + } + if (o.boxes > 0) { + auto city = b.node("city"); + cvc::geometry g; + const int side = static_cast(std::ceil(std::sqrt(double(o.boxes)))); + for (int k = 0; k < o.boxes; ++k) { + const double x = -half * 0.9 + 1.8 * half * (k % side) / side; + const double y = -half * 0.9 + 1.8 * half * (k / side) / side; + const double hgt = 2.0 + 10.0 * (0.5 + 0.5 * std::sin(k * 0.37)); + addBox(g, x, y, hgt, 2.5, 2.5, hgt); + } + city->setGeometry(g); + city->setUseSingleColor(true); + city->setColor(0.72, 0.70, 0.66); + } + for (int v = 0; v < o.vehicles; ++v) { + auto veh = b.node("veh" + std::to_string(v)); + cvc::geometry g; + addBox(g, 0, 0, 2.0, 4.0, 2.0, 1.5); + veh->setGeometry(g); + veh->setUseSingleColor(true); + veh->setColor(0.2 + 0.1 * v, 0.5, 0.9 - 0.1 * v); + b.vehicles.push_back(veh); + } + for (int r = 0; r < 4; ++r) { + auto rib = b.node("route" + std::to_string(r)); + cvc::geometry g; + for (int k = 0; k < 120; ++k) { + const double x = -400 + 7.0 * k, y = -200 + 120.0 * r + 25 * std::sin(k * 0.1); + g.points().push_back({x, y - 2.0, 4.0}); + g.points().push_back({x, y + 2.0, 4.0}); + } + for (unsigned k = 0; k + 1 < 120; ++k) { + g.tris().push_back({2 * k, 2 * k + 1, 2 * k + 3}); + g.tris().push_back({2 * k, 2 * k + 3, 2 * k + 2}); + } + rib->setGeometry(g); + rib->setUseSingleColor(true); + rib->setColor(0.98, 0.8, 0.25); + rib->setOpacity(0.55); + rib->setDepthOffset(2.0); + + auto trail = b.node("trail" + std::to_string(r)); + cvc::geometry t; + for (int k = 0; k < 200; ++k) { + t.points().push_back({-450 + 4.0 * k, -150 + 110.0 * r + 15 * std::cos(k * 0.07), 5.0}); + if (k) + t.lines().push_back({static_cast(k - 1), static_cast(k)}); + } + trail->setGeometry(t); + trail->setRenderMode(GeometryRenderMode::LINES); // after setGeometry, which picks a mode + trail->setUseSingleColor(true); + trail->setColor(0.3, 1.0, 0.4); + } + for (int e = 0; e < o.extra; ++e) { + auto n = b.node("extra" + std::to_string(e)); + cvc::geometry g; + addBox(g, -500 + 1000.0 * (e % 20) / 20, -500 + 1000.0 * (e / 20) / 20, 3, 3, 3, 3); + n->setGeometry(g); + n->setUseSingleColor(true); + n->setColor(0.9, 0.4, 0.3); + } + b.sg.addSpotLight(-500, -700, 900, 0, 0, 0, 45.0, 1.0, 0.97, 0.9, 1.1); + b.sg.addFillLight(600, 500, 600, 0, 0, 0, 0.8, 0.85, 1.0, 0.4); +} + +void poseVehicles(Bench &b, int frame) { + for (size_t v = 0; v < b.vehicles.size(); ++v) { + const double t = 0.02 * frame + 1.1 * v, c = std::cos(t), s = std::sin(t); + const double m[16] = { + c, -s, 0, 300 * std::cos(0.3 * t + v), s, c, 0, 250 * std::sin(0.4 * t), 0, 0, 1, 0, 0, + 0, 0, 1}; + b.vehicles[v]->setPoseMatrix(m); + } +} + +struct Run { + double frames = 0, frameMs = 0, drawMs = 0, draws = 0; + GLCalls frame, draw; + LowMemoryPolyDataMapper::Stats stats; +}; + +LowMemoryPolyDataMapper::Stats sumStats(Bench &b) { + LowMemoryPolyDataMapper::Stats s; + for (auto &n : b.nodes) + if (auto *m = LowMemoryPolyDataMapper::SafeDownCast(n->actor()->GetMapper())) { + const auto &t = m->stats(); + s.draws += t.draws; + s.cellTypesSkipped += t.cellTypesSkipped; + s.programLookups += t.programLookups; + s.programLookupsSkipped += t.programLookupsSkipped; + s.stockDraws += t.stockDraws; + } + return s; +} + +Run measure(Bench &b, DrawPath path, bool bake, int frames, int &frameNo) { + LowMemoryPolyDataMapper::setDrawPath(path); + Run r; + for (auto &n : b.nodes) + if (auto *m = LowMemoryPolyDataMapper::SafeDownCast(n->actor()->GetMapper())) + m->resetStats(); + for (int f = 0; f < frames; ++f) { + poseVehicles(b, frameNo++); + if (bake) + b.sg.invalidateShadowBake(); + g_drawCalls = GLCalls(); + g_drawMs = 0; + g_draws = 0; + const GLCalls before = glcallsRead(); + const auto t0 = clk::now(); + b.sr->render(); + r.frameMs += msSince(t0); + r.frame += glcallsRead() - before; + r.draw += g_drawCalls; + r.drawMs += g_drawMs; + r.draws += g_draws; + r.frames += 1; + } + r.stats = sumStats(b); + return r; +} + +std::vector rgba(Bench &b) { + vtkNew px; + const int *sz = b.sr->renderWindow()->GetSize(); + b.sr->renderWindow()->GetRGBACharPixelData(0, 0, sz[0] - 1, sz[1] - 1, 0, px); + const unsigned char *p = px->GetPointer(0); + return std::vector(p, p + px->GetNumberOfValues()); +} + +void report(const char *label, const Run &s, const Run &f) { + const double n = s.frames; + std::printf("\n== %s (%g frames each)\n", label, n); + std::printf(" draws/frame %8.1f %8.1f\n", s.draws / n, f.draws / n); + std::printf(" Stock Fast\n"); + std::printf(" GL calls/frame %8.1f %8.1f (%+.1f, %+.0f%%)\n", s.frame.total() / n, + f.frame.total() / n, (f.frame.total() - s.frame.total()) / n, + 100.0 * (f.frame.total() - s.frame.total()) / s.frame.total()); + std::printf(" in mapper draws %8.1f %8.1f\n", s.draw.total() / n, f.draw.total() / n); + std::printf(" uniforms %8.1f %8.1f (redundant %.1f -> %.1f)\n", + s.frame.uniforms() / n, f.frame.uniforms() / n, s.frame.uniformRedundant / n, + f.frame.uniformRedundant / n); + std::printf(" sync queries %8.1f %8.1f\n", s.frame.sync() / n, f.frame.sync() / n); + std::printf(" GL calls/draw %8.1f %8.1f\n", s.draw.total() / s.draws, + f.draw.total() / f.draws); + std::printf(" draw-stage CPU ms/frame %6.3f %8.3f\n", s.drawMs / n, f.drawMs / n); + std::printf(" draw-stage CPU ms/draw %6.4f %8.4f\n", s.drawMs / s.draws, f.drawMs / f.draws); + std::printf(" frame CPU ms (no finish)%6.2f %8.2f\n", s.frameMs / n, f.frameMs / n); + std::printf(" lookups/frame %8.1f %8.1f (skipped %.1f)\n", + double(s.stats.programLookups + s.stats.stockDraws) / n, + double(f.stats.programLookups) / n, double(f.stats.programLookupsSkipped) / n); + std::printf(" Stock per frame: %s\n", s.frame.str(n).c_str()); + std::printf(" Fast per frame: %s\n", f.frame.str(n).c_str()); + std::printf(" Stock per draw : %s\n", s.draw.str(s.draws).c_str()); + std::printf(" Fast per draw : %s\n", f.draw.str(f.draws).c_str()); +} + +} // namespace + +int main(int argc, char **argv) { +#ifdef _WIN32 + _putenv_s("CVCGL_LOWMEM_MAPPER", "force"); +#else + setenv("CVCGL_LOWMEM_MAPPER", "force", 1); + setenv("__GL_SYNC_TO_VBLANK", "0", 0); // NVIDIA throttles offscreen swaps otherwise + setenv("vblank_mode", "0", 0); +#endif + Options o; + for (int i = 1; i < argc; ++i) { + const std::string a = argv[i]; + auto val = [&](const char *k) { + return a.rfind(k, 0) == 0 ? a.c_str() + std::strlen(k) : nullptr; + }; + if (const char *v = val("--frames=")) + o.frames = std::atoi(v); + else if (const char *v = val("--boxes=")) + o.boxes = std::atoi(v); + else if (const char *v = val("--vehicles=")) + o.vehicles = std::atoi(v); + else if (const char *v = val("--extra=")) + o.extra = std::atoi(v); + else if (const char *v = val("--size=")) + std::sscanf(v, "%dx%d", &o.w, &o.h); + } + if (!LowMemoryPolyDataMapper::replicaActive()) { + std::printf("this VTK is not 9.5.0: LowMemoryPolyDataMapper draws the stock way\n"); + return 0; + } + cvc::app app; + Bench b(app); + build(b, o); + b.sr = std::make_unique(b.sg, o.w, o.h, /*offscreen=*/true, "bench"); + // No MSAA: with VTK's default 8x, NVIDIA's resolve varies by 1 LSB at a few edge + // pixels between identical frames, which would hide the Stock/Fast comparison. + b.sr->renderWindow()->SetMultiSamples(0); + b.sr->setBackground(0.2, 0.26, 0.36); + b.sr->setCamera(-700, -900, 650, 0, 0, 0, 0, 0, 1, 42.0, 10.0, 6000.0); + for (auto &n : b.nodes) + swapInTally(*n); + b.sg.setShadowsEnabled(true); + b.sg.setShadowResolution(2048); + b.sg.setShadowUpdateInterval(100000); // bake only when invalidated + b.sr->render(); + b.sr->render(); + cvcgl_test::glcallsInstall(); + std::printf("lowmem bench: %zu nodes, %d boxes merged, %d vehicles, %d extra nodes, %dx%d\n", + b.nodes.size(), o.boxes, o.vehicles, o.extra, o.w, o.h); + + int frameNo = 0; + for (int shadows = 1; shadows >= 0; --shadows) { + if (!shadows) + b.sg.setShadowsEnabled(false); + for (DrawPath p : {DrawPath::Stock, DrawPath::Fast}) // warm both paths + measure(b, p, shadows != 0, 2, frameNo); + const auto kinds = shadows ? std::vector{true, false} : std::vector{false}; + for (bool bake : kinds) { + // Interleave the two paths in halves so drift hits both alike. + Run s, f; + for (int half = 0; half < 2; ++half) { + const Run s1 = measure(b, DrawPath::Stock, bake, o.frames / 2, frameNo); + const Run f1 = measure(b, DrawPath::Fast, bake, o.frames / 2, frameNo); + for (auto [dst, src] : {std::make_pair(&s, &s1), std::make_pair(&f, &f1)}) { + dst->frames += src->frames; + dst->frameMs += src->frameMs; + dst->drawMs += src->drawMs; + dst->draws += src->draws; + dst->frame += src->frame; + dst->draw += src->draw; + dst->stats.programLookups += src->stats.programLookups; + dst->stats.programLookupsSkipped += src->stats.programLookupsSkipped; + dst->stats.stockDraws += src->stats.stockDraws; + } + } + report(shadows ? (bake ? "shadows ON, bake every frame" : "shadows ON, steady (no bake)") + : "shadows OFF", + s, f); + // Same pose on both paths, then compare the frames. + LowMemoryPolyDataMapper::setDrawPath(DrawPath::Stock); + poseVehicles(b, 7); + if (bake) + b.sg.invalidateShadowBake(); + b.sr->render(); + const auto a = rgba(b); + LowMemoryPolyDataMapper::setDrawPath(DrawPath::Fast); + if (bake) + b.sg.invalidateShadowBake(); + b.sr->render(); + const auto c = rgba(b); + long d = 0; + for (size_t i = 0; i < a.size() && i < c.size(); ++i) + d += a[i] != c[i]; + std::printf(" pixels Stock vs Fast: %s (%ld differing bytes of %zu)\n", + d == 0 && a.size() == c.size() ? "IDENTICAL" : "DIFFER", d, a.size()); + } + } + return 0; +} diff --git a/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp b/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp new file mode 100644 index 000000000..dc1258371 --- /dev/null +++ b/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp @@ -0,0 +1,969 @@ +/* + Copyright 2026 The University of Texas at Austin + + This file is part of libcvc. + + libcvc is free software; you can redistribute it and/or + modify it under the terms of the GNU Lesser General Public + License version 2.1 as published by the Free Software Foundation. +*/ + +// cvc::gl::LowMemoryPolyDataMapper -- the WebGL2/GLES mapper with the empty +// cell types and the per-draw shader-cache lookup taken out of its draw. +// +// Runs that mapper natively (policy Force) on cvcGL's real pipeline -- a +// SceneGraph with shadows on (camera -> strided shadow baker -> shadow map -> +// translucent -> volumetric -> overlay) and off -- over GeometryNodes of every +// kind a scene uses (lit single colour, textured ground, translucent with a +// depth offset, lines, points, per-vertex colour) plus raw VTK actors whose +// mapper counts the GL calls of its own draws. Checks: +// 1. the policy: CVCGL_LOWMEM_MAPPER is read, Force/Off/Auto pick the mapper; +// 2. the replica: DrawPath::AllCellTypes issues exactly VTK's GL calls (every +// entry point, whole frame, bake and plain frames, shadows on and off); +// 3. pixels: Stock, AllCellTypes and Fast frames are byte-identical (RGBA); +// 4. the saving: Fast draws a single-cell-type mesh in <= 2 VAO binds and a +// fraction of the stock calls, with no synchronous GL query in any draw, +// and a steady frame does no shader-cache lookup; +// 5. lifecycle: a shader rebuild (the shadow bake's pass change), a mapper's +// released resources, programs released in the shader cache and the scene +// re-opened in a new window all go back through the full lookup (or a +// recompile) and still match; +// 6. guards: an offset layout where skipping would change a drawn type's +// depth offset, picking and vertex visibility all draw the stock way. +// Frames are rendered without MSAA: with VTK's default 8x, NVIDIA's resolve +// varies by 1 LSB at a few edge pixels between identical GL streams. +// Renders for real; skips (rc 0) where nothing rasterises unless +// CVC_REQUIRE_RENDER=1. Set CVCGL_LOWMEM_REPORT=1 for the per-draw breakdown. +// +// NOT assert(): built Release, where NDEBUG makes assert() a no-op. +#include "gl_call_counter.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +using cvc::gl::GeometryNode; +using cvc::gl::GeometryRenderMode; +using cvc::gl::LowMemoryMapperPolicy; +using cvc::gl::LowMemoryPolyDataMapper; +using cvc::gl::SceneGraph; +using cvc::gl::SceneRenderer; +using cvcgl_test::GLCalls; +using cvcgl_test::glcallsInstall; +using cvcgl_test::glcallsRead; +using DrawPath = LowMemoryPolyDataMapper::DrawPath; + +namespace { + +int g_failures = 0, g_checks = 0; +bool g_report = false; + +bool check(bool ok, const std::string &what, const std::string &detail = "") { + ++g_checks; + if (!ok) + ++g_failures; + std::printf(" [%s] %s%s%s\n", ok ? "PASS" : "FAIL", what.c_str(), detail.empty() ? "" : " -- ", + detail.c_str()); + std::fflush(stdout); + return ok; +} + +std::string fmt(const char *f, double a, double b = 0, double c = 0) { + char buf[256]; + std::snprintf(buf, sizeof buf, f, a, b, c); + return buf; +} + +const char *pathName(DrawPath p) { + return p == DrawPath::Stock ? "Stock" : p == DrawPath::AllCellTypes ? "AllCellTypes" : "Fast"; +} + +// Counts VTK errors (shader compile/link failures arrive here, not as exceptions). +class ErrorCounter : public vtkOutputWindow { +public: + static ErrorCounter *New(); + vtkTypeMacro(ErrorCounter, vtkOutputWindow); + static int &errors() { + static int n = 0; + return n; + } + void DisplayErrorText(const char *t) override { + ++errors(); + std::fprintf(stderr, "VTK ERROR: %s\n", t); + } + void DisplayWarningText(const char *t) override { std::fprintf(stderr, "VTK WARNING: %s\n", t); } + void DisplayGenericWarningText(const char *t) override { + std::fprintf(stderr, "VTK WARNING: %s\n", t); + } + void DisplayText(const char *t) override { std::fprintf(stderr, "%s", t); } +}; +vtkStandardNewMacro(ErrorCounter); + +// The production mapper plus a GL-call tally of its own RenderPieceDraw calls. +class ProbeMapper : public LowMemoryPolyDataMapper { +public: + static ProbeMapper *New(); + vtkTypeMacro(ProbeMapper, LowMemoryPolyDataMapper); + GLCalls calls; + double draws = 0; + void clearTally() { + calls = GLCalls(); + draws = 0; + } + void RenderPieceDraw(vtkRenderer *ren, vtkActor *act) override { + const GLCalls before = glcallsRead(); + Superclass::RenderPieceDraw(ren, act); + calls += glcallsRead() - before; + draws += 1; + } + +protected: + ProbeMapper() = default; +}; +vtkStandardNewMacro(ProbeMapper); + +// A GeometryNode whose actor the test can reach. +class PeekNode : public GeometryNode { +public: + using GeometryNode::GeometryNode; + vtkActor *actor() { return vtkActor::SafeDownCast(getProp()); } +}; + +// ── geometry ───────────────────────────────────────────────────────────────── +// A closed, lit box (separate faces so the normals are flat). +cvc::geometry boxGeometry(double cx, double cy, double cz, double sx, double sy, double sz) { + cvc::geometry g; + const double x0 = cx - sx, x1 = cx + sx, y0 = cy - sy, y1 = cy + sy, z0 = cz - sz, z1 = cz + sz; + const double faces[6][4][3] = {{{x0, y0, z1}, {x1, y0, z1}, {x1, y1, z1}, {x0, y1, z1}}, // top + {{x0, y0, z0}, {x0, y1, z0}, {x1, y1, z0}, {x1, y0, z0}}, // bottom + {{x0, y0, z0}, {x1, y0, z0}, {x1, y0, z1}, {x0, y0, z1}}, // -y + {{x1, y1, z0}, {x0, y1, z0}, {x0, y1, z1}, {x1, y1, z1}}, // +y + {{x0, y1, z0}, {x0, y0, z0}, {x0, y0, z1}, {x0, y1, z1}}, // -x + {{x1, y0, z0}, {x1, y1, z0}, {x1, y1, z1}, {x1, y0, z1}}}; // +x + for (const auto &f : faces) { + const unsigned base = static_cast(g.points().size()); + for (const auto &p : f) + g.points().push_back({p[0], p[1], p[2]}); + g.tris().push_back({base, base + 1, base + 2}); + g.tris().push_back({base, base + 2, base + 3}); + } + return g; +} + +// An n x n height-field grid over [-s, s]^2 with UVs. +cvc::geometry groundGeometry(int n, double s) { + cvc::geometry g; + for (int j = 0; j < n; ++j) + for (int i = 0; i < n; ++i) { + const double x = -s + 2 * s * i / (n - 1), y = -s + 2 * s * j / (n - 1); + g.points().push_back({x, y, 0.6 * std::sin(x * 0.35) * std::cos(y * 0.3)}); + g.uvs().push_back({double(i) / (n - 1), double(j) / (n - 1)}); + } + for (int j = 0; j + 1 < n; ++j) + for (int i = 0; i + 1 < n; ++i) { + const unsigned a = j * n + i, b = a + 1, c = a + n, d = c + 1; + g.tris().push_back({a, b, d}); + g.tris().push_back({a, d, c}); + } + return g; +} + +cvc::geometry ribbonGeometry(int n, double x0, double y0, double z) { + cvc::geometry g; + for (int k = 0; k < n; ++k) { + g.points().push_back({x0 + 0.8 * k, y0 - 0.6, z}); + g.points().push_back({x0 + 0.8 * k, y0 + 0.6, z}); + } + for (int k = 0; k + 1 < n; ++k) { + const unsigned l0 = 2 * k, r0 = l0 + 1, l1 = l0 + 2, r1 = l0 + 3; + g.tris().push_back({l0, r0, r1}); + g.tris().push_back({l0, r1, l1}); + } + return g; +} + +cvc::geometry polylineGeometry(int n, double x0, double y0, double z) { + cvc::geometry g; + for (int k = 0; k < n; ++k) { + g.points().push_back({x0 + 0.5 * k, y0 + std::sin(k * 0.4), z}); + if (k) + g.lines().push_back({static_cast(k - 1), static_cast(k)}); + } + return g; +} + +cvc::geometry cloudGeometry(int n, double x0, double y0, double z) { + cvc::geometry g; + for (int k = 0; k < n; ++k) + g.points().push_back({x0 + 0.4 * (k % 10), y0 + 0.4 * (k / 10), z + 0.1 * (k % 3)}); + return g; +} + +cvc::geometry colouredGeometry(double cx, double cy) { + cvc::geometry g = boxGeometry(cx, cy, 1.0, 1.0, 1.0, 1.0); + for (size_t i = 0; i < g.points().size(); ++i) + g.colors().push_back({(i % 3) / 2.0, ((i / 3) % 3) / 2.0, ((i / 9) % 3) / 2.0}); + return g; +} + +cvc::image checkerImage(int n) { + cvc::image img(n, n, cvc::image::pixel_format::RGBA, cvc::image::data_type::u8); + unsigned char *p = img.data(); + for (int j = 0; j < n; ++j) + for (int i = 0; i < n; ++i) { + unsigned char *q = p + 4 * (j * n + i); + const bool on = ((i / 8) + (j / 8)) % 2 == 0; + q[0] = on ? 200 : 60; + q[1] = static_cast(i * 255 / n); + q[2] = static_cast(j * 255 / n); + q[3] = 255; + } + return img; +} + +// Raw VTK poly data for the probe actors. +vtkSmartPointer withNormals(vtkPolyData *pd) { + vtkNew f; + f->SetInputData(pd); + f->SplittingOff(); + f->ComputePointNormalsOn(); + f->ComputeCellNormalsOff(); + f->Update(); + auto out = vtkSmartPointer::New(); + out->ShallowCopy(pd); + out->GetPointData()->SetNormals(f->GetOutput()->GetPointData()->GetNormals()); + return out; +} + +vtkSmartPointer spherePoly(double r) { + vtkNew s; + s->SetRadius(r); + s->SetThetaResolution(24); + s->SetPhiResolution(18); + s->Update(); + auto pd = vtkSmartPointer::New(); + pd->DeepCopy(s->GetOutput()); + return pd; +} + +vtkSmartPointer terrainPoly(int n, double x0, double y0, double s) { + auto pts = vtkSmartPointer::New(); + pts->SetDataTypeToFloat(); + auto tc = vtkSmartPointer::New(); + tc->SetNumberOfComponents(2); + for (int j = 0; j < n; ++j) + for (int i = 0; i < n; ++i) { + pts->InsertNextPoint(x0 + s * i / (n - 1), y0 + s * j / (n - 1), 0.05 * ((i + j) % 3)); + tc->InsertNextTuple2(double(i) / (n - 1), double(j) / (n - 1)); + } + auto polys = vtkSmartPointer::New(); + for (int j = 0; j + 1 < n; ++j) + for (int i = 0; i + 1 < n; ++i) { + const vtkIdType a = j * n + i, b = a + 1, c = a + n, d = c + 1; + const vtkIdType t1[3] = {a, b, d}, t2[3] = {a, d, c}; + polys->InsertNextCell(3, t1); + polys->InsertNextCell(3, t2); + } + auto pd = vtkSmartPointer::New(); + pd->SetPoints(pts); + pd->SetPolys(polys); + pd->GetPointData()->SetTCoords(tc); + return withNormals(pd); +} + +vtkSmartPointer ribbonPoly(int n, double x0, double y0, double z) { + auto pts = vtkSmartPointer::New(); + pts->SetDataTypeToFloat(); + auto polys = vtkSmartPointer::New(); + for (int k = 0; k < n; ++k) { + pts->InsertNextPoint(x0 + 0.7 * k, y0 - 0.5, z); + pts->InsertNextPoint(x0 + 0.7 * k, y0 + 0.5, z); + } + for (int k = 0; k + 1 < n; ++k) { + const vtkIdType l0 = 2 * k, r0 = l0 + 1, l1 = l0 + 2, r1 = l0 + 3; + const vtkIdType t1[3] = {l0, r0, r1}, t2[3] = {l0, r1, l1}; + polys->InsertNextCell(3, t1); + polys->InsertNextCell(3, t2); + } + auto pd = vtkSmartPointer::New(); + pd->SetPoints(pts); + pd->SetPolys(polys); + return withNormals(pd); +} + +vtkSmartPointer linesPoly(int n, double x0, double y0, double z) { + auto pts = vtkSmartPointer::New(); + pts->SetDataTypeToFloat(); + auto lines = vtkSmartPointer::New(); + for (int k = 0; k < n; ++k) { + pts->InsertNextPoint(x0 + 0.6 * k, y0 + 0.8 * std::cos(k * 0.5), z); + if (k) { + const vtkIdType ids[2] = {k - 1, k}; + lines->InsertNextCell(2, ids); + } + } + auto pd = vtkSmartPointer::New(); + pd->SetPoints(pts); + pd->SetLines(lines); + return pd; +} + +// One cell type of a 3 x 4 point patch: 'v' verts, 's' a triangle strip. (One +// type per data set: VTK 9.5.0's low-memory mapper crashes on its first upload +// of a data set with two or more cell types -- vtkDrawTexturedElements:: +// AppendArrayToTexture flags a not-yet-created buffer dirty; fixed upstream in +// 9.5.1 by 26ef893b5d9. cvcGL nodes always produce a single cell type.) +vtkSmartPointer patchPoly(char kind, double x0, double y0, double z) { + auto pts = vtkSmartPointer::New(); + pts->SetDataTypeToFloat(); + for (int k = 0; k < 12; ++k) + pts->InsertNextPoint(x0 + (k % 4), y0 + (k / 4), z + 0.2 * (k % 2)); + auto cells = vtkSmartPointer::New(); + auto pd = vtkSmartPointer::New(); + pd->SetPoints(pts); + if (kind == 'v') { + for (vtkIdType k = 0; k < 12; ++k) + cells->InsertNextCell(1, &k); + pd->SetVerts(cells); + } else { + const vtkIdType a[8] = {0, 4, 1, 5, 2, 6, 3, 7}, b[8] = {4, 8, 5, 9, 6, 10, 7, 11}; + cells->InsertNextCell(8, a); + cells->InsertNextCell(8, b); + pd->SetStrips(cells); + } + return pd; +} + +vtkSmartPointer checkerTexture(int n) { + auto img = vtkSmartPointer::New(); + img->SetDimensions(n, n, 1); + img->AllocateScalars(VTK_UNSIGNED_CHAR, 4); + auto *px = static_cast(img->GetScalarPointer()); + for (int j = 0; j < n; ++j) + for (int i = 0; i < n; ++i) { + unsigned char *q = px + 4 * (j * n + i); + q[0] = static_cast(i * 255 / n); + q[1] = ((i / 4 + j / 4) % 2) ? 220 : 40; + q[2] = 128; + q[3] = 255; + } + auto tex = vtkSmartPointer::New(); + tex->SetInputData(img); + tex->InterpolateOn(); + return tex; +} + +// ── the scene ──────────────────────────────────────────────────────────────── +struct Probe { + std::string name; + vtkSmartPointer mapper; + vtkSmartPointer actor; +}; + +struct Scene { + explicit Scene(cvc::app &app) : sg(app, "lowmem") { sg.setDiagnosticChromeVisible(false); } + SceneGraph sg; + std::unique_ptr sr; + std::vector> nodes; + std::deque probes; // stable references across probe() + int w = 320, h = 240; + + void open() { + sr = std::make_unique(sg, w, h, /*offscreen=*/true, "lowmem"); + // No MSAA: with VTK's default 8x, NVIDIA's resolve differs by 1 LSB at a few + // edge pixels from one frame to the next for an identical GL stream, so two + // stock frames would not compare equal either. + sr->renderWindow()->SetMultiSamples(0); + sr->setBackground(0.18, 0.22, 0.30); + sr->setCamera(-14, -20, 16, 0, 0, 0, 0, 0, 1, 40.0, 1.0, 200.0); + for (auto &p : probes) + sr->renderer()->AddActor(p.actor); + } + + std::shared_ptr node(const std::string &name) { + auto n = sg.getGraphicsRoot()->addGraphicsChild(name); + nodes.push_back(n); + return n; + } + + Probe &probe(const std::string &name, vtkSmartPointer pd) { + Probe p; + p.name = name; + p.mapper = vtkSmartPointer::New(); + p.mapper->SetInputData(pd); + p.mapper->ScalarVisibilityOff(); + p.actor = vtkSmartPointer::New(); + p.actor->SetMapper(p.mapper); + p.actor->GetProperty()->SetSpecular(0.3); + p.actor->GetProperty()->SetSpecularPower(20); + probes.push_back(p); + return probes.back(); + } + + // Every LowMemoryPolyDataMapper drawing in this scene. + std::vector mappers() { + std::vector out; + vtkActorCollection *actors = sr->renderer()->GetActors(); + actors->InitTraversal(); + while (vtkActor *a = actors->GetNextActor()) + if (auto *m = LowMemoryPolyDataMapper::SafeDownCast(a->GetMapper())) + out.push_back(m); + return out; + } + LowMemoryPolyDataMapper::Stats stats() { + LowMemoryPolyDataMapper::Stats s; + for (auto *m : mappers()) { + const auto &t = m->stats(); + s.draws += t.draws; + s.stockDraws += t.stockDraws; + s.cellTypesSkipped += t.cellTypesSkipped; + s.coincidentFallbacks += t.coincidentFallbacks; + s.programLookups += t.programLookups; + s.programLookupsSkipped += t.programLookupsSkipped; + } + return s; + } + void resetCounters() { + for (auto *m : mappers()) + m->resetStats(); + for (auto &p : probes) + p.mapper->clearTally(); + } + + // Render one frame; the GL calls it issued. + GLCalls frame() { + const GLCalls before = glcallsRead(); + sr->render(); + return glcallsRead() - before; + } + std::vector rgba() { + vtkNew px; + sr->renderWindow()->GetRGBACharPixelData(0, 0, w - 1, h - 1, /*front=*/0, px); + const unsigned char *p = px->GetPointer(0); + return std::vector(p, p + px->GetNumberOfValues()); + } +}; + +void buildCity(Scene &s) { + // Nodes: every kind of GeometryNode a scene uses. + auto veh = s.node("veh"); + veh->setGeometry(boxGeometry(0, 0, 1.2, 1.6, 0.9, 0.7)); + veh->setUseSingleColor(true); + veh->setColor(0.3, 0.5, 0.9); + const double pose[16] = {0.8, -0.6, 0, 2.5, 0.6, 0.8, 0, -1.0, 0, 0, 1, 0.1, 0, 0, 0, 1}; + veh->setPoseMatrix(pose); + + auto ground = s.node("ground"); + ground->setGeometry(groundGeometry(40, 14.0)); + ground->setUseSingleColor(true); + ground->setColor(1, 1, 1); + ground->setAmbient(0.4); + ground->setDiffuse(0.7); + ground->setTexture(checkerImage(64)); + + auto ribbon = s.node("ribbon"); + ribbon->setGeometry(ribbonGeometry(30, -9, 4, 0.9)); + ribbon->setUseSingleColor(true); + ribbon->setColor(0.98, 0.85, 0.3); + ribbon->setOpacity(0.6); + ribbon->setDepthOffset(2.0); + + auto trail = s.node("trail"); + trail->setGeometry(polylineGeometry(40, -10, -6, 0.8)); + trail->setRenderMode(GeometryRenderMode::LINES); // after setGeometry, which picks a mode + trail->setUseSingleColor(true); + trail->setColor(0.2, 0.95, 0.3); + trail->setLineWidth(2.0); + + auto cloud = s.node("cloud"); + cloud->setGeometry(cloudGeometry(60, 5, -9, 1.5)); + cloud->setRenderMode(GeometryRenderMode::POINTS); + cloud->setUseSingleColor(true); + cloud->setColor(0.95, 0.3, 0.9); + cloud->setPointSize(4.0); + + auto vcol = s.node("vcol"); + vcol->setUseSingleColor(false); + vcol->setGeometry(colouredGeometry(-4, -3)); + + // Probes: the same kinds as raw actors whose draws are tallied. + { + Probe &p = s.probe("lit", spherePoly(1.4)); + p.actor->GetProperty()->SetColor(0.85, 0.35, 0.25); + vtkNew m; + m->SetElement(0, 3, 6.0); + m->SetElement(1, 3, 3.0); + m->SetElement(2, 3, 1.6); + p.actor->SetUserMatrix(m); + } + { + Probe &p = s.probe("textured", terrainPoly(24, -13, 6, 7)); + p.actor->GetProperty()->SetColor(1, 1, 1); + p.actor->SetTexture(checkerTexture(32)); + } + { + Probe &p = s.probe("translucent", ribbonPoly(24, -6, 8.5, 0.7)); + p.actor->GetProperty()->SetColor(0.3, 0.9, 0.95); + p.actor->GetProperty()->SetOpacity(0.5); + p.mapper->SetRelativeCoincidentTopologyPolygonOffsetParameters(0.0, -2.0); + p.mapper->SetRelativeCoincidentTopologyLineOffsetParameters(0.0, -2.0); + p.mapper->SetRelativeCoincidentTopologyPointOffsetParameter(-2.0); + } + { + Probe &p = s.probe("lines", linesPoly(30, -12, -10, 0.5)); + p.actor->GetProperty()->SetColor(0.95, 0.95, 0.2); + } + { + Probe &p = s.probe("verts", patchPoly('v', 8, -12, 0.8)); + p.actor->GetProperty()->SetColor(0.7, 0.7, 0.9); + p.actor->GetProperty()->SetPointSize(5.0); + } + { + Probe &p = s.probe("strips", patchPoly('s', 2, 9, 0.6)); + p.actor->GetProperty()->SetColor(0.5, 0.9, 0.6); + } +} + +void addLights(SceneGraph &sg) { + sg.addSpotLight(-12, -16, 26, 0, 0, 0, 40.0, 1.0, 0.97, 0.9, 1.0); + sg.addFillLight(14, 10, 20, 0, 0, 0, 0.8, 0.85, 1.0, 0.4); +} + +struct PathRun { + GLCalls first, steady; // whole frame + std::vector firstPx, steadyPx; // RGBA + LowMemoryPolyDataMapper::Stats firstStats, steadyStats; + std::vector probeCalls; // steady frame, per probe + std::vector probeDraws; // steady frame, per probe + std::vector probeFirst; // first frame, per probe + std::vector probeFirstDraws; +}; + +// Two frames on one draw path: `first` (a real shadow bake when shadows are on: +// every opaque mapper draws twice, rebuilding its shader for each pass) and +// `steady` (re-uses the bake; nothing rebuilt). +PathRun runPath(Scene &s, DrawPath path) { + LowMemoryPolyDataMapper::setDrawPath(path); + // One frame on this path first, so the counted frames start from the GL + // state this path leaves (on desktop the empty verts agent's glPointSize is + // cached state, so the first frame after a switch would differ by one call). + s.sr->render(); + PathRun r; + // vtkShadowMapBakerPass only re-bakes when a light or prop changed since the + // last bake; touch one so the first frame really bakes. + s.sg.invalidateShadowBake(); + if (vtkActor *a = s.sr->renderer()->GetActors()->GetLastActor()) + a->Modified(); + s.resetCounters(); + r.first = s.frame(); + r.firstStats = s.stats(); + for (auto &p : s.probes) { + r.probeFirst.push_back(p.mapper->calls); + r.probeFirstDraws.push_back(p.mapper->draws); + } + r.firstPx = s.rgba(); + s.resetCounters(); + r.steady = s.frame(); + r.steadyStats = s.stats(); + for (auto &p : s.probes) { + r.probeCalls.push_back(p.mapper->calls); + r.probeDraws.push_back(p.mapper->draws); + } + r.steadyPx = s.rgba(); + return r; +} + +long diffBytes(const std::vector &a, const std::vector &b) { + if (a.size() != b.size()) + return -1; + long n = 0; + for (size_t i = 0; i < a.size(); ++i) + n += a[i] != b[i]; + return n; +} + +bool samePixels(const std::string &what, const std::vector &a, + const std::vector &b) { + const long d = diffBytes(a, b); + return check(d == 0 && !a.empty(), what, d == 0 ? "" : fmt("%.0f differing bytes", double(d))); +} + +// Every pixel the background colour? (a frame that drew nothing) +bool drewSomething(const std::vector &px) { + if (px.size() < 8) + return false; + for (size_t i = 4; i < px.size(); i += 4) + if (px[i] != px[0] || px[i + 1] != px[1] || px[i + 2] != px[2]) + return true; + return false; +} + +// ── 1. policy ──────────────────────────────────────────────────────────────── +void testPolicy() { + std::printf("policy\n"); + check(cvc::gl::lowMemoryMapperPolicy() == LowMemoryMapperPolicy::Force, + "CVCGL_LOWMEM_MAPPER=force is read on first use"); + check(LowMemoryPolyDataMapper::SafeDownCast(cvc::gl::newPolyDataMapper()) != nullptr, + "Force: newPolyDataMapper() is a LowMemoryPolyDataMapper"); + const bool factoryLowMem = + vtkSmartPointer::New()->IsA("vtkOpenGLLowMemoryPolyDataMapper") != 0; + cvc::gl::setLowMemoryMapperPolicy(LowMemoryMapperPolicy::Off); + check(LowMemoryPolyDataMapper::SafeDownCast(cvc::gl::newPolyDataMapper()) == nullptr, + "Off: newPolyDataMapper() is the factory's mapper"); + cvc::gl::setLowMemoryMapperPolicy(LowMemoryMapperPolicy::Auto); + check((LowMemoryPolyDataMapper::SafeDownCast(cvc::gl::newPolyDataMapper()) != nullptr) == + factoryLowMem, + "Auto: LowMemoryPolyDataMapper exactly where the factory hands out the low-memory one", + factoryLowMem ? "factory: low-memory" : "factory: classic"); + cvc::gl::setLowMemoryMapperPolicy(LowMemoryMapperPolicy::Force); + if (!std::getenv("CVCGL_LOWMEM_DRAW")) + check(LowMemoryPolyDataMapper::drawPath() == DrawPath::Fast, "the default draw path is Fast"); +} + +// Can this build rasterise at all? Asked of a plain node in a scene of its own. +bool renderAvailable(cvc::app &app) { + bool drew = false; + { + Scene s(app); + auto n = s.node("control"); + n->setGeometry(boxGeometry(0, 0, 0, 5, 5, 0.5)); + n->setColor(1, 1, 1); + s.w = 48; + s.h = 48; + s.open(); + s.sr->setCamera(0, 0, 50, 0, 0, 0, 0, 1, 0, 30.0, 1.0, 200.0); + s.sr->render(); + drew = drewSomething(s.rgba()); + } + if (drew) + return true; + const char *require = std::getenv("CVC_REQUIRE_RENDER"); + if (require && *require && std::string(require) != "0") + check(false, "CVC_REQUIRE_RENDER is set, but this build did not rasterise"); + else + std::printf(" skipped: this build did not rasterise\n"); + return false; +} + +// ── 2-4. replica, pixels, savings ──────────────────────────────────────────── +void comparePaths(Scene &s, const std::string &label, bool shadows) { + std::printf("%s\n", label.c_str()); + // Settle: build every shader both ways first so no path pays a first compile. + for (DrawPath p : {DrawPath::Stock, DrawPath::Fast}) + runPath(s, p); + const PathRun stock = runPath(s, DrawPath::Stock); + const PathRun all = runPath(s, DrawPath::AllCellTypes); + const PathRun fast = runPath(s, DrawPath::Fast); + + check(drewSomething(stock.steadyPx), label + ": the scene draws"); + std::string diff; + const bool pinFirst = all.first.sameAs(stock.first, &diff); + check(pinFirst, label + ": AllCellTypes issues VTK's GL calls, first frame", diff); + diff.clear(); + const bool pinSteady = all.steady.sameAs(stock.steady, &diff); + check(pinSteady, label + ": AllCellTypes issues VTK's GL calls, steady frame", diff); + samePixels(label + ": AllCellTypes pixels == Stock, first frame", stock.firstPx, all.firstPx); + samePixels(label + ": AllCellTypes pixels == Stock, steady frame", stock.steadyPx, all.steadyPx); + samePixels(label + ": Fast pixels == Stock, first frame", stock.firstPx, fast.firstPx); + samePixels(label + ": Fast pixels == Stock, steady frame", stock.steadyPx, fast.steadyPx); + + std::printf(" frame GL calls, first : Stock %s\n", stock.first.str().c_str()); + std::printf(" Fast %s\n", fast.first.str().c_str()); + std::printf(" frame GL calls, steady : Stock %s\n", stock.steady.str().c_str()); + std::printf(" Fast %s\n", fast.steady.str().c_str()); + check(fast.steady.total() < 0.75 * stock.steady.total(), + label + ": Fast frame issues well under Stock's GL calls", + fmt("steady %.0f -> %.0f, first %.0f", stock.steady.total(), fast.steady.total(), + stock.first.total()) + + fmt(" -> %.0f", fast.first.total())); + check(fast.steady.sync() == stock.steady.sync(), + label + ": Fast adds no synchronous GL query to the frame", + fmt("sync %.0f vs %.0f", fast.steady.sync(), stock.steady.sync())); + check(fast.steadyStats.programLookups == 0 && fast.steadyStats.programLookupsSkipped > 0, + label + ": Fast steady frame re-binds every program without a lookup", + fmt("lookups %.0f, skipped %.0f", double(fast.steadyStats.programLookups), + double(fast.steadyStats.programLookupsSkipped))); + check(fast.steadyStats.cellTypesSkipped > 0 && fast.steadyStats.coincidentFallbacks == 0 && + fast.steadyStats.stockDraws == 0, + label + ": Fast skipped empty cell types, no fallback", + fmt("skipped %.0f of %.0f draws x 4", double(fast.steadyStats.cellTypesSkipped), + double(fast.steadyStats.draws))); + check(stock.steadyStats.stockDraws == stock.steadyStats.draws && stock.steadyStats.draws > 0, + label + ": Stock draws through VTK's own path"); + if (shadows) + check(fast.firstStats.programLookups > 0, + label + ": a shadow bake (render-pass change) takes the full lookup", + fmt("lookups %.0f", double(fast.firstStats.programLookups))); + + // Per draw, on the probes (steady frame: one draw per probe and pass). + for (size_t i = 0; i < s.probes.size(); ++i) { + const Probe &p = s.probes[i]; + const double nStock = stock.probeDraws[i], nFast = fast.probeDraws[i]; + if (nStock <= 0 || nFast <= 0) { + check(false, label + ": probe " + p.name + " drew"); + continue; + } + const GLCalls &cs = stock.probeCalls[i]; + const GLCalls &cf = fast.probeCalls[i]; + std::printf(" per draw %-11s Stock %s\n", p.name.c_str(), cs.str(nStock).c_str()); + std::printf(" %-20s Fast %s\n", "", cf.str(nFast).c_str()); + check(cs.sync() == 0 && cf.sync() == 0 && all.probeCalls[i].sync() == 0, + label + ": " + p.name + " draws issue no synchronous GL query"); + check(cf.count("UseProgram") <= 2 * nFast, label + ": " + p.name + " <= 2 glUseProgram/draw"); + check(cf.count("BindVertexArray") == 2 * nFast, + label + ": " + p.name + " binds the VAO once per draw (+ unbind)", + fmt("%.1f per draw", cf.count("BindVertexArray") / nFast)); + check(cf.count("DrawArraysInstanced") == cs.count("DrawArraysInstanced") && + cf.count("DrawArraysInstanced") == nFast, + label + ": " + p.name + " one draw call per draw, as Stock"); + const double perFast = cf.total() / nFast, perStock = cs.total() / nStock; + check(perFast <= 0.45 * perStock, label + ": " + p.name + " Fast <= 45% of Stock calls/draw", + fmt("%.1f -> %.1f", perStock, perFast)); + } + if (g_report) { + LowMemoryPolyDataMapper::setDrawPath(DrawPath::Fast); + s.resetCounters(); + s.sr->render(); + for (auto *m : s.mappers()) { + const auto &t = m->stats(); + vtkPolyData *in = m->GetInput(); + std::printf( + " mapper %-24s points %5lld v/l/p/s %lld/%lld/%lld/%lld: draws %llu skipped %llu " + "lookups %llu/%llu\n", + m->GetClassName(), static_cast(in->GetNumberOfPoints()), + static_cast(in->GetNumberOfVerts()), + static_cast(in->GetNumberOfLines()), + static_cast(in->GetNumberOfPolys()), + static_cast(in->GetNumberOfStrips()), static_cast(t.draws), + static_cast(t.cellTypesSkipped), + static_cast(t.programLookups), + static_cast(t.programLookupsSkipped)); + } + for (size_t i = 0; i < s.probes.size(); ++i) + std::printf(" first frame per draw %-11s Stock %s\n %-32s Fast %s\n", + s.probes[i].name.c_str(), + stock.probeFirst[i].str(std::max(1.0, stock.probeFirstDraws[i])).c_str(), "", + fast.probeFirst[i].str(std::max(1.0, fast.probeFirstDraws[i])).c_str()); + } +} + +// ── 5. lifecycle ───────────────────────────────────────────────────────────── +void testLifecycle(Scene &s) { + std::printf("lifecycle\n"); + LowMemoryPolyDataMapper::setDrawPath(DrawPath::Fast); + s.sr->render(); + + // A mapper's graphics resources released (what removing a prop from its + // renderer does): its next draw looks the program up afresh. + for (auto *m : s.mappers()) + m->ReleaseGraphicsResources(s.sr->renderWindow()); + s.resetCounters(); + s.sr->render(); + const auto rel = s.stats(); + check(rel.programLookups == rel.draws && rel.programLookupsSkipped == 0 && rel.draws > 0, + "after a mapper's ReleaseGraphicsResources its next draw looks the program up", + fmt("lookups %.0f of %.0f draws", double(rel.programLookups), double(rel.draws))); + { + const PathRun a = runPath(s, DrawPath::Stock); + const PathRun b = runPath(s, DrawPath::Fast); + samePixels("after mapper release: Fast pixels == Stock", a.steadyPx, b.steadyPx); + } + + // The shader cache's programs released and recompiled in place (the cache + // keeps the objects): the re-bound program is compiled again before use. + auto *win = vtkOpenGLRenderWindow::SafeDownCast(s.sr->renderWindow()); + const PathRun before = runPath(s, DrawPath::Stock); + win->GetShaderCache()->ReleaseGraphicsResources(win); + LowMemoryPolyDataMapper::setDrawPath(DrawPath::Fast); + s.sr->render(); + const PathRun after = runPath(s, DrawPath::Fast); + samePixels("programs released in the cache: Fast recompiles and matches Stock", before.steadyPx, + after.steadyPx); + + // The same scene -- the same nodes and mappers -- closed and re-opened in a + // new window (new context, new shader cache): the first draws resolve fully + // and the frames match the old window's. + s.sr.reset(); + s.open(); + s.sg.setShadowsEnabled(true); // the pass chain belongs to the renderer, not the scene + s.sg.setShadowUpdateInterval(1000); + s.resetCounters(); + s.sr->render(); + glcallsInstall(); // the new context re-loaded glad + const auto st = s.stats(); + check(st.programLookups > 0 && st.programLookupsSkipped == 0, + "re-opened in a new window: the first draws look their programs up", + fmt("lookups %.0f, skipped %.0f", double(st.programLookups), + double(st.programLookupsSkipped))); + const PathRun a = runPath(s, DrawPath::Stock); + const PathRun b = runPath(s, DrawPath::Fast); + samePixels("new window: Fast pixels == Stock", a.steadyPx, b.steadyPx); + samePixels("new window: same frame as the old window", before.steadyPx, b.steadyPx); +} + +// ── 6. guards ──────────────────────────────────────────────────────────────── +void testGuards(cvc::app &app) { + std::printf("guards\n"); + Scene s(app); + // Coincident offsets under the global POLYGON_OFFSET mode, where VTK's + // defaults give points -8 and lines -4 units but polygons 0: + // zeroPoly polys only, polygon offset 0 -> stock's empty verts/lines agents + // leave -4 in the program for the polys draw; skipping them would + // not. Must fall back to all four. + // shifted polys only, the same relative offset on every class -> the polys + // draw sets its own non-zero value. Safe to skip. + Probe &zero = s.probe("zeroPoly", spherePoly(1.5)); + zero.actor->GetProperty()->SetColor(0.9, 0.4, 0.3); + Probe &shifted = s.probe("shifted", ribbonPoly(12, -4, 3, 0.2)); + shifted.mapper->SetRelativeCoincidentTopologyPolygonOffsetParameters(0.0, -3.0); + shifted.mapper->SetRelativeCoincidentTopologyLineOffsetParameters(0.0, -3.0); + shifted.mapper->SetRelativeCoincidentTopologyPointOffsetParameter(-3.0); + Probe &vv = s.probe("vertexVis", terrainPoly(6, -5, -5, 4)); + vv.actor->GetProperty()->VertexVisibilityOn(); + vv.actor->GetProperty()->SetVertexColor(1, 0, 0); + s.open(); + s.sr->setCamera(0, -14, 10, 0, 0, 0, 0, 0, 1, 40.0, 1.0, 100.0); + s.sr->render(); + glcallsInstall(); + + const int savedMode = vtkMapper::GetResolveCoincidentTopology(); + vtkMapper::SetResolveCoincidentTopologyToPolygonOffset(); + for (DrawPath p : {DrawPath::Stock, DrawPath::Fast}) + runPath(s, p); + const PathRun stock = runPath(s, DrawPath::Stock); + const PathRun fast = runPath(s, DrawPath::Fast); + samePixels("POLYGON_OFFSET mode: Fast pixels == Stock", stock.steadyPx, fast.steadyPx); + const auto &zs = zero.mapper->stats(); + const auto &ss = shifted.mapper->stats(); + check(zs.coincidentFallbacks == zs.draws && zs.cellTypesSkipped == 0 && zs.draws > 0, + "zero polygon offset under defaults that offset points/lines: all four agents run", + fmt("fallbacks %.0f of %.0f draws", double(zs.coincidentFallbacks), double(zs.draws))); + check(ss.coincidentFallbacks == 0 && ss.cellTypesSkipped == 3 * ss.draws && ss.draws > 0, + "an offset the drawn type sets itself: empty types skipped", + fmt("skipped %.0f over %.0f draws", double(ss.cellTypesSkipped), double(ss.draws))); + vtkMapper::SetResolveCoincidentTopology(savedMode); + + const auto &vs = vv.mapper->stats(); + check(vs.stockDraws == vs.draws && vs.draws > 0, "vertex visibility draws the stock way"); + + // Picking: the selector's passes draw the stock way, and select the same. + auto select = [&](DrawPath p) { + LowMemoryPolyDataMapper::setDrawPath(p); + s.resetCounters(); + vtkNew sel; + sel->SetRenderer(s.sr->renderer()); + sel->SetArea(0, 0, s.w - 1, s.h - 1); + sel->SetFieldAssociation(vtkDataObject::FIELD_ASSOCIATION_CELLS); + vtkSmartPointer res = vtk::TakeSmartPointer(sel->Select()); + std::string sig; + for (unsigned i = 0; res && i < res->GetNumberOfNodes(); ++i) { + vtkSelectionNode *n = res->GetNode(i); + sig += fmt("[%.0f:%.0f]", n->GetProperties()->Get(vtkSelectionNode::PROP_ID()), + n->GetSelectionList() ? double(n->GetSelectionList()->GetNumberOfTuples()) : 0.0); + } + return std::make_pair(sig, s.stats()); + }; + const auto pickStock = select(DrawPath::Stock); + const auto pickFast = select(DrawPath::Fast); + check( + pickFast.second.stockDraws == pickFast.second.draws && pickFast.second.draws > 0, + "picking draws the stock way on the Fast path", + fmt("%.0f of %.0f draws", double(pickFast.second.stockDraws), double(pickFast.second.draws))); + check(pickFast.first == pickStock.first, "picking selects the same cells on both paths", + pickStock.first); + LowMemoryPolyDataMapper::setDrawPath(DrawPath::Fast); +} + +} // namespace + +int main() { +#ifdef _WIN32 + _putenv_s("CVCGL_LOWMEM_MAPPER", "force"); + if (!std::getenv("__GL_SYNC_TO_VBLANK")) + _putenv_s("__GL_SYNC_TO_VBLANK", "0"); + if (!std::getenv("vblank_mode")) + _putenv_s("vblank_mode", "0"); +#else + setenv("CVCGL_LOWMEM_MAPPER", "force", 1); + setenv("__GL_SYNC_TO_VBLANK", "0", 0); // NVIDIA throttles offscreen swaps otherwise + setenv("vblank_mode", "0", 0); +#endif + const char *rep = std::getenv("CVCGL_LOWMEM_REPORT"); + g_report = rep && *rep && std::string(rep) != "0"; + vtkOutputWindow::SetInstance(vtkSmartPointer::New()); + + testPolicy(); + if (!LowMemoryPolyDataMapper::replicaActive()) { + std::printf(" skipped: this VTK is not 9.5.0, so the mapper draws the stock way\n"); + std::printf("PASS: cvcgl_lowmem_fastdraw (%d checks)\n", g_checks); + return g_failures == 0 ? 0 : 1; + } + + cvc::app app; + if (renderAvailable(app)) { + // Shadows on and off in separate scenes: SceneGraph::setShadowsEnabled(false) + // drops the shadow passes without releasing their GL resources, which VTK + // reports as an error when they are destroyed. + for (bool shadows : {true, false}) { + Scene s(app); + buildCity(s); + addLights(s.sg); + s.open(); + if (shadows) { + check(s.sg.setShadowsEnabled(true), "shadows enabled"); + s.sg.setShadowUpdateInterval(1000); // bake only when invalidated + } + s.sr->render(); + glcallsInstall(); + check(s.mappers().size() == s.nodes.size() + s.probes.size(), + "every node and probe draws through LowMemoryPolyDataMapper", + fmt("%.0f of %.0f", double(s.mappers().size()), + double(s.nodes.size() + s.probes.size()))); + comparePaths(s, shadows ? "shadows on" : "shadows off", shadows); + if (shadows) + testLifecycle(s); + } + testGuards(app); + } + check(ErrorCounter::errors() == 0, "no VTK errors", + fmt("%.0f errors", double(ErrorCounter::errors()))); + std::printf("%s: cvcgl_lowmem_fastdraw (%d checks, %d failed)\n", + g_failures == 0 ? "PASS" : "FAIL", g_checks, g_failures); + return g_failures == 0 ? 0 : 1; +} diff --git a/src/cvcGL/test/gl_call_counter.cpp b/src/cvcGL/test/gl_call_counter.cpp new file mode 100644 index 000000000..bc5ecef42 --- /dev/null +++ b/src/cvcGL/test/gl_call_counter.cpp @@ -0,0 +1,383 @@ +/* + Copyright 2026 The University of Texas at Austin + + This file is part of libcvc. + + libcvc is free software; you can redistribute it and/or + modify it under the terms of the GNU Lesser General Public + License version 2.1 as published by the Free Software Foundation. +*/ + +#include "gl_call_counter.h" + +#include +#include +#include +#include +#include +#include + +namespace cvcgl_test { + +namespace { + +// Counted and forwarded unchanged. +#define CVC_GL_PLAIN(X) \ + X(ActiveTexture) \ + X(BindTexture) \ + X(TexParameteri) \ + X(TexParameterf) \ + X(TexImage2D) \ + X(TexImage3D) \ + X(TexSubImage2D) \ + X(TexSubImage3D) \ + X(TexBuffer) \ + X(GenerateMipmap) \ + X(GenTextures) \ + X(DeleteTextures) \ + X(BindBuffer) \ + X(BufferData) \ + X(BufferSubData) \ + X(CopyBufferSubData) \ + X(GenBuffers) \ + X(DeleteBuffers) \ + X(BindVertexArray) \ + X(GenVertexArrays) \ + X(DeleteVertexArrays) \ + X(EnableVertexAttribArray) \ + X(DisableVertexAttribArray) \ + X(VertexAttribPointer) \ + X(DrawArrays) \ + X(DrawArraysInstanced) \ + X(DrawElements) \ + X(DrawRangeElements) \ + X(DrawElementsInstanced) \ + X(Enable) \ + X(Disable) \ + X(DepthMask) \ + X(DepthFunc) \ + X(ColorMask) \ + X(BlendFunc) \ + X(BlendFuncSeparate) \ + X(BlendEquation) \ + X(BlendEquationSeparate) \ + X(Viewport) \ + X(Scissor) \ + X(Clear) \ + X(ClearColor) \ + X(ClearDepth) \ + X(ClearDepthf) \ + X(StencilFunc) \ + X(StencilMask) \ + X(StencilOp) \ + X(CullFace) \ + X(PolygonOffset) \ + X(LineWidth) \ + X(PointSize) \ + X(PixelStorei) \ + X(BindFramebuffer) \ + X(BindRenderbuffer) \ + X(FramebufferTexture2D) \ + X(FramebufferRenderbuffer) \ + X(BlitFramebuffer) \ + X(ReadBuffer) \ + X(DrawBuffer) \ + X(DrawBuffers) \ + X(CreateProgram) \ + X(CreateShader) \ + X(ShaderSource) \ + X(CompileShader) \ + X(AttachShader) \ + X(Flush) \ + X(Finish) + +// Round trips on WebGL (the caller waits for the GPU process). +#define CVC_GL_SYNC(X) \ + X(GetIntegerv) \ + X(GetFloatv) \ + X(GetBooleanv) \ + X(GetDoublev) \ + X(IsEnabled) \ + X(GetError) \ + X(GetUniformLocation) \ + X(GetAttribLocation) \ + X(CheckFramebufferStatus) \ + X(GetProgramiv) \ + X(GetShaderiv) \ + X(GetString) \ + X(GetStringi) \ + X(GetTexLevelParameteriv) \ + X(GetTexImage) \ + X(ReadPixels) \ + X(IsProgram) + +// Counted with their own wrappers below. +#define CVC_GL_OWN(X) \ + X(UseProgram) \ + X(LinkProgram) \ + X(DeleteProgram) \ + X(Uniform1i) \ + X(Uniform2i) \ + X(Uniform1f) \ + X(Uniform2f) \ + X(Uniform3f) \ + X(Uniform4f) \ + X(Uniform1iv) \ + X(Uniform1fv) \ + X(Uniform2fv) \ + X(Uniform3fv) \ + X(Uniform4fv) \ + X(UniformMatrix3fv) \ + X(UniformMatrix4fv) + +enum Entry : int { +#define CVC_E(name) E_##name, + CVC_GL_PLAIN(CVC_E) CVC_GL_SYNC(CVC_E) CVC_GL_OWN(CVC_E) +#undef CVC_E + E_COUNT +}; + +const char *const g_names[E_COUNT] = { +#define CVC_N(name) #name, + CVC_GL_PLAIN(CVC_N) CVC_GL_SYNC(CVC_N) CVC_GL_OWN(CVC_N) +#undef CVC_N +}; + +double g_n[E_COUNT]; +double g_uniformRedundant = 0; + +bool isSync(int i) { return i >= E_GetIntegerv && i <= E_IsProgram; } +bool isUniform(int i) { return i >= E_Uniform1i && i <= E_UniformMatrix4fv; } + +template struct Counted; +template struct Counted { + static inline R(GLAD_API_PTR *orig)(A...) = nullptr; + static R GLAD_API_PTR fn(A... a) { + g_n[Id] += 1; + return orig(a...); + } + static void hook(R(GLAD_API_PTR *&ptr)(A...)) { + if (ptr && ptr != &fn) { + orig = ptr; + ptr = &fn; + } + } +}; + +// Uniform values are per program: (program, location) -> last bytes sent. +GLuint g_program = 0; +std::map, std::string> g_last; + +void noteUniform(GLint loc, const void *p, size_t bytes) { + std::string v(static_cast(p), bytes); + auto key = std::make_pair(g_program, loc); + auto it = g_last.find(key); + if (it != g_last.end() && it->second == v) + g_uniformRedundant += 1; + else + g_last[key] = std::move(v); +} +void forgetProgram(GLuint p) { + for (auto it = g_last.begin(); it != g_last.end();) + it = it->first.first == p ? g_last.erase(it) : std::next(it); +} + +#define CVC_O(name) decltype(glad_gl##name) o_##name = nullptr; +CVC_GL_OWN(CVC_O) +#undef CVC_O + +void GLAD_API_PTR w_UseProgram(GLuint p) { + g_n[E_UseProgram] += 1; + g_program = p; + o_UseProgram(p); +} +void GLAD_API_PTR w_LinkProgram(GLuint p) { + g_n[E_LinkProgram] += 1; + forgetProgram(p); + o_LinkProgram(p); +} +void GLAD_API_PTR w_DeleteProgram(GLuint p) { + g_n[E_DeleteProgram] += 1; + forgetProgram(p); + o_DeleteProgram(p); +} +void GLAD_API_PTR w_Uniform1i(GLint l, GLint a) { + g_n[E_Uniform1i] += 1; + noteUniform(l, &a, sizeof a); + o_Uniform1i(l, a); +} +void GLAD_API_PTR w_Uniform2i(GLint l, GLint a, GLint b) { + g_n[E_Uniform2i] += 1; + const GLint v[2] = {a, b}; + noteUniform(l, v, sizeof v); + o_Uniform2i(l, a, b); +} +void GLAD_API_PTR w_Uniform1f(GLint l, GLfloat a) { + g_n[E_Uniform1f] += 1; + noteUniform(l, &a, sizeof a); + o_Uniform1f(l, a); +} +void GLAD_API_PTR w_Uniform2f(GLint l, GLfloat a, GLfloat b) { + g_n[E_Uniform2f] += 1; + const GLfloat v[2] = {a, b}; + noteUniform(l, v, sizeof v); + o_Uniform2f(l, a, b); +} +void GLAD_API_PTR w_Uniform3f(GLint l, GLfloat a, GLfloat b, GLfloat c) { + g_n[E_Uniform3f] += 1; + const GLfloat v[3] = {a, b, c}; + noteUniform(l, v, sizeof v); + o_Uniform3f(l, a, b, c); +} +void GLAD_API_PTR w_Uniform4f(GLint l, GLfloat a, GLfloat b, GLfloat c, GLfloat d) { + g_n[E_Uniform4f] += 1; + const GLfloat v[4] = {a, b, c, d}; + noteUniform(l, v, sizeof v); + o_Uniform4f(l, a, b, c, d); +} +void GLAD_API_PTR w_Uniform1iv(GLint l, GLsizei c, const GLint *v) { + g_n[E_Uniform1iv] += 1; + noteUniform(l, v, sizeof(GLint) * c); + o_Uniform1iv(l, c, v); +} +void GLAD_API_PTR w_Uniform1fv(GLint l, GLsizei c, const GLfloat *v) { + g_n[E_Uniform1fv] += 1; + noteUniform(l, v, sizeof(GLfloat) * c); + o_Uniform1fv(l, c, v); +} +void GLAD_API_PTR w_Uniform2fv(GLint l, GLsizei c, const GLfloat *v) { + g_n[E_Uniform2fv] += 1; + noteUniform(l, v, 2 * sizeof(GLfloat) * c); + o_Uniform2fv(l, c, v); +} +void GLAD_API_PTR w_Uniform3fv(GLint l, GLsizei c, const GLfloat *v) { + g_n[E_Uniform3fv] += 1; + noteUniform(l, v, 3 * sizeof(GLfloat) * c); + o_Uniform3fv(l, c, v); +} +void GLAD_API_PTR w_Uniform4fv(GLint l, GLsizei c, const GLfloat *v) { + g_n[E_Uniform4fv] += 1; + noteUniform(l, v, 4 * sizeof(GLfloat) * c); + o_Uniform4fv(l, c, v); +} +void GLAD_API_PTR w_UniformMatrix3fv(GLint l, GLsizei c, GLboolean t, const GLfloat *v) { + g_n[E_UniformMatrix3fv] += 1; + noteUniform(l, v, 9 * sizeof(GLfloat) * c); + o_UniformMatrix3fv(l, c, t, v); +} +void GLAD_API_PTR w_UniformMatrix4fv(GLint l, GLsizei c, GLboolean t, const GLfloat *v) { + g_n[E_UniformMatrix4fv] += 1; + noteUniform(l, v, 16 * sizeof(GLfloat) * c); + o_UniformMatrix4fv(l, c, t, v); +} + +} // namespace + +void glcallsInstall() { +#define CVC_H(name) Counted::hook(glad_gl##name); + CVC_GL_PLAIN(CVC_H) + CVC_GL_SYNC(CVC_H) +#undef CVC_H +#define CVC_W(name) \ + if (glad_gl##name && glad_gl##name != w_##name) { \ + o_##name = glad_gl##name; \ + glad_gl##name = w_##name; \ + } + CVC_GL_OWN(CVC_W) +#undef CVC_W +} + +GLCalls glcallsRead() { + GLCalls r; + r.n.assign(g_n, g_n + E_COUNT); + r.uniformRedundant = g_uniformRedundant; + return r; +} + +int glcallsEntries() { return E_COUNT; } +const char *glcallsName(int i) { return i >= 0 && i < E_COUNT ? g_names[i] : "?"; } + +GLCalls GLCalls::operator-(const GLCalls &o) const { + GLCalls r; + r.n.resize(n.size()); + for (size_t i = 0; i < n.size(); ++i) + r.n[i] = n[i] - (i < o.n.size() ? o.n[i] : 0.0); + r.uniformRedundant = uniformRedundant - o.uniformRedundant; + return r; +} + +GLCalls &GLCalls::operator+=(const GLCalls &o) { + if (n.size() < o.n.size()) + n.resize(o.n.size(), 0.0); + for (size_t i = 0; i < o.n.size(); ++i) + n[i] += o.n[i]; + uniformRedundant += o.uniformRedundant; + return *this; +} + +double GLCalls::total() const { + double s = 0; + for (double v : n) + s += v; + return s; +} + +double GLCalls::count(const std::string &entry) const { + for (int i = 0; i < E_COUNT && i < static_cast(n.size()); ++i) + if (entry == g_names[i]) + return n[i]; + return 0; +} + +double GLCalls::uniforms() const { + double s = 0; + for (int i = 0; i < static_cast(n.size()); ++i) + s += isUniform(i) ? n[i] : 0.0; + return s; +} + +double GLCalls::sync() const { + double s = 0; + for (int i = 0; i < static_cast(n.size()); ++i) + s += isSync(i) ? n[i] : 0.0; + return s; +} + +bool GLCalls::sameAs(const GLCalls &o, std::string *firstDifference) const { + for (int i = 0; i < E_COUNT; ++i) { + const double a = i < static_cast(n.size()) ? n[i] : 0.0; + const double b = i < static_cast(o.n.size()) ? o.n[i] : 0.0; + if (a != b) { + if (firstDifference) { + char buf[128]; + std::snprintf(buf, sizeof buf, "gl%s %g vs %g", g_names[i], a, b); + *firstDifference = buf; + } + return false; + } + } + return true; +} + +std::string GLCalls::str(double per, int top) const { + if (per <= 0) + per = 1; + char buf[160]; + std::snprintf(buf, sizeof buf, "total %.1f (uniforms %.1f, redundant %.1f, sync %.1f) |", + total() / per, uniforms() / per, uniformRedundant / per, sync() / per); + std::string out = buf; + std::vector> v; + for (int i = 0; i < static_cast(n.size()); ++i) + if (n[i] != 0) + v.emplace_back(n[i], i); + std::sort(v.begin(), v.end(), [](const auto &a, const auto &b) { + return a.first != b.first ? a.first > b.first : a.second < b.second; + }); + for (int k = 0; k < static_cast(v.size()) && k < top; ++k) { + std::snprintf(buf, sizeof buf, " %s=%.2f", g_names[v[k].second], v[k].first / per); + out += buf; + } + return out; +} + +} // namespace cvcgl_test diff --git a/src/cvcGL/test/gl_call_counter.h b/src/cvcGL/test/gl_call_counter.h new file mode 100644 index 000000000..f5622fd0c --- /dev/null +++ b/src/cvcGL/test/gl_call_counter.h @@ -0,0 +1,52 @@ +/* + Copyright 2026 The University of Texas at Austin + + This file is part of libcvc. + + libcvc is free software; you can redistribute it and/or + modify it under the terms of the GNU Lesser General Public + License version 2.1 as published by the Free Software Foundation. +*/ + +// Test-only GL call counter: swaps VTK's glad function pointers (glad_gl*) for +// counting wrappers, so a test can see every GL call the scene issues, by entry +// point. Uniform uploads are also checked against the last value sent to the +// same (program, location): a re-send of an unchanged value counts as +// redundant. Desktop GL only (the wasm VTK calls GLES directly, not via glad). +#ifndef CVCGL_TEST_GL_CALL_COUNTER_H +#define CVCGL_TEST_GL_CALL_COUNTER_H + +#include +#include + +namespace cvcgl_test { + +struct GLCalls { + std::vector n; // per entry point (index = glcallsName(i)) + double uniformRedundant = 0; // uniform uploads that repeated the current value + + GLCalls operator-(const GLCalls &o) const; + GLCalls &operator+=(const GLCalls &o); + double total() const; + double count(const std::string &entry) const; // entry without the "gl" prefix + double uniforms() const; // every glUniform* + // Calls that make WebGL wait for the GPU process: glGet*, glIsEnabled, + // glGetError, glGetUniformLocation, glCheckFramebufferStatus, glReadPixels... + double sync() const; + // Same count for every entry point; otherwise the first difference. + bool sameAs(const GLCalls &o, std::string *firstDifference = nullptr) const; + // "total T (uniforms U, redundant R, sync S) | Entry=k ..." with every number + // divided by `per`; the `top` largest entries. + std::string str(double per = 1.0, int top = 14) const; +}; + +// (Re)install the wrappers. Call once a GL context exists (VTK has loaded glad), +// and again after anything that may reload it; idempotent. +void glcallsInstall(); +GLCalls glcallsRead(); +int glcallsEntries(); +const char *glcallsName(int i); + +} // namespace cvcgl_test + +#endif From 01da6a6eeaa7302343c0d63e05a0559b4dc208d9 Mon Sep 17 00:00:00 2001 From: Joe Rivera Date: Thu, 1 Oct 2026 17:49:44 -0500 Subject: [PATCH 03/18] refactor(cvcGL): StreamingLowMemoryPolyDataMapper on LowMemoryPolyDataMapper #512 and the fast-draw mapper landed independently; re-parent the streaming low-memory mapper so streaming overlays (tracks, ribbons, draped links) get the per-draw cuts too. Its draw-range RenderPieceDraw narrows cell group 0 and then calls the base, which draws exactly that group. Its constructor's shift/scale initialisation moves to the base (now applied on every VTK 9.x: 9.5.1+ initialise only the bool). --- inc/cvc/gl/StreamingMappers.h | 12 ++++++++---- src/cvcGL/LowMemoryPolyDataMapper.cpp | 9 +++++---- src/cvcGL/StreamingMappers.cpp | 15 ++++----------- 3 files changed, 17 insertions(+), 19 deletions(-) diff --git a/inc/cvc/gl/StreamingMappers.h b/inc/cvc/gl/StreamingMappers.h index f26273a44..e600bab58 100644 --- a/inc/cvc/gl/StreamingMappers.h +++ b/inc/cvc/gl/StreamingMappers.h @@ -60,9 +60,9 @@ #include #include +#include #include // StreamingMapperKind #include -#include #include #include @@ -139,10 +139,14 @@ class StreamingPolyDataMapper : public vtkOpenGLPolyDataMapper { void operator=(const StreamingPolyDataMapper &) = delete; }; -class StreamingLowMemoryPolyDataMapper : public vtkOpenGLLowMemoryPolyDataMapper { +// On top of cvc::gl::LowMemoryPolyDataMapper, so a streaming overlay also gets +// its per-draw cuts (empty cell types skipped, cached program re-bound) and its +// shift/scale initialisation; the draw range below narrows cell group 0 before +// that draw runs. +class StreamingLowMemoryPolyDataMapper : public LowMemoryPolyDataMapper { public: static StreamingLowMemoryPolyDataMapper *New(); - vtkTypeMacro(StreamingLowMemoryPolyDataMapper, vtkOpenGLLowMemoryPolyDataMapper); + vtkTypeMacro(StreamingLowMemoryPolyDataMapper, LowMemoryPolyDataMapper); StreamingMapperCore &core() { return Core; } const StreamingMapperCore &core() const { return Core; } @@ -151,7 +155,7 @@ class StreamingLowMemoryPolyDataMapper : public vtkOpenGLLowMemoryPolyDataMapper void RenderPieceDraw(vtkRenderer *ren, vtkActor *act) override; protected: - StreamingLowMemoryPolyDataMapper(); + StreamingLowMemoryPolyDataMapper() = default; ~StreamingLowMemoryPolyDataMapper() override = default; void ComputeBounds() override; diff --git a/src/cvcGL/LowMemoryPolyDataMapper.cpp b/src/cvcGL/LowMemoryPolyDataMapper.cpp index cb4fde5e2..a4be54a81 100644 --- a/src/cvcGL/LowMemoryPolyDataMapper.cpp +++ b/src/cvcGL/LowMemoryPolyDataMapper.cpp @@ -125,10 +125,11 @@ CoincidentValue coincidentValue(vtkMapper *mapper, vtkProperty *prop, int slot) vtkStandardNewMacro(LowMemoryPolyDataMapper); LowMemoryPolyDataMapper::LowMemoryPolyDataMapper() { -#if CVC_GL_LOWMEM_REPLICA - // VTK 9.5.0 declares these without initialisers. With DISABLE_SHIFT_SCALE - // nothing ever assigns them, so whether the mapper uploaded a (garbage) - // shift/scale copy of the positions depended on a heap byte. +#if VTK_MAJOR_VERSION == 9 + // VTK 9.5.0 declares these without initialisers (9.5.1 initialises the bool + // only). With DISABLE_SHIFT_SCALE nothing ever assigns them, so whether the + // mapper uploaded a (garbage) shift/scale copy of the positions depended on a + // heap byte. this->CoordinateShiftAndScaleInUse = false; this->ShiftValues.fill(0.0); this->ScaleValues.fill(1.0); diff --git a/src/cvcGL/StreamingMappers.cpp b/src/cvcGL/StreamingMappers.cpp index dfcefe4c0..72a03f27a 100644 --- a/src/cvcGL/StreamingMappers.cpp +++ b/src/cvcGL/StreamingMappers.cpp @@ -190,19 +190,12 @@ void StreamingPolyDataMapper::RenderPieceDraw(vtkRenderer *ren, vtkActor *act) { } // ─────────────────────────────── low-memory mapper ────────────────────────── +// The base, LowMemoryPolyDataMapper, initialises the shift/scale members VTK +// 9.5.0 leaves uninitialised: with garbage there the mapper uploaded a +// shift/scale COPY of the positions, which also defeats streaming (the uploaded +// array must be the input's own). vtkStandardNewMacro(StreamingLowMemoryPolyDataMapper); -StreamingLowMemoryPolyDataMapper::StreamingLowMemoryPolyDataMapper() { - // VTK 9.5.0 never initialises these (9.5.2 fixes the bool only). With - // DISABLE_SHIFT_SCALE nothing ever assigns them, so the mapper applied a - // GARBAGE shift/scale copy of the positions whenever the heap byte under the - // bool happened to be non-zero -- nondeterministic, and it also defeats the - // streaming path (the uploaded array is then a copy, not the input's). - this->CoordinateShiftAndScaleInUse = false; - this->ShiftValues.fill(0.0); - this->ScaleValues.fill(1.0); -} - void StreamingLowMemoryPolyDataMapper::ComputeBounds() { if (Core.beforeBounds) Core.beforeBounds(); From da9dabeffc6006f583be1d84a8088a8ad6aed0cb Mon Sep 17 00:00:00 2001 From: Joe Rivera Date: Thu, 1 Oct 2026 19:05:28 -0500 Subject: [PATCH 04/18] fix(cvcGL): LowMemoryPolyDataMapper -- late shader replacements, replica guards, VTK attribution Review fixes for the fast-draw mapper: - Shader-property changes after the first draw: VTK's low-memory mapper (9.5 through 9.7) never compares the actor's shader property against its built program, so a GeometryNode::add*ShaderReplacement (or a custom uniform declared) after the first draw never reached the GPU on wasm. The GetShaderMTime() > ShaderBuildTimeStamp check that only StreamingLowMemoryPolyDataMapper had moves into LowMemoryPolyDataMapper::RenderPieceStart (every VTK version), and the streaming subclass's copy is removed. A rebuild bumps the stamp, so the next draw also takes the full shader-cache lookup. - The shader-cache skip's invariant is stated in the header: this->Shaders changes only in UpdateShaders, always followed by ShaderBuildTimeStamp.Modified(). vtkDrawTexturedElements::GetShader() is public, so the class hides it with a version that drops the cached program first (as upstream f59c2d0b297 does), and invalidateProgramCache() is there for any other route. - The 9.5.0 replica no longer trusts the version number alone. Contract for the libcvc-deps VTK recipe: any patch to the low-memory / DTE / GLSL-mod code defines VTK_CVC_LOWMEM_PATCHLEVEL in the installed vtkOpenGLLowMemoryPolyDataMapper.h, and the replica turns itself off when it is defined. cvcGL's configure step also hashes the three installed headers the replica builds on against pristine v9.5.0 (line endings normalised) and builds the replica off (CVC_GL_LOWMEM_REPLICA_DISABLED) on any difference. - Records in the header that VTK v9.7.0 (2026-08-14) contains f59c2d0b297 (DTE program re-bind) and 15ed6edc478 (per-program uniform location and value caches) but still runs all four agents, so a VTK upgrade must re-port the empty-agent skip; and that its d1cde2af1a8 adds two glCheckFramebufferStatus per vtkglBlitFramebuffer (4 synchronous calls a frame on WebGL) and should be reverted or avoided for the wasm build. - License: the regions of LowMemoryPolyDataMapper.cpp derived from VTK 9.5.0 (GLSLModCoincidentTopology::GetCoincidentParameters, RenderPieceDraw's agent loop, CellTypeAgent PreDraw/Draw and the Vertices/Lines/Polygons PreDrawInternal) are marked BEGIN/END VTK 9.5.0-DERIVED with VTK's BSD-3-Clause notice (Copyright.txt wording) at the top of the file. THIRD_PARTY_NOTICES.md records it (with the existing XmlRpc++ note), the README's License section points to it, and both the top-level and the standalone cvcGL installs ship it in the doc dir. - A stage observer (null by default; one relaxed load per stage) lets the test interleave the replica's draw stages with a GL trace. --- CMakeLists.txt | 2 +- README.md | 6 ++ THIRD_PARTY_NOTICES.md | 56 ++++++++++ inc/cvc/gl/LowMemoryPolyDataMapper.h | 88 +++++++++++++-- src/cvcGL/CMakeLists.txt | 79 +++++++++++--- src/cvcGL/LowMemoryPolyDataMapper.cpp | 150 +++++++++++++++++++++++--- src/cvcGL/StreamingMappers.cpp | 12 +-- 7 files changed, 352 insertions(+), 41 deletions(-) create mode 100644 THIRD_PARTY_NOTICES.md diff --git a/CMakeLists.txt b/CMakeLists.txt index b57f1f00a..608ff1bfc 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -390,7 +390,7 @@ if(CVC_BUILD_PYCVC) endif() # ************* Install documentation & resources ************* -install(FILES README.md USAGE.md LICENSE +install(FILES README.md USAGE.md LICENSE THIRD_PARTY_NOTICES.md DESTINATION ${CMAKE_INSTALL_DOCDIR} COMPONENT libcvc) diff --git a/README.md b/README.md index 2aabb4f71..a64085658 100644 --- a/README.md +++ b/README.md @@ -704,6 +704,12 @@ libcvc is licensed under the **GNU Lesser General Public License, version 2.1** The bundled XmlRpc++ sources under `inc/xmlrpc/` and `src/xmlrpc/` are Copyright (c) 2002-2003 Chris Morley and are used under LGPL-2.1-or-later. +Portions of `src/cvcGL/LowMemoryPolyDataMapper.cpp` are derived from VTK 9.5.0, +Copyright (c) Ken Martin, Will Schroeder, Bill Lorensen, and are used under the +BSD-3-Clause license. The full notices for third-party code are in +[THIRD_PARTY_NOTICES.md](THIRD_PARTY_NOTICES.md), which is installed with the +documentation. + ## Credits Based on research and software developed at the **Computational Visualization Center, University of Texas at Austin**. diff --git a/THIRD_PARTY_NOTICES.md b/THIRD_PARTY_NOTICES.md new file mode 100644 index 000000000..110c3364c --- /dev/null +++ b/THIRD_PARTY_NOTICES.md @@ -0,0 +1,56 @@ +# Third-party notices + +libcvc is licensed under the GNU Lesser General Public License, version 2.1 +(see [LICENSE](LICENSE)). Some of its source is derived from, or bundles, code +under other licenses. Those notices are recorded here and installed with the +documentation. + +## XmlRpc++ + +The bundled XmlRpc++ sources under `inc/xmlrpc/` and `src/xmlrpc/` are +Copyright (c) 2002-2003 Chris Morley and are used under the GNU Lesser General +Public License, version 2.1 or later. + +## VTK (The Visualization Toolkit) 9.5.0 + +Portions of `src/cvcGL/LowMemoryPolyDataMapper.cpp` -- the regions between +`BEGIN VTK 9.5.0-DERIVED` and `END VTK 9.5.0-DERIVED` -- are derived from +VTK 9.5.0 (`Rendering/OpenGL2`: `vtkGLSLModCoincidentTopology.cxx`, +`vtkOpenGLLowMemoryPolyDataMapper.cxx`, `vtkOpenGLLowMemoryCellTypeAgent.cxx`, +`vtkOpenGLLowMemoryVerticesAgent.cxx`, `vtkOpenGLLowMemoryLinesAgent.cxx`, +`vtkOpenGLLowMemoryPolygonsAgent.cxx`). They are used under VTK's license: + +``` +SPDX-FileCopyrightText: Copyright (c) Ken Martin, Will Schroeder, Bill Lorensen +SPDX-License-Identifier: BSD-3-Clause + +Copyright (c) 1993-2015 Ken Martin, Will Schroeder, Bill Lorensen +All rights reserved. + +Redistribution and use in source and binary forms, with or without +modification, are permitted provided that the following conditions are met: + + * Redistributions of source code must retain the above copyright notice, + this list of conditions and the following disclaimer. + + * Redistributions in binary form must reproduce the above copyright notice, + this list of conditions and the following disclaimer in the documentation + and/or other materials provided with the distribution. + + * Neither name of Ken Martin, Will Schroeder, or Bill Lorensen nor the names + of any contributors may be used to endorse or promote products derived + from this software without specific prior written permission. + +THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS ``AS IS'' +AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE +IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE +ARE DISCLAIMED. IN NO EVENT SHALL THE AUTHORS OR CONTRIBUTORS BE LIABLE FOR +ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL +DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR +SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER +CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, +OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE +OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. +``` + +The same notice is reproduced at the top of that source file. diff --git a/inc/cvc/gl/LowMemoryPolyDataMapper.h b/inc/cvc/gl/LowMemoryPolyDataMapper.h index 1bf72f753..bddf1fd5c 100644 --- a/inc/cvc/gl/LowMemoryPolyDataMapper.h +++ b/inc/cvc/gl/LowMemoryPolyDataMapper.h @@ -43,11 +43,56 @@ // // The cell-type agents are VTK private classes (headers not installed, symbols // hidden), so the drawn agent's PreDraw/Draw/PostDraw is replicated here from -// VTK 9.5.0. That replica compiles only against exactly 9.5.0; on any other -// VTK this class forwards to the stock draw and only the factory below remains. -// Picking, vertex visibility and a coincident-offset layout where skipping a -// cell type would change the depth offset a drawn one sees all take the stock -// (all four agents) draw. +// VTK 9.5.0 (BSD-3-Clause; the notice is in LowMemoryPolyDataMapper.cpp and +// THIRD_PARTY_NOTICES.md). Picking, vertex visibility and a coincident-offset +// layout where skipping a cell type would change the depth offset a drawn one +// sees all take the stock (all four agents) draw. +// +// The replica is active (replicaActive()) only against an UNPATCHED VTK 9.5.0. +// The version macros alone cannot tell a patched 9.5.0 from a pristine one, so: +// * Contract for the libcvc-deps VTK recipe: any patch that touches the +// Rendering/OpenGL2 low-memory mapper, its cell-type agents, +// vtkDrawTexturedElements or the GLSL mods must also add +// #define VTK_CVC_LOWMEM_PATCHLEVEL // n >= 1 +// to the installed vtkOpenGLLowMemoryPolyDataMapper.h. The replica turns +// itself off whenever that macro is defined (every DrawPath is then the +// stock draw) until it is re-validated against the patched code and taught +// that patch level. +// * At configure time cvcGL also hashes the installed +// vtkOpenGLLowMemoryPolyDataMapper.h, vtkDrawTexturedElements.h and +// vtkGLSLModCoincidentTopology.h against pristine 9.5.0 and builds the +// replica off (CVC_GL_LOWMEM_REPLICA_DISABLED) if any of them differs. +// cvcgl_lowmem_fastdraw pins the replica against the stock draw call for call, +// so it is the gate for any VTK recipe change. +// +// VTK v9.7.0 (tagged 2026-08-14) is NOT a drop-in replacement. It carries +// upstream f59c2d0b297 (vtkDrawTexturedElements re-binds its cached program +// unless GetShader() was called -- this class's cut 2) and 15ed6edc478 +// (per-program uniform-location and last-value caches in vtkShaderProgram, the +// GLSL mods and the low-memory mapper), but it still runs all four cell-type +// agents per draw. A VTK upgrade switches the replica off (version check), so +// the empty-agent skip (cut 1) must be re-ported against 9.7's agents or that +// saving is lost. 9.7.0 also contains d1cde2af1a8, which makes every +// vtkOpenGLState::vtkglBlitFramebuffer issue two glCheckFramebufferStatus calls +// first -- at a cvcGL frame's two blits, 4 synchronous round trips per frame on +// WebGL; revert or avoid it in the wasm VTK build. +// +// The shader-cache skip (cut 2) relies on one invariant: this->Shaders (the +// vtkShader sources vtkDrawTexturedElements resolves its program from) changes +// only inside UpdateShaders, and RenderPieceStart always follows UpdateShaders +// with ShaderBuildTimeStamp.Modified(). vtkDrawTexturedElements::GetShader() is +// public and hands out a mutable vtkShader, so this class hides it with a +// version that drops the cached program first (as upstream f59c2d0b297 does); +// anything that edits shader sources by another route must call +// invalidateProgramCache(). Nothing in cvcGL edits them outside +// UpdateShaders today. +// +// For every VTK version, RenderPieceStart also rebuilds the shader program when +// the actor's shader property (replacements, custom uniform declarations) +// changed after the program was built: VTK's low-memory mapper (9.5 through +// 9.7) ignores the shader property once its program exists, so a +// GeometryNode::add*ShaderReplacement made after the first draw would never +// reach the GPU on WebGL. // // It also zero-initialises the shift/scale members VTK 9.5.0 leaves // uninitialised (with DISABLE_SHIFT_SCALE nothing ever assigns them, so a @@ -59,6 +104,7 @@ #include #include #include +#include #include #include @@ -81,8 +127,8 @@ class LowMemoryPolyDataMapper : public vtkOpenGLLowMemoryPolyDataMapper { static void setDrawPath(DrawPath path); static DrawPath drawPath(); - // True when this build carries the VTK 9.5.0 replica. False means every path - // is the stock draw. + // True when this build carries the VTK 9.5.0 replica (an unpatched 9.5.0, see + // above). False means every path is the stock draw. static bool replicaActive(); // Per-mapper counters, read and reset on the render thread. @@ -97,6 +143,34 @@ class LowMemoryPolyDataMapper : public vtkOpenGLLowMemoryPolyDataMapper { const Stats &stats() const { return m_stats; } void resetStats() { m_stats = Stats{}; } + // Test instrumentation: when set, the replica reports where each stage of a + // draw begins and ends (the cvcgl_lowmem_fastdraw GL trace interleaves these + // with the GL calls to derive what Fast must leave out). Null -- the default + // -- costs one relaxed load per stage. Stock draws report nothing. + enum class DrawStage : int { + DrawBegin, // arg: 1 if Fast may skip empty cell types in this draw + LookupBegin, // full shader-cache lookup + LookupEnd, + RebindBegin, // cached program re-bound instead + RebindEnd, + AgentBegin, // arg: cell type 0-3 + AgentEnd, // arg: cell type 0-3 + AgentSkipped, + DrawEnd + }; + using StageObserver = void (*)(DrawStage stage, int arg); + static void setStageObserver(StageObserver observer); + + // Drop the program the last lookup resolved, so the next draw looks it up in + // the shader cache again. For anything that edits this->Shaders outside + // UpdateShaders (see the invariant above). + void invalidateProgramCache(); + // vtkDrawTexturedElements::GetShader (public, non-virtual) hands out a + // mutable source; this hides it so a caller holding this class also drops + // the cached program. + vtkShader *GetShader(vtkShader::Type shaderType); + + void RenderPieceStart(vtkRenderer *ren, vtkActor *act) override; void RenderPieceDraw(vtkRenderer *ren, vtkActor *act) override; void ReleaseGraphicsResources(vtkWindow *win) override; diff --git a/src/cvcGL/CMakeLists.txt b/src/cvcGL/CMakeLists.txt index b95df6472..662ee993b 100644 --- a/src/cvcGL/CMakeLists.txt +++ b/src/cvcGL/CMakeLists.txt @@ -54,6 +54,50 @@ target_include_directories(cvcGL PUBLIC ) target_link_libraries(cvcGL PUBLIC cvc::cvc ${VTK_LIBRARIES}) +# LowMemoryPolyDataMapper carries a replica of VTK 9.5.0's private low-memory +# cell-type agents, valid only against an UNPATCHED 9.5.0 (the contract is in +# inc/cvc/gl/LowMemoryPolyDataMapper.h). The version number cannot tell a +# libcvc-deps recipe patch from pristine 9.5.0, so: a patch to that code must +# define VTK_CVC_LOWMEM_PATCHLEVEL in the installed +# vtkOpenGLLowMemoryPolyDataMapper.h (checked in the source), and here the +# installed headers the replica builds on are hashed against v9.5.0's (line +# endings normalised) -- any difference builds the replica off. +if(VTK_VERSION VERSION_EQUAL "9.5.0") + set(_cvc_lowmem_pristine + "vtkOpenGLLowMemoryPolyDataMapper.h=a4100d6d90f72861ce4eacd6e29506bbab727fd28fe60d6fdb319d2695d61c1a" + "vtkDrawTexturedElements.h=07df517fba54da88b2ac1b1ad4ca847aea7b2eda72422bbeeafe0f8d20a520af" + "vtkGLSLModCoincidentTopology.h=708637ca8dc0695a60c1bb71269220b726cfe7f37af54c214430d4f8f2b42033") + set(_cvc_lowmem_inc "${VTK_PREFIX_PATH}/include/vtk-${VTK_VERSION_MAJOR}.${VTK_VERSION_MINOR}") + set(_cvc_lowmem_changed "") + set(_cvc_lowmem_missing "") + foreach(_entry IN LISTS _cvc_lowmem_pristine) + string(REPLACE "=" ";" _pair "${_entry}") + list(GET _pair 0 _hdr) + list(GET _pair 1 _want) + if(EXISTS "${_cvc_lowmem_inc}/${_hdr}") + file(READ "${_cvc_lowmem_inc}/${_hdr}" _text) + string(REPLACE "\r\n" "\n" _text "${_text}") + string(SHA256 _have "${_text}") + if(NOT _have STREQUAL _want) + list(APPEND _cvc_lowmem_changed "${_hdr}") + endif() + else() + list(APPEND _cvc_lowmem_missing "${_hdr}") + endif() + endforeach() + if(_cvc_lowmem_changed) + target_compile_definitions(cvcGL PRIVATE CVC_GL_LOWMEM_REPLICA_DISABLED=1) + message(WARNING "cvcGL: VTK 9.5.0 headers differ from upstream v9.5.0 (${_cvc_lowmem_changed}); " + "LowMemoryPolyDataMapper's replica is OFF (stock draw). Re-validate it with " + "cvcgl_lowmem_fastdraw against the patched VTK before re-enabling.") + elseif(_cvc_lowmem_missing) + message(STATUS "cvcGL: could not find ${_cvc_lowmem_missing} under ${_cvc_lowmem_inc}; " + "LowMemoryPolyDataMapper relies on VTK_CVC_LOWMEM_PATCHLEVEL alone") + else() + message(STATUS "cvcGL: VTK 9.5.0 low-memory mapper headers are pristine; replica enabled") + endif() +endif() + # --------------------------------------------------------------------------- # Dear ImGui overlay (optional): real menus/panels/widgets drawn inside the VTK # render, native AND WebAssembly. The cvcpkg imgui bundle ships no CMake config, @@ -238,6 +282,11 @@ install(FILES "${CMAKE_CURRENT_BINARY_DIR}/cvcGLConfigVersion.cmake" DESTINATION lib/cmake/cvcGL ) +# libcvcGL carries code derived from VTK 9.5.0 (BSD-3-Clause, in +# LowMemoryPolyDataMapper.cpp), whose notice a binary redistribution must +# reproduce in its documentation: the standalone cvcGL package ships it too. +install(FILES "${CMAKE_CURRENT_SOURCE_DIR}/../../THIRD_PARTY_NOTICES.md" + DESTINATION ${CMAKE_INSTALL_DOCDIR}) # ── functional smoke test ── enable_testing() @@ -596,19 +645,25 @@ if(NOT EMSCRIPTEN) endif() # LowMemoryPolyDataMapper (the GLES3/WebGL2 mapper with empty cell types and the -# per-draw shader-cache lookup taken out of its draw), forced on natively. Counts -# every GL call at the driver (test/gl_call_counter.cpp swaps VTK's glad function -# pointers) to pin the VTK 9.5.0 replica call-for-call against the stock draw and -# measure the saving, and compares RGBA frames byte for byte -- shadows on and -# off, bake and steady frames, released resources, a re-opened window, the -# offset / picking / vertex-visibility guards. Renders for real; skips where -# nothing rasterises unless CVC_REQUIRE_RENDER=1. CVCGL_LOWMEM_REPORT=1 prints -# per-draw breakdowns. Desktop GL only: the wasm VTK calls GLES directly, not -# through glad. +# per-draw shader-cache lookup taken out of its draw), forced on natively. Records +# an ordered trace of every GL call at the driver -- entry point, bound program / +# texture unit, argument bytes, uniform values (test/gl_call_counter.cpp swaps +# VTK's glad function pointers) -- and asserts the VTK 9.5.0 replica's trace +# equals the stock draw's exactly and Fast's equals it with the empty cell-type +# blocks removed (derived from the replica's stage markers); compares RGBA frames +# byte for byte -- shadows on and off, bake and steady frames, every +# representation / scalar / edge / line kind, the hardware selector, released +# resources, a re-opened window, the offset / picking / vertex-visibility guards, +# and shader replacements made after the first draw. This is the gate for any +# libcvc-deps VTK recipe change. Renders for real; skips where nothing +# rasterises unless CVC_REQUIRE_RENDER=1. CVCGL_LOWMEM_REPORT=1 prints per-draw +# breakdowns; CVCGL_LOWMEM_DUMP= writes the traces of a failing comparison. +# Desktop GL only: the wasm VTK calls GLES directly, not through glad. # -# cvcgl_lowmem_bench (not a test) prints GL calls and draw-stage CPU per frame -# and per draw, Stock vs Fast, for a city of single-colour box nodes on a -# textured ground with overlays, shadows on and off. +# cvcgl_lowmem_bench (not a test) prints GL calls (counted pass) and draw-stage / +# frame CPU (timed pass, no GL wrappers) per frame and per draw, Stock vs Fast, +# for a city of single-colour box nodes on a textured ground with overlays, +# shadows on and off. if(NOT EMSCRIPTEN) add_executable(cvcgl_lowmem_fastdraw test/cvcgl_lowmem_fastdraw.cpp test/gl_call_counter.cpp) target_link_libraries(cvcgl_lowmem_fastdraw PRIVATE cvcGL) diff --git a/src/cvcGL/LowMemoryPolyDataMapper.cpp b/src/cvcGL/LowMemoryPolyDataMapper.cpp index a4be54a81..04a89f7cd 100644 --- a/src/cvcGL/LowMemoryPolyDataMapper.cpp +++ b/src/cvcGL/LowMemoryPolyDataMapper.cpp @@ -17,6 +17,49 @@ Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA */ +/* + Portions of this file -- the regions between "BEGIN VTK 9.5.0-DERIVED" and + "END VTK 9.5.0-DERIVED" below -- are derived from VTK 9.5.0 + (Rendering/OpenGL2: vtkGLSLModCoincidentTopology.cxx, + vtkOpenGLLowMemoryPolyDataMapper.cxx, vtkOpenGLLowMemoryCellTypeAgent.cxx, + vtkOpenGLLowMemoryVerticesAgent.cxx, vtkOpenGLLowMemoryLinesAgent.cxx, + vtkOpenGLLowMemoryPolygonsAgent.cxx) and are used under VTK's license: + + SPDX-FileCopyrightText: Copyright (c) Ken Martin, Will Schroeder, Bill Lorensen + SPDX-License-Identifier: BSD-3-Clause + + Copyright (c) 1993-2015 Ken Martin, Will Schroeder, Bill Lorensen + All rights reserved. + + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions are met: + + * Redistributions of source code must retain the above copyright notice, + this list of conditions and the following disclaimer. + + * Redistributions in binary form must reproduce the above copyright notice, + this list of conditions and the following disclaimer in the documentation + and/or other materials provided with the distribution. + + * Neither name of Ken Martin, Will Schroeder, or Bill Lorensen nor the names + of any contributors may be used to endorse or promote products derived + from this software without specific prior written permission. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS ``AS IS'' + AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE + IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE + ARE DISCLAIMED. IN NO EVENT SHALL THE AUTHORS OR CONTRIBUTORS BE LIABLE FOR + ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL + DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR + SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER + CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, + OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE + OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + + The rest of the file is libcvc's own (LGPL-2.1, above). The notice is also + recorded in THIRD_PARTY_NOTICES.md. +*/ + // No vtkOpenGL*ErrorMacro anywhere in this file: those macros are inline in the // including translation unit, so a cvcGL build without NDEBUG would issue a // glGetError per draw -- a synchronous round trip on WebGL. @@ -33,9 +76,19 @@ #include #include #include +#include #include -#if VTK_MAJOR_VERSION == 9 && VTK_MINOR_VERSION == 5 && VTK_BUILD_VERSION == 0 +// The replica is VTK 9.5.0's own agent code, so it is on only for an UNPATCHED +// 9.5.0 (the contract is in the class header): +// VTK_CVC_LOWMEM_PATCHLEVEL defined by any libcvc-deps VTK recipe patch +// to the low-memory / vtkDrawTexturedElements +// code, in the installed +// vtkOpenGLLowMemoryPolyDataMapper.h; +// CVC_GL_LOWMEM_REPLICA_DISABLED set by cvcGL's configure-time check when an +// installed header differs from 9.5.0's. +#if VTK_MAJOR_VERSION == 9 && VTK_MINOR_VERSION == 5 && VTK_BUILD_VERSION == 0 && \ + !defined(VTK_CVC_LOWMEM_PATCHLEVEL) && !defined(CVC_GL_LOWMEM_REPLICA_DISABLED) #define CVC_GL_LOWMEM_REPLICA 1 #else #define CVC_GL_LOWMEM_REPLICA 0 @@ -74,6 +127,9 @@ std::atomic &policyStore() { return policy; } +// Constant-initialised: no guard on the per-stage load. +std::atomic g_stageObserver{nullptr}; + // Does VTK's object factory hand out the low-memory mapper on this build? Asked // once, on the first node construction (the OpenGL2 factory is registered by // then: cvcGL's sources carry the module auto-init). @@ -84,6 +140,12 @@ bool factoryIsLowMemory() { } #if CVC_GL_LOWMEM_REPLICA +// ---- BEGIN VTK 9.5.0-DERIVED ------------------------------------------------ +// Portions derived from VTK 9.5.0, Copyright (c) Ken Martin, Will Schroeder, +// Bill Lorensen; All rights reserved. SPDX-License-Identifier: BSD-3-Clause +// (full notice at the top of this file). coincidentValue() below is +// vtkGLSLModCoincidentTopology::GetCoincidentParameters. +// // The depth offset vtkGLSLModCoincidentTopology::GetCoincidentParameters (VTK // 9.5.0) computes for one primitive class, as the floats it would upload. // slot: 0 points, 1 lines, 2 polygons. `set` false: the mod uploads nothing. @@ -118,6 +180,7 @@ CoincidentValue coincidentValue(vtkMapper *mapper, vtkProperty *prop, int slot) v.offset = offset; return v; } +// ---- END VTK 9.5.0-DERIVED -------------------------------------------------- #endif } // namespace @@ -146,10 +209,40 @@ LowMemoryPolyDataMapper::DrawPath LowMemoryPolyDataMapper::drawPath() { bool LowMemoryPolyDataMapper::replicaActive() { return CVC_GL_LOWMEM_REPLICA != 0; } -void LowMemoryPolyDataMapper::ReleaseGraphicsResources(vtkWindow *win) { - // The cached program belongs to that window's context; look it up afresh. +void LowMemoryPolyDataMapper::setStageObserver(StageObserver observer) { + g_stageObserver.store(observer, std::memory_order_relaxed); +} + +void LowMemoryPolyDataMapper::invalidateProgramCache() { m_cache = nullptr; m_builtStamp = 0; +} + +vtkShader *LowMemoryPolyDataMapper::GetShader(vtkShader::Type shaderType) { + // The caller may edit the source: the next draw must resolve it again. + invalidateProgramCache(); + return this->vtkDrawTexturedElements::GetShader(shaderType); +} + +void LowMemoryPolyDataMapper::RenderPieceStart(vtkRenderer *ren, vtkActor *act) { + // VTK's low-memory mapper (9.5 through 9.7) never looks at the actor's shader + // property once its program is built: IsShaderUpToDate ignores it, where the + // classic mapper compares GetShaderMTime(). A replacement added or cleared + // after the first draw (GeometryNode::add*ShaderReplacement / + // clearShaderReplacements), or a custom uniform declared after it, would + // never reach the GPU. Dropping the program makes the base rebuild it + // (UpdateShaders + a new ShaderBuildTimeStamp, so the draw that follows also + // takes the full shader-cache lookup). Uniform VALUE changes do not move + // GetShaderMTime, so they never cost a rebuild. + if (vtkShaderProperty *sp = act->GetShaderProperty()) + if (sp->GetShaderMTime() > this->ShaderBuildTimeStamp.GetMTime()) + this->ShaderProgram = nullptr; + Superclass::RenderPieceStart(ren, act); +} + +void LowMemoryPolyDataMapper::ReleaseGraphicsResources(vtkWindow *win) { + // The cached program belongs to that window's context; look it up afresh. + invalidateProgramCache(); Superclass::ReleaseGraphicsResources(win); } @@ -170,6 +263,22 @@ void LowMemoryPolyDataMapper::agentDraw(int, vtkRenderer *, vtkActor *) {} #else +namespace { +inline void observeStage(LowMemoryPolyDataMapper::StageObserver observe, + LowMemoryPolyDataMapper::DrawStage stage, int arg = 0) { + if (observe) + observe(stage, arg); +} +} // namespace + +// ---- BEGIN VTK 9.5.0-DERIVED ------------------------------------------------ +// Portions derived from VTK 9.5.0, Copyright (c) Ken Martin, Will Schroeder, +// Bill Lorensen; All rights reserved. SPDX-License-Identifier: BSD-3-Clause +// (full notice at the top of this file). RenderPieceDraw's agent loop is +// vtkOpenGLLowMemoryPolyDataMapper::RenderPieceDraw; renderable(), agentPreDraw() +// and agentDraw() are vtkOpenGLLowMemoryCellTypeAgent::PreDraw / Draw and the +// vtkOpenGLLowMemory{Vertices,Lines,Polygons}Agent::PreDrawInternal they call. +// readyProgram() and coincidentSkipSafe() are libcvc's own. void LowMemoryPolyDataMapper::RenderPieceDraw(vtkRenderer *ren, vtkActor *act) { ++m_stats.draws; const DrawPath path = drawPath(); @@ -181,45 +290,61 @@ void LowMemoryPolyDataMapper::RenderPieceDraw(vtkRenderer *ren, vtkActor *act) { return; } const bool fast = path == DrawPath::Fast; + const StageObserver observe = g_stageObserver.load(std::memory_order_relaxed); + // CPU only (no GL); asked on AllCellTypes too when observed, so a test can + // derive from an AllCellTypes trace what Fast must leave out. + const bool skipSafe = (fast || observe) && coincidentSkipSafe(act); + observeStage(observe, DrawStage::DrawBegin, skipSafe ? 1 : 0); readyProgram(ren, fast); // Uniforms common to every cell type; also fires UpdateShaderEvent, which // GeometryNode's custom shader textures hang off. this->SetShaderParameters(ren, act); - if (!this->ShaderProgram) + if (!this->ShaderProgram) { + observeStage(observe, DrawStage::DrawEnd); return; // compile/link failure: VTK's agents would dereference null here - bool skip = fast; - if (skip && !coincidentSkipSafe(act)) { - skip = false; - ++m_stats.coincidentFallbacks; } + if (fast && !skipSafe) + ++m_stats.coincidentFallbacks; + const bool skip = fast && skipSafe; for (int t = 0; t < 4; ++t) { // verts, lines, polys, strips: VTK's order if (skip && !renderable(t)) { ++m_stats.cellTypesSkipped; + observeStage(observe, DrawStage::AgentSkipped, t); continue; } + observeStage(observe, DrawStage::AgentBegin, t); agentPreDraw(t, ren, act); agentDraw(t, ren, act); // vtkOpenGLLowMemoryCellTypeAgent::PostDraw; every 9.5.0 PostDrawInternal is empty. this->vtkDrawTexturedElements::PostDraw(ren, act, this); + observeStage(observe, DrawStage::AgentEnd, t); } + observeStage(observe, DrawStage::DrawEnd); } void LowMemoryPolyDataMapper::readyProgram(vtkRenderer *ren, bool allowSkip) { + const StageObserver observe = g_stageObserver.load(std::memory_order_relaxed); auto *win = vtkOpenGLRenderWindow::SafeDownCast(ren->GetRenderWindow()); vtkOpenGLShaderCache *cache = win ? win->GetShaderCache() : nullptr; const vtkMTimeType stamp = this->ShaderBuildTimeStamp.GetMTime(); // this->Shaders only changes in UpdateShaders, which RenderPieceStart follows - // with ShaderBuildTimeStamp.Modified(); while the stamp and the cache are the - // ones the last lookup saw, that lookup would find the same program again. - // ReadyShaderProgram(program) still compiles it if its context was released - // and binds it exactly as the lookup's tail would. + // with ShaderBuildTimeStamp.Modified() (GetShader() and + // invalidateProgramCache() clear m_cache for any other route); while the + // stamp and the cache are the ones the last lookup saw, that lookup would + // find the same program again. ReadyShaderProgram(program) still compiles it + // if its context was released and binds it exactly as the lookup's tail + // would, so the two issue the same GL calls. if (allowSkip && cache && this->ShaderProgram && cache == m_cache.GetPointer() && stamp == m_builtStamp && this->ElementType != AbstractPatches) { + observeStage(observe, DrawStage::RebindBegin); this->ShaderProgram = cache->ReadyShaderProgram(this->ShaderProgram.GetPointer()); + observeStage(observe, DrawStage::RebindEnd); ++m_stats.programLookupsSkipped; return; } + observeStage(observe, DrawStage::LookupBegin); this->vtkDrawTexturedElements::ReadyShaderProgram(ren); // copy, substitute, MD5, find, bind + observeStage(observe, DrawStage::LookupEnd); ++m_stats.programLookups; m_cache = cache; m_builtStamp = stamp; @@ -340,6 +465,7 @@ void LowMemoryPolyDataMapper::agentDraw(int t, vtkRenderer *ren, vtkActor *act) program->SetUniformi("usesEdgeValues", group.UsesEdgeValueBuffer); this->vtkDrawTexturedElements::DrawInstancedElementsImpl(ren, act, this); } +// ---- END VTK 9.5.0-DERIVED -------------------------------------------------- #endif // CVC_GL_LOWMEM_REPLICA diff --git a/src/cvcGL/StreamingMappers.cpp b/src/cvcGL/StreamingMappers.cpp index 72a03f27a..dce607fd2 100644 --- a/src/cvcGL/StreamingMappers.cpp +++ b/src/cvcGL/StreamingMappers.cpp @@ -36,7 +36,6 @@ #include #include #include -#include #include #include #include @@ -209,14 +208,9 @@ void StreamingLowMemoryPolyDataMapper::ComputeBounds() { void StreamingLowMemoryPolyDataMapper::RenderPieceStart(vtkRenderer *ren, vtkActor *act) { if (Core.beforeDraw) Core.beforeDraw(ren); - // VTK 9.5's low-memory mapper never looks at the actor's shader property once - // its program is built: IsShaderUpToDate ignores it, where the classic mapper - // compares GetShaderMTime(). A replacement added or cleared after the first - // draw (GeometryNode::add*ShaderReplacement / clearShaderReplacements) would - // never reach the GPU. Dropping the program makes the base rebuild it. - if (vtkShaderProperty *sp = act->GetShaderProperty()) - if (sp->GetShaderMTime() > this->ShaderBuildTimeStamp.GetMTime()) - this->ShaderProgram = nullptr; + // (The base, LowMemoryPolyDataMapper::RenderPieceStart, rebuilds the program + // when the actor's shader property changed after it was built -- VTK's + // low-memory mapper ignores that property once its program exists.) // IsUpToDate() false => the base deletes EVERY array texture and re-binds them // all (the per-frame churn this mapper exists to avoid). That only happens on // a real input/topology/shader change; streaming writes never touch the From 8b8fe1bfc48207bee9fe3d40e3f6969a2835dc7a Mon Sep 17 00:00:00 2001 From: Joe Rivera Date: Thu, 1 Oct 2026 19:05:41 -0500 Subject: [PATCH 05/18] test(cvcGL): pin LowMemoryPolyDataMapper by ordered GL trace; time the bench without wrappers gl_call_counter now wraps every GL entry point VTK 9.5.0's Rendering/OpenGL2, Rendering/VolumeOpenGL2 and Rendering/Core and cvcGL call (a grep of their sources against glad: glVertexAttribDivisor, glBindBufferBase, queries, renderbuffers, glMapBuffer*, glUniform2iv ...) plus the GLES3/GL4 core families they may grow into (glUniform*ui*, the full glUniform*v / glUniformMatrix* set, glClearBuffer*, glTexStorage*, glColorMaski, glBindBufferRange ...), so frame totals are complete. Its "sync" column is now "desktop round-trips (upper bound)": a desktop classification that Firefox partly answers client-side, not a WebGL census. While a trace is open it records every call in order with the bound program (uniforms, draws) or active texture unit (texture state), its argument bytes and the uniform values it sends. glcallsUninstall() restores VTK's pointers. cvcgl_lowmem_fastdraw: - asserts the AllCellTypes frame's trace equals Stock's exactly (bake and steady frames, shadows on and off, the hardware selector's cell and point picking), and that Fast's equals it with the removed work DERIVED from the AllCellTypes trace: the replica marks each draw's stages, and every cell-type block that issued no draw call goes, in every draw where the depth-offset rules allow skipping. A skipped shader-cache lookup removes no GL call (its GL calls are the bind tail the re-bind issues too); the test checks the lookups are accounted for and the stats show the CPU saving. Two desktop-only refinements, both documented in the test: glPointSize/glLineWidth go through vtkOpenGLState's cache, so they are compared as the state each points/lines draw sees; and float values are recorded with -0.0 as +0.0 (VTK's camera key-matrix cache hands out either sign for a zero normal-matrix entry depending on earlier renders -- two Stock selections in a row differ that way). - a Stock-twice control proves the trace is reproducible. - the scene now covers VTK_WIREFRAME and VTK_POINTS representations of a triangle mesh, edge visibility (one that must fall back to all four cell types, one that skips), cell colours on quads (cell map), point colours, texture coordinates, translucency, flat / lit / wide lines, lines as points, a QUADS GeometryNode, the shadow bake and the hardware selector. - GetShader() and invalidateProgramCache() force exactly one full lookup. - a plain GeometryNode's shader replacement added after its first draw, and cleared again, reaches the GPU on every draw path (forced low-memory mapper on desktop; fails without the RenderPieceStart fix). - CVCGL_LOWMEM_DUMP= writes the traces of a failing comparison. cvcgl_lowmem_bench: every configuration runs a timed pass with the wrappers uninstalled (each mapper draw only reads the clock) and a counted pass for the call counts, so the uniform-redundancy bookkeeping no longer inflates Stock's CPU time. Steady frame, shadows on: draw stage 1.65 -> 0.41 ms, frame 2.33 -> 1.05 ms (was reported 1.93 -> 0.49 / 2.67 -> 1.19 with the wrappers in the timed loop); GL calls unchanged at 2532 -> 831. --- src/cvcGL/test/cvcgl_lowmem_bench.cpp | 148 +++-- src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp | 666 +++++++++++++++++++--- src/cvcGL/test/gl_call_counter.cpp | 672 +++++++++++++++++------ src/cvcGL/test/gl_call_counter.h | 64 ++- 4 files changed, 1243 insertions(+), 307 deletions(-) diff --git a/src/cvcGL/test/cvcgl_lowmem_bench.cpp b/src/cvcGL/test/cvcgl_lowmem_bench.cpp index a358a73a7..e15c1beb2 100644 --- a/src/cvcGL/test/cvcgl_lowmem_bench.cpp +++ b/src/cvcGL/test/cvcgl_lowmem_bench.cpp @@ -17,13 +17,21 @@ // line trails, and optionally many extra small single-colour nodes; shadows on // (camera -> strided shadow baker -> shadow map -> translucent -> volumetric -> // overlay) and off. Each node's mapper is swapped for one that tallies the GL -// calls and CPU time of its own draws, so "per draw" is exactly +// calls or the CPU time of its own draws, so "per draw" is exactly // RenderPieceDraw. Frames are rendered without glFinish; times are CPU. // +// Every configuration runs twice per path: a TIMED pass with VTK's own GL +// function pointers (no counting wrapper on any call; each mapper draw only +// reads the clock) and a COUNTED pass with the wrappers installed, which also +// track uniform redundancy per (program, location) -- that costs CPU in +// proportion to the calls, so its times are not reported. Times come from the +// timed pass, call counts from the counted one. +// // cvcgl_lowmem_bench [--frames=40] [--boxes=20000] [--vehicles=6] [--extra=0] // [--size=960x540] #include "gl_call_counter.h" +#include #include #include #include @@ -62,7 +70,9 @@ double msSince(clk::time_point t0) { return std::chrono::duration(clk::now() - t0).count(); } -// Tallies the GL calls and CPU time of every draw of every TallyMapper. +// Tallies the GL calls (counted pass) or the CPU time (timed pass) of every +// draw of every TallyMapper. +bool g_counting = false; GLCalls g_drawCalls; double g_drawMs = 0, g_draws = 0; @@ -71,12 +81,16 @@ class TallyMapper : public LowMemoryPolyDataMapper { static TallyMapper *New(); vtkTypeMacro(TallyMapper, LowMemoryPolyDataMapper); void RenderPieceDraw(vtkRenderer *ren, vtkActor *act) override { - const GLCalls before = glcallsRead(); + g_draws += 1; + if (g_counting) { + const GLCalls before = glcallsRead(); + Superclass::RenderPieceDraw(ren, act); + g_drawCalls += glcallsRead() - before; + return; + } const auto t0 = clk::now(); Superclass::RenderPieceDraw(ren, act); g_drawMs += msSince(t0); - g_drawCalls += glcallsRead() - before; - g_draws += 1; } protected: @@ -256,10 +270,23 @@ void poseVehicles(Bench &b, int frame) { } } +// One pass: timed (frameMs, drawMs) or counted (frame, draw). struct Run { double frames = 0, frameMs = 0, drawMs = 0, draws = 0; GLCalls frame, draw; LowMemoryPolyDataMapper::Stats stats; + Run &operator+=(const Run &o) { + frames += o.frames; + frameMs += o.frameMs; + drawMs += o.drawMs; + draws += o.draws; + frame += o.frame; + draw += o.draw; + stats.programLookups += o.stats.programLookups; + stats.programLookupsSkipped += o.stats.programLookupsSkipped; + stats.stockDraws += o.stats.stockDraws; + return *this; + } }; LowMemoryPolyDataMapper::Stats sumStats(Bench &b) { @@ -276,8 +303,21 @@ LowMemoryPolyDataMapper::Stats sumStats(Bench &b) { return s; } -Run measure(Bench &b, DrawPath path, bool bake, int frames, int &frameNo) { +Run measure(Bench &b, DrawPath path, bool bake, int frames, int &frameNo, bool counted) { LowMemoryPolyDataMapper::setDrawPath(path); + if (counted) + cvcgl_test::glcallsInstall(); + else + cvcgl_test::glcallsUninstall(); // VTK's own pointers: no wrapper cost in the timing + g_counting = counted; + if (counted) { + // One frame on this path first: the first frame after a path switch differs + // by a call or two of vtkOpenGLState-cached state (glPointSize). + poseVehicles(b, frameNo++); + if (bake) + b.sg.invalidateShadowBake(); + b.sr->render(); + } Run r; for (auto &n : b.nodes) if (auto *m = LowMemoryPolyDataMapper::SafeDownCast(n->actor()->GetMapper())) @@ -289,17 +329,22 @@ Run measure(Bench &b, DrawPath path, bool bake, int frames, int &frameNo) { g_drawCalls = GLCalls(); g_drawMs = 0; g_draws = 0; - const GLCalls before = glcallsRead(); - const auto t0 = clk::now(); - b.sr->render(); - r.frameMs += msSince(t0); - r.frame += glcallsRead() - before; - r.draw += g_drawCalls; - r.drawMs += g_drawMs; + if (counted) { + const GLCalls before = glcallsRead(); + b.sr->render(); + r.frame += glcallsRead() - before; + r.draw += g_drawCalls; + } else { + const auto t0 = clk::now(); + b.sr->render(); + r.frameMs += msSince(t0); + r.drawMs += g_drawMs; + } r.draws += g_draws; r.frames += 1; } r.stats = sumStats(b); + g_counting = false; return r; } @@ -311,31 +356,36 @@ std::vector rgba(Bench &b) { return std::vector(p, p + px->GetNumberOfValues()); } -void report(const char *label, const Run &s, const Run &f) { - const double n = s.frames; - std::printf("\n== %s (%g frames each)\n", label, n); - std::printf(" draws/frame %8.1f %8.1f\n", s.draws / n, f.draws / n); +// st/ft: the timed passes; sc/fc: the counted passes (Stock / Fast). +void report(const char *label, const Run &st, const Run &ft, const Run &sc, const Run &fc) { + const double n = st.frames, nc = sc.frames; + std::printf("\n== %s (%g timed + %g counted frames per path)\n", label, n, nc); std::printf(" Stock Fast\n"); - std::printf(" GL calls/frame %8.1f %8.1f (%+.1f, %+.0f%%)\n", s.frame.total() / n, - f.frame.total() / n, (f.frame.total() - s.frame.total()) / n, - 100.0 * (f.frame.total() - s.frame.total()) / s.frame.total()); - std::printf(" in mapper draws %8.1f %8.1f\n", s.draw.total() / n, f.draw.total() / n); + std::printf(" draws/frame %8.1f %8.1f\n", st.draws / n, ft.draws / n); + std::printf(" GL calls/frame %8.1f %8.1f (%+.1f, %+.0f%%)\n", sc.frame.total() / nc, + fc.frame.total() / nc, (fc.frame.total() - sc.frame.total()) / nc, + 100.0 * (fc.frame.total() - sc.frame.total()) / sc.frame.total()); + std::printf(" in mapper draws %8.1f %8.1f\n", sc.draw.total() / nc, + fc.draw.total() / nc); std::printf(" uniforms %8.1f %8.1f (redundant %.1f -> %.1f)\n", - s.frame.uniforms() / n, f.frame.uniforms() / n, s.frame.uniformRedundant / n, - f.frame.uniformRedundant / n); - std::printf(" sync queries %8.1f %8.1f\n", s.frame.sync() / n, f.frame.sync() / n); - std::printf(" GL calls/draw %8.1f %8.1f\n", s.draw.total() / s.draws, - f.draw.total() / f.draws); - std::printf(" draw-stage CPU ms/frame %6.3f %8.3f\n", s.drawMs / n, f.drawMs / n); - std::printf(" draw-stage CPU ms/draw %6.4f %8.4f\n", s.drawMs / s.draws, f.drawMs / f.draws); - std::printf(" frame CPU ms (no finish)%6.2f %8.2f\n", s.frameMs / n, f.frameMs / n); + sc.frame.uniforms() / nc, fc.frame.uniforms() / nc, sc.frame.uniformRedundant / nc, + fc.frame.uniformRedundant / nc); + std::printf(" desktop round-trips (upper bound) %8.1f %8.1f\n", sc.frame.roundTrips() / nc, + fc.frame.roundTrips() / nc); + std::printf(" GL calls/draw %8.1f %8.1f\n", sc.draw.total() / sc.draws, + fc.draw.total() / fc.draws); + std::printf(" timed, no GL wrappers:\n"); + std::printf(" draw-stage CPU ms/frame %6.3f %8.3f\n", st.drawMs / n, ft.drawMs / n); + std::printf(" draw-stage CPU ms/draw %6.4f %8.4f\n", st.drawMs / st.draws, + ft.drawMs / ft.draws); + std::printf(" frame CPU ms (no finish)%6.2f %8.2f\n", st.frameMs / n, ft.frameMs / n); std::printf(" lookups/frame %8.1f %8.1f (skipped %.1f)\n", - double(s.stats.programLookups + s.stats.stockDraws) / n, - double(f.stats.programLookups) / n, double(f.stats.programLookupsSkipped) / n); - std::printf(" Stock per frame: %s\n", s.frame.str(n).c_str()); - std::printf(" Fast per frame: %s\n", f.frame.str(n).c_str()); - std::printf(" Stock per draw : %s\n", s.draw.str(s.draws).c_str()); - std::printf(" Fast per draw : %s\n", f.draw.str(f.draws).c_str()); + double(st.stats.programLookups + st.stats.stockDraws) / n, + double(ft.stats.programLookups) / n, double(ft.stats.programLookupsSkipped) / n); + std::printf(" Stock per frame: %s\n", sc.frame.str(nc).c_str()); + std::printf(" Fast per frame: %s\n", fc.frame.str(nc).c_str()); + std::printf(" Stock per draw : %s\n", sc.draw.str(sc.draws).c_str()); + std::printf(" Fast per draw : %s\n", fc.draw.str(fc.draws).c_str()); } } // namespace @@ -394,29 +444,23 @@ int main(int argc, char **argv) { if (!shadows) b.sg.setShadowsEnabled(false); for (DrawPath p : {DrawPath::Stock, DrawPath::Fast}) // warm both paths - measure(b, p, shadows != 0, 2, frameNo); + measure(b, p, shadows != 0, 2, frameNo, false); const auto kinds = shadows ? std::vector{true, false} : std::vector{false}; for (bool bake : kinds) { - // Interleave the two paths in halves so drift hits both alike. - Run s, f; + // Timed: interleave the two paths in halves so drift hits both alike. + Run st, ft, sc, fc; for (int half = 0; half < 2; ++half) { - const Run s1 = measure(b, DrawPath::Stock, bake, o.frames / 2, frameNo); - const Run f1 = measure(b, DrawPath::Fast, bake, o.frames / 2, frameNo); - for (auto [dst, src] : {std::make_pair(&s, &s1), std::make_pair(&f, &f1)}) { - dst->frames += src->frames; - dst->frameMs += src->frameMs; - dst->drawMs += src->drawMs; - dst->draws += src->draws; - dst->frame += src->frame; - dst->draw += src->draw; - dst->stats.programLookups += src->stats.programLookups; - dst->stats.programLookupsSkipped += src->stats.programLookupsSkipped; - dst->stats.stockDraws += src->stats.stockDraws; - } + st += measure(b, DrawPath::Stock, bake, o.frames / 2, frameNo, false); + ft += measure(b, DrawPath::Fast, bake, o.frames / 2, frameNo, false); } + // Counted: GL call counts are the same every frame of a kind; a few do. + const int countedFrames = std::max(2, o.frames / 8); + sc = measure(b, DrawPath::Stock, bake, countedFrames, frameNo, true); + fc = measure(b, DrawPath::Fast, bake, countedFrames, frameNo, true); + cvcgl_test::glcallsUninstall(); report(shadows ? (bake ? "shadows ON, bake every frame" : "shadows ON, steady (no bake)") : "shadows OFF", - s, f); + st, ft, sc, fc); // Same pose on both paths, then compare the frames. LowMemoryPolyDataMapper::setDrawPath(DrawPath::Stock); poseVehicles(b, 7); diff --git a/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp b/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp index dc1258371..b71aa8da7 100644 --- a/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp +++ b/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp @@ -15,21 +15,36 @@ // SceneGraph with shadows on (camera -> strided shadow baker -> shadow map -> // translucent -> volumetric -> overlay) and off -- over GeometryNodes of every // kind a scene uses (lit single colour, textured ground, translucent with a -// depth offset, lines, points, per-vertex colour) plus raw VTK actors whose -// mapper counts the GL calls of its own draws. Checks: +// depth offset, wireframe lines, points, per-vertex colour, quads) plus raw VTK +// actors whose mapper counts the GL calls of its own draws: surface, wireframe +// and points representations of a triangle mesh, edge visibility (with and +// without its own offset), cell colours on quads, point colours, texture +// coordinates, translucency, verts, strips, and lines flat / lit / wide / as +// points. Checks: // 1. the policy: CVCGL_LOWMEM_MAPPER is read, Force/Off/Auto pick the mapper; -// 2. the replica: DrawPath::AllCellTypes issues exactly VTK's GL calls (every -// entry point, whole frame, bake and plain frames, shadows on and off); -// 3. pixels: Stock, AllCellTypes and Fast frames are byte-identical (RGBA); -// 4. the saving: Fast draws a single-cell-type mesh in <= 2 VAO binds and a -// fraction of the stock calls, with no synchronous GL query in any draw, +// 2. the replica: the ordered GL trace of a DrawPath::AllCellTypes frame -- +// every call, its bound program or texture unit, its argument bytes and +// the uniform values it sends -- equals the Stock frame's exactly (bake +// and plain frames, shadows on and off, and the hardware selector's cell +// and point picking passes); +// 3. Fast's trace equals that same trace with what Fast leaves out removed, +// DERIVED from the AllCellTypes trace (the replica marks where each draw's +// stages begin and end): every cell-type block that issued no draw call, +// in every draw where skipping is allowed. A skipped shader-cache lookup +// removes no GL call -- the lookup's GL calls are the bind tail the +// re-bind issues too; what it saves is CPU, which the stats check; +// 4. pixels: Stock, AllCellTypes and Fast frames are byte-identical (RGBA); +// 5. the saving: Fast draws a single-cell-type mesh in <= 2 VAO binds and a +// fraction of the stock calls, with no desktop round-trip in any draw, // and a steady frame does no shader-cache lookup; -// 5. lifecycle: a shader rebuild (the shadow bake's pass change), a mapper's -// released resources, programs released in the shader cache and the scene -// re-opened in a new window all go back through the full lookup (or a -// recompile) and still match; -// 6. guards: an offset layout where skipping would change a drawn type's -// depth offset, picking and vertex visibility all draw the stock way. +// 6. lifecycle: a shader rebuild (the shadow bake's pass change), a mapper's +// released resources, GetShader(), programs released in the shader cache +// and the scene re-opened in a new window all go back through the full +// lookup (or a recompile) and still match; +// 7. guards: an offset layout where skipping would change a drawn type's +// depth offset, picking and vertex visibility all draw the stock way; +// 8. a plain GeometryNode's shader replacement added (and cleared) after its +// first draw reaches the GPU, on every draw path. // Frames are rendered without MSAA: with VTK's default 8x, NVIDIA's resolve // varies by 1 LSB at a few edge pixels between identical GL streams. // Renders for real; skips (rc 0) where nothing rasterises unless @@ -42,6 +57,7 @@ #include #include #include +#include #include #include #include @@ -56,6 +72,7 @@ #include #include #include +#include #include #include #include @@ -74,9 +91,11 @@ #include #include #include +#include #include #include #include +#include // GL_POINTS ... (desktop-only test) using cvc::gl::GeometryNode; using cvc::gl::GeometryRenderMode; @@ -87,7 +106,11 @@ using cvc::gl::SceneRenderer; using cvcgl_test::GLCalls; using cvcgl_test::glcallsInstall; using cvcgl_test::glcallsRead; +using cvcgl_test::kTraceMarkBase; +using cvcgl_test::Trace; +using cvcgl_test::TraceRec; using DrawPath = LowMemoryPolyDataMapper::DrawPath; +using DrawStage = LowMemoryPolyDataMapper::DrawStage; namespace { @@ -135,6 +158,143 @@ class ErrorCounter : public vtkOutputWindow { }; vtkStandardNewMacro(ErrorCounter); +// ── GL traces ──────────────────────────────────────────────────────────────── +// The replica's stage observer: each stage boundary becomes a trace marker. +void onStage(DrawStage stage, int arg) { cvcgl_test::glcallsMark(static_cast(stage), arg); } + +bool isStage(const TraceRec &r, DrawStage stage) { + return r.marker() && r.entry - kTraceMarkBase == static_cast(stage); +} + +int countStage(const Trace &t, DrawStage stage) { + int n = 0; + for (const auto &r : t) + n += isStage(r, stage); + return n; +} + +// The GL calls alone. +Trace glOnly(const Trace &t) { + Trace out; + out.reserve(t.size()); + for (const auto &r : t) + if (!r.marker()) + out.push_back(r); + return out; +} + +// Equal GL traces (every call, in order, with its context and bytes)? If not, +// where they first part. +bool sameTrace(const Trace &a, const Trace &b, std::string *why) { + const size_t n = std::min(a.size(), b.size()); + size_t i = 0; + while (i < n && a[i] == b[i]) + ++i; + if (i == n && a.size() == b.size()) + return true; + if (why) { + *why = fmt("%.0f vs %.0f calls, first difference at #%.0f: ", double(a.size()), + double(b.size()), double(i)); + *why += i < a.size() ? cvcgl_test::glcallsDescribe(a[i]) : std::string("(end)"); + *why += " vs "; + *why += i < b.size() ? cvcgl_test::glcallsDescribe(b[i]) : std::string("(end)"); + } + return false; +} + +// vtkOpenGLState caches glPointSize and glLineWidth (desktop GL only: GLES 3 / +// WebGL have neither call). Whether a kept block's vtkglPointSize reaches GL +// therefore depends on what the blocks before it set, so leaving out an empty +// block can add or drop one of these calls without changing what any draw +// sees. Compare them as that: drop the calls, and append to every GL_POINTS +// draw the point size it draws with and to every line draw the line width +// (tracked from the trace's initial-state markers and its calls). +Trace asDrawState(const Trace &t) { + static const int pointSize = cvcgl_test::glcallsEntry("PointSize"); + static const int lineWidth = cvcgl_test::glcallsEntry("LineWidth"); + static const int dispatch = cvcgl_test::glcallsEntry("DispatchCompute"); + std::string ps(4, '\0'), lw(4, '\0'); + Trace out; + out.reserve(t.size()); + for (const auto &r : t) { + if (r.marker()) { + if (r.entry == kTraceMarkBase + cvcgl_test::kTraceInitPointSize) + ps = r.args; + else if (r.entry == kTraceMarkBase + cvcgl_test::kTraceInitLineWidth) + lw = r.args; + else + out.push_back(r); + continue; + } + if (r.entry == pointSize) { + ps = r.args; + continue; + } + if (r.entry == lineWidth) { + lw = r.args; + continue; + } + out.push_back(r); + if (cvcgl_test::glcallsIsDraw(r.entry) && r.entry != dispatch && r.args.size() >= 4) { + GLenum mode = 0; + std::memcpy(&mode, r.args.data(), sizeof mode); + if (mode == GL_POINTS) + out.back().args += ps; + else if (mode == GL_LINES || mode == GL_LINE_STRIP || mode == GL_LINE_LOOP) + out.back().args += lw; + } + } + return out; +} + +// What Fast's GL trace must be, derived from an AllCellTypes trace and its +// stage markers. In every draw where skipping is allowed (DrawBegin arg 1), +// each cell-type block (AgentBegin..AgentEnd) that issued no draw call is +// removed. Everything else stays -- including each full lookup's GL calls, +// which are the ReadyShaderProgram(program) tail (compile if released, bind) +// that the re-bind replacing it issues too. +struct Expected { + Trace gl; + int blocksRemoved = 0, blocksKept = 0, lookups = 0, draws = 0, unskippableDraws = 0; +}; + +Expected expectFast(const Trace &all) { + Expected e; + bool skipAllowed = false, inBlock = false, blockDraws = false; + Trace block; + for (const auto &r : all) { + if (!r.marker()) { + if (inBlock) { + block.push_back(r); + blockDraws = blockDraws || cvcgl_test::glcallsIsDraw(r.entry); + } else { + e.gl.push_back(r); + } + continue; + } + if (isStage(r, DrawStage::DrawBegin)) { + ++e.draws; + skipAllowed = r.ctx != 0; + e.unskippableDraws += !skipAllowed; + } else if (isStage(r, DrawStage::LookupBegin)) { + ++e.lookups; + } else if (isStage(r, DrawStage::AgentBegin)) { + inBlock = true; + blockDraws = false; + block.clear(); + } else if (isStage(r, DrawStage::AgentEnd)) { + inBlock = false; + if (skipAllowed && !blockDraws) { + ++e.blocksRemoved; + } else { + ++e.blocksKept; + e.gl.insert(e.gl.end(), block.begin(), block.end()); + } + } + } + return e; +} + // The production mapper plus a GL-call tally of its own RenderPieceDraw calls. class ProbeMapper : public LowMemoryPolyDataMapper { public: @@ -370,6 +530,63 @@ vtkSmartPointer patchPoly(char kind, double x0, double y0, double z return pd; } +// An n x n patch of quads (4-point polygons: VTK triangulates them, so the +// draw needs the cell map) with one RGB colour per quad. +vtkSmartPointer quadsPoly(int n, double x0, double y0, double z) { + auto pts = vtkSmartPointer::New(); + pts->SetDataTypeToFloat(); + for (int j = 0; j <= n; ++j) + for (int i = 0; i <= n; ++i) + pts->InsertNextPoint(x0 + 0.8 * i, y0 + 0.8 * j, z + 0.15 * ((i + j) % 2)); + auto quads = vtkSmartPointer::New(); + auto colours = vtkSmartPointer::New(); + colours->SetNumberOfComponents(3); + colours->SetName("quadColours"); + for (int j = 0; j < n; ++j) + for (int i = 0; i < n; ++i) { + const vtkIdType a = j * (n + 1) + i, ids[4] = {a, a + 1, a + n + 2, a + n + 1}; + quads->InsertNextCell(4, ids); + const unsigned char c[3] = {static_cast(40 + 200 * i / n), + static_cast(40 + 200 * j / n), + static_cast((i + j) % 2 ? 220 : 60)}; + colours->InsertNextTypedTuple(c); + } + auto pd = vtkSmartPointer::New(); + pd->SetPoints(pts); + pd->SetPolys(quads); + pd->GetCellData()->SetScalars(colours); + return withNormals(pd); +} + +// One RGB colour per point. +vtkSmartPointer withPointColours(vtkSmartPointer pd) { + auto colours = vtkSmartPointer::New(); + colours->SetNumberOfComponents(3); + colours->SetName("pointColours"); + for (vtkIdType k = 0; k < pd->GetNumberOfPoints(); ++k) { + const unsigned char c[3] = {static_cast(37 * k % 256), + static_cast(91 * k % 256), + static_cast(151 * k % 256)}; + colours->InsertNextTypedTuple(c); + } + pd->GetPointData()->SetScalars(colours); + return pd; +} + +// Lines with a point normal each (VTK lights lines only with normals and +// non-flat interpolation). +vtkSmartPointer withLineNormals(vtkSmartPointer pd) { + auto normals = vtkSmartPointer::New(); + normals->SetNumberOfComponents(3); + normals->SetName("Normals"); + for (vtkIdType k = 0; k < pd->GetNumberOfPoints(); ++k) { + const double a = 0.3 * k; + normals->InsertNextTuple3(0.3 * std::cos(a), 0.3 * std::sin(a), 0.9); + } + pd->GetPointData()->SetNormals(normals); + return pd; +} + vtkSmartPointer checkerTexture(int n) { auto img = vtkSmartPointer::New(); img->SetDimensions(n, n, 1); @@ -394,6 +611,9 @@ struct Probe { std::string name; vtkSmartPointer mapper; vtkSmartPointer actor; + // Expected to take the coincident-offset fallback (all four cell types) on + // the Fast path. + bool fallback = false; }; struct Scene { @@ -522,6 +742,23 @@ void buildCity(Scene &s) { vcol->setUseSingleColor(false); vcol->setGeometry(colouredGeometry(-4, -3)); + auto quads = s.node("quads"); + { + cvc::geometry g; + for (int j = 0; j < 4; ++j) + for (int i = 0; i < 4; ++i) + g.points().push_back({1.0 + 0.9 * i, 3.0 + 0.9 * j, 0.4 + 0.1 * ((i + j) % 2)}); + for (unsigned j = 0; j + 1 < 4; ++j) + for (unsigned i = 0; i + 1 < 4; ++i) { + const unsigned a = j * 4 + i; + g.quads().push_back({a, a + 1, a + 5, a + 4}); + } + quads->setGeometry(g); + } + quads->setRenderMode(GeometryRenderMode::QUADS); + quads->setUseSingleColor(true); + quads->setColor(0.6, 0.4, 0.95); + // Probes: the same kinds as raw actors whose draws are tallied. { Probe &p = s.probe("lit", spherePoly(1.4)); @@ -558,6 +795,74 @@ void buildCity(Scene &s) { Probe &p = s.probe("strips", patchPoly('s', 2, 9, 0.6)); p.actor->GetProperty()->SetColor(0.5, 0.9, 0.6); } + // Representations of a triangle mesh other than the surface. + { + Probe &p = s.probe("wireframe", spherePoly(1.3)); + p.actor->GetProperty()->SetRepresentationToWireframe(); + p.actor->GetProperty()->SetColor(0.95, 0.6, 0.2); + p.actor->SetPosition(-8, 0, 2); + } + { + Probe &p = s.probe("pointsRep", spherePoly(1.3)); // lit: point normals, Gouraud + p.actor->GetProperty()->SetRepresentationToPoints(); + p.actor->GetProperty()->SetPointSize(3.0); + p.actor->GetProperty()->SetColor(0.3, 0.95, 0.95); + p.actor->SetPosition(8, -4, 2); + } + // Edge visibility on a surface puts every draw under the coincident-offset + // rules: with VTK's defaults the polys draw sees the -4 units the empty lines + // block left in the program, so Fast must fall back to all four ... + { + Probe &p = s.probe("edges", spherePoly(1.2)); + p.actor->GetProperty()->EdgeVisibilityOn(); + p.actor->GetProperty()->SetEdgeColor(0.1, 0.1, 0.1); + p.actor->GetProperty()->SetColor(0.8, 0.8, 0.5); + p.actor->SetPosition(0, 6, 2); + p.fallback = true; + } + // ... while one whose own polygon offset is non-zero skips them. + { + Probe &p = s.probe("edgesOffset", ribbonPoly(10, -7, -12, 0.5)); + p.actor->GetProperty()->EdgeVisibilityOn(); + p.actor->GetProperty()->SetEdgeColor(0.9, 0.1, 0.1); + p.actor->GetProperty()->SetColor(0.4, 0.5, 0.9); + p.mapper->SetRelativeCoincidentTopologyPolygonOffsetParameters(0.0, -2.0); + p.mapper->SetRelativeCoincidentTopologyLineOffsetParameters(0.0, -2.0); + p.mapper->SetRelativeCoincidentTopologyPointOffsetParameter(-2.0); + } + // Scalars: cell colours on quads (cell map), point colours on a sphere. + { + Probe &p = s.probe("cellColours", quadsPoly(5, -3, -9, 0.3)); + p.mapper->ScalarVisibilityOn(); + p.mapper->SetScalarModeToUseCellData(); + p.mapper->SetColorModeToDirectScalars(); + } + { + Probe &p = s.probe("pointColours", withPointColours(spherePoly(1.1))); + p.mapper->ScalarVisibilityOn(); + p.mapper->SetScalarModeToUsePointData(); + p.mapper->SetColorModeToDirectScalars(); + p.actor->SetPosition(-3, 3, 3); + } + // Lines: flat with normals (unlit), Gouraud with normals and wide (lit, + // instanced), and drawn as points. + { + Probe &p = s.probe("flatLines", withLineNormals(linesPoly(20, -9, 8, 1.0))); + p.actor->GetProperty()->SetInterpolationToFlat(); + p.actor->GetProperty()->SetColor(0.9, 0.3, 0.3); + } + { + Probe &p = s.probe("litLines", withLineNormals(linesPoly(20, -9, 10, 1.2))); + p.actor->GetProperty()->SetInterpolationToGouraud(); + p.actor->GetProperty()->SetLineWidth(3.0); + p.actor->GetProperty()->SetColor(0.3, 0.3, 0.9); + } + { + Probe &p = s.probe("linePoints", linesPoly(20, 0, -6, 1.4)); + p.actor->GetProperty()->SetRepresentationToPoints(); + p.actor->GetProperty()->SetPointSize(4.0); + p.actor->GetProperty()->SetColor(0.95, 0.95, 0.95); + } } void addLights(SceneGraph &sg) { @@ -567,6 +872,7 @@ void addLights(SceneGraph &sg) { struct PathRun { GLCalls first, steady; // whole frame + Trace firstTrace, steadyTrace; // whole frame, with stage markers std::vector firstPx, steadyPx; // RGBA LowMemoryPolyDataMapper::Stats firstStats, steadyStats; std::vector probeCalls; // steady frame, per probe @@ -591,7 +897,9 @@ PathRun runPath(Scene &s, DrawPath path) { if (vtkActor *a = s.sr->renderer()->GetActors()->GetLastActor()) a->Modified(); s.resetCounters(); + cvcgl_test::glcallsTraceStart(); r.first = s.frame(); + r.firstTrace = cvcgl_test::glcallsTraceStop(); r.firstStats = s.stats(); for (auto &p : s.probes) { r.probeFirst.push_back(p.mapper->calls); @@ -599,7 +907,9 @@ PathRun runPath(Scene &s, DrawPath path) { } r.firstPx = s.rgba(); s.resetCounters(); + cvcgl_test::glcallsTraceStart(); r.steady = s.frame(); + r.steadyTrace = cvcgl_test::glcallsTraceStop(); r.steadyStats = s.stats(); for (auto &p : s.probes) { r.probeCalls.push_back(p.mapper->calls); @@ -681,7 +991,160 @@ bool renderAvailable(cvc::app &app) { return false; } -// ── 2-4. replica, pixels, savings ──────────────────────────────────────────── +// ── 2-5. replica, derived Fast trace, pixels, savings ──────────────────────── +// The replica (AllCellTypes) against Stock, call for call, and Fast against the +// trace derived from it, for one frame of each kind. +// CVCGL_LOWMEM_DUMP=: write the traces of each failing comparison there +// (_.txt, one call per line), for a diff. +void dumpTraces(const std::string &what, const std::vector> &ts) { + static int dumped = 0; + const char *dir = std::getenv("CVCGL_LOWMEM_DUMP"); + if (!dir || !*dir) + return; + ++dumped; + for (const auto &[name, t] : ts) { + const std::string path = std::string(dir) + "/" + std::to_string(dumped) + "_" + name + ".txt"; + if (FILE *f = std::fopen(path.c_str(), "w")) { + std::fprintf(f, "# %s\n", what.c_str()); + for (const auto &r : t) + std::fprintf(f, "%s\n", cvcgl_test::glcallsDescribe(r).c_str()); + std::fclose(f); + } + } +} + +// mayDraw: whether the frame has replica draws at all (a cell-picking trace is +// selection passes only, every one of them the stock draw). +void checkTraces(const std::string &what, const Trace &stock, const Trace &all, const Trace &fast, + const LowMemoryPolyDataMapper::Stats &fastStats, bool mayDraw = true) { + std::string why; + check(countStage(stock, DrawStage::DrawBegin) == 0 && + countStage(all, DrawStage::RebindBegin) == 0 && + countStage(all, DrawStage::AgentSkipped) == 0 && + (countStage(all, DrawStage::DrawBegin) > 0) == mayDraw, + what + ": Stock reports no stages; AllCellTypes looks every program up and runs every " + "cell type"); + // (Each comparison runs before its message is built: argument order is unspecified.) + const bool replica = sameTrace(glOnly(all), glOnly(stock), &why); + if (!replica) + dumpTraces(what, {{"stock", stock}, {"all", all}}); + check(replica, + what + ": AllCellTypes trace == Stock trace (every call, program/unit, bytes, values)", + replica ? fmt("%.0f calls", double(stock.size())) : why); + why.clear(); + // The draw state is read off the full AllCellTypes trace, before the blocks + // go: what each draw saw there is what it must see on Fast. + const Expected e = expectFast(asDrawState(all)); + const Trace got = glOnly(asDrawState(fast)); + const bool derived = sameTrace(got, glOnly(e.gl), &why); + if (!derived) + dumpTraces(what, {{"all", all}, {"fast", fast}, {"expected", e.gl}, {"fast_state", got}}); + check(derived, + what + ": Fast trace == AllCellTypes trace minus its empty cell-type blocks (cached " + "point size / line width compared at the draws)", + derived ? fmt("%.0f calls; %.0f blocks removed, %.0f kept", double(got.size()), + double(e.blocksRemoved), double(e.blocksKept)) + : why); + const int skipped = countStage(fast, DrawStage::AgentSkipped); + check(e.blocksRemoved == skipped && (skipped > 0) == mayDraw && + static_cast(skipped) == double(fastStats.cellTypesSkipped), + what + ": the blocks derived as removable are exactly the ones Fast skipped", + fmt("derived %.0f, Fast skipped %.0f (stats %.0f)", double(e.blocksRemoved), + double(skipped), double(fastStats.cellTypesSkipped))); + const int looked = countStage(fast, DrawStage::LookupBegin); + const int rebound = countStage(fast, DrawStage::RebindBegin); + check(looked + rebound == e.lookups && countStage(fast, DrawStage::DrawBegin) == e.draws && + static_cast(e.unskippableDraws) == double(fastStats.coincidentFallbacks), + what + ": every AllCellTypes lookup is a Fast lookup or re-bind; same draws, same " + "fallbacks", + fmt("lookups %.0f = %.0f + %.0f re-bound", double(e.lookups), double(looked), + double(rebound)) + + fmt(", fallbacks %.0f", double(e.unskippableDraws))); +} + +// The hardware selector's passes on one draw path: what it selected, the GL +// trace of the selection, the stats. +struct Pick { + std::string signature; + Trace trace; + LowMemoryPolyDataMapper::Stats stats; +}; + +Pick pick(Scene &s, DrawPath p, int association) { + LowMemoryPolyDataMapper::setDrawPath(p); + s.sr->render(); // the same state before each path's selection + s.resetCounters(); + vtkNew sel; + sel->SetRenderer(s.sr->renderer()); + sel->SetArea(0, 0, s.w - 1, s.h - 1); + sel->SetFieldAssociation(association); + cvcgl_test::glcallsTraceStart(); + vtkSmartPointer res = vtk::TakeSmartPointer(sel->Select()); + Pick out; + out.trace = cvcgl_test::glcallsTraceStop(); + std::vector> hits; // (prop id, items), in prop-id order + for (unsigned i = 0; res && i < res->GetNumberOfNodes(); ++i) { + vtkSelectionNode *n = res->GetNode(i); + hits.emplace_back(n->GetProperties()->Get(vtkSelectionNode::PROP_ID()), + n->GetSelectionList() ? double(n->GetSelectionList()->GetNumberOfTuples()) + : 0.0); + } + std::sort(hits.begin(), hits.end()); + for (const auto &h : hits) + out.signature += fmt("[%.0f:%.0f]", h.first, h.second); + out.stats = s.stats(); + return out; +} + +// Picking takes the stock draw on every path: identical selections and GL. +// (Point picking first renders one ordinary frame for the depth buffer -- +// vtkOpenGLHardwareSelector::BeginSelection -- with no selector set, so those +// draws take the path under test; the trace check covers them like a frame.) +void comparePicking(Scene &s, const std::string &label) { + // Replica draws in one ordinary frame (vertex visibility is stock anyway). + LowMemoryPolyDataMapper::setDrawPath(DrawPath::Fast); + s.sr->render(); + s.resetCounters(); + s.sr->render(); + const double frameDraws = double(s.stats().draws - s.stats().stockDraws); + for (int assoc : + {vtkDataObject::FIELD_ASSOCIATION_CELLS, vtkDataObject::FIELD_ASSOCIATION_POINTS}) { + const bool points = assoc == vtkDataObject::FIELD_ASSOCIATION_POINTS; + const std::string what = label + (points ? ": point picking" : ": cell picking"); + // Warm: the selection passes' shader variants are compiled by the first + // selection on any path; compare the ones after. + for (DrawPath p : {DrawPath::Stock, DrawPath::Fast}) + pick(s, p, assoc); + const Pick stock = pick(s, DrawPath::Stock, assoc); + const Pick all = pick(s, DrawPath::AllCellTypes, assoc); + const Pick fast = pick(s, DrawPath::Fast, assoc); + const Pick stock2 = pick(s, DrawPath::Stock, assoc); + { + std::string why; + const bool reproducible = sameTrace(glOnly(stock2.trace), glOnly(stock.trace), &why); + if (!reproducible) + dumpTraces(what + " (Stock twice)", {{"stock", stock.trace}, {"stock2", stock2.trace}}); + check(reproducible, what + ": the selection trace is reproducible (Stock twice)", + reproducible ? fmt("%.0f calls", double(stock.trace.size())) : why); + } + const double ordinary = points ? frameDraws : 0.0; + check(double(fast.stats.draws - fast.stats.stockDraws) == ordinary && + double(all.stats.draws - all.stats.stockDraws) == ordinary && + fast.stats.stockDraws > 0, + what + ": every selection-pass draw takes the stock path, on every path", + fmt("stock %.0f of %.0f draws", double(fast.stats.stockDraws), double(fast.stats.draws)) + + fmt(" (+%.0f in the ordinary depth frame)", ordinary)); + const bool sameSelection = !stock.signature.empty() && fast.signature == stock.signature && + all.signature == stock.signature; + check(sameSelection, what + " selects the same on every path", + sameSelection ? stock.signature + : "Stock " + stock.signature + " AllCellTypes " + all.signature + " Fast " + + fast.signature); + checkTraces(what, stock.trace, all.trace, fast.trace, fast.stats, points); + } + LowMemoryPolyDataMapper::setDrawPath(DrawPath::Fast); +} + void comparePaths(Scene &s, const std::string &label, bool shadows) { std::printf("%s\n", label.c_str()); // Settle: build every shader both ways first so no path pays a first compile. @@ -690,14 +1153,18 @@ void comparePaths(Scene &s, const std::string &label, bool shadows) { const PathRun stock = runPath(s, DrawPath::Stock); const PathRun all = runPath(s, DrawPath::AllCellTypes); const PathRun fast = runPath(s, DrawPath::Fast); + const PathRun stock2 = runPath(s, DrawPath::Stock); check(drewSomething(stock.steadyPx), label + ": the scene draws"); - std::string diff; - const bool pinFirst = all.first.sameAs(stock.first, &diff); - check(pinFirst, label + ": AllCellTypes issues VTK's GL calls, first frame", diff); - diff.clear(); - const bool pinSteady = all.steady.sameAs(stock.steady, &diff); - check(pinSteady, label + ": AllCellTypes issues VTK's GL calls, steady frame", diff); + std::string why; + const bool reproducible = sameTrace(stock2.firstTrace, stock.firstTrace, &why) && + sameTrace(stock2.steadyTrace, stock.steadyTrace, &why); + check(reproducible, label + ": the GL trace is reproducible (Stock twice, same trace)", + reproducible ? fmt("%.0f calls", double(stock.steadyTrace.size())) : why); + checkTraces(label + ", first frame" + (shadows ? " (shadow bake)" : ""), stock.firstTrace, + all.firstTrace, fast.firstTrace, fast.firstStats); + checkTraces(label + ", steady frame", stock.steadyTrace, all.steadyTrace, fast.steadyTrace, + fast.steadyStats); samePixels(label + ": AllCellTypes pixels == Stock, first frame", stock.firstPx, all.firstPx); samePixels(label + ": AllCellTypes pixels == Stock, steady frame", stock.steadyPx, all.steadyPx); samePixels(label + ": Fast pixels == Stock, first frame", stock.firstPx, fast.firstPx); @@ -712,18 +1179,24 @@ void comparePaths(Scene &s, const std::string &label, bool shadows) { fmt("steady %.0f -> %.0f, first %.0f", stock.steady.total(), fast.steady.total(), stock.first.total()) + fmt(" -> %.0f", fast.first.total())); - check(fast.steady.sync() == stock.steady.sync(), - label + ": Fast adds no synchronous GL query to the frame", - fmt("sync %.0f vs %.0f", fast.steady.sync(), stock.steady.sync())); + check(fast.steady.roundTrips() == stock.steady.roundTrips(), + label + ": Fast adds no desktop round-trip to the frame", + fmt("%.0f vs %.0f", fast.steady.roundTrips(), stock.steady.roundTrips())); check(fast.steadyStats.programLookups == 0 && fast.steadyStats.programLookupsSkipped > 0, label + ": Fast steady frame re-binds every program without a lookup", fmt("lookups %.0f, skipped %.0f", double(fast.steadyStats.programLookups), double(fast.steadyStats.programLookupsSkipped))); - check(fast.steadyStats.cellTypesSkipped > 0 && fast.steadyStats.coincidentFallbacks == 0 && + // The only fallbacks are the probes built to need one. + double expectFallbacks = 0; + for (size_t i = 0; i < s.probes.size(); ++i) + expectFallbacks += s.probes[i].fallback ? fast.probeDraws[i] : 0.0; + check(fast.steadyStats.cellTypesSkipped > 0 && + double(fast.steadyStats.coincidentFallbacks) == expectFallbacks && fast.steadyStats.stockDraws == 0, - label + ": Fast skipped empty cell types, no fallback", - fmt("skipped %.0f of %.0f draws x 4", double(fast.steadyStats.cellTypesSkipped), - double(fast.steadyStats.draws))); + label + ": Fast skipped empty cell types; falls back only where an offset needs it", + fmt("skipped %.0f of %.0f draws x 4, fallbacks %.0f", + double(fast.steadyStats.cellTypesSkipped), double(fast.steadyStats.draws), + double(fast.steadyStats.coincidentFallbacks))); check(stock.steadyStats.stockDraws == stock.steadyStats.draws && stock.steadyStats.draws > 0, label + ": Stock draws through VTK's own path"); if (shadows) @@ -741,17 +1214,23 @@ void comparePaths(Scene &s, const std::string &label, bool shadows) { } const GLCalls &cs = stock.probeCalls[i]; const GLCalls &cf = fast.probeCalls[i]; - std::printf(" per draw %-11s Stock %s\n", p.name.c_str(), cs.str(nStock).c_str()); - std::printf(" %-20s Fast %s\n", "", cf.str(nFast).c_str()); - check(cs.sync() == 0 && cf.sync() == 0 && all.probeCalls[i].sync() == 0, - label + ": " + p.name + " draws issue no synchronous GL query"); + std::printf(" per draw %-12s Stock %s\n", p.name.c_str(), cs.str(nStock).c_str()); + std::printf(" %-21s Fast %s\n", "", cf.str(nFast).c_str()); + check(cs.roundTrips() == 0 && cf.roundTrips() == 0 && all.probeCalls[i].roundTrips() == 0, + label + ": " + p.name + " draws issue no desktop round-trip"); + check(cf.count("DrawArraysInstanced") == cs.count("DrawArraysInstanced") && + cf.count("DrawArraysInstanced") == nFast, + label + ": " + p.name + " one draw call per draw, as Stock"); + if (p.fallback) { + check(cf.total() / nFast == cs.total() / nStock, + label + ": " + p.name + " falls back to all four cell types (Stock's calls)", + fmt("%.1f vs %.1f per draw", cf.total() / nFast, cs.total() / nStock)); + continue; + } check(cf.count("UseProgram") <= 2 * nFast, label + ": " + p.name + " <= 2 glUseProgram/draw"); check(cf.count("BindVertexArray") == 2 * nFast, label + ": " + p.name + " binds the VAO once per draw (+ unbind)", fmt("%.1f per draw", cf.count("BindVertexArray") / nFast)); - check(cf.count("DrawArraysInstanced") == cs.count("DrawArraysInstanced") && - cf.count("DrawArraysInstanced") == nFast, - label + ": " + p.name + " one draw call per draw, as Stock"); const double perFast = cf.total() / nFast, perStock = cs.total() / nStock; check(perFast <= 0.45 * perStock, label + ": " + p.name + " Fast <= 45% of Stock calls/draw", fmt("%.1f -> %.1f", perStock, perFast)); @@ -765,30 +1244,57 @@ void comparePaths(Scene &s, const std::string &label, bool shadows) { vtkPolyData *in = m->GetInput(); std::printf( " mapper %-24s points %5lld v/l/p/s %lld/%lld/%lld/%lld: draws %llu skipped %llu " - "lookups %llu/%llu\n", + "fallbacks %llu lookups %llu/%llu\n", m->GetClassName(), static_cast(in->GetNumberOfPoints()), static_cast(in->GetNumberOfVerts()), static_cast(in->GetNumberOfLines()), static_cast(in->GetNumberOfPolys()), static_cast(in->GetNumberOfStrips()), static_cast(t.draws), static_cast(t.cellTypesSkipped), + static_cast(t.coincidentFallbacks), static_cast(t.programLookups), static_cast(t.programLookupsSkipped)); } for (size_t i = 0; i < s.probes.size(); ++i) - std::printf(" first frame per draw %-11s Stock %s\n %-32s Fast %s\n", + std::printf(" first frame per draw %-12s Stock %s\n %-33s Fast %s\n", s.probes[i].name.c_str(), stock.probeFirst[i].str(std::max(1.0, stock.probeFirstDraws[i])).c_str(), "", fast.probeFirst[i].str(std::max(1.0, fast.probeFirstDraws[i])).c_str()); } + // The hardware selector over the whole probe set. Not with shadows on: + // cvcGL's shadow pass chain renders props outside the selector's + // BeginRenderProp/EndRenderProp, which VTK rejects ("Too many props") on + // every path alike. + if (!shadows) + comparePicking(s, label); } -// ── 5. lifecycle ───────────────────────────────────────────────────────────── +// ── 6. lifecycle ───────────────────────────────────────────────────────────── void testLifecycle(Scene &s) { std::printf("lifecycle\n"); LowMemoryPolyDataMapper::setDrawPath(DrawPath::Fast); s.sr->render(); + // GetShader() hands out a mutable shader source: that mapper's next draw + // resolves its program in the cache again; nobody else's does. + { + s.sr->render(); + s.resetCounters(); + LowMemoryPolyDataMapper *m = s.probes.front().mapper; + m->GetShader(vtkShader::Fragment); + s.sr->render(); + const auto all = s.stats(); + check(m->stats().programLookups == 1 && all.programLookups == 1, + "GetShader(): that mapper's next draw looks its program up (and only that one)", + fmt("lookups: mapper %.0f, scene %.0f", double(m->stats().programLookups), + double(all.programLookups))); + s.resetCounters(); + m->invalidateProgramCache(); + s.sr->render(); + check(m->stats().programLookups == 1 && s.stats().programLookups == 1, + "invalidateProgramCache(): the same"); + } + // A mapper's graphics resources released (what removing a prop from its // renderer does): its next draw looks the program up afresh. for (auto *m : s.mappers()) @@ -837,7 +1343,7 @@ void testLifecycle(Scene &s) { samePixels("new window: same frame as the old window", before.steadyPx, b.steadyPx); } -// ── 6. guards ──────────────────────────────────────────────────────────────── +// ── 7. guards ──────────────────────────────────────────────────────────────── void testGuards(cvc::app &app) { std::printf("guards\n"); Scene s(app); @@ -867,8 +1373,11 @@ void testGuards(cvc::app &app) { for (DrawPath p : {DrawPath::Stock, DrawPath::Fast}) runPath(s, p); const PathRun stock = runPath(s, DrawPath::Stock); + const PathRun all = runPath(s, DrawPath::AllCellTypes); const PathRun fast = runPath(s, DrawPath::Fast); samePixels("POLYGON_OFFSET mode: Fast pixels == Stock", stock.steadyPx, fast.steadyPx); + checkTraces("POLYGON_OFFSET mode", stock.steadyTrace, all.steadyTrace, fast.steadyTrace, + fast.steadyStats); const auto &zs = zero.mapper->stats(); const auto &ss = shifted.mapper->stats(); check(zs.coincidentFallbacks == zs.draws && zs.cellTypesSkipped == 0 && zs.draws > 0, @@ -883,30 +1392,54 @@ void testGuards(cvc::app &app) { check(vs.stockDraws == vs.draws && vs.draws > 0, "vertex visibility draws the stock way"); // Picking: the selector's passes draw the stock way, and select the same. - auto select = [&](DrawPath p) { + comparePicking(s, "guards scene"); +} + +// ── 8. shader replacements after the first draw ───────────────────────────── +// VTK's low-memory mapper ignores its actor's shader property once the program +// is built; LowMemoryPolyDataMapper::RenderPieceStart rebuilds it. A plain +// GeometryNode (no streaming), every draw path. +void testLateShaderReplacement(cvc::app &app) { + std::printf("shader replacement after the first draw\n"); + for (DrawPath p : {DrawPath::Stock, DrawPath::AllCellTypes, DrawPath::Fast}) { LowMemoryPolyDataMapper::setDrawPath(p); + Scene s(app); + auto n = s.node("plain"); + n->setGeometry(boxGeometry(0, 0, 0, 5, 5, 0.5)); + n->setUseSingleColor(true); + n->setColor(1, 1, 1); + s.w = 48; + s.h = 48; + s.open(); + s.sr->setCamera(0, 0, 50, 0, 0, 0, 0, 1, 0, 30.0, 1.0, 200.0); + s.sr->render(); + s.sr->render(); + const auto centre = [&](const std::vector &px) { + const size_t i = 4 * (static_cast(s.h / 2) * s.w + s.w / 2); + return std::vector(px.begin() + i, px.begin() + i + 3); + }; + const auto corner = [](const std::vector &px) { + return std::vector(px.begin(), px.begin() + 3); + }; + const auto drawn = s.rgba(); + const bool visible = centre(drawn) != corner(drawn); + + n->addFragmentShaderReplacement("//VTK::UniformFlow::Impl", + "//VTK::UniformFlow::Impl\n discard;\n"); s.resetCounters(); - vtkNew sel; - sel->SetRenderer(s.sr->renderer()); - sel->SetArea(0, 0, s.w - 1, s.h - 1); - sel->SetFieldAssociation(vtkDataObject::FIELD_ASSOCIATION_CELLS); - vtkSmartPointer res = vtk::TakeSmartPointer(sel->Select()); - std::string sig; - for (unsigned i = 0; res && i < res->GetNumberOfNodes(); ++i) { - vtkSelectionNode *n = res->GetNode(i); - sig += fmt("[%.0f:%.0f]", n->GetProperties()->Get(vtkSelectionNode::PROP_ID()), - n->GetSelectionList() ? double(n->GetSelectionList()->GetNumberOfTuples()) : 0.0); - } - return std::make_pair(sig, s.stats()); - }; - const auto pickStock = select(DrawPath::Stock); - const auto pickFast = select(DrawPath::Fast); - check( - pickFast.second.stockDraws == pickFast.second.draws && pickFast.second.draws > 0, - "picking draws the stock way on the Fast path", - fmt("%.0f of %.0f draws", double(pickFast.second.stockDraws), double(pickFast.second.draws))); - check(pickFast.first == pickStock.first, "picking selects the same cells on both paths", - pickStock.first); + s.sr->render(); + const auto discarded = s.rgba(); + const auto st = s.stats(); + check(visible && centre(discarded) == corner(discarded), + std::string(pathName(p)) + ": a replacement added after the first draw reaches the GPU", + fmt("lookups %.0f", double(st.programLookups + st.stockDraws))); + + n->clearShaderReplacements(); + s.sr->render(); + const auto back = s.rgba(); + check(diffBytes(back, drawn) == 0, + std::string(pathName(p)) + ": cleared again, the original frame comes back"); + } LowMemoryPolyDataMapper::setDrawPath(DrawPath::Fast); } @@ -929,13 +1462,18 @@ int main() { vtkOutputWindow::SetInstance(vtkSmartPointer::New()); testPolicy(); + cvc::app app; if (!LowMemoryPolyDataMapper::replicaActive()) { - std::printf(" skipped: this VTK is not 9.5.0, so the mapper draws the stock way\n"); - std::printf("PASS: cvcgl_lowmem_fastdraw (%d checks)\n", g_checks); + // The shader-property rebuild is not part of the replica: every VTK. + std::printf(" replica off (VTK is not an unpatched 9.5.0): every path draws the stock way\n"); + if (renderAvailable(app)) + testLateShaderReplacement(app); + std::printf("%s: cvcgl_lowmem_fastdraw (%d checks, %d failed)\n", + g_failures == 0 ? "PASS" : "FAIL", g_checks, g_failures); return g_failures == 0 ? 0 : 1; } - cvc::app app; + LowMemoryPolyDataMapper::setStageObserver(&onStage); if (renderAvailable(app)) { // Shadows on and off in separate scenes: SceneGraph::setShadowsEnabled(false) // drops the shadow passes without releasing their GL resources, which VTK @@ -960,7 +1498,9 @@ int main() { testLifecycle(s); } testGuards(app); + testLateShaderReplacement(app); } + LowMemoryPolyDataMapper::setStageObserver(nullptr); check(ErrorCounter::errors() == 0, "no VTK errors", fmt("%.0f errors", double(ErrorCounter::errors()))); std::printf("%s: cvcgl_lowmem_fastdraw (%d checks, %d failed)\n", diff --git a/src/cvcGL/test/gl_call_counter.cpp b/src/cvcGL/test/gl_call_counter.cpp index bc5ecef42..509f1bba6 100644 --- a/src/cvcGL/test/gl_call_counter.cpp +++ b/src/cvcGL/test/gl_call_counter.cpp @@ -14,6 +14,7 @@ #include #include #include +#include #include #include @@ -21,155 +22,279 @@ namespace cvcgl_test { namespace { -// Counted and forwarded unchanged. +// ── the wrapped entry points ──────────────────────────────────────────────── +// Every GL entry point VTK 9.5.0's Rendering/OpenGL2, Rendering/VolumeOpenGL2 +// and Rendering/Core call (a grep of their sources against glad), cvcGL's own, +// and the GLES3/GL4 core families those may grow into. + +// Counted and traced with no context. #define CVC_GL_PLAIN(X) \ - X(ActiveTexture) \ - X(BindTexture) \ - X(TexParameteri) \ - X(TexParameterf) \ - X(TexImage2D) \ - X(TexImage3D) \ - X(TexSubImage2D) \ - X(TexSubImage3D) \ - X(TexBuffer) \ - X(GenerateMipmap) \ - X(GenTextures) \ - X(DeleteTextures) \ + X(AttachShader) \ + X(BeginQuery) \ + X(BeginQueryIndexed) \ + X(BeginTransformFeedback) \ X(BindBuffer) \ - X(BufferData) \ - X(BufferSubData) \ - X(CopyBufferSubData) \ - X(GenBuffers) \ - X(DeleteBuffers) \ + X(BindBufferBase) \ + X(BindBufferRange) \ + X(BindFragDataLocation) \ + X(BindFramebuffer) \ + X(BindRenderbuffer) \ X(BindVertexArray) \ - X(GenVertexArrays) \ - X(DeleteVertexArrays) \ - X(EnableVertexAttribArray) \ - X(DisableVertexAttribArray) \ - X(VertexAttribPointer) \ - X(DrawArrays) \ - X(DrawArraysInstanced) \ - X(DrawElements) \ - X(DrawRangeElements) \ - X(DrawElementsInstanced) \ - X(Enable) \ - X(Disable) \ - X(DepthMask) \ - X(DepthFunc) \ - X(ColorMask) \ - X(BlendFunc) \ - X(BlendFuncSeparate) \ X(BlendEquation) \ X(BlendEquationSeparate) \ - X(Viewport) \ - X(Scissor) \ + X(BlendFunc) \ + X(BlendFuncSeparate) \ + X(BlitFramebuffer) \ + X(BufferData) \ + X(BufferSubData) \ + X(ClampColor) \ X(Clear) \ + X(ClearBufferfi) \ + X(ClearBufferfv) \ + X(ClearBufferiv) \ + X(ClearBufferuiv) \ X(ClearColor) \ X(ClearDepth) \ X(ClearDepthf) \ - X(StencilFunc) \ - X(StencilMask) \ - X(StencilOp) \ + X(ClearStencil) \ + X(ColorMask) \ + X(ColorMaski) \ + X(CompileShader) \ + X(CopyBufferSubData) \ + X(CreateProgram) \ + X(CreateShader) \ X(CullFace) \ - X(PolygonOffset) \ + X(DebugMessageCallback) \ + X(DebugMessageControl) \ + X(DebugMessageInsert) \ + X(DeleteBuffers) \ + X(DeleteFramebuffers) \ + X(DeleteQueries) \ + X(DeleteRenderbuffers) \ + X(DeleteShader) \ + X(DeleteSync) \ + X(DeleteTextures) \ + X(DeleteVertexArrays) \ + X(DepthFunc) \ + X(DepthMask) \ + X(DetachShader) \ + X(Disable) \ + X(DisableVertexAttribArray) \ + X(Disablei) \ + X(DrawBuffer) \ + X(DrawBuffers) \ + X(Enable) \ + X(EnableVertexAttribArray) \ + X(Enablei) \ + X(EndQuery) \ + X(EndQueryIndexed) \ + X(EndTransformFeedback) \ + X(FenceSync) \ + X(Flush) \ + X(FramebufferRenderbuffer) \ + X(FramebufferTexture2D) \ + X(FramebufferTexture3D) \ + X(FramebufferTextureLayer) \ + X(GenBuffers) \ + X(GenFramebuffers) \ + X(GenQueries) \ + X(GenRenderbuffers) \ + X(GenTextures) \ + X(GenVertexArrays) \ + X(InvalidateFramebuffer) \ X(LineWidth) \ - X(PointSize) \ + X(MemoryBarrier) \ + X(PatchParameteri) \ X(PixelStorei) \ - X(BindFramebuffer) \ - X(BindRenderbuffer) \ - X(FramebufferTexture2D) \ - X(FramebufferRenderbuffer) \ - X(BlitFramebuffer) \ + X(PointSize) \ + X(PolygonOffset) \ + X(QueryCounter) \ X(ReadBuffer) \ - X(DrawBuffer) \ - X(DrawBuffers) \ - X(CreateProgram) \ - X(CreateShader) \ + X(RenderbufferStorage) \ + X(RenderbufferStorageMultisample) \ + X(Scissor) \ X(ShaderSource) \ - X(CompileShader) \ - X(AttachShader) \ - X(Flush) \ - X(Finish) + X(StencilFunc) \ + X(StencilFuncSeparate) \ + X(StencilMask) \ + X(StencilMaskSeparate) \ + X(StencilOp) \ + X(StencilOpSeparate) \ + X(TransformFeedbackVaryings) \ + X(UniformBlockBinding) \ + X(UnmapBuffer) \ + X(VertexAttribDivisor) \ + X(VertexAttribDivisorARB) \ + X(VertexAttribIPointer) \ + X(VertexAttribPointer) \ + X(Viewport) + +// Texture-state calls: act on the active unit's binding (traced with the unit). +#define CVC_GL_UNIT(X) \ + X(BindTexture) \ + X(CopyTexImage2D) \ + X(CopyTexSubImage2D) \ + X(GenerateMipmap) \ + X(TexBuffer) \ + X(TexImage1D) \ + X(TexImage2D) \ + X(TexImage2DMultisample) \ + X(TexImage3D) \ + X(TexParameterf) \ + X(TexParameterfv) \ + X(TexParameteri) \ + X(TexStorage2D) \ + X(TexStorage2DMultisample) \ + X(TexStorage3D) \ + X(TexSubImage2D) \ + X(TexSubImage3D) + +// Draw / dispatch calls (traced with the bound program). +#define CVC_GL_DRAW(X) \ + X(DispatchCompute) \ + X(DrawArrays) \ + X(DrawArraysInstanced) \ + X(DrawArraysInstancedARB) \ + X(DrawElements) \ + X(DrawElementsInstanced) \ + X(DrawElementsInstancedARB) \ + X(DrawRangeElements) -// Round trips on WebGL (the caller waits for the GPU process). +// Desktop round-trips (upper bound for WebGL): the caller waits on the driver. #define CVC_GL_SYNC(X) \ - X(GetIntegerv) \ - X(GetFloatv) \ + X(CheckFramebufferStatus) \ + X(ClientWaitSync) \ + X(Finish) \ + X(GetAttribLocation) \ X(GetBooleanv) \ + X(GetBufferPointerv) \ X(GetDoublev) \ - X(IsEnabled) \ X(GetError) \ - X(GetUniformLocation) \ - X(GetAttribLocation) \ - X(CheckFramebufferStatus) \ + X(GetFloatv) \ + X(GetFramebufferAttachmentParameteriv) \ + X(GetIntegerv) \ + X(GetNamedRenderbufferParameteriv) \ + X(GetProgramInfoLog) \ X(GetProgramiv) \ + X(GetQueryObjectiv) \ + X(GetQueryObjectui64v) \ + X(GetQueryObjectuiv) \ + X(GetRenderbufferParameteriv) \ + X(GetShaderInfoLog) \ X(GetShaderiv) \ X(GetString) \ X(GetStringi) \ - X(GetTexLevelParameteriv) \ X(GetTexImage) \ - X(ReadPixels) \ - X(IsProgram) + X(GetTexLevelParameteriv) \ + X(GetTextureLevelParameteriv) \ + X(GetUniformBlockIndex) \ + X(GetUniformLocation) \ + X(IsEnabled) \ + X(IsProgram) \ + X(MapBuffer) \ + X(MapBufferRange) \ + X(ReadPixels) -// Counted with their own wrappers below. -#define CVC_GL_OWN(X) \ - X(UseProgram) \ - X(LinkProgram) \ - X(DeleteProgram) \ - X(Uniform1i) \ - X(Uniform2i) \ +// glUniform{1,2,3,4}{f,i,ui}: every argument is a scalar. +#define CVC_GL_USCALAR(X) \ X(Uniform1f) \ X(Uniform2f) \ X(Uniform3f) \ X(Uniform4f) \ - X(Uniform1iv) \ - X(Uniform1fv) \ - X(Uniform2fv) \ - X(Uniform3fv) \ - X(Uniform4fv) \ - X(UniformMatrix3fv) \ - X(UniformMatrix4fv) + X(Uniform1i) \ + X(Uniform2i) \ + X(Uniform3i) \ + X(Uniform4i) \ + X(Uniform1ui) \ + X(Uniform2ui) \ + X(Uniform3ui) \ + X(Uniform4ui) + +// glUniform{1,2,3,4}{f,i,ui}v(location, count, const T *v): (name, T, components). +#define CVC_GL_UVEC(X) \ + X(Uniform1fv, GLfloat, 1) \ + X(Uniform2fv, GLfloat, 2) \ + X(Uniform3fv, GLfloat, 3) \ + X(Uniform4fv, GLfloat, 4) \ + X(Uniform1iv, GLint, 1) \ + X(Uniform2iv, GLint, 2) \ + X(Uniform3iv, GLint, 3) \ + X(Uniform4iv, GLint, 4) \ + X(Uniform1uiv, GLuint, 1) \ + X(Uniform2uiv, GLuint, 2) \ + X(Uniform3uiv, GLuint, 3) \ + X(Uniform4uiv, GLuint, 4) + +// glUniformMatrix*fv(location, count, transpose, const GLfloat *v): (name, floats). +#define CVC_GL_UMAT(X) \ + X(UniformMatrix2fv, 4) \ + X(UniformMatrix3fv, 9) \ + X(UniformMatrix4fv, 16) \ + X(UniformMatrix2x3fv, 6) \ + X(UniformMatrix3x2fv, 6) \ + X(UniformMatrix2x4fv, 8) \ + X(UniformMatrix4x2fv, 8) \ + X(UniformMatrix3x4fv, 12) \ + X(UniformMatrix4x3fv, 12) + +// Tracked state: the bound program and the active texture unit. +#define CVC_GL_STATE(X) \ + X(UseProgram) \ + X(LinkProgram) \ + X(DeleteProgram) \ + X(ActiveTexture) + +enum Cls : unsigned char { kPlain, kUnit, kDraw, kSync, kUniform, kState }; enum Entry : int { -#define CVC_E(name) E_##name, - CVC_GL_PLAIN(CVC_E) CVC_GL_SYNC(CVC_E) CVC_GL_OWN(CVC_E) +#define CVC_E(name, ...) E_##name, + CVC_GL_PLAIN(CVC_E) CVC_GL_UNIT(CVC_E) CVC_GL_DRAW(CVC_E) CVC_GL_SYNC(CVC_E) CVC_GL_USCALAR(CVC_E) + CVC_GL_UVEC(CVC_E) CVC_GL_UMAT(CVC_E) CVC_GL_STATE(CVC_E) #undef CVC_E - E_COUNT + E_COUNT }; const char *const g_names[E_COUNT] = { -#define CVC_N(name) #name, - CVC_GL_PLAIN(CVC_N) CVC_GL_SYNC(CVC_N) CVC_GL_OWN(CVC_N) +#define CVC_N(name, ...) #name, + CVC_GL_PLAIN(CVC_N) CVC_GL_UNIT(CVC_N) CVC_GL_DRAW(CVC_N) CVC_GL_SYNC(CVC_N) + CVC_GL_USCALAR(CVC_N) CVC_GL_UVEC(CVC_N) CVC_GL_UMAT(CVC_N) CVC_GL_STATE(CVC_N) #undef CVC_N }; +constexpr Cls g_cls[E_COUNT] = { +#define CVC_PLAIN(...) kPlain, +#define CVC_UNIT(...) kUnit, +#define CVC_DRAW(...) kDraw, +#define CVC_SYNC(...) kSync, +#define CVC_UNI(...) kUniform, +#define CVC_STATE(...) kState, + CVC_GL_PLAIN(CVC_PLAIN) CVC_GL_UNIT(CVC_UNIT) CVC_GL_DRAW(CVC_DRAW) CVC_GL_SYNC(CVC_SYNC) + CVC_GL_USCALAR(CVC_UNI) CVC_GL_UVEC(CVC_UNI) CVC_GL_UMAT(CVC_UNI) CVC_GL_STATE(CVC_STATE) +#undef CVC_PLAIN +#undef CVC_UNIT +#undef CVC_DRAW +#undef CVC_SYNC +#undef CVC_UNI +#undef CVC_STATE +}; + double g_n[E_COUNT]; double g_uniformRedundant = 0; -bool isSync(int i) { return i >= E_GetIntegerv && i <= E_IsProgram; } -bool isUniform(int i) { return i >= E_Uniform1i && i <= E_UniformMatrix4fv; } - -template struct Counted; -template struct Counted { - static inline R(GLAD_API_PTR *orig)(A...) = nullptr; - static R GLAD_API_PTR fn(A... a) { - g_n[Id] += 1; - return orig(a...); - } - static void hook(R(GLAD_API_PTR *&ptr)(A...)) { - if (ptr && ptr != &fn) { - orig = ptr; - ptr = &fn; - } - } -}; +// Tracked GL state, and the open trace. +GLuint g_program = 0; +unsigned g_unit = 0; +bool g_tracing = false; +Trace g_trace; // Uniform values are per program: (program, location) -> last bytes sent. -GLuint g_program = 0; std::map, std::string> g_last; -void noteUniform(GLint loc, const void *p, size_t bytes) { - std::string v(static_cast(p), bytes); +// argBytes starts with the GLint location; the rest is the value as sent. +void noteUniform(const std::string &argBytes) { + GLint loc = 0; + std::memcpy(&loc, argBytes.data(), sizeof loc); + std::string v = argBytes.substr(sizeof loc); auto key = std::make_pair(g_program, loc); auto it = g_last.find(key); if (it != g_last.end() && it->second == v) @@ -182,111 +307,230 @@ void forgetProgram(GLuint p) { it = it->first.first == p ? g_last.erase(it) : std::next(it); } -#define CVC_O(name) decltype(glad_gl##name) o_##name = nullptr; -CVC_GL_OWN(CVC_O) +void record(int id, std::string &&args) { + TraceRec r; + r.entry = id; + switch (g_cls[id]) { + case kUnit: + r.ctx = g_unit; + break; + case kDraw: + case kUniform: + r.ctx = g_program; + break; + default: + r.ctx = 0; + } + r.args = std::move(args); + g_trace.push_back(std::move(r)); +} + +// Floats are recorded as GL computes with them: -0.0 as +0.0. (VTK's camera +// and actor key-matrix caches hand out a zero normal-matrix entry with either +// sign depending on which earlier render last recomputed them -- history, not +// the draw path: two Stock hardware selections in a row differ that way.) +template F canonicalZero(F v) { return v == F(0) ? F(0) : v; } + +template void appendArg(std::string &s, T v) { + if constexpr (std::is_floating_point_v) { + const T c = canonicalZero(v); + char b[sizeof(T)]; + std::memcpy(b, &c, sizeof(T)); + s.append(b, sizeof(T)); + } else if constexpr (std::is_same_v) { + // Location / attribute names: the string, so the lookup is identified. + if (v) + s.append(v, std::strlen(v) + 1); + else + s.push_back('\0'); + } else if constexpr (std::is_pointer_v) { + s.push_back(v ? '\1' : '\0'); // bulk data / out-params: only whether it is there + } else { + char b[sizeof(T)]; + std::memcpy(b, &v, sizeof(T)); + s.append(b, sizeof(T)); + } +} + +// The generic wrapper: count, note uniform values, trace. +template struct Wrap; +template struct Wrap { + static constexpr bool kIsUniform = g_cls[Id] == kUniform; + static inline R(GLAD_API_PTR *orig)(A...) = nullptr; + static R GLAD_API_PTR fn(A... a) { + g_n[Id] += 1; + if (kIsUniform || g_tracing) { + std::string bytes; + (appendArg(bytes, a), ...); + if constexpr (kIsUniform) + noteUniform(bytes); + if (g_tracing) + record(Id, std::move(bytes)); + } + return orig(a...); + } + static void hook(R(GLAD_API_PTR *&ptr)(A...)) { + if (ptr && ptr != &fn) { + orig = ptr; + ptr = &fn; + } + } + static void unhook(R(GLAD_API_PTR *&ptr)(A...)) { + if (ptr == &fn) + ptr = orig; + } +}; + +// glUniform*v / glUniformMatrix*: the values behind the pointer. +#define CVC_O(name, ...) decltype(glad_gl##name) o_##name = nullptr; +CVC_GL_UVEC(CVC_O) +CVC_GL_UMAT(CVC_O) +CVC_GL_STATE(CVC_O) #undef CVC_O +void uniformArray(int id, GLint l, GLsizei c, const std::string &prefix, const void *v, + size_t bytesPerElement, bool isFloat) { + g_n[id] += 1; + std::string b; + appendArg(b, l); + appendArg(b, c); + b += prefix; + if (v && c > 0) { + const size_t n = bytesPerElement * static_cast(c); + if (isFloat) { + for (size_t k = 0; k < n; k += sizeof(GLfloat)) { + GLfloat f; + std::memcpy(&f, static_cast(v) + k, sizeof f); + appendArg(b, f); + } + } else { + b.append(static_cast(v), n); + } + } + noteUniform(b); + if (g_tracing) + record(id, std::move(b)); +} + +#define CVC_UV_W(name, T, k) \ + void GLAD_API_PTR w_##name(GLint l, GLsizei c, const T *v) { \ + uniformArray(E_##name, l, c, std::string(), v, sizeof(T) * (k), std::is_same_v); \ + o_##name(l, c, v); \ + } +CVC_GL_UVEC(CVC_UV_W) +#undef CVC_UV_W + +#define CVC_UM_W(name, k) \ + void GLAD_API_PTR w_##name(GLint l, GLsizei c, GLboolean t, const GLfloat *v) { \ + uniformArray(E_##name, l, c, std::string(1, static_cast(t)), v, sizeof(GLfloat) * (k), \ + true); \ + o_##name(l, c, t, v); \ + } +CVC_GL_UMAT(CVC_UM_W) +#undef CVC_UM_W + void GLAD_API_PTR w_UseProgram(GLuint p) { g_n[E_UseProgram] += 1; g_program = p; + if (g_tracing) { + std::string b; + appendArg(b, p); + record(E_UseProgram, std::move(b)); + } o_UseProgram(p); } void GLAD_API_PTR w_LinkProgram(GLuint p) { g_n[E_LinkProgram] += 1; forgetProgram(p); + if (g_tracing) { + std::string b; + appendArg(b, p); + record(E_LinkProgram, std::move(b)); + } o_LinkProgram(p); } void GLAD_API_PTR w_DeleteProgram(GLuint p) { g_n[E_DeleteProgram] += 1; forgetProgram(p); + if (g_tracing) { + std::string b; + appendArg(b, p); + record(E_DeleteProgram, std::move(b)); + } o_DeleteProgram(p); } -void GLAD_API_PTR w_Uniform1i(GLint l, GLint a) { - g_n[E_Uniform1i] += 1; - noteUniform(l, &a, sizeof a); - o_Uniform1i(l, a); -} -void GLAD_API_PTR w_Uniform2i(GLint l, GLint a, GLint b) { - g_n[E_Uniform2i] += 1; - const GLint v[2] = {a, b}; - noteUniform(l, v, sizeof v); - o_Uniform2i(l, a, b); -} -void GLAD_API_PTR w_Uniform1f(GLint l, GLfloat a) { - g_n[E_Uniform1f] += 1; - noteUniform(l, &a, sizeof a); - o_Uniform1f(l, a); -} -void GLAD_API_PTR w_Uniform2f(GLint l, GLfloat a, GLfloat b) { - g_n[E_Uniform2f] += 1; - const GLfloat v[2] = {a, b}; - noteUniform(l, v, sizeof v); - o_Uniform2f(l, a, b); -} -void GLAD_API_PTR w_Uniform3f(GLint l, GLfloat a, GLfloat b, GLfloat c) { - g_n[E_Uniform3f] += 1; - const GLfloat v[3] = {a, b, c}; - noteUniform(l, v, sizeof v); - o_Uniform3f(l, a, b, c); -} -void GLAD_API_PTR w_Uniform4f(GLint l, GLfloat a, GLfloat b, GLfloat c, GLfloat d) { - g_n[E_Uniform4f] += 1; - const GLfloat v[4] = {a, b, c, d}; - noteUniform(l, v, sizeof v); - o_Uniform4f(l, a, b, c, d); -} -void GLAD_API_PTR w_Uniform1iv(GLint l, GLsizei c, const GLint *v) { - g_n[E_Uniform1iv] += 1; - noteUniform(l, v, sizeof(GLint) * c); - o_Uniform1iv(l, c, v); -} -void GLAD_API_PTR w_Uniform1fv(GLint l, GLsizei c, const GLfloat *v) { - g_n[E_Uniform1fv] += 1; - noteUniform(l, v, sizeof(GLfloat) * c); - o_Uniform1fv(l, c, v); -} -void GLAD_API_PTR w_Uniform2fv(GLint l, GLsizei c, const GLfloat *v) { - g_n[E_Uniform2fv] += 1; - noteUniform(l, v, 2 * sizeof(GLfloat) * c); - o_Uniform2fv(l, c, v); -} -void GLAD_API_PTR w_Uniform3fv(GLint l, GLsizei c, const GLfloat *v) { - g_n[E_Uniform3fv] += 1; - noteUniform(l, v, 3 * sizeof(GLfloat) * c); - o_Uniform3fv(l, c, v); -} -void GLAD_API_PTR w_Uniform4fv(GLint l, GLsizei c, const GLfloat *v) { - g_n[E_Uniform4fv] += 1; - noteUniform(l, v, 4 * sizeof(GLfloat) * c); - o_Uniform4fv(l, c, v); +void GLAD_API_PTR w_ActiveTexture(GLenum t) { + g_n[E_ActiveTexture] += 1; + g_unit = t - GL_TEXTURE0; + if (g_tracing) { + std::string b; + appendArg(b, t); + record(E_ActiveTexture, std::move(b)); + } + o_ActiveTexture(t); } -void GLAD_API_PTR w_UniformMatrix3fv(GLint l, GLsizei c, GLboolean t, const GLfloat *v) { - g_n[E_UniformMatrix3fv] += 1; - noteUniform(l, v, 9 * sizeof(GLfloat) * c); - o_UniformMatrix3fv(l, c, t, v); + +// The real glGetIntegerv / glGetFloatv, whether or not they are wrapped now. +void rawGetIntegerv(GLenum what, GLint *v) { + using W = Wrap; + auto fn = glad_glGetIntegerv == &W::fn ? W::orig : glad_glGetIntegerv; + if (fn) + fn(what, v); } -void GLAD_API_PTR w_UniformMatrix4fv(GLint l, GLsizei c, GLboolean t, const GLfloat *v) { - g_n[E_UniformMatrix4fv] += 1; - noteUniform(l, v, 16 * sizeof(GLfloat) * c); - o_UniformMatrix4fv(l, c, t, v); +void rawGetFloatv(GLenum what, GLfloat *v) { + using W = Wrap; + auto fn = glad_glGetFloatv == &W::fn ? W::orig : glad_glGetFloatv; + if (fn) + fn(what, v); } } // namespace +bool TraceRec::marker() const { return entry >= kTraceMarkBase; } + void glcallsInstall() { -#define CVC_H(name) Counted::hook(glad_gl##name); + // Start from the context's real state (calls made while uninstalled, or in + // another context, were not tracked). + GLint program = 0, unit = GL_TEXTURE0; + rawGetIntegerv(GL_CURRENT_PROGRAM, &program); + rawGetIntegerv(GL_ACTIVE_TEXTURE, &unit); + g_program = static_cast(program); + g_unit = static_cast(unit - GL_TEXTURE0); +#define CVC_H(name, ...) Wrap::hook(glad_gl##name); CVC_GL_PLAIN(CVC_H) + CVC_GL_UNIT(CVC_H) + CVC_GL_DRAW(CVC_H) CVC_GL_SYNC(CVC_H) + CVC_GL_USCALAR(CVC_H) #undef CVC_H -#define CVC_W(name) \ +#define CVC_W(name, ...) \ if (glad_gl##name && glad_gl##name != w_##name) { \ o_##name = glad_gl##name; \ glad_gl##name = w_##name; \ } - CVC_GL_OWN(CVC_W) + CVC_GL_UVEC(CVC_W) + CVC_GL_UMAT(CVC_W) + CVC_GL_STATE(CVC_W) #undef CVC_W } +void glcallsUninstall() { +#define CVC_U(name, ...) Wrap::unhook(glad_gl##name); + CVC_GL_PLAIN(CVC_U) + CVC_GL_UNIT(CVC_U) + CVC_GL_DRAW(CVC_U) + CVC_GL_SYNC(CVC_U) + CVC_GL_USCALAR(CVC_U) +#undef CVC_U +#define CVC_R(name, ...) \ + if (glad_gl##name == w_##name) \ + glad_gl##name = o_##name; + CVC_GL_UVEC(CVC_R) + CVC_GL_UMAT(CVC_R) + CVC_GL_STATE(CVC_R) +#undef CVC_R +} + GLCalls glcallsRead() { GLCalls r; r.n.assign(g_n, g_n + E_COUNT); @@ -297,6 +541,66 @@ GLCalls glcallsRead() { int glcallsEntries() { return E_COUNT; } const char *glcallsName(int i) { return i >= 0 && i < E_COUNT ? g_names[i] : "?"; } +void glcallsTraceStart() { + g_trace.clear(); + g_tracing = true; + for (auto [kind, what] : {std::make_pair(kTraceInitPointSize, GLenum(GL_POINT_SIZE)), + std::make_pair(kTraceInitLineWidth, GLenum(GL_LINE_WIDTH))}) { + GLfloat v = 0; + rawGetFloatv(what, &v); + TraceRec r; + r.entry = kTraceMarkBase + kind; + appendArg(r.args, v); + g_trace.push_back(std::move(r)); + } +} + +Trace glcallsTraceStop() { + g_tracing = false; + Trace t; + t.swap(g_trace); + return t; +} + +void glcallsMark(int kind, int arg) { + if (!g_tracing) + return; + TraceRec r; + r.entry = kTraceMarkBase + kind; + r.ctx = static_cast(arg); + g_trace.push_back(std::move(r)); +} + +bool glcallsIsDraw(int entry) { return entry >= 0 && entry < E_COUNT && g_cls[entry] == kDraw; } + +int glcallsEntry(const std::string &name) { + for (int i = 0; i < E_COUNT; ++i) + if (name == g_names[i]) + return i; + return -1; +} + +std::string glcallsDescribe(const TraceRec &r) { + char buf[96]; + if (r.marker()) { + std::snprintf(buf, sizeof buf, "mark %d (%u)", r.entry - kTraceMarkBase, r.ctx); + return buf; + } + const Cls c = r.entry >= 0 && r.entry < E_COUNT ? g_cls[r.entry] : kPlain; + std::snprintf(buf, sizeof buf, "gl%s%s%u [", glcallsName(r.entry), + c == kUnit ? " unit=" + : (c == kDraw || c == kUniform) ? " prog=" + : " ", + r.ctx); + std::string out = buf; + for (size_t i = 0; i < r.args.size() && i < 48; ++i) { + std::snprintf(buf, sizeof buf, i ? " %02x" : "%02x", static_cast(r.args[i])); + out += buf; + } + out += r.args.size() > 48 ? " ...]" : "]"; + return out; +} + GLCalls GLCalls::operator-(const GLCalls &o) const { GLCalls r; r.n.resize(n.size()); @@ -331,15 +635,15 @@ double GLCalls::count(const std::string &entry) const { double GLCalls::uniforms() const { double s = 0; - for (int i = 0; i < static_cast(n.size()); ++i) - s += isUniform(i) ? n[i] : 0.0; + for (int i = 0; i < static_cast(n.size()) && i < E_COUNT; ++i) + s += g_cls[i] == kUniform ? n[i] : 0.0; return s; } -double GLCalls::sync() const { +double GLCalls::roundTrips() const { double s = 0; - for (int i = 0; i < static_cast(n.size()); ++i) - s += isSync(i) ? n[i] : 0.0; + for (int i = 0; i < static_cast(n.size()) && i < E_COUNT; ++i) + s += g_cls[i] == kSync ? n[i] : 0.0; return s; } @@ -362,9 +666,11 @@ bool GLCalls::sameAs(const GLCalls &o, std::string *firstDifference) const { std::string GLCalls::str(double per, int top) const { if (per <= 0) per = 1; - char buf[160]; - std::snprintf(buf, sizeof buf, "total %.1f (uniforms %.1f, redundant %.1f, sync %.1f) |", - total() / per, uniforms() / per, uniformRedundant / per, sync() / per); + char buf[192]; + std::snprintf(buf, sizeof buf, + "total %.1f (uniforms %.1f, redundant %.1f, desktop round-trips (upper bound) " + "%.1f) |", + total() / per, uniforms() / per, uniformRedundant / per, roundTrips() / per); std::string out = buf; std::vector> v; for (int i = 0; i < static_cast(n.size()); ++i) diff --git a/src/cvcGL/test/gl_call_counter.h b/src/cvcGL/test/gl_call_counter.h index f5622fd0c..ed397f070 100644 --- a/src/cvcGL/test/gl_call_counter.h +++ b/src/cvcGL/test/gl_call_counter.h @@ -8,11 +8,18 @@ License version 2.1 as published by the Free Software Foundation. */ -// Test-only GL call counter: swaps VTK's glad function pointers (glad_gl*) for -// counting wrappers, so a test can see every GL call the scene issues, by entry -// point. Uniform uploads are also checked against the last value sent to the +// Test-only GL call counter and tracer: swaps VTK's glad function pointers +// (glad_gl*) for wrappers, so a test can see every GL call the scene issues -- +// as counts per entry point, and (while a trace is open) as an ordered trace of +// (entry point, bound program / active texture unit, argument bytes, uniform +// values). Uniform uploads are also checked against the last value sent to the // same (program, location): a re-send of an unchanged value counts as -// redundant. Desktop GL only (the wasm VTK calls GLES directly, not via glad). +// redundant. Every entry point VTK 9.5.0's Rendering/OpenGL2 and +// RenderingVolumeOpenGL2 and cvcGL call is wrapped, plus the GLES3-core +// families they may grow into (glUniform*ui*, glClearBuffer*, glTexStorage*, +// ...), so frame totals are complete. Desktop GL only (the wasm VTK calls GLES +// directly, not via glad). glcallsUninstall() puts VTK's pointers back, for +// timing without the wrappers' cost. #ifndef CVCGL_TEST_GL_CALL_COUNTER_H #define CVCGL_TEST_GL_CALL_COUNTER_H @@ -30,23 +37,62 @@ struct GLCalls { double total() const; double count(const std::string &entry) const; // entry without the "gl" prefix double uniforms() const; // every glUniform* - // Calls that make WebGL wait for the GPU process: glGet*, glIsEnabled, - // glGetError, glGetUniformLocation, glCheckFramebufferStatus, glReadPixels... - double sync() const; + // Desktop round-trips (upper bound): calls that wait on the driver on desktop + // GL -- glGet*, glIs*, glGetError, glGet*Location, glCheckFramebufferStatus, + // glReadPixels, glMapBuffer*, query results, glFinish. NOT a WebGL census: + // Firefox answers some of them (cached limits, glIsProgram, ...) in the + // content process, so on WebGL this is an upper bound. + double roundTrips() const; // Same count for every entry point; otherwise the first difference. bool sameAs(const GLCalls &o, std::string *firstDifference = nullptr) const; - // "total T (uniforms U, redundant R, sync S) | Entry=k ..." with every number - // divided by `per`; the `top` largest entries. + // "total T (uniforms U, redundant R, desktop round-trips (upper bound) S) | + // Entry=k ..." with every number divided by `per`; the `top` largest entries. std::string str(double per = 1.0, int top = 14) const; }; +// One traced GL call, or a marker the test inserted (glcallsMark). +struct TraceRec { + int entry = 0; // glcallsName(entry); >= kTraceMarkBase: marker kTraceMarkBase + kind + unsigned ctx = 0; // the bound program for glUniform* and draw calls, the active texture + // unit (0-based) for texture-state calls, 0 for the rest; a marker's arg + std::string args; // the arguments' bytes as passed (a pointer: whether it is null), the + // values a glUniform*v / glUniformMatrix* sends, the name a + // glGet*Location looks up + bool marker() const; + bool operator==(const TraceRec &o) const { + return entry == o.entry && ctx == o.ctx && args == o.args; + } + bool operator!=(const TraceRec &o) const { return !(*this == o); } +}; +using Trace = std::vector; +constexpr int kTraceMarkBase = 1 << 20; +// Marker kinds glcallsTraceStart() puts first: the GL state the trace starts +// from that vtkOpenGLState caches, args = the float value. +constexpr int kTraceInitPointSize = 1000, kTraceInitLineWidth = 1001; + // (Re)install the wrappers. Call once a GL context exists (VTK has loaded glad), // and again after anything that may reload it; idempotent. void glcallsInstall(); +// Put VTK's own function pointers back (no wrapper on any call); idempotent. +void glcallsUninstall(); GLCalls glcallsRead(); int glcallsEntries(); const char *glcallsName(int i); +// Start recording every wrapped call in order (clears the previous trace; the +// trace opens with kTraceInitPointSize / kTraceInitLineWidth markers); stop and +// take it. +void glcallsTraceStart(); +Trace glcallsTraceStop(); +// Insert a marker at this point of the open trace (no-op when none is open). +void glcallsMark(int kind, int arg); +// glDraw* / glDispatchCompute. +bool glcallsIsDraw(int entry); +// The index of an entry point ("PointSize"), or -1. +int glcallsEntry(const std::string &name); +// "glUniform1i prog=3 [05 00 00 00 01 00 00 00]" / "mark 5 (2)". +std::string glcallsDescribe(const TraceRec &r); + } // namespace cvcgl_test #endif From c9e255f7b0e4f6cf8bf91931094e57f1abe378e9 Mon Sep 17 00:00:00 2001 From: Joe Rivera Date: Thu, 1 Oct 2026 19:58:43 -0500 Subject: [PATCH 06/18] test(cvcGL): derive Fast's expected trace from replayed uniform state, not coincidentSkipSafe cvcgl_lowmem_fastdraw derived "Fast trace == AllCellTypes minus its empty cell-type blocks" only in the draws whose DrawBegin marker said skipping was allowed -- and that marker was coincidentSkipSafe()'s own answer. With coincidentSkipSafe forced to return true, every ordinary-frame trace check still passed. The oracle no longer asks the mapper: - gl_call_counter decodes a traced glUniform* call into the per-location writes it makes (glcallsUniformWrites; an n-element array upload is n writes, and each value carries its type, so Uniform1i and Uniform1iv agree). - The test replays a trace's uniform writes into a (program, location) -> value map and records, at every draw call, the bound program's state. Two traces see the same uniforms when each draw call sees the same written value at every location (or both see an unwritten one), and -- for the locations a draw reads before the frame writes them -- the frame leaves the same value for the next frame. End values nobody reads before writing are not compared: a lines-only draw on AllCellTypes leaves the empty strips block's cellType/primitiveSize behind, and every block writes both first. - expectFast() removes a draw's empty blocks iff removing them (that draw's alone, its program state on entry treated as unknown) changes no uniform any draw call sees. Fast's GL trace must equal that, Fast's own trace replayed must show every draw call AllCellTypes' uniform state, and the mapper's markers and stats are only counted against the derivation (blocks skipped, fallbacks). Mutation check: coincidentSkipSafe forced true now fails 30 checks, among them the derived-trace and uniform-state checks of every ordinary frame (shadows on/off, first and steady) and the POLYGON_OFFSET frame; forced false fails 82 (the derivation expects the skips). The new oracle also showed the guards scene's POLYGON_OFFSET case was vacuous: the mode was switched on after the shaders were built, and vtkGLSLModCoincidentTopology declares cOffset/cFactor only when a shader is built under an offset, so no offset ever reached the GPU and the expected fallback was for nothing. The scene now sets the mode before its first render and checks that the programs carry the offset uniforms. LowMemoryPolyDataMapper: coincidentSkipSafe() is asked on Fast only (it was also asked on AllCellTypes when observed, for the old derivation); the DrawBegin arg is the draw's decision, for counting. Its comment notes it is conservative where a program lacks the offset uniforms. No change to Fast's GL output. cvcgl_lowmem_fastdraw: 264/264 on the GTX 1650 and on llvmpipe (CVC_REQUIRE_RENDER=1). cvcGL ctest suite 42/42 under default, CVCGL_LOWMEM_MAPPER=force + CVCGL_LOWMEM_DRAW=stock, and force + fast. --- inc/cvc/gl/LowMemoryPolyDataMapper.h | 6 +- src/cvcGL/CMakeLists.txt | 6 +- src/cvcGL/LowMemoryPolyDataMapper.cpp | 14 +- src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp | 340 +++++++++++++++++++---- src/cvcGL/test/gl_call_counter.cpp | 39 +++ src/cvcGL/test/gl_call_counter.h | 12 + 6 files changed, 355 insertions(+), 62 deletions(-) diff --git a/inc/cvc/gl/LowMemoryPolyDataMapper.h b/inc/cvc/gl/LowMemoryPolyDataMapper.h index bddf1fd5c..899c7a4e1 100644 --- a/inc/cvc/gl/LowMemoryPolyDataMapper.h +++ b/inc/cvc/gl/LowMemoryPolyDataMapper.h @@ -145,10 +145,10 @@ class LowMemoryPolyDataMapper : public vtkOpenGLLowMemoryPolyDataMapper { // Test instrumentation: when set, the replica reports where each stage of a // draw begins and ends (the cvcgl_lowmem_fastdraw GL trace interleaves these - // with the GL calls to derive what Fast must leave out). Null -- the default - // -- costs one relaxed load per stage. Stock draws report nothing. + // with the GL calls to cut a trace into draws and cell-type blocks). Null -- + // the default -- costs one relaxed load per stage. Stock draws report nothing. enum class DrawStage : int { - DrawBegin, // arg: 1 if Fast may skip empty cell types in this draw + DrawBegin, // arg: 1 if this draw skips its empty cell types (Fast only) LookupBegin, // full shader-cache lookup LookupEnd, RebindBegin, // cached program re-bound instead diff --git a/src/cvcGL/CMakeLists.txt b/src/cvcGL/CMakeLists.txt index 662ee993b..861ed35c3 100644 --- a/src/cvcGL/CMakeLists.txt +++ b/src/cvcGL/CMakeLists.txt @@ -650,8 +650,10 @@ endif() # texture unit, argument bytes, uniform values (test/gl_call_counter.cpp swaps # VTK's glad function pointers) -- and asserts the VTK 9.5.0 replica's trace # equals the stock draw's exactly and Fast's equals it with the empty cell-type -# blocks removed (derived from the replica's stage markers); compares RGBA frames -# byte for byte -- shadows on and off, bake and steady frames, every +# blocks removed wherever removing them leaves every draw call's uniform state +# unchanged (derived by replaying the trace's glUniform* calls, not from the +# mapper's own decision) and that Fast's draw calls see that state; compares +# RGBA frames byte for byte -- shadows on and off, bake and steady frames, every # representation / scalar / edge / line kind, the hardware selector, released # resources, a re-opened window, the offset / picking / vertex-visibility guards, # and shader replacements made after the first draw. This is the gate for any diff --git a/src/cvcGL/LowMemoryPolyDataMapper.cpp b/src/cvcGL/LowMemoryPolyDataMapper.cpp index 04a89f7cd..ba6762275 100644 --- a/src/cvcGL/LowMemoryPolyDataMapper.cpp +++ b/src/cvcGL/LowMemoryPolyDataMapper.cpp @@ -291,10 +291,8 @@ void LowMemoryPolyDataMapper::RenderPieceDraw(vtkRenderer *ren, vtkActor *act) { } const bool fast = path == DrawPath::Fast; const StageObserver observe = g_stageObserver.load(std::memory_order_relaxed); - // CPU only (no GL); asked on AllCellTypes too when observed, so a test can - // derive from an AllCellTypes trace what Fast must leave out. - const bool skipSafe = (fast || observe) && coincidentSkipSafe(act); - observeStage(observe, DrawStage::DrawBegin, skipSafe ? 1 : 0); + const bool skip = fast && coincidentSkipSafe(act); // CPU only (no GL) + observeStage(observe, DrawStage::DrawBegin, skip ? 1 : 0); readyProgram(ren, fast); // Uniforms common to every cell type; also fires UpdateShaderEvent, which // GeometryNode's custom shader textures hang off. @@ -303,9 +301,8 @@ void LowMemoryPolyDataMapper::RenderPieceDraw(vtkRenderer *ren, vtkActor *act) { observeStage(observe, DrawStage::DrawEnd); return; // compile/link failure: VTK's agents would dereference null here } - if (fast && !skipSafe) + if (fast && !skip) ++m_stats.coincidentFallbacks; - const bool skip = fast && skipSafe; for (int t = 0; t < 4; ++t) { // verts, lines, polys, strips: VTK's order if (skip && !renderable(t)) { ++m_stats.cellTypesSkipped; @@ -363,7 +360,10 @@ bool LowMemoryPolyDataMapper::coincidentSkipSafe(vtkActor *act) const { // what an earlier cell type -- possibly one with nothing to draw -- left // behind, and the last one leaves a value for the next draw of that program. // Skipping is safe when every drawn type, and the program afterwards, ends up - // with the same value either way. + // with the same value either way. Conservative: the mod declares the offset + // uniforms only if the shader was BUILT under a non-zero offset, which this + // does not ask -- with the global mode switched on after the build (an offset + // VTK then never applies) it falls back for nothing. vtkProperty *prop = act->GetProperty(); const int mode = vtkMapper::GetResolveCoincidentTopology(); if (mode != VTK_RESOLVE_POLYGON_OFFSET && mode != VTK_RESOLVE_SHIFT_ZBUFFER && diff --git a/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp b/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp index b71aa8da7..e1075b562 100644 --- a/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp +++ b/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp @@ -28,11 +28,17 @@ // and plain frames, shadows on and off, and the hardware selector's cell // and point picking passes); // 3. Fast's trace equals that same trace with what Fast leaves out removed, -// DERIVED from the AllCellTypes trace (the replica marks where each draw's -// stages begin and end): every cell-type block that issued no draw call, -// in every draw where skipping is allowed. A skipped shader-cache lookup -// removes no GL call -- the lookup's GL calls are the bind tail the -// re-bind issues too; what it saves is CPU, which the stats check; +// DERIVED from the AllCellTypes trace alone (the replica marks where each +// draw's stages begin and end): a draw's cell-type blocks that issued no +// draw call go iff replaying the trace's glUniform* calls without them +// leaves every draw call's uniform state unchanged -- never from the +// mapper's own skip decision, whose markers are only counted. And Fast's +// own trace, replayed the same way, shows every draw call the uniform +// state (every location of its program) it sees on AllCellTypes, and +// leaves each program what the next frame's draws read before writing. A +// skipped shader-cache lookup removes no GL call -- the lookup's GL calls +// are the bind tail the re-bind issues too; what it saves is CPU, which +// the stats check; // 4. pixels: Stock, AllCellTypes and Fast frames are byte-identical (RGBA); // 5. the saving: Fast draws a single-cell-type mesh in <= 2 VAO binds and a // fraction of the stock calls, with no desktop round-trip in any draw, @@ -66,8 +72,11 @@ #include #include #include +#include #include +#include #include +#include #include #include #include @@ -92,6 +101,7 @@ #include #include #include +#include #include #include #include @@ -247,51 +257,251 @@ Trace asDrawState(const Trace &t) { return out; } -// What Fast's GL trace must be, derived from an AllCellTypes trace and its -// stage markers. In every draw where skipping is allowed (DrawBegin arg 1), -// each cell-type block (AgentBegin..AgentEnd) that issued no draw call is -// removed. Everything else stays -- including each full lookup's GL calls, -// which are the ReadyShaderProgram(program) tail (compile if released, bind) -// that the re-bind replacing it issues too. -struct Expected { - Trace gl; - int blocksRemoved = 0, blocksKept = 0, lookups = 0, draws = 0, unskippableDraws = 0; +// ── the uniforms each draw call sees ───────────────────────────────────────── +// A trace's glUniform* calls replayed into (program, location) -> value: at +// every draw call, the bound program's locations the trace has written so far +// (a location it has not written holds what the program had before the trace). +struct UniformView { + struct DrawCall { + size_t at = 0; // index in the trace + unsigned program = 0; + std::map written; + }; + std::vector calls; + std::map> end; // each program, at the trace's end }; -Expected expectFast(const Trace &all) { - Expected e; - bool skipAllowed = false, inBlock = false, blockDraws = false; - Trace block; - for (const auto &r : all) { - if (!r.marker()) { - if (inBlock) { - block.push_back(r); - blockDraws = blockDraws || cvcgl_test::glcallsIsDraw(r.entry); - } else { - e.gl.push_back(r); +// resetDraw >= 0: forget what the program of the trace's resetDraw-th replica +// draw (DrawBegin) holds when that draw first touches it -- that draw then sees +// only what it writes itself, whatever an earlier draw left. +UniformView replayUniforms(const Trace &t, int resetDraw = -1) { + static const int link = cvcgl_test::glcallsEntry("LinkProgram"); + static const int del = cvcgl_test::glcallsEntry("DeleteProgram"); + UniformView v; + auto &state = v.end; + int draw = -1; + bool resetPending = false; + for (size_t i = 0; i < t.size(); ++i) { + const TraceRec &r = t[i]; + if (r.marker()) { + if (isStage(r, DrawStage::DrawBegin) && ++draw == resetDraw) + resetPending = true; + continue; + } + const bool uniform = cvcgl_test::glcallsIsUniform(r.entry); + const bool drawCall = cvcgl_test::glcallsIsDraw(r.entry); + if (resetPending && (uniform || drawCall)) { + state.erase(r.ctx); + resetPending = false; + } + if ((r.entry == link || r.entry == del) && r.args.size() >= sizeof(unsigned)) { + unsigned program = 0; // a (re)link resets every uniform; a deleted name is reused + std::memcpy(&program, r.args.data(), sizeof program); + state.erase(program); + } else if (uniform) { + for (auto &w : cvcgl_test::glcallsUniformWrites(r)) + state[r.ctx][w.location] = std::move(w.value); + } else if (drawCall) { + v.calls.push_back({i, r.ctx, state[r.ctx]}); + } + } + return v; +} + +// A UniformWrite value ("3f:") as its type and numbers. +std::string uniformValue(const std::string &v) { + const size_t colon = v.find(':'); + if (colon == std::string::npos) + return v; + const std::string type = v.substr(0, colon); + const bool f = type.find('f') != std::string::npos; // "3f", "Matrix4f", "Matrix4fT" + const bool u = type.find("ui") != std::string::npos; + std::string out = type + " ("; + char buf[32]; + for (size_t at = colon + 1, k = 0; at + 4 <= v.size() && k < 16; at += 4, ++k) { + if (f) { + float x; + std::memcpy(&x, v.data() + at, sizeof x); + std::snprintf(buf, sizeof buf, k ? " %g" : "%g", double(x)); + } else if (u) { + unsigned x; + std::memcpy(&x, v.data() + at, sizeof x); + std::snprintf(buf, sizeof buf, k ? " %u" : "%u", x); + } else { + int x; + std::memcpy(&x, v.data() + at, sizeof x); + std::snprintf(buf, sizeof buf, k ? " %d" : "%d", x); + } + out += buf; + } + return out + ")"; +} + +// Do two traces' draw calls see the same uniforms? The traces issue the same +// draw calls in the same order (the trace checks pin that), and at each one the +// bound program must hold the same value at every location either trace wrote: +// written to the same value in both, or written in neither so far (both see +// the value from before the trace). Where a draw call sees such an inherited +// value it reads what the previous frame left -- for a repeated frame, what this +// frame leaves -- so for those (program, location)s the two frames' end values +// must agree too. End values no draw call reads before writing are not +// compared: on AllCellTypes a lines-only draw leaves the empty polys/strips +// blocks' cellType and primitiveSize in its program, Fast leaves the lines +// values, and every cell-type block writes both before it draws. +bool sameUniformsSeen(const UniformView &a, const UniformView &b, const Trace &ta, + std::string *why) { + const auto describe = [](const std::map &m, int loc) { + const auto it = m.find(loc); + return it == m.end() ? std::string("(not written yet)") : uniformValue(it->second); + }; + if (a.calls.size() != b.calls.size()) { + if (why) + *why = fmt("%.0f vs %.0f draw calls", double(a.calls.size()), double(b.calls.size())); + return false; + } + static const std::map kNone; + const auto endOf = [](const UniformView &v, + unsigned program) -> const std::map & { + const auto it = v.end.find(program); + return it == v.end.end() ? kNone : it->second; + }; + std::set> inherited; + for (size_t k = 0; k < a.calls.size(); ++k) { + const auto &x = a.calls[k], &y = b.calls[k]; + if (x.program != y.program) { + if (why) + *why = fmt("draw call %.0f: program %.0f vs %.0f", double(k), double(x.program), + double(y.program)); + return false; + } + std::set locations; + for (const auto *m : {&x.written, &y.written, &endOf(a, x.program), &endOf(b, x.program)}) + for (const auto &kv : *m) + locations.insert(kv.first); + for (int loc : locations) { + const auto i = x.written.find(loc), j = y.written.find(loc); + if (i == x.written.end() && j == y.written.end()) { + inherited.insert({x.program, loc}); + continue; } + if (i != x.written.end() && j != y.written.end() && i->second == j->second) + continue; + if (why) + *why = fmt("draw call %.0f (trace #%.0f, ", double(k), double(x.at)) + + cvcgl_test::glcallsDescribe(ta[x.at]) + fmt("): location %.0f: ", double(loc)) + + describe(x.written, loc) + " vs " + describe(y.written, loc); + return false; + } + } + for (const auto &[program, loc] : inherited) { + const auto &ea = endOf(a, program), &eb = endOf(b, program); + const auto i = ea.find(loc), j = eb.find(loc); + const bool same = (i == ea.end() && j == eb.end()) || + (i != ea.end() && j != eb.end() && i->second == j->second); + if (!same) { + if (why) + *why = fmt("program %.0f location %.0f, read before the frame writes it: the frame " + "leaves ", + double(program), double(loc)) + + describe(ea, loc) + " vs " + describe(eb, loc); + return false; + } + } + return true; +} + +// For each replica draw (DrawBegin) of a trace: its cell-type blocks that issued +// no draw call. +std::vector> emptyBlocks(const Trace &t) { + std::vector> out; + bool inBlock = false, drew = false; + int type = 0; + for (const auto &r : t) { + if (!r.marker()) { + drew = drew || (inBlock && cvcgl_test::glcallsIsDraw(r.entry)); continue; } if (isStage(r, DrawStage::DrawBegin)) { - ++e.draws; - skipAllowed = r.ctx != 0; - e.unskippableDraws += !skipAllowed; - } else if (isStage(r, DrawStage::LookupBegin)) { - ++e.lookups; + out.emplace_back(); } else if (isStage(r, DrawStage::AgentBegin)) { inBlock = true; - blockDraws = false; - block.clear(); + drew = false; + type = static_cast(r.ctx); } else if (isStage(r, DrawStage::AgentEnd)) { inBlock = false; - if (skipAllowed && !blockDraws) { - ++e.blocksRemoved; - } else { - ++e.blocksKept; - e.gl.insert(e.gl.end(), block.begin(), block.end()); - } + if (!drew && !out.empty()) + out.back().push_back(type); + } + } + return out; +} + +// `t` without the cell-type blocks (AgentBegin..AgentEnd, markers included) +// that drop(draw, type) selects; draw counts the trace's replica draws. +template Trace without(const Trace &t, Drop drop) { + Trace out; + out.reserve(t.size()); + int draw = -1; + bool dropping = false; + for (const auto &r : t) { + if (isStage(r, DrawStage::DrawBegin)) + ++draw; + if (isStage(r, DrawStage::AgentBegin) && drop(draw, static_cast(r.ctx))) + dropping = true; + if (!dropping) + out.push_back(r); + if (isStage(r, DrawStage::AgentEnd)) + dropping = false; + } + return out; +} + +// What Fast's GL trace must be, derived from an AllCellTypes trace alone -- +// never from the mapper's own skip decision (coincidentSkipSafe). A replica +// draw's empty cell-type blocks may go iff removing them (that draw's alone) +// changes no uniform any draw call sees, with the draw's program state on +// entry treated as unknown (what earlier draws leave in a shared program is not +// the draw's to rely on): the blocks' writes either reach no draw call or are +// written again, to the same value, before one reads them. Everything else +// stays -- including each full lookup's GL calls, which are the +// ReadyShaderProgram(program) tail (compile if released, bind) that the re-bind +// replacing it issues too. +struct Expected { + Trace trace; // with the markers of what stays + int blocksRemoved = 0, blocksKept = 0, lookups = 0, draws = 0, fallbacks = 0; + std::string firstFallback; // why the first draw that keeps its empty blocks must +}; + +Expected expectFast(const Trace &all) { + Expected e; + const auto empty = emptyBlocks(all); + const auto isEmpty = [&](int d, int type) { + const auto &v = empty[static_cast(d)]; + return std::find(v.begin(), v.end(), type) != v.end(); + }; + std::vector removable(empty.size(), false); + for (size_t d = 0; d < empty.size(); ++d) { + if (empty[d].empty()) + continue; + const int draw = static_cast(d); + const Trace removed = + without(all, [&](int dd, int type) { return dd == draw && isEmpty(dd, type); }); + std::string why; + removable[d] = + sameUniformsSeen(replayUniforms(all, draw), replayUniforms(removed, draw), all, &why); + if (!removable[d]) { + ++e.fallbacks; + if (e.firstFallback.empty()) + e.firstFallback = fmt("draw %.0f: ", double(d)) + why; } } + e.trace = without(all, [&](int d, int type) { + return d >= 0 && removable[static_cast(d)] && isEmpty(d, type); + }); + e.draws = static_cast(empty.size()); + for (size_t d = 0; d < empty.size(); ++d) + (removable[d] ? e.blocksRemoved : e.blocksKept) += static_cast(empty[d].size()); + e.lookups = countStage(all, DrawStage::LookupBegin); return e; } @@ -1034,32 +1244,53 @@ void checkTraces(const std::string &what, const Trace &stock, const Trace &all, why.clear(); // The draw state is read off the full AllCellTypes trace, before the blocks // go: what each draw saw there is what it must see on Fast. - const Expected e = expectFast(asDrawState(all)); - const Trace got = glOnly(asDrawState(fast)); - const bool derived = sameTrace(got, glOnly(e.gl), &why); + const Trace allState = asDrawState(all), fastState = asDrawState(fast); + const Expected e = expectFast(allState); + const Trace got = glOnly(fastState); + const bool derived = sameTrace(got, glOnly(e.trace), &why); if (!derived) - dumpTraces(what, {{"all", all}, {"fast", fast}, {"expected", e.gl}, {"fast_state", got}}); + dumpTraces(what, {{"all", all}, {"fast", fast}, {"expected", e.trace}, {"fast_state", got}}); check(derived, - what + ": Fast trace == AllCellTypes trace minus its empty cell-type blocks (cached " - "point size / line width compared at the draws)", + what + ": Fast trace == AllCellTypes trace minus the empty cell-type blocks whose " + "removal no draw call's uniforms see (cached point size / line width compared " + "at the draws)", derived ? fmt("%.0f calls; %.0f blocks removed, %.0f kept", double(got.size()), - double(e.blocksRemoved), double(e.blocksKept)) + double(e.blocksRemoved), double(e.blocksKept)) + + (e.firstFallback.empty() ? "" : "; kept e.g. " + e.firstFallback) : why); + why.clear(); + // Fast's own trace, replayed: every draw call sees the uniforms it sees on + // AllCellTypes, and the frame leaves what the next frame's draws read. + const UniformView allUniforms = replayUniforms(allState), + fastUniforms = replayUniforms(fastState); + const bool uniforms = sameUniformsSeen(allUniforms, fastUniforms, allState, &why); + if (!uniforms) + dumpTraces(what + " (uniforms)", {{"all", allState}, {"fast", fastState}}); + check(uniforms, + what + ": every Fast draw call sees AllCellTypes' uniform state (every location of " + "its program), and the frame leaves the values the next frame reads", + uniforms ? fmt("%.0f draw calls", double(fastUniforms.calls.size())) : why); + // The mapper's own markers and stats: counts only. const int skipped = countStage(fast, DrawStage::AgentSkipped); check(e.blocksRemoved == skipped && (skipped > 0) == mayDraw && static_cast(skipped) == double(fastStats.cellTypesSkipped), - what + ": the blocks derived as removable are exactly the ones Fast skipped", + what + ": Fast skipped as many blocks as derived", fmt("derived %.0f, Fast skipped %.0f (stats %.0f)", double(e.blocksRemoved), double(skipped), double(fastStats.cellTypesSkipped))); const int looked = countStage(fast, DrawStage::LookupBegin); const int rebound = countStage(fast, DrawStage::RebindBegin); + int fastFallbacks = 0; + for (const auto &r : fast) + fastFallbacks += isStage(r, DrawStage::DrawBegin) && r.ctx == 0; check(looked + rebound == e.lookups && countStage(fast, DrawStage::DrawBegin) == e.draws && - static_cast(e.unskippableDraws) == double(fastStats.coincidentFallbacks), + e.fallbacks == fastFallbacks && + static_cast(e.fallbacks) == double(fastStats.coincidentFallbacks), what + ": every AllCellTypes lookup is a Fast lookup or re-bind; same draws, same " "fallbacks", fmt("lookups %.0f = %.0f + %.0f re-bound", double(e.lookups), double(looked), double(rebound)) + - fmt(", fallbacks %.0f", double(e.unskippableDraws))); + fmt(", fallbacks derived %.0f, Fast %.0f (stats %.0f)", double(e.fallbacks), + double(fastFallbacks), double(fastStats.coincidentFallbacks))); } // The hardware selector's passes on one draw path: what it selected, the GL @@ -1363,13 +1594,22 @@ void testGuards(cvc::app &app) { Probe &vv = s.probe("vertexVis", terrainPoly(6, -5, -5, 4)); vv.actor->GetProperty()->VertexVisibilityOn(); vv.actor->GetProperty()->SetVertexColor(1, 0, 0); + // The mode is set before the first render: vtkGLSLModCoincidentTopology + // declares the offset uniforms only if the shader is BUILT under an offset, + // and nothing rebuilds it when the global mode changes later -- switched on + // after the build, the offsets never reach the GPU on any path. + const int savedMode = vtkMapper::GetResolveCoincidentTopology(); + vtkMapper::SetResolveCoincidentTopologyToPolygonOffset(); s.open(); s.sr->setCamera(0, -14, 10, 0, 0, 0, 0, 0, 1, 40.0, 1.0, 100.0); s.sr->render(); glcallsInstall(); + for (const Probe *p : {&zero, &shifted}) { + vtkShaderProgram *prog = p->mapper->vtkDrawTexturedElements::GetShaderProgram(); + check(prog && prog->IsUniformUsed("cOffset") && prog->IsUniformUsed("cFactor"), + "POLYGON_OFFSET mode: " + p->name + "'s program carries the depth offset uniforms"); + } - const int savedMode = vtkMapper::GetResolveCoincidentTopology(); - vtkMapper::SetResolveCoincidentTopologyToPolygonOffset(); for (DrawPath p : {DrawPath::Stock, DrawPath::Fast}) runPath(s, p); const PathRun stock = runPath(s, DrawPath::Stock); diff --git a/src/cvcGL/test/gl_call_counter.cpp b/src/cvcGL/test/gl_call_counter.cpp index 509f1bba6..66098e38e 100644 --- a/src/cvcGL/test/gl_call_counter.cpp +++ b/src/cvcGL/test/gl_call_counter.cpp @@ -573,6 +573,45 @@ void glcallsMark(int kind, int arg) { bool glcallsIsDraw(int entry) { return entry >= 0 && entry < E_COUNT && g_cls[entry] == kDraw; } +bool glcallsIsUniform(int entry) { + return entry >= 0 && entry < E_COUNT && g_cls[entry] == kUniform; +} + +std::vector glcallsUniformWrites(const TraceRec &r) { + std::vector out; + if (r.marker() || !glcallsIsUniform(r.entry) || r.args.size() < sizeof(GLint)) + return out; + // The args as recorded: location, then count for the *v forms, then the + // transpose byte for the matrices, then the values. The type is the name less + // "Uniform" and the trailing "v": "3f", "1ui", "Matrix4x3f". + const std::string name = g_names[r.entry]; + const bool matrix = name.compare(0, 13, "UniformMatrix") == 0; + const bool array = name.back() == 'v'; + std::string type = name.substr(7, name.size() - 7 - (array ? 1 : 0)); + GLint location = 0; + std::memcpy(&location, r.args.data(), sizeof location); + size_t at = sizeof location; + GLsizei count = 1; + if (array) { + if (r.args.size() < at + sizeof count) + return out; + std::memcpy(&count, r.args.data() + at, sizeof count); + at += sizeof count; + } + if (matrix) { + if (r.args.size() < at + 1) + return out; + type += r.args[at] ? "T" : ""; + ++at; + } + if (location < 0 || count <= 0) + return out; + const size_t each = (r.args.size() - at) / static_cast(count); + for (GLsizei k = 0; k < count; ++k) + out.push_back({location + k, type + ':' + r.args.substr(at + k * each, each)}); + return out; +} + int glcallsEntry(const std::string &name) { for (int i = 0; i < E_COUNT; ++i) if (name == g_names[i]) diff --git a/src/cvcGL/test/gl_call_counter.h b/src/cvcGL/test/gl_call_counter.h index ed397f070..b93adff91 100644 --- a/src/cvcGL/test/gl_call_counter.h +++ b/src/cvcGL/test/gl_call_counter.h @@ -88,6 +88,18 @@ Trace glcallsTraceStop(); void glcallsMark(int kind, int arg); // glDraw* / glDispatchCompute. bool glcallsIsDraw(int entry); +// glUniform* (scalar, vector or matrix). +bool glcallsIsUniform(int entry); +// A traced glUniform* call as the uniform locations it sets (in its bound +// program, the record's ctx): an upload of n array elements is n writes, at +// location, location + 1, ... Each value is the element's type and bytes, so +// glUniform1i(l, v) and glUniform1iv(l, 1, &v) set the same value. Empty for any +// other record, and for location -1 (which GL ignores). +struct UniformWrite { + int location = 0; + std::string value; +}; +std::vector glcallsUniformWrites(const TraceRec &r); // The index of an entry point ("PointSize"), or -1. int glcallsEntry(const std::string &name); // "glUniform1i prog=3 [05 00 00 00 01 00 00 00]" / "mark 5 (2)". From 1410a72d09bdda58f1c412fa022e38083cd07385 Mon Sep 17 00:00:00 2001 From: Joe Rivera Date: Thu, 1 Oct 2026 19:58:47 -0500 Subject: [PATCH 07/18] build(cvcGL): re-run the low-memory replica guard when the hashed VTK headers change The configure-time check hashes the installed vtkOpenGLLowMemoryPolyDataMapper.h, vtkDrawTexturedElements.h and vtkGLSLModCoincidentTopology.h, but only when CMake runs: a patched VTK re-installed into the same prefix under an existing build directory left the replica on until someone re-ran the configure. The headers that exist are now CMAKE_CONFIGURE_DEPENDS, so the next build re-runs the configure and the check (build.ninja lists all three as RERUN_CMAKE inputs). --- src/cvcGL/CMakeLists.txt | 3 +++ 1 file changed, 3 insertions(+) diff --git a/src/cvcGL/CMakeLists.txt b/src/cvcGL/CMakeLists.txt index 861ed35c3..0681328d2 100644 --- a/src/cvcGL/CMakeLists.txt +++ b/src/cvcGL/CMakeLists.txt @@ -75,6 +75,9 @@ if(VTK_VERSION VERSION_EQUAL "9.5.0") list(GET _pair 0 _hdr) list(GET _pair 1 _want) if(EXISTS "${_cvc_lowmem_inc}/${_hdr}") + # A VTK re-installed into the same prefix (a patched recipe) re-runs the + # configure, and with it this check, on the next build. + set_property(DIRECTORY APPEND PROPERTY CMAKE_CONFIGURE_DEPENDS "${_cvc_lowmem_inc}/${_hdr}") file(READ "${_cvc_lowmem_inc}/${_hdr}" _text) string(REPLACE "\r\n" "\n" _text "${_text}") string(SHA256 _have "${_text}") From f01d2f5b471d3e23af3cb724a714ad482529ee00 Mon Sep 17 00:00:00 2001 From: Joe Rivera Date: Thu, 1 Oct 2026 19:58:57 -0500 Subject: [PATCH 08/18] docs: list every third-party component in THIRD_PARTY_NOTICES.md The file said the notices for code under other licenses "are recorded here" but listed only XmlRpc++ and the VTK-derived mapper code. An audit of the tree (third_party dirs, license and copyright headers, "derived from / ported from" notes) and of what the libraries compile in adds, each with its notice exactly as the source carries it (checked mechanically against the files): - stb_image 2.30 / stb_image_write 1.16 (src/cvc/image/third_party/; MIT or public domain, Copyright (c) 2017 Sean Barrett) -- always compiled into libcvc by stb_io.cpp. - PocketFFT (BSD-3-Clause; not vendored -- the cvcpkg pocketfft package, commit c90e55b3d5...) -- its header-only templates are compiled into libcvc's volume_ops with CVC_FFT_PROVIDER=pocketfft, the default. - readvtk/writevtk in src/cvc/volume/vtk_io.cpp (Technical University of Denmark; no license terms stated). - The contour library (Dan Schikore, Emilio Camahort; no license terms stated) and kazlib's dict.c/dict.h (Kaz Kylheku's free software license), compiled with CVC_ENABLE_MESHER (default ON). - sdf (Xiaoyu Zhang, LGPL-2.1) and mtxlib (Dante Treglia II and Mark A. DeLoura, 2000), compiled with CVC_ENABLE_SDF (default ON). - VTK's marching-cubes triCases table in inc/cvc/volren/detail/mc_tables.h, under the same VTK BSD-3 notice as the mapper code. - ABYSSAL (MIT, Copyright (c) 2026 Davi (Token-Gremlin)), ported in src/cvcGL/examples/OceanFFT.{h,cpp} for lsystem_coast and cvcgl_ocean_fft; the notice is upstream's LICENSE verbatim. - The vendored CMake modules (Kitware BSD, FindPETSc, NVIDIA/SCI FindCUDA MIT), marked as build scripts that nothing compiles or installs. The XmlRpc++ entry now quotes its header notice and says which files carry it (the headers; the .cpp files carry none, base64.h only its author line). The README's license paragraph points at the full list. --- README.md | 9 +- THIRD_PARTY_NOTICES.md | 423 +++++++++++++++++++++++++++++++++++++++-- 2 files changed, 415 insertions(+), 17 deletions(-) diff --git a/README.md b/README.md index a64085658..c8df33341 100644 --- a/README.md +++ b/README.md @@ -706,9 +706,12 @@ Copyright (c) 2002-2003 Chris Morley and are used under LGPL-2.1-or-later. Portions of `src/cvcGL/LowMemoryPolyDataMapper.cpp` are derived from VTK 9.5.0, Copyright (c) Ken Martin, Will Schroeder, Bill Lorensen, and are used under the -BSD-3-Clause license. The full notices for third-party code are in -[THIRD_PARTY_NOTICES.md](THIRD_PARTY_NOTICES.md), which is installed with the -documentation. +BSD-3-Clause license. That and the other third-party code in, or compiled into, +libcvc and cvcGL -- stb_image, PocketFFT, the contour library, kazlib, sdf, +mtxlib, the DTU `.vtk` reader, VTK's marching-cubes table, the ABYSSAL ocean +port in the cvcGL examples, and the vendored CMake modules -- are listed with +their notices in [THIRD_PARTY_NOTICES.md](THIRD_PARTY_NOTICES.md), which is +installed with the documentation. ## Credits diff --git a/THIRD_PARTY_NOTICES.md b/THIRD_PARTY_NOTICES.md index 110c3364c..f83f4be29 100644 --- a/THIRD_PARTY_NOTICES.md +++ b/THIRD_PARTY_NOTICES.md @@ -1,24 +1,284 @@ # Third-party notices libcvc is licensed under the GNU Lesser General Public License, version 2.1 -(see [LICENSE](LICENSE)). Some of its source is derived from, or bundles, code -under other licenses. Those notices are recorded here and installed with the -documentation. +(see [LICENSE](LICENSE)). This file records the code in this repository -- and +the code compiled into its libraries from a dependency -- that comes from, or +is derived from, authors other than The University of Texas at Austin, each +with its notice exactly as the source carries it. It is installed with the +documentation by the libcvc and cvcGL installs and ships in the cvcgl-examples +package. + +Compiled into libcvc: + +- [stb_image / stb_image_write](#stb_image-230-and-stb_image_write-116) +- [PocketFFT](#pocketfft) +- [XmlRpc++](#xmlrpc) +- [readvtk / writevtk](#readvtk--writevtk-technical-university-of-denmark) +- [The contour library](#the-contour-library) and + [kazlib's dictionary](#kazlib-dictionary) +- [sdf](#sdf-signed-distance-function) and [mtxlib](#mtxlib) +- [VTK](#vtk-the-visualization-toolkit) (also compiled into libcvcGL) + +Compiled into the cvcGL examples: [ABYSSAL](#abyssal). + +In the build scripts only: [CMake modules](#cmake-modules-build-scripts-only). + +## stb_image 2.30 and stb_image_write 1.16 + +`src/cvc/image/third_party/stb_image.h` and `stb_image_write.h` (Sean Barrett +and contributors), compiled into libcvc by `src/cvc/image/stb_io.cpp`. Both +files end with this notice: + +``` +This software is available under 2 licenses -- choose whichever you prefer. +------------------------------------------------------------------------------ +ALTERNATIVE A - MIT License +Copyright (c) 2017 Sean Barrett +Permission is hereby granted, free of charge, to any person obtaining a copy of +this software and associated documentation files (the "Software"), to deal in +the Software without restriction, including without limitation the rights to +use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +of the Software, and to permit persons to whom the Software is furnished to do +so, subject to the following conditions: +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. +------------------------------------------------------------------------------ +ALTERNATIVE B - Public Domain (www.unlicense.org) +This is free and unencumbered software released into the public domain. +Anyone is free to copy, modify, publish, use, compile, sell, or distribute this +software, either in source code form or as a compiled binary, for any purpose, +commercial or non-commercial, and by any means. +In jurisdictions that recognize copyright laws, the author or authors of this +software dedicate any and all copyright interest in the software to the public +domain. We make this dedication for the benefit of the public at large and to +the detriment of our heirs and successors. We intend this dedication to be an +overt act of relinquishment in perpetuity of all present and future rights to +this software under copyright law. +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN +ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION +WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. +``` + +## PocketFFT + +Not vendored: `pocketfft_hdronly.h` comes from the cvcpkg `pocketfft` package +(github.com/mreineck/pocketfft, `cpp` branch, commit +`c90e55b3d529f8efa40ed01a20de22405f45fc65`). Being header-only, its code is +compiled into libcvc (`src/cvc/volume/volume_ops.cpp`) whenever +`CVC_FFT_PROVIDER=pocketfft`, the default. The header's notice: + +``` +This file is part of pocketfft. + +Copyright (C) 2010-2024 Max-Planck-Society +Copyright (C) 2019-2020 Peter Bell + +For the odd-sized DCT-IV transforms: + Copyright (C) 2003, 2007-14 Matteo Frigo + Copyright (C) 2003, 2007-14 Massachusetts Institute of Technology + +For the prev_good_size search: + Copyright (C) 2024 Tan Ping Liang, Peter Bell + +For the safeguards against integer overflow in good_size search: + Copyright (C) 2024 Cris Luengo + +Authors: Martin Reinecke, Peter Bell + +All rights reserved. + +Redistribution and use in source and binary forms, with or without modification, +are permitted provided that the following conditions are met: + +* Redistributions of source code must retain the above copyright notice, this + list of conditions and the following disclaimer. +* Redistributions in binary form must reproduce the above copyright notice, this + list of conditions and the following disclaimer in the documentation and/or + other materials provided with the distribution. +* Neither the name of the copyright holder nor the names of its contributors may + be used to endorse or promote products derived from this software without + specific prior written permission. + +THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND +ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED +WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE +DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE FOR +ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES +(INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; +LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON +ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT +(INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS +SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. +``` ## XmlRpc++ -The bundled XmlRpc++ sources under `inc/xmlrpc/` and `src/xmlrpc/` are -Copyright (c) 2002-2003 Chris Morley and are used under the GNU Lesser General -Public License, version 2.1 or later. +The bundled XmlRpc++ sources under `inc/xmlrpc/` and `src/xmlrpc/`, built only +with `CVC_USING_XMLRPC=ON` (default OFF), are used under the GNU Lesser General +Public License, version 2.1 or later. The headers in `inc/xmlrpc/` carry this +notice (the `.cpp` files carry none of their own): -## VTK (The Visualization Toolkit) 9.5.0 +``` +// XmlRpc++ Copyright (c) 2002-2003 by Chris Morley +// This library is free software; you can redistribute it and/or +// modify it under the terms of the GNU Lesser General Public +// License as published by the Free Software Foundation; either +// version 2.1 of the License, or (at your option) any later version. +// +// This library is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU +// Lesser General Public License for more details. +// +// You should have received a copy of the GNU Lesser General Public +// License along with this library; if not, write to the Free Software +// Foundation, Inc., 59 Temple Place, Suite 330, Boston, MA 02111-1307 +``` + +Its `inc/xmlrpc/base64.h` carries only: + +``` +// base64.hpp +// Autor Konstantin Pilipchuk +// mailto:lostd@ukr.net +``` + +## readvtk / writevtk (Technical University of Denmark) + +The legacy `.vtk` reader and writer in `src/cvc/volume/vtk_io.cpp`, compiled +into libcvc, carry this notice and no license terms of their own: + +``` +/* readvtk/writevtk: + * Copyright (c) The Technical University of Denmark + * Author: Ojaswa Sharma + * E-mail: os@imm.dtu.dk + * File: vtkIO.h + * .vtk I/O + */ +``` + +## The contour library -Portions of `src/cvcGL/LowMemoryPolyDataMapper.cpp` -- the regions between -`BEGIN VTK 9.5.0-DERIVED` and `END VTK 9.5.0-DERIVED` -- are derived from -VTK 9.5.0 (`Rendering/OpenGL2`: `vtkGLSLModCoincidentTopology.cxx`, -`vtkOpenGLLowMemoryPolyDataMapper.cxx`, `vtkOpenGLLowMemoryCellTypeAgent.cxx`, -`vtkOpenGLLowMemoryVerticesAgent.cxx`, `vtkOpenGLLowMemoryLinesAgent.cxx`, -`vtkOpenGLLowMemoryPolygonsAgent.cxx`). They are used under VTK's license: +`src/cvc/geometry/cvc-mesher/contour/` (with copies of its `cubes.h` in +`cvc-mesher/LBIE/` and `SDF/SignDistanceFunction_v2/`), compiled into libcvc +with `CVC_ENABLE_MESHER` (default ON). Its files carry these lines and no +license terms of their own; files without a header belong to the same library: + +``` +Copyright (c) 1997 Dan Schikore +Copyright (c) 1997 Dan Schikore - modified by Emilio Camahort, 1999 +Copyright (c) 1997 Dan Schikore - updated by Emilio Camahort, 1998 +Copyright (c) 1997 Dan Schikore - updated by Emilio Camahort, 1999 +Copyright (c) 1998 Emilio Camahort +Copyright (c) 1998 Emilio Camahort, Dan Schikore +Copyright (c) 1998 Emilio Camahort - ecamahor@cs.utexas.edu +Copyright (c) 1999 Emilio Camahort +``` + +and, in the priority-queue headers (`ipqueue.h` and its siblings): + +``` + * Author: Dan Schikore + * + * Adapted from pqueue.h + * Ported to C++ by Raymund Merkert - June 1995 + * Changes by Fausto Bernardini - Sept 1995 +``` + +## kazlib dictionary + +`src/cvc/geometry/cvc-mesher/contour/dict.c` and `dict.h`, compiled with the +contour library: + +``` +/* + * Dictionary Abstract Data Type + * Copyright (C) 1997 Kaz Kylheku + * + * Free Software License: + * + * All rights are reserved by the author, with the following exceptions: + * Permission is granted to freely reproduce and distribute this software, + * possibly in exchange for a fee, provided that this copyright notice appears + * intact. Permission is also granted to adapt this software to produce + * derivative works, as long as the modified versions carry this copyright + * notice and additional notices stating that the work has been modified. + * This source code may be translated into executable form and incorporated + * into proprietary software; there is no requirement for such software to + * contain a copyright notice related to this source. +``` + +## sdf (signed distance function) + +`src/cvc/geometry/SDF/SignDistanceFunction_v2/`, compiled into libcvc with +`CVC_ENABLE_SDF` (default ON), is used under the GNU Lesser General Public +License, version 2.1. Its files that carry a header carry: + +``` + Copyright (c): Xiaoyu Zhang (xiaoyu@csusm.edu) + + This file is part of sdf (signed distance function). + + sdf is free software; you can redistribute it and/or + modify it under the terms of the GNU Lesser General Public + License version 2.1 as published by the Free Software Foundation. + + sdf is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU + Lesser General Public License for more details. + + You should have received a copy of the GNU Lesser General Public + License along with this library; if not, write to the Free Software + Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA +``` + +and, in `diskio.h` and `geom.h`, `(c) 2000 Xiaoyu Zhang`. + +## mtxlib + +`src/cvc/geometry/SDF/SignDistanceFunction_v2/mtxlib.h` and `mtxlib.cpp` (C++ +Matrix Library 2.6), compiled with sdf. Portions Copyright (C) Dante Treglia II +and Mark A. DeLoura, 2000. The notice, as `mtxlib.h` carries it (`mtxlib.cpp` +has the same text): + +``` +/* Copyright (C) Dante Treglia II and Mark A. DeLoura, 2000. + * All rights reserved worldwide. + * + * This software is provided "as is" without express or implied + * warranties. You may freely copy and compile this source into + * applications you distribute provided that the copyright text + * below is included in the resulting source code, for example: + * "Portions Copyright (C) Dante Treglia II and Mark A. DeLoura, 2000" + */ +``` + +## VTK (The Visualization Toolkit) + +- Portions of `src/cvcGL/LowMemoryPolyDataMapper.cpp` (libcvcGL) -- the + regions between `BEGIN VTK 9.5.0-DERIVED` and `END VTK 9.5.0-DERIVED` -- are + derived from VTK 9.5.0 (`Rendering/OpenGL2`: `vtkGLSLModCoincidentTopology.cxx`, + `vtkOpenGLLowMemoryPolyDataMapper.cxx`, `vtkOpenGLLowMemoryCellTypeAgent.cxx`, + `vtkOpenGLLowMemoryVerticesAgent.cxx`, `vtkOpenGLLowMemoryLinesAgent.cxx`, + `vtkOpenGLLowMemoryPolygonsAgent.cxx`). The same notice is reproduced at the + top of that source file. +- The marching-cubes triangulation table in `inc/cvc/volren/detail/mc_tables.h` + (libcvc's volume renderer) is VTK's `triCases` from `vtkMarchingCubesCases.h`, + by way of volrover's libiso. + +Both are used under VTK's license: ``` SPDX-FileCopyrightText: Copyright (c) Ken Martin, Will Schroeder, Bill Lorensen @@ -53,4 +313,139 @@ OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. ``` -The same notice is reproduced at the top of that source file. +## ABYSSAL + +The GPU FFT ocean in `src/cvcGL/examples/OceanFFT.h` and `OceanFFT.cpp` -- its +spectrum, time evolution, butterfly IFFT and assembly -- is a C++/VTK port of +ABYSSAL (github.com/Token-Gremlin/natural-disasters). It is built into the +`lsystem_coast` example (the cvcgl-examples package) and the `cvcgl_ocean_fft` +test. Upstream's `LICENSE`: + +``` +MIT License + +Copyright (c) 2026 Davi (Token-Gremlin) + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. +``` + +## CMake modules (build scripts only) + +These run at configure time; nothing from them is compiled or installed. + +`CMake/BundleUtilities.cmake`: + +``` +#============================================================================= +# CMake - Cross Platform Makefile Generator +# Copyright 2000-2009 Kitware, Inc., Insight Software Consortium +# All rights reserved. +# +# Redistribution and use in source and binary forms, with or without +# modification, are permitted provided that the following conditions +# are met: +# +# * Redistributions of source code must retain the above copyright +# notice, this list of conditions and the following disclaimer. +# +# * Redistributions in binary form must reproduce the above copyright +# notice, this list of conditions and the following disclaimer in the +# documentation and/or other materials provided with the distribution. +# +# * Neither the names of Kitware, Inc., the Insight Software Consortium, +# nor the names of their contributors may be used to endorse or promote +# products derived from this software without specific prior written +# permission. +# +# THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS +# "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT +# LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR +# A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT +# HOLDER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, +# SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT +# LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, +# DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY +# THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT +# (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE +# OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. +#============================================================================= +``` + +`CMake/GetPrerequisites.cmake` (the `Copyright.txt` it refers to is not in this +repository): + +``` +#============================================================================= +# Copyright 2008-2009 Kitware, Inc. +# +# Distributed under the OSI-approved BSD License (the "License"); +# see accompanying file Copyright.txt for details. +# +# This software is distributed WITHOUT ANY WARRANTY; without even the +# implied warranty of MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. +# See the License for more information. +#============================================================================= +# (To distribute this file outside of CMake, substitute the full +# License text for the above reference.) +``` + +`CMake/FindPETSc.cmake` (the `COPYING-CMAKE-SCRIPTS` it refers to is not in this +repository): + +``` +# Redistribution and use is allowed according to the terms of the BSD license. +# For details see the accompanying COPYING-CMAKE-SCRIPTS file. +``` + +`CMake/cuda/FindCUDA.cmake` and `CMake/cuda/FindCUDA/{make2cmake,parse_cubin,run_nvcc}.cmake` +(`run_nvcc.cmake` names NVIDIA only): + +``` +# James Bigler, NVIDIA Corp (nvidia.com - jbigler) +# Abe Stephens, SCI Institute -- http://www.sci.utah.edu/~abe/FindCuda.html +# +# Copyright (c) 2008 - 2009 NVIDIA Corporation. All rights reserved. +# +# Copyright (c) 2007-2009 +# Scientific Computing and Imaging Institute, University of Utah +# +# This code is licensed under the MIT License. See the FindCUDA.cmake script +# for the text of the license. + +# The MIT License +# +# License for the specific language governing rights and limitations under +# Permission is hereby granted, free of charge, to any person obtaining a +# copy of this software and associated documentation files (the "Software"), +# to deal in the Software without restriction, including without limitation +# the rights to use, copy, modify, merge, publish, distribute, sublicense, +# and/or sell copies of the Software, and to permit persons to whom the +# Software is furnished to do so, subject to the following conditions: +# +# The above copyright notice and this permission notice shall be included +# in all copies or substantial portions of the Software. +# +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS +# OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL +# THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING +# FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER +# DEALINGS IN THE SOFTWARE. +``` From 75ce09d0ef12199c35acde1ef6c4f7fc421fc170 Mon Sep 17 00:00:00 2001 From: Joe Rivera Date: Thu, 1 Oct 2026 19:59:06 -0500 Subject: [PATCH 09/18] cvcpkg: declare BSD-3-Clause on the cvcGL packages; ship the notices with cvcgl-examples License metadata for the packages that carry the VTK-derived code in libcvcGL (LowMemoryPolyDataMapper.cpp), as SPDX expressions (the recipe schema's license field is an SPDX expression; jack, libjpeg-turbo and others already use AND): - cvcgl, cvcgl-cuda: "LGPL-2.1-only AND BSD-3-Clause". - cvcgl-examples: "LGPL-2.1-only AND BSD-3-Clause AND MIT" -- the demos link cvcGL statically, and lsystem_coast's OceanFFT is a port of ABYSSAL (MIT). libcvc keeps LGPL-2.1-only here: its package does not contain cvcGL (the recipe leaves CVC_BUILD_CVCGL at its OFF default). cvcgl-examples did not ship THIRD_PARTY_NOTICES.md: build.sh installs only the cvcgl-examples component, and the wasm build installs nothing but the gallery. Now: - native (build.sh / build.ps1): src/cvcGL/examples installs THIRD_PARTY_NOTICES.md and the LICENSE it links to into share/cvcgl-examples/ as part of the cvcgl-examples component; - wasm (build-wasm.sh): the same two files go beside the gallery in share/cvcgl-examples/web/, so the served payload carries them. cvc_revision bumps (this commit): cvcgl and libcvc 15 -> 16 (cvc.15 is already published without the fast-draw mapper), cvcgl-cuda 1 -> 2, cvcgl-examples 3 -> 4. --- cvcpkg/recipes/cvcgl-cuda/recipe.yaml | 8 ++++++-- cvcpkg/recipes/cvcgl-examples/build-wasm.sh | 5 +++++ cvcpkg/recipes/cvcgl-examples/recipe.yaml | 11 +++++++++-- cvcpkg/recipes/cvcgl/recipe.yaml | 11 +++++++++-- cvcpkg/recipes/libcvc/recipe.yaml | 6 +++++- src/cvcGL/examples/CMakeLists.txt | 8 ++++++++ 6 files changed, 42 insertions(+), 7 deletions(-) diff --git a/cvcpkg/recipes/cvcgl-cuda/recipe.yaml b/cvcpkg/recipes/cvcgl-cuda/recipe.yaml index 71a0a9cb5..9ea5a909d 100644 --- a/cvcpkg/recipes/cvcgl-cuda/recipe.yaml +++ b/cvcpkg/recipes/cvcgl-cuda/recipe.yaml @@ -4,13 +4,17 @@ recipe: upstream_version: "3.4.0" # Reset to 1 for the 3.3.0 upstream version: cvc_revision counts # revisions OF an upstream version, so a version bump restarts it. - cvc_revision: 1 + # Bumped 1 -> 2: carries cvcGL's fast-draw low-memory mapper and the BSD-3-Clause license + # metadata for its VTK 9.5.0-derived code (same as cvcgl 3.4.0+cvc.16). + cvc_revision: 2 maintainer: "cvcpkg group" maintainer_email: "info@cvcpkg.org" maintainer_url: "https://cvcpkg.org" description: "cvcGL (CUDA/GPU build) — the Qt-free VTK scene-graph library built against the CUDA libcvc-cuda SDK, so its GPU-only code paths (GL <-> CUDA interop: mapping a volume's CUDA unified/device buffer into a GL 3D texture for on-device volume rendering) compile in. Same cmake package (cvc::cvcGL) and same API as the CPU `cvcgl` package." homepage: https://github.com/transfix/libcvc - license: LGPL-2.1-only + # BSD-3-Clause: libcvcGL carries code derived from VTK 9.5.0 + # (LowMemoryPolyDataMapper.cpp); THIRD_PARTY_NOTICES.md ships in share/doc/cvcGL. + license: "LGPL-2.1-only AND BSD-3-Clause" tags: [scientific, volumetric, scene, vtk, graphics, cuda, gpu] # The GPU build of cvcGL — identical sources to the CPU `cvcgl` recipe, built diff --git a/cvcpkg/recipes/cvcgl-examples/build-wasm.sh b/cvcpkg/recipes/cvcgl-examples/build-wasm.sh index dfce801de..c143dea29 100755 --- a/cvcpkg/recipes/cvcgl-examples/build-wasm.sh +++ b/cvcpkg/recipes/cvcgl-examples/build-wasm.sh @@ -75,3 +75,8 @@ python3 "$CVC_SOURCE_DIR/src/cvcGL/examples/wasm/build-pages.py" \ --bin "$CVC_BUILD_DIR/bin" \ --out "$WEB" \ --assets "$CVC_SOURCE_DIR/src/cvcGL/examples/wasm/gallery-assets" +# The .wasm files embed third-party code (VTK-derived code in cvcGL, BSD-3-Clause; +# stb, sdf / mtxlib and the DTU .vtk reader in the cvc core) whose notices a binary +# redistribution must reproduce: ship them, and the LICENSE they link to, beside +# the gallery. +install -m 644 "$CVC_SOURCE_DIR/THIRD_PARTY_NOTICES.md" "$CVC_SOURCE_DIR/LICENSE" "$WEB/" diff --git a/cvcpkg/recipes/cvcgl-examples/recipe.yaml b/cvcpkg/recipes/cvcgl-examples/recipe.yaml index b5b04e9eb..83021a256 100644 --- a/cvcpkg/recipes/cvcgl-examples/recipe.yaml +++ b/cvcpkg/recipes/cvcgl-examples/recipe.yaml @@ -17,13 +17,20 @@ recipe: # Bumped 2 -> 3: build-wasm.sh no longer forces state_exec OFF (the # CVC_STATE_EXEC option is gone; state_exec is always built), so the wasm # gallery carries the program lanes that Ariadne apps run on. - cvc_revision: 3 + # Bumped 3 -> 4: ships THIRD_PARTY_NOTICES.md + LICENSE (native share/cvcgl-examples/ and + # the wasm web/ payload), declares LGPL-2.1-only AND BSD-3-Clause AND MIT, and rebuilds the + # demos against cvcgl 3.4.0+cvc.16 (fast-draw low-memory mapper). + cvc_revision: 4 maintainer: "cvcpkg group" maintainer_email: "info@cvcpkg.org" maintainer_url: "https://cvcpkg.org" description: "cvcGL example programs as ready-to-run commands: the GRL-SNAM navigation demos (nav_city_swarm / nav_city_drive / nav_fog_ghost / nav_finale — the torch-free cvc::nav reactive swarm rendered natively) plus terrain_lab, as native binaries in bin/, AND the full navigation gallery as a threaded (wasm-mt) WebAssembly browser build (share/cvcgl-examples/web/) with a cvcgl-examples-web launcher that serves it cross-origin-isolated (COOP/COEP) and opens a browser." homepage: https://github.com/transfix/libcvc - license: LGPL-2.1-only + # BSD-3-Clause: the demos link cvcGL statically, which carries code derived + # from VTK 9.5.0. MIT: lsystem_coast's OceanFFT is a port of ABYSSAL. Both + # notices are in THIRD_PARTY_NOTICES.md, shipped in share/cvcgl-examples/ (and + # share/cvcgl-examples/web/ for the browser build). + license: "LGPL-2.1-only AND BSD-3-Clause AND MIT" tags: [scientific, graphics, examples, vtk, wasm] # The examples are demos of cvcGL-as-a-toolkit, packaged separately from the diff --git a/cvcpkg/recipes/cvcgl/recipe.yaml b/cvcpkg/recipes/cvcgl/recipe.yaml index 310aaa3dd..a29b19c3b 100644 --- a/cvcpkg/recipes/cvcgl/recipe.yaml +++ b/cvcpkg/recipes/cvcgl/recipe.yaml @@ -69,13 +69,20 @@ recipe: # _rect partial texture uploads, caster-aware shadow baking (setCastsShadow), and the cheaper # scene-bounds walk deferred under hidden chrome. ABI-stale: SceneGraph/GraphicsNode/ # NullGraphicNode layouts changed, so the published +cvc.14 must not be mixed with this build. - cvc_revision: 15 + # Bumped 15 -> 16: the fast-draw low-memory mapper (LowMemoryPolyDataMapper: skips empty + # cell-type agents and re-binds the cached shader program; ~2/3 fewer GL calls per frame on + # WebGL2 with byte-identical pixels; CVCGL_LOWMEM_DRAW / setDrawPath for A/B), now the base of + # StreamingLowMemoryPolyDataMapper; late shader replacements reach the GPU on the low-memory + # path. License metadata now LGPL-2.1-only AND BSD-3-Clause (VTK 9.5.0-derived code). + cvc_revision: 16 maintainer: "cvcpkg group" maintainer_email: "info@cvcpkg.org" maintainer_url: "https://cvcpkg.org" description: "cvcGL — the Qt-free VTK scene-graph library extracted from volrover3: a reactive SceneGraph of GraphicsNode/GeometryNode/VolumeNode over cvc::state that renders cvc::geometry / cvc::volume via VTK. Ships libcvcGL.a + headers + a cmake package config; provides the cvcGL cmake package (cvc::cvcGL target). Consumed by pycvc-gl and other non-Qt scene consumers." homepage: https://github.com/transfix/libcvc - license: LGPL-2.1-only + # BSD-3-Clause: libcvcGL carries code derived from VTK 9.5.0 + # (LowMemoryPolyDataMapper.cpp); THIRD_PARTY_NOTICES.md ships in share/doc/cvcGL. + license: "LGPL-2.1-only AND BSD-3-Clause" tags: [scientific, volumetric, scene, vtk, graphics] # cvcGL is its OWN package, separate from libcvc: a plain libcvc bundle is a diff --git a/cvcpkg/recipes/libcvc/recipe.yaml b/cvcpkg/recipes/libcvc/recipe.yaml index 41bb24235..68fdaf60f 100644 --- a/cvcpkg/recipes/libcvc/recipe.yaml +++ b/cvcpkg/recipes/libcvc/recipe.yaml @@ -134,7 +134,11 @@ recipe: # caster-aware shadow baking, the cheaper bounds walk). ABI: SceneGraph, GraphicsNode and # NullGraphicNode layouts changed (#512 and the frame-cost fixes) with the SOVERSION # unchanged, so dependents must rebuild. cvcgl bumps in lockstep (14 -> 15). - cvc_revision: 15 + # Bumped 15 -> 16: lockstep with cvcgl's fast-draw low-memory mapper + # (LowMemoryPolyDataMapper, ~2/3 fewer GL calls per frame on WebGL2) and the expanded + # THIRD_PARTY_NOTICES.md. libcvc's own sources are unchanged by it; the bump keeps the pair's + # revisions aligned for consumers that pin both. + cvc_revision: 16 maintainer: "cvcpkg group" maintainer_email: "info@cvcpkg.org" maintainer_url: "https://cvcpkg.org" diff --git a/src/cvcGL/examples/CMakeLists.txt b/src/cvcGL/examples/CMakeLists.txt index e3b75aa7f..e2c3eeb34 100644 --- a/src/cvcGL/examples/CMakeLists.txt +++ b/src/cvcGL/examples/CMakeLists.txt @@ -176,6 +176,14 @@ set_target_properties(${_cvcgl_example_bins} PROPERTIES INSTALL_RPATH_USE_LINK_PATH ON) install(TARGETS ${_cvcgl_example_bins} RUNTIME DESTINATION bin COMPONENT cvcgl-examples) +# The bins carry third-party code whose notices a binary redistribution must +# reproduce -- VTK-derived code in the statically linked cvcGL (BSD-3-Clause) and +# lsystem_coast's OceanFFT, a port of ABYSSAL (MIT) -- so the component ships +# THIRD_PARTY_NOTICES.md, with the LICENSE it links to. (The browser build gets +# the same two files from cvcpkg/recipes/cvcgl-examples/build-wasm.sh.) +install(FILES "${CMAKE_CURRENT_SOURCE_DIR}/../../../THIRD_PARTY_NOTICES.md" + "${CMAKE_CURRENT_SOURCE_DIR}/../../../LICENSE" + DESTINATION share/cvcgl-examples COMPONENT cvcgl-examples) # Unit tests for nav_common's bounds-safe compositing (the minimap-overflow crash # guard) + the occupancy rasterizer. Only when GoogleTest is available (CVC_BUILD_TESTS From a7bb082acb7c65d298d10b2a9288bdeb24bd7b36 Mon Sep 17 00:00:00 2001 From: Joe Rivera Date: Thu, 1 Oct 2026 21:56:06 -0500 Subject: [PATCH 10/18] test(cvcGL): lowmem fastdraw -- probe round-trips bounded by Stock's, not zero On macOS CI (GL 4.1, Apple's software renderer, a core profile) the litLines probe failed "draws issue no desktop round-trip" on every run, shadows on and off. Its draws issue one glGetString per draw on every path, Stock included: VTK 9.5.0's vtkDrawTexturedElements::PreDraw sends a line width above vtkOpenGLRenderWindow::GetMaximumHardwareLineWidth() (1 on a core profile without wide lines) to ReportUnsupportedLineWidth, which logs the warning with glGetString(GL_VERSION). litLines draws at width 3; NVIDIA (10) and llvmpipe (255) draw it, so Stock has none there. The per-probe check now asks that Fast and AllCellTypes add no round-trip to Stock's draw, and that Stock's own be none or exactly that warning (only glGetString, on a probe whose width the driver lacks) -- still zero on every probe on NVIDIA and llvmpipe. The test also prints the GL renderer, version and maximum hardware line width it ran on. --- src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp | 46 +++++++++++++++++++++--- 1 file changed, 41 insertions(+), 5 deletions(-) diff --git a/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp b/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp index e1075b562..d11055d4d 100644 --- a/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp +++ b/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp @@ -41,8 +41,9 @@ // the stats check; // 4. pixels: Stock, AllCellTypes and Fast frames are byte-identical (RGBA); // 5. the saving: Fast draws a single-cell-type mesh in <= 2 VAO binds and a -// fraction of the stock calls, with no desktop round-trip in any draw, -// and a steady frame does no shader-cache lookup; +// fraction of the stock calls, with no desktop round-trip in any draw +// beyond Stock's (none, but VTK's own warning on a line width the driver +// cannot draw), and a steady frame does no shader-cache lookup; // 6. lifecycle: a shader rebuild (the shadow bake's pass change), a mapper's // released resources, GetShader(), programs released in the shader cache // and the scene re-opened in a new window all go back through the full @@ -1190,6 +1191,18 @@ bool renderAvailable(cvc::app &app) { s.sr->setCamera(0, 0, 50, 0, 0, 0, 0, 1, 0, 30.0, 1.0, 200.0); s.sr->render(); drew = drewSomething(s.rgba()); + // Which renderer the pixel and round-trip checks below ran on. + if (auto *win = vtkOpenGLRenderWindow::SafeDownCast(s.sr->renderWindow())) { + const char *report = win->ReportCapabilities(); + const std::string caps = report ? report : ""; + for (const char *key : {"OpenGL renderer string:", "OpenGL version string:"}) { + const size_t at = caps.find(key); + if (at != std::string::npos) + std::printf(" %s\n", caps.substr(at, caps.find('\n', at) - at).c_str()); + } + std::printf(" maximum hardware line width: %g\n", + double(win->GetMaximumHardwareLineWidth())); + } } if (drew) return true; @@ -1436,19 +1449,42 @@ void comparePaths(Scene &s, const std::string &label, bool shadows) { fmt("lookups %.0f", double(fast.firstStats.programLookups))); // Per draw, on the probes (steady frame: one draw per probe and pass). + auto *glWin = vtkOpenGLRenderWindow::SafeDownCast(s.sr->renderWindow()); + const double maxLineWidth = glWin ? glWin->GetMaximumHardwareLineWidth() : 0.0; for (size_t i = 0; i < s.probes.size(); ++i) { const Probe &p = s.probes[i]; const double nStock = stock.probeDraws[i], nFast = fast.probeDraws[i]; - if (nStock <= 0 || nFast <= 0) { + const double nAll = all.probeDraws[i]; + if (nStock <= 0 || nFast <= 0 || nAll <= 0) { check(false, label + ": probe " + p.name + " drew"); continue; } const GLCalls &cs = stock.probeCalls[i]; const GLCalls &cf = fast.probeCalls[i]; + const GLCalls &ca = all.probeCalls[i]; std::printf(" per draw %-12s Stock %s\n", p.name.c_str(), cs.str(nStock).c_str()); std::printf(" %-21s Fast %s\n", "", cf.str(nFast).c_str()); - check(cs.roundTrips() == 0 && cf.roundTrips() == 0 && all.probeCalls[i].roundTrips() == 0, - label + ": " + p.name + " draws issue no desktop round-trip"); + // Desktop round-trips: Stock's draw has none but VTK's own warning about a + // line width the driver cannot draw -- vtkDrawTexturedElements::PreDraw + // (9.5.0) logs it with glGetString(GL_VERSION) on every such draw. A core + // profile without wide lines (macOS: GetMaximumHardwareLineWidth() 1) does + // that for litLines (width 3) on every path; where the driver draws the + // width (NVIDIA, llvmpipe) Stock has none. Fast and AllCellTypes add none. + const double lineWidth = p.actor->GetProperty()->GetLineWidth(); + const bool vtkWarning = lineWidth > maxLineWidth && cs.roundTrips() == cs.count("GetString"); + const bool stockExplained = cs.roundTrips() == 0 || vtkWarning; + const double tripsStock = cs.roundTrips() / nStock; + check(stockExplained && cf.roundTrips() / nFast <= tripsStock && + ca.roundTrips() / nAll <= tripsStock, + label + ": " + p.name + + " draws issue no desktop round-trip beyond Stock's (none, but VTK's warning on a " + "line width the driver lacks)", + fmt("per draw: Stock %.1f, AllCellTypes %.1f", tripsStock, ca.roundTrips() / nAll) + + fmt(", Fast %.1f", cf.roundTrips() / nFast) + + (cs.roundTrips() == 0 + ? std::string() + : fmt("; Stock's glGetString %.1f, line width %g, driver max %g", + cs.count("GetString") / nStock, lineWidth, maxLineWidth))); check(cf.count("DrawArraysInstanced") == cs.count("DrawArraysInstanced") && cf.count("DrawArraysInstanced") == nFast, label + ": " + p.name + " one draw call per draw, as Stock"); From ae6771265841c59d568fbc9d23f5a3b631104775 Mon Sep 17 00:00:00 2001 From: Joe Rivera Date: Thu, 1 Oct 2026 21:57:48 -0500 Subject: [PATCH 11/18] test(cvcGL): lowmem fastdraw -- judge pixels against the renderer's measured self-variance On macOS CI (Apple's software renderer, GL 4.1) the pixel-exact checks failed in some runs and not in others, shadows on only: AllCellTypes vs Stock 1-4 differing bytes although the two GL traces were identical call for call, Fast vs Stock up to 5, and Stock before a shader-cache release vs Fast after it, or vs Fast in a new window, up to 9. Three of the four ctest reruns across the two jobs had no pixel difference at all. That renderer is not bit-deterministic for one GL stream; NVIDIA and llvmpipe are. The ordered GL-trace equalities stay the exact oracle everywhere. Pixels are now judged against a bound measured in-test: pairs of Stock frames whose traces are identical -- three Stock runs interleaved with the AllCellTypes and Fast runs in each scene, Stock twice across the lifecycle in one window, the old window's frame against the new window's, the guards scene twice -- give the most bytes and the largest per-byte step the renderer differed by on one stream. Every pixel check (Fast/AllCellTypes vs Stock, across a recompile or a new window, a shader replacement cleared) is recorded and judged at the end against the whole run's bound: no more differing bytes, no larger a step. A noisy pair is printed with where it differs; a pair whose traces differ is not a sample. A last check keeps the bound itself small (<= 0.1% of a frame). On NVIDIA and llvmpipe all 18 pairs are byte-identical, so the bound is 0 and every pixel check is still byte-exact (run 3x each with CVC_REQUIRE_RENDER=1). Injected noise in a third of the Stock runs passes within the bound; the same noise in Fast alone, with clean Stock runs, fails every Fast check. --- src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp | 201 +++++++++++++++++++---- 1 file changed, 171 insertions(+), 30 deletions(-) diff --git a/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp b/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp index d11055d4d..195ab38bc 100644 --- a/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp +++ b/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp @@ -39,7 +39,10 @@ // skipped shader-cache lookup removes no GL call -- the lookup's GL calls // are the bind tail the re-bind issues too; what it saves is CPU, which // the stats check; -// 4. pixels: Stock, AllCellTypes and Fast frames are byte-identical (RGBA); +// 4. pixels: Stock, AllCellTypes and Fast frames (RGBA) differ by no more +// than two Stock frames of one GL trace do -- the renderer's self-variance, +// measured through the run: none (byte-identical frames) on NVIDIA and +// llvmpipe, a few bytes now and then on Apple's software renderer; // 5. the saving: Fast draws a single-cell-type mesh in <= 2 VAO binds and a // fraction of the stock calls, with no desktop round-trip in any draw // beyond Stock's (none, but VTK's own warning on a line width the driver @@ -1130,19 +1133,138 @@ PathRun runPath(Scene &s, DrawPath path) { return r; } -long diffBytes(const std::vector &a, const std::vector &b) { - if (a.size() != b.size()) - return -1; - long n = 0; - for (size_t i = 0; i < a.size(); ++i) - n += a[i] != b[i]; - return n; +// ── pixels, against the renderer's own self-variance ───────────────────────── +// Two RGBA frames: how many bytes differ, by how much at most, and where first. +struct PixelDiff { + long bytes = 0; // -1: frames of different sizes, or empty + int maxDelta = 0; + long first = -1; // the pixel (index) of the first differing byte +}; + +PixelDiff pixelDiff(const std::vector &a, const std::vector &b) { + PixelDiff d; + if (a.size() != b.size() || a.empty()) { + d.bytes = -1; + return d; + } + for (size_t i = 0; i < a.size(); ++i) { + const int delta = std::abs(int(a[i]) - int(b[i])); + if (!delta) + continue; + if (d.first < 0) + d.first = static_cast(i / 4); + ++d.bytes; + d.maxDelta = std::max(d.maxDelta, delta); + } + return d; +} + +std::string describe(const PixelDiff &d, int w) { + if (d.bytes < 0) + return "frames of different sizes (or none)"; + if (d.bytes == 0) + return "byte-identical"; + return fmt("%.0f differing bytes, max delta %.0f", double(d.bytes), double(d.maxDelta)) + + fmt(", first at pixel (%.0f, %.0f)", double(d.first % w), double(d.first / w)); +} + +// Pixels are judged against what the renderer itself does with one GL stream. +// Two Stock frames whose GL traces are identical -- VTK's own draws, every +// call with its arguments and uniform values, in order, in one window -- can +// differ only by what the rasteriser does with that stream. NVIDIA and Mesa +// llvmpipe give the same bytes every time: the measured bound stays 0 and every +// pixel check is byte-exact. Apple's software renderer (macOS CI, GL 4.1) does +// not: with shadows on, AllCellTypes and Stock frames -- identical traces -- +// came out up to 4 bytes apart, and frames across a recompile or a new window +// up to 9, in some runs and not in others (never with shadows off). So a pixel +// check passes when its frames differ by no more bytes, and by no larger a +// step, than two Stock frames of one trace were seen to differ by. The samples +// are taken all through the test -- each Stock run, like each compared frame, +// follows a switch from another path -- and every check is judged at the end, +// against the bound of the whole run. The GL traces stay the exact oracle on +// every renderer. +struct SelfVariance { + long bytes = 0; // the most bytes two Stock frames of one trace differed by + int maxDelta = 0; + long frameBytes = 0; // the largest frame sampled + int pairs = 0, noisy = 0, skipped = 0; +}; +SelfVariance g_noise; + +struct PixelCheck { + std::string what; + PixelDiff d; + int w; +}; +std::vector g_pixelChecks; + +void sampleSelfVariance(const std::string &what, const Trace &ta, const Trace &tb, + const std::vector &a, const std::vector &b, + int w) { + std::string why; + if (!sameTrace(ta, tb, &why)) { + ++g_noise.skipped; // not one GL stream: what the frames differ by is not noise + std::printf(" self-variance: %s not sampled, GL traces differ -- %s\n", what.c_str(), + why.c_str()); + return; + } + const PixelDiff d = pixelDiff(a, b); + if (d.bytes < 0) + return; + ++g_noise.pairs; + g_noise.frameBytes = std::max(g_noise.frameBytes, static_cast(a.size())); + if (d.bytes == 0) + return; + ++g_noise.noisy; + g_noise.bytes = std::max(g_noise.bytes, d.bytes); + g_noise.maxDelta = std::max(g_noise.maxDelta, d.maxDelta); + std::printf(" self-variance: %s -- %s, identical GL traces\n", what.c_str(), + describe(d, w).c_str()); } -bool samePixels(const std::string &what, const std::vector &a, - const std::vector &b) { - const long d = diffBytes(a, b); - return check(d == 0 && !a.empty(), what, d == 0 ? "" : fmt("%.0f differing bytes", double(d))); +// Two Stock runs of one draw path: the first frames (a shadow bake when +// shadows are on) and the steady frames, which sample that bake -- so the +// steady pair counts only if the first frames' traces match too. +void sampleSelfVariance(const std::string &what, const PathRun &a, const PathRun &b, int w) { + const bool firstSame = sameTrace(a.firstTrace, b.firstTrace, nullptr); + sampleSelfVariance(what + ", first frame", a.firstTrace, b.firstTrace, a.firstPx, b.firstPx, w); + if (firstSame) + sampleSelfVariance(what + ", steady frame", a.steadyTrace, b.steadyTrace, a.steadyPx, + b.steadyPx, w); + else + ++g_noise.skipped; +} + +// Deferred: judged by checkPixels() once the whole run's self-variance is in. +void samePixels(const std::string &what, const std::vector &a, + const std::vector &b, int w) { + g_pixelChecks.push_back({what, pixelDiff(a, b), w}); +} + +void checkPixels() { + if (g_pixelChecks.empty()) + return; + const bool exact = g_noise.bytes == 0; + std::string head = exact ? std::string("none, so every check is byte-exact") + : fmt("up to %.0f bytes, max delta %.0f", double(g_noise.bytes), + double(g_noise.maxDelta)) + + fmt(" (in %.0f pairs)", double(g_noise.noisy)); + head += fmt(", over %.0f Stock frame pairs of identical GL traces", double(g_noise.pairs)); + if (g_noise.skipped) + head += fmt("; %.0f not sampled: traces differ", double(g_noise.skipped)); + std::printf("pixels (renderer self-variance: %s)\n", head.c_str()); + for (const auto &c : g_pixelChecks) { + const bool ok = + c.d.bytes >= 0 && c.d.bytes <= g_noise.bytes && c.d.maxDelta <= g_noise.maxDelta; + std::string detail = describe(c.d, c.w); + if (!exact && c.d.bytes != 0) + detail += std::string(ok ? ": within" : ": BEYOND") + " the renderer's self-variance"; + check(ok, c.what, detail); + } + // A renderer noisy enough to blur a real difference is not one to judge on. + check(g_noise.bytes * 1000 <= g_noise.frameBytes, + "the renderer's self-variance, if any, is a few bytes (<= 0.1% of a frame)", + fmt("%.0f of %.0f bytes", double(g_noise.bytes), double(g_noise.frameBytes))); } // Every pixel the background colour? (a frame that drew nothing) @@ -1394,25 +1516,35 @@ void comparePaths(Scene &s, const std::string &label, bool shadows) { // Settle: build every shader both ways first so no path pays a first compile. for (DrawPath p : {DrawPath::Stock, DrawPath::Fast}) runPath(s, p); + // Stock between and after the others: the renderer's self-variance is + // sampled where the compared frames are rendered, each after a path switch. const PathRun stock = runPath(s, DrawPath::Stock); const PathRun all = runPath(s, DrawPath::AllCellTypes); - const PathRun fast = runPath(s, DrawPath::Fast); const PathRun stock2 = runPath(s, DrawPath::Stock); + const PathRun fast = runPath(s, DrawPath::Fast); + const PathRun stock3 = runPath(s, DrawPath::Stock); check(drewSomething(stock.steadyPx), label + ": the scene draws"); std::string why; - const bool reproducible = sameTrace(stock2.firstTrace, stock.firstTrace, &why) && - sameTrace(stock2.steadyTrace, stock.steadyTrace, &why); - check(reproducible, label + ": the GL trace is reproducible (Stock twice, same trace)", + bool reproducible = true; + for (const PathRun *again : {&stock2, &stock3}) + reproducible = reproducible && sameTrace(again->firstTrace, stock.firstTrace, &why) && + sameTrace(again->steadyTrace, stock.steadyTrace, &why); + check(reproducible, label + ": the GL trace is reproducible (Stock three times, same trace)", reproducible ? fmt("%.0f calls", double(stock.steadyTrace.size())) : why); + sampleSelfVariance(label + ": Stock twice", stock, stock2, s.w); + sampleSelfVariance(label + ": Stock twice", stock, stock3, s.w); + sampleSelfVariance(label + ": Stock twice", stock2, stock3, s.w); checkTraces(label + ", first frame" + (shadows ? " (shadow bake)" : ""), stock.firstTrace, all.firstTrace, fast.firstTrace, fast.firstStats); checkTraces(label + ", steady frame", stock.steadyTrace, all.steadyTrace, fast.steadyTrace, fast.steadyStats); - samePixels(label + ": AllCellTypes pixels == Stock, first frame", stock.firstPx, all.firstPx); - samePixels(label + ": AllCellTypes pixels == Stock, steady frame", stock.steadyPx, all.steadyPx); - samePixels(label + ": Fast pixels == Stock, first frame", stock.firstPx, fast.firstPx); - samePixels(label + ": Fast pixels == Stock, steady frame", stock.steadyPx, fast.steadyPx); + samePixels(label + ": AllCellTypes pixels == Stock, first frame", stock.firstPx, all.firstPx, + s.w); + samePixels(label + ": AllCellTypes pixels == Stock, steady frame", stock.steadyPx, all.steadyPx, + s.w); + samePixels(label + ": Fast pixels == Stock, first frame", stock.firstPx, fast.firstPx, s.w); + samePixels(label + ": Fast pixels == Stock, steady frame", stock.steadyPx, fast.steadyPx, s.w); std::printf(" frame GL calls, first : Stock %s\n", stock.first.str().c_str()); std::printf(" Fast %s\n", fast.first.str().c_str()); @@ -1572,22 +1704,23 @@ void testLifecycle(Scene &s) { check(rel.programLookups == rel.draws && rel.programLookupsSkipped == 0 && rel.draws > 0, "after a mapper's ReleaseGraphicsResources its next draw looks the program up", fmt("lookups %.0f of %.0f draws", double(rel.programLookups), double(rel.draws))); + const PathRun released = runPath(s, DrawPath::Stock); { - const PathRun a = runPath(s, DrawPath::Stock); const PathRun b = runPath(s, DrawPath::Fast); - samePixels("after mapper release: Fast pixels == Stock", a.steadyPx, b.steadyPx); + samePixels("after mapper release: Fast pixels == Stock", released.steadyPx, b.steadyPx, s.w); } // The shader cache's programs released and recompiled in place (the cache // keeps the objects): the re-bound program is compiled again before use. auto *win = vtkOpenGLRenderWindow::SafeDownCast(s.sr->renderWindow()); const PathRun before = runPath(s, DrawPath::Stock); + sampleSelfVariance("lifecycle: Stock twice", released, before, s.w); win->GetShaderCache()->ReleaseGraphicsResources(win); LowMemoryPolyDataMapper::setDrawPath(DrawPath::Fast); s.sr->render(); const PathRun after = runPath(s, DrawPath::Fast); samePixels("programs released in the cache: Fast recompiles and matches Stock", before.steadyPx, - after.steadyPx); + after.steadyPx, s.w); // The same scene -- the same nodes and mappers -- closed and re-opened in a // new window (new context, new shader cache): the first draws resolve fully @@ -1606,8 +1739,13 @@ void testLifecycle(Scene &s) { double(st.programLookupsSkipped))); const PathRun a = runPath(s, DrawPath::Stock); const PathRun b = runPath(s, DrawPath::Fast); - samePixels("new window: Fast pixels == Stock", a.steadyPx, b.steadyPx); - samePixels("new window: same frame as the old window", before.steadyPx, b.steadyPx); + // The same Stock frame in the fresh window, against the old window's. (Not + // the window's first Stock run: its bake frame starts from a fresh context's + // GL state, so its trace differs from every later one by a cached enable.) + const PathRun a2 = runPath(s, DrawPath::Stock); + sampleSelfVariance("Stock, old window and new", before, a2, s.w); + samePixels("new window: Fast pixels == Stock", a.steadyPx, b.steadyPx, s.w); + samePixels("new window: same frame as the old window", before.steadyPx, b.steadyPx, s.w); } // ── 7. guards ──────────────────────────────────────────────────────────────── @@ -1650,8 +1788,10 @@ void testGuards(cvc::app &app) { runPath(s, p); const PathRun stock = runPath(s, DrawPath::Stock); const PathRun all = runPath(s, DrawPath::AllCellTypes); - const PathRun fast = runPath(s, DrawPath::Fast); - samePixels("POLYGON_OFFSET mode: Fast pixels == Stock", stock.steadyPx, fast.steadyPx); + const PathRun stock2 = runPath(s, DrawPath::Stock); + const PathRun fast = runPath(s, DrawPath::Fast); // last: the mapper stats below are Fast's + sampleSelfVariance("POLYGON_OFFSET mode: Stock twice", stock, stock2, s.w); + samePixels("POLYGON_OFFSET mode: Fast pixels == Stock", stock.steadyPx, fast.steadyPx, s.w); checkTraces("POLYGON_OFFSET mode", stock.steadyTrace, all.steadyTrace, fast.steadyTrace, fast.steadyStats); const auto &zs = zero.mapper->stats(); @@ -1712,9 +1852,8 @@ void testLateShaderReplacement(cvc::app &app) { n->clearShaderReplacements(); s.sr->render(); - const auto back = s.rgba(); - check(diffBytes(back, drawn) == 0, - std::string(pathName(p)) + ": cleared again, the original frame comes back"); + samePixels(std::string(pathName(p)) + ": cleared again, the original frame comes back", drawn, + s.rgba(), s.w); } LowMemoryPolyDataMapper::setDrawPath(DrawPath::Fast); } @@ -1744,6 +1883,7 @@ int main() { std::printf(" replica off (VTK is not an unpatched 9.5.0): every path draws the stock way\n"); if (renderAvailable(app)) testLateShaderReplacement(app); + checkPixels(); std::printf("%s: cvcgl_lowmem_fastdraw (%d checks, %d failed)\n", g_failures == 0 ? "PASS" : "FAIL", g_checks, g_failures); return g_failures == 0 ? 0 : 1; @@ -1777,6 +1917,7 @@ int main() { testLateShaderReplacement(app); } LowMemoryPolyDataMapper::setStageObserver(nullptr); + checkPixels(); check(ErrorCounter::errors() == 0, "no VTK errors", fmt("%.0f errors", double(ErrorCounter::errors()))); std::printf("%s: cvcgl_lowmem_fastdraw (%d checks, %d failed)\n", From 00f0a468be1cb4c755dbe43e445111523cdc26d5 Mon Sep 17 00:00:00 2001 From: Joe Rivera Date: Thu, 1 Oct 2026 22:57:57 -0500 Subject: [PATCH 12/18] test(cvcGL): lowmem fastdraw -- self-variance samples that cannot excuse what they judge Review of 63b77d8c1: the old window's Stock frame against the new window's fed the noise bound, while "new window: same frame as the old window" judged nearly that pair, so the check excused itself on every renderer. A cell colour changed by +40 on re-open (scratch build, every path alike) read as 18 bytes / step 24 of self-variance on NVIDIA and passed, and that bound then relaxed every Fast-vs-Stock pixel check too. Which pairs may be samples: - two Stock frames of identical GL traces in ONE window; nothing across the re-open. The new window samples its own Stock twice, and its Stock and Fast frames are judged against the old window's Stock frame; - neither right after a Fast run: whatever a Fast frame leaves that the trace cannot see would otherwise count as noise. The settles end on Stock, and the Stock run right after Fast is judged ("a Stock frame right after Fast == Stock"), in each scene and in the lifecycle; - no compared pair is a sampled pair, and every reference frame of a check is a member of a sampled pair. How a check is judged: against one observed pair in the same scene (shadows on, shadows off, guards) that differed by at least as many bytes AND at least as large a step -- not the most bytes of one pair with the largest step of another, nor another scene's noise. A sampled pair beyond 64 bytes or a step of 8 (was 0.1% of a frame, 307 bytes, no step limit) excuses nothing and fails the run. The old check logged byte counts only (1-9 on macOS); counts that are no multiple of 3 are pixels changed in some channels and not others, i.e. rounding, so the step cap is inferred, not measured -- the next macOS run prints each pair's bytes and step. testLateShaderReplacement's "cleared again" check is exact again (48x48, no shadows, never seen to vary, nothing sampled there). litLines round-trips: without a vtkOpenGLRenderWindow the line-width limit is +inf (fail closed), and VTK's warning explains Stock's round-trips only on a probe that draws lines, as exactly one glGetString per draw. NVIDIA and llvmpipe, CVC_REQUIRE_RENDER=1, 5 runs each: 271 checks pass, all 18 sampled pairs byte-identical, so every pixel check is byte-exact. The +40 mutation now fails on both ("BEYOND every pair of Stock frames sampled in shadows on", 18 bytes, step 24). Injected noise (scratch build): in the reference Stock frames -- the macOS pattern -- passes; the same noise only in Fast, only in the Stock frame after Fast, or in another scene fails; 30 bytes / step 5 fails against pairs of (60, 1) and (2, 8); a sampled pair of 100 bytes, or of step 30, fails the cap. --- src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp | 293 +++++++++++++++-------- 1 file changed, 190 insertions(+), 103 deletions(-) diff --git a/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp b/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp index 195ab38bc..aa9734407 100644 --- a/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp +++ b/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp @@ -39,10 +39,13 @@ // skipped shader-cache lookup removes no GL call -- the lookup's GL calls // are the bind tail the re-bind issues too; what it saves is CPU, which // the stats check; -// 4. pixels: Stock, AllCellTypes and Fast frames (RGBA) differ by no more -// than two Stock frames of one GL trace do -- the renderer's self-variance, -// measured through the run: none (byte-identical frames) on NVIDIA and -// llvmpipe, a few bytes now and then on Apple's software renderer; +// 4. pixels: AllCellTypes and Fast frames (RGBA) match Stock's -- as do a +// Stock frame right after Fast, and frames across a recompile and across a +// new window -- byte for byte, or by no more than one pair of Stock frames +// of one GL trace (same scene and window, neither right after Fast) +// differed in the same run: the renderer's self-variance. None on NVIDIA +// and llvmpipe, so byte-exact there; a few bytes now and then on Apple's +// software renderer; // 5. the saving: Fast draws a single-cell-type mesh in <= 2 VAO binds and a // fraction of the stock calls, with no desktop round-trip in any draw // beyond Stock's (none, but VTK's own warning on a line width the driver @@ -76,6 +79,7 @@ #include #include #include +#include #include #include #include @@ -1172,52 +1176,75 @@ std::string describe(const PixelDiff &d, int w) { // Two Stock frames whose GL traces are identical -- VTK's own draws, every // call with its arguments and uniform values, in order, in one window -- can // differ only by what the rasteriser does with that stream. NVIDIA and Mesa -// llvmpipe give the same bytes every time: the measured bound stays 0 and every -// pixel check is byte-exact. Apple's software renderer (macOS CI, GL 4.1) does -// not: with shadows on, AllCellTypes and Stock frames -- identical traces -- -// came out up to 4 bytes apart, and frames across a recompile or a new window -// up to 9, in some runs and not in others (never with shadows off). So a pixel -// check passes when its frames differ by no more bytes, and by no larger a -// step, than two Stock frames of one trace were seen to differ by. The samples -// are taken all through the test -- each Stock run, like each compared frame, -// follows a switch from another path -- and every check is judged at the end, -// against the bound of the whole run. The GL traces stay the exact oracle on -// every renderer. -struct SelfVariance { - long bytes = 0; // the most bytes two Stock frames of one trace differed by - int maxDelta = 0; - long frameBytes = 0; // the largest frame sampled - int pairs = 0, noisy = 0, skipped = 0; +// llvmpipe give the same bytes every time: no sample differs and every pixel +// check is byte-exact. Apple's software renderer (macOS CI, GL 4.1) does not: +// with shadows on, AllCellTypes and Stock frames -- identical traces -- came +// out 1 to 4 bytes apart, and frames across a recompile or a new window up to +// 9, in some runs and not in others (never with shadows off). +// +// So a pixel check passes when one sampled pair -- in the same scene, so the +// shadowed scene's noise excuses nothing elsewhere -- differed by at least as +// many bytes AND by at least as large a step: an envelope an actual pair +// filled, not the most bytes of one pair with the largest step of another. +// Which pairs may be samples is what keeps the bound from excusing a real +// difference: +// - both frames Stock, in one window (a pair across a re-open would measure +// as noise the very difference the cross-window check looks for); +// - neither right after a Fast run (whatever a Fast frame leaves that the +// trace cannot see -- buffer contents, the untraced warm-up frame -- would +// count as noise); a Stock frame right after Fast is judged instead; +// - no compared pair is a sampled pair. +// Every check is judged at the end, once the run's samples are in. The GL +// traces stay the exact oracle on every renderer. +// +// The most a sampled pair may differ by for the renderer to be judged on at +// all -- a noisier pair excuses nothing and fails the run: 64 bytes, ~7x the +// most seen on macOS (9), and a step of a few LSB. The old check logged byte +// counts only, not the step; counts that are no multiple of 3 (1, 2, 4, 5) +// mean pixels that changed in some colour channels and not the others, which +// is rounding, not a shading change. +constexpr long kSelfVarianceMaxBytes = 64; +constexpr int kSelfVarianceMaxDelta = 8; + +bool withinCap(const PixelDiff &d) { + return d.bytes >= 0 && d.bytes <= kSelfVarianceMaxBytes && d.maxDelta <= kSelfVarianceMaxDelta; +} + +struct SelfVarianceSample { + std::string scope, what; + PixelDiff d; + int w; }; -SelfVariance g_noise; +std::vector g_noisy; // the sampled pairs that differed +std::map g_pairs; // pairs sampled, per scope +std::map g_pairsSkipped; // not sampled (traces differ), per scope struct PixelCheck { - std::string what; + std::string scope, what; PixelDiff d; int w; }; std::vector g_pixelChecks; -void sampleSelfVariance(const std::string &what, const Trace &ta, const Trace &tb, - const std::vector &a, const std::vector &b, - int w) { +void sampleSelfVariance(const std::string &scope, const std::string &what, const Trace &ta, + const Trace &tb, const std::vector &a, + const std::vector &b, int w) { std::string why; if (!sameTrace(ta, tb, &why)) { - ++g_noise.skipped; // not one GL stream: what the frames differ by is not noise + ++g_pairsSkipped[scope]; // not one GL stream: what the frames differ by is not noise std::printf(" self-variance: %s not sampled, GL traces differ -- %s\n", what.c_str(), why.c_str()); return; } const PixelDiff d = pixelDiff(a, b); - if (d.bytes < 0) + if (d.bytes < 0) { + ++g_pairsSkipped[scope]; return; - ++g_noise.pairs; - g_noise.frameBytes = std::max(g_noise.frameBytes, static_cast(a.size())); + } + ++g_pairs[scope]; if (d.bytes == 0) return; - ++g_noise.noisy; - g_noise.bytes = std::max(g_noise.bytes, d.bytes); - g_noise.maxDelta = std::max(g_noise.maxDelta, d.maxDelta); + g_noisy.push_back({scope, what, d, w}); std::printf(" self-variance: %s -- %s, identical GL traces\n", what.c_str(), describe(d, w).c_str()); } @@ -1225,46 +1252,73 @@ void sampleSelfVariance(const std::string &what, const Trace &ta, const Trace &t // Two Stock runs of one draw path: the first frames (a shadow bake when // shadows are on) and the steady frames, which sample that bake -- so the // steady pair counts only if the first frames' traces match too. -void sampleSelfVariance(const std::string &what, const PathRun &a, const PathRun &b, int w) { +void sampleSelfVariance(const std::string &scope, const std::string &what, const PathRun &a, + const PathRun &b, int w) { const bool firstSame = sameTrace(a.firstTrace, b.firstTrace, nullptr); - sampleSelfVariance(what + ", first frame", a.firstTrace, b.firstTrace, a.firstPx, b.firstPx, w); + sampleSelfVariance(scope, what + ", first frame", a.firstTrace, b.firstTrace, a.firstPx, + b.firstPx, w); if (firstSame) - sampleSelfVariance(what + ", steady frame", a.steadyTrace, b.steadyTrace, a.steadyPx, + sampleSelfVariance(scope, what + ", steady frame", a.steadyTrace, b.steadyTrace, a.steadyPx, b.steadyPx, w); else - ++g_noise.skipped; + ++g_pairsSkipped[scope]; } -// Deferred: judged by checkPixels() once the whole run's self-variance is in. -void samePixels(const std::string &what, const std::vector &a, - const std::vector &b, int w) { - g_pixelChecks.push_back({what, pixelDiff(a, b), w}); +// Deferred: judged by checkPixels() once the run's samples are in. +void samePixels(const std::string &scope, const std::string &what, + const std::vector &a, const std::vector &b, int w) { + g_pixelChecks.push_back({scope, what, pixelDiff(a, b), w}); } void checkPixels() { if (g_pixelChecks.empty()) return; - const bool exact = g_noise.bytes == 0; - std::string head = exact ? std::string("none, so every check is byte-exact") - : fmt("up to %.0f bytes, max delta %.0f", double(g_noise.bytes), - double(g_noise.maxDelta)) + - fmt(" (in %.0f pairs)", double(g_noise.noisy)); - head += fmt(", over %.0f Stock frame pairs of identical GL traces", double(g_noise.pairs)); - if (g_noise.skipped) - head += fmt("; %.0f not sampled: traces differ", double(g_noise.skipped)); - std::printf("pixels (renderer self-variance: %s)\n", head.c_str()); + std::set scopes; + for (const auto &c : g_pixelChecks) + scopes.insert(c.scope); + std::printf("pixels (the renderer's self-variance: pairs of Stock frames of identical GL " + "traces, in one window, neither right after Fast)\n"); + for (const auto &scope : scopes) { + int noisy = 0; + for (const auto &n : g_noisy) + noisy += n.scope == scope; + std::string line = " " + scope + fmt(": %.0f pairs, ", double(g_pairs[scope])); + line += noisy ? fmt("%.0f differ (listed above)", double(noisy)) + : std::string("none differ, so its checks are byte-exact"); + if (g_pairsSkipped[scope]) + line += fmt("; %.0f not sampled", double(g_pairsSkipped[scope])); + std::printf("%s\n", line.c_str()); + } for (const auto &c : g_pixelChecks) { - const bool ok = - c.d.bytes >= 0 && c.d.bytes <= g_noise.bytes && c.d.maxDelta <= g_noise.maxDelta; + const SelfVarianceSample *excuse = nullptr; + for (const auto &n : g_noisy) + if (c.d.bytes > 0 && n.scope == c.scope && withinCap(n.d) && n.d.bytes >= c.d.bytes && + n.d.maxDelta >= c.d.maxDelta) { + excuse = &n; + break; + } std::string detail = describe(c.d, c.w); - if (!exact && c.d.bytes != 0) - detail += std::string(ok ? ": within" : ": BEYOND") + " the renderer's self-variance"; - check(ok, c.what, detail); + if (excuse) + detail += ": within " + excuse->what + " (" + describe(excuse->d, excuse->w) + ")"; + else if (c.d.bytes > 0) + detail += ": BEYOND every pair of Stock frames sampled in " + c.scope; + check(c.d.bytes == 0 || excuse, c.what, detail); } // A renderer noisy enough to blur a real difference is not one to judge on. - check(g_noise.bytes * 1000 <= g_noise.frameBytes, - "the renderer's self-variance, if any, is a few bytes (<= 0.1% of a frame)", - fmt("%.0f of %.0f bytes", double(g_noise.bytes), double(g_noise.frameBytes))); + const SelfVarianceSample *beyond = nullptr, *largest = nullptr; + for (const auto &n : g_noisy) { + if (!beyond && !withinCap(n.d)) + beyond = &n; + if (!largest || std::make_pair(n.d.bytes, n.d.maxDelta) > + std::make_pair(largest->d.bytes, largest->d.maxDelta)) + largest = &n; + } + const SelfVarianceSample *shown = beyond ? beyond : largest; + check(!beyond, + fmt("the renderer's self-variance, if any, is a few LSB at a few pixels (every sampled " + "pair <= %.0f bytes, max delta <= %.0f)", + double(kSelfVarianceMaxBytes), double(kSelfVarianceMaxDelta)), + shown ? shown->what + ": " + describe(shown->d, shown->w) : std::string("none differ")); } // Every pixel the background colour? (a frame that drew nothing) @@ -1513,38 +1567,47 @@ void comparePicking(Scene &s, const std::string &label) { void comparePaths(Scene &s, const std::string &label, bool shadows) { std::printf("%s\n", label.c_str()); - // Settle: build every shader both ways first so no path pays a first compile. - for (DrawPath p : {DrawPath::Stock, DrawPath::Fast}) + // Settle: build every shader both ways first so no path pays a first compile + // (Stock last: the first counted Stock run must not follow a Fast one). + for (DrawPath p : {DrawPath::Fast, DrawPath::Stock}) runPath(s, p); - // Stock between and after the others: the renderer's self-variance is - // sampled where the compared frames are rendered, each after a path switch. + // The renderer's self-variance is sampled from the Stock runs before Fast's + // (one after AllCellTypes); the Stock run right after Fast is judged, like + // the AllCellTypes and Fast frames. const PathRun stock = runPath(s, DrawPath::Stock); - const PathRun all = runPath(s, DrawPath::AllCellTypes); const PathRun stock2 = runPath(s, DrawPath::Stock); - const PathRun fast = runPath(s, DrawPath::Fast); + const PathRun all = runPath(s, DrawPath::AllCellTypes); const PathRun stock3 = runPath(s, DrawPath::Stock); + const PathRun fast = runPath(s, DrawPath::Fast); + const PathRun afterFast = runPath(s, DrawPath::Stock); check(drewSomething(stock.steadyPx), label + ": the scene draws"); std::string why; bool reproducible = true; - for (const PathRun *again : {&stock2, &stock3}) + for (const PathRun *again : {&stock2, &stock3, &afterFast}) reproducible = reproducible && sameTrace(again->firstTrace, stock.firstTrace, &why) && sameTrace(again->steadyTrace, stock.steadyTrace, &why); - check(reproducible, label + ": the GL trace is reproducible (Stock three times, same trace)", + check(reproducible, label + ": the GL trace is reproducible (Stock four times, same trace)", reproducible ? fmt("%.0f calls", double(stock.steadyTrace.size())) : why); - sampleSelfVariance(label + ": Stock twice", stock, stock2, s.w); - sampleSelfVariance(label + ": Stock twice", stock, stock3, s.w); - sampleSelfVariance(label + ": Stock twice", stock2, stock3, s.w); + sampleSelfVariance(label, label + ": Stock twice", stock, stock2, s.w); + sampleSelfVariance(label, label + ": Stock twice", stock, stock3, s.w); + sampleSelfVariance(label, label + ": Stock twice", stock2, stock3, s.w); checkTraces(label + ", first frame" + (shadows ? " (shadow bake)" : ""), stock.firstTrace, all.firstTrace, fast.firstTrace, fast.firstStats); checkTraces(label + ", steady frame", stock.steadyTrace, all.steadyTrace, fast.steadyTrace, fast.steadyStats); - samePixels(label + ": AllCellTypes pixels == Stock, first frame", stock.firstPx, all.firstPx, + samePixels(label, label + ": AllCellTypes pixels == Stock, first frame", stock.firstPx, + all.firstPx, s.w); + samePixels(label, label + ": AllCellTypes pixels == Stock, steady frame", stock.steadyPx, + all.steadyPx, s.w); + samePixels(label, label + ": Fast pixels == Stock, first frame", stock.firstPx, fast.firstPx, s.w); - samePixels(label + ": AllCellTypes pixels == Stock, steady frame", stock.steadyPx, all.steadyPx, + samePixels(label, label + ": Fast pixels == Stock, steady frame", stock.steadyPx, fast.steadyPx, s.w); - samePixels(label + ": Fast pixels == Stock, first frame", stock.firstPx, fast.firstPx, s.w); - samePixels(label + ": Fast pixels == Stock, steady frame", stock.steadyPx, fast.steadyPx, s.w); + samePixels(label, label + ": a Stock frame right after Fast == Stock, first frame", stock.firstPx, + afterFast.firstPx, s.w); + samePixels(label, label + ": a Stock frame right after Fast == Stock, steady frame", + stock.steadyPx, afterFast.steadyPx, s.w); std::printf(" frame GL calls, first : Stock %s\n", stock.first.str().c_str()); std::printf(" Fast %s\n", fast.first.str().c_str()); @@ -1581,8 +1644,10 @@ void comparePaths(Scene &s, const std::string &label, bool shadows) { fmt("lookups %.0f", double(fast.firstStats.programLookups))); // Per draw, on the probes (steady frame: one draw per probe and pass). + // Without the window's limit no width counts as one the driver lacks. auto *glWin = vtkOpenGLRenderWindow::SafeDownCast(s.sr->renderWindow()); - const double maxLineWidth = glWin ? glWin->GetMaximumHardwareLineWidth() : 0.0; + const double maxLineWidth = + glWin ? glWin->GetMaximumHardwareLineWidth() : std::numeric_limits::infinity(); for (size_t i = 0; i < s.probes.size(); ++i) { const Probe &p = s.probes[i]; const double nStock = stock.probeDraws[i], nFast = fast.probeDraws[i]; @@ -1598,12 +1663,18 @@ void comparePaths(Scene &s, const std::string &label, bool shadows) { std::printf(" %-21s Fast %s\n", "", cf.str(nFast).c_str()); // Desktop round-trips: Stock's draw has none but VTK's own warning about a // line width the driver cannot draw -- vtkDrawTexturedElements::PreDraw - // (9.5.0) logs it with glGetString(GL_VERSION) on every such draw. A core - // profile without wide lines (macOS: GetMaximumHardwareLineWidth() 1) does - // that for litLines (width 3) on every path; where the driver draws the - // width (NVIDIA, llvmpipe) Stock has none. Fast and AllCellTypes add none. + // (9.5.0) logs it with one glGetString(GL_VERSION) per draw of lines. A + // core profile without wide lines (macOS: GetMaximumHardwareLineWidth() 1) + // does that for litLines (width 3) on every path; where the driver draws + // the width (NVIDIA, llvmpipe) Stock has none. Fast and AllCellTypes add + // none. const double lineWidth = p.actor->GetProperty()->GetLineWidth(); - const bool vtkWarning = lineWidth > maxLineWidth && cs.roundTrips() == cs.count("GetString"); + vtkPolyData *input = p.mapper->GetInput(); + const bool drawsLines = (input && input->GetNumberOfLines() > 0) || + p.actor->GetProperty()->GetRepresentation() == VTK_WIREFRAME; + const bool vtkWarning = drawsLines && lineWidth > maxLineWidth && + cs.roundTrips() == cs.count("GetString") && + cs.count("GetString") == nStock; const bool stockExplained = cs.roundTrips() == 0 || vtkWarning; const double tripsStock = cs.roundTrips() / nStock; check(stockExplained && cf.roundTrips() / nFast <= tripsStock && @@ -1669,7 +1740,8 @@ void comparePaths(Scene &s, const std::string &label, bool shadows) { } // ── 6. lifecycle ───────────────────────────────────────────────────────────── -void testLifecycle(Scene &s) { +// scope: the scene's label, whose self-variance samples judge these pixels. +void testLifecycle(Scene &s, const std::string &scope) { std::printf("lifecycle\n"); LowMemoryPolyDataMapper::setDrawPath(DrawPath::Fast); s.sr->render(); @@ -1704,23 +1776,27 @@ void testLifecycle(Scene &s) { check(rel.programLookups == rel.draws && rel.programLookupsSkipped == 0 && rel.draws > 0, "after a mapper's ReleaseGraphicsResources its next draw looks the program up", fmt("lookups %.0f of %.0f draws", double(rel.programLookups), double(rel.draws))); - const PathRun released = runPath(s, DrawPath::Stock); - { - const PathRun b = runPath(s, DrawPath::Fast); - samePixels("after mapper release: Fast pixels == Stock", released.steadyPx, b.steadyPx, s.w); - } + const PathRun released = runPath(s, DrawPath::Fast); + // Stock to compare against: two runs sampled for the renderer's + // self-variance, after one that follows Fast (judged, not sampled). + const PathRun afterFast = runPath(s, DrawPath::Stock); + const PathRun before = runPath(s, DrawPath::Stock); + const PathRun before2 = runPath(s, DrawPath::Stock); + sampleSelfVariance(scope, "lifecycle: Stock twice", before, before2, s.w); + samePixels(scope, "after mapper release: Fast pixels == Stock", before.steadyPx, + released.steadyPx, s.w); + samePixels(scope, "lifecycle: a Stock frame right after Fast == Stock", before.steadyPx, + afterFast.steadyPx, s.w); // The shader cache's programs released and recompiled in place (the cache // keeps the objects): the re-bound program is compiled again before use. auto *win = vtkOpenGLRenderWindow::SafeDownCast(s.sr->renderWindow()); - const PathRun before = runPath(s, DrawPath::Stock); - sampleSelfVariance("lifecycle: Stock twice", released, before, s.w); win->GetShaderCache()->ReleaseGraphicsResources(win); LowMemoryPolyDataMapper::setDrawPath(DrawPath::Fast); s.sr->render(); const PathRun after = runPath(s, DrawPath::Fast); - samePixels("programs released in the cache: Fast recompiles and matches Stock", before.steadyPx, - after.steadyPx, s.w); + samePixels(scope, "programs released in the cache: Fast recompiles and matches Stock", + before.steadyPx, after.steadyPx, s.w); // The same scene -- the same nodes and mappers -- closed and re-opened in a // new window (new context, new shader cache): the first draws resolve fully @@ -1737,15 +1813,20 @@ void testLifecycle(Scene &s) { "re-opened in a new window: the first draws look their programs up", fmt("lookups %.0f, skipped %.0f", double(st.programLookups), double(st.programLookupsSkipped))); + // The window's first Stock run follows that Fast frame: neither sampled nor + // compared. The two after it are sampled -- in this window, never with the + // old window's: a pair across the re-open would measure as noise the very + // difference the cross-window checks look for. + runPath(s, DrawPath::Stock); const PathRun a = runPath(s, DrawPath::Stock); - const PathRun b = runPath(s, DrawPath::Fast); - // The same Stock frame in the fresh window, against the old window's. (Not - // the window's first Stock run: its bake frame starts from a fresh context's - // GL state, so its trace differs from every later one by a cached enable.) const PathRun a2 = runPath(s, DrawPath::Stock); - sampleSelfVariance("Stock, old window and new", before, a2, s.w); - samePixels("new window: Fast pixels == Stock", a.steadyPx, b.steadyPx, s.w); - samePixels("new window: same frame as the old window", before.steadyPx, b.steadyPx, s.w); + const PathRun b = runPath(s, DrawPath::Fast); + sampleSelfVariance(scope, "new window: Stock twice", a, a2, s.w); + samePixels(scope, "new window: Fast pixels == Stock", a.steadyPx, b.steadyPx, s.w); + samePixels(scope, "new window: the Stock frame == the old window's", before.steadyPx, a.steadyPx, + s.w); + samePixels(scope, "new window: the Fast frame == the old window's Stock frame", before.steadyPx, + b.steadyPx, s.w); } // ── 7. guards ──────────────────────────────────────────────────────────────── @@ -1784,14 +1865,16 @@ void testGuards(cvc::app &app) { "POLYGON_OFFSET mode: " + p->name + "'s program carries the depth offset uniforms"); } - for (DrawPath p : {DrawPath::Stock, DrawPath::Fast}) + // Stock last in the settle, so that neither sampled Stock run follows Fast. + for (DrawPath p : {DrawPath::Fast, DrawPath::Stock}) runPath(s, p); const PathRun stock = runPath(s, DrawPath::Stock); const PathRun all = runPath(s, DrawPath::AllCellTypes); const PathRun stock2 = runPath(s, DrawPath::Stock); const PathRun fast = runPath(s, DrawPath::Fast); // last: the mapper stats below are Fast's - sampleSelfVariance("POLYGON_OFFSET mode: Stock twice", stock, stock2, s.w); - samePixels("POLYGON_OFFSET mode: Fast pixels == Stock", stock.steadyPx, fast.steadyPx, s.w); + sampleSelfVariance("guards", "POLYGON_OFFSET mode: Stock twice", stock, stock2, s.w); + samePixels("guards", "POLYGON_OFFSET mode: Fast pixels == Stock", stock.steadyPx, fast.steadyPx, + s.w); checkTraces("POLYGON_OFFSET mode", stock.steadyTrace, all.steadyTrace, fast.steadyTrace, fast.steadyStats); const auto &zs = zero.mapper->stats(); @@ -1850,10 +1933,14 @@ void testLateShaderReplacement(cvc::app &app) { std::string(pathName(p)) + ": a replacement added after the first draw reaches the GPU", fmt("lookups %.0f", double(st.programLookups + st.stockDraws))); + // Exact on every renderer: an unshadowed 48x48 scene, where no renderer has + // been seen to vary, and no self-variance is sampled here. n->clearShaderReplacements(); s.sr->render(); - samePixels(std::string(pathName(p)) + ": cleared again, the original frame comes back", drawn, - s.rgba(), s.w); + const PixelDiff back = pixelDiff(drawn, s.rgba()); + check(back.bytes == 0, + std::string(pathName(p)) + ": cleared again, the original frame comes back", + describe(back, s.w)); } LowMemoryPolyDataMapper::setDrawPath(DrawPath::Fast); } @@ -1883,7 +1970,6 @@ int main() { std::printf(" replica off (VTK is not an unpatched 9.5.0): every path draws the stock way\n"); if (renderAvailable(app)) testLateShaderReplacement(app); - checkPixels(); std::printf("%s: cvcgl_lowmem_fastdraw (%d checks, %d failed)\n", g_failures == 0 ? "PASS" : "FAIL", g_checks, g_failures); return g_failures == 0 ? 0 : 1; @@ -1909,9 +1995,10 @@ int main() { "every node and probe draws through LowMemoryPolyDataMapper", fmt("%.0f of %.0f", double(s.mappers().size()), double(s.nodes.size() + s.probes.size()))); - comparePaths(s, shadows ? "shadows on" : "shadows off", shadows); + const std::string label = shadows ? "shadows on" : "shadows off"; + comparePaths(s, label, shadows); if (shadows) - testLifecycle(s); + testLifecycle(s, label); } testGuards(app); testLateShaderReplacement(app); From ac6a69ada5e059b693a781c8aa2f2de6434a738a Mon Sep 17 00:00:00 2001 From: Joe Rivera Date: Thu, 1 Oct 2026 22:51:51 -0500 Subject: [PATCH 13/18] cvcGL tests: one GL test at a time (CTest RESOURCE_LOCK cvcgl_gl_context) Every cvcGL test that renders through a GL context now takes the CTest resource lock cvcgl_gl_context, so ctest never runs two of them at once; the 23 headless tests (listed by name) keep running in parallel. The lock is assigned by exclusion -- every test in the directory minus the headless list -- so a GL test added later is locked unless someone lists it as headless. Why: on the macOS runners every failed attempt of cvcgl_shadow_casters (4 of the 12 attempts across the 8 package-macos jobs of #521-#523) ran alongside cvcgl_shadow_caster_growth and other GL tests under ctest --parallel; no passing attempt overlapped growth, and each retry passed. That is a correlation, not a reproduction -- no Mac reproduced the flake. On every platform, on purpose: the cost is wall time only. On macOS Debug the GL tests' own times summed to ~370 s (cvcgl_volren_node 230 s of it) where they overlapped into ~250 s, so that ctest step can grow by up to ~2 minutes of ~9. --- src/cvcGL/CMakeLists.txt | 21 +++++++++++++++++++++ 1 file changed, 21 insertions(+) diff --git a/src/cvcGL/CMakeLists.txt b/src/cvcGL/CMakeLists.txt index 0681328d2..5c82d297c 100644 --- a/src/cvcGL/CMakeLists.txt +++ b/src/cvcGL/CMakeLists.txt @@ -684,6 +684,27 @@ add_executable(cvcgl_sdl_input test/cvcgl_sdl_input.cpp) target_link_libraries(cvcgl_sdl_input PRIVATE cvcGL) add_test(NAME cvcgl_sdl_input COMMAND cvcgl_sdl_input) +# ── GL tests: one at a time ── +# Every cvcGL test that renders through a GL context takes the CTest resource +# lock cvcgl_gl_context, so ctest never runs two of them at once; the headless +# tests listed below keep running in parallel. Why: on the macOS runners every +# failed attempt of cvcgl_shadow_casters (4 of 12 seen) ran alongside +# cvcgl_shadow_caster_growth and other GL tests under ctest --parallel, no +# passing attempt overlapped growth, and each retry passed. On every platform, +# on purpose: it costs only wall time (the GL tests run end to end instead of +# overlapped) and takes GL contention out of every run. A test added above is +# locked unless it is listed here as headless. +get_property(_cvcgl_gl_tests DIRECTORY PROPERTY TESTS) +list(REMOVE_ITEM _cvcgl_gl_tests + cvcgl_smoke cvcgl_metadata_mirror cvcgl_volume_range cvcgl_traversal + cvcgl_grid_bounds cvcgl_bounds_walk cvcgl_scene_chrome cvcgl_remove + cvcgl_teardown cvcgl_light_state_thread cvcgl_pose_echo + cvcgl_node_outlives_scene cvcgl_texture cvcgl_texture_zerocopy + cvcgl_stream_texture cvcgl_state_publisher cvcgl_nav_stats_publish + cvcgl_camera cvcgl_track_parity cvcgl_fps_hud cvcgl_stage_setters + cvcgl_streaming_races cvcgl_sdl_input) +set_tests_properties(${_cvcgl_gl_tests} PROPERTIES RESOURCE_LOCK cvcgl_gl_context) + # ── examples (opt-in, OFF by default) ── # Standalone C++ programs that drive cvcGL directly — SceneGraph + SceneRenderer # (the onscreen "basic window") + CameraController (orbit / Quake-fly / cinematic From 130f8c921893769b44e74f0a998458ebac5aed7c Mon Sep 17 00:00:00 2001 From: Joe Rivera Date: Thu, 1 Oct 2026 23:31:10 -0500 Subject: [PATCH 14/18] test(cvcGL): lowmem fastdraw -- self-variance relaxation on Apple's renderer only Review of a802747fd: the sample rules narrowed what a self-variance pair could excuse but did not close it. A sampled pair can still share a frame with the checks it judges: comparePaths samples (stock, stock2) and (stock, stock3) while all six of its pixel checks compare against stock; the lifecycle samples (a, a2) while judging (before, a) and (a, b). If that one frame alone is wrong and the other member is right, the pair measures exactly the check's difference and excuses it. Shown on NVIDIA and llvmpipe, both byte-exact renderers: - cellColours cell 0 +4 after the window re-open, undone before a2: "new window: ... within new window: Stock twice", run passed; - cellColours cell 0 +4 during only the first counted Stock run of the shadows-on scene: all six shadows-on pixel checks passed "within shadows on: Stock twice". The relaxation is now Apple-only, since a shared frame could still self-excuse. renderAvailable() keeps the GL renderer and version strings from ReportCapabilities(); Apple-tolerance mode is on when either names Apple (macOS CI: GL "4.1 APPLE-23.1.1"). Everywhere else (exact mode): - a sampled pair of Stock frames of one GL trace is itself a check, "... Stock twice, is byte-identical on a deterministic renderer", and fails the run if it differs; - every pixel check is byte for byte. On Apple the per-pair, same-scope, 64-byte / step-8 logic is unchanged and each differing pair still prints its bytes and step. The mode is printed once at startup with the renderer and version: pixel checks: exact|apple-tolerance (OpenGL renderer "...", version "...") Both mutations above now fail on NVIDIA and llvmpipe (4 and 10 checks), as does the earlier +40 on re-open (2). With the mode forced to apple-tolerance in a scratch build, every check line matches the previous commit's, mutations included. --- src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp | 91 +++++++++++++++++------- 1 file changed, 66 insertions(+), 25 deletions(-) diff --git a/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp b/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp index aa9734407..4c8b9089d 100644 --- a/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp +++ b/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp @@ -41,11 +41,11 @@ // the stats check; // 4. pixels: AllCellTypes and Fast frames (RGBA) match Stock's -- as do a // Stock frame right after Fast, and frames across a recompile and across a -// new window -- byte for byte, or by no more than one pair of Stock frames -// of one GL trace (same scene and window, neither right after Fast) -// differed in the same run: the renderer's self-variance. None on NVIDIA -// and llvmpipe, so byte-exact there; a few bytes now and then on Apple's -// software renderer; +// new window -- byte for byte, and so do pairs of Stock frames of one GL +// trace. Only on Apple's renderer (GL version or renderer string naming +// Apple), which varies by a few bytes now and then, may a check differ by +// no more than one such pair (same scene and window, neither right after +// Fast) differed in the same run; // 5. the saving: Fast draws a single-cell-type mesh in <= 2 VAO binds and a // fraction of the stock calls, with no desktop round-trip in any draw // beyond Stock's (none, but VTK's own warning on a line width the driver @@ -1176,24 +1176,33 @@ std::string describe(const PixelDiff &d, int w) { // Two Stock frames whose GL traces are identical -- VTK's own draws, every // call with its arguments and uniform values, in order, in one window -- can // differ only by what the rasteriser does with that stream. NVIDIA and Mesa -// llvmpipe give the same bytes every time: no sample differs and every pixel -// check is byte-exact. Apple's software renderer (macOS CI, GL 4.1) does not: -// with shadows on, AllCellTypes and Stock frames -- identical traces -- came -// out 1 to 4 bytes apart, and frames across a recompile or a new window up to -// 9, in some runs and not in others (never with shadows off). +// llvmpipe give the same bytes every time. Apple's software renderer (macOS +// CI, GL 4.1) does not: with shadows on, AllCellTypes and Stock frames -- +// identical traces -- came out 1 to 4 bytes apart, and frames across a +// recompile or a new window up to 9, in some runs and not in others (never +// with shadows off). // -// So a pixel check passes when one sampled pair -- in the same scene, so the +// Exact mode (every renderer but Apple's): a sampled pair -- two Stock frames +// of one GL trace -- that differs fails the run on its own, and every pixel +// check is byte for byte. +// +// Apple-tolerance mode (the GL version or renderer string names Apple): a +// pixel check passes when one sampled pair -- in the same scene, so the // shadowed scene's noise excuses nothing elsewhere -- differed by at least as // many bytes AND by at least as large a step: an envelope an actual pair // filled, not the most bytes of one pair with the largest step of another. -// Which pairs may be samples is what keeps the bound from excusing a real -// difference: +// Which pairs may be samples narrows what they can excuse: // - both frames Stock, in one window (a pair across a re-open would measure // as noise the very difference the cross-window check looks for); // - neither right after a Fast run (whatever a Fast frame leaves that the // trace cannot see -- buffer contents, the untraced warm-up frame -- would // count as noise); a Stock frame right after Fast is judged instead; // - no compared pair is a sampled pair. +// That narrows it, no more: a sampled pair and a check may share a frame (a +// scene's first counted Stock run is in two samples and in all six of its +// checks; the new window's first sampled run is in two of its checks), and if +// that frame alone is wrong the pair measures exactly the check's difference +// and excuses it. Hence the relaxation is Apple's alone, where it is needed. // Every check is judged at the end, once the run's samples are in. The GL // traces stay the exact oracle on every renderer. // @@ -1210,6 +1219,11 @@ bool withinCap(const PixelDiff &d) { return d.bytes >= 0 && d.bytes <= kSelfVarianceMaxBytes && d.maxDelta <= kSelfVarianceMaxDelta; } +// The renderer the frames are drawn by, as renderAvailable() reads it, and +// whether its pixels are judged in Apple-tolerance mode (exact otherwise). +std::string g_glRenderer, g_glVersion; +bool g_appleTolerance = false; + struct SelfVarianceSample { std::string scope, what; PixelDiff d; @@ -1242,6 +1256,11 @@ void sampleSelfVariance(const std::string &scope, const std::string &what, const return; } ++g_pairs[scope]; + if (!g_appleTolerance) { // exact mode: the pair is itself a check, never an excuse + check(d.bytes == 0, what + " is byte-identical on a deterministic renderer", + d.bytes ? describe(d, w) + ", identical GL traces" : std::string()); + return; + } if (d.bytes == 0) return; g_noisy.push_back({scope, what, d, w}); @@ -1273,11 +1292,17 @@ void samePixels(const std::string &scope, const std::string &what, void checkPixels() { if (g_pixelChecks.empty()) return; + if (!g_appleTolerance) { + std::printf("pixels (exact mode: byte for byte)\n"); + for (const auto &c : g_pixelChecks) + check(c.d.bytes == 0, c.what, describe(c.d, c.w)); + return; + } std::set scopes; for (const auto &c : g_pixelChecks) scopes.insert(c.scope); - std::printf("pixels (the renderer's self-variance: pairs of Stock frames of identical GL " - "traces, in one window, neither right after Fast)\n"); + std::printf("pixels (apple-tolerance mode, the renderer's self-variance: pairs of Stock frames " + "of identical GL traces, in one window, neither right after Fast)\n"); for (const auto &scope : scopes) { int noisy = 0; for (const auto &n : g_noisy) @@ -1367,18 +1392,34 @@ bool renderAvailable(cvc::app &app) { s.sr->setCamera(0, 0, 50, 0, 0, 0, 0, 1, 0, 30.0, 1.0, 200.0); s.sr->render(); drew = drewSomething(s.rgba()); - // Which renderer the pixel and round-trip checks below ran on. + // Which renderer the pixel and round-trip checks below ran on, and so how + // its pixels are judged (see checkPixels()). if (auto *win = vtkOpenGLRenderWindow::SafeDownCast(s.sr->renderWindow())) { const char *report = win->ReportCapabilities(); const std::string caps = report ? report : ""; - for (const char *key : {"OpenGL renderer string:", "OpenGL version string:"}) { + const auto value = [&caps](const std::string &key) { const size_t at = caps.find(key); - if (at != std::string::npos) - std::printf(" %s\n", caps.substr(at, caps.find('\n', at) - at).c_str()); - } + if (at == std::string::npos) + return std::string(); + const size_t end = std::min(caps.find('\n', at), caps.size()); + const size_t from = std::min(caps.find_first_not_of(' ', at + key.size()), end); + return caps.substr(from, end - from); + }; + g_glRenderer = value("OpenGL renderer string:"); + g_glVersion = value("OpenGL version string:"); + const auto apple = [](const std::string &v) { + return v.find("APPLE") != std::string::npos || v.find("Apple") != std::string::npos; + }; + g_appleTolerance = apple(g_glVersion) || apple(g_glRenderer); std::printf(" maximum hardware line width: %g\n", double(win->GetMaximumHardwareLineWidth())); } + const char *mode = g_appleTolerance ? "apple-tolerance" : "exact"; + const char *rule = g_appleTolerance + ? "a check may differ by no more than a sampled pair of Stock frames" + : "every pair of Stock frames and every check byte for byte"; + std::printf(" pixel checks: %s (OpenGL renderer \"%s\", version \"%s\") -- %s\n", mode, + g_glRenderer.c_str(), g_glVersion.c_str(), rule); } if (drew) return true; @@ -1571,9 +1612,9 @@ void comparePaths(Scene &s, const std::string &label, bool shadows) { // (Stock last: the first counted Stock run must not follow a Fast one). for (DrawPath p : {DrawPath::Fast, DrawPath::Stock}) runPath(s, p); - // The renderer's self-variance is sampled from the Stock runs before Fast's - // (one after AllCellTypes); the Stock run right after Fast is judged, like - // the AllCellTypes and Fast frames. + // Pairs of the Stock runs before Fast's (one after AllCellTypes) are checked + // byte-identical -- on Apple's renderer, sampled as its self-variance; the + // Stock run right after Fast is judged, like the AllCellTypes and Fast frames. const PathRun stock = runPath(s, DrawPath::Stock); const PathRun stock2 = runPath(s, DrawPath::Stock); const PathRun all = runPath(s, DrawPath::AllCellTypes); @@ -1777,8 +1818,8 @@ void testLifecycle(Scene &s, const std::string &scope) { "after a mapper's ReleaseGraphicsResources its next draw looks the program up", fmt("lookups %.0f of %.0f draws", double(rel.programLookups), double(rel.draws))); const PathRun released = runPath(s, DrawPath::Fast); - // Stock to compare against: two runs sampled for the renderer's - // self-variance, after one that follows Fast (judged, not sampled). + // Stock to compare against: two runs, a self-variance pair (byte-identical + // but on Apple's renderer), after one that follows Fast (judged, not sampled). const PathRun afterFast = runPath(s, DrawPath::Stock); const PathRun before = runPath(s, DrawPath::Stock); const PathRun before2 = runPath(s, DrawPath::Stock); From 8a1adecd8ee3f8499e040d9857f00c57f8e70e02 Mon Sep 17 00:00:00 2001 From: Joe Rivera Date: Fri, 2 Oct 2026 04:47:09 -0500 Subject: [PATCH 15/18] cvcpkg: fast-draw reship at libcvc/cvcgl 17, cvcgl-examples 16 libcvc and cvcgl 3.4.0+cvc.16 went out as a native publish of master without the fast-draw mapper, and cvcgl-examples is at +cvc.15 natively. The wasm publish packs the recipe revision verbatim, so take the next number above every platform's. --- cvcpkg/recipes/cvcgl-cuda/recipe.yaml | 2 +- cvcpkg/recipes/cvcgl-examples/recipe.yaml | 7 ++++--- cvcpkg/recipes/cvcgl/recipe.yaml | 5 +++-- cvcpkg/recipes/libcvc/recipe.yaml | 5 +++-- 4 files changed, 11 insertions(+), 8 deletions(-) diff --git a/cvcpkg/recipes/cvcgl-cuda/recipe.yaml b/cvcpkg/recipes/cvcgl-cuda/recipe.yaml index 9ea5a909d..6647e3b75 100644 --- a/cvcpkg/recipes/cvcgl-cuda/recipe.yaml +++ b/cvcpkg/recipes/cvcgl-cuda/recipe.yaml @@ -5,7 +5,7 @@ recipe: # Reset to 1 for the 3.3.0 upstream version: cvc_revision counts # revisions OF an upstream version, so a version bump restarts it. # Bumped 1 -> 2: carries cvcGL's fast-draw low-memory mapper and the BSD-3-Clause license - # metadata for its VTK 9.5.0-derived code (same as cvcgl 3.4.0+cvc.16). + # metadata for its VTK 9.5.0-derived code (same as cvcgl 3.4.0+cvc.17). cvc_revision: 2 maintainer: "cvcpkg group" maintainer_email: "info@cvcpkg.org" diff --git a/cvcpkg/recipes/cvcgl-examples/recipe.yaml b/cvcpkg/recipes/cvcgl-examples/recipe.yaml index 83021a256..b471751cd 100644 --- a/cvcpkg/recipes/cvcgl-examples/recipe.yaml +++ b/cvcpkg/recipes/cvcgl-examples/recipe.yaml @@ -17,10 +17,11 @@ recipe: # Bumped 2 -> 3: build-wasm.sh no longer forces state_exec OFF (the # CVC_STATE_EXEC option is gone; state_exec is always built), so the wasm # gallery carries the program lanes that Ariadne apps run on. - # Bumped 3 -> 4: ships THIRD_PARTY_NOTICES.md + LICENSE (native share/cvcgl-examples/ and + # Bumped 3 -> 16 (above the native 3.4.0+cvc.15 already published; revisions stay monotonic + # across platforms): ships THIRD_PARTY_NOTICES.md + LICENSE (native share/cvcgl-examples/ and # the wasm web/ payload), declares LGPL-2.1-only AND BSD-3-Clause AND MIT, and rebuilds the - # demos against cvcgl 3.4.0+cvc.16 (fast-draw low-memory mapper). - cvc_revision: 4 + # demos against cvcgl 3.4.0+cvc.17 (fast-draw low-memory mapper). + cvc_revision: 16 maintainer: "cvcpkg group" maintainer_email: "info@cvcpkg.org" maintainer_url: "https://cvcpkg.org" diff --git a/cvcpkg/recipes/cvcgl/recipe.yaml b/cvcpkg/recipes/cvcgl/recipe.yaml index a29b19c3b..7159f89b3 100644 --- a/cvcpkg/recipes/cvcgl/recipe.yaml +++ b/cvcpkg/recipes/cvcgl/recipe.yaml @@ -69,12 +69,13 @@ recipe: # _rect partial texture uploads, caster-aware shadow baking (setCastsShadow), and the cheaper # scene-bounds walk deferred under hidden chrome. ABI-stale: SceneGraph/GraphicsNode/ # NullGraphicNode layouts changed, so the published +cvc.14 must not be mixed with this build. - # Bumped 15 -> 16: the fast-draw low-memory mapper (LowMemoryPolyDataMapper: skips empty + # Bumped 15 -> 17 (16 went out as a native publish of master without this): + # the fast-draw low-memory mapper (LowMemoryPolyDataMapper: skips empty # cell-type agents and re-binds the cached shader program; ~2/3 fewer GL calls per frame on # WebGL2 with byte-identical pixels; CVCGL_LOWMEM_DRAW / setDrawPath for A/B), now the base of # StreamingLowMemoryPolyDataMapper; late shader replacements reach the GPU on the low-memory # path. License metadata now LGPL-2.1-only AND BSD-3-Clause (VTK 9.5.0-derived code). - cvc_revision: 16 + cvc_revision: 17 maintainer: "cvcpkg group" maintainer_email: "info@cvcpkg.org" maintainer_url: "https://cvcpkg.org" diff --git a/cvcpkg/recipes/libcvc/recipe.yaml b/cvcpkg/recipes/libcvc/recipe.yaml index 68fdaf60f..ac2dae670 100644 --- a/cvcpkg/recipes/libcvc/recipe.yaml +++ b/cvcpkg/recipes/libcvc/recipe.yaml @@ -134,11 +134,12 @@ recipe: # caster-aware shadow baking, the cheaper bounds walk). ABI: SceneGraph, GraphicsNode and # NullGraphicNode layouts changed (#512 and the frame-cost fixes) with the SOVERSION # unchanged, so dependents must rebuild. cvcgl bumps in lockstep (14 -> 15). - # Bumped 15 -> 16: lockstep with cvcgl's fast-draw low-memory mapper + # Bumped 15 -> 17 (16 went out as a native publish of master without this): + # lockstep with cvcgl's fast-draw low-memory mapper # (LowMemoryPolyDataMapper, ~2/3 fewer GL calls per frame on WebGL2) and the expanded # THIRD_PARTY_NOTICES.md. libcvc's own sources are unchanged by it; the bump keeps the pair's # revisions aligned for consumers that pin both. - cvc_revision: 16 + cvc_revision: 17 maintainer: "cvcpkg group" maintainer_email: "info@cvcpkg.org" maintainer_url: "https://cvcpkg.org" From 10a6ea0daa22f9d7126d50939b4daeda3a46a371 Mon Sep 17 00:00:00 2001 From: Joe Rivera Date: Fri, 2 Oct 2026 14:57:45 -0500 Subject: [PATCH 16/18] test(cvcGL): lowmem fastdraw -- Apple LSB envelope across program caches and windows On macOS CI ("Apple Software Renderer", GL 4.1 APPLE-23.1.1) the first attempt of every package-macos job on the last three heads failed the same three pixel checks: Fast after the shader cache's programs were released and recompiled, the new window's Stock frame against the old window's Stock frame, and the new window's Fast frame against it -- 7 to 10 differing bytes, every one at max delta 1, more than any pair of Stock frames sampled in one window that run. The Stock-against-Stock failure never touches the fast path: Apple's software renderer rounds differently across a program recompile or a new context than within one window. Apple-tolerance mode only: a check whose two frames come from different program caches or windows is now judged by a fixed envelope, pure LSB rounding -- max per-byte delta <= 1 and <= 64 differing bytes -- that no in-window sample widens, so no frame can excuse itself. In-window checks keep the sampled-pair rule. Every pixel check prints the rule it was judged by. Point picking on Apple's renderer may differ by 1 point per prop: Stock picked 1118 points of one prop where Fast (once also AllCellTypes, whose GL trace equals Stock's) picked 1117. Cell picking, and every other renderer, stay exact. Exact mode (NVIDIA, Mesa llvmpipe) is unchanged: every pair and every check byte for byte, 288 checks. --- src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp | 181 +++++++++++++++++------ 1 file changed, 138 insertions(+), 43 deletions(-) diff --git a/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp b/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp index 4c8b9089d..df49a1860 100644 --- a/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp +++ b/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp @@ -43,9 +43,11 @@ // Stock frame right after Fast, and frames across a recompile and across a // new window -- byte for byte, and so do pairs of Stock frames of one GL // trace. Only on Apple's renderer (GL version or renderer string naming -// Apple), which varies by a few bytes now and then, may a check differ by -// no more than one such pair (same scene and window, neither right after -// Fast) differed in the same run; +// Apple), which varies by a few bytes now and then, may a check in one +// window differ by no more than one such pair (same scene and window, +// neither right after Fast) differed in the same run, and a check across +// a recompile or a new window by LSB rounding only (no byte by more than +// 1, at most 64 bytes); see checkPixels(); // 5. the saving: Fast draws a single-cell-type mesh in <= 2 VAO binds and a // fraction of the stock calls, with no desktop round-trip in any draw // beyond Stock's (none, but VTK's own warning on a line width the driver @@ -55,7 +57,9 @@ // and the scene re-opened in a new window all go back through the full // lookup (or a recompile) and still match; // 7. guards: an offset layout where skipping would change a drawn type's -// depth offset, picking and vertex visibility all draw the stock way; +// depth offset, picking and vertex visibility all draw the stock way (and +// pick the same; on Apple's renderer point picking may differ by 1 point +// per prop, see comparePicking()); // 8. a plain GeometryNode's shader replacement added (and cleared) after its // first draw reaches the GPU, on every draw path. // Frames are rendered without MSAA: with VTK's default 8x, NVIDIA's resolve @@ -1177,48 +1181,80 @@ std::string describe(const PixelDiff &d, int w) { // call with its arguments and uniform values, in order, in one window -- can // differ only by what the rasteriser does with that stream. NVIDIA and Mesa // llvmpipe give the same bytes every time. Apple's software renderer (macOS -// CI, GL 4.1) does not: with shadows on, AllCellTypes and Stock frames -- -// identical traces -- came out 1 to 4 bytes apart, and frames across a -// recompile or a new window up to 9, in some runs and not in others (never -// with shadows off). +// CI, "Apple Software Renderer", GL 4.1) does not: with shadows on, +// AllCellTypes and Stock frames -- identical traces -- came out 1 to 7 bytes +// apart in one window, in some runs and not in others (never with shadows +// off). Across a program recompile (the shader cache's programs released) or a +// new window (new context, new cache) it rounds differently again: on the +// first attempt of every macOS CI run over three heads, Stock against Stock +// across a new window -- the fast path not involved -- and Fast across a +// recompile or a new window came out 7 to 10 bytes apart (1 to 10 over all +// attempts), every one by a single LSB (max delta 1), and more than any pair +// sampled in one window that run. // // Exact mode (every renderer but Apple's): a sampled pair -- two Stock frames // of one GL trace -- that differs fails the run on its own, and every pixel // check is byte for byte. // -// Apple-tolerance mode (the GL version or renderer string names Apple): a -// pixel check passes when one sampled pair -- in the same scene, so the -// shadowed scene's noise excuses nothing elsewhere -- differed by at least as -// many bytes AND by at least as large a step: an envelope an actual pair -// filled, not the most bytes of one pair with the largest step of another. -// Which pairs may be samples narrows what they can excuse: +// Apple-tolerance mode (the GL version or renderer string names Apple), by +// where a check's two frames come from (FramePair): +// +// Across program caches or windows (FramePair::AcrossPrograms): a fixed +// envelope, pure LSB rounding -- no byte off by more than 1, at most 64 bytes +// (kSelfVarianceMaxBytes). It is the renderer's, not the run's: no sampled +// pair widens it, so no frame can excuse itself, and a shading change (a +// colour off by 2 or more) or a changed region (more than 64 bytes) fails. +// Pairs across a recompile or a re-open are never samples: they would measure +// as noise the very difference these checks look for. +// +// In one window and one shader cache (FramePair::InWindow): a pixel check +// passes when one sampled pair -- in the same scene, so the shadowed scene's +// noise excuses nothing elsewhere -- differed by at least as many bytes AND by +// at least as large a step: an envelope an actual pair filled, not the most +// bytes of one pair with the largest step of another. Which pairs may be +// samples narrows what they can excuse: // - both frames Stock, in one window (a pair across a re-open would measure // as noise the very difference the cross-window check looks for); // - neither right after a Fast run (whatever a Fast frame leaves that the // trace cannot see -- buffer contents, the untraced warm-up frame -- would // count as noise); a Stock frame right after Fast is judged instead; // - no compared pair is a sampled pair. -// That narrows it, no more: a sampled pair and a check may share a frame (a -// scene's first counted Stock run is in two samples and in all six of its -// checks; the new window's first sampled run is in two of its checks), and if -// that frame alone is wrong the pair measures exactly the check's difference -// and excuses it. Hence the relaxation is Apple's alone, where it is needed. +// That narrows it, no more: a sampled pair and an in-window check may share a +// frame (a scene's first counted Stock run is in two samples and in all six of +// its checks; the new window's first sampled run is in one), and if that frame +// alone is wrong the pair measures exactly the check's difference and excuses +// it -- in one window; a check of that frame across the re-open is held to the +// LSB envelope. Hence the relaxation is Apple's alone, where it is needed. // Every check is judged at the end, once the run's samples are in. The GL // traces stay the exact oracle on every renderer. // // The most a sampled pair may differ by for the renderer to be judged on at -// all -- a noisier pair excuses nothing and fails the run: 64 bytes, ~7x the -// most seen on macOS (9), and a step of a few LSB. The old check logged byte +// all -- a noisier pair excuses nothing and fails the run: 64 bytes, ~6x the +// most seen on macOS (10), and a step of a few LSB. The old check logged byte // counts only, not the step; counts that are no multiple of 3 (1, 2, 4, 5) // mean pixels that changed in some colour channels and not the others, which -// is rounding, not a shading change. +// is rounding, not a shading change. The same byte cap bounds the LSB envelope +// across program caches and windows, whose step is 1: what macOS measured. constexpr long kSelfVarianceMaxBytes = 64; constexpr int kSelfVarianceMaxDelta = 8; +constexpr int kAppleLsbMaxDelta = 1; bool withinCap(const PixelDiff &d) { return d.bytes >= 0 && d.bytes <= kSelfVarianceMaxBytes && d.maxDelta <= kSelfVarianceMaxDelta; } +// Apple-tolerance mode, frames across program caches or windows: LSB rounding. +bool withinAppleLsbEnvelope(const PixelDiff &d) { + return d.bytes >= 0 && d.bytes <= kSelfVarianceMaxBytes && d.maxDelta <= kAppleLsbMaxDelta; +} + +// Where a pixel check's two frames come from, and so (Apple-tolerance mode +// only) the rule that judges them; exact mode judges both byte for byte. +enum class FramePair { + InWindow, // one window, one shader cache: against the sampled pairs + AcrossPrograms, // across a program recompile or a new window: LSB envelope +}; + // The renderer the frames are drawn by, as renderAvailable() reads it, and // whether its pixels are judged in Apple-tolerance mode (exact otherwise). std::string g_glRenderer, g_glVersion; @@ -1237,6 +1273,7 @@ struct PixelCheck { std::string scope, what; PixelDiff d; int w; + FramePair pair; }; std::vector g_pixelChecks; @@ -1285,8 +1322,9 @@ void sampleSelfVariance(const std::string &scope, const std::string &what, const // Deferred: judged by checkPixels() once the run's samples are in. void samePixels(const std::string &scope, const std::string &what, - const std::vector &a, const std::vector &b, int w) { - g_pixelChecks.push_back({scope, what, pixelDiff(a, b), w}); + const std::vector &a, const std::vector &b, int w, + FramePair pair = FramePair::InWindow) { + g_pixelChecks.push_back({scope, what, pixelDiff(a, b), w, pair}); } void checkPixels() { @@ -1301,20 +1339,36 @@ void checkPixels() { std::set scopes; for (const auto &c : g_pixelChecks) scopes.insert(c.scope); - std::printf("pixels (apple-tolerance mode, the renderer's self-variance: pairs of Stock frames " - "of identical GL traces, in one window, neither right after Fast)\n"); + const std::string envelope = fmt("Apple LSB envelope (delta<=%.0f, bytes<=%.0f)", + double(kAppleLsbMaxDelta), double(kSelfVarianceMaxBytes)); + std::printf("pixels (apple-tolerance mode: in one window, the renderer's self-variance -- pairs " + "of Stock frames of identical GL traces, in one window, neither right after Fast; " + "across a recompile or a new window, the %s)\n", + envelope.c_str()); for (const auto &scope : scopes) { int noisy = 0; for (const auto &n : g_noisy) noisy += n.scope == scope; std::string line = " " + scope + fmt(": %.0f pairs, ", double(g_pairs[scope])); line += noisy ? fmt("%.0f differ (listed above)", double(noisy)) - : std::string("none differ, so its checks are byte-exact"); + : std::string("none differ, so its in-window checks are byte-exact"); if (g_pairsSkipped[scope]) line += fmt("; %.0f not sampled", double(g_pairsSkipped[scope])); std::printf("%s\n", line.c_str()); } for (const auto &c : g_pixelChecks) { + std::string detail = describe(c.d, c.w); + if (c.pair == FramePair::AcrossPrograms) { + // The fixed envelope alone: no sampled pair widens it. + const bool lsb = withinAppleLsbEnvelope(c.d); + if (c.d.bytes == 0) + detail += " (across program caches / windows: " + envelope + ")"; + else + detail += std::string(lsb ? ": within " : ": BEYOND ") + envelope + + ", across program caches / windows"; + check(c.d.bytes == 0 || lsb, c.what, detail); + continue; + } const SelfVarianceSample *excuse = nullptr; for (const auto &n : g_noisy) if (c.d.bytes > 0 && n.scope == c.scope && withinCap(n.d) && n.d.bytes >= c.d.bytes && @@ -1322,11 +1376,13 @@ void checkPixels() { excuse = &n; break; } - std::string detail = describe(c.d, c.w); - if (excuse) - detail += ": within " + excuse->what + " (" + describe(excuse->d, excuse->w) + ")"; - else if (c.d.bytes > 0) - detail += ": BEYOND every pair of Stock frames sampled in " + c.scope; + if (c.d.bytes == 0) + detail += " (in one window: a sampled pair of Stock frames)"; + else if (excuse) + detail += + ": in one window, within " + excuse->what + " (" + describe(excuse->d, excuse->w) + ")"; + else + detail += ": in one window, BEYOND every pair of Stock frames sampled in " + c.scope; check(c.d.bytes == 0 || excuse, c.what, detail); } // A renderer noisy enough to blur a real difference is not one to judge on. @@ -1415,9 +1471,12 @@ bool renderAvailable(cvc::app &app) { double(win->GetMaximumHardwareLineWidth())); } const char *mode = g_appleTolerance ? "apple-tolerance" : "exact"; - const char *rule = g_appleTolerance - ? "a check may differ by no more than a sampled pair of Stock frames" - : "every pair of Stock frames and every check byte for byte"; + const char *rule = + g_appleTolerance + ? "a check in one window may differ by no more than a sampled pair of Stock frames, " + "one across a recompile or a new window by LSB rounding only (delta<=1, " + "bytes<=64); point picking by <= 1 point per prop" + : "every pair of Stock frames and every check byte for byte"; std::printf(" pixel checks: %s (OpenGL renderer \"%s\", version \"%s\") -- %s\n", mode, g_glRenderer.c_str(), g_glVersion.c_str(), rule); } @@ -1527,10 +1586,28 @@ void checkTraces(const std::string &what, const Trace &stock, const Trace &all, // trace of the selection, the stats. struct Pick { std::string signature; + std::vector> hits; // (prop id, items), in prop-id order Trace trace; LowMemoryPolyDataMapper::Stats stats; }; +// Apple-tolerance mode, point picking: the same props, each picked within +// `tolerance` items of the reference. Where the counts differ, *diff lists them. +bool pickCountsWithin(const Pick &ref, const Pick &p, double tolerance, const char *name, + std::string *diff) { + if (p.hits.size() != ref.hits.size()) + return false; + for (size_t i = 0; i < ref.hits.size(); ++i) { + if (p.hits[i].first != ref.hits[i].first || + std::abs(p.hits[i].second - ref.hits[i].second) > tolerance) + return false; + if (p.hits[i].second != ref.hits[i].second) + *diff += fmt("; prop %.0f: Stock %.0f, ", ref.hits[i].first, ref.hits[i].second) + name + + fmt(" %.0f", p.hits[i].second); + } + return true; +} + Pick pick(Scene &s, DrawPath p, int association) { LowMemoryPolyDataMapper::setDrawPath(p); s.sr->render(); // the same state before each path's selection @@ -1543,7 +1620,7 @@ Pick pick(Scene &s, DrawPath p, int association) { vtkSmartPointer res = vtk::TakeSmartPointer(sel->Select()); Pick out; out.trace = cvcgl_test::glcallsTraceStop(); - std::vector> hits; // (prop id, items), in prop-id order + auto &hits = out.hits; for (unsigned i = 0; res && i < res->GetNumberOfNodes(); ++i) { vtkSelectionNode *n = res->GetNode(i); hits.emplace_back(n->GetProperties()->Get(vtkSelectionNode::PROP_ID()), @@ -1597,10 +1674,26 @@ void comparePicking(Scene &s, const std::string &label) { fmt(" (+%.0f in the ordinary depth frame)", ordinary)); const bool sameSelection = !stock.signature.empty() && fast.signature == stock.signature && all.signature == stock.signature; - check(sameSelection, what + " selects the same on every path", - sameSelection ? stock.signature - : "Stock " + stock.signature + " AllCellTypes " + all.signature + " Fast " + - fast.signature); + // Apple's renderer, point picking only: the selector's point pass picked + // 1117 points of a prop on one path and 1118 on another, in runs where + // AllCellTypes' GL trace equalled Stock's -- the rasteriser's rounding at + // a point's edge, like its pixels. There a path may pick 1 point more or + // fewer per prop (the same props); cell picking, and every other renderer, + // stay exact. + std::string within; + const bool nearSelection = !sameSelection && g_appleTolerance && points && + !stock.signature.empty() && + pickCountsWithin(stock, all, 1.0, "AllCellTypes", &within) && + pickCountsWithin(stock, fast, 1.0, "Fast", &within); + std::string detail = sameSelection ? stock.signature + : "Stock " + stock.signature + " AllCellTypes " + + all.signature + " Fast " + fast.signature; + if (nearSelection) + detail += ": within 1 point per prop on Apple's renderer" + within; + else if (g_appleTolerance && points) + detail += sameSelection ? " (Apple's renderer: <= 1 point per prop)" + : ": BEYOND 1 point per prop on Apple's renderer"; + check(sameSelection || nearSelection, what + " selects the same on every path", detail); checkTraces(what, stock.trace, all.trace, fast.trace, fast.stats, points); } LowMemoryPolyDataMapper::setDrawPath(DrawPath::Fast); @@ -1836,8 +1929,9 @@ void testLifecycle(Scene &s, const std::string &scope) { LowMemoryPolyDataMapper::setDrawPath(DrawPath::Fast); s.sr->render(); const PathRun after = runPath(s, DrawPath::Fast); + // Across the recompile: on Apple's renderer, the LSB envelope (checkPixels()). samePixels(scope, "programs released in the cache: Fast recompiles and matches Stock", - before.steadyPx, after.steadyPx, s.w); + before.steadyPx, after.steadyPx, s.w, FramePair::AcrossPrograms); // The same scene -- the same nodes and mappers -- closed and re-opened in a // new window (new context, new shader cache): the first draws resolve fully @@ -1864,10 +1958,11 @@ void testLifecycle(Scene &s, const std::string &scope) { const PathRun b = runPath(s, DrawPath::Fast); sampleSelfVariance(scope, "new window: Stock twice", a, a2, s.w); samePixels(scope, "new window: Fast pixels == Stock", a.steadyPx, b.steadyPx, s.w); + // Across the windows: on Apple's renderer, the LSB envelope (checkPixels()). samePixels(scope, "new window: the Stock frame == the old window's", before.steadyPx, a.steadyPx, - s.w); + s.w, FramePair::AcrossPrograms); samePixels(scope, "new window: the Fast frame == the old window's Stock frame", before.steadyPx, - b.steadyPx, s.w); + b.steadyPx, s.w, FramePair::AcrossPrograms); } // ── 7. guards ──────────────────────────────────────────────────────────────── From f2fc174d91f87545aa6ad227e84d979610c354c2 Mon Sep 17 00:00:00 2001 From: Joe Rivera Date: Fri, 2 Oct 2026 17:16:52 -0500 Subject: [PATCH 17/18] test(cvcGL): lowmem fastdraw -- one fixed Apple LSB envelope for every pixel check macOS CI ("Apple Software Renderer", GL 4.1 APPLE-23.1.1) still failed first attempts on pixel checks in one window, which the sampled-pair rule judged: four shadows-on checks (Fast against Stock, and a Stock frame right after Fast against Stock, first and steady frames) 1 to 2 bytes apart at max delta 1, in a run where all ten pairs of Stock frames sampled in that window were byte-identical; on earlier heads, Fast after a mapper's resources were released, 1 to 8 bytes at max delta 1. Apple's renderer rounds by 1 LSB nondeterministically within a window too, so no pair a run samples can bound what its other frames do. Apple-tolerance mode only: every pixel check -- in one window, across a program recompile, across a new window, and the late shader replacement's cleared frame -- now passes iff its two frames are byte-identical, or differ by LSB rounding alone: max per-byte delta <= 1 (kAppleLsbMaxDelta) and at most 64 differing bytes (kSelfVarianceMaxBytes). The envelope is fixed, so no frame can excuse itself. Every difference measured on macOS so far fits it: 1 to 8 bytes in one window, 1 to 10 across a recompile or a new window, all at max delta 1. The pairs of Stock frames are still sampled and printed, as diagnostics only; they excuse nothing (the sampled-pair excuse, its cap check and FramePair are gone). The startup "pixel checks:" line states the rule. Point picking on Apple's renderer may still differ by 1 point per prop; cell picking stays exact. Exact mode (NVIDIA, Mesa llvmpipe) is unchanged: every pair and every check byte for byte, 288 checks, the same output line for line. --- src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp | 307 +++++++++-------------- 1 file changed, 121 insertions(+), 186 deletions(-) diff --git a/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp b/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp index df49a1860..2b52bf3d5 100644 --- a/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp +++ b/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp @@ -43,11 +43,11 @@ // Stock frame right after Fast, and frames across a recompile and across a // new window -- byte for byte, and so do pairs of Stock frames of one GL // trace. Only on Apple's renderer (GL version or renderer string naming -// Apple), which varies by a few bytes now and then, may a check in one -// window differ by no more than one such pair (same scene and window, -// neither right after Fast) differed in the same run, and a check across -// a recompile or a new window by LSB rounding only (no byte by more than -// 1, at most 64 bytes); see checkPixels(); +// Apple), which rounds a few bytes 1 LSB apart now and then, in one window +// and across windows, is every pixel check held to a fixed envelope +// instead: identical, or no byte off by more than 1 and at most 64 bytes; +// there the pairs of Stock frames are printed, never judged; see +// checkPixels(); // 5. the saving: Fast draws a single-cell-type mesh in <= 2 VAO binds and a // fraction of the stock calls, with no desktop round-trip in any draw // beyond Stock's (none, but VTK's own warning on a line width the driver @@ -1181,85 +1181,68 @@ std::string describe(const PixelDiff &d, int w) { // call with its arguments and uniform values, in order, in one window -- can // differ only by what the rasteriser does with that stream. NVIDIA and Mesa // llvmpipe give the same bytes every time. Apple's software renderer (macOS -// CI, "Apple Software Renderer", GL 4.1) does not: with shadows on, -// AllCellTypes and Stock frames -- identical traces -- came out 1 to 7 bytes -// apart in one window, in some runs and not in others (never with shadows -// off). Across a program recompile (the shader cache's programs released) or a -// new window (new context, new cache) it rounds differently again: on the -// first attempt of every macOS CI run over three heads, Stock against Stock -// across a new window -- the fast path not involved -- and Fast across a -// recompile or a new window came out 7 to 10 bytes apart (1 to 10 over all -// attempts), every one by a single LSB (max delta 1), and more than any pair -// sampled in one window that run. +// CI, "Apple Software Renderer", GL 4.1 APPLE-23.1.1) does not, and not only +// across windows. Measured on the first attempts of macOS CI runs, every +// difference by a single LSB (max delta 1), in some runs and not in others: +// - in one window, shadows on (never off): AllCellTypes, Fast and a Stock +// frame right after Fast against Stock 1 to 7 bytes apart, Fast after a +// mapper's resources were released 1 to 8 -- in runs where every pair of +// Stock frames sampled in that window was byte-identical, so no pair a run +// samples can bound what its other frames do; +// - across a program recompile (the shader cache's programs released) or a +// new window (new context, new cache): Fast, and Stock against Stock -- +// the fast path not involved -- 1 to 10 bytes apart. // -// Exact mode (every renderer but Apple's): a sampled pair -- two Stock frames -// of one GL trace -- that differs fails the run on its own, and every pixel -// check is byte for byte. +// Exact mode (every renderer but Apple's): every pixel check is byte for byte, +// and a sampled pair -- two Stock frames of one GL trace, in one window, +// neither right after Fast -- that differs fails the run on its own. // -// Apple-tolerance mode (the GL version or renderer string names Apple), by -// where a check's two frames come from (FramePair): -// -// Across program caches or windows (FramePair::AcrossPrograms): a fixed -// envelope, pure LSB rounding -- no byte off by more than 1, at most 64 bytes -// (kSelfVarianceMaxBytes). It is the renderer's, not the run's: no sampled -// pair widens it, so no frame can excuse itself, and a shading change (a -// colour off by 2 or more) or a changed region (more than 64 bytes) fails. -// Pairs across a recompile or a re-open are never samples: they would measure -// as noise the very difference these checks look for. -// -// In one window and one shader cache (FramePair::InWindow): a pixel check -// passes when one sampled pair -- in the same scene, so the shadowed scene's -// noise excuses nothing elsewhere -- differed by at least as many bytes AND by -// at least as large a step: an envelope an actual pair filled, not the most -// bytes of one pair with the largest step of another. Which pairs may be -// samples narrows what they can excuse: -// - both frames Stock, in one window (a pair across a re-open would measure -// as noise the very difference the cross-window check looks for); -// - neither right after a Fast run (whatever a Fast frame leaves that the -// trace cannot see -- buffer contents, the untraced warm-up frame -- would -// count as noise); a Stock frame right after Fast is judged instead; -// - no compared pair is a sampled pair. -// That narrows it, no more: a sampled pair and an in-window check may share a -// frame (a scene's first counted Stock run is in two samples and in all six of -// its checks; the new window's first sampled run is in one), and if that frame -// alone is wrong the pair measures exactly the check's difference and excuses -// it -- in one window; a check of that frame across the re-open is held to the -// LSB envelope. Hence the relaxation is Apple's alone, where it is needed. -// Every check is judged at the end, once the run's samples are in. The GL -// traces stay the exact oracle on every renderer. -// -// The most a sampled pair may differ by for the renderer to be judged on at -// all -- a noisier pair excuses nothing and fails the run: 64 bytes, ~6x the -// most seen on macOS (10), and a step of a few LSB. The old check logged byte -// counts only, not the step; counts that are no multiple of 3 (1, 2, 4, 5) -// mean pixels that changed in some colour channels and not the others, which -// is rounding, not a shading change. The same byte cap bounds the LSB envelope -// across program caches and windows, whose step is 1: what macOS measured. +// Apple-tolerance mode (the GL version or renderer string names Apple): every +// pixel check -- in one window, across a recompile, across a new window -- +// passes iff its two frames are byte-identical or differ by LSB rounding +// alone: no byte off by more than 1 (kAppleLsbMaxDelta), at most 64 bytes +// (kSelfVarianceMaxBytes, ~6x the most measured). The envelope is the +// renderer's, fixed, not the run's: no frame widens it, so no frame can +// excuse itself, and a shading change (a colour off by 2 or more) or a +// changed region (more than 64 bytes) fails. The pairs of Stock frames are +// still sampled and printed, as a diagnostic of the renderer's self-variance +// in that run; they excuse nothing. Byte counts that are no multiple of 3 (1, +// 2, 4, 5, ...) mean pixels that changed in some colour channels and not the +// others: rounding, not a shading change. Hence the relaxation is Apple's +// alone, where it is needed; the GL traces stay the exact oracle on every +// renderer, and point picking has its own rule (comparePicking()). constexpr long kSelfVarianceMaxBytes = 64; -constexpr int kSelfVarianceMaxDelta = 8; constexpr int kAppleLsbMaxDelta = 1; -bool withinCap(const PixelDiff &d) { - return d.bytes >= 0 && d.bytes <= kSelfVarianceMaxBytes && d.maxDelta <= kSelfVarianceMaxDelta; -} - -// Apple-tolerance mode, frames across program caches or windows: LSB rounding. +// Apple-tolerance mode: identical, or LSB rounding at a few pixels. bool withinAppleLsbEnvelope(const PixelDiff &d) { return d.bytes >= 0 && d.bytes <= kSelfVarianceMaxBytes && d.maxDelta <= kAppleLsbMaxDelta; } -// Where a pixel check's two frames come from, and so (Apple-tolerance mode -// only) the rule that judges them; exact mode judges both byte for byte. -enum class FramePair { - InWindow, // one window, one shader cache: against the sampled pairs - AcrossPrograms, // across a program recompile or a new window: LSB envelope -}; +std::string appleLsbEnvelope() { + return fmt("Apple LSB envelope (delta<=%.0f, bytes<=%.0f)", double(kAppleLsbMaxDelta), + double(kSelfVarianceMaxBytes)); +} // The renderer the frames are drawn by, as renderAvailable() reads it, and // whether its pixels are judged in Apple-tolerance mode (exact otherwise). std::string g_glRenderer, g_glVersion; bool g_appleTolerance = false; +// Every pixel check's verdict, and its detail: byte for byte, or in +// Apple-tolerance mode the fixed LSB envelope -- the same rule wherever the +// two frames come from. +bool pixelsMatch(const PixelDiff &d) { + return g_appleTolerance ? withinAppleLsbEnvelope(d) : d.bytes == 0; +} + +std::string pixelVerdict(const PixelDiff &d, int w) { + std::string detail = describe(d, w); + if (g_appleTolerance && d.bytes > 0) + detail += (pixelsMatch(d) ? ": within the " : ": BEYOND the ") + appleLsbEnvelope(); + return detail; +} + struct SelfVarianceSample { std::string scope, what; PixelDiff d; @@ -1270,10 +1253,9 @@ std::map g_pairs; // pairs sampled, per scope std::map g_pairsSkipped; // not sampled (traces differ), per scope struct PixelCheck { - std::string scope, what; + std::string what; PixelDiff d; int w; - FramePair pair; }; std::vector g_pixelChecks; @@ -1293,16 +1275,16 @@ void sampleSelfVariance(const std::string &scope, const std::string &what, const return; } ++g_pairs[scope]; - if (!g_appleTolerance) { // exact mode: the pair is itself a check, never an excuse + if (!g_appleTolerance) { // exact mode: the pair is itself a check check(d.bytes == 0, what + " is byte-identical on a deterministic renderer", d.bytes ? describe(d, w) + ", identical GL traces" : std::string()); return; } if (d.bytes == 0) return; - g_noisy.push_back({scope, what, d, w}); - std::printf(" self-variance: %s -- %s, identical GL traces\n", what.c_str(), - describe(d, w).c_str()); + g_noisy.push_back({scope, what, d, w}); // a diagnostic: it excuses nothing + std::printf(" self-variance (diagnostic): %s -- %s, identical GL traces\n", what.c_str(), + pixelVerdict(d, w).c_str()); } // Two Stock runs of one draw path: the first frames (a shadow bake when @@ -1320,11 +1302,10 @@ void sampleSelfVariance(const std::string &scope, const std::string &what, const ++g_pairsSkipped[scope]; } -// Deferred: judged by checkPixels() once the run's samples are in. -void samePixels(const std::string &scope, const std::string &what, - const std::vector &a, const std::vector &b, int w, - FramePair pair = FramePair::InWindow) { - g_pixelChecks.push_back({scope, what, pixelDiff(a, b), w, pair}); +// Deferred: judged by checkPixels(), after the run's samples are printed. +void samePixels(const std::string &what, const std::vector &a, + const std::vector &b, int w) { + g_pixelChecks.push_back({what, pixelDiff(a, b), w}); } void checkPixels() { @@ -1332,74 +1313,30 @@ void checkPixels() { return; if (!g_appleTolerance) { std::printf("pixels (exact mode: byte for byte)\n"); - for (const auto &c : g_pixelChecks) - check(c.d.bytes == 0, c.what, describe(c.d, c.w)); - return; + } else { + std::printf("pixels (apple-tolerance mode: every check, in one window or across a recompile " + "or a new window, byte-identical or within the %s; the pairs of Stock frames " + "sampled are a diagnostic and excuse nothing)\n", + appleLsbEnvelope().c_str()); + std::set scopes; + for (const auto &p : g_pairs) + scopes.insert(p.first); + for (const auto &p : g_pairsSkipped) + scopes.insert(p.first); + for (const auto &scope : scopes) { + int noisy = 0; + for (const auto &n : g_noisy) + noisy += n.scope == scope; + std::string line = " self-variance (diagnostic): " + scope + + fmt(": %.0f pairs of Stock frames, ", double(g_pairs[scope])); + line += noisy ? fmt("%.0f differ (listed above)", double(noisy)) : std::string("none differ"); + if (g_pairsSkipped[scope]) + line += fmt("; %.0f not sampled", double(g_pairsSkipped[scope])); + std::printf("%s\n", line.c_str()); + } } - std::set scopes; for (const auto &c : g_pixelChecks) - scopes.insert(c.scope); - const std::string envelope = fmt("Apple LSB envelope (delta<=%.0f, bytes<=%.0f)", - double(kAppleLsbMaxDelta), double(kSelfVarianceMaxBytes)); - std::printf("pixels (apple-tolerance mode: in one window, the renderer's self-variance -- pairs " - "of Stock frames of identical GL traces, in one window, neither right after Fast; " - "across a recompile or a new window, the %s)\n", - envelope.c_str()); - for (const auto &scope : scopes) { - int noisy = 0; - for (const auto &n : g_noisy) - noisy += n.scope == scope; - std::string line = " " + scope + fmt(": %.0f pairs, ", double(g_pairs[scope])); - line += noisy ? fmt("%.0f differ (listed above)", double(noisy)) - : std::string("none differ, so its in-window checks are byte-exact"); - if (g_pairsSkipped[scope]) - line += fmt("; %.0f not sampled", double(g_pairsSkipped[scope])); - std::printf("%s\n", line.c_str()); - } - for (const auto &c : g_pixelChecks) { - std::string detail = describe(c.d, c.w); - if (c.pair == FramePair::AcrossPrograms) { - // The fixed envelope alone: no sampled pair widens it. - const bool lsb = withinAppleLsbEnvelope(c.d); - if (c.d.bytes == 0) - detail += " (across program caches / windows: " + envelope + ")"; - else - detail += std::string(lsb ? ": within " : ": BEYOND ") + envelope + - ", across program caches / windows"; - check(c.d.bytes == 0 || lsb, c.what, detail); - continue; - } - const SelfVarianceSample *excuse = nullptr; - for (const auto &n : g_noisy) - if (c.d.bytes > 0 && n.scope == c.scope && withinCap(n.d) && n.d.bytes >= c.d.bytes && - n.d.maxDelta >= c.d.maxDelta) { - excuse = &n; - break; - } - if (c.d.bytes == 0) - detail += " (in one window: a sampled pair of Stock frames)"; - else if (excuse) - detail += - ": in one window, within " + excuse->what + " (" + describe(excuse->d, excuse->w) + ")"; - else - detail += ": in one window, BEYOND every pair of Stock frames sampled in " + c.scope; - check(c.d.bytes == 0 || excuse, c.what, detail); - } - // A renderer noisy enough to blur a real difference is not one to judge on. - const SelfVarianceSample *beyond = nullptr, *largest = nullptr; - for (const auto &n : g_noisy) { - if (!beyond && !withinCap(n.d)) - beyond = &n; - if (!largest || std::make_pair(n.d.bytes, n.d.maxDelta) > - std::make_pair(largest->d.bytes, largest->d.maxDelta)) - largest = &n; - } - const SelfVarianceSample *shown = beyond ? beyond : largest; - check(!beyond, - fmt("the renderer's self-variance, if any, is a few LSB at a few pixels (every sampled " - "pair <= %.0f bytes, max delta <= %.0f)", - double(kSelfVarianceMaxBytes), double(kSelfVarianceMaxDelta)), - shown ? shown->what + ": " + describe(shown->d, shown->w) : std::string("none differ")); + check(pixelsMatch(c.d), c.what, pixelVerdict(c.d, c.w)); } // Every pixel the background colour? (a frame that drew nothing) @@ -1471,14 +1408,15 @@ bool renderAvailable(cvc::app &app) { double(win->GetMaximumHardwareLineWidth())); } const char *mode = g_appleTolerance ? "apple-tolerance" : "exact"; - const char *rule = - g_appleTolerance - ? "a check in one window may differ by no more than a sampled pair of Stock frames, " - "one across a recompile or a new window by LSB rounding only (delta<=1, " - "bytes<=64); point picking by <= 1 point per prop" - : "every pair of Stock frames and every check byte for byte"; + std::string rule = "every pair of Stock frames and every check byte for byte"; + if (g_appleTolerance) + rule = "every check, in one window or across a recompile or a new window, byte-identical or " + "within the " + + appleLsbEnvelope() + + ", pairs of Stock frames printed only (they excuse nothing); point picking by <= 1 " + "point per prop, cell picking exact"; std::printf(" pixel checks: %s (OpenGL renderer \"%s\", version \"%s\") -- %s\n", mode, - g_glRenderer.c_str(), g_glVersion.c_str(), rule); + g_glRenderer.c_str(), g_glVersion.c_str(), rule.c_str()); } if (drew) return true; @@ -1706,7 +1644,7 @@ void comparePaths(Scene &s, const std::string &label, bool shadows) { for (DrawPath p : {DrawPath::Fast, DrawPath::Stock}) runPath(s, p); // Pairs of the Stock runs before Fast's (one after AllCellTypes) are checked - // byte-identical -- on Apple's renderer, sampled as its self-variance; the + // byte-identical -- on Apple's renderer, printed as its self-variance; the // Stock run right after Fast is judged, like the AllCellTypes and Fast frames. const PathRun stock = runPath(s, DrawPath::Stock); const PathRun stock2 = runPath(s, DrawPath::Stock); @@ -1730,18 +1668,16 @@ void comparePaths(Scene &s, const std::string &label, bool shadows) { all.firstTrace, fast.firstTrace, fast.firstStats); checkTraces(label + ", steady frame", stock.steadyTrace, all.steadyTrace, fast.steadyTrace, fast.steadyStats); - samePixels(label, label + ": AllCellTypes pixels == Stock, first frame", stock.firstPx, - all.firstPx, s.w); - samePixels(label, label + ": AllCellTypes pixels == Stock, steady frame", stock.steadyPx, - all.steadyPx, s.w); - samePixels(label, label + ": Fast pixels == Stock, first frame", stock.firstPx, fast.firstPx, + samePixels(label + ": AllCellTypes pixels == Stock, first frame", stock.firstPx, all.firstPx, s.w); - samePixels(label, label + ": Fast pixels == Stock, steady frame", stock.steadyPx, fast.steadyPx, + samePixels(label + ": AllCellTypes pixels == Stock, steady frame", stock.steadyPx, all.steadyPx, s.w); - samePixels(label, label + ": a Stock frame right after Fast == Stock, first frame", stock.firstPx, + samePixels(label + ": Fast pixels == Stock, first frame", stock.firstPx, fast.firstPx, s.w); + samePixels(label + ": Fast pixels == Stock, steady frame", stock.steadyPx, fast.steadyPx, s.w); + samePixels(label + ": a Stock frame right after Fast == Stock, first frame", stock.firstPx, afterFast.firstPx, s.w); - samePixels(label, label + ": a Stock frame right after Fast == Stock, steady frame", - stock.steadyPx, afterFast.steadyPx, s.w); + samePixels(label + ": a Stock frame right after Fast == Stock, steady frame", stock.steadyPx, + afterFast.steadyPx, s.w); std::printf(" frame GL calls, first : Stock %s\n", stock.first.str().c_str()); std::printf(" Fast %s\n", fast.first.str().c_str()); @@ -1911,15 +1847,15 @@ void testLifecycle(Scene &s, const std::string &scope) { "after a mapper's ReleaseGraphicsResources its next draw looks the program up", fmt("lookups %.0f of %.0f draws", double(rel.programLookups), double(rel.draws))); const PathRun released = runPath(s, DrawPath::Fast); - // Stock to compare against: two runs, a self-variance pair (byte-identical - // but on Apple's renderer), after one that follows Fast (judged, not sampled). + // Stock to compare against: two runs, a self-variance pair (checked + // byte-identical but on Apple's renderer, where it is printed), after one + // that follows Fast (judged, not sampled). const PathRun afterFast = runPath(s, DrawPath::Stock); const PathRun before = runPath(s, DrawPath::Stock); const PathRun before2 = runPath(s, DrawPath::Stock); sampleSelfVariance(scope, "lifecycle: Stock twice", before, before2, s.w); - samePixels(scope, "after mapper release: Fast pixels == Stock", before.steadyPx, - released.steadyPx, s.w); - samePixels(scope, "lifecycle: a Stock frame right after Fast == Stock", before.steadyPx, + samePixels("after mapper release: Fast pixels == Stock", before.steadyPx, released.steadyPx, s.w); + samePixels("lifecycle: a Stock frame right after Fast == Stock", before.steadyPx, afterFast.steadyPx, s.w); // The shader cache's programs released and recompiled in place (the cache @@ -1929,9 +1865,10 @@ void testLifecycle(Scene &s, const std::string &scope) { LowMemoryPolyDataMapper::setDrawPath(DrawPath::Fast); s.sr->render(); const PathRun after = runPath(s, DrawPath::Fast); - // Across the recompile: on Apple's renderer, the LSB envelope (checkPixels()). - samePixels(scope, "programs released in the cache: Fast recompiles and matches Stock", - before.steadyPx, after.steadyPx, s.w, FramePair::AcrossPrograms); + // Across the recompile: byte for byte, or on Apple's renderer the LSB + // envelope, like every pixel check (checkPixels()). + samePixels("programs released in the cache: Fast recompiles and matches Stock", before.steadyPx, + after.steadyPx, s.w); // The same scene -- the same nodes and mappers -- closed and re-opened in a // new window (new context, new shader cache): the first draws resolve fully @@ -1949,20 +1886,19 @@ void testLifecycle(Scene &s, const std::string &scope) { fmt("lookups %.0f, skipped %.0f", double(st.programLookups), double(st.programLookupsSkipped))); // The window's first Stock run follows that Fast frame: neither sampled nor - // compared. The two after it are sampled -- in this window, never with the - // old window's: a pair across the re-open would measure as noise the very - // difference the cross-window checks look for. + // compared. The two after it are a self-variance pair -- in this window, + // never with the old window's, which the cross-window checks compare. runPath(s, DrawPath::Stock); const PathRun a = runPath(s, DrawPath::Stock); const PathRun a2 = runPath(s, DrawPath::Stock); const PathRun b = runPath(s, DrawPath::Fast); sampleSelfVariance(scope, "new window: Stock twice", a, a2, s.w); - samePixels(scope, "new window: Fast pixels == Stock", a.steadyPx, b.steadyPx, s.w); - // Across the windows: on Apple's renderer, the LSB envelope (checkPixels()). - samePixels(scope, "new window: the Stock frame == the old window's", before.steadyPx, a.steadyPx, - s.w, FramePair::AcrossPrograms); - samePixels(scope, "new window: the Fast frame == the old window's Stock frame", before.steadyPx, - b.steadyPx, s.w, FramePair::AcrossPrograms); + samePixels("new window: Fast pixels == Stock", a.steadyPx, b.steadyPx, s.w); + // Across the windows: byte for byte, or on Apple's renderer the LSB + // envelope, like every pixel check (checkPixels()). + samePixels("new window: the Stock frame == the old window's", before.steadyPx, a.steadyPx, s.w); + samePixels("new window: the Fast frame == the old window's Stock frame", before.steadyPx, + b.steadyPx, s.w); } // ── 7. guards ──────────────────────────────────────────────────────────────── @@ -2009,8 +1945,7 @@ void testGuards(cvc::app &app) { const PathRun stock2 = runPath(s, DrawPath::Stock); const PathRun fast = runPath(s, DrawPath::Fast); // last: the mapper stats below are Fast's sampleSelfVariance("guards", "POLYGON_OFFSET mode: Stock twice", stock, stock2, s.w); - samePixels("guards", "POLYGON_OFFSET mode: Fast pixels == Stock", stock.steadyPx, fast.steadyPx, - s.w); + samePixels("POLYGON_OFFSET mode: Fast pixels == Stock", stock.steadyPx, fast.steadyPx, s.w); checkTraces("POLYGON_OFFSET mode", stock.steadyTrace, all.steadyTrace, fast.steadyTrace, fast.steadyStats); const auto &zs = zero.mapper->stats(); @@ -2069,14 +2004,14 @@ void testLateShaderReplacement(cvc::app &app) { std::string(pathName(p)) + ": a replacement added after the first draw reaches the GPU", fmt("lookups %.0f", double(st.programLookups + st.stockDraws))); - // Exact on every renderer: an unshadowed 48x48 scene, where no renderer has - // been seen to vary, and no self-variance is sampled here. + // Byte for byte, or on Apple's renderer the LSB envelope, like every pixel + // check (no renderer has been seen to vary on this unshadowed 48x48 scene). n->clearShaderReplacements(); s.sr->render(); const PixelDiff back = pixelDiff(drawn, s.rgba()); - check(back.bytes == 0, + check(pixelsMatch(back), std::string(pathName(p)) + ": cleared again, the original frame comes back", - describe(back, s.w)); + pixelVerdict(back, s.w)); } LowMemoryPolyDataMapper::setDrawPath(DrawPath::Fast); } From bd3911d6ab941224ace611d56a770528fa640b80 Mon Sep 17 00:00:00 2001 From: Joe Rivera Date: Fri, 2 Oct 2026 17:54:14 -0500 Subject: [PATCH 18/18] test(cvcGL): lowmem fastdraw -- hold Apple's pairs of Stock frames to the LSB envelope too In Apple-tolerance mode the sampled pairs of Stock frames were only printed. The Stock frame drawn right after AllCellTypes in comparePaths() is compared only through such pairs, so on Apple's renderer it could be off by any amount and pass. Each pair is now a check there as well, against the same fixed envelope as every other pixel check (no byte off by more than 1, at most 64 bytes); a pair that differs at all is still printed as the renderer's self-variance. Exact mode is unchanged, and both modes now run the same 288 checks. Also: kSelfVarianceMaxBytes is renamed kAppleLsbMaxBytes (it bounds the Apple envelope only), and the comments and startup/summary lines that still said the pairs are printed only or judge other frames are updated. --- src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp | 49 ++++++++++++++---------- 1 file changed, 28 insertions(+), 21 deletions(-) diff --git a/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp b/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp index 2b52bf3d5..e1b9ae161 100644 --- a/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp +++ b/src/cvcGL/test/cvcgl_lowmem_fastdraw.cpp @@ -44,10 +44,9 @@ // new window -- byte for byte, and so do pairs of Stock frames of one GL // trace. Only on Apple's renderer (GL version or renderer string naming // Apple), which rounds a few bytes 1 LSB apart now and then, in one window -// and across windows, is every pixel check held to a fixed envelope -// instead: identical, or no byte off by more than 1 and at most 64 bytes; -// there the pairs of Stock frames are printed, never judged; see -// checkPixels(); +// and across windows, is every pixel check -- the pairs of Stock frames +// too -- held to a fixed envelope instead: identical, or no byte off by +// more than 1 and at most 64 bytes; see checkPixels(); // 5. the saving: Fast draws a single-cell-type mesh in <= 2 VAO binds and a // fraction of the stock calls, with no desktop round-trip in any draw // beyond Stock's (none, but VTK's own warning on a line width the driver @@ -1201,27 +1200,28 @@ std::string describe(const PixelDiff &d, int w) { // pixel check -- in one window, across a recompile, across a new window -- // passes iff its two frames are byte-identical or differ by LSB rounding // alone: no byte off by more than 1 (kAppleLsbMaxDelta), at most 64 bytes -// (kSelfVarianceMaxBytes, ~6x the most measured). The envelope is the +// (kAppleLsbMaxBytes, ~6x the most measured). The envelope is the // renderer's, fixed, not the run's: no frame widens it, so no frame can // excuse itself, and a shading change (a colour off by 2 or more) or a // changed region (more than 64 bytes) fails. The pairs of Stock frames are -// still sampled and printed, as a diagnostic of the renderer's self-variance -// in that run; they excuse nothing. Byte counts that are no multiple of 3 (1, -// 2, 4, 5, ...) mean pixels that changed in some colour channels and not the -// others: rounding, not a shading change. Hence the relaxation is Apple's -// alone, where it is needed; the GL traces stay the exact oracle on every -// renderer, and point picking has its own rule (comparePicking()). -constexpr long kSelfVarianceMaxBytes = 64; +// held to the same envelope, and those that differ are printed as well, a +// diagnostic of the renderer's self-variance in that run; they excuse +// nothing. Byte counts that are no multiple of 3 (1, 2, 4, 5, ...) mean +// pixels that changed in some colour channels and not the others: rounding, +// not a shading change. Hence the relaxation is Apple's alone, where it is +// needed; the GL traces stay the exact oracle on every renderer, and point +// picking has its own rule (comparePicking()). +constexpr long kAppleLsbMaxBytes = 64; constexpr int kAppleLsbMaxDelta = 1; // Apple-tolerance mode: identical, or LSB rounding at a few pixels. bool withinAppleLsbEnvelope(const PixelDiff &d) { - return d.bytes >= 0 && d.bytes <= kSelfVarianceMaxBytes && d.maxDelta <= kAppleLsbMaxDelta; + return d.bytes >= 0 && d.bytes <= kAppleLsbMaxBytes && d.maxDelta <= kAppleLsbMaxDelta; } std::string appleLsbEnvelope() { return fmt("Apple LSB envelope (delta<=%.0f, bytes<=%.0f)", double(kAppleLsbMaxDelta), - double(kSelfVarianceMaxBytes)); + double(kAppleLsbMaxBytes)); } // The renderer the frames are drawn by, as renderAvailable() reads it, and @@ -1280,6 +1280,11 @@ void sampleSelfVariance(const std::string &scope, const std::string &what, const d.bytes ? describe(d, w) + ", identical GL traces" : std::string()); return; } + // Apple-tolerance mode: the pair is held to the same fixed envelope as every + // other pixel check, and printed when it differs at all. + check(withinAppleLsbEnvelope(d), + what + " is byte-identical or within the " + appleLsbEnvelope() + " on Apple's renderer", + d.bytes ? describe(d, w) + ", identical GL traces" : std::string()); if (d.bytes == 0) return; g_noisy.push_back({scope, what, d, w}); // a diagnostic: it excuses nothing @@ -1316,7 +1321,7 @@ void checkPixels() { } else { std::printf("pixels (apple-tolerance mode: every check, in one window or across a recompile " "or a new window, byte-identical or within the %s; the pairs of Stock frames " - "sampled are a diagnostic and excuse nothing)\n", + "sampled are held to the same envelope and excuse nothing)\n", appleLsbEnvelope().c_str()); std::set scopes; for (const auto &p : g_pairs) @@ -1413,8 +1418,8 @@ bool renderAvailable(cvc::app &app) { rule = "every check, in one window or across a recompile or a new window, byte-identical or " "within the " + appleLsbEnvelope() + - ", pairs of Stock frames printed only (they excuse nothing); point picking by <= 1 " - "point per prop, cell picking exact"; + ", pairs of Stock frames held to the same envelope (they excuse nothing); point " + "picking by <= 1 point per prop, cell picking exact"; std::printf(" pixel checks: %s (OpenGL renderer \"%s\", version \"%s\") -- %s\n", mode, g_glRenderer.c_str(), g_glVersion.c_str(), rule.c_str()); } @@ -1644,8 +1649,9 @@ void comparePaths(Scene &s, const std::string &label, bool shadows) { for (DrawPath p : {DrawPath::Fast, DrawPath::Stock}) runPath(s, p); // Pairs of the Stock runs before Fast's (one after AllCellTypes) are checked - // byte-identical -- on Apple's renderer, printed as its self-variance; the - // Stock run right after Fast is judged, like the AllCellTypes and Fast frames. + // byte-identical -- on Apple's renderer, within its LSB envelope, and printed + // as its self-variance; the Stock run right after Fast is judged, like the + // AllCellTypes and Fast frames. const PathRun stock = runPath(s, DrawPath::Stock); const PathRun stock2 = runPath(s, DrawPath::Stock); const PathRun all = runPath(s, DrawPath::AllCellTypes); @@ -1810,7 +1816,8 @@ void comparePaths(Scene &s, const std::string &label, bool shadows) { } // ── 6. lifecycle ───────────────────────────────────────────────────────────── -// scope: the scene's label, whose self-variance samples judge these pixels. +// scope: the scene's label, under which this lifecycle's pairs of Stock frames are counted +// and printed. void testLifecycle(Scene &s, const std::string &scope) { std::printf("lifecycle\n"); LowMemoryPolyDataMapper::setDrawPath(DrawPath::Fast); @@ -1848,7 +1855,7 @@ void testLifecycle(Scene &s, const std::string &scope) { fmt("lookups %.0f of %.0f draws", double(rel.programLookups), double(rel.draws))); const PathRun released = runPath(s, DrawPath::Fast); // Stock to compare against: two runs, a self-variance pair (checked - // byte-identical but on Apple's renderer, where it is printed), after one + // byte-identical, or on Apple's renderer within its LSB envelope), after one // that follows Fast (judged, not sampled). const PathRun afterFast = runPath(s, DrawPath::Stock); const PathRun before = runPath(s, DrawPath::Stock);