Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 2 additions & 1 deletion benchmarks/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -17,6 +17,7 @@ cost. Runs in CI (informational, non-gating) and locally via `just bench`.
| G5 | Cross-scope resolve, REQUEST -> APP dep | `find_container` traversal |
| G6 | `build_child_container(REQUEST)` | per-request setup |
| G7 | Full lifecycle batch: K=100 x (build REQUEST -> sync-init cached resolve -> `await close_async()`) | real per-request cost incl. async teardown |
| G7b | One request cycle: build REQUEST -> first-resolve one cached REQUEST provider -> `close_sync()` | per-request cost of a cached item, no event loop |
| G7c | Control: K=100 empty awaits in one loop entry | residual event-loop floor inside G7 |
| G8 | Cold first-resolve: build root container + compile + resolve, depth 6 | construction + first-compile cost |
| G8b | G8 with every provider `cache=True` | the cached template's cold-miss `build`/`create`, read against G8 |
Expand All @@ -27,7 +28,7 @@ cost. Runs in CI (informational, non-gating) and locally via `just bench`.
| G13 | Per-request cycle finalizing 10 cached resources (`close_sync`) | LIFO teardown at scale |
| G13b | Batch of K=100 request cycles, 10 finalizer-less cached REQUEST providers, `await close_async()` | the async close loop when there is nothing to finalize |
| G14 | Concurrent cached-hit throughput, N threads (lock-free read) | free-threaded read scaling |
| G15 | Concurrent first-resolve, N threads (double-checked creation lock) | free-threaded creation-lock contention |
| G15 | Concurrent first-resolve, N threads (per-item double-checked creation lock) | free-threaded creation-lock contention |
| G16 | Warm by-type `resolve(SomeType)`, small graph | `find_provider` lookup on the integration/`@inject` path |
| G17 | Warm by-type `resolve(SomeType)`, 200-provider registry | lookup cost at realistic registry scale |
| G18 | Warm resolve through an `Alias` to a cached source | the alias hop, read against G2 |
Expand Down
14 changes: 8 additions & 6 deletions benchmarks/test_guard_concurrency.py
Original file line number Diff line number Diff line change
Expand Up @@ -10,8 +10,8 @@
across N threads. The cached-hit path is lock-free, so on a free-threaded build (PEP 703) the
batch time should *drop* as N rises (throughput scales); under the GIL it stays flat.
- G15 concurrent first-resolve: N threads each race to resolve the *same* K cold singletons, so
they contend on the double-checked creation lock (`CacheItem.get_or_create`). Singleton
creation is serialized by design, so this is expected *not* to scale even free-threaded — the
they contend on each item's double-checked creation lock (`CacheItem.get_or_create`). Creating
one singleton is serialized by design, so this is expected *not* to scale even free-threaded: the
measured cost is the contention itself (the known trade-off vs lock-free-slot rivals).

Read the batch-time-vs-thread-count trend, not the absolutes. The GIL vs free-threaded comparison
Expand Down Expand Up @@ -124,10 +124,12 @@ def _worker() -> None:

@pytest.mark.parametrize("n_threads", _THREAD_COUNTS)
def test_g15b_concurrent_first_resolve_sibling_children(benchmark, n_threads):
# Each thread builds its OWN REQUEST child and first-resolves K cached providers in it, so
# every creation is a cold miss in a container no other thread touches. With a lock per
# container these creations never contend; with one lock per tree they serialize. This is the
# scenario G15 does not cover -- G15 races on one root, whose lock is shared either way.
"""Each thread builds its own REQUEST child and first-resolves K cached providers in it.

Every creation is a cold miss in a container no other thread touches. Each cache item has its
own lock, so these creations never contend; under a lock shared by the tree they would
serialize. G15 does not cover this: it races on one root's items, whose locks are shared.
"""
check = Container(scope=Scope.APP, groups=[_REQUEST_GROUP])
check.open()
probe = check.build_child_container(scope=Scope.REQUEST)
Expand Down
31 changes: 30 additions & 1 deletion benchmarks/test_guard_lifecycle.py
Original file line number Diff line number Diff line change
Expand Up @@ -7,13 +7,14 @@
K times per loop entry to isolate DI work from the ~27us event-loop entry cost.
Divide G7's number by 100 for per-request cost. G7c is the control (K=100 empty
awaits on the same shape) so the residual loop overhead is visible (~15% at K=100).
G7b is one request cycle with a synchronous close, so no event loop is in the number.
See benchmarks/README.md.
"""

import asyncio
import dataclasses

from benchmarks._pinned import ITER_UNDER_1US, ROUNDS
from benchmarks._pinned import ITER_UNDER_1US, ITER_UNDER_2US, ROUNDS
from modern_di import Container, Group, Scope, providers


Expand Down Expand Up @@ -117,6 +118,34 @@ def _run_batch() -> None:
loop.close()


# --- G7b: one request cycle, cached REQUEST provider without a finalizer, sync close ---
@dataclasses.dataclass(slots=True)
class RequestService:
pass


class RequestCycleGroup(Group):
svc = providers.Factory(creator=RequestService, scope=Scope.REQUEST, cache=True)


def test_g7b_request_cycle_sync(benchmark):
"""Build a REQUEST child, first-resolve one cached REQUEST provider, close it synchronously.

Every cycle creates a fresh cache item and its lock, so this is where a per-item cost lands.
"""
app = Container(scope=Scope.APP, groups=[RequestCycleGroup])
app.open()

def _one_request() -> RequestService:
req = app.build_child_container(scope=Scope.REQUEST)
svc = req.resolve_provider(RequestCycleGroup.svc)
req.close_sync()
return svc

result = benchmark.pedantic(_one_request, rounds=ROUNDS, iterations=ITER_UNDER_2US)
assert isinstance(result, RequestService)


# --- G13: teardown at scale -- 10 cached REQUEST resources, sync finalizers ---
# G7 finalizes one resource; a real request closes several. G13 measures the per-request cycle
# with 10 cached REQUEST providers (each a sync finalizer) so the LIFO close loop is exercised.
Expand Down
5 changes: 3 additions & 2 deletions docs/introduction/design-decisions.md
Original file line number Diff line number Diff line change
Expand Up @@ -10,11 +10,12 @@ Async resolution will not be added.

## 2. Cached factories are thread-safe

Cached `Factory` providers use one reentrant lock (`threading.RLock`) per container tree, created by the root and shared by every child, so concurrent resolves in multiple threads still produce exactly one instance per cache. The lock is taken only on a cache miss; a resolve that finds the instance already cached never touches it.
Each cache item, the cached instance of one `Factory` in one container, has its own reentrant lock (`threading.RLock`), so concurrent resolves in multiple threads still produce exactly one instance per cache. The lock is taken only on a cache miss; a resolve that finds the instance already cached never touches it. On a miss, the dependencies are resolved and the creator is called while the lock is held, so threads that miss together build the dependencies once, and the others wait for that instance.

### The thread-safety boundary

- Cached / singleton creation is locked. The tree-wide reentrant lock guards the create-and-store step, so two threads racing to resolve the same cached provider get the same single instance.
- Cached / singleton creation is locked per cached provider. Two threads racing to resolve the same cached provider get the same single instance, and transient dependencies of that provider are built once for it. Creations of different cached providers do not wait for each other, so a creator can hand a resolve of another cached type to a worker thread and wait for the result. A creator that waits on another thread resolving the provider it is creating, directly or through its dependencies, still deadlocks: that is a cycle.
- Call `validate()` at startup if the graph might contain a cycle. Without it, a single thread resolving a cyclic graph gets `CircularDependencyError` from the runtime guard. Two threads that cold-resolve different providers of the same cycle at the same time can each hold one cache item's lock while waiting for the other's, and block forever.
- Provider registration is safe. `ProvidersRegistry` mutations (`register`, `add_providers`) are guarded by the registry's own lock, and iteration snapshots the provider dict (`iter(list(...))`), so registering providers concurrently, or while another thread iterates, will not corrupt the registry or raise "dict changed size during iteration".
- Registration belongs to the setup phase. The registry is lock-guarded
against corruption, but the supported model is to register every provider
Expand Down
6 changes: 6 additions & 0 deletions docs/introduction/performance.md
Original file line number Diff line number Diff line change
Expand Up @@ -322,6 +322,12 @@ resolver, so a hop through an alias runs no frame of its own (−33% on an alias
now matches a plain cached resolve). An error that crosses an alias still shows the alias in its
chain: the parent puts the hop back when it adds its own step.

4.0 also replaced the tree lock from #541 with one lock per cache item (#569), so creating one
cached factory no longer waits for another. A child build still allocates no lock, because the
lock comes with the cache item. The cost moved to the first resolve of a cached factory in each
container. A request that resolves one request-scoped cached factory pays about 120 ns to
allocate its lock: +7.7% on G7b, one request cycle with a sync close, and +4.6% on G7.

## Reproduce it yourself

```bash
Expand Down
21 changes: 20 additions & 1 deletion docs/migration/to-4.x.md
Original file line number Diff line number Diff line change
Expand Up @@ -11,10 +11,29 @@ Every `Container` argument after `scope` is keyword-only: `Container(Scope.APP,

### `use_lock` is removed

`Container(use_lock=...)` raises `TypeError`; drop the argument. Every container tree is now
`Container(use_lock=...)` raises `TypeError`; drop the argument. Cached factories are always
locked, and the lock is taken only on a cache miss (see
[Design decisions](../introduction/design-decisions.md#2-cached-factories-are-thread-safe)).

### Each cached factory has its own lock

In 3.x, with the default `use_lock=True`, every cached creator ran under a shared lock: the
container's lock, or from 3.6 the lock of the whole container tree. Two cached factories that
shared the lock were never created at the same time. In 4.0 each cached factory has its own lock
in each container, so creators of different cached factories can run at the same time on
different threads. If two creators share state that is not thread-safe, guard that state with
your own lock.

On a cache miss the lock is now held while the dependencies are resolved as well. Threads that
miss together build the dependencies once, where 3.x built them once per thread and discarded all
but one. A cached creator can also wait on another thread that resolves a different cached type,
which deadlocked in 3.x.

One case that raised in 3.x can now block. If the graph has a cycle and `validate()` was never
called, two threads that cold-resolve different providers of that cycle at the same time can each
hold one lock and wait for the other forever. In 3.x both got `CircularDependencyError`. A single
thread still gets that error. Call `validate()` at startup to catch cycles before serving.

### Resolving on a closed container raises

In 3.x, resolving from a closed container, or through a child whose resolve reached a closed
Expand Down
6 changes: 4 additions & 2 deletions docs/providers/advanced-api.md
Original file line number Diff line number Diff line change
Expand Up @@ -38,5 +38,7 @@ inspect or iterate all providers declared on a group hierarchy.
`Container` subclass does not redirect navigation. `resolve` and `resolve_provider` are entry
points, not hooks either: a compiled resolver calls its dependencies' resolvers directly, so an
override of either sees only the top-level call.
- The root creates one `threading.RLock`, and every child shares it. A cached `Factory` holds
that lock while it builds on a cold cache miss, so one instance is created per cache key.
- Each cached `Factory` gets its own `threading.RLock` in each container that caches it, created
with the cache item on the first resolve there. Building a child allocates no lock. On a cold
cache miss the factory holds its lock while it resolves its dependencies and calls the creator,
so one instance is created per cache key. A warm resolve does not take the lock.
3 changes: 2 additions & 1 deletion docs/providers/errors-and-exceptions.md
Original file line number Diff line number Diff line change
Expand Up @@ -121,7 +121,8 @@ to render the chain programmatically.
[Troubleshooting: ArgumentResolutionError](../troubleshooting/argument-resolution-error.md).
- `CircularDependencyError` is raised when the provider graph contains a cycle (A → B → A); the
message shows the cycle path. Raised eagerly by `validate()`, and also by a bare `resolve()` on an
unvalidated cyclic graph via a runtime guard; see
unvalidated cyclic graph via a runtime guard. The guard covers one thread resolving: concurrent
first resolves of an unvalidated cyclic graph can block instead, so call `validate()` at startup. See
[Troubleshooting: Circular dependency](../troubleshooting/circular-dependency.md#the-runtime-cycle-guard-without-validate).
- `CreatorCallError` is raised when a creator's dependencies all resolved but argument binding
failed while calling it (the assembled arguments don't match the signature, typically a `kwargs` /
Expand Down
2 changes: 1 addition & 1 deletion docs/providers/factories.md
Original file line number Diff line number Diff line change
Expand Up @@ -51,7 +51,7 @@ This is modern-di's Singleton. There is no separate `Singleton` provider class:
[Where is Singleton?](../introduction/comparison.md#where-is-singleton-cross-framework-vocabulary)
for the full cross-framework mapping.

The caching mechanism is thread-safe: when multiple threads resolve the same cached factory simultaneously, only one instance is created.
The caching mechanism is thread-safe: when multiple threads resolve the same cached factory simultaneously, only one instance is created, and its dependencies are resolved once for it. Other threads wait for that instance; resolves of other cached factories do not.

```python
import random
Expand Down
5 changes: 5 additions & 0 deletions docs/troubleshooting/circular-dependency.md
Original file line number Diff line number Diff line change
Expand Up @@ -31,6 +31,11 @@ re-walks the static graph from the failing provider, and, since a cycle is reach
`RecursionError`. A creator that merely recurses on its own, with no actual cycle in the provider
graph, still raises the original `RecursionError` unchanged. This guard runs on every resolve, whether or not `validate()` was ever called.

The guard covers resolution on one thread. Each cached factory locks its cache item while it is
created, so two threads that cold-resolve different providers of the same cycle at the same time
can each hold one of those locks and wait for the other forever, and neither reaches the guard.
If the graph might have a cycle, call `validate()` at startup, before any thread resolves.

### Cycle detection with `validate()`

Calling `validate()` up front finds the *same* cycle earlier, and finds *every* issue in the graph
Expand Down
4 changes: 0 additions & 4 deletions modern_di/container.py
Original file line number Diff line number Diff line change
@@ -1,6 +1,5 @@
import copy
import enum
import threading
import typing

from modern_di import dependency_graph, exceptions, types
Expand Down Expand Up @@ -40,7 +39,6 @@ class Container:
"_cache_registry",
"_closed",
"_context_registry",
"_lock",
"_providers_registry",
"_scope_map",
"parent_container",
Expand Down Expand Up @@ -87,10 +85,8 @@ def __init__(
self._context_registry = ContextRegistry(copy.copy(context) if context is not None else {})
self._providers_registry: ProvidersRegistry
if parent_container:
self._lock = parent_container._lock # noqa: SLF001
self._providers_registry = parent_container._providers_registry # noqa: SLF001
else:
self._lock = threading.RLock()
self._providers_registry = ProvidersRegistry()
self._providers_registry.register(Container, container_provider)
if groups:
Expand Down
20 changes: 8 additions & 12 deletions modern_di/registries/cache_registry.py
Original file line number Diff line number Diff line change
@@ -1,15 +1,12 @@
import dataclasses
import inspect
import threading
import typing

from modern_di import exceptions, types
from modern_di.providers import CacheSettings, Factory


if typing.TYPE_CHECKING:
import threading


_R = typing.TypeVar("_R")
_V = typing.TypeVar("_V")

Expand All @@ -19,6 +16,7 @@ class CacheItem:
settings: CacheSettings[typing.Any]
cache: typing.Any = types.UNSET
finalized: bool = False
lock: threading.RLock = dataclasses.field(default_factory=threading.RLock, repr=False, compare=False)

def clear(self) -> None:
if self.settings.clear_cache:
Expand All @@ -27,22 +25,20 @@ def clear(self) -> None:

def get_or_create(
self,
lock: "threading.RLock",
resolve: typing.Callable[[], _R],
create: typing.Callable[[_R], _V],
) -> tuple[_V, bool]:
"""Return the memoized singleton, or resolve-and-create it once under `lock`.
"""Return the memoized singleton, or resolve-and-create it once under this item's lock.

`resolve()` runs unlocked — recursive resolution must not hold the lock; creation and
the store are double-checked under it. `created` is True only for the caller that built.
A hit never takes the lock. A miss resolves and creates under it, so concurrent misses
build the value and its dependencies once. `created` is True only for the caller that built.
"""
if self.cache is not types.UNSET:
return self.cache, False
resolved = resolve()
with lock:
with self.lock:
if self.cache is not types.UNSET:
return self.cache, False
value = create(resolved)
value = create(resolve())
self.cache = value
return value, True

Expand Down Expand Up @@ -92,7 +88,7 @@ def cached_count(self) -> int:
return sum(1 for item in self._items.values() if item.cache is not types.UNSET)

def fetch_cache_item(self, provider: Factory[typing.Any]) -> CacheItem:
"""Return the cache slot for a cached ``provider``, creating it on first use."""
"""Return the cache item for a cached ``provider``, creating it on first use."""
# Get before setdefault: a bare setdefault builds a throwaway CacheItem on every hit.
item = self._items.get(provider.provider_id)
if item is not None:
Expand Down
4 changes: 2 additions & 2 deletions modern_di/resolver_compiler.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,7 +9,7 @@
``ProvidersRegistry.drop_resolvers``). Why a template and not shared helpers: see
docs/adr/0001-resolver-hot-path-generated-source.md.

The template reaches into `Container._lock`/`_scope_map` and `CacheRegistry._items` to stay
The template reaches into `Container._scope_map` and `CacheRegistry._items` to stay
within that frame budget. No linter sees the template, so those reaches are outside every suppression here.
"""

Expand Down Expand Up @@ -114,7 +114,7 @@ def compile_resolver(provider: "AbstractProvider[typing.Any]", registry: "Provid
cached = cache_item.cache
if cached is not UNSET:
return cached
value, created = cache_item.get_or_create(target._lock, resolve=partial(build, target), create=create)
value, created = cache_item.get_or_create(partial(build, target), create)
if created:
cache_registry.mark_created(cache_item)
return value
Expand Down
10 changes: 10 additions & 0 deletions tests/helpers.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,10 @@
import typing

from modern_di import Container
from modern_di.providers import Factory
from modern_di.registries.cache_registry import CacheItem


def cache_item(container: Container, provider: Factory[typing.Any]) -> CacheItem:
"""Return `container`'s cache item for the cached `provider`, creating it if needed."""
return container._cache_registry.fetch_cache_item(provider)
Loading
Loading