From 8cb42ca7b4a61bf6d32f4b3a0a21ab557e84fceb Mon Sep 17 00:00:00 2001 From: wzj1228516103 <154067345+wzj1228516103@users.noreply.github.com> Date: Tue, 8 Sep 2026 18:21:56 +0800 Subject: [PATCH] fix(proxy): increase spend counter cache capacity --- litellm/proxy/proxy_server.py | 7 ++++++- tests/test_litellm/proxy/test_proxy_server.py | 18 ++++++++++++++++++ 2 files changed, 24 insertions(+), 1 deletion(-) diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 32b6b841af7a..69b3b890c226 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -240,6 +240,7 @@ def generate_feedback_box(): from litellm import Router from litellm._logging import _redact_string, verbose_proxy_logger, verbose_router_logger from litellm.caching.caching import DualCache, RedisCache +from litellm.caching.in_memory_cache import InMemoryCache from litellm.caching.redis_cluster_cache import RedisClusterCache from litellm.constants import ( _REALTIME_BODY_CACHE_SIZE, @@ -2301,7 +2302,11 @@ async def root_redirect(): user_api_key_cache: UserApiKeyCache = UserApiKeyCache( default_in_memory_ttl=UserAPIKeyCacheTTLEnum.in_memory_cache_ttl.value ) -spend_counter_cache: Final = DualCache(default_in_memory_ttl=UserAPIKeyCacheTTLEnum.in_memory_cache_ttl.value) +SPEND_COUNTER_CACHE_MAX_SIZE: Final = 10_000 +spend_counter_cache: Final = DualCache( + in_memory_cache=InMemoryCache(max_size_in_memory=SPEND_COUNTER_CACHE_MAX_SIZE), + default_in_memory_ttl=UserAPIKeyCacheTTLEnum.in_memory_cache_ttl.value, +) cli_sso_session_cache: Final = DualCache(default_in_memory_ttl=CLI_SSO_SESSION_TTL_SECONDS) model_max_budget_limiter: Final = _PROXY_VirtualKeyModelMaxBudgetLimiter(dual_cache=spend_counter_cache) litellm.logging_callback_manager.add_litellm_callback(model_max_budget_limiter) diff --git a/tests/test_litellm/proxy/test_proxy_server.py b/tests/test_litellm/proxy/test_proxy_server.py index b7bb58378d40..63ce73b3224a 100644 --- a/tests/test_litellm/proxy/test_proxy_server.py +++ b/tests/test_litellm/proxy/test_proxy_server.py @@ -7542,6 +7542,24 @@ async def test_store_model_in_db_db_failure_graceful(monkeypatch): # ===================================================================== +def test_spend_counter_cache_keeps_active_counters_warm(monkeypatch): + from litellm.caching.in_memory_cache import InMemoryCache + from litellm.proxy.proxy_server import spend_counter_cache + + cache = InMemoryCache( + max_size_in_memory=spend_counter_cache.in_memory_cache.max_size_in_memory, + default_ttl=60, + ) + monkeypatch.setattr(spend_counter_cache, "in_memory_cache", cache) + + sentinel = "spend:key:sentinel" + cache.set_cache(key=sentinel, value=1.0) + for index in range(300): + cache.set_cache(key=f"spend:end_user:scope-{index}", value=1.0) + + assert cache.get_cache(key=sentinel) == 1.0 + + @pytest.mark.asyncio async def test_get_current_spend_reads_redis_first(): """get_current_spend should prefer Redis over in-memory."""