From ef07978f4a4e4f5085a99d2f1a8244d9c88255b5 Mon Sep 17 00:00:00 2001 From: "avi@robusta.dev" Date: Wed, 13 May 2026 17:00:20 +0300 Subject: [PATCH 1/5] reducing litellm noise Signed-off-by: avi@robusta.dev --- holmes/core/llm.py | 32 ++++++++++++++++++++------------ 1 file changed, 20 insertions(+), 12 deletions(-) diff --git a/holmes/core/llm.py b/holmes/core/llm.py index 6e33484e18..1b36d6cac9 100644 --- a/holmes/core/llm.py +++ b/holmes/core/llm.py @@ -55,6 +55,8 @@ OVERRIDE_MAX_OUTPUT_TOKEN = environ_get_safe_int("OVERRIDE_MAX_OUTPUT_TOKEN") OVERRIDE_MAX_CONTENT_SIZE = environ_get_safe_int("OVERRIDE_MAX_CONTENT_SIZE") +_warned_missing_model_lookups: set[tuple[str, str]] = set() + def get_context_window_compaction_threshold_pct() -> int: """Get the compaction threshold percentage at runtime to support test overrides.""" @@ -407,12 +409,15 @@ def get_context_window_size(self) -> int: except Exception: continue - # Log which lookups we tried - logging.warning( - f"Couldn't find model {self.model} in litellm's model list (tried: {', '.join(self._get_model_name_variants_for_lookup())}), " - f"using default {FALLBACK_CONTEXT_WINDOW_SIZE} tokens for max_input_tokens. " - f"To override, set OVERRIDE_MAX_CONTENT_SIZE environment variable to the correct value for your model." - ) + # Log which lookups we tried (once per model to avoid log spam) + warn_key = (self.model, "max_input_tokens") + if warn_key not in _warned_missing_model_lookups: + _warned_missing_model_lookups.add(warn_key) + logging.warning( + f"Couldn't find model {self.model} in litellm's model list (tried: {', '.join(self._get_model_name_variants_for_lookup())}), " + f"using default {FALLBACK_CONTEXT_WINDOW_SIZE} tokens for max_input_tokens. " + f"To override, set OVERRIDE_MAX_CONTENT_SIZE environment variable to the correct value for your model." + ) return FALLBACK_CONTEXT_WINDOW_SIZE def _is_anthropic_model(self) -> bool: @@ -642,12 +647,15 @@ def get_maximum_output_token(self) -> int: except Exception: continue - # Log which lookups we tried - logging.warning( - f"Couldn't find model {self.model} in litellm's model list (tried: {', '.join(self._get_model_name_variants_for_lookup())}), " - f"using {max_output_tokens} tokens for max_output_tokens. " - f"To override, set OVERRIDE_MAX_OUTPUT_TOKEN environment variable to the correct value for your model." - ) + # Log which lookups we tried (once per model to avoid log spam) + warn_key = (self.model, "max_output_tokens") + if warn_key not in _warned_missing_model_lookups: + _warned_missing_model_lookups.add(warn_key) + logging.warning( + f"Couldn't find model {self.model} in litellm's model list (tried: {', '.join(self._get_model_name_variants_for_lookup())}), " + f"using {max_output_tokens} tokens for max_output_tokens. " + f"To override, set OVERRIDE_MAX_OUTPUT_TOKEN environment variable to the correct value for your model." + ) return max_output_tokens From 592b4b6bccdbb99b969bd2c1dd9709c92fcfed3d Mon Sep 17 00:00:00 2001 From: "avi@robusta.dev" Date: Wed, 13 May 2026 17:24:21 +0300 Subject: [PATCH 2/5] debug: surface raw Loki response on JSON parse failure MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit When Loki returns a malformed JSON body, the toolset previously surfaced only the parse error message (e.g. "Expecting ',' delimiter: line 1 column 2811"), giving no insight into what was actually returned. This is particularly painful when a proxy (e.g. nginx with body_filter_by_lua scrubbing) is corrupting the response in transit — the error position is on the corrupted body, not the original. Catch the ValueError from response.json() and re-raise with the raw response text, length, and content-type included so the LLM (and operator) can see the actual bytes that came back. Signed-off-by: avi@robusta.dev --- holmes/plugins/toolsets/grafana/loki_api.py | 11 ++++++++++- 1 file changed, 10 insertions(+), 1 deletion(-) diff --git a/holmes/plugins/toolsets/grafana/loki_api.py b/holmes/plugins/toolsets/grafana/loki_api.py index 98b12f93e8..0861e346d8 100644 --- a/holmes/plugins/toolsets/grafana/loki_api.py +++ b/holmes/plugins/toolsets/grafana/loki_api.py @@ -65,7 +65,16 @@ def _make_request(): try: response = _make_request() - result = response.json() + try: + result = response.json() + except ValueError as parse_err: + raw = response.text + raise Exception( + f"Failed to parse Loki response as JSON: {parse_err}\n" + f"--- raw response ({len(raw)} chars, content-type={response.headers.get('Content-Type', 'unknown')}) ---\n" + f"{raw}\n" + f"--- end raw response ---" + ) if "data" in result and "result" in result["data"]: return parse_loki_response(result["data"]["result"]) return [] From dde8dbddb9122f7c2d9a59f363d210ee241091bc Mon Sep 17 00:00:00 2001 From: "avi@robusta.dev" Date: Wed, 13 May 2026 18:02:52 +0300 Subject: [PATCH 3/5] debug: broaden raw-response dump to any post-request exception MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The previous commit only included the raw response when JSON parsing failed. Widen the net to any exception raised after the HTTP response is in hand (e.g. unexpected response shape in parse_loki_response, KeyError on missing fields), so the raw bytes are always available for debugging — not just for JSONDecodeError. The split into two try/except blocks keeps the RequestException path (no response object yet) separate from the response-handling path (response.text is available). Signed-off-by: avi@robusta.dev --- holmes/plugins/toolsets/grafana/loki_api.py | 26 ++++++++++----------- 1 file changed, 13 insertions(+), 13 deletions(-) diff --git a/holmes/plugins/toolsets/grafana/loki_api.py b/holmes/plugins/toolsets/grafana/loki_api.py index 0861e346d8..ac1bc3ca65 100644 --- a/holmes/plugins/toolsets/grafana/loki_api.py +++ b/holmes/plugins/toolsets/grafana/loki_api.py @@ -65,19 +65,19 @@ def _make_request(): try: response = _make_request() - try: - result = response.json() - except ValueError as parse_err: - raw = response.text - raise Exception( - f"Failed to parse Loki response as JSON: {parse_err}\n" - f"--- raw response ({len(raw)} chars, content-type={response.headers.get('Content-Type', 'unknown')}) ---\n" - f"{raw}\n" - f"--- end raw response ---" - ) + except requests.exceptions.RequestException as e: + raise Exception(f"Failed to query Loki logs: {str(e)}") + + try: + result = response.json() if "data" in result and "result" in result["data"]: return parse_loki_response(result["data"]["result"]) return [] - - except requests.exceptions.RequestException as e: - raise Exception(f"Failed to query Loki logs: {str(e)}") + except Exception as e: + raw = response.text + raise Exception( + f"Failed to process Loki response: {e}\n" + f"--- raw response ({len(raw)} chars, content-type={response.headers.get('Content-Type', 'unknown')}) ---\n" + f"{raw}\n" + f"--- end raw response ---" + ) From 27a2c7840101c26989ae6fbf6cb89671f3551f05 Mon Sep 17 00:00:00 2001 From: "avi@robusta.dev" Date: Thu, 14 May 2026 11:19:34 +0300 Subject: [PATCH 4/5] Revert "reducing litellm noise" This reverts commit ef07978f4a4e4f5085a99d2f1a8244d9c88255b5. Signed-off-by: avi@robusta.dev --- holmes/core/llm.py | 32 ++++++++++++-------------------- 1 file changed, 12 insertions(+), 20 deletions(-) diff --git a/holmes/core/llm.py b/holmes/core/llm.py index 1b36d6cac9..6e33484e18 100644 --- a/holmes/core/llm.py +++ b/holmes/core/llm.py @@ -55,8 +55,6 @@ OVERRIDE_MAX_OUTPUT_TOKEN = environ_get_safe_int("OVERRIDE_MAX_OUTPUT_TOKEN") OVERRIDE_MAX_CONTENT_SIZE = environ_get_safe_int("OVERRIDE_MAX_CONTENT_SIZE") -_warned_missing_model_lookups: set[tuple[str, str]] = set() - def get_context_window_compaction_threshold_pct() -> int: """Get the compaction threshold percentage at runtime to support test overrides.""" @@ -409,15 +407,12 @@ def get_context_window_size(self) -> int: except Exception: continue - # Log which lookups we tried (once per model to avoid log spam) - warn_key = (self.model, "max_input_tokens") - if warn_key not in _warned_missing_model_lookups: - _warned_missing_model_lookups.add(warn_key) - logging.warning( - f"Couldn't find model {self.model} in litellm's model list (tried: {', '.join(self._get_model_name_variants_for_lookup())}), " - f"using default {FALLBACK_CONTEXT_WINDOW_SIZE} tokens for max_input_tokens. " - f"To override, set OVERRIDE_MAX_CONTENT_SIZE environment variable to the correct value for your model." - ) + # Log which lookups we tried + logging.warning( + f"Couldn't find model {self.model} in litellm's model list (tried: {', '.join(self._get_model_name_variants_for_lookup())}), " + f"using default {FALLBACK_CONTEXT_WINDOW_SIZE} tokens for max_input_tokens. " + f"To override, set OVERRIDE_MAX_CONTENT_SIZE environment variable to the correct value for your model." + ) return FALLBACK_CONTEXT_WINDOW_SIZE def _is_anthropic_model(self) -> bool: @@ -647,15 +642,12 @@ def get_maximum_output_token(self) -> int: except Exception: continue - # Log which lookups we tried (once per model to avoid log spam) - warn_key = (self.model, "max_output_tokens") - if warn_key not in _warned_missing_model_lookups: - _warned_missing_model_lookups.add(warn_key) - logging.warning( - f"Couldn't find model {self.model} in litellm's model list (tried: {', '.join(self._get_model_name_variants_for_lookup())}), " - f"using {max_output_tokens} tokens for max_output_tokens. " - f"To override, set OVERRIDE_MAX_OUTPUT_TOKEN environment variable to the correct value for your model." - ) + # Log which lookups we tried + logging.warning( + f"Couldn't find model {self.model} in litellm's model list (tried: {', '.join(self._get_model_name_variants_for_lookup())}), " + f"using {max_output_tokens} tokens for max_output_tokens. " + f"To override, set OVERRIDE_MAX_OUTPUT_TOKEN environment variable to the correct value for your model." + ) return max_output_tokens From cb1bd9b101285a95e0f6473afd750f6e419476e4 Mon Sep 17 00:00:00 2001 From: "avi@robusta.dev" Date: Sun, 17 May 2026 13:59:09 +0300 Subject: [PATCH 5/5] test: cover raw-response dump on malformed Loki JSON Add a unit test that mocks the Loki query_range endpoint with a malformed JSON body and verifies the exception raised by execute_loki_query includes both the JSON parser error and the raw response text (with content-type and byte count), matching the debug behavior added earlier on this branch. Also scope the existing module-level "Grafana not running" skip to the integration tests via a @needs_grafana marker, so the new unit test is not skipped on machines without a local Grafana. Co-Authored-By: Claude Opus 4.7 (1M context) Signed-off-by: avi@robusta.dev --- .../plugins/toolsets/grafana/test_grafana.py | 52 +++++++++++++++++-- 1 file changed, 48 insertions(+), 4 deletions(-) diff --git a/tests/plugins/toolsets/grafana/test_grafana.py b/tests/plugins/toolsets/grafana/test_grafana.py index 385f37e7f0..f3500460d2 100644 --- a/tests/plugins/toolsets/grafana/test_grafana.py +++ b/tests/plugins/toolsets/grafana/test_grafana.py @@ -1,18 +1,24 @@ import pytest +import responses from holmes.core.tools import ToolsetStatusEnum from holmes.plugins.toolsets.grafana.loki.toolset_grafana_loki import ( GrafanaLokiToolset, ) +from holmes.plugins.toolsets.grafana.loki_api import execute_loki_query from holmes.plugins.toolsets.grafana.toolset_grafana import GrafanaToolset from tests.plugins.toolsets.grafana.conftest import check_service_running -# Skip all tests in this module if Grafana and loki are not running. use loki/docker-compose.yaml -skip_reason = check_service_running("Grafana", 3000) -if skip_reason: - pytestmark = pytest.mark.skip(reason=skip_reason) +# Skip integration tests that require Grafana/Loki running on localhost:3000. +# Use loki/docker-compose.yaml to bring them up. +_grafana_skip_reason = check_service_running("Grafana", 3000) +needs_grafana = pytest.mark.skipif( + _grafana_skip_reason is not None, + reason=_grafana_skip_reason or "", +) +@needs_grafana def test_grafana_toolset_direct_health_check(): toolset = GrafanaToolset() toolset.config = {"api_url": "http://localhost:3000/"} @@ -22,6 +28,7 @@ def test_grafana_toolset_direct_health_check(): assert toolset.status == ToolsetStatusEnum.ENABLED +@needs_grafana def test_grafana_toolset_error_health_check(): toolset = GrafanaToolset() toolset.config = {"api_url": "http://localhost:2000/"} @@ -34,6 +41,7 @@ def test_grafana_toolset_error_health_check(): assert toolset.status == ToolsetStatusEnum.FAILED +@needs_grafana def test_loki_toolset_direct_health_check(): toolset = GrafanaLokiToolset() toolset.config = {"api_url": "http://localhost:3100/"} @@ -43,6 +51,7 @@ def test_loki_toolset_direct_health_check(): assert toolset.status == ToolsetStatusEnum.ENABLED +@needs_grafana def test_loki_datasource_toolset_health_check(): toolset = GrafanaLokiToolset() toolset.config = { @@ -55,6 +64,7 @@ def test_loki_datasource_toolset_health_check(): assert toolset.status == ToolsetStatusEnum.ENABLED +@needs_grafana def test_loki_datasource_toolset_error_health_check(): toolset = GrafanaLokiToolset() toolset.config = { @@ -68,3 +78,37 @@ def test_loki_datasource_toolset_error_health_check(): in toolset.error ) assert toolset.status == ToolsetStatusEnum.FAILED + + +def test_execute_loki_query_includes_raw_body_on_malformed_json(): + """When Loki returns malformed JSON, the raised exception should include + both the JSON parser error and the raw response body so an operator (and + the LLM) can see what actually came back over the wire.""" + base_url = "http://loki.example.com" + raw_body = '{"data": {"result": [oops not json' + + with responses.RequestsMock() as rsps: + rsps.add( + responses.GET, + f"{base_url}/loki/api/v1/query_range", + body=raw_body, + status=200, + content_type="application/json", + ) + + with pytest.raises(Exception) as exc_info: + execute_loki_query( + base_url=base_url, + api_key=None, + headers=None, + query='{job="foo"}', + start=0, + end=1, + limit=10, + ) + + message = str(exc_info.value) + assert "Failed to process Loki response" in message + assert raw_body in message + assert "content-type=application/json" in message + assert f"{len(raw_body)} chars" in message