From 38b98b84bcce6b0f8d943b1822fd678f494750eb Mon Sep 17 00:00:00 2001 From: Yongan Zhang <374456248@qq.com> Date: Sun, 17 May 2026 17:12:00 +0800 Subject: [PATCH] hash: silence 19 bandit B324 weak-hash warnings (cache keys + protocol fields) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `bandit -r .` reports 19 HIGH-severity B324 hits for `hashlib.md5(...)` and `hashlib.sha1(...)` uses. Every one of them is functionally safe — they fall into two distinct categories that the current code does not document: 1. **Non-security content fingerprints / cache keys (11 sites)** — `agent/codex_responses_adapter.py`, `agent/context_compressor.py`, `gateway/platforms/msgraph_webhook.py` (idempotency dedup), `gateway/platforms/weixin.py:1395` (sender-content cache key), `gateway/platforms/yuanbao_media.py:110` (`md5_hex` content helper), `tools/skills_hub.py` (×5 cache keys), `tools/skills_sync.py` (`_dir_hash` directory-change detection). For these, Python 3.9+ accepts `hashlib.md5(data, usedforsecurity=False)` which both documents intent and lets `hashlib` skip FIPS-disallowed algorithm checks. No behavioral change. 2. **Third-party protocol-required digests (8 sites)** — `gateway/platforms/qqbot/chunked_upload.py` (×4, QQ Bot rich-media upload protocol), `gateway/platforms/wecom.py:1161` (WeCom media `md5` field), `gateway/platforms/wecom_crypto.py:63` (WeChat enterprise message-encrypt signature uses SHA-1 per spec), `gateway/platforms/weixin.py:1907` (WeChat media `rawfilemd5` protocol field), `gateway/platforms/yuanbao_media.py:311` (Tencent Cloud COS V4 signature requires SHA-1 of HttpString). Switching algorithm here would break interoperability with the upstream service. Annotated each with `# nosec B324 -- ` so future contributors don't try to "fix" them. After this change `bandit -r . -x tests,skills/red-teaming,scripts,website,docs --severity-level high -t B324` reports zero hits (was 19). --- agent/codex_responses_adapter.py | 2 +- agent/context_compressor.py | 2 +- gateway/platforms/msgraph_webhook.py | 2 +- gateway/platforms/qqbot/chunked_upload.py | 8 ++++---- gateway/platforms/wecom.py | 2 +- gateway/platforms/wecom_crypto.py | 2 +- gateway/platforms/weixin.py | 4 ++-- gateway/platforms/yuanbao_media.py | 4 ++-- tools/skills_hub.py | 10 +++++----- tools/skills_sync.py | 2 +- 10 files changed, 19 insertions(+), 19 deletions(-) diff --git a/agent/codex_responses_adapter.py b/agent/codex_responses_adapter.py index 6fe9dc5bc6492..ff9835bd96545 100644 --- a/agent/codex_responses_adapter.py +++ b/agent/codex_responses_adapter.py @@ -194,7 +194,7 @@ def _derive_responses_function_call_id( return f"fc_{sanitized[:48]}" seed = source or str(response_item_id or "") or uuid.uuid4().hex - digest = hashlib.sha1(seed.encode("utf-8")).hexdigest()[:24] + digest = hashlib.sha1(seed.encode("utf-8"), usedforsecurity=False).hexdigest()[:24] return f"fc_{digest}" diff --git a/agent/context_compressor.py b/agent/context_compressor.py index 8eadcf26ef8a3..2bca951dde250 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -720,7 +720,7 @@ def _prune_old_tool_results( continue if len(content) < 200: continue - h = hashlib.md5(content.encode("utf-8", errors="replace")).hexdigest()[:12] + h = hashlib.md5(content.encode("utf-8", errors="replace"), usedforsecurity=False).hexdigest()[:12] if h in content_hashes: # This is an older duplicate — replace with back-reference result[i] = {**msg, "content": "[Duplicate tool output — same content as a more recent call]"} diff --git a/gateway/platforms/msgraph_webhook.py b/gateway/platforms/msgraph_webhook.py index 46430a25bc744..18d4024c5a83c 100644 --- a/gateway/platforms/msgraph_webhook.py +++ b/gateway/platforms/msgraph_webhook.py @@ -331,7 +331,7 @@ def _build_message_event( notification: Dict[str, Any], receipt_key: Optional[str], ) -> MessageEvent: - message_id = receipt_key or f"sha1:{sha1(json.dumps(notification, sort_keys=True).encode('utf-8')).hexdigest()}" + message_id = receipt_key or f"sha1:{sha1(json.dumps(notification, sort_keys=True).encode('utf-8'), usedforsecurity=False).hexdigest()}" source = self.build_source( chat_id=f"msgraph:{notification.get('subscriptionId', 'unknown')}", chat_name="msgraph/webhook", diff --git a/gateway/platforms/qqbot/chunked_upload.py b/gateway/platforms/qqbot/chunked_upload.py index 416dfc52a9808..41978bc0a5967 100644 --- a/gateway/platforms/qqbot/chunked_upload.py +++ b/gateway/platforms/qqbot/chunked_upload.py @@ -366,7 +366,7 @@ async def _upload_one_part( data = await asyncio.get_running_loop().run_in_executor( None, _read_file_chunk, file_path, offset, length ) - md5_hex = hashlib.md5(data).hexdigest() + md5_hex = hashlib.md5(data).hexdigest() # nosec B324 -- QQ Bot rich-media upload protocol requires MD5 digest of chunk payload logger.debug( "[%s] Part %d/%d: uploading %s (offset=%d md5=%s)", @@ -558,9 +558,9 @@ def _read_file_chunk(file_path: str, offset: int, length: int) -> bytes: def _compute_file_hashes(file_path: str, file_size: int) -> Dict[str, str]: """Compute md5, sha1, and md5_10m in a single pass.""" - md5 = hashlib.md5() - sha1 = hashlib.sha1() - md5_10m = hashlib.md5() + md5 = hashlib.md5() # nosec B324 -- QQ Bot chunked upload protocol field + sha1 = hashlib.sha1() # nosec B324 -- QQ Bot chunked upload protocol field + md5_10m = hashlib.md5() # nosec B324 -- QQ Bot chunked upload first-10MB MD5 protocol field need_10m = file_size > _MD5_10M_SIZE bytes_read = 0 diff --git a/gateway/platforms/wecom.py b/gateway/platforms/wecom.py index 96769ea59b1f5..8a71569e1bca9 100644 --- a/gateway/platforms/wecom.py +++ b/gateway/platforms/wecom.py @@ -1158,7 +1158,7 @@ async def _upload_media_bytes(self, data: bytes, media_type: str, filename: str) "filename": filename, "total_size": total_size, "total_chunks": total_chunks, - "md5": hashlib.md5(data).hexdigest(), + "md5": hashlib.md5(data).hexdigest(), # nosec B324 -- WeCom media upload protocol requires the md5 field }, ) self._raise_for_wecom_error(init_response, "media upload init") diff --git a/gateway/platforms/wecom_crypto.py b/gateway/platforms/wecom_crypto.py index f984ca80c3eb0..a79f34d915d1f 100644 --- a/gateway/platforms/wecom_crypto.py +++ b/gateway/platforms/wecom_crypto.py @@ -60,7 +60,7 @@ def decode(cls, decrypted: bytes) -> bytes: def _sha1_signature(token: str, timestamp: str, nonce: str, encrypt: str) -> str: parts = sorted([token, timestamp, nonce, encrypt]) - return hashlib.sha1("".join(parts).encode("utf-8")).hexdigest() + return hashlib.sha1("".join(parts).encode("utf-8")).hexdigest() # nosec B324 -- WeChat enterprise message-encrypt signature protocol uses SHA-1 class WXBizMsgCrypt: diff --git a/gateway/platforms/weixin.py b/gateway/platforms/weixin.py index 1c9fec0af7fb2..23984d518a489 100644 --- a/gateway/platforms/weixin.py +++ b/gateway/platforms/weixin.py @@ -1392,7 +1392,7 @@ async def _process_message(self, message: Dict[str, Any]) -> None: item_list = message.get("item_list") or [] text = _extract_text(item_list) if text: - content_key = f"content:{sender_id}:{hashlib.md5(text.encode()).hexdigest()}" + content_key = f"content:{sender_id}:{hashlib.md5(text.encode(), usedforsecurity=False).hexdigest()}" if self._dedup.is_duplicate(content_key): logger.debug("[%s] Content-dedup: skipping duplicate message from %s", self.name, sender_id) return @@ -1904,7 +1904,7 @@ async def _send_file( filekey = secrets.token_hex(16) aes_key = secrets.token_bytes(16) rawsize = len(plaintext) - rawfilemd5 = hashlib.md5(plaintext).hexdigest() + rawfilemd5 = hashlib.md5(plaintext).hexdigest() # nosec B324 -- WeChat media upload protocol field name `rawfilemd5` requires MD5 upload_response = await _get_upload_url( self._send_session, base_url=self._base_url, diff --git a/gateway/platforms/yuanbao_media.py b/gateway/platforms/yuanbao_media.py index 87eefcddae2c8..348ef322b134f 100644 --- a/gateway/platforms/yuanbao_media.py +++ b/gateway/platforms/yuanbao_media.py @@ -107,7 +107,7 @@ def get_image_format(mime_type: str) -> int: def md5_hex(data: bytes) -> str: """计算 MD5 十六进制摘要。""" - return hashlib.md5(data).hexdigest() + return hashlib.md5(data, usedforsecurity=False).hexdigest() def generate_file_id() -> str: @@ -308,7 +308,7 @@ def _cos_sign( ]) # Step 3: StringToSign = sha1 hash of HttpString - sha1_of_http = hashlib.sha1(http_string.encode("utf-8")).hexdigest() + sha1_of_http = hashlib.sha1(http_string.encode("utf-8")).hexdigest() # nosec B324 -- Tencent Cloud COS V4 signature requires SHA-1 of HttpString string_to_sign = "\n".join([ "sha1", q_sign_time, diff --git a/tools/skills_hub.py b/tools/skills_hub.py index 35cec56e08e84..d035b51bb5fa4 100644 --- a/tools/skills_hub.py +++ b/tools/skills_hub.py @@ -925,7 +925,7 @@ def _parse_identifier(self, identifier: str) -> Optional[dict]: } def _parse_index(self, index_url: str) -> Optional[dict]: - cache_key = f"well_known_index_{hashlib.md5(index_url.encode()).hexdigest()}" + cache_key = f"well_known_index_{hashlib.md5(index_url.encode(), usedforsecurity=False).hexdigest()}" cached = _read_index_cache(cache_key) if isinstance(cached, dict) and isinstance(cached.get("skills"), list): return cached @@ -1177,7 +1177,7 @@ def search(self, query: str, limit: int = 10) -> List[SkillMeta]: if not query.strip(): return self._featured_skills(limit) - cache_key = f"skills_sh_search_{hashlib.md5(f'{query}|{limit}'.encode()).hexdigest()}" + cache_key = f"skills_sh_search_{hashlib.md5(f'{query}|{limit}'.encode(), usedforsecurity=False).hexdigest()}" cached = _read_index_cache(cache_key) if cached is not None: return [SkillMeta(**item) for item in cached][:limit] @@ -1313,7 +1313,7 @@ def _meta_from_search_item(self, item: dict) -> Optional[SkillMeta]: ) def _fetch_detail_page(self, identifier: str) -> Optional[dict]: - cache_key = f"skills_sh_detail_{hashlib.md5(identifier.encode()).hexdigest()}" + cache_key = f"skills_sh_detail_{hashlib.md5(identifier.encode(), usedforsecurity=False).hexdigest()}" cached = _read_index_cache(cache_key) if isinstance(cached, dict): return cached @@ -1790,7 +1790,7 @@ def search(self, query: str, limit: int = 10) -> List[SkillMeta]: return results # Empty query or catalog fallback failure: use the lightweight listing API. - cache_key = f"clawhub_search_listing_v1_{hashlib.md5(query.encode()).hexdigest()}_{limit}" + cache_key = f"clawhub_search_listing_v1_{hashlib.md5(query.encode(), usedforsecurity=False).hexdigest()}_{limit}" cached = _read_index_cache(cache_key) if cached is not None: return self._finalize_search_results( @@ -1896,7 +1896,7 @@ def inspect(self, identifier: str) -> Optional[SkillMeta]: ) def _search_catalog(self, query: str, limit: int = 10) -> List[SkillMeta]: - cache_key = f"clawhub_search_catalog_v1_{hashlib.md5(f'{query}|{limit}'.encode()).hexdigest()}" + cache_key = f"clawhub_search_catalog_v1_{hashlib.md5(f'{query}|{limit}'.encode(), usedforsecurity=False).hexdigest()}" cached = _read_index_cache(cache_key) if cached is not None: return [SkillMeta(**s) for s in cached][:limit] diff --git a/tools/skills_sync.py b/tools/skills_sync.py index 0c65b6281c77c..d535c098cc1ac 100644 --- a/tools/skills_sync.py +++ b/tools/skills_sync.py @@ -162,7 +162,7 @@ def _compute_relative_dest(skill_dir: Path, bundled_dir: Path) -> Path: def _dir_hash(directory: Path) -> str: """Compute a hash of all file contents in a directory for change detection.""" - hasher = hashlib.md5() + hasher = hashlib.md5(usedforsecurity=False) try: for fpath in sorted(directory.rglob("*")): if fpath.is_file():