Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 8 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -252,6 +252,14 @@

- **mcp:** move `enforceScopes` guard before `MCP_TOOL_MAP` lookup, add inline `scopes` parameter to `withScopeEnforcement()`, and declare scopes on all 24 dynamic tool definitions (memory, skills, plugins, gamification, compression) to fix scope enforcement for dynamic MCP tool groups (#2958)

### ✨ New Features

- **notion:** add Notion as an MCP context source — 6 tools (`notion_search`, `notion_list_databases`, `notion_get_database`, `notion_query_database`, `notion_read`, `notion_append_blocks`) scoped under `read:notion` / `write:notion`, with dashboard "Context Sources" tab, settings API, and token persistence in `key_value` table (#2959)

### 🔧 Bug Fixes

- **mcp:** move `enforceScopes` guard before `MCP_TOOL_MAP` lookup, add inline `scopes` parameter to `withScopeEnforcement()`, and declare scopes on all 24 dynamic tool definitions (memory, skills, plugins, gamification, compression) to fix scope enforcement for dynamic MCP tool groups (#2958)

---

## [3.8.7] — 2026-05-29
Expand Down
63 changes: 61 additions & 2 deletions open-sse/handlers/chatCore.ts
Original file line number Diff line number Diff line change
Expand Up @@ -229,6 +229,11 @@ import {

const MEMORY_EXTRACTION_TEXT_LIMIT = 64 * 1024;

// ── Global memory pressure guard ────────────────────────────────────────
// Prevents OOM by rejecting new requests when V8 heap exceeds threshold.
// Self-healing: no counters to leak, no cleanup needed.
const HEAP_PRESSURE_THRESHOLD_MB = parseInt(process.env.HEAP_PRESSURE_THRESHOLD_MB || "200", 10);

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

P2 Badge Tie heap-pressure threshold to the configured heap limit

With the new hard-coded 200 MB default, any deployment using the repo defaults can start returning 503s while still far from OOM: Docker sets OMNIROUTE_MEMORY_MB=1024 in Dockerfile, and omniroute serve defaults/clamps to 512 MB, but the guard rejects all chat requests once heapUsed crosses 200 MB unless operators discover and set this separate env var. This makes normal high-memory-but-healthy processes unavailable; derive the threshold from the actual V8 heap limit / OMNIROUTE_MEMORY_MB or default it proportionally.

Useful? React with 👍 / 👎.


function capMemoryExtractionText(value: string): string {
if (value.length <= MEMORY_EXTRACTION_TEXT_LIMIT) return value;
return value.slice(-MEMORY_EXTRACTION_TEXT_LIMIT);
Expand Down Expand Up @@ -280,6 +285,32 @@ function cloneBoundedChatLogPayload(value: unknown, depth = 0): unknown {

import { estimateSizeFast, isSmallEnoughForSemanticCache } from "../utils/estimateSize.ts";

const MAX_LOG_BODY_CHARS = 8 * 1024; // 8KB cap for logged request/response bodies
/**
* Truncate a large object for logging. If its JSON representation exceeds
* MAX_LOG_BODY_CHARS, return a lightweight summary instead of the full clone.
* This prevents persistAttemptLogs from holding multi-MB references to
* translatedBody across 17 call sites per request.
*/
function truncateForLog(value: unknown): Record<string, unknown> | null | undefined {
if (value === null || value === undefined) return value as null | undefined;
if (typeof value !== "object") return value as unknown as Record<string, unknown>;
const estimatedSize = estimateSizeFast(value);
if (estimatedSize <= MAX_LOG_BODY_CHARS) return value as Record<string, unknown>;
// Object is too large — return a summary instead of a deep clone
const obj = value as Record<string, unknown>;
const summary: Record<string, unknown> = {
_truncated: true,
_originalBytes: estimatedSize,
};
if (typeof obj.model === "string") summary.model = obj.model;
if (typeof obj.provider === "string") summary.provider = obj.provider;
if (Array.isArray(obj.messages)) summary.messageCount = obj.messages.length;
if (Array.isArray(obj.contents)) summary.contentCount = obj.contents.length;
if (typeof obj.stream === "boolean") summary.stream = obj.stream;
return summary;
}
Comment thread
soyelmismo marked this conversation as resolved.

function extractMemoryTextFromResponse(
response: Record<string, unknown> | null | undefined
): string {
Expand Down Expand Up @@ -1485,6 +1516,34 @@ export async function handleChatCore({
skipUpstreamRetry = false,
}) {
let { provider, model, extendedContext } = modelInfo;
// ── Memory pressure guard ────────────────────────────────────────────
// Reject early if V8 heap is already near the 256MB limit. Prevents
// cascading OOM when many large-context requests arrive concurrently.
try {
const heapUsedMB = process.memoryUsage().heapUsed / (1024 * 1024);
if (heapUsedMB > HEAP_PRESSURE_THRESHOLD_MB) {
// Internal telemetry only — never expose the heap figure to clients (Hard Rule #12).
console.warn(
`[chatCore] heap pressure guard tripped: ${Math.round(heapUsedMB)}MB > ${HEAP_PRESSURE_THRESHOLD_MB}MB; returning 503`
);
return {
success: false,
status: 503,
error: "Service temporarily unavailable due to resource pressure. Retry shortly.",
response: new Response(
JSON.stringify({
error: {
message: "Service temporarily unavailable due to resource pressure. Retry shortly.",
type: "server_error",
code: "heap_pressure",
},
}),
{ status: 503, headers: { "Content-Type": "application/json", "Retry-After": "5" } }
),
};
}
} catch { /* memoryUsage() never throws */ }
Comment thread
soyelmismo marked this conversation as resolved.

// apiFormat is an optional custom-model marker injected by getModelInfo for
// providers whose models can route to /chat/completions or /responses
// (Azure AI Foundry, OCI generic OpenAI). It's not on the base ModelInfo
Expand Down Expand Up @@ -2040,12 +2099,12 @@ export async function handleChatCore({
duration: Date.now() - startTime,
tokens: tokens || {},
requestBody: cloneBoundedChatLogPayload(
attachLogMeta((body as Record<string, unknown>) ?? undefined, {
attachLogMeta(truncateForLog(body as Record<string, unknown>), {
claudePromptCache: claudeCacheMeta,
})
),
responseBody: cloneBoundedChatLogPayload(
attachLogMeta((responseBody as Record<string, unknown>) ?? undefined, {
attachLogMeta(truncateForLog(responseBody as Record<string, unknown>), {
claudePromptCache: claudeCacheMeta
? {
applied: claudeCacheMeta.applied,
Expand Down
80 changes: 80 additions & 0 deletions tests/unit/chatcore-memory-pressure.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,80 @@
import test from "node:test";
import assert from "node:assert/strict";

// ── estimateSizeFast (truncateForLog dependency) ───────────────────────

test("estimateSizeFast estimates small objects correctly", async () => {
const { estimateSizeFast } = await import("../../open-sse/utils/estimateSize.ts");
const small = { model: "gpt-4", messages: [{ role: "user", content: "hello" }] };
const size = estimateSizeFast(small);
assert.ok(size > 0, "Size should be positive");
assert.ok(size < 1024, `Small object should be < 1KB, got ${size}`);
});

test("estimateSizeFast estimates large objects correctly", async () => {
const { estimateSizeFast } = await import("../../open-sse/utils/estimateSize.ts");
const largeContent = "x".repeat(100 * 1024);
const large = { model: "gpt-4", messages: [{ role: "user", content: largeContent }] };
const size = estimateSizeFast(large);
assert.ok(size > 100 * 1024, `Large object should be > 100KB, got ${size}`);
});

test("estimateSizeFast handles null and primitives", async () => {
const { estimateSizeFast } = await import("../../open-sse/utils/estimateSize.ts");
assert.equal(estimateSizeFast(null), 0);
assert.equal(estimateSizeFast(undefined), 0);
assert.equal(estimateSizeFast("hello"), 5);
assert.equal(estimateSizeFast(42), 8);
assert.equal(estimateSizeFast(true), 4);
});

test("estimateSizeFast handles circular references", async () => {
const { estimateSizeFast } = await import("../../open-sse/utils/estimateSize.ts");
const obj: Record<string, unknown> = { a: 1 };
obj.self = obj;
const size = estimateSizeFast(obj);
assert.ok(size > 0, "Should handle circular refs");
assert.ok(size < 1000, `Simple circular ref should be small, got ${size}`);
});

// ── HEAP_PRESSURE_THRESHOLD_MB default ─────────────────────────────────

test("HEAP_PRESSURE_THRESHOLD_MB defaults to 200 when env is unset", () => {
const val = parseInt(process.env.HEAP_PRESSURE_THRESHOLD_MB || "200", 10);
assert.equal(val, 200);
});

// ── estimateSizeFast vs MAX_LOG_BODY_CHARS threshold ───────────────────
// This validates the logic that truncateForLog uses: if estimateSizeFast
// returns <= 8KB, the object is kept as-is; otherwise it's summarized.

test("estimateSizeFast distinguishes small vs large for 8KB threshold", async () => {
const { estimateSizeFast } = await import("../../open-sse/utils/estimateSize.ts");
const MAX_LOG_BODY_CHARS = 8 * 1024;

// Small: 50 messages with short content
const small = {
model: "gpt-4",
messages: Array.from({ length: 50 }, (_, i) => ({
role: "user",
content: `message ${i}`,
})),
};
assert.ok(
estimateSizeFast(small) <= MAX_LOG_BODY_CHARS,
`Small payload (${estimateSizeFast(small)}B) should be <= ${MAX_LOG_BODY_CHARS}`
);

// Large: 500 messages with long content
const large = {
model: "gpt-4",
messages: Array.from({ length: 500 }, (_, i) => ({
role: "user",
content: `x`.repeat(500),
})),
};
assert.ok(
estimateSizeFast(large) > MAX_LOG_BODY_CHARS,
`Large payload (${estimateSizeFast(large)}B) should be > ${MAX_LOG_BODY_CHARS}`
);
});