fix reload&refrehs
This commit is contained in:
@@ -232,6 +232,8 @@ function mcpCapability(mcp, configuration, pollRunningSeconds=0) {
|
|||||||
if (status === 'connected') return {state: 'on', detail: 'connected'};
|
if (status === 'connected') return {state: 'on', detail: 'connected'};
|
||||||
if (status === 'partial') return {state: 'warn', detail: 'partial'};
|
if (status === 'partial') return {state: 'warn', detail: 'partial'};
|
||||||
if (status === 'refreshing') {
|
if (status === 'refreshing') {
|
||||||
|
const pollBudget = Number(mcp.poll_timeout_seconds || configuration.graylog_mcp_poll_timeout_seconds || 120);
|
||||||
|
if (pollRunningSeconds > pollBudget) return {state: 'warn', detail: `poll over budget ${pollRunningSeconds}s`};
|
||||||
const previous = String(mcp.previous_status || '');
|
const previous = String(mcp.previous_status || '');
|
||||||
if (['connected', 'partial'].includes(previous) || Number(mcp.previous_events_fetched || mcp.events_fetched || 0) > 0 || Number(mcp.previous_aggregate_events || mcp.aggregate_events || 0) > 0) {
|
if (['connected', 'partial'].includes(previous) || Number(mcp.previous_events_fetched || mcp.events_fetched || 0) > 0 || Number(mcp.previous_aggregate_events || mcp.aggregate_events || 0) > 0) {
|
||||||
const suffix = pollRunningSeconds ? `polling ${pollRunningSeconds}s` : 'polling';
|
const suffix = pollRunningSeconds ? `polling ${pollRunningSeconds}s` : 'polling';
|
||||||
@@ -605,7 +607,7 @@ async function refresh() {
|
|||||||
`MCP status: ${esc(mcp.status || 'unknown')}`,
|
`MCP status: ${esc(mcp.status || 'unknown')}`,
|
||||||
mcp.status === 'refreshing' && pollStartedAt ? `MCP poll running: ${esc(pollRunningSeconds)}s` : '',
|
mcp.status === 'refreshing' && pollStartedAt ? `MCP poll running: ${esc(pollRunningSeconds)}s` : '',
|
||||||
pollCompletedAt ? `Last completed MCP poll: ${esc(pollCompletedAge)}s ago` : '',
|
pollCompletedAt ? `Last completed MCP poll: ${esc(pollCompletedAge)}s ago` : '',
|
||||||
mcp.status === 'refreshing' && pollRunningSeconds > Math.max(120, Number(configuration.graylog_range_seconds || 300)) ? `<span class="sev-high">MCP poll has been refreshing for ${esc(pollRunningSeconds)}s. It may be slow or stuck.</span>` : '',
|
mcp.status === 'refreshing' && pollRunningSeconds > Number(mcp.poll_timeout_seconds || configuration.graylog_mcp_poll_timeout_seconds || 120) ? `<span class="sev-high">MCP poll has been refreshing for ${esc(pollRunningSeconds)}s, which is over the configured poll budget. Restart monitor or lower enabled streams/sample size if this keeps happening.</span>` : '',
|
||||||
mcp.error ? `<span class="sev-high">MCP error: ${esc(mcp.error)}</span>` : '',
|
mcp.error ? `<span class="sev-high">MCP error: ${esc(mcp.error)}</span>` : '',
|
||||||
`Enabled streams: ${esc(enabledStreams.length)}`,
|
`Enabled streams: ${esc(enabledStreams.length)}`,
|
||||||
`MCP fetch mode: ${esc(displayedFetchMode)}${mcp.status === 'refreshing' ? ' (refreshing, showing previous counters)' : ''}`,
|
`MCP fetch mode: ${esc(displayedFetchMode)}${mcp.status === 'refreshing' ? ' (refreshing, showing previous counters)' : ''}`,
|
||||||
|
|||||||
@@ -517,7 +517,7 @@ def write_status(status: dict[str, object], output: str) -> None:
|
|||||||
tmp_path.replace(output_path)
|
tmp_path.replace(output_path)
|
||||||
|
|
||||||
|
|
||||||
def write_refreshing_status(output: str, *, cache_path: str | None = None) -> None:
|
def write_refreshing_status(output: str, *, cache_path: str | None = None, call_timeout_seconds: int = 0, poll_timeout_seconds: int = 0) -> None:
|
||||||
output_path = Path(output)
|
output_path = Path(output)
|
||||||
now = int(time.time())
|
now = int(time.time())
|
||||||
try:
|
try:
|
||||||
@@ -561,6 +561,8 @@ def write_refreshing_status(output: str, *, cache_path: str | None = None) -> No
|
|||||||
"previous_poll_completed_at": previous_completed_at,
|
"previous_poll_completed_at": previous_completed_at,
|
||||||
"fetch_mode": previous_fetch_mode,
|
"fetch_mode": previous_fetch_mode,
|
||||||
"coverage_status": previous_coverage_status or "refreshing",
|
"coverage_status": previous_coverage_status or "refreshing",
|
||||||
|
"call_timeout_seconds": call_timeout_seconds or previous_mcp.get("call_timeout_seconds", 0),
|
||||||
|
"poll_timeout_seconds": poll_timeout_seconds or previous_mcp.get("poll_timeout_seconds", 0),
|
||||||
}
|
}
|
||||||
current["status_cache"] = {"served_from_cache": used_cache, "reason": "refreshing"}
|
current["status_cache"] = {"served_from_cache": used_cache, "reason": "refreshing"}
|
||||||
write_status(current, output)
|
write_status(current, output)
|
||||||
@@ -590,13 +592,29 @@ def monitor_loop(
|
|||||||
runtime = ConfigStore(config_path).read() if config_path and Path(config_path).exists() else {}
|
runtime = ConfigStore(config_path).read() if config_path and Path(config_path).exists() else {}
|
||||||
effective_llm = bool(runtime.get("llm_enabled")) if runtime else llm
|
effective_llm = bool(runtime.get("llm_enabled")) if runtime else llm
|
||||||
effective_model = str(runtime.get("llm_model") or llm_model or "")
|
effective_model = str(runtime.get("llm_model") or llm_model or "")
|
||||||
|
mcp_call_timeout = max(1, int(runtime.get("graylog_mcp_call_timeout_seconds", 8) or 8))
|
||||||
|
mcp_poll_timeout = max(60, int(runtime.get("graylog_mcp_poll_timeout_seconds", 120) or 120))
|
||||||
if runtime.get("log_source") == "graylog_mcp":
|
if runtime.get("log_source") == "graylog_mcp":
|
||||||
write_refreshing_status(output, cache_path=status_cache_path)
|
write_refreshing_status(output, cache_path=status_cache_path, call_timeout_seconds=mcp_call_timeout, poll_timeout_seconds=mcp_poll_timeout)
|
||||||
|
try:
|
||||||
status = build_status(
|
status = build_status(
|
||||||
log_path, policy_path=policy_path, anomaly_limit=anomaly_limit,
|
log_path, policy_path=policy_path, anomaly_limit=anomaly_limit,
|
||||||
baseline_path=baseline_path, config_path=config_path, history_path=history_path,
|
baseline_path=baseline_path, config_path=config_path, history_path=history_path,
|
||||||
status_cache_path=status_cache_path,
|
status_cache_path=status_cache_path,
|
||||||
)
|
)
|
||||||
|
except Exception as exc:
|
||||||
|
error_status = {"status": "error", "error": f"monitor_error: {exc}", "call_timeout_seconds": mcp_call_timeout, "poll_timeout_seconds": mcp_poll_timeout}
|
||||||
|
status = cached_status_with_error(status_cache_path, error_status) if status_cache_path else None
|
||||||
|
if not status:
|
||||||
|
status = {
|
||||||
|
"status_schema": 2,
|
||||||
|
"generated_at": int(time.time()),
|
||||||
|
"stale": True,
|
||||||
|
"stale_reason": "monitor_error",
|
||||||
|
"capabilities": {"graylog_mcp": error_status},
|
||||||
|
"summary": {"total": 0},
|
||||||
|
"status_cache": {"served_from_cache": False, "reason": "monitor_error"},
|
||||||
|
}
|
||||||
mcp = status.get("capabilities", {}).get("graylog_mcp", {}) if isinstance(status.get("capabilities"), dict) else {}
|
mcp = status.get("capabilities", {}).get("graylog_mcp", {}) if isinstance(status.get("capabilities"), dict) else {}
|
||||||
if status_cache_path and isinstance(mcp, dict) and mcp.get("status") == "error":
|
if status_cache_path and isinstance(mcp, dict) and mcp.get("status") == "error":
|
||||||
cached = cached_status_with_error(status_cache_path, mcp)
|
cached = cached_status_with_error(status_cache_path, mcp)
|
||||||
|
|||||||
@@ -55,7 +55,7 @@ class MonitorTests(unittest.TestCase):
|
|||||||
encoding="utf-8",
|
encoding="utf-8",
|
||||||
)
|
)
|
||||||
|
|
||||||
write_refreshing_status(str(output))
|
write_refreshing_status(str(output), call_timeout_seconds=8, poll_timeout_seconds=120)
|
||||||
|
|
||||||
status = json.loads(output.read_text(encoding="utf-8"))
|
status = json.loads(output.read_text(encoding="utf-8"))
|
||||||
self.assertFalse(status["stale"])
|
self.assertFalse(status["stale"])
|
||||||
@@ -66,6 +66,8 @@ class MonitorTests(unittest.TestCase):
|
|||||||
self.assertEqual(status["capabilities"]["graylog_mcp"]["previous_events_fetched"], 99)
|
self.assertEqual(status["capabilities"]["graylog_mcp"]["previous_events_fetched"], 99)
|
||||||
self.assertEqual(status["capabilities"]["graylog_mcp"]["previous_aggregate_events"], 1234)
|
self.assertEqual(status["capabilities"]["graylog_mcp"]["previous_aggregate_events"], 1234)
|
||||||
self.assertEqual(status["capabilities"]["graylog_mcp"]["previous_fetch_mode"], "aggregate")
|
self.assertEqual(status["capabilities"]["graylog_mcp"]["previous_fetch_mode"], "aggregate")
|
||||||
|
self.assertEqual(status["capabilities"]["graylog_mcp"]["call_timeout_seconds"], 8)
|
||||||
|
self.assertEqual(status["capabilities"]["graylog_mcp"]["poll_timeout_seconds"], 120)
|
||||||
self.assertFalse(status["status_cache"]["served_from_cache"])
|
self.assertFalse(status["status_cache"]["served_from_cache"])
|
||||||
|
|
||||||
def test_write_refreshing_status_uses_last_good_cache(self):
|
def test_write_refreshing_status_uses_last_good_cache(self):
|
||||||
|
|||||||
Reference in New Issue
Block a user