keep discoverds fileds in separate sqlite

This commit is contained in:
larssand
2026-07-02 12:51:36 +02:00
parent cc6c97ec79
commit d48d34f96c
5 changed files with 134 additions and 3 deletions

View File

@@ -337,6 +337,8 @@ async function refresh() {
`Baseline sources ready: ${esc((data.baseline || {}).sources_ready || 0)}`,
`Baseline training days: ${esc((data.baseline || {}).training_days || 0)}`,
`Baseline DB size: ${esc(bytes((data.baseline || {}).size_bytes || 0))}`,
`Discovery fields recorded: ${esc((data.baseline || {}).discovery_fields_recorded || 0)}`,
`Discovery cache events: ${esc((data.baseline || {}).discovery_cache_events || 0)}`,
`MCP status: ${esc(mcp.status || 'unknown')}`,
mcp.error ? `<span class="sev-high">MCP error: ${esc(mcp.error)}</span>` : '',
`Enabled streams: ${esc(enabledStreams.length)}`,

View File

@@ -5,6 +5,8 @@ import sqlite3
import time
from pathlib import Path
from .profile_suggestions import IGNORED_DISCOVERY_FIELDS
class HistoryStore:
def __init__(self, path: str) -> None:
@@ -58,3 +60,86 @@ class StatusSnapshotStore:
if isinstance(payload["status_cache"], dict):
payload["status_cache"]["stored_at"] = int(row[0])
return payload
class FieldDiscoveryStore:
def __init__(self, path: str) -> None:
self.path = Path(path)
self.path.parent.mkdir(parents=True, exist_ok=True)
with sqlite3.connect(self.path) as connection:
connection.execute(
"""create table if not exists field_discovery (
stream_id text not null, stream_name text not null, field text not null,
seen_count integer not null, numeric_count integer not null,
sample_values text not null, first_seen integer not null, last_seen integer not null,
primary key (stream_id, field)
)"""
)
def ingest(self, events: list[object], *, max_values: int = 20, retention_days: int = 30) -> int:
now = int(time.time())
pending: dict[tuple[str, str, str], dict[str, object]] = {}
for event in events:
fields = getattr(event, "fields", {})
if not isinstance(fields, dict):
continue
stream_id = str(fields.get("fgai_stream_id") or fields.get("fgai_stream") or "local")
stream_name = str(fields.get("fgai_stream") or fields.get("fgai_stream_name") or stream_id)
for field, value in fields.items():
if field in IGNORED_DISCOVERY_FIELDS:
continue
text = str(value).strip()
if not text or text.lower() in {"-", "--", "unknown", "n/a", "none", "null", "nil", "undefined", "[]", "{}"}:
continue
key = (stream_id, stream_name, str(field))
row = pending.setdefault(key, {"seen_count": 0, "numeric_count": 0, "sample_values": set()})
row["seen_count"] = int(row["seen_count"]) + 1
try:
float(text)
row["numeric_count"] = int(row["numeric_count"]) + 1
except (TypeError, ValueError):
pass
values = row["sample_values"]
if isinstance(values, set) and len(values) < max_values:
values.add(text)
with sqlite3.connect(self.path) as connection:
for (stream_id, stream_name, field), row in pending.items():
existing = connection.execute("select sample_values from field_discovery where stream_id=? and field=?", (stream_id, field)).fetchone()
values = set(row["sample_values"] if isinstance(row["sample_values"], set) else set())
if existing:
try:
values.update(str(item) for item in json.loads(str(existing[0]))[:max_values])
except (json.JSONDecodeError, TypeError):
pass
connection.execute(
"""insert into field_discovery values (?, ?, ?, ?, ?, ?, ?, ?)
on conflict(stream_id, field) do update set
stream_name=excluded.stream_name,
seen_count=field_discovery.seen_count+excluded.seen_count,
numeric_count=field_discovery.numeric_count+excluded.numeric_count,
sample_values=excluded.sample_values,
last_seen=excluded.last_seen""",
(stream_id, stream_name, field, int(row["seen_count"]), int(row["numeric_count"]), json.dumps(sorted(values)[:max_values]), now, now),
)
connection.execute("delete from field_discovery where last_seen < ?", (now - retention_days * 86400,))
return len(pending)
def synthetic_events(self) -> list[object]:
from .models import LogEvent
events: list[LogEvent] = []
with sqlite3.connect(self.path) as connection:
rows = connection.execute("select stream_id, stream_name, field, seen_count, numeric_count, sample_values from field_discovery").fetchall()
for stream_id, stream_name, field, seen_count, numeric_count, sample_values in rows:
try:
values = [str(item) for item in json.loads(str(sample_values))]
except (json.JSONDecodeError, TypeError):
values = []
if not values:
values = ["1" if int(numeric_count or 0) else "observed"]
event_count = max(1, min(5, len(values)))
for index in range(event_count):
value = values[index % len(values)]
fields = {"fgai_stream_id": str(stream_id), "fgai_stream": str(stream_name), str(field): value}
events.append(LogEvent(raw=json.dumps(fields, sort_keys=True), fields=fields))
return events

View File

@@ -13,7 +13,7 @@ from .feedback import FeedbackStore
from .graylog_aggregate import GraylogAggregateSource
from .graylog_mcp import GraylogMcpClient
from .graylog_source import GraylogStreamSource
from .history import HistoryStore, StatusSnapshotStore
from .history import FieldDiscoveryStore, HistoryStore, StatusSnapshotStore
from .incidents import IncidentStore, build_incidents
from .data_quality import assess_data_quality
from .llm import ollama_dashboard_assessment, ollama_profile_advice
@@ -262,7 +262,13 @@ def build_status(
for item in profile_readiness
]
stream_coverage = _stream_coverage(runtime_values, stream_profiles, mcp_status, profile_readiness, stream_titles)
profile_suggestions = suggest_stream_profiles(events, existing_profiles=stream_profiles)
discovery_cache_events = []
discovered_profile_fields = 0
if history_path:
discovery_store = FieldDiscoveryStore(history_path)
discovered_profile_fields = discovery_store.ingest(events)
discovery_cache_events = discovery_store.synthetic_events()
profile_suggestions = suggest_stream_profiles([*discovery_cache_events, *events], existing_profiles=stream_profiles)
profile_advisor_status = {"enabled": bool(runtime_values.get("profile_advisor_enabled")), "status": "disabled"}
if runtime_values.get("profile_advisor_enabled") and profile_suggestions:
try:
@@ -329,7 +335,7 @@ def build_status(
"policy_path": policy_path,
"summary": summary,
"anomaly_summary": anomaly_summary(anomalies),
"baseline": {"enabled": bool(baseline), "sources_ready": len(profiles), "training_days": baseline_training_days, "new_events_recorded": baseline_events, "profile_fields_recorded": profile_baseline_fields, "maintenance": baseline_maintenance, "size_bytes": baseline_maintenance.get("size_bytes", 0) if isinstance(baseline_maintenance, dict) else 0},
"baseline": {"enabled": bool(baseline), "sources_ready": len(profiles), "training_days": baseline_training_days, "new_events_recorded": baseline_events, "profile_fields_recorded": profile_baseline_fields, "discovery_fields_recorded": discovered_profile_fields, "discovery_cache_events": len(discovery_cache_events), "maintenance": baseline_maintenance, "size_bytes": baseline_maintenance.get("size_bytes", 0) if isinstance(baseline_maintenance, dict) else 0},
"capabilities": {"threat_intel": threat_intel_status, "graylog_mcp": mcp_status, "profile_advisor": profile_advisor_status},
"configuration": runtime_config,
"stream_profiles": [{"stream_id": item.stream_id, "name": _profile_name(item.stream_id, stream_titles, item), "stream_name": _stream_name(item.stream_id, stream_titles, item), "entity_field": item.entity_field, "entity_fields": list(item.entity_fields), "timestamp_field": item.timestamp_field, "categorical_fields": list(item.categorical_fields), "numeric_fields": list(item.numeric_fields), "detectors": item.detectors, "field_weights": item.field_weights} for item in stream_profiles.values()],