Added the Profile Discovery report.

This commit is contained in:
larssand
2026-06-30 16:15:54 +02:00
parent 70191d93fb
commit 78f11af302
3 changed files with 117 additions and 0 deletions

View File

@@ -298,6 +298,7 @@ async function refresh() {
document.getElementById('profileSuggestions').innerHTML = visibleProfileSuggestions.length ? table(visibleProfileSuggestions, [ document.getElementById('profileSuggestions').innerHTML = visibleProfileSuggestions.length ? table(visibleProfileSuggestions, [
{label:'Stream', key:'stream_name'}, {label:'Stream', key:'stream_name'},
{label:'Events', key:'events'}, {label:'Events', key:'events'},
{label:'Type', render:r => esc(r.detected_log_type || (r.discovery || {}).log_type || '-')},
{label:'Confidence', key:'confidence'}, {label:'Confidence', key:'confidence'},
{label:'Advisor', render:r => esc((r.profile_advisor || {}).status || (advisor.enabled ? advisor.status : 'heuristic'))}, {label:'Advisor', render:r => esc((r.profile_advisor || {}).status || (advisor.enabled ? advisor.status : 'heuristic'))},
{label:'Profile', render:r => r.profile_exists ? 'exists' : 'new'}, {label:'Profile', render:r => r.profile_exists ? 'exists' : 'new'},
@@ -306,6 +307,7 @@ async function refresh() {
{label:'Baseline fields', render:r => esc([...(r.categorical_fields || []), ...(r.numeric_fields || [])].slice(0,8).join(', ') || '-')}, {label:'Baseline fields', render:r => esc([...(r.categorical_fields || []), ...(r.numeric_fields || [])].slice(0,8).join(', ') || '-')},
{label:'Detectors', render:r => esc(Object.keys(r.detectors || {}).join(', ') || '-')}, {label:'Detectors', render:r => esc(Object.keys(r.detectors || {}).join(', ') || '-')},
{label:'Common denominators', render:r => esc((r.common_fields || []).slice(0,5).map(item => `${item.field} ${(item.coverage*100).toFixed(0)}%`).join(', ') || '-')}, {label:'Common denominators', render:r => esc((r.common_fields || []).slice(0,5).map(item => `${item.field} ${(item.coverage*100).toFixed(0)}%`).join(', ') || '-')},
{label:'Discovery', render:r => { const d=r.discovery || {}; const selected=d.selected_fields || {}; const top=(d.top_fields || []).slice(0,8).map(item => `${item.field} ${(Number(item.coverage || 0)*100).toFixed(0)}%/${item.unique_values}`).join(', '); const rejected=(d.rejected_fields || []).slice(0,8).map(item => `${item.field}: ${item.reason}`).join('; '); const reasons=(d.reasons || []).join('; '); return `<details><summary>${esc(d.field_count || 0)} fields analyzed</summary><p><b>Selected</b><br>Entity: ${esc((selected.entity || []).join(', ') || '-')}<br>Time: ${esc((selected.time || []).join(', ') || '-')}<br>Categorical: ${esc((selected.categorical || []).join(', ') || '-')}<br>Numeric: ${esc((selected.numeric || []).join(', ') || '-')}</p><p><b>Top fields</b><br>${esc(top || '-')}</p><p><b>Rejected</b><br>${esc(rejected || '-')}</p><p>${esc(reasons || '')}</p></details>`; }},
{label:'Action', render:r => r.profile_exists ? '<button type="button" disabled>Applied</button>' : `<button type="button" class="apply-suggested-profile" data-stream-id="${esc(r.stream_id)}">Apply profile</button>`} {label:'Action', render:r => r.profile_exists ? '<button type="button" disabled>Applied</button>' : `<button type="button" class="apply-suggested-profile" data-stream-id="${esc(r.stream_id)}">Apply profile</button>`}
], 'profile-suggestions') : '<p class="muted">No missing stream profiles. Enable "show existing profiles" to inspect already configured profiles.</p>'; ], 'profile-suggestions') : '<p class="muted">No missing stream profiles. Enable "show existing profiles" to inspect already configured profiles.</p>';
const llmText = llm.text ? esc(llm.text).replace(/\\n/g, '<br>') : esc(llm.error || 'LLM assessment disabled or waiting for first run.'); const llmText = llm.text ? esc(llm.text).replace(/\\n/g, '<br>') : esc(llm.error || 'LLM assessment disabled or waiting for first run.');

View File

@@ -242,6 +242,77 @@ def _compact_windows_fields(fields: list[str]) -> list[str]:
return output return output
def _detected_log_type(stream_id: str, stream_name: str, coverage: Counter[str]) -> str:
name = f"{stream_id} {stream_name}".lower()
if _looks_like_windows_stream(stream_id, stream_name, coverage):
return "windows"
if any(token in name for token in ("fortigate", "firewall", "pan-os", "checkpoint", "iptables", "ufw")) or (
coverage["policyid"] and (coverage["dstip"] or coverage["dstport"]) and coverage["action"]
):
return "firewall"
if any(token in name for token in ("dns", "adguard", "bind", "unbound")) or coverage["dns_query"] or coverage["query_domain"] or coverage["qh"]:
return "dns"
if any(token in name for token in ("squid", "proxy", "nginx", "apache", "iis", "web")) or coverage["url"] or coverage["request_uri"] or coverage["http_method"]:
return "web_proxy"
if any(token in name for token in ("audit", "app", "application", "serverlog", "prod")):
return "application"
return "unknown"
def _field_stats(coverage: Counter[str], unique_values: dict[str, set[str]], numeric_counts: Counter[str], total: int) -> list[dict[str, object]]:
rows = []
for field, count in coverage.most_common():
if field in IGNORED_DISCOVERY_FIELDS:
continue
numeric_ratio = numeric_counts[field] / max(1, count)
rows.append({
"field": field,
"coverage": round(count / max(1, total), 2),
"unique_values": len(unique_values[field]),
"numeric_ratio": round(numeric_ratio, 2),
})
return rows
def _rejected_fields(
coverage: Counter[str],
unique_values: dict[str, set[str]],
numeric_counts: Counter[str],
total: int,
selected_fields: set[str],
*,
limit: int = 12,
) -> list[dict[str, object]]:
rejected = []
for field, count in coverage.most_common():
if len(rejected) >= limit:
break
coverage_ratio = count / max(1, total)
cardinality = len(unique_values[field])
reason = ""
if field in selected_fields:
continue
if field in IGNORED_DISCOVERY_FIELDS:
reason = "raw/internal field"
elif coverage_ratio < 0.1:
reason = "low coverage"
elif cardinality <= 1:
reason = "constant value"
elif cardinality > min(200, max(8, int(total * 0.7))):
reason = "too high cardinality"
elif numeric_counts[field] and numeric_counts[field] / max(1, count) < 0.95:
reason = "mixed numeric/text values"
else:
reason = "lower signal than selected fields"
rejected.append({
"field": field,
"reason": reason,
"coverage": round(coverage_ratio, 2),
"unique_values": cardinality,
})
return rejected
def _generic_entity_fields(coverage: Counter[str], unique_values: dict[str, set[str]], total: int, selected: list[str]) -> list[str]: def _generic_entity_fields(coverage: Counter[str], unique_values: dict[str, set[str]], total: int, selected: list[str]) -> list[str]:
output = list(selected) output = list(selected)
for field, count in coverage.most_common(): for field, count in coverage.most_common():
@@ -376,6 +447,7 @@ def suggest_stream_profiles(events: list[LogEvent], *, existing_profiles: dict[s
stream_name = names.get(stream_id, stream_id) stream_name = names.get(stream_id, stream_id)
is_windows = _looks_like_windows_stream(stream_id, stream_name, coverage) is_windows = _looks_like_windows_stream(stream_id, stream_name, coverage)
log_type = _detected_log_type(stream_id, stream_name, coverage)
entity_priority = WINDOWS_ENTITY_PRIORITY if is_windows else ENTITY_PRIORITY entity_priority = WINDOWS_ENTITY_PRIORITY if is_windows else ENTITY_PRIORITY
categorical_priority = WINDOWS_CATEGORICAL_PRIORITY if is_windows else CATEGORICAL_PRIORITY categorical_priority = WINDOWS_CATEGORICAL_PRIORITY if is_windows else CATEGORICAL_PRIORITY
numeric_priority = WINDOWS_NUMERIC_PRIORITY if is_windows else NUMERIC_PRIORITY numeric_priority = WINDOWS_NUMERIC_PRIORITY if is_windows else NUMERIC_PRIORITY
@@ -433,6 +505,24 @@ def suggest_stream_profiles(events: list[LogEvent], *, existing_profiles: dict[s
{"field": field, "coverage": round(count / max(1, total), 2), "unique_values": len(unique_values[field])} {"field": field, "coverage": round(count / max(1, total), 2), "unique_values": len(unique_values[field])}
for field, count in coverage.most_common(12) for field, count in coverage.most_common(12)
] ]
selected_fields = {timestamp, *entity_fields, *categorical, *numeric}
discovery = {
"log_type": log_type,
"field_count": len([field for field in coverage if field not in IGNORED_DISCOVERY_FIELDS]),
"selected_fields": {
"entity": entity_fields,
"time": [timestamp] if timestamp else [],
"categorical": categorical,
"numeric": numeric,
},
"top_fields": _field_stats(coverage, unique_values, numeric_counts, total)[:16],
"rejected_fields": _rejected_fields(coverage, unique_values, numeric_counts, total, selected_fields),
"reasons": [
f"detected {log_type} log pattern",
f"selected {len(entity_fields)} entity field(s), {len(categorical)} categorical field(s), and {len(numeric)} numeric field(s)",
"ignored raw/internal, constant, sparse, and very high-cardinality fields",
],
}
existing = existing_profiles.get(stream_id) existing = existing_profiles.get(stream_id)
score = min(100, 30 + len(entity_fields) * 15 + min(20, len(categorical) * 3) + min(15, len(detectors) * 5)) score = min(100, 30 + len(entity_fields) * 15 + min(20, len(categorical) * 3) + min(15, len(detectors) * 5))
suggestions.append({ suggestions.append({
@@ -448,6 +538,8 @@ def suggest_stream_profiles(events: list[LogEvent], *, existing_profiles: dict[s
"numeric_fields": numeric, "numeric_fields": numeric,
"detectors": detectors, "detectors": detectors,
"common_fields": high_coverage, "common_fields": high_coverage,
"discovery": discovery,
"detected_log_type": log_type,
"profile": { "profile": {
"stream_id": stream_id, "stream_id": stream_id,
"name": f"{names.get(stream_id, stream_id)} recommended profile", "name": f"{names.get(stream_id, stream_id)} recommended profile",

View File

@@ -111,6 +111,29 @@ class ProfileSuggestionTests(unittest.TestCase):
self.assertIn("workflow_state", profile["categorical_fields"]) self.assertIn("workflow_state", profile["categorical_fields"])
self.assertIn("risk_points", profile["numeric_fields"]) self.assertIn("risk_points", profile["numeric_fields"])
def test_profile_discovery_explains_selected_and_rejected_fields(self):
events = [
parse_log_line(
f"fgai_stream_id=app fgai_stream=BillingApp tenant_id=t{index % 4} actor_id=user{index % 7} "
f"workflow_state={'approved' if index % 2 else 'rejected'} request_id=req-{index} "
f"constant_field=same raw='{{json}}' risk_points={index % 9} created_at=2026-06-29T10:00:{index:02d}Z"
)
for index in range(1, 40)
]
suggestion = suggest_stream_profiles(events)[0]
discovery = suggestion["discovery"]
self.assertEqual(suggestion["detected_log_type"], "application")
self.assertEqual(discovery["log_type"], "application")
self.assertIn("actor_id", discovery["selected_fields"]["entity"])
self.assertIn("workflow_state", discovery["selected_fields"]["categorical"])
self.assertIn("risk_points", discovery["selected_fields"]["numeric"])
rejected = {item["field"]: item["reason"] for item in discovery["rejected_fields"]}
self.assertEqual(rejected["raw"], "raw/internal field")
self.assertEqual(rejected["constant_field"], "constant value")
self.assertEqual(rejected["request_id"], "too high cardinality")
def test_applies_valid_llm_advice_and_rejects_unknown_fields(self): def test_applies_valid_llm_advice_and_rejects_unknown_fields(self):
suggestion = suggest_stream_profiles([ suggestion = suggest_stream_profiles([
parse_log_line("fgai_stream_id=windows fgai_stream=Windows username=alice hostname=host01 eventid=4625 action=failure") parse_log_line("fgai_stream_id=windows fgai_stream=Windows username=alice hostname=host01 eventid=4625 action=failure")