Normalize CAN IDs number of bits
This commit is contained in:
+20
-1
@@ -12,6 +12,20 @@ import plotly.graph_objects as go
|
||||
from utils.extractor import load_data
|
||||
from utils.extractor import to_int
|
||||
|
||||
def _format_can_id(x):
|
||||
"""Safely cleans CAN ID strings without altering their length or value."""
|
||||
if pd.isna(x):
|
||||
return "UNKNOWN"
|
||||
|
||||
s = str(x).strip()
|
||||
if not s:
|
||||
return "UNKNOWN"
|
||||
|
||||
if s.lower().startswith('0x'):
|
||||
s = s[2:]
|
||||
|
||||
return s.upper()
|
||||
|
||||
def calculate_byte_entropy(df: pd.DataFrame) -> pd.DataFrame:
|
||||
"""Calculates Shannon entropy per byte position for each identifier."""
|
||||
byte_cols = [f"b{i}" for i in range(8)]
|
||||
@@ -19,6 +33,9 @@ def calculate_byte_entropy(df: pd.DataFrame) -> pd.DataFrame:
|
||||
if not available_cols:
|
||||
raise ValueError("No byte columns (b0-b7) found in the DataFrame")
|
||||
|
||||
can_id_col = 'ID' if 'ID' in df.columns else 'Identifier'
|
||||
identifiers = df[can_id_col].apply(_format_can_id)
|
||||
|
||||
df_bytes = df[available_cols].copy()
|
||||
for col in available_cols:
|
||||
df_bytes[col] = df_bytes[col].apply(to_int)
|
||||
@@ -30,7 +47,9 @@ def calculate_byte_entropy(df: pd.DataFrame) -> pd.DataFrame:
|
||||
p = s.value_counts(normalize=True)
|
||||
return -np.sum(p * np.log2(p))
|
||||
|
||||
return df.groupby("Identifier")[available_cols].agg(entropy)
|
||||
result = df_bytes.groupby(identifiers)[available_cols].agg(entropy)
|
||||
result.index.name = 'Identifier'
|
||||
return result
|
||||
|
||||
|
||||
def plot_entropy_heatmap(entropy_df: pd.DataFrame, title: str) -> go.Figure:
|
||||
|
||||
Reference in New Issue
Block a user