Refactor statistical analysis modules for performance
- Optimize data processing pipelines across files by replacing iterative pandas operations with vectorized NumPy routines
This commit is contained in:
+66
-105
@@ -12,57 +12,76 @@ import plotly.graph_objects as go
|
||||
from utils.extractor import load_data
|
||||
from utils.extractor import to_int
|
||||
|
||||
def _format_can_id(x):
|
||||
"""Safely cleans CAN ID strings without altering their length or value."""
|
||||
if pd.isna(x):
|
||||
return "UNKNOWN"
|
||||
def _format_can_id_vec(s: pd.Series) -> pd.Series:
|
||||
s = s.astype('string').str.strip()
|
||||
s = s.str.replace(r'^0x', '', case=False, regex=True)
|
||||
s = s.str.upper()
|
||||
return s.fillna('UNKNOWN').replace('', 'UNKNOWN')
|
||||
|
||||
s = str(x).strip()
|
||||
if not s:
|
||||
return "UNKNOWN"
|
||||
|
||||
if s.lower().startswith('0x'):
|
||||
s = s[2:]
|
||||
|
||||
return s.upper()
|
||||
def _entropy_col(a: np.ndarray) -> float:
|
||||
a = a[~np.isnan(a)]
|
||||
if a.size == 0:
|
||||
return 0.0
|
||||
a = a.astype(np.int64)
|
||||
lo, hi = a.min(), a.max()
|
||||
span = hi - lo + 1
|
||||
if span <= 0:
|
||||
return 0.0
|
||||
if span > 1 << 20:
|
||||
_, counts = np.unique(a, return_counts=True)
|
||||
else:
|
||||
counts = np.bincount(a - lo, minlength=span)
|
||||
counts = counts[counts > 0]
|
||||
p = counts / counts.sum()
|
||||
return float(-np.sum(p * np.log2(p)))
|
||||
|
||||
def calculate_byte_entropy(df: pd.DataFrame) -> pd.DataFrame:
|
||||
"""Calculates Shannon entropy per byte position for each identifier."""
|
||||
byte_cols = [f"b{i}" for i in range(8)]
|
||||
available_cols = [col for col in byte_cols if col in df.columns]
|
||||
byte_cols = [f"b{i}" for i in range(8) if f"b{i}" in df.columns]
|
||||
available_cols = byte_cols
|
||||
if not available_cols:
|
||||
raise ValueError("No byte columns (b0-b7) found in the DataFrame")
|
||||
|
||||
can_id_col = 'ID' if 'ID' in df.columns else 'Identifier'
|
||||
identifiers = df[can_id_col].apply(_format_can_id)
|
||||
identifiers = _format_can_id_vec(df[can_id_col]).to_numpy()
|
||||
|
||||
df_bytes = df[available_cols].copy()
|
||||
for col in available_cols:
|
||||
df_bytes[col] = df_bytes[col].apply(to_int)
|
||||
needs = [c for c in available_cols if not pd.api.types.is_numeric_dtype(df[c])]
|
||||
if needs:
|
||||
df = df.copy()
|
||||
for c in needs:
|
||||
df[c] = df[c].apply(to_int)
|
||||
|
||||
def entropy(s: pd.Series) -> float:
|
||||
s = s.dropna()
|
||||
if s.empty:
|
||||
return 0.0
|
||||
p = s.value_counts(normalize=True)
|
||||
return -np.sum(p * np.log2(p))
|
||||
data = df[available_cols].to_numpy(dtype=np.float64, copy=False)
|
||||
unique_ids, inverse = np.unique(identifiers, return_inverse=True)
|
||||
n_cols = len(available_cols)
|
||||
|
||||
result = df_bytes.groupby(identifiers)[available_cols].agg(entropy)
|
||||
sort_idx = np.argsort(inverse, kind='stable')
|
||||
data_sorted = data[sort_idx]
|
||||
inverse_sorted = inverse[sort_idx]
|
||||
|
||||
if len(inverse_sorted) > 0:
|
||||
split_points = np.flatnonzero(np.diff(inverse_sorted)) + 1
|
||||
groups = np.split(data_sorted, split_points)
|
||||
else:
|
||||
groups = []
|
||||
|
||||
out = np.zeros((len(unique_ids), n_cols), dtype=np.float64)
|
||||
for gi, sub in enumerate(groups):
|
||||
for ci in range(n_cols):
|
||||
out[gi, ci] = _entropy_col(sub[:, ci])
|
||||
|
||||
result = pd.DataFrame(out, index=unique_ids, columns=available_cols)
|
||||
result.index.name = 'Identifier'
|
||||
return result
|
||||
|
||||
|
||||
def plot_entropy_heatmap(entropy_df: pd.DataFrame, title: str) -> go.Figure:
|
||||
"""Generates an interactive heatmap of byte-level Shannon entropy."""
|
||||
x = entropy_df.columns.tolist()
|
||||
y = entropy_df.index.tolist()
|
||||
z = entropy_df.values
|
||||
|
||||
fig = go.Figure(
|
||||
data=go.Heatmap(
|
||||
z=z,
|
||||
x=x,
|
||||
y=y,
|
||||
z=z, x=x, y=y,
|
||||
colorscale=[
|
||||
[0.0, "#ffffff"],
|
||||
[0.15, "#fff7ec"],
|
||||
@@ -71,112 +90,54 @@ def plot_entropy_heatmap(entropy_df: pd.DataFrame, title: str) -> go.Figure:
|
||||
[0.75, "#fdbb84"],
|
||||
[1.0, "#ef6548"],
|
||||
],
|
||||
xgap=3,
|
||||
ygap=3,
|
||||
xgap=3, ygap=3,
|
||||
text=np.round(z, 2),
|
||||
texttemplate="%{text}",
|
||||
textfont={
|
||||
"size": 11,
|
||||
"color": "#2a2a2a",
|
||||
"family": "Segoe UI, Arial, sans-serif",
|
||||
},
|
||||
textfont={"size": 11, "color": "#2a2a2a", "family": "Segoe UI, Arial, sans-serif"},
|
||||
hoverongaps=False,
|
||||
hovertemplate=(
|
||||
"<b>%{y}</b><br>"
|
||||
"Byte %{x}: %{z:.2f} bits<extra></extra>"
|
||||
),
|
||||
hovertemplate="<b>%{y}</b><br>Byte %{x}: %{z:.2f} bits<extra></extra>",
|
||||
colorbar=dict(
|
||||
title=dict(
|
||||
text="Entropy (bits)",
|
||||
side="top",
|
||||
font=dict(size=13, color="#1a1a1a"),
|
||||
),
|
||||
orientation="h",
|
||||
thickness=15,
|
||||
len=0.35,
|
||||
x=1.0,
|
||||
xanchor="right",
|
||||
y=1.02,
|
||||
yanchor="bottom",
|
||||
title=dict(text="Entropy (bits)", side="top", font=dict(size=13, color="#1a1a1a")),
|
||||
orientation="h", thickness=15, len=0.35,
|
||||
x=1.0, xanchor="right", y=1.02, yanchor="bottom",
|
||||
tickfont=dict(size=11, color="#2a2a2a"),
|
||||
tickformat=".1f",
|
||||
outlinewidth=0.5,
|
||||
outlinecolor="#cccccc",
|
||||
tickformat=".1f", outlinewidth=0.5, outlinecolor="#cccccc",
|
||||
),
|
||||
)
|
||||
)
|
||||
|
||||
fig.update_layout(
|
||||
title=dict(
|
||||
text=title,
|
||||
font=dict(size=20, color="#1a1a1a"),
|
||||
x=0.5,
|
||||
xanchor="center",
|
||||
pad=dict(b=20),
|
||||
),
|
||||
title=dict(text=title, font=dict(size=20, color="#1a1a1a"), x=0.5, xanchor="center", pad=dict(b=20)),
|
||||
height=max(600, len(y) * 28 + 150),
|
||||
autosize=True,
|
||||
template="plotly_white",
|
||||
xaxis=dict(
|
||||
title=dict(text="Byte Position", font=dict(size=13, color="#1a1a1a")),
|
||||
side="top",
|
||||
dtick=1,
|
||||
showgrid=False,
|
||||
linecolor="#bdbdbd",
|
||||
tickfont=dict(size=12, color="#2a2a2a"),
|
||||
ticks="outside",
|
||||
ticklen=4,
|
||||
tickcolor="#cccccc",
|
||||
side="top", dtick=1, showgrid=False, linecolor="#bdbdbd",
|
||||
tickfont=dict(size=12, color="#2a2a2a"), ticks="outside", ticklen=4, tickcolor="#cccccc",
|
||||
),
|
||||
yaxis=dict(
|
||||
title=dict(text="PGN or CAN ID", font=dict(size=13, color="#1a1a1a")),
|
||||
autorange="reversed",
|
||||
showgrid=False,
|
||||
linecolor="#bdbdbd",
|
||||
tickfont=dict(size=12, color="#2a2a2a"),
|
||||
ticks="outside",
|
||||
ticklen=4,
|
||||
tickcolor="#cccccc",
|
||||
automargin=True,
|
||||
autorange="reversed", showgrid=False, linecolor="#bdbdbd",
|
||||
tickfont=dict(size=12, color="#2a2a2a"), ticks="outside", ticklen=4, tickcolor="#cccccc", automargin=True,
|
||||
),
|
||||
font=dict(family="Segoe UI, Arial, sans-serif", size=12, color="#2a2a2a"),
|
||||
hoverlabel=dict(
|
||||
bgcolor="white",
|
||||
font_size=13,
|
||||
font_family="Segoe UI",
|
||||
bordercolor="#cccccc",
|
||||
),
|
||||
hoverlabel=dict(bgcolor="white", font_size=13, font_family="Segoe UI", bordercolor="#cccccc"),
|
||||
margin=dict(l=200, r=40, t=120, b=60),
|
||||
)
|
||||
return fig
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Analyze CAN bus byte-level entropy"
|
||||
)
|
||||
parser.add_argument(
|
||||
"input", type=Path, help="Path to the input CAN log file"
|
||||
)
|
||||
parser.add_argument(
|
||||
"output",
|
||||
type=Path,
|
||||
nargs="?",
|
||||
default=Path("entropy_report.html"),
|
||||
help="Path to the output HTML report",
|
||||
)
|
||||
parser.add_argument(
|
||||
"title",
|
||||
nargs="?",
|
||||
default="CAN Bus Byte-Level Entropy",
|
||||
help="Title for the HTML report",
|
||||
)
|
||||
parser = argparse.ArgumentParser(description="Analyze CAN bus byte-level entropy")
|
||||
parser.add_argument("input", type=Path, help="Path to the input CAN log file")
|
||||
parser.add_argument("output", type=Path, nargs="?", default=Path("entropy_report.html"))
|
||||
parser.add_argument("title", nargs="?", default="CAN Bus Byte-Level Entropy")
|
||||
args = parser.parse_args()
|
||||
|
||||
df = load_data(args.input)
|
||||
entropy_df = calculate_byte_entropy(df)
|
||||
fig = plot_entropy_heatmap(entropy_df, title=args.title)
|
||||
|
||||
config = {
|
||||
"responsive": True,
|
||||
"displaylogo": False,
|
||||
|
||||
Reference in New Issue
Block a user