Implement entropy heatmap for CAN frames #2

Merged
eeeck merged 9 commits from dev-entropy-heatmap into main 2026-07-13 21:17:34 +02:00
7 changed files with 335 additions and 50 deletions
+4
View File
@@ -1,3 +1,7 @@
# File: decoder.py
# Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later
import argparse import argparse
import polars as pl import polars as pl
+3
View File
@@ -0,0 +1,3 @@
# File: main.py
# Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later
+4
View File
@@ -1,3 +1,7 @@
# File: parser.py
# Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later
import re import re
import csv import csv
import polars as pl import polars as pl
+180
View File
@@ -0,0 +1,180 @@
# File: entropy.py
# Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later
import argparse
from pathlib import Path
import numpy as np
import pandas as pd
import plotly.graph_objects as go
from utils.extractor import load_data
def _to_int(x):
"""Convert a hex string or integer to int, returning NaN on failure."""
if isinstance(x, (int, np.integer)):
return int(x)
if isinstance(x, str):
try:
return int(x, 16)
except ValueError:
return np.nan
return np.nan
def calculate_byte_entropy(df: pd.DataFrame) -> pd.DataFrame:
"""Calculates Shannon entropy per byte position for each identifier."""
byte_cols = [f"b{i}" for i in range(8)]
available_cols = [col for col in byte_cols if col in df.columns]
if not available_cols:
raise ValueError("No byte columns (b0-b7) found in the DataFrame")
df_bytes = df[available_cols].copy()
for col in available_cols:
df_bytes[col] = df_bytes[col].apply(_to_int)
def entropy(s: pd.Series) -> float:
s = s.dropna()
if s.empty:
return 0.0
p = s.value_counts(normalize=True)
return -np.sum(p * np.log2(p))
return df.groupby("Identifier")[available_cols].agg(entropy)
def plot_entropy_heatmap(entropy_df: pd.DataFrame, title: str) -> go.Figure:
"""Generates an interactive heatmap of byte-level Shannon entropy."""
x = entropy_df.columns.tolist()
y = entropy_df.index.tolist()
z = entropy_df.values
fig = go.Figure(
data=go.Heatmap(
z=z,
x=x,
y=y,
colorscale=[
[0.0, "#ffffff"],
[0.15, "#fff7ec"],
[0.35, "#fee8c8"],
[0.55, "#fdd49e"],
[0.75, "#fdbb84"],
[1.0, "#ef6548"],
],
xgap=3,
ygap=3,
text=np.round(z, 2),
texttemplate="%{text}",
textfont={
"size": 11,
"color": "#2a2a2a",
"family": "Segoe UI, Arial, sans-serif",
},
hoverongaps=False,
hovertemplate=(
"<b>%{y}</b><br>"
"Byte %{x}: %{z:.2f} bits<extra></extra>"
),
colorbar=dict(
title=dict(
text="Entropy (bits)",
side="top",
font=dict(size=13, color="#1a1a1a"),
),
orientation="h",
thickness=15,
len=0.35,
x=1.0,
xanchor="right",
y=1.02,
yanchor="bottom",
tickfont=dict(size=11, color="#2a2a2a"),
tickformat=".1f",
outlinewidth=0.5,
outlinecolor="#cccccc",
),
)
)
fig.update_layout(
title=dict(
text=title,
font=dict(size=20, color="#1a1a1a"),
x=0.5,
xanchor="center",
pad=dict(b=20),
),
height=max(600, len(y) * 28 + 150),
autosize=True,
template="plotly_white",
xaxis=dict(
title=dict(text="Byte Position", font=dict(size=13, color="#1a1a1a")),
side="top",
dtick=1,
showgrid=False,
linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"),
ticks="outside",
ticklen=4,
tickcolor="#cccccc",
),
yaxis=dict(
title=dict(text="PGN or CAN ID", font=dict(size=13, color="#1a1a1a")),
autorange="reversed",
showgrid=False,
linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"),
ticks="outside",
ticklen=4,
tickcolor="#cccccc",
automargin=True,
),
font=dict(family="Segoe UI, Arial, sans-serif", size=12, color="#2a2a2a"),
hoverlabel=dict(
bgcolor="white",
font_size=13,
font_family="Segoe UI",
bordercolor="#cccccc",
),
margin=dict(l=200, r=40, t=120, b=60),
)
return fig
if __name__ == "__main__":
parser = argparse.ArgumentParser(
description="Analyze CAN bus byte-level entropy"
)
parser.add_argument(
"input", type=Path, help="Path to the input CAN log file"
)
parser.add_argument(
"output",
type=Path,
nargs="?",
default=Path("entropy_report.html"),
help="Path to the output HTML report",
)
parser.add_argument(
"title",
nargs="?",
default="CAN Bus Byte-Level Entropy",
help="Title for the HTML report",
)
args = parser.parse_args()
df = load_data(args.input)
entropy_df = calculate_byte_entropy(df)
fig = plot_entropy_heatmap(entropy_df, title=args.title)
config = {
"responsive": True,
"displaylogo": False,
"scrollZoom": True,
"modeBarButtonsToAdd": ["toggleSpikelines"],
"toImageButtonOptions": {"format": "png", "scale": 2},
}
fig.write_html(str(args.output), include_plotlyjs="cdn", config=config)
+90 -50
View File
@@ -1,76 +1,116 @@
# File: frequency.py
# Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later
import argparse import argparse
import json
from pathlib import Path from pathlib import Path
import fastparquet
import pandas as pd import pandas as pd
import plotly.express as px import plotly.express as px
import plotly.graph_objects as go import plotly.graph_objects as go
from utils.extractor import load_data
def calc_freq(df: pd.DataFrame) -> pd.DataFrame:
def _extract_identifier(row: pd.Series) -> str:
"""Extracts PGN from metadata or falls back to CAN ID."""
meta = row.get('j1939_metadata')
if pd.isna(meta):
return f"{row['ID']} "
if isinstance(meta, str):
try:
meta = json.loads(meta)
except json.JSONDecodeError:
return f"{row['ID']}"
if isinstance(meta, dict) and 'PGN' in meta:
return f"PGN: {meta['PGN']} "
return f"{row['ID']}"
def load_data(file_path: Path) -> pd.DataFrame:
"""Loads Parquet file and adds an Identifier column."""
df = pd.read_parquet(file_path)
df['Identifier'] = df.apply(_extract_identifier, axis=1)
return df
def calculate_frequency(df: pd.DataFrame) -> pd.DataFrame:
"""Calculates frequency counts and percentages for identifiers.""" """Calculates frequency counts and percentages for identifiers."""
freq_df = df['Identifier'].value_counts().reset_index() freq_df = df['Identifier'].value_counts().reset_index()
freq_df.columns = ['Identifier', 'Count'] freq_df.columns = ['Identifier', 'Count']
total_messages = freq_df['Count'].sum() total = freq_df['Count'].sum()
freq_df['Percentage'] = (freq_df['Count'] / total_messages * 100).round(2) freq_df['Percentage'] = (freq_df['Count'] / total * 100).round(2)
return freq_df.sort_values('Count', ascending=True) return freq_df.sort_values('Count', ascending=True)
def plot_freq(stats_df: pd.DataFrame, title: str) -> go.Figure:
def visualize_frequency(stats_df: pd.DataFrame, title: str = "CAN Bus Message Frequency") -> go.Figure: """Generates interactive horizontal bar chart with log x-axis."""
"""Generates an interactive Plotly horizontal bar chart with a logarithmic x-axis."""
fig = px.bar( fig = px.bar(
stats_df, y='Identifier', x='Count', orientation='h', title=title, log_x=True, stats_df, y='Identifier', x='Count', orientation='h', title=title, log_x=True,
labels={'Identifier': 'PGN / CAN ID', 'Count': 'Message Count'}, labels={'Identifier': 'PGN / CAN ID', 'Count': 'Message Count'},
color='Count', color_continuous_scale='Turbo', color='Count', color_continuous_scale='Turbo',
hover_data={'Percentage': ':.2f', 'Count': True, 'Identifier': True} range_color=(stats_df['Count'].min(), stats_df['Count'].max()),
hover_data={'Percentage': ':.2f', 'Count': ':,', 'Identifier': True}
) )
fig.update_layout( fig.update_layout(
height=max(600, len(stats_df) * 18), width=1000, height=max(600, len(stats_df) * 18),
xaxis_title='Total Message Count (Log Scale)', yaxis_title='PGN or CAN ID', autosize=True,
yaxis={'categoryorder': 'total ascending'}, template='plotly_white',
plot_bgcolor='rgba(0,0,0,0)', paper_bgcolor='white', xaxis=dict(
font=dict(family="Segoe UI, Arial, sans-serif", size=12), title=dict(text="Message count [log scale]", font=dict(size=13, color="#1a1a1a")),
hoverlabel=dict(bgcolor="white", font_size=13, font_family="Segoe UI"), side="top",
margin=dict(l=250, r=50, t=80, b=50), dtick=1,
title=dict(font=dict(size=20), x=0.5) showgrid=False,
linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"),
ticks="outside",
ticklen=4,
tickcolor="#cccccc",
),
yaxis=dict(
title=dict(text="PGN or CAN ID", font=dict(size=13, color="#1a1a1a")),
#autorange="",
showgrid=False,
linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"),
ticks="outside",
ticklen=4,
tickcolor="#cccccc",
automargin=True,
),
font=dict(family="Segoe UI, Arial, sans-serif", size=12, color='#2a2a2a'),
hoverlabel=dict(bgcolor="white", font_size=13, font_family="Segoe UI",
bordercolor='#cccccc'),
margin=dict(l=200, r=40, t=120, b=60),
bargap=0.35,
coloraxis_colorbar=dict(
title=dict(text='Message Count', side='top'),
orientation='h',
thickness=15,
len=0.35,
x=1.0,
xanchor='right',
y=1.02,
yanchor='bottom',
tickformat=',',
outlinecolor='#cccccc',
outlinewidth=0.5
),
title=dict(font=dict(size=20, color='#1a1a1a'), x=0.5, xanchor='center',
pad=dict(b=20))
)
fig.update_xaxes(
showgrid=True, gridwidth=0.5, gridcolor='#e8e8e8',
zeroline=False, linecolor='#bdbdbd', mirror=False,
tickformat=',',
minor=dict(showgrid=True, gridcolor='#f4f4f4', gridwidth=0.5)
)
fig.update_yaxes(
showgrid=False, zeroline=False, linecolor='#bdbdbd',
ticks='outside', ticklen=4, tickcolor='#cccccc',
automargin=True
) )
fig.update_traces( fig.update_traces(
hovertemplate="<b>%{y}</b><br>Count: %{x:,}<br>Share: %{customdata[0]}%<extra></extra>" hovertemplate="<b>%{y}</b><br>Count: %{x:,}<br>Share: %{customdata[0]}%<extra></extra>",
marker_line_width=0,
texttemplate='%{x:,}',
textposition='outside',
textfont=dict(size=10, color='#666666'),
cliponaxis=False,
selected=dict(marker=dict(opacity=0.6)),
unselected=dict(marker=dict(opacity=0.2))
) )
return fig return fig
if __name__ == "__main__": if __name__ == "__main__":
parser = argparse.ArgumentParser(description="Analyze CAN bus Parquet data.") parser = argparse.ArgumentParser(description="Analyze CAN bus message frequency")
parser.add_argument("-i", "--input", type=Path, required=True) parser.add_argument("input", type=Path, help="Path to the input CAN log file")
parser.add_argument("-o", "--output", type=Path, default=Path("can_analysis_report.html")) parser.add_argument("output", type=Path, nargs="?", default=Path("freq_report.html"), help="Path to the output HTML report")
parser.add_argument("-t", "--title", type=str, default="CAN Bus Message Frequency by PGN / ID") parser.add_argument("title", nargs="?", default="CAN Bus Message Frequency", help="Title for the HTML report")
args = parser.parse_args() args = parser.parse_args()
df = load_data(args.input) df = load_data(args.input)
stats_df = calculate_frequency(df) stats = calc_freq(df)
fig = visualize_frequency(stats_df, title=args.title) fig = plot_freq(stats, title=args.title)
fig.write_html(str(args.output), include_plotlyjs='cdn') config = {
'responsive': True,
'displaylogo': False,
'scrollZoom': True,
'modeBarButtonsToAdd': ['toggleSpikelines'],
'toImageButtonOptions': {'format': 'png', 'scale': 2}
}
fig.write_html(str(args.output), include_plotlyjs='cdn', config=config)
+27
View File
@@ -0,0 +1,27 @@
# File: extractor.py
# Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later
import json
from pathlib import Path
import pandas as pd
def extract_id(row: pd.Series) -> str:
"""Extracts PGN from metadata or falls back to CAN ID."""
meta = row.get('j1939_metadata')
if pd.isna(meta):
return f"ID: {row['ID']}"
if isinstance(meta, str):
try:
meta = json.loads(meta)
except json.JSONDecodeError:
return f"ID: {row['ID']}"
if isinstance(meta, dict) and 'PGN' in meta:
return f"PGN: {meta['PGN']}"
return f"ID: {row['ID']}"
def load_data(file_path: Path) -> pd.DataFrame:
"""Loads Parquet file and adds an Identifier column."""
df = pd.read_parquet(file_path)
df['Identifier'] = df.apply(extract_id, axis=1)
return df
+27
View File
@@ -0,0 +1,27 @@
# File: extractor.py
# Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later
import json
from pathlib import Path
import pandas as pd
def extract_id(row: pd.Series) -> str:
"""Extracts PGN from metadata or falls back to CAN ID."""
meta = row.get('j1939_metadata')
if pd.isna(meta):
return f"ID: {row['ID']}"
if isinstance(meta, str):
try:
meta = json.loads(meta)
except json.JSONDecodeError:
return f"ID: {row['ID']}"
if isinstance(meta, dict) and 'PGN' in meta:
return f"PGN: {meta['PGN']}"
return f"ID: {row['ID']}"
def load_data(file_path: Path) -> pd.DataFrame:
"""Loads Parquet file and adds an Identifier column."""
df = pd.read_parquet(file_path)
df['Identifier'] = df.apply(extract_id, axis=1)
return df