16 Commits

Author SHA1 Message Date
eeeck fe4bc0e46a Merge pull request 'Implement Pearson and Spearman per-bit correleration' (#3) from dev-inter-byte-correlation into main
Reviewed-on: erickahmed/CANveyor#3
2026-07-13 21:54:38 +02:00
eeeck c9ae482174 Add positional argument to choose correlation methods (Pearson or
Spearman)
2026-07-13 21:53:20 +02:00
eeeck 5e6f81b50b Add hex converter utility
- To move to separate utility file in the future
2026-07-13 21:50:32 +02:00
eeeck ce74c7fcd9 Treat undefined correlation as zero correlation 2026-07-13 21:43:24 +02:00
eeeck 0f25c0671b Implement Pearson correlation 2026-07-13 21:43:17 +02:00
eeeck d0e70dad2a Remove leftovers 2026-07-13 21:20:07 +02:00
eeeck f39d23fc64 Merge pull request 'Implement entropy heatmap for CAN frames' (#2) from dev-entropy-heatmap into main
Reviewed-on: erickahmed/CANveyor#2
2026-07-13 21:17:33 +02:00
eeeck 7daf4c8e08 Add entropy heatmap analysis
- Same style of frequency analysis for consistency
2026-07-13 21:16:18 +02:00
eeeck 8a3fc81d12 Use more consistent styling
- X axis on top
- Refactor figure generation logic
2026-07-13 21:15:36 +02:00
eeeck db7e5eb40e Add header informations 2026-07-13 20:08:29 +02:00
eeeck 8ccb97afae Remove requirement for positional arguments 2026-07-13 20:07:10 +02:00
eeeck 13d91d566b Move to subfolder
- Makes python recognize it as a submodule
2026-07-13 20:06:45 +02:00
eeeck 118f1567a2 Merge pull request 'Add header with copyright and SPDX licensing information' (#1) from code-header into dev-entropy-heatmap
Reviewed-on: erickahmed/CANveyor#1
2026-07-13 19:54:10 +02:00
eeeck cbba800536 Merge branch 'dev-entropy-heatmap' into code-header 2026-07-13 19:53:59 +02:00
eeeck 17340746ce Move data extraction functions to utility library 2026-07-13 19:51:22 +02:00
eeeck ad8334c915 Add header with copyright and SPDX licensing information 2026-07-13 19:48:46 +02:00
7 changed files with 461 additions and 50 deletions
+4
View File
@@ -1,3 +1,7 @@
# File: decoder.py
# Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later
import argparse import argparse
import polars as pl import polars as pl
+3
View File
@@ -0,0 +1,3 @@
# File: main.py
# Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later
+4
View File
@@ -1,3 +1,7 @@
# File: parser.py
# Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later
import re import re
import csv import csv
import polars as pl import polars as pl
+144
View File
@@ -0,0 +1,144 @@
# File: correlation.py
# Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later
import argparse
from pathlib import Path
import numpy as np
import pandas as pd
import plotly.graph_objects as go
from utils.extractor import load_data
def _to_int(x):
"""Convert a hex string or integer to int, returning NaN on failure."""
if isinstance(x, (int, np.integer)):
return int(x)
if isinstance(x, str):
try:
return int(x, 16)
except ValueError:
return np.nan
return np.nan
def calculate_correlation(df: pd.DataFrame, method: str, target_id: str | None = None) -> pd.DataFrame:
"""Calculates inter-byte correlation grouped by identifier."""
byte_cols = [f"b{i}" for i in range(8)]
available_cols = [col for col in byte_cols if col in df.columns]
if not available_cols:
raise ValueError("No byte columns (b0-b7) found in the DataFrame")
df_bytes = df[["Identifier"] + available_cols].copy()
for col in available_cols:
df_bytes[col] = df_bytes[col].apply(_to_int)
if target_id:
group = df_bytes[df_bytes["Identifier"] == target_id]
if group.empty:
raise ValueError(f"Identifier '{target_id}' not found in data")
return group[available_cols].corr(method=method).fillna(0.0)
def max_abs_corr(group: pd.DataFrame) -> pd.Series:
corr_arr = np.abs(group.corr(method=method).to_numpy().copy())
np.fill_diagonal(corr_arr, 0.0)
return pd.Series(corr_arr.max(axis=0), index=group.columns).fillna(0.0)
return df_bytes.groupby("Identifier")[available_cols].apply(max_abs_corr)
def plot_correlation_heatmap(corr_df: pd.DataFrame, target_id: str | None, title: str) -> go.Figure:
"""Generates an interactive heatmap of inter-byte correlation."""
is_8x8 = target_id is not None
if is_8x8:
x = corr_df.columns.tolist()
y = corr_df.index.tolist()
z = corr_df.values
z_min, z_max = -1.0, 1.0
colorscale = [
[0.0, "#2c7bb6"], [0.25, "#abd9e9"], [0.5, "#ffffff"],
[0.75, "#fdae61"], [1.0, "#d7191c"]
]
hover_template = "<b>%{y}</b> vs <b>%{x}</b><br>Correlation: %{z:.2f}<extra></extra>"
else:
x = corr_df.columns.tolist()
y = corr_df.index.tolist()
z = corr_df.values
z_min, z_max = 0.0, 1.0
colorscale = [
[0.0, "#ffffff"], [0.2, "#fff5f0"], [0.4, "#fecc5c"],
[0.6, "#fd8d3c"], [0.8, "#e31a1c"], [1.0, "#800026"]
]
hover_template = "<b>%{y}</b><br>Byte %{x} max correlation: %{z:.2f}<extra></extra>"
fig = go.Figure(
data=go.Heatmap(
z=z, x=x, y=y,
zmin=z_min, zmax=z_max,
colorscale=colorscale,
xgap=3, ygap=3,
text=np.round(z, 2),
texttemplate="%{text}",
textfont={"size": 11, "color": "#2a2a2a", "family": "Segoe UI, Arial, sans-serif"},
hoverongaps=False,
hovertemplate=hover_template,
colorbar=dict(
title=dict(text="Correlation", side="top", font=dict(size=13, color="#1a1a1a")),
orientation="h", thickness=15, len=0.35,
x=1.0, xanchor="right", y=1.02, yanchor="bottom",
tickfont=dict(size=11, color="#2a2a2a"),
tickformat=".1f", outlinewidth=0.5, outlinecolor="#cccccc",
),
)
)
fig.update_layout(
title=dict(text=title, font=dict(size=20, color="#1a1a1a"), x=0.5, xanchor="center", pad=dict(b=20)),
height=max(600, len(y) * 28 + 150) if not is_8x8 else 600,
autosize=True,
template="plotly_white",
xaxis=dict(
title=dict(text="Byte Position", font=dict(size=13, color="#1a1a1a")),
side="top" if not is_8x8 else "bottom",
dtick=1, showgrid=False, linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"), ticks="outside", ticklen=4, tickcolor="#cccccc",
),
yaxis=dict(
title=dict(text="PGN or CAN ID" if not is_8x8 else "Byte Position", font=dict(size=13, color="#1a1a1a")),
autorange="reversed", showgrid=False, linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"), ticks="outside", ticklen=4, tickcolor="#cccccc", automargin=True,
),
font=dict(family="Segoe UI, Arial, sans-serif", size=12, color="#2a2a2a"),
hoverlabel=dict(bgcolor="white", font_size=13, font_family="Segoe UI", bordercolor="#cccccc"),
margin=dict(l=200, r=40, t=120, b=60),
)
return fig
if __name__ == "__main__":
parser = argparse.ArgumentParser(description="Analyze CAN bus inter-byte correlation")
parser.add_argument("method", choices=["pearson", "spearman"], help="Correlation method to use")
parser.add_argument("input", type=Path, help="Path to the input CAN log file")
parser.add_argument("output", type=Path, nargs="?", default=Path("correlation_report.html"), help="Path to the output HTML report")
parser.add_argument("title", nargs="?", default="CAN Bus Inter-Byte Correlation", help="Title for the HTML report")
parser.add_argument("--identifier", type=str, default=None, help="Specific PGN/CAN ID to analyze (e.g., 'PGN: 65331'). If omitted, shows max correlation per byte for all IDs.")
args = parser.parse_args()
df = load_data(args.input)
corr_df = calculate_correlation(df, method=args.method, target_id=args.identifier)
display_title = f"{args.title} ({args.identifier})" if args.identifier else args.title
fig = plot_correlation_heatmap(corr_df, target_id=args.identifier, title=display_title)
config = {
"responsive": True,
"displaylogo": False,
"scrollZoom": True,
"modeBarButtonsToAdd": ["toggleSpikelines"],
"toImageButtonOptions": {"format": "png", "scale": 2},
}
fig.write_html(str(args.output), include_plotlyjs="cdn", config=config)
+180
View File
@@ -0,0 +1,180 @@
# File: entropy.py
# Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later
import argparse
from pathlib import Path
import numpy as np
import pandas as pd
import plotly.graph_objects as go
from utils.extractor import load_data
def _to_int(x):
"""Convert a hex string or integer to int, returning NaN on failure."""
if isinstance(x, (int, np.integer)):
return int(x)
if isinstance(x, str):
try:
return int(x, 16)
except ValueError:
return np.nan
return np.nan
def calculate_byte_entropy(df: pd.DataFrame) -> pd.DataFrame:
"""Calculates Shannon entropy per byte position for each identifier."""
byte_cols = [f"b{i}" for i in range(8)]
available_cols = [col for col in byte_cols if col in df.columns]
if not available_cols:
raise ValueError("No byte columns (b0-b7) found in the DataFrame")
df_bytes = df[available_cols].copy()
for col in available_cols:
df_bytes[col] = df_bytes[col].apply(_to_int)
def entropy(s: pd.Series) -> float:
s = s.dropna()
if s.empty:
return 0.0
p = s.value_counts(normalize=True)
return -np.sum(p * np.log2(p))
return df.groupby("Identifier")[available_cols].agg(entropy)
def plot_entropy_heatmap(entropy_df: pd.DataFrame, title: str) -> go.Figure:
"""Generates an interactive heatmap of byte-level Shannon entropy."""
x = entropy_df.columns.tolist()
y = entropy_df.index.tolist()
z = entropy_df.values
fig = go.Figure(
data=go.Heatmap(
z=z,
x=x,
y=y,
colorscale=[
[0.0, "#ffffff"],
[0.15, "#fff7ec"],
[0.35, "#fee8c8"],
[0.55, "#fdd49e"],
[0.75, "#fdbb84"],
[1.0, "#ef6548"],
],
xgap=3,
ygap=3,
text=np.round(z, 2),
texttemplate="%{text}",
textfont={
"size": 11,
"color": "#2a2a2a",
"family": "Segoe UI, Arial, sans-serif",
},
hoverongaps=False,
hovertemplate=(
"<b>%{y}</b><br>"
"Byte %{x}: %{z:.2f} bits<extra></extra>"
),
colorbar=dict(
title=dict(
text="Entropy (bits)",
side="top",
font=dict(size=13, color="#1a1a1a"),
),
orientation="h",
thickness=15,
len=0.35,
x=1.0,
xanchor="right",
y=1.02,
yanchor="bottom",
tickfont=dict(size=11, color="#2a2a2a"),
tickformat=".1f",
outlinewidth=0.5,
outlinecolor="#cccccc",
),
)
)
fig.update_layout(
title=dict(
text=title,
font=dict(size=20, color="#1a1a1a"),
x=0.5,
xanchor="center",
pad=dict(b=20),
),
height=max(600, len(y) * 28 + 150),
autosize=True,
template="plotly_white",
xaxis=dict(
title=dict(text="Byte Position", font=dict(size=13, color="#1a1a1a")),
side="top",
dtick=1,
showgrid=False,
linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"),
ticks="outside",
ticklen=4,
tickcolor="#cccccc",
),
yaxis=dict(
title=dict(text="PGN or CAN ID", font=dict(size=13, color="#1a1a1a")),
autorange="reversed",
showgrid=False,
linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"),
ticks="outside",
ticklen=4,
tickcolor="#cccccc",
automargin=True,
),
font=dict(family="Segoe UI, Arial, sans-serif", size=12, color="#2a2a2a"),
hoverlabel=dict(
bgcolor="white",
font_size=13,
font_family="Segoe UI",
bordercolor="#cccccc",
),
margin=dict(l=200, r=40, t=120, b=60),
)
return fig
if __name__ == "__main__":
parser = argparse.ArgumentParser(
description="Analyze CAN bus byte-level entropy"
)
parser.add_argument(
"input", type=Path, help="Path to the input CAN log file"
)
parser.add_argument(
"output",
type=Path,
nargs="?",
default=Path("entropy_report.html"),
help="Path to the output HTML report",
)
parser.add_argument(
"title",
nargs="?",
default="CAN Bus Byte-Level Entropy",
help="Title for the HTML report",
)
args = parser.parse_args()
df = load_data(args.input)
entropy_df = calculate_byte_entropy(df)
fig = plot_entropy_heatmap(entropy_df, title=args.title)
config = {
"responsive": True,
"displaylogo": False,
"scrollZoom": True,
"modeBarButtonsToAdd": ["toggleSpikelines"],
"toImageButtonOptions": {"format": "png", "scale": 2},
}
fig.write_html(str(args.output), include_plotlyjs="cdn", config=config)
+90 -50
View File
@@ -1,76 +1,116 @@
# File: frequency.py
# Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later
import argparse import argparse
import json
from pathlib import Path from pathlib import Path
import fastparquet
import pandas as pd import pandas as pd
import plotly.express as px import plotly.express as px
import plotly.graph_objects as go import plotly.graph_objects as go
from utils.extractor import load_data
def calc_freq(df: pd.DataFrame) -> pd.DataFrame:
def _extract_identifier(row: pd.Series) -> str:
"""Extracts PGN from metadata or falls back to CAN ID."""
meta = row.get('j1939_metadata')
if pd.isna(meta):
return f"{row['ID']} "
if isinstance(meta, str):
try:
meta = json.loads(meta)
except json.JSONDecodeError:
return f"{row['ID']}"
if isinstance(meta, dict) and 'PGN' in meta:
return f"PGN: {meta['PGN']} "
return f"{row['ID']}"
def load_data(file_path: Path) -> pd.DataFrame:
"""Loads Parquet file and adds an Identifier column."""
df = pd.read_parquet(file_path)
df['Identifier'] = df.apply(_extract_identifier, axis=1)
return df
def calculate_frequency(df: pd.DataFrame) -> pd.DataFrame:
"""Calculates frequency counts and percentages for identifiers.""" """Calculates frequency counts and percentages for identifiers."""
freq_df = df['Identifier'].value_counts().reset_index() freq_df = df['Identifier'].value_counts().reset_index()
freq_df.columns = ['Identifier', 'Count'] freq_df.columns = ['Identifier', 'Count']
total_messages = freq_df['Count'].sum() total = freq_df['Count'].sum()
freq_df['Percentage'] = (freq_df['Count'] / total_messages * 100).round(2) freq_df['Percentage'] = (freq_df['Count'] / total * 100).round(2)
return freq_df.sort_values('Count', ascending=True) return freq_df.sort_values('Count', ascending=True)
def plot_freq(stats_df: pd.DataFrame, title: str) -> go.Figure:
def visualize_frequency(stats_df: pd.DataFrame, title: str = "CAN Bus Message Frequency") -> go.Figure: """Generates interactive horizontal bar chart with log x-axis."""
"""Generates an interactive Plotly horizontal bar chart with a logarithmic x-axis."""
fig = px.bar( fig = px.bar(
stats_df, y='Identifier', x='Count', orientation='h', title=title, log_x=True, stats_df, y='Identifier', x='Count', orientation='h', title=title, log_x=True,
labels={'Identifier': 'PGN / CAN ID', 'Count': 'Message Count'}, labels={'Identifier': 'PGN / CAN ID', 'Count': 'Message Count'},
color='Count', color_continuous_scale='Turbo', color='Count', color_continuous_scale='Turbo',
hover_data={'Percentage': ':.2f', 'Count': True, 'Identifier': True} range_color=(stats_df['Count'].min(), stats_df['Count'].max()),
hover_data={'Percentage': ':.2f', 'Count': ':,', 'Identifier': True}
) )
fig.update_layout( fig.update_layout(
height=max(600, len(stats_df) * 18), width=1000, height=max(600, len(stats_df) * 18),
xaxis_title='Total Message Count (Log Scale)', yaxis_title='PGN or CAN ID', autosize=True,
yaxis={'categoryorder': 'total ascending'}, template='plotly_white',
plot_bgcolor='rgba(0,0,0,0)', paper_bgcolor='white', xaxis=dict(
font=dict(family="Segoe UI, Arial, sans-serif", size=12), title=dict(text="Message count [log scale]", font=dict(size=13, color="#1a1a1a")),
hoverlabel=dict(bgcolor="white", font_size=13, font_family="Segoe UI"), side="top",
margin=dict(l=250, r=50, t=80, b=50), dtick=1,
title=dict(font=dict(size=20), x=0.5) showgrid=False,
linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"),
ticks="outside",
ticklen=4,
tickcolor="#cccccc",
),
yaxis=dict(
title=dict(text="PGN or CAN ID", font=dict(size=13, color="#1a1a1a")),
#autorange="",
showgrid=False,
linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"),
ticks="outside",
ticklen=4,
tickcolor="#cccccc",
automargin=True,
),
font=dict(family="Segoe UI, Arial, sans-serif", size=12, color='#2a2a2a'),
hoverlabel=dict(bgcolor="white", font_size=13, font_family="Segoe UI",
bordercolor='#cccccc'),
margin=dict(l=200, r=40, t=120, b=60),
bargap=0.35,
coloraxis_colorbar=dict(
title=dict(text='Message Count', side='top'),
orientation='h',
thickness=15,
len=0.35,
x=1.0,
xanchor='right',
y=1.02,
yanchor='bottom',
tickformat=',',
outlinecolor='#cccccc',
outlinewidth=0.5
),
title=dict(font=dict(size=20, color='#1a1a1a'), x=0.5, xanchor='center',
pad=dict(b=20))
)
fig.update_xaxes(
showgrid=True, gridwidth=0.5, gridcolor='#e8e8e8',
zeroline=False, linecolor='#bdbdbd', mirror=False,
tickformat=',',
minor=dict(showgrid=True, gridcolor='#f4f4f4', gridwidth=0.5)
)
fig.update_yaxes(
showgrid=False, zeroline=False, linecolor='#bdbdbd',
ticks='outside', ticklen=4, tickcolor='#cccccc',
automargin=True
) )
fig.update_traces( fig.update_traces(
hovertemplate="<b>%{y}</b><br>Count: %{x:,}<br>Share: %{customdata[0]}%<extra></extra>" hovertemplate="<b>%{y}</b><br>Count: %{x:,}<br>Share: %{customdata[0]}%<extra></extra>",
marker_line_width=0,
texttemplate='%{x:,}',
textposition='outside',
textfont=dict(size=10, color='#666666'),
cliponaxis=False,
selected=dict(marker=dict(opacity=0.6)),
unselected=dict(marker=dict(opacity=0.2))
) )
return fig return fig
if __name__ == "__main__": if __name__ == "__main__":
parser = argparse.ArgumentParser(description="Analyze CAN bus Parquet data.") parser = argparse.ArgumentParser(description="Analyze CAN bus message frequency")
parser.add_argument("-i", "--input", type=Path, required=True) parser.add_argument("input", type=Path, help="Path to the input CAN log file")
parser.add_argument("-o", "--output", type=Path, default=Path("can_analysis_report.html")) parser.add_argument("output", type=Path, nargs="?", default=Path("freq_report.html"), help="Path to the output HTML report")
parser.add_argument("-t", "--title", type=str, default="CAN Bus Message Frequency by PGN / ID") parser.add_argument("title", nargs="?", default="CAN Bus Message Frequency", help="Title for the HTML report")
args = parser.parse_args() args = parser.parse_args()
df = load_data(args.input) df = load_data(args.input)
stats_df = calculate_frequency(df) stats = calc_freq(df)
fig = visualize_frequency(stats_df, title=args.title) fig = plot_freq(stats, title=args.title)
fig.write_html(str(args.output), include_plotlyjs='cdn') config = {
'responsive': True,
'displaylogo': False,
'scrollZoom': True,
'modeBarButtonsToAdd': ['toggleSpikelines'],
'toImageButtonOptions': {'format': 'png', 'scale': 2}
}
fig.write_html(str(args.output), include_plotlyjs='cdn', config=config)
+36
View File
@@ -0,0 +1,36 @@
import json
from pathlib import Path
import numpy as np
import pandas as pd
def to_int(x):
"""Convert a hex string or integer to int, returning NaN on failure."""
if isinstance(x, (int, np.integer)):
return int(x)
if isinstance(x, str):
try:
return int(x, 16)
except ValueError:
return np.nan
return np.nan
def extract_id(row: pd.Series) -> str:
"""Extracts PGN from metadata or falls back to CAN ID."""
meta = row.get('j1939_metadata')
if pd.isna(meta):
return f"ID: {row['ID']}"
if isinstance(meta, str):
try:
meta = json.loads(meta)
except json.JSONDecodeError:
return f"ID: {row['ID']}"
if isinstance(meta, dict) and 'PGN' in meta:
return f"PGN: {meta['PGN']}"
return f"ID: {row['ID']}"
def load_data(file_path: Path) -> pd.DataFrame:
"""Loads Parquet file and adds an Identifier column."""
df = pd.read_parquet(file_path)
df['Identifier'] = df.apply(extract_id, axis=1)
return df