From ad8334c915ff9f35cbd4a9717a4810042c8e722a Mon Sep 17 00:00:00 2001 From: Erick Ahmed Date: Mon, 13 Jul 2026 19:48:46 +0200 Subject: [PATCH 1/7] Add header with copyright and SPDX licensing information --- decoder.py | 4 ++++ main.py | 3 +++ parser.py | 4 ++++ stat/frequency.py | 4 ++++ 4 files changed, 15 insertions(+) diff --git a/decoder.py b/decoder.py index 84c6ea1..f8341ec 100644 --- a/decoder.py +++ b/decoder.py @@ -1,3 +1,7 @@ +# File: decoder.py +# Copyright (C) 2026 Erick Ahmed +# SPDX-License-Identifier: AGPL-3.0-or-later + import argparse import polars as pl diff --git a/main.py b/main.py index e69de29..3735861 100644 --- a/main.py +++ b/main.py @@ -0,0 +1,3 @@ +# File: main.py +# Copyright (C) 2026 Erick Ahmed +# SPDX-License-Identifier: AGPL-3.0-or-later diff --git a/parser.py b/parser.py index 5cd641f..b219036 100644 --- a/parser.py +++ b/parser.py @@ -1,3 +1,7 @@ +# File: parser.py +# Copyright (C) 2026 Erick Ahmed +# SPDX-License-Identifier: AGPL-3.0-or-later + import re import csv import polars as pl diff --git a/stat/frequency.py b/stat/frequency.py index 48f5cb3..9766cbe 100644 --- a/stat/frequency.py +++ b/stat/frequency.py @@ -1,3 +1,7 @@ +# File: frequency.py +# Copyright (C) 2026 Erick Ahmed +# SPDX-License-Identifier: AGPL-3.0-or-later + import argparse import json from pathlib import Path From 17340746ce1a392c10b29dd9c94e235b8ca0658f Mon Sep 17 00:00:00 2001 From: Erick Ahmed Date: Mon, 13 Jul 2026 19:51:22 +0200 Subject: [PATCH 2/7] Move data extraction functions to utility library --- stat/frequency.py | 53 +++++++++++----------------------------------- utils/extractor.py | 27 +++++++++++++++++++++++ 2 files changed, 39 insertions(+), 41 deletions(-) create mode 100644 utils/extractor.py diff --git a/stat/frequency.py b/stat/frequency.py index 48f5cb3..c4fd929 100644 --- a/stat/frequency.py +++ b/stat/frequency.py @@ -1,52 +1,26 @@ import argparse -import json from pathlib import Path -import fastparquet import pandas as pd import plotly.express as px import plotly.graph_objects as go +from utils.extractor import load_data - -def _extract_identifier(row: pd.Series) -> str: - """Extracts PGN from metadata or falls back to CAN ID.""" - meta = row.get('j1939_metadata') - if pd.isna(meta): - return f"{row['ID']} " - if isinstance(meta, str): - try: - meta = json.loads(meta) - except json.JSONDecodeError: - return f"{row['ID']}" - if isinstance(meta, dict) and 'PGN' in meta: - return f"PGN: {meta['PGN']} " - return f"{row['ID']}" - - -def load_data(file_path: Path) -> pd.DataFrame: - """Loads Parquet file and adds an Identifier column.""" - df = pd.read_parquet(file_path) - df['Identifier'] = df.apply(_extract_identifier, axis=1) - return df - - -def calculate_frequency(df: pd.DataFrame) -> pd.DataFrame: +def calc_freq(df: pd.DataFrame) -> pd.DataFrame: """Calculates frequency counts and percentages for identifiers.""" freq_df = df['Identifier'].value_counts().reset_index() freq_df.columns = ['Identifier', 'Count'] - total_messages = freq_df['Count'].sum() - freq_df['Percentage'] = (freq_df['Count'] / total_messages * 100).round(2) + total = freq_df['Count'].sum() + freq_df['Percentage'] = (freq_df['Count'] / total * 100).round(2) return freq_df.sort_values('Count', ascending=True) - -def visualize_frequency(stats_df: pd.DataFrame, title: str = "CAN Bus Message Frequency") -> go.Figure: - """Generates an interactive Plotly horizontal bar chart with a logarithmic x-axis.""" +def plot_freq(stats_df: pd.DataFrame, title: str) -> go.Figure: + """Generates interactive horizontal bar chart with log x-axis.""" fig = px.bar( stats_df, y='Identifier', x='Count', orientation='h', title=title, log_x=True, labels={'Identifier': 'PGN / CAN ID', 'Count': 'Message Count'}, color='Count', color_continuous_scale='Turbo', hover_data={'Percentage': ':.2f', 'Count': True, 'Identifier': True} ) - fig.update_layout( height=max(600, len(stats_df) * 18), width=1000, xaxis_title='Total Message Count (Log Scale)', yaxis_title='PGN or CAN ID', @@ -57,20 +31,17 @@ def visualize_frequency(stats_df: pd.DataFrame, title: str = "CAN Bus Message Fr margin=dict(l=250, r=50, t=80, b=50), title=dict(font=dict(size=20), x=0.5) ) - - fig.update_traces( - hovertemplate="%{y}
Count: %{x:,}
Share: %{customdata[0]}%" - ) + fig.update_traces(hovertemplate="%{y}
Count: %{x:,}
Share: %{customdata[0]}%") return fig if __name__ == "__main__": - parser = argparse.ArgumentParser(description="Analyze CAN bus Parquet data.") + parser = argparse.ArgumentParser(description="Analyze CAN bus frequency.") parser.add_argument("-i", "--input", type=Path, required=True) - parser.add_argument("-o", "--output", type=Path, default=Path("can_analysis_report.html")) - parser.add_argument("-t", "--title", type=str, default="CAN Bus Message Frequency by PGN / ID") + parser.add_argument("-o", "--output", type=Path, default=Path("freq_report.html")) + parser.add_argument("-t", "--title", type=str, default="CAN Bus Message Frequency") args = parser.parse_args() df = load_data(args.input) - stats_df = calculate_frequency(df) - fig = visualize_frequency(stats_df, title=args.title) + stats = calc_freq(df) + fig = plot_freq(stats, title=args.title) fig.write_html(str(args.output), include_plotlyjs='cdn') diff --git a/utils/extractor.py b/utils/extractor.py new file mode 100644 index 0000000..8cdd07e --- /dev/null +++ b/utils/extractor.py @@ -0,0 +1,27 @@ +# File: extractor.py +# Copyright (C) 2026 Erick Ahmed +# SPDX-License-Identifier: AGPL-3.0-or-later + +import json +from pathlib import Path +import pandas as pd + +def extract_id(row: pd.Series) -> str: + """Extracts PGN from metadata or falls back to CAN ID.""" + meta = row.get('j1939_metadata') + if pd.isna(meta): + return f"ID: {row['ID']}" + if isinstance(meta, str): + try: + meta = json.loads(meta) + except json.JSONDecodeError: + return f"ID: {row['ID']}" + if isinstance(meta, dict) and 'PGN' in meta: + return f"PGN: {meta['PGN']}" + return f"ID: {row['ID']}" + +def load_data(file_path: Path) -> pd.DataFrame: + """Loads Parquet file and adds an Identifier column.""" + df = pd.read_parquet(file_path) + df['Identifier'] = df.apply(extract_id, axis=1) + return df From 13d91d566ba87755b995488d229b6198189e9b1e Mon Sep 17 00:00:00 2001 From: Erick Ahmed Date: Mon, 13 Jul 2026 20:06:45 +0200 Subject: [PATCH 3/7] Move to subfolder - Makes python recognize it as a submodule --- stat/utils/extractor.py | 23 +++++++++++++++++++++++ 1 file changed, 23 insertions(+) create mode 100644 stat/utils/extractor.py diff --git a/stat/utils/extractor.py b/stat/utils/extractor.py new file mode 100644 index 0000000..a5dd172 --- /dev/null +++ b/stat/utils/extractor.py @@ -0,0 +1,23 @@ +import json +from pathlib import Path +import pandas as pd + +def extract_id(row: pd.Series) -> str: + """Extracts PGN from metadata or falls back to CAN ID.""" + meta = row.get('j1939_metadata') + if pd.isna(meta): + return f"ID: {row['ID']}" + if isinstance(meta, str): + try: + meta = json.loads(meta) + except json.JSONDecodeError: + return f"ID: {row['ID']}" + if isinstance(meta, dict) and 'PGN' in meta: + return f"PGN: {meta['PGN']}" + return f"ID: {row['ID']}" + +def load_data(file_path: Path) -> pd.DataFrame: + """Loads Parquet file and adds an Identifier column.""" + df = pd.read_parquet(file_path) + df['Identifier'] = df.apply(extract_id, axis=1) + return df From 8ccb97afae5e1d6807a686266f9d9564d0d3293d Mon Sep 17 00:00:00 2001 From: Erick Ahmed Date: Mon, 13 Jul 2026 20:07:10 +0200 Subject: [PATCH 4/7] Remove requirement for positional arguments --- stat/frequency.py | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/stat/frequency.py b/stat/frequency.py index d92c8fc..5633799 100644 --- a/stat/frequency.py +++ b/stat/frequency.py @@ -39,10 +39,10 @@ def plot_freq(stats_df: pd.DataFrame, title: str) -> go.Figure: return fig if __name__ == "__main__": - parser = argparse.ArgumentParser(description="Analyze CAN bus frequency.") - parser.add_argument("-i", "--input", type=Path, required=True) - parser.add_argument("-o", "--output", type=Path, default=Path("freq_report.html")) - parser.add_argument("-t", "--title", type=str, default="CAN Bus Message Frequency") + parser = argparse.ArgumentParser(description="Analyze CAN bus message frequency") + parser.add_argument("input", type=Path, help="Path to the input CAN log file") + parser.add_argument("output", type=Path, nargs="?", default=Path("freq_report.html"), help="Path to the output HTML report") + parser.add_argument("title", nargs="?", default="CAN Bus Message Frequency", help="Title for the HTML report") args = parser.parse_args() df = load_data(args.input) From db7e5eb40edf0d41a2c149e371b61ae109f5dbc5 Mon Sep 17 00:00:00 2001 From: Erick Ahmed Date: Mon, 13 Jul 2026 20:08:29 +0200 Subject: [PATCH 5/7] Add header informations --- stat/utils/extractor.py | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/stat/utils/extractor.py b/stat/utils/extractor.py index a5dd172..8cdd07e 100644 --- a/stat/utils/extractor.py +++ b/stat/utils/extractor.py @@ -1,3 +1,7 @@ +# File: extractor.py +# Copyright (C) 2026 Erick Ahmed +# SPDX-License-Identifier: AGPL-3.0-or-later + import json from pathlib import Path import pandas as pd From 8a3fc81d12b9fe0b18671344930471cd95371e3f Mon Sep 17 00:00:00 2001 From: Erick Ahmed Date: Mon, 13 Jul 2026 21:15:36 +0200 Subject: [PATCH 6/7] Use more consistent styling - X axis on top - Refactor figure generation logic --- stat/frequency.py | 87 +++++++++++++++++++++++++++++++++++++++++------ 1 file changed, 76 insertions(+), 11 deletions(-) diff --git a/stat/frequency.py b/stat/frequency.py index 5633799..b3e185a 100644 --- a/stat/frequency.py +++ b/stat/frequency.py @@ -23,19 +23,77 @@ def plot_freq(stats_df: pd.DataFrame, title: str) -> go.Figure: stats_df, y='Identifier', x='Count', orientation='h', title=title, log_x=True, labels={'Identifier': 'PGN / CAN ID', 'Count': 'Message Count'}, color='Count', color_continuous_scale='Turbo', - hover_data={'Percentage': ':.2f', 'Count': True, 'Identifier': True} + range_color=(stats_df['Count'].min(), stats_df['Count'].max()), + hover_data={'Percentage': ':.2f', 'Count': ':,', 'Identifier': True} ) fig.update_layout( - height=max(600, len(stats_df) * 18), width=1000, - xaxis_title='Total Message Count (Log Scale)', yaxis_title='PGN or CAN ID', - yaxis={'categoryorder': 'total ascending'}, - plot_bgcolor='rgba(0,0,0,0)', paper_bgcolor='white', - font=dict(family="Segoe UI, Arial, sans-serif", size=12), - hoverlabel=dict(bgcolor="white", font_size=13, font_family="Segoe UI"), - margin=dict(l=250, r=50, t=80, b=50), - title=dict(font=dict(size=20), x=0.5) + height=max(600, len(stats_df) * 18), + autosize=True, + template='plotly_white', + xaxis=dict( + title=dict(text="Message count [log scale]", font=dict(size=13, color="#1a1a1a")), + side="top", + dtick=1, + showgrid=False, + linecolor="#bdbdbd", + tickfont=dict(size=12, color="#2a2a2a"), + ticks="outside", + ticklen=4, + tickcolor="#cccccc", + ), + yaxis=dict( + title=dict(text="PGN or CAN ID", font=dict(size=13, color="#1a1a1a")), + #autorange="", + showgrid=False, + linecolor="#bdbdbd", + tickfont=dict(size=12, color="#2a2a2a"), + ticks="outside", + ticklen=4, + tickcolor="#cccccc", + automargin=True, + ), + font=dict(family="Segoe UI, Arial, sans-serif", size=12, color='#2a2a2a'), + hoverlabel=dict(bgcolor="white", font_size=13, font_family="Segoe UI", + bordercolor='#cccccc'), + margin=dict(l=200, r=40, t=120, b=60), + bargap=0.35, + coloraxis_colorbar=dict( + title=dict(text='Message Count', side='top'), + orientation='h', + thickness=15, + len=0.35, + x=1.0, + xanchor='right', + y=1.02, + yanchor='bottom', + tickformat=',', + outlinecolor='#cccccc', + outlinewidth=0.5 + ), + title=dict(font=dict(size=20, color='#1a1a1a'), x=0.5, xanchor='center', + pad=dict(b=20)) + ) + fig.update_xaxes( + showgrid=True, gridwidth=0.5, gridcolor='#e8e8e8', + zeroline=False, linecolor='#bdbdbd', mirror=False, + tickformat=',', + minor=dict(showgrid=True, gridcolor='#f4f4f4', gridwidth=0.5) + ) + fig.update_yaxes( + showgrid=False, zeroline=False, linecolor='#bdbdbd', + ticks='outside', ticklen=4, tickcolor='#cccccc', + automargin=True + ) + fig.update_traces( + hovertemplate="%{y}
Count: %{x:,}
Share: %{customdata[0]}%", + marker_line_width=0, + texttemplate='%{x:,}', + textposition='outside', + textfont=dict(size=10, color='#666666'), + cliponaxis=False, + selected=dict(marker=dict(opacity=0.6)), + unselected=dict(marker=dict(opacity=0.2)) ) - fig.update_traces(hovertemplate="%{y}
Count: %{x:,}
Share: %{customdata[0]}%") return fig if __name__ == "__main__": @@ -48,4 +106,11 @@ if __name__ == "__main__": df = load_data(args.input) stats = calc_freq(df) fig = plot_freq(stats, title=args.title) - fig.write_html(str(args.output), include_plotlyjs='cdn') + config = { + 'responsive': True, + 'displaylogo': False, + 'scrollZoom': True, + 'modeBarButtonsToAdd': ['toggleSpikelines'], + 'toImageButtonOptions': {'format': 'png', 'scale': 2} + } + fig.write_html(str(args.output), include_plotlyjs='cdn', config=config) From 7daf4c8e088bb2fea675ec8478a8d4e77101c802 Mon Sep 17 00:00:00 2001 From: Erick Ahmed Date: Mon, 13 Jul 2026 21:16:18 +0200 Subject: [PATCH 7/7] Add entropy heatmap analysis - Same style of frequency analysis for consistency --- stat/entropy.py | 180 ++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 180 insertions(+) create mode 100644 stat/entropy.py diff --git a/stat/entropy.py b/stat/entropy.py new file mode 100644 index 0000000..ac19a22 --- /dev/null +++ b/stat/entropy.py @@ -0,0 +1,180 @@ +# File: entropy.py +# Copyright (C) 2026 Erick Ahmed +# SPDX-License-Identifier: AGPL-3.0-or-later + +import argparse +from pathlib import Path + +import numpy as np +import pandas as pd +import plotly.graph_objects as go + +from utils.extractor import load_data + + +def _to_int(x): + """Convert a hex string or integer to int, returning NaN on failure.""" + if isinstance(x, (int, np.integer)): + return int(x) + if isinstance(x, str): + try: + return int(x, 16) + except ValueError: + return np.nan + return np.nan + + +def calculate_byte_entropy(df: pd.DataFrame) -> pd.DataFrame: + """Calculates Shannon entropy per byte position for each identifier.""" + byte_cols = [f"b{i}" for i in range(8)] + available_cols = [col for col in byte_cols if col in df.columns] + if not available_cols: + raise ValueError("No byte columns (b0-b7) found in the DataFrame") + + df_bytes = df[available_cols].copy() + for col in available_cols: + df_bytes[col] = df_bytes[col].apply(_to_int) + + def entropy(s: pd.Series) -> float: + s = s.dropna() + if s.empty: + return 0.0 + p = s.value_counts(normalize=True) + return -np.sum(p * np.log2(p)) + + return df.groupby("Identifier")[available_cols].agg(entropy) + + +def plot_entropy_heatmap(entropy_df: pd.DataFrame, title: str) -> go.Figure: + """Generates an interactive heatmap of byte-level Shannon entropy.""" + x = entropy_df.columns.tolist() + y = entropy_df.index.tolist() + z = entropy_df.values + + fig = go.Figure( + data=go.Heatmap( + z=z, + x=x, + y=y, + colorscale=[ + [0.0, "#ffffff"], + [0.15, "#fff7ec"], + [0.35, "#fee8c8"], + [0.55, "#fdd49e"], + [0.75, "#fdbb84"], + [1.0, "#ef6548"], + ], + xgap=3, + ygap=3, + text=np.round(z, 2), + texttemplate="%{text}", + textfont={ + "size": 11, + "color": "#2a2a2a", + "family": "Segoe UI, Arial, sans-serif", + }, + hoverongaps=False, + hovertemplate=( + "%{y}
" + "Byte %{x}: %{z:.2f} bits" + ), + colorbar=dict( + title=dict( + text="Entropy (bits)", + side="top", + font=dict(size=13, color="#1a1a1a"), + ), + orientation="h", + thickness=15, + len=0.35, + x=1.0, + xanchor="right", + y=1.02, + yanchor="bottom", + tickfont=dict(size=11, color="#2a2a2a"), + tickformat=".1f", + outlinewidth=0.5, + outlinecolor="#cccccc", + ), + ) + ) + + fig.update_layout( + title=dict( + text=title, + font=dict(size=20, color="#1a1a1a"), + x=0.5, + xanchor="center", + pad=dict(b=20), + ), + height=max(600, len(y) * 28 + 150), + autosize=True, + template="plotly_white", + xaxis=dict( + title=dict(text="Byte Position", font=dict(size=13, color="#1a1a1a")), + side="top", + dtick=1, + showgrid=False, + linecolor="#bdbdbd", + tickfont=dict(size=12, color="#2a2a2a"), + ticks="outside", + ticklen=4, + tickcolor="#cccccc", + ), + yaxis=dict( + title=dict(text="PGN or CAN ID", font=dict(size=13, color="#1a1a1a")), + autorange="reversed", + showgrid=False, + linecolor="#bdbdbd", + tickfont=dict(size=12, color="#2a2a2a"), + ticks="outside", + ticklen=4, + tickcolor="#cccccc", + automargin=True, + ), + font=dict(family="Segoe UI, Arial, sans-serif", size=12, color="#2a2a2a"), + hoverlabel=dict( + bgcolor="white", + font_size=13, + font_family="Segoe UI", + bordercolor="#cccccc", + ), + margin=dict(l=200, r=40, t=120, b=60), + ) + return fig + + +if __name__ == "__main__": + parser = argparse.ArgumentParser( + description="Analyze CAN bus byte-level entropy" + ) + parser.add_argument( + "input", type=Path, help="Path to the input CAN log file" + ) + parser.add_argument( + "output", + type=Path, + nargs="?", + default=Path("entropy_report.html"), + help="Path to the output HTML report", + ) + parser.add_argument( + "title", + nargs="?", + default="CAN Bus Byte-Level Entropy", + help="Title for the HTML report", + ) + args = parser.parse_args() + + df = load_data(args.input) + entropy_df = calculate_byte_entropy(df) + fig = plot_entropy_heatmap(entropy_df, title=args.title) + + config = { + "responsive": True, + "displaylogo": False, + "scrollZoom": True, + "modeBarButtonsToAdd": ["toggleSpikelines"], + "toImageButtonOptions": {"format": "png", "scale": 2}, + } + fig.write_html(str(args.output), include_plotlyjs="cdn", config=config)