diff --git a/decoder.py b/decoder.py
index 84c6ea1..f8341ec 100644
--- a/decoder.py
+++ b/decoder.py
@@ -1,3 +1,7 @@
+# File: decoder.py
+# Copyright (C) 2026 Erick Ahmed
+# SPDX-License-Identifier: AGPL-3.0-or-later
+
import argparse
import polars as pl
diff --git a/main.py b/main.py
index e69de29..3735861 100644
--- a/main.py
+++ b/main.py
@@ -0,0 +1,3 @@
+# File: main.py
+# Copyright (C) 2026 Erick Ahmed
+# SPDX-License-Identifier: AGPL-3.0-or-later
diff --git a/parser.py b/parser.py
index 5cd641f..b219036 100644
--- a/parser.py
+++ b/parser.py
@@ -1,3 +1,7 @@
+# File: parser.py
+# Copyright (C) 2026 Erick Ahmed
+# SPDX-License-Identifier: AGPL-3.0-or-later
+
import re
import csv
import polars as pl
diff --git a/stat/entropy.py b/stat/entropy.py
new file mode 100644
index 0000000..ac19a22
--- /dev/null
+++ b/stat/entropy.py
@@ -0,0 +1,180 @@
+# File: entropy.py
+# Copyright (C) 2026 Erick Ahmed
+# SPDX-License-Identifier: AGPL-3.0-or-later
+
+import argparse
+from pathlib import Path
+
+import numpy as np
+import pandas as pd
+import plotly.graph_objects as go
+
+from utils.extractor import load_data
+
+
+def _to_int(x):
+ """Convert a hex string or integer to int, returning NaN on failure."""
+ if isinstance(x, (int, np.integer)):
+ return int(x)
+ if isinstance(x, str):
+ try:
+ return int(x, 16)
+ except ValueError:
+ return np.nan
+ return np.nan
+
+
+def calculate_byte_entropy(df: pd.DataFrame) -> pd.DataFrame:
+ """Calculates Shannon entropy per byte position for each identifier."""
+ byte_cols = [f"b{i}" for i in range(8)]
+ available_cols = [col for col in byte_cols if col in df.columns]
+ if not available_cols:
+ raise ValueError("No byte columns (b0-b7) found in the DataFrame")
+
+ df_bytes = df[available_cols].copy()
+ for col in available_cols:
+ df_bytes[col] = df_bytes[col].apply(_to_int)
+
+ def entropy(s: pd.Series) -> float:
+ s = s.dropna()
+ if s.empty:
+ return 0.0
+ p = s.value_counts(normalize=True)
+ return -np.sum(p * np.log2(p))
+
+ return df.groupby("Identifier")[available_cols].agg(entropy)
+
+
+def plot_entropy_heatmap(entropy_df: pd.DataFrame, title: str) -> go.Figure:
+ """Generates an interactive heatmap of byte-level Shannon entropy."""
+ x = entropy_df.columns.tolist()
+ y = entropy_df.index.tolist()
+ z = entropy_df.values
+
+ fig = go.Figure(
+ data=go.Heatmap(
+ z=z,
+ x=x,
+ y=y,
+ colorscale=[
+ [0.0, "#ffffff"],
+ [0.15, "#fff7ec"],
+ [0.35, "#fee8c8"],
+ [0.55, "#fdd49e"],
+ [0.75, "#fdbb84"],
+ [1.0, "#ef6548"],
+ ],
+ xgap=3,
+ ygap=3,
+ text=np.round(z, 2),
+ texttemplate="%{text}",
+ textfont={
+ "size": 11,
+ "color": "#2a2a2a",
+ "family": "Segoe UI, Arial, sans-serif",
+ },
+ hoverongaps=False,
+ hovertemplate=(
+ "%{y}
"
+ "Byte %{x}: %{z:.2f} bits"
+ ),
+ colorbar=dict(
+ title=dict(
+ text="Entropy (bits)",
+ side="top",
+ font=dict(size=13, color="#1a1a1a"),
+ ),
+ orientation="h",
+ thickness=15,
+ len=0.35,
+ x=1.0,
+ xanchor="right",
+ y=1.02,
+ yanchor="bottom",
+ tickfont=dict(size=11, color="#2a2a2a"),
+ tickformat=".1f",
+ outlinewidth=0.5,
+ outlinecolor="#cccccc",
+ ),
+ )
+ )
+
+ fig.update_layout(
+ title=dict(
+ text=title,
+ font=dict(size=20, color="#1a1a1a"),
+ x=0.5,
+ xanchor="center",
+ pad=dict(b=20),
+ ),
+ height=max(600, len(y) * 28 + 150),
+ autosize=True,
+ template="plotly_white",
+ xaxis=dict(
+ title=dict(text="Byte Position", font=dict(size=13, color="#1a1a1a")),
+ side="top",
+ dtick=1,
+ showgrid=False,
+ linecolor="#bdbdbd",
+ tickfont=dict(size=12, color="#2a2a2a"),
+ ticks="outside",
+ ticklen=4,
+ tickcolor="#cccccc",
+ ),
+ yaxis=dict(
+ title=dict(text="PGN or CAN ID", font=dict(size=13, color="#1a1a1a")),
+ autorange="reversed",
+ showgrid=False,
+ linecolor="#bdbdbd",
+ tickfont=dict(size=12, color="#2a2a2a"),
+ ticks="outside",
+ ticklen=4,
+ tickcolor="#cccccc",
+ automargin=True,
+ ),
+ font=dict(family="Segoe UI, Arial, sans-serif", size=12, color="#2a2a2a"),
+ hoverlabel=dict(
+ bgcolor="white",
+ font_size=13,
+ font_family="Segoe UI",
+ bordercolor="#cccccc",
+ ),
+ margin=dict(l=200, r=40, t=120, b=60),
+ )
+ return fig
+
+
+if __name__ == "__main__":
+ parser = argparse.ArgumentParser(
+ description="Analyze CAN bus byte-level entropy"
+ )
+ parser.add_argument(
+ "input", type=Path, help="Path to the input CAN log file"
+ )
+ parser.add_argument(
+ "output",
+ type=Path,
+ nargs="?",
+ default=Path("entropy_report.html"),
+ help="Path to the output HTML report",
+ )
+ parser.add_argument(
+ "title",
+ nargs="?",
+ default="CAN Bus Byte-Level Entropy",
+ help="Title for the HTML report",
+ )
+ args = parser.parse_args()
+
+ df = load_data(args.input)
+ entropy_df = calculate_byte_entropy(df)
+ fig = plot_entropy_heatmap(entropy_df, title=args.title)
+
+ config = {
+ "responsive": True,
+ "displaylogo": False,
+ "scrollZoom": True,
+ "modeBarButtonsToAdd": ["toggleSpikelines"],
+ "toImageButtonOptions": {"format": "png", "scale": 2},
+ }
+ fig.write_html(str(args.output), include_plotlyjs="cdn", config=config)
diff --git a/stat/frequency.py b/stat/frequency.py
index 48f5cb3..b3e185a 100644
--- a/stat/frequency.py
+++ b/stat/frequency.py
@@ -1,76 +1,116 @@
+# File: frequency.py
+# Copyright (C) 2026 Erick Ahmed
+# SPDX-License-Identifier: AGPL-3.0-or-later
+
import argparse
-import json
from pathlib import Path
-import fastparquet
import pandas as pd
import plotly.express as px
import plotly.graph_objects as go
+from utils.extractor import load_data
-
-def _extract_identifier(row: pd.Series) -> str:
- """Extracts PGN from metadata or falls back to CAN ID."""
- meta = row.get('j1939_metadata')
- if pd.isna(meta):
- return f"{row['ID']} "
- if isinstance(meta, str):
- try:
- meta = json.loads(meta)
- except json.JSONDecodeError:
- return f"{row['ID']}"
- if isinstance(meta, dict) and 'PGN' in meta:
- return f"PGN: {meta['PGN']} "
- return f"{row['ID']}"
-
-
-def load_data(file_path: Path) -> pd.DataFrame:
- """Loads Parquet file and adds an Identifier column."""
- df = pd.read_parquet(file_path)
- df['Identifier'] = df.apply(_extract_identifier, axis=1)
- return df
-
-
-def calculate_frequency(df: pd.DataFrame) -> pd.DataFrame:
+def calc_freq(df: pd.DataFrame) -> pd.DataFrame:
"""Calculates frequency counts and percentages for identifiers."""
freq_df = df['Identifier'].value_counts().reset_index()
freq_df.columns = ['Identifier', 'Count']
- total_messages = freq_df['Count'].sum()
- freq_df['Percentage'] = (freq_df['Count'] / total_messages * 100).round(2)
+ total = freq_df['Count'].sum()
+ freq_df['Percentage'] = (freq_df['Count'] / total * 100).round(2)
return freq_df.sort_values('Count', ascending=True)
-
-def visualize_frequency(stats_df: pd.DataFrame, title: str = "CAN Bus Message Frequency") -> go.Figure:
- """Generates an interactive Plotly horizontal bar chart with a logarithmic x-axis."""
+def plot_freq(stats_df: pd.DataFrame, title: str) -> go.Figure:
+ """Generates interactive horizontal bar chart with log x-axis."""
fig = px.bar(
stats_df, y='Identifier', x='Count', orientation='h', title=title, log_x=True,
labels={'Identifier': 'PGN / CAN ID', 'Count': 'Message Count'},
color='Count', color_continuous_scale='Turbo',
- hover_data={'Percentage': ':.2f', 'Count': True, 'Identifier': True}
+ range_color=(stats_df['Count'].min(), stats_df['Count'].max()),
+ hover_data={'Percentage': ':.2f', 'Count': ':,', 'Identifier': True}
)
-
fig.update_layout(
- height=max(600, len(stats_df) * 18), width=1000,
- xaxis_title='Total Message Count (Log Scale)', yaxis_title='PGN or CAN ID',
- yaxis={'categoryorder': 'total ascending'},
- plot_bgcolor='rgba(0,0,0,0)', paper_bgcolor='white',
- font=dict(family="Segoe UI, Arial, sans-serif", size=12),
- hoverlabel=dict(bgcolor="white", font_size=13, font_family="Segoe UI"),
- margin=dict(l=250, r=50, t=80, b=50),
- title=dict(font=dict(size=20), x=0.5)
+ height=max(600, len(stats_df) * 18),
+ autosize=True,
+ template='plotly_white',
+ xaxis=dict(
+ title=dict(text="Message count [log scale]", font=dict(size=13, color="#1a1a1a")),
+ side="top",
+ dtick=1,
+ showgrid=False,
+ linecolor="#bdbdbd",
+ tickfont=dict(size=12, color="#2a2a2a"),
+ ticks="outside",
+ ticklen=4,
+ tickcolor="#cccccc",
+ ),
+ yaxis=dict(
+ title=dict(text="PGN or CAN ID", font=dict(size=13, color="#1a1a1a")),
+ #autorange="",
+ showgrid=False,
+ linecolor="#bdbdbd",
+ tickfont=dict(size=12, color="#2a2a2a"),
+ ticks="outside",
+ ticklen=4,
+ tickcolor="#cccccc",
+ automargin=True,
+ ),
+ font=dict(family="Segoe UI, Arial, sans-serif", size=12, color='#2a2a2a'),
+ hoverlabel=dict(bgcolor="white", font_size=13, font_family="Segoe UI",
+ bordercolor='#cccccc'),
+ margin=dict(l=200, r=40, t=120, b=60),
+ bargap=0.35,
+ coloraxis_colorbar=dict(
+ title=dict(text='Message Count', side='top'),
+ orientation='h',
+ thickness=15,
+ len=0.35,
+ x=1.0,
+ xanchor='right',
+ y=1.02,
+ yanchor='bottom',
+ tickformat=',',
+ outlinecolor='#cccccc',
+ outlinewidth=0.5
+ ),
+ title=dict(font=dict(size=20, color='#1a1a1a'), x=0.5, xanchor='center',
+ pad=dict(b=20))
+ )
+ fig.update_xaxes(
+ showgrid=True, gridwidth=0.5, gridcolor='#e8e8e8',
+ zeroline=False, linecolor='#bdbdbd', mirror=False,
+ tickformat=',',
+ minor=dict(showgrid=True, gridcolor='#f4f4f4', gridwidth=0.5)
+ )
+ fig.update_yaxes(
+ showgrid=False, zeroline=False, linecolor='#bdbdbd',
+ ticks='outside', ticklen=4, tickcolor='#cccccc',
+ automargin=True
)
-
fig.update_traces(
- hovertemplate="%{y}
Count: %{x:,}
Share: %{customdata[0]}%"
+ hovertemplate="%{y}
Count: %{x:,}
Share: %{customdata[0]}%",
+ marker_line_width=0,
+ texttemplate='%{x:,}',
+ textposition='outside',
+ textfont=dict(size=10, color='#666666'),
+ cliponaxis=False,
+ selected=dict(marker=dict(opacity=0.6)),
+ unselected=dict(marker=dict(opacity=0.2))
)
return fig
if __name__ == "__main__":
- parser = argparse.ArgumentParser(description="Analyze CAN bus Parquet data.")
- parser.add_argument("-i", "--input", type=Path, required=True)
- parser.add_argument("-o", "--output", type=Path, default=Path("can_analysis_report.html"))
- parser.add_argument("-t", "--title", type=str, default="CAN Bus Message Frequency by PGN / ID")
+ parser = argparse.ArgumentParser(description="Analyze CAN bus message frequency")
+ parser.add_argument("input", type=Path, help="Path to the input CAN log file")
+ parser.add_argument("output", type=Path, nargs="?", default=Path("freq_report.html"), help="Path to the output HTML report")
+ parser.add_argument("title", nargs="?", default="CAN Bus Message Frequency", help="Title for the HTML report")
args = parser.parse_args()
df = load_data(args.input)
- stats_df = calculate_frequency(df)
- fig = visualize_frequency(stats_df, title=args.title)
- fig.write_html(str(args.output), include_plotlyjs='cdn')
+ stats = calc_freq(df)
+ fig = plot_freq(stats, title=args.title)
+ config = {
+ 'responsive': True,
+ 'displaylogo': False,
+ 'scrollZoom': True,
+ 'modeBarButtonsToAdd': ['toggleSpikelines'],
+ 'toImageButtonOptions': {'format': 'png', 'scale': 2}
+ }
+ fig.write_html(str(args.output), include_plotlyjs='cdn', config=config)
diff --git a/stat/utils/extractor.py b/stat/utils/extractor.py
new file mode 100644
index 0000000..8cdd07e
--- /dev/null
+++ b/stat/utils/extractor.py
@@ -0,0 +1,27 @@
+# File: extractor.py
+# Copyright (C) 2026 Erick Ahmed
+# SPDX-License-Identifier: AGPL-3.0-or-later
+
+import json
+from pathlib import Path
+import pandas as pd
+
+def extract_id(row: pd.Series) -> str:
+ """Extracts PGN from metadata or falls back to CAN ID."""
+ meta = row.get('j1939_metadata')
+ if pd.isna(meta):
+ return f"ID: {row['ID']}"
+ if isinstance(meta, str):
+ try:
+ meta = json.loads(meta)
+ except json.JSONDecodeError:
+ return f"ID: {row['ID']}"
+ if isinstance(meta, dict) and 'PGN' in meta:
+ return f"PGN: {meta['PGN']}"
+ return f"ID: {row['ID']}"
+
+def load_data(file_path: Path) -> pd.DataFrame:
+ """Loads Parquet file and adds an Identifier column."""
+ df = pd.read_parquet(file_path)
+ df['Identifier'] = df.apply(extract_id, axis=1)
+ return df
diff --git a/utils/extractor.py b/utils/extractor.py
new file mode 100644
index 0000000..8cdd07e
--- /dev/null
+++ b/utils/extractor.py
@@ -0,0 +1,27 @@
+# File: extractor.py
+# Copyright (C) 2026 Erick Ahmed
+# SPDX-License-Identifier: AGPL-3.0-or-later
+
+import json
+from pathlib import Path
+import pandas as pd
+
+def extract_id(row: pd.Series) -> str:
+ """Extracts PGN from metadata or falls back to CAN ID."""
+ meta = row.get('j1939_metadata')
+ if pd.isna(meta):
+ return f"ID: {row['ID']}"
+ if isinstance(meta, str):
+ try:
+ meta = json.loads(meta)
+ except json.JSONDecodeError:
+ return f"ID: {row['ID']}"
+ if isinstance(meta, dict) and 'PGN' in meta:
+ return f"PGN: {meta['PGN']}"
+ return f"ID: {row['ID']}"
+
+def load_data(file_path: Path) -> pd.DataFrame:
+ """Loads Parquet file and adds an Identifier column."""
+ df = pd.read_parquet(file_path)
+ df['Identifier'] = df.apply(extract_id, axis=1)
+ return df