36 Commits

Author SHA1 Message Date
eeeck fe4bc0e46a Merge pull request 'Implement Pearson and Spearman per-bit correleration' (#3) from dev-inter-byte-correlation into main
Reviewed-on: erickahmed/CANveyor#3
2026-07-13 21:54:38 +02:00
eeeck c9ae482174 Add positional argument to choose correlation methods (Pearson or
Spearman)
2026-07-13 21:53:20 +02:00
eeeck 5e6f81b50b Add hex converter utility
- To move to separate utility file in the future
2026-07-13 21:50:32 +02:00
eeeck ce74c7fcd9 Treat undefined correlation as zero correlation 2026-07-13 21:43:24 +02:00
eeeck 0f25c0671b Implement Pearson correlation 2026-07-13 21:43:17 +02:00
eeeck d0e70dad2a Remove leftovers 2026-07-13 21:20:07 +02:00
eeeck f39d23fc64 Merge pull request 'Implement entropy heatmap for CAN frames' (#2) from dev-entropy-heatmap into main
Reviewed-on: erickahmed/CANveyor#2
2026-07-13 21:17:33 +02:00
eeeck 7daf4c8e08 Add entropy heatmap analysis
- Same style of frequency analysis for consistency
2026-07-13 21:16:18 +02:00
eeeck 8a3fc81d12 Use more consistent styling
- X axis on top
- Refactor figure generation logic
2026-07-13 21:15:36 +02:00
eeeck db7e5eb40e Add header informations 2026-07-13 20:08:29 +02:00
eeeck 8ccb97afae Remove requirement for positional arguments 2026-07-13 20:07:10 +02:00
eeeck 13d91d566b Move to subfolder
- Makes python recognize it as a submodule
2026-07-13 20:06:45 +02:00
eeeck 118f1567a2 Merge pull request 'Add header with copyright and SPDX licensing information' (#1) from code-header into dev-entropy-heatmap
Reviewed-on: erickahmed/CANveyor#1
2026-07-13 19:54:10 +02:00
eeeck cbba800536 Merge branch 'dev-entropy-heatmap' into code-header 2026-07-13 19:53:59 +02:00
eeeck 17340746ce Move data extraction functions to utility library 2026-07-13 19:51:22 +02:00
eeeck ad8334c915 Add header with copyright and SPDX licensing information 2026-07-13 19:48:46 +02:00
eeeck 9a5a231841 Add simple CAN message frequency analyzer 2026-07-13 19:28:52 +02:00
eeeck 74b393934c Fix NoneType error 2026-07-13 18:53:27 +02:00
eeeck 4a701e34c5 Remove unused variable 2026-07-13 18:48:45 +02:00
eeeck c42afb34b3 Specify that code is licensed under AGPLv3-or-later 2026-07-13 15:24:52 +02:00
eeeck 70abf7b5ce Use last pre-release tag version 2026-07-13 14:35:52 +02:00
eeeck f975cba4f8 Directory restruture
- Essential files in .
- Analysis or utilities related scripts on respective directories
2026-07-13 14:31:27 +02:00
eeeck 663d75174e Remove J1939 stub for data calculation
- To be done separately
2026-07-13 14:17:12 +02:00
eeeck 324aa5652c Use AGPLv3 license 2026-07-11 09:52:41 +02:00
eeeck e27351aa28 Delete src/analyzer.py 2026-07-10 01:27:54 +02:00
eeeck 6e386fcb4b Add stub to decode J1939 and calculate useful values 2026-07-10 01:27:21 +02:00
eeeck ca670c9bb7 Fix issues with metadata not correctly computed 2026-07-10 00:46:42 +02:00
eeeck 725c9ccbc1 Use more clear function name 2026-07-10 00:03:47 +02:00
eeeck 9d5cd74ef7 Decode J1939 metadata for compliant frames and add a struct with J1939
metadata
2026-07-10 00:03:04 +02:00
eeeck 6ef0571e4e Add J1939 decoder 2026-07-09 23:57:37 +02:00
eeeck c758aca6c6 Drop data column, keeping only the 8 bytes 2026-07-09 23:45:45 +02:00
eeeck 21172e020b Keep parquet data in hexadecimal format 2026-07-09 23:39:55 +02:00
eeeck ce20587591 Move to /src 2026-07-09 23:29:25 +02:00
eeeck 2fcaef5571 Remove file meant to be local only 2026-07-06 18:26:28 +02:00
eeeck ecef35918e Add parquet optimizer for faster data analysis in the future 2026-07-06 18:24:33 +02:00
eeeck 86b14af3c7 Add python project related files 2026-07-06 18:08:53 +02:00
11 changed files with 822 additions and 202 deletions
+1
View File
@@ -0,0 +1 @@
3.14
+9
View File
@@ -0,0 +1,9 @@
# Copyright
Copyright © 2026 Erick Ahmed
The source code in this repository is licensed under the **GNU Affero General Public License v3.0 or later (AGPL-3.0-or-later)**.
A copy of the license is provided in the `LICENSE` file. If any discrepancy exists between this notice and the `LICENSE` file, the `LICENSE` file shall prevail.
Any third-party components included in this repository at any point during developement remain the property of their respective copyright holders and are subject to their own license terms.
+2 -2
View File
@@ -633,8 +633,8 @@ the "copyright" line and a pointer to where the full notice is found.
Copyright (C) <year> <name of author> Copyright (C) <year> <name of author>
This program is free software: you can redistribute it and/or modify This program is free software: you can redistribute it and/or modify
it under the terms of the GNU Affero General Public License as published by it under the terms of the GNU Affero General Public License as published
the Free Software Foundation, either version 3 of the License, or by the Free Software Foundation, either version 3 of the License, or
(at your option) any later version. (at your option) any later version.
This program is distributed in the hope that it will be useful, This program is distributed in the hope that it will be useful,
+86
View File
@@ -0,0 +1,86 @@
# File: decoder.py
# Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later
import argparse
import polars as pl
def get_j1939_mask() -> pl.Expr:
"""
Returns a Polars expression representing the strict J1939 filtering rules.
"""
id_int = pl.col("ID").str.to_integer(base=16).cast(pl.UInt32)
return id_int > 0x7FF
def decode_j1939_metadata(lf: pl.LazyFrame) -> pl.LazyFrame:
"""
Decodes J1939 fields and bundles them into a Struct column.
"""
id_int = pl.col("ID").str.to_integer(base=16).cast(pl.UInt32)
priority = ((id_int // 67108864) % 8).cast(pl.UInt8)
pf = ((id_int // 65536) % 256).cast(pl.UInt8)
ps = ((id_int // 256) % 256).cast(pl.UInt8)
sa = (id_int % 256).cast(pl.UInt8)
da = pl.when(pf < 240).then(ps).otherwise(pl.lit(255, dtype=pl.UInt8)).cast(pl.UInt8)
pgn = pl.when(pf < 240).then(
((id_int // 256) & 0x3FF00)
).otherwise(
((id_int // 256) & 0x3FFFF)
).cast(pl.UInt32)
return lf.with_columns(
pl.struct([
priority.alias("Priority"),
pf.alias("PF"),
ps.alias("PS"),
sa.alias("SA"),
da.alias("DA"),
pgn.alias("PGN")
]).alias("j1939_metadata")
)
def decode_j1939_frames(df: pl.DataFrame) -> pl.DataFrame:
id_int = pl.col("ID").str.to_integer(base=16).cast(pl.UInt32)
is_j1939 = id_int > 0x7FF
priority = ((id_int // 67108864) % 8).cast(pl.UInt8)
pf = ((id_int // 65536) % 256).cast(pl.UInt8)
ps = ((id_int // 256) % 256).cast(pl.UInt8)
sa = (id_int % 256).cast(pl.UInt8)
da = pl.when(pf < 240).then(ps).otherwise(pl.lit(255, dtype=pl.UInt8)).cast(pl.UInt8)
pgn = pl.when(pf < 240).then(
((id_int // 256) & 0x3FF00)
).otherwise(
((id_int // 256) & 0x3FFFF)
).cast(pl.UInt32)
j1939_meta = pl.when(is_j1939).then(
pl.struct([
priority.alias("Priority"),
pf.alias("PF"),
ps.alias("PS"),
sa.alias("SA"),
da.alias("DA"),
pgn.alias("PGN")
])
).otherwise(None)
return df.with_columns(j1939_meta.alias("j1939_metadata"))
if __name__ == "__main__":
parser = argparse.ArgumentParser(description="J1939 decoder")
parser.add_argument("input_parquet", help="Path to the raw .parquet file")
parser.add_argument("output_parquet", help="Path to save the decoded .parquet file")
args = parser.parse_args()
df = pl.scan_parquet(args.input_parquet).collect()
decoded_df = decode_j1939_frames(df)
decoded_df.write_parquet(args.output_parquet)
+3
View File
@@ -0,0 +1,3 @@
# File: main.py
# Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later
+55 -17
View File
@@ -1,21 +1,25 @@
# File: parser.py
# Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later
import re import re
import csv import csv
import polars as pl
from pathlib import Path from pathlib import Path
from typing import Union
def parse_can_log(input_path: str | Path, out_bus1: str | Path, out_bus2: str | Path) -> None: PathLike = Union[str, Path]
def parse_log(input_path: PathLike, out_bus1: PathLike, out_bus2: PathLike) -> None:
""" """
Parses a CAN bus log file from CANdigger and saves valid frames to separate CSV files for Bus 1 and Bus 2. Parses a raw CAN bus log from CANdigger using regex and saves valid frames to separate CSV.
Corrupted, incomplete, or debug frames are silently discarded. Corrupted, incomplete, or debug frames are silently discarded.
""" """
input_file = Path(input_path) input_file = Path(input_path)
out1_file = Path(out_bus1) out1_file = Path(out_bus1)
out2_file = Path(out_bus2) out2_file = Path(out_bus2)
# Group 1: Bus (C1 or C2)
# Group 2: ID (1 to 8 hex chars)
# Group 3: DLC (1 to 2 hex chars)
start_pattern = re.compile(r'(C[12]):([0-9A-Fa-f]{1,8})\s+([0-9A-Fa-f]{1,2})\s+') start_pattern = re.compile(r'(C[12]):([0-9A-Fa-f]{1,8})\s+([0-9A-Fa-f]{1,2})\s+')
byte_pattern = re.compile(r'^[0-9A-Fa-f]{2}$') byte_pattern = re.compile(r'^[0-9A-Fa-f]{2}$')
with input_file.open('r', encoding='utf-8') as f_in, \ with input_file.open('r', encoding='utf-8') as f_in, \
@@ -61,21 +65,55 @@ def parse_can_log(input_path: str | Path, out_bus1: str | Path, out_bus2: str |
elif bus == 'C2': elif bus == 'C2':
writer2.writerow(row) writer2.writerow(row)
if __name__ == '__main__': def parse_csv(csv_path: PathLike) -> pl.LazyFrame:
# Example usage: """
# parse_can_log('can_traffic.txt', 'bus1_output.csv', 'bus2_output.csv') Ingests a parsed CSV file, unpacks hex strings into 8 hex columns,
# or on CLI: generates sequential timestamps if missing, and returns a Polars LazyFrame.
# python3 can_parser.py can_traffic.txt' bus1_output.csv bus2_output.csv """
lf = pl.scan_csv(csv_path, schema_overrides={"ID": pl.String, "Data": pl.String})
if "Timestamp" not in lf.columns:
lf = lf.with_row_index("Timestamp")
byte_exprs = []
for i in range(8):
expr = (
pl.col("Data").str.strip_chars().str.split(" ")
.list.get(i, null_on_oob=True)
.alias(f"b{i}")
)
byte_exprs.append(expr)
lf = lf.with_columns(byte_exprs).drop("Data")
return lf.with_columns([
pl.col("DLC").cast(pl.UInt8),
pl.col("Timestamp").cast(pl.Float64)
])
if __name__ == '__main__': if __name__ == '__main__':
import argparse import argparse
parser = argparse.ArgumentParser(description="Parse CAN bus logs to separate CSV files") parser = argparse.ArgumentParser(description="CAN Bus Data Engine & Parser")
parser.add_argument("input", help="Path to the input .txt log file") subparsers = parser.add_subparsers(dest="command", required=True, help="Available commands")
parser.add_argument("out_bus1", help="Output CSV filename for Bus 1 (C1)")
parser.add_argument("out_bus2", help="Output CSV filename for Bus 2 (C2)") parser_csv = subparsers.add_parser("csv", help="Parse raw text log into Bus 1 and Bus 2 CSVs")
parser_csv.add_argument("input", help="Path to the .txt log file from CANdigger")
parser_csv.add_argument("out_bus1", help="Output CSV filename for Bus 1 (C1)")
parser_csv.add_argument("out_bus2", help="Output CSV filename for Bus 2 (C2)")
parser_parquet = subparsers.add_parser("parquet", help="Convert a parsed CSV into an optimized Parquet file")
parser_parquet.add_argument("input_csv", help="Path to the input .csv file")
parser_parquet.add_argument("output_parquet", help="Path to the output .parquet file")
args = parser.parse_args() args = parser.parse_args()
parse_can_log(args.input, args.out_bus1, args.out_bus2)
pass if args.command == "csv":
parse_log(args.input, args.out_bus1, args.out_bus2)
print(f"[+] Saved csv file to {args.out_bus1} and {args.out_bus2}")
elif args.command == "parquet":
print(f"[*] Processing {args.input_csv}...")
lf = parse_csv(args.input_csv)
lf.sink_parquet(args.output_parquet)
print(f"[+] Saved parquet file to {args.output_parquet}")
+7
View File
@@ -0,0 +1,7 @@
[project]
name = "CANveyor"
version = "0.0.1"
description = "J1939 CAN bus parser that works in pair with CANdigger"
readme = "README.md"
requires-python = ">=3.14"
dependencies = ["polars", "pathlib", "typing"]
+144
View File
@@ -0,0 +1,144 @@
# File: correlation.py
# Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later
import argparse
from pathlib import Path
import numpy as np
import pandas as pd
import plotly.graph_objects as go
from utils.extractor import load_data
def _to_int(x):
"""Convert a hex string or integer to int, returning NaN on failure."""
if isinstance(x, (int, np.integer)):
return int(x)
if isinstance(x, str):
try:
return int(x, 16)
except ValueError:
return np.nan
return np.nan
def calculate_correlation(df: pd.DataFrame, method: str, target_id: str | None = None) -> pd.DataFrame:
"""Calculates inter-byte correlation grouped by identifier."""
byte_cols = [f"b{i}" for i in range(8)]
available_cols = [col for col in byte_cols if col in df.columns]
if not available_cols:
raise ValueError("No byte columns (b0-b7) found in the DataFrame")
df_bytes = df[["Identifier"] + available_cols].copy()
for col in available_cols:
df_bytes[col] = df_bytes[col].apply(_to_int)
if target_id:
group = df_bytes[df_bytes["Identifier"] == target_id]
if group.empty:
raise ValueError(f"Identifier '{target_id}' not found in data")
return group[available_cols].corr(method=method).fillna(0.0)
def max_abs_corr(group: pd.DataFrame) -> pd.Series:
corr_arr = np.abs(group.corr(method=method).to_numpy().copy())
np.fill_diagonal(corr_arr, 0.0)
return pd.Series(corr_arr.max(axis=0), index=group.columns).fillna(0.0)
return df_bytes.groupby("Identifier")[available_cols].apply(max_abs_corr)
def plot_correlation_heatmap(corr_df: pd.DataFrame, target_id: str | None, title: str) -> go.Figure:
"""Generates an interactive heatmap of inter-byte correlation."""
is_8x8 = target_id is not None
if is_8x8:
x = corr_df.columns.tolist()
y = corr_df.index.tolist()
z = corr_df.values
z_min, z_max = -1.0, 1.0
colorscale = [
[0.0, "#2c7bb6"], [0.25, "#abd9e9"], [0.5, "#ffffff"],
[0.75, "#fdae61"], [1.0, "#d7191c"]
]
hover_template = "<b>%{y}</b> vs <b>%{x}</b><br>Correlation: %{z:.2f}<extra></extra>"
else:
x = corr_df.columns.tolist()
y = corr_df.index.tolist()
z = corr_df.values
z_min, z_max = 0.0, 1.0
colorscale = [
[0.0, "#ffffff"], [0.2, "#fff5f0"], [0.4, "#fecc5c"],
[0.6, "#fd8d3c"], [0.8, "#e31a1c"], [1.0, "#800026"]
]
hover_template = "<b>%{y}</b><br>Byte %{x} max correlation: %{z:.2f}<extra></extra>"
fig = go.Figure(
data=go.Heatmap(
z=z, x=x, y=y,
zmin=z_min, zmax=z_max,
colorscale=colorscale,
xgap=3, ygap=3,
text=np.round(z, 2),
texttemplate="%{text}",
textfont={"size": 11, "color": "#2a2a2a", "family": "Segoe UI, Arial, sans-serif"},
hoverongaps=False,
hovertemplate=hover_template,
colorbar=dict(
title=dict(text="Correlation", side="top", font=dict(size=13, color="#1a1a1a")),
orientation="h", thickness=15, len=0.35,
x=1.0, xanchor="right", y=1.02, yanchor="bottom",
tickfont=dict(size=11, color="#2a2a2a"),
tickformat=".1f", outlinewidth=0.5, outlinecolor="#cccccc",
),
)
)
fig.update_layout(
title=dict(text=title, font=dict(size=20, color="#1a1a1a"), x=0.5, xanchor="center", pad=dict(b=20)),
height=max(600, len(y) * 28 + 150) if not is_8x8 else 600,
autosize=True,
template="plotly_white",
xaxis=dict(
title=dict(text="Byte Position", font=dict(size=13, color="#1a1a1a")),
side="top" if not is_8x8 else "bottom",
dtick=1, showgrid=False, linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"), ticks="outside", ticklen=4, tickcolor="#cccccc",
),
yaxis=dict(
title=dict(text="PGN or CAN ID" if not is_8x8 else "Byte Position", font=dict(size=13, color="#1a1a1a")),
autorange="reversed", showgrid=False, linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"), ticks="outside", ticklen=4, tickcolor="#cccccc", automargin=True,
),
font=dict(family="Segoe UI, Arial, sans-serif", size=12, color="#2a2a2a"),
hoverlabel=dict(bgcolor="white", font_size=13, font_family="Segoe UI", bordercolor="#cccccc"),
margin=dict(l=200, r=40, t=120, b=60),
)
return fig
if __name__ == "__main__":
parser = argparse.ArgumentParser(description="Analyze CAN bus inter-byte correlation")
parser.add_argument("method", choices=["pearson", "spearman"], help="Correlation method to use")
parser.add_argument("input", type=Path, help="Path to the input CAN log file")
parser.add_argument("output", type=Path, nargs="?", default=Path("correlation_report.html"), help="Path to the output HTML report")
parser.add_argument("title", nargs="?", default="CAN Bus Inter-Byte Correlation", help="Title for the HTML report")
parser.add_argument("--identifier", type=str, default=None, help="Specific PGN/CAN ID to analyze (e.g., 'PGN: 65331'). If omitted, shows max correlation per byte for all IDs.")
args = parser.parse_args()
df = load_data(args.input)
corr_df = calculate_correlation(df, method=args.method, target_id=args.identifier)
display_title = f"{args.title} ({args.identifier})" if args.identifier else args.title
fig = plot_correlation_heatmap(corr_df, target_id=args.identifier, title=display_title)
config = {
"responsive": True,
"displaylogo": False,
"scrollZoom": True,
"modeBarButtonsToAdd": ["toggleSpikelines"],
"toImageButtonOptions": {"format": "png", "scale": 2},
}
fig.write_html(str(args.output), include_plotlyjs="cdn", config=config)
+180
View File
@@ -0,0 +1,180 @@
# File: entropy.py
# Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later
import argparse
from pathlib import Path
import numpy as np
import pandas as pd
import plotly.graph_objects as go
from utils.extractor import load_data
def _to_int(x):
"""Convert a hex string or integer to int, returning NaN on failure."""
if isinstance(x, (int, np.integer)):
return int(x)
if isinstance(x, str):
try:
return int(x, 16)
except ValueError:
return np.nan
return np.nan
def calculate_byte_entropy(df: pd.DataFrame) -> pd.DataFrame:
"""Calculates Shannon entropy per byte position for each identifier."""
byte_cols = [f"b{i}" for i in range(8)]
available_cols = [col for col in byte_cols if col in df.columns]
if not available_cols:
raise ValueError("No byte columns (b0-b7) found in the DataFrame")
df_bytes = df[available_cols].copy()
for col in available_cols:
df_bytes[col] = df_bytes[col].apply(_to_int)
def entropy(s: pd.Series) -> float:
s = s.dropna()
if s.empty:
return 0.0
p = s.value_counts(normalize=True)
return -np.sum(p * np.log2(p))
return df.groupby("Identifier")[available_cols].agg(entropy)
def plot_entropy_heatmap(entropy_df: pd.DataFrame, title: str) -> go.Figure:
"""Generates an interactive heatmap of byte-level Shannon entropy."""
x = entropy_df.columns.tolist()
y = entropy_df.index.tolist()
z = entropy_df.values
fig = go.Figure(
data=go.Heatmap(
z=z,
x=x,
y=y,
colorscale=[
[0.0, "#ffffff"],
[0.15, "#fff7ec"],
[0.35, "#fee8c8"],
[0.55, "#fdd49e"],
[0.75, "#fdbb84"],
[1.0, "#ef6548"],
],
xgap=3,
ygap=3,
text=np.round(z, 2),
texttemplate="%{text}",
textfont={
"size": 11,
"color": "#2a2a2a",
"family": "Segoe UI, Arial, sans-serif",
},
hoverongaps=False,
hovertemplate=(
"<b>%{y}</b><br>"
"Byte %{x}: %{z:.2f} bits<extra></extra>"
),
colorbar=dict(
title=dict(
text="Entropy (bits)",
side="top",
font=dict(size=13, color="#1a1a1a"),
),
orientation="h",
thickness=15,
len=0.35,
x=1.0,
xanchor="right",
y=1.02,
yanchor="bottom",
tickfont=dict(size=11, color="#2a2a2a"),
tickformat=".1f",
outlinewidth=0.5,
outlinecolor="#cccccc",
),
)
)
fig.update_layout(
title=dict(
text=title,
font=dict(size=20, color="#1a1a1a"),
x=0.5,
xanchor="center",
pad=dict(b=20),
),
height=max(600, len(y) * 28 + 150),
autosize=True,
template="plotly_white",
xaxis=dict(
title=dict(text="Byte Position", font=dict(size=13, color="#1a1a1a")),
side="top",
dtick=1,
showgrid=False,
linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"),
ticks="outside",
ticklen=4,
tickcolor="#cccccc",
),
yaxis=dict(
title=dict(text="PGN or CAN ID", font=dict(size=13, color="#1a1a1a")),
autorange="reversed",
showgrid=False,
linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"),
ticks="outside",
ticklen=4,
tickcolor="#cccccc",
automargin=True,
),
font=dict(family="Segoe UI, Arial, sans-serif", size=12, color="#2a2a2a"),
hoverlabel=dict(
bgcolor="white",
font_size=13,
font_family="Segoe UI",
bordercolor="#cccccc",
),
margin=dict(l=200, r=40, t=120, b=60),
)
return fig
if __name__ == "__main__":
parser = argparse.ArgumentParser(
description="Analyze CAN bus byte-level entropy"
)
parser.add_argument(
"input", type=Path, help="Path to the input CAN log file"
)
parser.add_argument(
"output",
type=Path,
nargs="?",
default=Path("entropy_report.html"),
help="Path to the output HTML report",
)
parser.add_argument(
"title",
nargs="?",
default="CAN Bus Byte-Level Entropy",
help="Title for the HTML report",
)
args = parser.parse_args()
df = load_data(args.input)
entropy_df = calculate_byte_entropy(df)
fig = plot_entropy_heatmap(entropy_df, title=args.title)
config = {
"responsive": True,
"displaylogo": False,
"scrollZoom": True,
"modeBarButtonsToAdd": ["toggleSpikelines"],
"toImageButtonOptions": {"format": "png", "scale": 2},
}
fig.write_html(str(args.output), include_plotlyjs="cdn", config=config)
+116
View File
@@ -0,0 +1,116 @@
# File: frequency.py
# Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later
import argparse
from pathlib import Path
import pandas as pd
import plotly.express as px
import plotly.graph_objects as go
from utils.extractor import load_data
def calc_freq(df: pd.DataFrame) -> pd.DataFrame:
"""Calculates frequency counts and percentages for identifiers."""
freq_df = df['Identifier'].value_counts().reset_index()
freq_df.columns = ['Identifier', 'Count']
total = freq_df['Count'].sum()
freq_df['Percentage'] = (freq_df['Count'] / total * 100).round(2)
return freq_df.sort_values('Count', ascending=True)
def plot_freq(stats_df: pd.DataFrame, title: str) -> go.Figure:
"""Generates interactive horizontal bar chart with log x-axis."""
fig = px.bar(
stats_df, y='Identifier', x='Count', orientation='h', title=title, log_x=True,
labels={'Identifier': 'PGN / CAN ID', 'Count': 'Message Count'},
color='Count', color_continuous_scale='Turbo',
range_color=(stats_df['Count'].min(), stats_df['Count'].max()),
hover_data={'Percentage': ':.2f', 'Count': ':,', 'Identifier': True}
)
fig.update_layout(
height=max(600, len(stats_df) * 18),
autosize=True,
template='plotly_white',
xaxis=dict(
title=dict(text="Message count [log scale]", font=dict(size=13, color="#1a1a1a")),
side="top",
dtick=1,
showgrid=False,
linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"),
ticks="outside",
ticklen=4,
tickcolor="#cccccc",
),
yaxis=dict(
title=dict(text="PGN or CAN ID", font=dict(size=13, color="#1a1a1a")),
#autorange="",
showgrid=False,
linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"),
ticks="outside",
ticklen=4,
tickcolor="#cccccc",
automargin=True,
),
font=dict(family="Segoe UI, Arial, sans-serif", size=12, color='#2a2a2a'),
hoverlabel=dict(bgcolor="white", font_size=13, font_family="Segoe UI",
bordercolor='#cccccc'),
margin=dict(l=200, r=40, t=120, b=60),
bargap=0.35,
coloraxis_colorbar=dict(
title=dict(text='Message Count', side='top'),
orientation='h',
thickness=15,
len=0.35,
x=1.0,
xanchor='right',
y=1.02,
yanchor='bottom',
tickformat=',',
outlinecolor='#cccccc',
outlinewidth=0.5
),
title=dict(font=dict(size=20, color='#1a1a1a'), x=0.5, xanchor='center',
pad=dict(b=20))
)
fig.update_xaxes(
showgrid=True, gridwidth=0.5, gridcolor='#e8e8e8',
zeroline=False, linecolor='#bdbdbd', mirror=False,
tickformat=',',
minor=dict(showgrid=True, gridcolor='#f4f4f4', gridwidth=0.5)
)
fig.update_yaxes(
showgrid=False, zeroline=False, linecolor='#bdbdbd',
ticks='outside', ticklen=4, tickcolor='#cccccc',
automargin=True
)
fig.update_traces(
hovertemplate="<b>%{y}</b><br>Count: %{x:,}<br>Share: %{customdata[0]}%<extra></extra>",
marker_line_width=0,
texttemplate='%{x:,}',
textposition='outside',
textfont=dict(size=10, color='#666666'),
cliponaxis=False,
selected=dict(marker=dict(opacity=0.6)),
unselected=dict(marker=dict(opacity=0.2))
)
return fig
if __name__ == "__main__":
parser = argparse.ArgumentParser(description="Analyze CAN bus message frequency")
parser.add_argument("input", type=Path, help="Path to the input CAN log file")
parser.add_argument("output", type=Path, nargs="?", default=Path("freq_report.html"), help="Path to the output HTML report")
parser.add_argument("title", nargs="?", default="CAN Bus Message Frequency", help="Title for the HTML report")
args = parser.parse_args()
df = load_data(args.input)
stats = calc_freq(df)
fig = plot_freq(stats, title=args.title)
config = {
'responsive': True,
'displaylogo': False,
'scrollZoom': True,
'modeBarButtonsToAdd': ['toggleSpikelines'],
'toImageButtonOptions': {'format': 'png', 'scale': 2}
}
fig.write_html(str(args.output), include_plotlyjs='cdn', config=config)
+36
View File
@@ -0,0 +1,36 @@
import json
from pathlib import Path
import numpy as np
import pandas as pd
def to_int(x):
"""Convert a hex string or integer to int, returning NaN on failure."""
if isinstance(x, (int, np.integer)):
return int(x)
if isinstance(x, str):
try:
return int(x, 16)
except ValueError:
return np.nan
return np.nan
def extract_id(row: pd.Series) -> str:
"""Extracts PGN from metadata or falls back to CAN ID."""
meta = row.get('j1939_metadata')
if pd.isna(meta):
return f"ID: {row['ID']}"
if isinstance(meta, str):
try:
meta = json.loads(meta)
except json.JSONDecodeError:
return f"ID: {row['ID']}"
if isinstance(meta, dict) and 'PGN' in meta:
return f"PGN: {meta['PGN']}"
return f"ID: {row['ID']}"
def load_data(file_path: Path) -> pd.DataFrame:
"""Loads Parquet file and adds an Identifier column."""
df = pd.read_parquet(file_path)
df['Identifier'] = df.apply(extract_id, axis=1)
return df