10 Commits

Author SHA1 Message Date
eeeck 7a0e2efba1 Integrate Dash background callbacks to handle computations in async 2026-07-15 00:18:27 +02:00
eeeck a2ac49c79f Refactor statistical analysis modules for performance
- Optimize data processing pipelines across files by replacing iterative
  pandas operations with vectorized NumPy routines
2026-07-15 00:17:59 +02:00
eeeck 0780a61d78 Improve bit selection 2026-07-14 23:46:09 +02:00
eeeck fae85d1af9 Change title 2026-07-14 14:28:15 +02:00
eeeck c9461868cc Implement CAN ID visualization at the bit level 2026-07-14 14:14:09 +02:00
eeeck 2408c7a963 Use better function names 2026-07-14 00:37:47 +02:00
eeeck 179ec6e56b Add header 2026-07-14 00:30:05 +02:00
eeeck d8105e2da3 Normalize CAN IDs number of bits 2026-07-14 00:21:03 +02:00
eeeck 9ddc6ea0d5 Use utility function instead of internal 2026-07-13 22:36:31 +02:00
eeeck df3e7b1f0e Update software version 2026-07-13 22:12:44 +02:00
8 changed files with 428 additions and 203 deletions
+48
View File
@@ -1,3 +1,51 @@
# File: main.py # File: main.py
# Copyright (C) 2026 Erick Ahmed # Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later # SPDX-License-Identifier: AGPL-3.0-or-later
import diskcache
import flask_caching
import dash
from dash import Input, Output, State, dcc, html, no_update
import dash_bootstrap_components as dbc
CACHE_DATA_DIR = ".cache_data"
background_callback_manager = dash.DiskcacheManager(cache_dir=CACHE_DATA_DIR)
data_cache = flask_caching.Cache(config={'CACHE_TYPE': 'FileSystemCache', 'CACHE_DIR': CACHE_DATA_DIR})
app = dash.Dash(
__name__,
external_stylesheets=[dbc.themes.BOOTSTRAP],
background_callback_manager=background_callback_manager
)
data_cache.init_app(app.server)
def _get_df(session_data):
if not session_data or "token" not in session_data:
return None
return data_cache.get(session_data["token"])
@app.callback(
Output("graph-correlation", "figure"),
Input("corr-method", "value"),
Input("corr-target", "value"),
Input("session-store", "data"),
background=True,
prevent_initial_call=True,
)
def update_correlation(method, target, session_data):
df = _get_df(session_data)
if df is None or not method:
return no_update
target_id = None if target == "all" else target
try:
corr_df = calculate_correlation(df, method=method, target_id=target_id)
title = f"Inter-Byte Correlation ({method.capitalize()})"
if target_id:
title += f" - {target_id}"
return plot_correlation_heatmap(corr_df, target_id=target_id, title=title)
except Exception as exc:
fig = dash.go.Figure()
fig.update_layout(title=f"Error: {exc}")
return fig
+7 -3
View File
@@ -19,7 +19,7 @@ def parse_log(input_path: PathLike, out_bus1: PathLike, out_bus2: PathLike) -> N
out1_file = Path(out_bus1) out1_file = Path(out_bus1)
out2_file = Path(out_bus2) out2_file = Path(out_bus2)
start_pattern = re.compile(r'(C[12]):([0-9A-Fa-f]{1,8})\s+([0-9A-Fa-f]{1,2})\s+') start_pattern = re.compile(r'(C[12]):([0-9A-Fa-f]{7,8})\s+([0-9A-Fa-f]{1,2})\s+')
byte_pattern = re.compile(r'^[0-9A-Fa-f]{2}$') byte_pattern = re.compile(r'^[0-9A-Fa-f]{2}$')
with input_file.open('r', encoding='utf-8') as f_in, \ with input_file.open('r', encoding='utf-8') as f_in, \
@@ -35,7 +35,12 @@ def parse_log(input_path: PathLike, out_bus1: PathLike, out_bus2: PathLike) -> N
for line in f_in: for line in f_in:
for match in start_pattern.finditer(line): for match in start_pattern.finditer(line):
bus = match.group(1) bus = match.group(1)
can_id = match.group(2).upper()
can_id = match.group(2).upper().zfill(8)
if int(can_id, 16) > 0x1FFFFFFF:
continue
dlc_str = match.group(3) dlc_str = match.group(3)
try: try:
@@ -116,4 +121,3 @@ if __name__ == '__main__':
print(f"[*] Processing {args.input_csv}...") print(f"[*] Processing {args.input_csv}...")
lf = parse_csv(args.input_csv) lf = parse_csv(args.input_csv)
lf.sink_parquet(args.output_parquet) lf.sink_parquet(args.output_parquet)
print(f"[+] Saved parquet file to {args.output_parquet}")
+1 -1
View File
@@ -1,6 +1,6 @@
[project] [project]
name = "CANveyor" name = "CANveyor"
version = "0.0.1" version = "0.0.4"
description = "J1939 CAN bus parser that works in pair with CANdigger" description = "J1939 CAN bus parser that works in pair with CANdigger"
readme = "README.md" readme = "README.md"
requires-python = ">=3.14" requires-python = ">=3.14"
+75 -48
View File
@@ -10,69 +10,98 @@ import pandas as pd
import plotly.graph_objects as go import plotly.graph_objects as go
from utils.extractor import load_data from utils.extractor import load_data
from utils.extractor import to_int
def _format_can_id_vec(s: pd.Series) -> pd.Series:
s = s.astype('string').str.strip()
s = s.str.replace(r'^0x', '', case=False, regex=True)
s = s.str.upper()
return s.fillna('UNKNOWN').replace('', 'UNKNOWN')
def _to_int(x): def _ensure_int_bytes(df: pd.DataFrame, cols: list) -> pd.DataFrame:
"""Convert a hex string or integer to int, returning NaN on failure.""" needs = [c for c in cols if not pd.api.types.is_numeric_dtype(df[c])]
if isinstance(x, (int, np.integer)): if needs:
return int(x) df = df.copy()
if isinstance(x, str): for c in needs:
try: df[c] = df[c].apply(to_int)
return int(x, 16) return df
except ValueError:
return np.nan
return np.nan
def calculate_correlation(df: pd.DataFrame, method: str, target_id: str | None = None) -> pd.DataFrame: def calculate_correlation(df: pd.DataFrame, method: str, target_id: str | None = None) -> pd.DataFrame:
"""Calculates inter-byte correlation grouped by identifier.""" byte_cols = [f"b{i}" for i in range(8) if f"b{i}" in df.columns]
byte_cols = [f"b{i}" for i in range(8)]
available_cols = [col for col in byte_cols if col in df.columns] available_cols = [col for col in byte_cols if col in df.columns]
if not available_cols: if not available_cols:
raise ValueError("No byte columns (b0-b7) found in the DataFrame") raise ValueError("No byte columns (b0-b7) found in the DataFrame")
df_bytes = df[["Identifier"] + available_cols].copy() can_id_col = 'ID' if 'ID' in df.columns else 'Identifier'
for col in available_cols: identifiers = _format_can_id_vec(df[can_id_col]).to_numpy()
df_bytes[col] = df_bytes[col].apply(_to_int)
if target_id: df_bytes = _ensure_int_bytes(df, available_cols)[available_cols]
group = df_bytes[df_bytes["Identifier"] == target_id] data = df_bytes.to_numpy(dtype=np.float64, copy=False)
if group.empty:
if target_id is not None:
target_id = _format_can_id_vec(pd.Series([target_id])).iloc[0]
mask = identifiers == target_id
if not mask.any():
raise ValueError(f"Identifier '{target_id}' not found in data") raise ValueError(f"Identifier '{target_id}' not found in data")
return group[available_cols].corr(method=method).fillna(0.0) sub = data[mask]
mask = ~np.isnan(sub).any(axis=1)
sub = sub[mask]
if method == 'spearman' and sub.shape[0] > 1:
sub = pd.DataFrame(sub).rank().to_numpy()
if sub.shape[0] > 1:
with np.errstate(divide='ignore', invalid='ignore'):
c = np.corrcoef(sub, rowvar=False)
np.nan_to_num(c, copy=False, nan=0.0)
else:
c = np.zeros((len(available_cols), len(available_cols)))
return pd.DataFrame(c, index=available_cols, columns=available_cols)
def max_abs_corr(group: pd.DataFrame) -> pd.Series: unique_ids, inverse = np.unique(identifiers, return_inverse=True)
corr_arr = np.abs(group.corr(method=method).to_numpy().copy()) n_cols = len(available_cols)
np.fill_diagonal(corr_arr, 0.0)
return pd.Series(corr_arr.max(axis=0), index=group.columns).fillna(0.0)
return df_bytes.groupby("Identifier")[available_cols].apply(max_abs_corr) sort_idx = np.argsort(inverse, kind='stable')
data_sorted = data[sort_idx]
inverse_sorted = inverse[sort_idx]
if len(inverse_sorted) > 0:
split_points = np.flatnonzero(np.diff(inverse_sorted)) + 1
groups = np.split(data_sorted, split_points)
else:
groups = []
out = np.zeros((len(unique_ids), n_cols), dtype=np.float64)
for gi, sub in enumerate(groups):
mask = ~np.isnan(sub).any(axis=1)
sub = sub[mask]
if len(sub) > 1:
if method == 'spearman':
sub = pd.DataFrame(sub).rank().to_numpy()
with np.errstate(divide='ignore', invalid='ignore'):
c = np.abs(np.corrcoef(sub, rowvar=False))
np.nan_to_num(c, copy=False, nan=0.0)
np.fill_diagonal(c, 0.0)
out[gi] = c.max(axis=0)
result = pd.DataFrame(out, index=unique_ids, columns=available_cols)
result.index.name = 'Identifier'
return result
def plot_correlation_heatmap(corr_df: pd.DataFrame, target_id: str | None, title: str) -> go.Figure: def plot_correlation_heatmap(corr_df: pd.DataFrame, target_id: str | None, title: str) -> go.Figure:
"""Generates an interactive heatmap of inter-byte correlation."""
is_8x8 = target_id is not None is_8x8 = target_id is not None
x = corr_df.columns.tolist()
y = corr_df.index.tolist()
z = corr_df.values
if is_8x8: if is_8x8:
x = corr_df.columns.tolist()
y = corr_df.index.tolist()
z = corr_df.values
z_min, z_max = -1.0, 1.0 z_min, z_max = -1.0, 1.0
colorscale = [ colorscale = [[0.0, "#2c7bb6"], [0.25, "#abd9e9"], [0.5, "#ffffff"], [0.75, "#fdae61"], [1.0, "#d7191c"]]
[0.0, "#2c7bb6"], [0.25, "#abd9e9"], [0.5, "#ffffff"],
[0.75, "#fdae61"], [1.0, "#d7191c"]
]
hover_template = "<b>%{y}</b> vs <b>%{x}</b><br>Correlation: %{z:.2f}<extra></extra>" hover_template = "<b>%{y}</b> vs <b>%{x}</b><br>Correlation: %{z:.2f}<extra></extra>"
else: else:
x = corr_df.columns.tolist()
y = corr_df.index.tolist()
z = corr_df.values
z_min, z_max = 0.0, 1.0 z_min, z_max = 0.0, 1.0
colorscale = [ colorscale = [[0.0, "#ffffff"], [0.2, "#fff5f0"], [0.4, "#fecc5c"], [0.6, "#fd8d3c"], [0.8, "#e31a1c"], [1.0, "#800026"]]
[0.0, "#ffffff"], [0.2, "#fff5f0"], [0.4, "#fecc5c"],
[0.6, "#fd8d3c"], [0.8, "#e31a1c"], [1.0, "#800026"]
]
hover_template = "<b>%{y}</b><br>Byte %{x} max correlation: %{z:.2f}<extra></extra>" hover_template = "<b>%{y}</b><br>Byte %{x} max correlation: %{z:.2f}<extra></extra>"
fig = go.Figure( fig = go.Figure(
@@ -98,17 +127,17 @@ def plot_correlation_heatmap(corr_df: pd.DataFrame, target_id: str | None, title
fig.update_layout( fig.update_layout(
title=dict(text=title, font=dict(size=20, color="#1a1a1a"), x=0.5, xanchor="center", pad=dict(b=20)), title=dict(text=title, font=dict(size=20, color="#1a1a1a"), x=0.5, xanchor="center", pad=dict(b=20)),
height=max(600, len(y) * 28 + 150) if not is_8x8 else 600, height=600 if is_8x8 else max(600, len(y) * 28 + 150),
autosize=True, autosize=True,
template="plotly_white", template="plotly_white",
xaxis=dict( xaxis=dict(
title=dict(text="Byte Position", font=dict(size=13, color="#1a1a1a")), title=dict(text="Byte Position", font=dict(size=13, color="#1a1a1a")),
side="top" if not is_8x8 else "bottom", side="bottom" if is_8x8 else "top",
dtick=1, showgrid=False, linecolor="#bdbdbd", dtick=1, showgrid=False, linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"), ticks="outside", ticklen=4, tickcolor="#cccccc", tickfont=dict(size=12, color="#2a2a2a"), ticks="outside", ticklen=4, tickcolor="#cccccc",
), ),
yaxis=dict( yaxis=dict(
title=dict(text="PGN or CAN ID" if not is_8x8 else "Byte Position", font=dict(size=13, color="#1a1a1a")), title=dict(text="Byte Position" if is_8x8 else "PGN or CAN ID", font=dict(size=13, color="#1a1a1a")),
autorange="reversed", showgrid=False, linecolor="#bdbdbd", autorange="reversed", showgrid=False, linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"), ticks="outside", ticklen=4, tickcolor="#cccccc", automargin=True, tickfont=dict(size=12, color="#2a2a2a"), ticks="outside", ticklen=4, tickcolor="#cccccc", automargin=True,
), ),
@@ -123,17 +152,15 @@ if __name__ == "__main__":
parser = argparse.ArgumentParser(description="Analyze CAN bus inter-byte correlation") parser = argparse.ArgumentParser(description="Analyze CAN bus inter-byte correlation")
parser.add_argument("method", choices=["pearson", "spearman"], help="Correlation method to use") parser.add_argument("method", choices=["pearson", "spearman"], help="Correlation method to use")
parser.add_argument("input", type=Path, help="Path to the input CAN log file") parser.add_argument("input", type=Path, help="Path to the input CAN log file")
parser.add_argument("output", type=Path, nargs="?", default=Path("correlation_report.html"), help="Path to the output HTML report") parser.add_argument("output", type=Path, nargs="?", default=Path("correlation_report.html"))
parser.add_argument("title", nargs="?", default="CAN Bus Inter-Byte Correlation", help="Title for the HTML report") parser.add_argument("title", nargs="?", default="CAN Bus Inter-Byte Correlation")
parser.add_argument("--identifier", type=str, default=None, help="Specific PGN/CAN ID to analyze (e.g., 'PGN: 65331'). If omitted, shows max correlation per byte for all IDs.") parser.add_argument("--identifier", type=str, default=None)
args = parser.parse_args() args = parser.parse_args()
df = load_data(args.input) df = load_data(args.input)
corr_df = calculate_correlation(df, method=args.method, target_id=args.identifier) corr_df = calculate_correlation(df, method=args.method, target_id=args.identifier)
display_title = f"{args.title} ({args.identifier})" if args.identifier else args.title display_title = f"{args.title} ({args.identifier})" if args.identifier else args.title
fig = plot_correlation_heatmap(corr_df, target_id=args.identifier, title=display_title) fig = plot_correlation_heatmap(corr_df, target_id=args.identifier, title=display_title)
config = { config = {
"responsive": True, "responsive": True,
"displaylogo": False, "displaylogo": False,
+71 -103
View File
@@ -10,52 +10,78 @@ import pandas as pd
import plotly.graph_objects as go import plotly.graph_objects as go
from utils.extractor import load_data from utils.extractor import load_data
from utils.extractor import to_int
def _format_can_id_vec(s: pd.Series) -> pd.Series:
s = s.astype('string').str.strip()
s = s.str.replace(r'^0x', '', case=False, regex=True)
s = s.str.upper()
return s.fillna('UNKNOWN').replace('', 'UNKNOWN')
def _to_int(x): def _entropy_col(a: np.ndarray) -> float:
"""Convert a hex string or integer to int, returning NaN on failure.""" a = a[~np.isnan(a)]
if isinstance(x, (int, np.integer)): if a.size == 0:
return int(x) return 0.0
if isinstance(x, str): a = a.astype(np.int64)
try: lo, hi = a.min(), a.max()
return int(x, 16) span = hi - lo + 1
except ValueError: if span <= 0:
return np.nan return 0.0
return np.nan if span > 1 << 20:
_, counts = np.unique(a, return_counts=True)
else:
counts = np.bincount(a - lo, minlength=span)
counts = counts[counts > 0]
p = counts / counts.sum()
return float(-np.sum(p * np.log2(p)))
def calculate_byte_entropy(df: pd.DataFrame) -> pd.DataFrame: def calculate_byte_entropy(df: pd.DataFrame) -> pd.DataFrame:
"""Calculates Shannon entropy per byte position for each identifier.""" byte_cols = [f"b{i}" for i in range(8) if f"b{i}" in df.columns]
byte_cols = [f"b{i}" for i in range(8)] available_cols = byte_cols
available_cols = [col for col in byte_cols if col in df.columns]
if not available_cols: if not available_cols:
raise ValueError("No byte columns (b0-b7) found in the DataFrame") raise ValueError("No byte columns (b0-b7) found in the DataFrame")
df_bytes = df[available_cols].copy() can_id_col = 'ID' if 'ID' in df.columns else 'Identifier'
for col in available_cols: identifiers = _format_can_id_vec(df[can_id_col]).to_numpy()
df_bytes[col] = df_bytes[col].apply(_to_int)
def entropy(s: pd.Series) -> float: needs = [c for c in available_cols if not pd.api.types.is_numeric_dtype(df[c])]
s = s.dropna() if needs:
if s.empty: df = df.copy()
return 0.0 for c in needs:
p = s.value_counts(normalize=True) df[c] = df[c].apply(to_int)
return -np.sum(p * np.log2(p))
return df.groupby("Identifier")[available_cols].agg(entropy) data = df[available_cols].to_numpy(dtype=np.float64, copy=False)
unique_ids, inverse = np.unique(identifiers, return_inverse=True)
n_cols = len(available_cols)
sort_idx = np.argsort(inverse, kind='stable')
data_sorted = data[sort_idx]
inverse_sorted = inverse[sort_idx]
if len(inverse_sorted) > 0:
split_points = np.flatnonzero(np.diff(inverse_sorted)) + 1
groups = np.split(data_sorted, split_points)
else:
groups = []
out = np.zeros((len(unique_ids), n_cols), dtype=np.float64)
for gi, sub in enumerate(groups):
for ci in range(n_cols):
out[gi, ci] = _entropy_col(sub[:, ci])
result = pd.DataFrame(out, index=unique_ids, columns=available_cols)
result.index.name = 'Identifier'
return result
def plot_entropy_heatmap(entropy_df: pd.DataFrame, title: str) -> go.Figure: def plot_entropy_heatmap(entropy_df: pd.DataFrame, title: str) -> go.Figure:
"""Generates an interactive heatmap of byte-level Shannon entropy."""
x = entropy_df.columns.tolist() x = entropy_df.columns.tolist()
y = entropy_df.index.tolist() y = entropy_df.index.tolist()
z = entropy_df.values z = entropy_df.values
fig = go.Figure( fig = go.Figure(
data=go.Heatmap( data=go.Heatmap(
z=z, z=z, x=x, y=y,
x=x,
y=y,
colorscale=[ colorscale=[
[0.0, "#ffffff"], [0.0, "#ffffff"],
[0.15, "#fff7ec"], [0.15, "#fff7ec"],
@@ -64,112 +90,54 @@ def plot_entropy_heatmap(entropy_df: pd.DataFrame, title: str) -> go.Figure:
[0.75, "#fdbb84"], [0.75, "#fdbb84"],
[1.0, "#ef6548"], [1.0, "#ef6548"],
], ],
xgap=3, xgap=3, ygap=3,
ygap=3,
text=np.round(z, 2), text=np.round(z, 2),
texttemplate="%{text}", texttemplate="%{text}",
textfont={ textfont={"size": 11, "color": "#2a2a2a", "family": "Segoe UI, Arial, sans-serif"},
"size": 11,
"color": "#2a2a2a",
"family": "Segoe UI, Arial, sans-serif",
},
hoverongaps=False, hoverongaps=False,
hovertemplate=( hovertemplate="<b>%{y}</b><br>Byte %{x}: %{z:.2f} bits<extra></extra>",
"<b>%{y}</b><br>"
"Byte %{x}: %{z:.2f} bits<extra></extra>"
),
colorbar=dict( colorbar=dict(
title=dict( title=dict(text="Entropy (bits)", side="top", font=dict(size=13, color="#1a1a1a")),
text="Entropy (bits)", orientation="h", thickness=15, len=0.35,
side="top", x=1.0, xanchor="right", y=1.02, yanchor="bottom",
font=dict(size=13, color="#1a1a1a"),
),
orientation="h",
thickness=15,
len=0.35,
x=1.0,
xanchor="right",
y=1.02,
yanchor="bottom",
tickfont=dict(size=11, color="#2a2a2a"), tickfont=dict(size=11, color="#2a2a2a"),
tickformat=".1f", tickformat=".1f", outlinewidth=0.5, outlinecolor="#cccccc",
outlinewidth=0.5,
outlinecolor="#cccccc",
), ),
) )
) )
fig.update_layout( fig.update_layout(
title=dict( title=dict(text=title, font=dict(size=20, color="#1a1a1a"), x=0.5, xanchor="center", pad=dict(b=20)),
text=title,
font=dict(size=20, color="#1a1a1a"),
x=0.5,
xanchor="center",
pad=dict(b=20),
),
height=max(600, len(y) * 28 + 150), height=max(600, len(y) * 28 + 150),
autosize=True, autosize=True,
template="plotly_white", template="plotly_white",
xaxis=dict( xaxis=dict(
title=dict(text="Byte Position", font=dict(size=13, color="#1a1a1a")), title=dict(text="Byte Position", font=dict(size=13, color="#1a1a1a")),
side="top", side="top", dtick=1, showgrid=False, linecolor="#bdbdbd",
dtick=1, tickfont=dict(size=12, color="#2a2a2a"), ticks="outside", ticklen=4, tickcolor="#cccccc",
showgrid=False,
linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"),
ticks="outside",
ticklen=4,
tickcolor="#cccccc",
), ),
yaxis=dict( yaxis=dict(
title=dict(text="PGN or CAN ID", font=dict(size=13, color="#1a1a1a")), title=dict(text="PGN or CAN ID", font=dict(size=13, color="#1a1a1a")),
autorange="reversed", autorange="reversed", showgrid=False, linecolor="#bdbdbd",
showgrid=False, tickfont=dict(size=12, color="#2a2a2a"), ticks="outside", ticklen=4, tickcolor="#cccccc", automargin=True,
linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"),
ticks="outside",
ticklen=4,
tickcolor="#cccccc",
automargin=True,
), ),
font=dict(family="Segoe UI, Arial, sans-serif", size=12, color="#2a2a2a"), font=dict(family="Segoe UI, Arial, sans-serif", size=12, color="#2a2a2a"),
hoverlabel=dict( hoverlabel=dict(bgcolor="white", font_size=13, font_family="Segoe UI", bordercolor="#cccccc"),
bgcolor="white",
font_size=13,
font_family="Segoe UI",
bordercolor="#cccccc",
),
margin=dict(l=200, r=40, t=120, b=60), margin=dict(l=200, r=40, t=120, b=60),
) )
return fig return fig
if __name__ == "__main__": if __name__ == "__main__":
parser = argparse.ArgumentParser( parser = argparse.ArgumentParser(description="Analyze CAN bus byte-level entropy")
description="Analyze CAN bus byte-level entropy" parser.add_argument("input", type=Path, help="Path to the input CAN log file")
) parser.add_argument("output", type=Path, nargs="?", default=Path("entropy_report.html"))
parser.add_argument( parser.add_argument("title", nargs="?", default="CAN Bus Byte-Level Entropy")
"input", type=Path, help="Path to the input CAN log file"
)
parser.add_argument(
"output",
type=Path,
nargs="?",
default=Path("entropy_report.html"),
help="Path to the output HTML report",
)
parser.add_argument(
"title",
nargs="?",
default="CAN Bus Byte-Level Entropy",
help="Title for the HTML report",
)
args = parser.parse_args() args = parser.parse_args()
df = load_data(args.input) df = load_data(args.input)
entropy_df = calculate_byte_entropy(df) entropy_df = calculate_byte_entropy(df)
fig = plot_entropy_heatmap(entropy_df, title=args.title) fig = plot_entropy_heatmap(entropy_df, title=args.title)
config = { config = {
"responsive": True, "responsive": True,
"displaylogo": False, "displaylogo": False,
+53 -42
View File
@@ -4,33 +4,57 @@
import argparse import argparse
from pathlib import Path from pathlib import Path
import numpy as np
import pandas as pd import pandas as pd
import plotly.express as px
import plotly.graph_objects as go import plotly.graph_objects as go
from utils.extractor import load_data from utils.extractor import load_data
def calc_freq(df: pd.DataFrame) -> pd.DataFrame: def _format_can_id_vec(s: pd.Series) -> pd.Series:
"""Calculates frequency counts and percentages for identifiers.""" s = s.astype('string').str.strip()
freq_df = df['Identifier'].value_counts().reset_index() s = s.str.replace(r'^0x', '', case=False, regex=True)
freq_df.columns = ['Identifier', 'Count'] s = s.str.upper()
total = freq_df['Count'].sum() return s.fillna('UNKNOWN').replace('', 'UNKNOWN')
freq_df['Percentage'] = (freq_df['Count'] / total * 100).round(2)
return freq_df.sort_values('Count', ascending=True) def calculate_frequency(df: pd.DataFrame) -> pd.DataFrame:
can_id_col = 'ID' if 'ID' in df.columns else 'Identifier'
formatted = _format_can_id_vec(df[can_id_col])
df['Formatted_ID'] = formatted
counts = formatted.value_counts()
freq_df = pd.DataFrame({
'Identifier': counts.index,
'Count': counts.to_numpy(),
})
total = counts.sum()
freq_df['Percentage'] = np.round(freq_df['Count'] / total * 100, 2) if total else 0.0
return freq_df.sort_values('Count', ascending=True).reset_index(drop=True)
def plot_frequency(stats_df: pd.DataFrame, title: str) -> go.Figure:
n = len(stats_df)
fig = go.Figure(go.Bar(
y=stats_df['Identifier'],
x=stats_df['Count'],
orientation='h',
marker=dict(
color=stats_df['Count'],
colorscale='Turbo',
cmin=int(stats_df['Count'].min()) if n else 0,
cmax=int(stats_df['Count'].max()) if n else 1,
line_width=0,
),
customdata=stats_df[['Percentage']].to_numpy(),
hovertemplate="<b>%{y}</b><br>Count: %{x:,}<br>Share: %{customdata[0]}%<extra></extra>",
texttemplate='%{x:,}',
textposition='outside',
cliponaxis=False,
))
def plot_freq(stats_df: pd.DataFrame, title: str) -> go.Figure:
"""Generates interactive horizontal bar chart with log x-axis."""
fig = px.bar(
stats_df, y='Identifier', x='Count', orientation='h', title=title, log_x=True,
labels={'Identifier': 'PGN / CAN ID', 'Count': 'Message Count'},
color='Count', color_continuous_scale='Turbo',
range_color=(stats_df['Count'].min(), stats_df['Count'].max()),
hover_data={'Percentage': ':.2f', 'Count': ':,', 'Identifier': True}
)
fig.update_layout( fig.update_layout(
height=max(600, len(stats_df) * 18), height=max(600, n * 18),
autosize=True, autosize=True,
template='plotly_white', template='plotly_white',
xaxis=dict( xaxis=dict(
type='log',
title=dict(text="Message count [log scale]", font=dict(size=13, color="#1a1a1a")), title=dict(text="Message count [log scale]", font=dict(size=13, color="#1a1a1a")),
side="top", side="top",
dtick=1, dtick=1,
@@ -43,7 +67,6 @@ def plot_freq(stats_df: pd.DataFrame, title: str) -> go.Figure:
), ),
yaxis=dict( yaxis=dict(
title=dict(text="PGN or CAN ID", font=dict(size=13, color="#1a1a1a")), title=dict(text="PGN or CAN ID", font=dict(size=13, color="#1a1a1a")),
#autorange="",
showgrid=False, showgrid=False,
linecolor="#bdbdbd", linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"), tickfont=dict(size=12, color="#2a2a2a"),
@@ -51,10 +74,10 @@ def plot_freq(stats_df: pd.DataFrame, title: str) -> go.Figure:
ticklen=4, ticklen=4,
tickcolor="#cccccc", tickcolor="#cccccc",
automargin=True, automargin=True,
type='category',
), ),
font=dict(family="Segoe UI, Arial, sans-serif", size=12, color='#2a2a2a'), font=dict(family="Segoe UI, Arial, sans-serif", size=12, color='#2a2a2a'),
hoverlabel=dict(bgcolor="white", font_size=13, font_family="Segoe UI", hoverlabel=dict(bgcolor="white", font_size=13, font_family="Segoe UI", bordercolor='#cccccc'),
bordercolor='#cccccc'),
margin=dict(l=200, r=40, t=120, b=60), margin=dict(l=200, r=40, t=120, b=60),
bargap=0.35, bargap=0.35,
coloraxis_colorbar=dict( coloraxis_colorbar=dict(
@@ -68,49 +91,37 @@ def plot_freq(stats_df: pd.DataFrame, title: str) -> go.Figure:
yanchor='bottom', yanchor='bottom',
tickformat=',', tickformat=',',
outlinecolor='#cccccc', outlinecolor='#cccccc',
outlinewidth=0.5 outlinewidth=0.5,
), ),
title=dict(font=dict(size=20, color='#1a1a1a'), x=0.5, xanchor='center', title=dict(text=title, font=dict(size=20, color='#1a1a1a'), x=0.5, xanchor='center', pad=dict(b=20)),
pad=dict(b=20))
) )
fig.update_xaxes( fig.update_xaxes(
showgrid=True, gridwidth=0.5, gridcolor='#e8e8e8', showgrid=True, gridwidth=0.5, gridcolor='#e8e8e8',
zeroline=False, linecolor='#bdbdbd', mirror=False, zeroline=False, linecolor='#bdbdbd', mirror=False,
tickformat=',', tickformat=',',
minor=dict(showgrid=True, gridcolor='#f4f4f4', gridwidth=0.5) minor=dict(showgrid=True, gridcolor='#f4f4f4', gridwidth=0.5),
) )
fig.update_yaxes( fig.update_yaxes(
showgrid=False, zeroline=False, linecolor='#bdbdbd', showgrid=False, zeroline=False, linecolor='#bdbdbd',
ticks='outside', ticklen=4, tickcolor='#cccccc', ticks='outside', ticklen=4, tickcolor='#cccccc', automargin=True,
automargin=True
)
fig.update_traces(
hovertemplate="<b>%{y}</b><br>Count: %{x:,}<br>Share: %{customdata[0]}%<extra></extra>",
marker_line_width=0,
texttemplate='%{x:,}',
textposition='outside',
textfont=dict(size=10, color='#666666'),
cliponaxis=False,
selected=dict(marker=dict(opacity=0.6)),
unselected=dict(marker=dict(opacity=0.2))
) )
return fig return fig
if __name__ == "__main__": if __name__ == "__main__":
parser = argparse.ArgumentParser(description="Analyze CAN bus message frequency") parser = argparse.ArgumentParser(description="Analyze CAN bus message frequency")
parser.add_argument("input", type=Path, help="Path to the input CAN log file") parser.add_argument("input", type=Path, help="Path to the input CAN log file")
parser.add_argument("output", type=Path, nargs="?", default=Path("freq_report.html"), help="Path to the output HTML report") parser.add_argument("output", type=Path, nargs="?", default=Path("freq_report.html"))
parser.add_argument("title", nargs="?", default="CAN Bus Message Frequency", help="Title for the HTML report") parser.add_argument("title", nargs="?", default="Frequency")
args = parser.parse_args() args = parser.parse_args()
df = load_data(args.input) df = load_data(args.input)
stats = calc_freq(df) stats = calculate_frequency(df)
fig = plot_freq(stats, title=args.title) fig = plot_frequency(stats, title=args.title)
config = { config = {
'responsive': True, 'responsive': True,
'displaylogo': False, 'displaylogo': False,
'scrollZoom': True, 'scrollZoom': True,
'modeBarButtonsToAdd': ['toggleSpikelines'], 'modeBarButtonsToAdd': ['toggleSpikelines'],
'toImageButtonOptions': {'format': 'png', 'scale': 2} 'toImageButtonOptions': {'format': 'png', 'scale': 2},
} }
fig.write_html(str(args.output), include_plotlyjs='cdn', config=config) fig.write_html(str(args.output), include_plotlyjs='cdn', config=config)
+140
View File
@@ -0,0 +1,140 @@
# File: id_viewer.py
# Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later
import argparse
from pathlib import Path
import numpy as np
import pandas as pd
import plotly.graph_objects as go
from utils.extractor import load_data
def _format_can_id_vec(s: pd.Series) -> pd.Series:
s = s.astype('string').str.strip()
s = s.str.replace(r'^0x', '', case=False, regex=True)
s = s.str.upper()
return s.fillna('UNKNOWN').replace('', 'UNKNOWN')
def prepare_data(df, target_id):
can_id_col = 'ID' if 'ID' in df.columns else 'Identifier'
formatted = _format_can_id_vec(df[can_id_col])
df = df.assign(Formatted_ID=formatted)
target_id_clean = _format_can_id_vec(pd.Series([target_id])).iloc[0]
filtered = df[df['Formatted_ID'] == target_id_clean]
byte_cols = [f"b{i}" for i in range(8) if f"b{i}" in filtered.columns]
if filtered.empty:
return filtered, byte_cols
for col in byte_cols:
if not pd.api.types.is_numeric_dtype(filtered[col]):
filtered = filtered.assign(**{col: pd.to_numeric(filtered[col], errors='coerce').astype('float32')})
filtered = filtered.sort_values('Timestamp', kind='stable')
arr = filtered[byte_cols].to_numpy(dtype=np.float32, copy=False)
if len(arr) > 1:
changed = np.any(arr[1:] != arr[:-1], axis=1)
keep = np.concatenate(([True], changed))
filtered = filtered.iloc[keep]
return filtered, byte_cols
def plot_bits(df, byte_cols, can_id, title):
fig = go.Figure()
colors = ['#e41a1c', '#377eb8', '#4daf4a', '#984ea3', '#ff7f00', '#ffff33', '#a65628', '#f781bf']
n = len(byte_cols)
x = df['Timestamp'].to_numpy() if not df.empty else np.array([])
for i, col in enumerate(byte_cols):
y = df[col].to_numpy(dtype=np.float32, copy=False) if not df.empty else np.array([])
fig.add_trace(go.Scattergl(
x=x,
y=y,
mode='lines',
line=dict(shape='hv', width=2, color=colors[i % len(colors)]),
name=col.upper(),
legendgroup=col.upper(),
hovertemplate=f"<b>{col.upper()}</b><br>Time: %{{x}}<br>Value: %{{y}}<extra></extra>",
))
all_button = dict(label='ALL', method='restyle', args=[{'visible': [True] * n}])
none_button = dict(label='NONE', method='restyle', args=[{'visible': ['legendonly'] * n}])
fig.update_layout(
height=600,
autosize=True,
template='plotly_white',
title=dict(
text=f"{title} - ID: {can_id}",
font=dict(size=20, color='#1a1a1a'),
x=0.5, xanchor='center',
pad=dict(b=20),
),
font=dict(family="Segoe UI, Arial, sans-serif", size=12, color='#2a2a2a'),
hoverlabel=dict(bgcolor="white", font_size=13, font_family="Segoe UI", bordercolor='#cccccc'),
margin=dict(l=60, r=40, t=120, b=140),
legend=dict(
orientation='h',
x=0.5, xanchor='center',
y=-0.18, yanchor='top',
title=None,
bgcolor='white',
bordercolor='#cccccc',
borderwidth=1,
font=dict(size=12, color="#2a2a2a"),
itemsizing='constant',
itemclick='toggle',
itemdoubleclick='toggleothers',
),
xaxis=dict(
title=dict(text="Timestamp", font=dict(size=13, color="#1a1a1a")),
showgrid=True, gridwidth=0.5, gridcolor='#e8e8e8',
zeroline=False, linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"),
ticks="outside", ticklen=4, tickcolor="#cccccc",
minor=dict(showgrid=True, gridcolor='#f4f4f4', gridwidth=0.5),
),
yaxis=dict(
title=dict(text="Byte Value", font=dict(size=13, color="#1a1a1a")),
showgrid=True, gridwidth=0.5, gridcolor='#e8e8e8',
zeroline=False, linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"),
ticks="outside", ticklen=4, tickcolor="#cccccc",
),
updatemenus=[
dict(
type='buttons',
direction='right',
x=0.5, xanchor='center',
y=-0.06, yanchor='top',
buttons=[all_button, none_button],
bgcolor='white',
bordercolor='#cccccc',
borderwidth=1,
font=dict(size=11, color='#2a2a2a'),
pad=dict(l=5, r=5, t=5, b=5),
)
],
)
return fig
if __name__ == "__main__":
parser = argparse.ArgumentParser(description="Visualize CAN bus byte changes over time")
parser.add_argument("input", type=Path, help="Path to the input CAN log file")
parser.add_argument("can_id", type=str, help="CAN ID to visualize")
parser.add_argument("output", type=Path, nargs="?", default=Path("bits_report.html"))
parser.add_argument("title", nargs="?", default="Byte Visualization")
args = parser.parse_args()
df = load_data(args.input)
filtered_df, byte_cols = prepare_data(df, args.can_id)
fig = plot_bits(filtered_df, byte_cols, args.can_id, title=args.title)
config = {
'responsive': True,
'displaylogo': False,
'scrollZoom': True,
'modeBarButtonsToAdd': ['toggleSpikelines'],
'toImageButtonOptions': {'format': 'png', 'scale': 2},
}
fig.write_html(str(args.output), include_plotlyjs='cdn', config=config)
+33 -6
View File
@@ -1,11 +1,15 @@
# File: extractor.py
# Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later
import json import json
from pathlib import Path from pathlib import Path
import numpy as np import numpy as np
import pandas as pd import pandas as pd
import polars as pl
def to_int(x): def to_int(x):
"""Convert a hex string or integer to int, returning NaN on failure."""
if isinstance(x, (int, np.integer)): if isinstance(x, (int, np.integer)):
return int(x) return int(x)
if isinstance(x, str): if isinstance(x, str):
@@ -16,7 +20,6 @@ def to_int(x):
return np.nan return np.nan
def extract_id(row: pd.Series) -> str: def extract_id(row: pd.Series) -> str:
"""Extracts PGN from metadata or falls back to CAN ID."""
meta = row.get('j1939_metadata') meta = row.get('j1939_metadata')
if pd.isna(meta): if pd.isna(meta):
return f"ID: {row['ID']}" return f"ID: {row['ID']}"
@@ -30,7 +33,31 @@ def extract_id(row: pd.Series) -> str:
return f"ID: {row['ID']}" return f"ID: {row['ID']}"
def load_data(file_path: Path) -> pd.DataFrame: def load_data(file_path: Path) -> pd.DataFrame:
"""Loads Parquet file and adds an Identifier column.""" lf = pl.scan_parquet(file_path)
df = pd.read_parquet(file_path) schema = lf.collect_schema()
df['Identifier'] = df.apply(extract_id, axis=1) names = schema.names()
return df
byte_cols = [f"b{i}" for i in range(8) if f"b{i}" in names]
if byte_cols:
lf = lf.with_columns([
pl.col(c).str.to_integer(base=16, strict=False).cast(pl.Int16).alias(c)
for c in byte_cols
])
id_col = 'ID' if 'ID' in names else 'Identifier'
id_expr = pl.col(id_col).cast(pl.Utf8)
if 'j1939_metadata' in names:
try:
lf = lf.with_columns(
pl.when(pl.col('j1939_metadata').is_not_null())
.then(pl.lit('PGN: ') + pl.col('j1939_metadata').struct.field('PGN').cast(pl.Utf8))
.otherwise(pl.lit('ID: ') + id_expr)
.alias('Identifier')
)
except Exception:
lf = lf.with_columns((pl.lit('ID: ') + id_expr).alias('Identifier'))
else:
lf = lf.with_columns((pl.lit('ID: ') + id_expr).alias('Identifier'))
return lf.collect().to_pandas()