13 Commits

Author SHA1 Message Date
eeeck 2408c7a963 Use better function names 2026-07-14 00:37:47 +02:00
eeeck 179ec6e56b Add header 2026-07-14 00:30:05 +02:00
eeeck d8105e2da3 Normalize CAN IDs number of bits 2026-07-14 00:21:03 +02:00
eeeck 9ddc6ea0d5 Use utility function instead of internal 2026-07-13 22:36:31 +02:00
eeeck df3e7b1f0e Update software version 2026-07-13 22:12:44 +02:00
eeeck fe4bc0e46a Merge pull request 'Implement Pearson and Spearman per-bit correleration' (#3) from dev-inter-byte-correlation into main
Reviewed-on: erickahmed/CANveyor#3
2026-07-13 21:54:38 +02:00
eeeck c9ae482174 Add positional argument to choose correlation methods (Pearson or
Spearman)
2026-07-13 21:53:20 +02:00
eeeck 5e6f81b50b Add hex converter utility
- To move to separate utility file in the future
2026-07-13 21:50:32 +02:00
eeeck ce74c7fcd9 Treat undefined correlation as zero correlation 2026-07-13 21:43:24 +02:00
eeeck 0f25c0671b Implement Pearson correlation 2026-07-13 21:43:17 +02:00
eeeck d0e70dad2a Remove leftovers 2026-07-13 21:20:07 +02:00
eeeck f39d23fc64 Merge pull request 'Implement entropy heatmap for CAN frames' (#2) from dev-entropy-heatmap into main
Reviewed-on: erickahmed/CANveyor#2
2026-07-13 21:17:33 +02:00
eeeck 7daf4c8e08 Add entropy heatmap analysis
- Same style of frequency analysis for consistency
2026-07-13 21:16:18 +02:00
7 changed files with 385 additions and 37 deletions
+7 -3
View File
@@ -19,7 +19,7 @@ def parse_log(input_path: PathLike, out_bus1: PathLike, out_bus2: PathLike) -> N
out1_file = Path(out_bus1)
out2_file = Path(out_bus2)
start_pattern = re.compile(r'(C[12]):([0-9A-Fa-f]{1,8})\s+([0-9A-Fa-f]{1,2})\s+')
start_pattern = re.compile(r'(C[12]):([0-9A-Fa-f]{7,8})\s+([0-9A-Fa-f]{1,2})\s+')
byte_pattern = re.compile(r'^[0-9A-Fa-f]{2}$')
with input_file.open('r', encoding='utf-8') as f_in, \
@@ -35,7 +35,12 @@ def parse_log(input_path: PathLike, out_bus1: PathLike, out_bus2: PathLike) -> N
for line in f_in:
for match in start_pattern.finditer(line):
bus = match.group(1)
can_id = match.group(2).upper()
can_id = match.group(2).upper().zfill(8)
if int(can_id, 16) > 0x1FFFFFFF:
continue
dlc_str = match.group(3)
try:
@@ -116,4 +121,3 @@ if __name__ == '__main__':
print(f"[*] Processing {args.input_csv}...")
lf = parse_csv(args.input_csv)
lf.sink_parquet(args.output_parquet)
print(f"[+] Saved parquet file to {args.output_parquet}")
+1 -1
View File
@@ -1,6 +1,6 @@
[project]
name = "CANveyor"
version = "0.0.1"
version = "0.0.4"
description = "J1939 CAN bus parser that works in pair with CANdigger"
readme = "README.md"
requires-python = ">=3.14"
+152
View File
@@ -0,0 +1,152 @@
# File: correlation.py
# Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later
import argparse
from pathlib import Path
import numpy as np
import pandas as pd
import plotly.graph_objects as go
from utils.extractor import load_data
from utils.extractor import to_int
def _format_can_id(x):
"""Safely cleans CAN ID strings without altering their length or value."""
if pd.isna(x):
return "UNKNOWN"
s = str(x).strip()
if not s:
return "UNKNOWN"
if s.lower().startswith('0x'):
s = s[2:]
return s.upper()
def calculate_correlation(df: pd.DataFrame, method: str, target_id: str | None = None) -> pd.DataFrame:
"""Calculates inter-byte correlation grouped by identifier."""
byte_cols = [f"b{i}" for i in range(8)]
available_cols = [col for col in byte_cols if col in df.columns]
if not available_cols:
raise ValueError("No byte columns (b0-b7) found in the DataFrame")
can_id_col = 'ID' if 'ID' in df.columns else 'Identifier'
identifiers = df[can_id_col].apply(_format_can_id)
df_bytes = df[available_cols].copy()
for col in available_cols:
df_bytes[col] = df_bytes[col].apply(to_int)
if target_id is not None:
target_id = _format_can_id(target_id)
group = df_bytes[identifiers == target_id]
if group.empty:
raise ValueError(f"Identifier '{target_id}' not found in data")
return group[available_cols].corr(method=method).fillna(0.0)
def max_abs_corr(group: pd.DataFrame) -> pd.Series:
corr_arr = np.abs(group.corr(method=method).to_numpy().copy())
np.fill_diagonal(corr_arr, 0.0)
return pd.Series(corr_arr.max(axis=0), index=group.columns).fillna(0.0)
result = df_bytes.groupby(identifiers)[available_cols].apply(max_abs_corr)
result.index.name = 'Identifier'
return result
def plot_correlation_heatmap(corr_df: pd.DataFrame, target_id: str | None, title: str) -> go.Figure:
"""Generates an interactive heatmap of inter-byte correlation."""
is_8x8 = target_id is not None
if is_8x8:
x = corr_df.columns.tolist()
y = corr_df.index.tolist()
z = corr_df.values
z_min, z_max = -1.0, 1.0
colorscale = [
[0.0, "#2c7bb6"], [0.25, "#abd9e9"], [0.5, "#ffffff"],
[0.75, "#fdae61"], [1.0, "#d7191c"]
]
hover_template = "<b>%{y}</b> vs <b>%{x}</b><br>Correlation: %{z:.2f}<extra></extra>"
else:
x = corr_df.columns.tolist()
y = corr_df.index.tolist()
z = corr_df.values
z_min, z_max = 0.0, 1.0
colorscale = [
[0.0, "#ffffff"], [0.2, "#fff5f0"], [0.4, "#fecc5c"],
[0.6, "#fd8d3c"], [0.8, "#e31a1c"], [1.0, "#800026"]
]
hover_template = "<b>%{y}</b><br>Byte %{x} max correlation: %{z:.2f}<extra></extra>"
fig = go.Figure(
data=go.Heatmap(
z=z, x=x, y=y,
zmin=z_min, zmax=z_max,
colorscale=colorscale,
xgap=3, ygap=3,
text=np.round(z, 2),
texttemplate="%{text}",
textfont={"size": 11, "color": "#2a2a2a", "family": "Segoe UI, Arial, sans-serif"},
hoverongaps=False,
hovertemplate=hover_template,
colorbar=dict(
title=dict(text="Correlation", side="top", font=dict(size=13, color="#1a1a1a")),
orientation="h", thickness=15, len=0.35,
x=1.0, xanchor="right", y=1.02, yanchor="bottom",
tickfont=dict(size=11, color="#2a2a2a"),
tickformat=".1f", outlinewidth=0.5, outlinecolor="#cccccc",
),
)
)
fig.update_layout(
title=dict(text=title, font=dict(size=20, color="#1a1a1a"), x=0.5, xanchor="center", pad=dict(b=20)),
height=max(600, len(y) * 28 + 150) if not is_8x8 else 600,
autosize=True,
template="plotly_white",
xaxis=dict(
title=dict(text="Byte Position", font=dict(size=13, color="#1a1a1a")),
side="top" if not is_8x8 else "bottom",
dtick=1, showgrid=False, linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"), ticks="outside", ticklen=4, tickcolor="#cccccc",
),
yaxis=dict(
title=dict(text="PGN or CAN ID" if not is_8x8 else "Byte Position", font=dict(size=13, color="#1a1a1a")),
autorange="reversed", showgrid=False, linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"), ticks="outside", ticklen=4, tickcolor="#cccccc", automargin=True,
),
font=dict(family="Segoe UI, Arial, sans-serif", size=12, color="#2a2a2a"),
hoverlabel=dict(bgcolor="white", font_size=13, font_family="Segoe UI", bordercolor="#cccccc"),
margin=dict(l=200, r=40, t=120, b=60),
)
return fig
if __name__ == "__main__":
parser = argparse.ArgumentParser(description="Analyze CAN bus inter-byte correlation")
parser.add_argument("method", choices=["pearson", "spearman"], help="Correlation method to use")
parser.add_argument("input", type=Path, help="Path to the input CAN log file")
parser.add_argument("output", type=Path, nargs="?", default=Path("correlation_report.html"), help="Path to the output HTML report")
parser.add_argument("title", nargs="?", default="CAN Bus Inter-Byte Correlation", help="Title for the HTML report")
parser.add_argument("--identifier", type=str, default=None, help="Specific PGN/CAN ID to analyze (e.g., 'PGN: 65331'). If omitted, shows max correlation per byte for all IDs.")
args = parser.parse_args()
df = load_data(args.input)
corr_df = calculate_correlation(df, method=args.method, target_id=args.identifier)
display_title = f"{args.title} ({args.identifier})" if args.identifier else args.title
fig = plot_correlation_heatmap(corr_df, target_id=args.identifier, title=display_title)
config = {
"responsive": True,
"displaylogo": False,
"scrollZoom": True,
"modeBarButtonsToAdd": ["toggleSpikelines"],
"toImageButtonOptions": {"format": "png", "scale": 2},
}
fig.write_html(str(args.output), include_plotlyjs="cdn", config=config)
+187
View File
@@ -0,0 +1,187 @@
# File: entropy.py
# Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later
import argparse
from pathlib import Path
import numpy as np
import pandas as pd
import plotly.graph_objects as go
from utils.extractor import load_data
from utils.extractor import to_int
def _format_can_id(x):
"""Safely cleans CAN ID strings without altering their length or value."""
if pd.isna(x):
return "UNKNOWN"
s = str(x).strip()
if not s:
return "UNKNOWN"
if s.lower().startswith('0x'):
s = s[2:]
return s.upper()
def calculate_byte_entropy(df: pd.DataFrame) -> pd.DataFrame:
"""Calculates Shannon entropy per byte position for each identifier."""
byte_cols = [f"b{i}" for i in range(8)]
available_cols = [col for col in byte_cols if col in df.columns]
if not available_cols:
raise ValueError("No byte columns (b0-b7) found in the DataFrame")
can_id_col = 'ID' if 'ID' in df.columns else 'Identifier'
identifiers = df[can_id_col].apply(_format_can_id)
df_bytes = df[available_cols].copy()
for col in available_cols:
df_bytes[col] = df_bytes[col].apply(to_int)
def entropy(s: pd.Series) -> float:
s = s.dropna()
if s.empty:
return 0.0
p = s.value_counts(normalize=True)
return -np.sum(p * np.log2(p))
result = df_bytes.groupby(identifiers)[available_cols].agg(entropy)
result.index.name = 'Identifier'
return result
def plot_entropy_heatmap(entropy_df: pd.DataFrame, title: str) -> go.Figure:
"""Generates an interactive heatmap of byte-level Shannon entropy."""
x = entropy_df.columns.tolist()
y = entropy_df.index.tolist()
z = entropy_df.values
fig = go.Figure(
data=go.Heatmap(
z=z,
x=x,
y=y,
colorscale=[
[0.0, "#ffffff"],
[0.15, "#fff7ec"],
[0.35, "#fee8c8"],
[0.55, "#fdd49e"],
[0.75, "#fdbb84"],
[1.0, "#ef6548"],
],
xgap=3,
ygap=3,
text=np.round(z, 2),
texttemplate="%{text}",
textfont={
"size": 11,
"color": "#2a2a2a",
"family": "Segoe UI, Arial, sans-serif",
},
hoverongaps=False,
hovertemplate=(
"<b>%{y}</b><br>"
"Byte %{x}: %{z:.2f} bits<extra></extra>"
),
colorbar=dict(
title=dict(
text="Entropy (bits)",
side="top",
font=dict(size=13, color="#1a1a1a"),
),
orientation="h",
thickness=15,
len=0.35,
x=1.0,
xanchor="right",
y=1.02,
yanchor="bottom",
tickfont=dict(size=11, color="#2a2a2a"),
tickformat=".1f",
outlinewidth=0.5,
outlinecolor="#cccccc",
),
)
)
fig.update_layout(
title=dict(
text=title,
font=dict(size=20, color="#1a1a1a"),
x=0.5,
xanchor="center",
pad=dict(b=20),
),
height=max(600, len(y) * 28 + 150),
autosize=True,
template="plotly_white",
xaxis=dict(
title=dict(text="Byte Position", font=dict(size=13, color="#1a1a1a")),
side="top",
dtick=1,
showgrid=False,
linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"),
ticks="outside",
ticklen=4,
tickcolor="#cccccc",
),
yaxis=dict(
title=dict(text="PGN or CAN ID", font=dict(size=13, color="#1a1a1a")),
autorange="reversed",
showgrid=False,
linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"),
ticks="outside",
ticklen=4,
tickcolor="#cccccc",
automargin=True,
),
font=dict(family="Segoe UI, Arial, sans-serif", size=12, color="#2a2a2a"),
hoverlabel=dict(
bgcolor="white",
font_size=13,
font_family="Segoe UI",
bordercolor="#cccccc",
),
margin=dict(l=200, r=40, t=120, b=60),
)
return fig
if __name__ == "__main__":
parser = argparse.ArgumentParser(
description="Analyze CAN bus byte-level entropy"
)
parser.add_argument(
"input", type=Path, help="Path to the input CAN log file"
)
parser.add_argument(
"output",
type=Path,
nargs="?",
default=Path("entropy_report.html"),
help="Path to the output HTML report",
)
parser.add_argument(
"title",
nargs="?",
default="CAN Bus Byte-Level Entropy",
help="Title for the HTML report",
)
args = parser.parse_args()
df = load_data(args.input)
entropy_df = calculate_byte_entropy(df)
fig = plot_entropy_heatmap(entropy_df, title=args.title)
config = {
"responsive": True,
"displaylogo": False,
"scrollZoom": True,
"modeBarButtonsToAdd": ["toggleSpikelines"],
"toImageButtonOptions": {"format": "png", "scale": 2},
}
fig.write_html(str(args.output), include_plotlyjs="cdn", config=config)
+25 -6
View File
@@ -9,15 +9,34 @@ import plotly.express as px
import plotly.graph_objects as go
from utils.extractor import load_data
def calc_freq(df: pd.DataFrame) -> pd.DataFrame:
def _format_can_id(x):
"""Safely cleans CAN ID strings without altering their length or value."""
if pd.isna(x):
return "UNKNOWN"
s = str(x).strip()
if not s:
return "UNKNOWN"
if s.lower().startswith('0x'):
s = s[2:]
return s.upper()
def calculate_frequency(df: pd.DataFrame) -> pd.DataFrame:
"""Calculates frequency counts and percentages for identifiers."""
freq_df = df['Identifier'].value_counts().reset_index()
can_id_col = 'ID' if 'ID' in df.columns else 'Identifier'
df['Formatted_ID'] = df[can_id_col].apply(_format_can_id)
freq_df = df['Formatted_ID'].value_counts().reset_index()
freq_df.columns = ['Identifier', 'Count']
total = freq_df['Count'].sum()
freq_df['Percentage'] = (freq_df['Count'] / total * 100).round(2)
return freq_df.sort_values('Count', ascending=True)
def plot_freq(stats_df: pd.DataFrame, title: str) -> go.Figure:
def plot_frequency(stats_df: pd.DataFrame, title: str) -> go.Figure:
"""Generates interactive horizontal bar chart with log x-axis."""
fig = px.bar(
stats_df, y='Identifier', x='Count', orientation='h', title=title, log_x=True,
@@ -43,7 +62,6 @@ def plot_freq(stats_df: pd.DataFrame, title: str) -> go.Figure:
),
yaxis=dict(
title=dict(text="PGN or CAN ID", font=dict(size=13, color="#1a1a1a")),
#autorange="",
showgrid=False,
linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"),
@@ -51,6 +69,7 @@ def plot_freq(stats_df: pd.DataFrame, title: str) -> go.Figure:
ticklen=4,
tickcolor="#cccccc",
automargin=True,
type='category'
),
font=dict(family="Segoe UI, Arial, sans-serif", size=12, color='#2a2a2a'),
hoverlabel=dict(bgcolor="white", font_size=13, font_family="Segoe UI",
@@ -104,8 +123,8 @@ if __name__ == "__main__":
args = parser.parse_args()
df = load_data(args.input)
stats = calc_freq(df)
fig = plot_freq(stats, title=args.title)
stats = calculate_frequency(df)
fig = plot_frequency(stats, title=args.title)
config = {
'responsive': True,
'displaylogo': False,
+13
View File
@@ -4,8 +4,21 @@
import json
from pathlib import Path
import numpy as np
import pandas as pd
def to_int(x):
"""Convert a hex string or integer to int, returning NaN on failure."""
if isinstance(x, (int, np.integer)):
return int(x)
if isinstance(x, str):
try:
return int(x, 16)
except ValueError:
return np.nan
return np.nan
def extract_id(row: pd.Series) -> str:
"""Extracts PGN from metadata or falls back to CAN ID."""
meta = row.get('j1939_metadata')
-27
View File
@@ -1,27 +0,0 @@
# File: extractor.py
# Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later
import json
from pathlib import Path
import pandas as pd
def extract_id(row: pd.Series) -> str:
"""Extracts PGN from metadata or falls back to CAN ID."""
meta = row.get('j1939_metadata')
if pd.isna(meta):
return f"ID: {row['ID']}"
if isinstance(meta, str):
try:
meta = json.loads(meta)
except json.JSONDecodeError:
return f"ID: {row['ID']}"
if isinstance(meta, dict) and 'PGN' in meta:
return f"PGN: {meta['PGN']}"
return f"ID: {row['ID']}"
def load_data(file_path: Path) -> pd.DataFrame:
"""Loads Parquet file and adds an Identifier column."""
df = pd.read_parquet(file_path)
df['Identifier'] = df.apply(extract_id, axis=1)
return df