Compare commits
1 Commits
v0.0.4
..
4be733d6e3
| Author | SHA1 | Date | |
|---|---|---|---|
| 4be733d6e3 |
@@ -1 +0,0 @@
|
||||
3.14
|
||||
@@ -1,9 +0,0 @@
|
||||
# Copyright
|
||||
|
||||
Copyright © 2026 Erick Ahmed
|
||||
|
||||
The source code in this repository is licensed under the **GNU Affero General Public License v3.0 or later (AGPL-3.0-or-later)**.
|
||||
|
||||
A copy of the license is provided in the `LICENSE` file. If any discrepancy exists between this notice and the `LICENSE` file, the `LICENSE` file shall prevail.
|
||||
|
||||
Any third-party components included in this repository at any point during developement remain the property of their respective copyright holders and are subject to their own license terms.
|
||||
@@ -633,8 +633,8 @@ the "copyright" line and a pointer to where the full notice is found.
|
||||
Copyright (C) <year> <name of author>
|
||||
|
||||
This program is free software: you can redistribute it and/or modify
|
||||
it under the terms of the GNU Affero General Public License as published
|
||||
by the Free Software Foundation, either version 3 of the License, or
|
||||
it under the terms of the GNU Affero General Public License as published by
|
||||
the Free Software Foundation, either version 3 of the License, or
|
||||
(at your option) any later version.
|
||||
|
||||
This program is distributed in the hope that it will be useful,
|
||||
|
||||
-86
@@ -1,86 +0,0 @@
|
||||
# File: decoder.py
|
||||
# Copyright (C) 2026 Erick Ahmed
|
||||
# SPDX-License-Identifier: AGPL-3.0-or-later
|
||||
|
||||
import argparse
|
||||
import polars as pl
|
||||
|
||||
def get_j1939_mask() -> pl.Expr:
|
||||
"""
|
||||
Returns a Polars expression representing the strict J1939 filtering rules.
|
||||
"""
|
||||
id_int = pl.col("ID").str.to_integer(base=16).cast(pl.UInt32)
|
||||
return id_int > 0x7FF
|
||||
|
||||
def decode_j1939_metadata(lf: pl.LazyFrame) -> pl.LazyFrame:
|
||||
"""
|
||||
Decodes J1939 fields and bundles them into a Struct column.
|
||||
"""
|
||||
id_int = pl.col("ID").str.to_integer(base=16).cast(pl.UInt32)
|
||||
|
||||
priority = ((id_int // 67108864) % 8).cast(pl.UInt8)
|
||||
pf = ((id_int // 65536) % 256).cast(pl.UInt8)
|
||||
ps = ((id_int // 256) % 256).cast(pl.UInt8)
|
||||
sa = (id_int % 256).cast(pl.UInt8)
|
||||
|
||||
da = pl.when(pf < 240).then(ps).otherwise(pl.lit(255, dtype=pl.UInt8)).cast(pl.UInt8)
|
||||
|
||||
pgn = pl.when(pf < 240).then(
|
||||
((id_int // 256) & 0x3FF00)
|
||||
).otherwise(
|
||||
((id_int // 256) & 0x3FFFF)
|
||||
).cast(pl.UInt32)
|
||||
|
||||
return lf.with_columns(
|
||||
pl.struct([
|
||||
priority.alias("Priority"),
|
||||
pf.alias("PF"),
|
||||
ps.alias("PS"),
|
||||
sa.alias("SA"),
|
||||
da.alias("DA"),
|
||||
pgn.alias("PGN")
|
||||
]).alias("j1939_metadata")
|
||||
)
|
||||
|
||||
def decode_j1939_frames(df: pl.DataFrame) -> pl.DataFrame:
|
||||
id_int = pl.col("ID").str.to_integer(base=16).cast(pl.UInt32)
|
||||
|
||||
is_j1939 = id_int > 0x7FF
|
||||
|
||||
priority = ((id_int // 67108864) % 8).cast(pl.UInt8)
|
||||
pf = ((id_int // 65536) % 256).cast(pl.UInt8)
|
||||
ps = ((id_int // 256) % 256).cast(pl.UInt8)
|
||||
sa = (id_int % 256).cast(pl.UInt8)
|
||||
|
||||
da = pl.when(pf < 240).then(ps).otherwise(pl.lit(255, dtype=pl.UInt8)).cast(pl.UInt8)
|
||||
|
||||
pgn = pl.when(pf < 240).then(
|
||||
((id_int // 256) & 0x3FF00)
|
||||
).otherwise(
|
||||
((id_int // 256) & 0x3FFFF)
|
||||
).cast(pl.UInt32)
|
||||
|
||||
j1939_meta = pl.when(is_j1939).then(
|
||||
pl.struct([
|
||||
priority.alias("Priority"),
|
||||
pf.alias("PF"),
|
||||
ps.alias("PS"),
|
||||
sa.alias("SA"),
|
||||
da.alias("DA"),
|
||||
pgn.alias("PGN")
|
||||
])
|
||||
).otherwise(None)
|
||||
|
||||
return df.with_columns(j1939_meta.alias("j1939_metadata"))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
parser = argparse.ArgumentParser(description="J1939 decoder")
|
||||
parser.add_argument("input_parquet", help="Path to the raw .parquet file")
|
||||
parser.add_argument("output_parquet", help="Path to save the decoded .parquet file")
|
||||
args = parser.parse_args()
|
||||
|
||||
df = pl.scan_parquet(args.input_parquet).collect()
|
||||
decoded_df = decode_j1939_frames(df)
|
||||
|
||||
decoded_df.write_parquet(args.output_parquet)
|
||||
@@ -1,3 +0,0 @@
|
||||
# File: main.py
|
||||
# Copyright (C) 2026 Erick Ahmed
|
||||
# SPDX-License-Identifier: AGPL-3.0-or-later
|
||||
@@ -1,25 +1,21 @@
|
||||
# File: parser.py
|
||||
# Copyright (C) 2026 Erick Ahmed
|
||||
# SPDX-License-Identifier: AGPL-3.0-or-later
|
||||
|
||||
import re
|
||||
import csv
|
||||
import polars as pl
|
||||
from pathlib import Path
|
||||
from typing import Union
|
||||
|
||||
PathLike = Union[str, Path]
|
||||
|
||||
def parse_log(input_path: PathLike, out_bus1: PathLike, out_bus2: PathLike) -> None:
|
||||
def parse_can_log(input_path: str | Path, out_bus1: str | Path, out_bus2: str | Path) -> None:
|
||||
"""
|
||||
Parses a raw CAN bus log from CANdigger using regex and saves valid frames to separate CSV.
|
||||
Parses a CAN bus log file from CANdigger and saves valid frames to separate CSV files for Bus 1 and Bus 2.
|
||||
Corrupted, incomplete, or debug frames are silently discarded.
|
||||
"""
|
||||
input_file = Path(input_path)
|
||||
out1_file = Path(out_bus1)
|
||||
out2_file = Path(out_bus2)
|
||||
|
||||
# Group 1: Bus (C1 or C2)
|
||||
# Group 2: ID (1 to 8 hex chars)
|
||||
# Group 3: DLC (1 to 2 hex chars)
|
||||
start_pattern = re.compile(r'(C[12]):([0-9A-Fa-f]{1,8})\s+([0-9A-Fa-f]{1,2})\s+')
|
||||
|
||||
byte_pattern = re.compile(r'^[0-9A-Fa-f]{2}$')
|
||||
|
||||
with input_file.open('r', encoding='utf-8') as f_in, \
|
||||
@@ -65,55 +61,21 @@ def parse_log(input_path: PathLike, out_bus1: PathLike, out_bus2: PathLike) -> N
|
||||
elif bus == 'C2':
|
||||
writer2.writerow(row)
|
||||
|
||||
def parse_csv(csv_path: PathLike) -> pl.LazyFrame:
|
||||
"""
|
||||
Ingests a parsed CSV file, unpacks hex strings into 8 hex columns,
|
||||
generates sequential timestamps if missing, and returns a Polars LazyFrame.
|
||||
"""
|
||||
lf = pl.scan_csv(csv_path, schema_overrides={"ID": pl.String, "Data": pl.String})
|
||||
|
||||
if "Timestamp" not in lf.columns:
|
||||
lf = lf.with_row_index("Timestamp")
|
||||
|
||||
byte_exprs = []
|
||||
for i in range(8):
|
||||
expr = (
|
||||
pl.col("Data").str.strip_chars().str.split(" ")
|
||||
.list.get(i, null_on_oob=True)
|
||||
.alias(f"b{i}")
|
||||
)
|
||||
byte_exprs.append(expr)
|
||||
|
||||
lf = lf.with_columns(byte_exprs).drop("Data")
|
||||
|
||||
return lf.with_columns([
|
||||
pl.col("DLC").cast(pl.UInt8),
|
||||
pl.col("Timestamp").cast(pl.Float64)
|
||||
])
|
||||
if __name__ == '__main__':
|
||||
# Example usage:
|
||||
# parse_can_log('can_traffic.txt', 'bus1_output.csv', 'bus2_output.csv')
|
||||
# or on CLI:
|
||||
# python3 can_parser.py can_traffic.txt' bus1_output.csv bus2_output.csv
|
||||
|
||||
if __name__ == '__main__':
|
||||
import argparse
|
||||
|
||||
parser = argparse.ArgumentParser(description="CAN Bus Data Engine & Parser")
|
||||
subparsers = parser.add_subparsers(dest="command", required=True, help="Available commands")
|
||||
|
||||
parser_csv = subparsers.add_parser("csv", help="Parse raw text log into Bus 1 and Bus 2 CSVs")
|
||||
parser_csv.add_argument("input", help="Path to the .txt log file from CANdigger")
|
||||
parser_csv.add_argument("out_bus1", help="Output CSV filename for Bus 1 (C1)")
|
||||
parser_csv.add_argument("out_bus2", help="Output CSV filename for Bus 2 (C2)")
|
||||
|
||||
parser_parquet = subparsers.add_parser("parquet", help="Convert a parsed CSV into an optimized Parquet file")
|
||||
parser_parquet.add_argument("input_csv", help="Path to the input .csv file")
|
||||
parser_parquet.add_argument("output_parquet", help="Path to the output .parquet file")
|
||||
parser = argparse.ArgumentParser(description="Parse CAN bus logs to separate CSV files")
|
||||
parser.add_argument("input", help="Path to the input .txt log file")
|
||||
parser.add_argument("out_bus1", help="Output CSV filename for Bus 1 (C1)")
|
||||
parser.add_argument("out_bus2", help="Output CSV filename for Bus 2 (C2)")
|
||||
|
||||
args = parser.parse_args()
|
||||
parse_can_log(args.input, args.out_bus1, args.out_bus2)
|
||||
|
||||
if args.command == "csv":
|
||||
parse_log(args.input, args.out_bus1, args.out_bus2)
|
||||
print(f"[+] Saved csv file to {args.out_bus1} and {args.out_bus2}")
|
||||
|
||||
elif args.command == "parquet":
|
||||
print(f"[*] Processing {args.input_csv}...")
|
||||
lf = parse_csv(args.input_csv)
|
||||
lf.sink_parquet(args.output_parquet)
|
||||
print(f"[+] Saved parquet file to {args.output_parquet}")
|
||||
pass
|
||||
|
||||
@@ -1,7 +0,0 @@
|
||||
[project]
|
||||
name = "CANveyor"
|
||||
version = "0.0.1"
|
||||
description = "J1939 CAN bus parser that works in pair with CANdigger"
|
||||
readme = "README.md"
|
||||
requires-python = ">=3.14"
|
||||
dependencies = ["polars", "pathlib", "typing"]
|
||||
@@ -1,144 +0,0 @@
|
||||
# File: correlation.py
|
||||
# Copyright (C) 2026 Erick Ahmed
|
||||
# SPDX-License-Identifier: AGPL-3.0-or-later
|
||||
|
||||
import argparse
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import plotly.graph_objects as go
|
||||
|
||||
from utils.extractor import load_data
|
||||
|
||||
|
||||
def _to_int(x):
|
||||
"""Convert a hex string or integer to int, returning NaN on failure."""
|
||||
if isinstance(x, (int, np.integer)):
|
||||
return int(x)
|
||||
if isinstance(x, str):
|
||||
try:
|
||||
return int(x, 16)
|
||||
except ValueError:
|
||||
return np.nan
|
||||
return np.nan
|
||||
|
||||
|
||||
def calculate_correlation(df: pd.DataFrame, method: str, target_id: str | None = None) -> pd.DataFrame:
|
||||
"""Calculates inter-byte correlation grouped by identifier."""
|
||||
byte_cols = [f"b{i}" for i in range(8)]
|
||||
available_cols = [col for col in byte_cols if col in df.columns]
|
||||
|
||||
if not available_cols:
|
||||
raise ValueError("No byte columns (b0-b7) found in the DataFrame")
|
||||
|
||||
df_bytes = df[["Identifier"] + available_cols].copy()
|
||||
for col in available_cols:
|
||||
df_bytes[col] = df_bytes[col].apply(_to_int)
|
||||
|
||||
if target_id:
|
||||
group = df_bytes[df_bytes["Identifier"] == target_id]
|
||||
if group.empty:
|
||||
raise ValueError(f"Identifier '{target_id}' not found in data")
|
||||
return group[available_cols].corr(method=method).fillna(0.0)
|
||||
|
||||
def max_abs_corr(group: pd.DataFrame) -> pd.Series:
|
||||
corr_arr = np.abs(group.corr(method=method).to_numpy().copy())
|
||||
np.fill_diagonal(corr_arr, 0.0)
|
||||
return pd.Series(corr_arr.max(axis=0), index=group.columns).fillna(0.0)
|
||||
|
||||
return df_bytes.groupby("Identifier")[available_cols].apply(max_abs_corr)
|
||||
|
||||
|
||||
def plot_correlation_heatmap(corr_df: pd.DataFrame, target_id: str | None, title: str) -> go.Figure:
|
||||
"""Generates an interactive heatmap of inter-byte correlation."""
|
||||
is_8x8 = target_id is not None
|
||||
|
||||
if is_8x8:
|
||||
x = corr_df.columns.tolist()
|
||||
y = corr_df.index.tolist()
|
||||
z = corr_df.values
|
||||
z_min, z_max = -1.0, 1.0
|
||||
colorscale = [
|
||||
[0.0, "#2c7bb6"], [0.25, "#abd9e9"], [0.5, "#ffffff"],
|
||||
[0.75, "#fdae61"], [1.0, "#d7191c"]
|
||||
]
|
||||
hover_template = "<b>%{y}</b> vs <b>%{x}</b><br>Correlation: %{z:.2f}<extra></extra>"
|
||||
else:
|
||||
x = corr_df.columns.tolist()
|
||||
y = corr_df.index.tolist()
|
||||
z = corr_df.values
|
||||
z_min, z_max = 0.0, 1.0
|
||||
colorscale = [
|
||||
[0.0, "#ffffff"], [0.2, "#fff5f0"], [0.4, "#fecc5c"],
|
||||
[0.6, "#fd8d3c"], [0.8, "#e31a1c"], [1.0, "#800026"]
|
||||
]
|
||||
hover_template = "<b>%{y}</b><br>Byte %{x} max correlation: %{z:.2f}<extra></extra>"
|
||||
|
||||
fig = go.Figure(
|
||||
data=go.Heatmap(
|
||||
z=z, x=x, y=y,
|
||||
zmin=z_min, zmax=z_max,
|
||||
colorscale=colorscale,
|
||||
xgap=3, ygap=3,
|
||||
text=np.round(z, 2),
|
||||
texttemplate="%{text}",
|
||||
textfont={"size": 11, "color": "#2a2a2a", "family": "Segoe UI, Arial, sans-serif"},
|
||||
hoverongaps=False,
|
||||
hovertemplate=hover_template,
|
||||
colorbar=dict(
|
||||
title=dict(text="Correlation", side="top", font=dict(size=13, color="#1a1a1a")),
|
||||
orientation="h", thickness=15, len=0.35,
|
||||
x=1.0, xanchor="right", y=1.02, yanchor="bottom",
|
||||
tickfont=dict(size=11, color="#2a2a2a"),
|
||||
tickformat=".1f", outlinewidth=0.5, outlinecolor="#cccccc",
|
||||
),
|
||||
)
|
||||
)
|
||||
|
||||
fig.update_layout(
|
||||
title=dict(text=title, font=dict(size=20, color="#1a1a1a"), x=0.5, xanchor="center", pad=dict(b=20)),
|
||||
height=max(600, len(y) * 28 + 150) if not is_8x8 else 600,
|
||||
autosize=True,
|
||||
template="plotly_white",
|
||||
xaxis=dict(
|
||||
title=dict(text="Byte Position", font=dict(size=13, color="#1a1a1a")),
|
||||
side="top" if not is_8x8 else "bottom",
|
||||
dtick=1, showgrid=False, linecolor="#bdbdbd",
|
||||
tickfont=dict(size=12, color="#2a2a2a"), ticks="outside", ticklen=4, tickcolor="#cccccc",
|
||||
),
|
||||
yaxis=dict(
|
||||
title=dict(text="PGN or CAN ID" if not is_8x8 else "Byte Position", font=dict(size=13, color="#1a1a1a")),
|
||||
autorange="reversed", showgrid=False, linecolor="#bdbdbd",
|
||||
tickfont=dict(size=12, color="#2a2a2a"), ticks="outside", ticklen=4, tickcolor="#cccccc", automargin=True,
|
||||
),
|
||||
font=dict(family="Segoe UI, Arial, sans-serif", size=12, color="#2a2a2a"),
|
||||
hoverlabel=dict(bgcolor="white", font_size=13, font_family="Segoe UI", bordercolor="#cccccc"),
|
||||
margin=dict(l=200, r=40, t=120, b=60),
|
||||
)
|
||||
return fig
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
parser = argparse.ArgumentParser(description="Analyze CAN bus inter-byte correlation")
|
||||
parser.add_argument("method", choices=["pearson", "spearman"], help="Correlation method to use")
|
||||
parser.add_argument("input", type=Path, help="Path to the input CAN log file")
|
||||
parser.add_argument("output", type=Path, nargs="?", default=Path("correlation_report.html"), help="Path to the output HTML report")
|
||||
parser.add_argument("title", nargs="?", default="CAN Bus Inter-Byte Correlation", help="Title for the HTML report")
|
||||
parser.add_argument("--identifier", type=str, default=None, help="Specific PGN/CAN ID to analyze (e.g., 'PGN: 65331'). If omitted, shows max correlation per byte for all IDs.")
|
||||
args = parser.parse_args()
|
||||
|
||||
df = load_data(args.input)
|
||||
corr_df = calculate_correlation(df, method=args.method, target_id=args.identifier)
|
||||
|
||||
display_title = f"{args.title} ({args.identifier})" if args.identifier else args.title
|
||||
fig = plot_correlation_heatmap(corr_df, target_id=args.identifier, title=display_title)
|
||||
|
||||
config = {
|
||||
"responsive": True,
|
||||
"displaylogo": False,
|
||||
"scrollZoom": True,
|
||||
"modeBarButtonsToAdd": ["toggleSpikelines"],
|
||||
"toImageButtonOptions": {"format": "png", "scale": 2},
|
||||
}
|
||||
fig.write_html(str(args.output), include_plotlyjs="cdn", config=config)
|
||||
-180
@@ -1,180 +0,0 @@
|
||||
# File: entropy.py
|
||||
# Copyright (C) 2026 Erick Ahmed
|
||||
# SPDX-License-Identifier: AGPL-3.0-or-later
|
||||
|
||||
import argparse
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import plotly.graph_objects as go
|
||||
|
||||
from utils.extractor import load_data
|
||||
|
||||
|
||||
def _to_int(x):
|
||||
"""Convert a hex string or integer to int, returning NaN on failure."""
|
||||
if isinstance(x, (int, np.integer)):
|
||||
return int(x)
|
||||
if isinstance(x, str):
|
||||
try:
|
||||
return int(x, 16)
|
||||
except ValueError:
|
||||
return np.nan
|
||||
return np.nan
|
||||
|
||||
|
||||
def calculate_byte_entropy(df: pd.DataFrame) -> pd.DataFrame:
|
||||
"""Calculates Shannon entropy per byte position for each identifier."""
|
||||
byte_cols = [f"b{i}" for i in range(8)]
|
||||
available_cols = [col for col in byte_cols if col in df.columns]
|
||||
if not available_cols:
|
||||
raise ValueError("No byte columns (b0-b7) found in the DataFrame")
|
||||
|
||||
df_bytes = df[available_cols].copy()
|
||||
for col in available_cols:
|
||||
df_bytes[col] = df_bytes[col].apply(_to_int)
|
||||
|
||||
def entropy(s: pd.Series) -> float:
|
||||
s = s.dropna()
|
||||
if s.empty:
|
||||
return 0.0
|
||||
p = s.value_counts(normalize=True)
|
||||
return -np.sum(p * np.log2(p))
|
||||
|
||||
return df.groupby("Identifier")[available_cols].agg(entropy)
|
||||
|
||||
|
||||
def plot_entropy_heatmap(entropy_df: pd.DataFrame, title: str) -> go.Figure:
|
||||
"""Generates an interactive heatmap of byte-level Shannon entropy."""
|
||||
x = entropy_df.columns.tolist()
|
||||
y = entropy_df.index.tolist()
|
||||
z = entropy_df.values
|
||||
|
||||
fig = go.Figure(
|
||||
data=go.Heatmap(
|
||||
z=z,
|
||||
x=x,
|
||||
y=y,
|
||||
colorscale=[
|
||||
[0.0, "#ffffff"],
|
||||
[0.15, "#fff7ec"],
|
||||
[0.35, "#fee8c8"],
|
||||
[0.55, "#fdd49e"],
|
||||
[0.75, "#fdbb84"],
|
||||
[1.0, "#ef6548"],
|
||||
],
|
||||
xgap=3,
|
||||
ygap=3,
|
||||
text=np.round(z, 2),
|
||||
texttemplate="%{text}",
|
||||
textfont={
|
||||
"size": 11,
|
||||
"color": "#2a2a2a",
|
||||
"family": "Segoe UI, Arial, sans-serif",
|
||||
},
|
||||
hoverongaps=False,
|
||||
hovertemplate=(
|
||||
"<b>%{y}</b><br>"
|
||||
"Byte %{x}: %{z:.2f} bits<extra></extra>"
|
||||
),
|
||||
colorbar=dict(
|
||||
title=dict(
|
||||
text="Entropy (bits)",
|
||||
side="top",
|
||||
font=dict(size=13, color="#1a1a1a"),
|
||||
),
|
||||
orientation="h",
|
||||
thickness=15,
|
||||
len=0.35,
|
||||
x=1.0,
|
||||
xanchor="right",
|
||||
y=1.02,
|
||||
yanchor="bottom",
|
||||
tickfont=dict(size=11, color="#2a2a2a"),
|
||||
tickformat=".1f",
|
||||
outlinewidth=0.5,
|
||||
outlinecolor="#cccccc",
|
||||
),
|
||||
)
|
||||
)
|
||||
|
||||
fig.update_layout(
|
||||
title=dict(
|
||||
text=title,
|
||||
font=dict(size=20, color="#1a1a1a"),
|
||||
x=0.5,
|
||||
xanchor="center",
|
||||
pad=dict(b=20),
|
||||
),
|
||||
height=max(600, len(y) * 28 + 150),
|
||||
autosize=True,
|
||||
template="plotly_white",
|
||||
xaxis=dict(
|
||||
title=dict(text="Byte Position", font=dict(size=13, color="#1a1a1a")),
|
||||
side="top",
|
||||
dtick=1,
|
||||
showgrid=False,
|
||||
linecolor="#bdbdbd",
|
||||
tickfont=dict(size=12, color="#2a2a2a"),
|
||||
ticks="outside",
|
||||
ticklen=4,
|
||||
tickcolor="#cccccc",
|
||||
),
|
||||
yaxis=dict(
|
||||
title=dict(text="PGN or CAN ID", font=dict(size=13, color="#1a1a1a")),
|
||||
autorange="reversed",
|
||||
showgrid=False,
|
||||
linecolor="#bdbdbd",
|
||||
tickfont=dict(size=12, color="#2a2a2a"),
|
||||
ticks="outside",
|
||||
ticklen=4,
|
||||
tickcolor="#cccccc",
|
||||
automargin=True,
|
||||
),
|
||||
font=dict(family="Segoe UI, Arial, sans-serif", size=12, color="#2a2a2a"),
|
||||
hoverlabel=dict(
|
||||
bgcolor="white",
|
||||
font_size=13,
|
||||
font_family="Segoe UI",
|
||||
bordercolor="#cccccc",
|
||||
),
|
||||
margin=dict(l=200, r=40, t=120, b=60),
|
||||
)
|
||||
return fig
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Analyze CAN bus byte-level entropy"
|
||||
)
|
||||
parser.add_argument(
|
||||
"input", type=Path, help="Path to the input CAN log file"
|
||||
)
|
||||
parser.add_argument(
|
||||
"output",
|
||||
type=Path,
|
||||
nargs="?",
|
||||
default=Path("entropy_report.html"),
|
||||
help="Path to the output HTML report",
|
||||
)
|
||||
parser.add_argument(
|
||||
"title",
|
||||
nargs="?",
|
||||
default="CAN Bus Byte-Level Entropy",
|
||||
help="Title for the HTML report",
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
df = load_data(args.input)
|
||||
entropy_df = calculate_byte_entropy(df)
|
||||
fig = plot_entropy_heatmap(entropy_df, title=args.title)
|
||||
|
||||
config = {
|
||||
"responsive": True,
|
||||
"displaylogo": False,
|
||||
"scrollZoom": True,
|
||||
"modeBarButtonsToAdd": ["toggleSpikelines"],
|
||||
"toImageButtonOptions": {"format": "png", "scale": 2},
|
||||
}
|
||||
fig.write_html(str(args.output), include_plotlyjs="cdn", config=config)
|
||||
@@ -1,116 +0,0 @@
|
||||
# File: frequency.py
|
||||
# Copyright (C) 2026 Erick Ahmed
|
||||
# SPDX-License-Identifier: AGPL-3.0-or-later
|
||||
|
||||
import argparse
|
||||
from pathlib import Path
|
||||
import pandas as pd
|
||||
import plotly.express as px
|
||||
import plotly.graph_objects as go
|
||||
from utils.extractor import load_data
|
||||
|
||||
def calc_freq(df: pd.DataFrame) -> pd.DataFrame:
|
||||
"""Calculates frequency counts and percentages for identifiers."""
|
||||
freq_df = df['Identifier'].value_counts().reset_index()
|
||||
freq_df.columns = ['Identifier', 'Count']
|
||||
total = freq_df['Count'].sum()
|
||||
freq_df['Percentage'] = (freq_df['Count'] / total * 100).round(2)
|
||||
return freq_df.sort_values('Count', ascending=True)
|
||||
|
||||
def plot_freq(stats_df: pd.DataFrame, title: str) -> go.Figure:
|
||||
"""Generates interactive horizontal bar chart with log x-axis."""
|
||||
fig = px.bar(
|
||||
stats_df, y='Identifier', x='Count', orientation='h', title=title, log_x=True,
|
||||
labels={'Identifier': 'PGN / CAN ID', 'Count': 'Message Count'},
|
||||
color='Count', color_continuous_scale='Turbo',
|
||||
range_color=(stats_df['Count'].min(), stats_df['Count'].max()),
|
||||
hover_data={'Percentage': ':.2f', 'Count': ':,', 'Identifier': True}
|
||||
)
|
||||
fig.update_layout(
|
||||
height=max(600, len(stats_df) * 18),
|
||||
autosize=True,
|
||||
template='plotly_white',
|
||||
xaxis=dict(
|
||||
title=dict(text="Message count [log scale]", font=dict(size=13, color="#1a1a1a")),
|
||||
side="top",
|
||||
dtick=1,
|
||||
showgrid=False,
|
||||
linecolor="#bdbdbd",
|
||||
tickfont=dict(size=12, color="#2a2a2a"),
|
||||
ticks="outside",
|
||||
ticklen=4,
|
||||
tickcolor="#cccccc",
|
||||
),
|
||||
yaxis=dict(
|
||||
title=dict(text="PGN or CAN ID", font=dict(size=13, color="#1a1a1a")),
|
||||
#autorange="",
|
||||
showgrid=False,
|
||||
linecolor="#bdbdbd",
|
||||
tickfont=dict(size=12, color="#2a2a2a"),
|
||||
ticks="outside",
|
||||
ticklen=4,
|
||||
tickcolor="#cccccc",
|
||||
automargin=True,
|
||||
),
|
||||
font=dict(family="Segoe UI, Arial, sans-serif", size=12, color='#2a2a2a'),
|
||||
hoverlabel=dict(bgcolor="white", font_size=13, font_family="Segoe UI",
|
||||
bordercolor='#cccccc'),
|
||||
margin=dict(l=200, r=40, t=120, b=60),
|
||||
bargap=0.35,
|
||||
coloraxis_colorbar=dict(
|
||||
title=dict(text='Message Count', side='top'),
|
||||
orientation='h',
|
||||
thickness=15,
|
||||
len=0.35,
|
||||
x=1.0,
|
||||
xanchor='right',
|
||||
y=1.02,
|
||||
yanchor='bottom',
|
||||
tickformat=',',
|
||||
outlinecolor='#cccccc',
|
||||
outlinewidth=0.5
|
||||
),
|
||||
title=dict(font=dict(size=20, color='#1a1a1a'), x=0.5, xanchor='center',
|
||||
pad=dict(b=20))
|
||||
)
|
||||
fig.update_xaxes(
|
||||
showgrid=True, gridwidth=0.5, gridcolor='#e8e8e8',
|
||||
zeroline=False, linecolor='#bdbdbd', mirror=False,
|
||||
tickformat=',',
|
||||
minor=dict(showgrid=True, gridcolor='#f4f4f4', gridwidth=0.5)
|
||||
)
|
||||
fig.update_yaxes(
|
||||
showgrid=False, zeroline=False, linecolor='#bdbdbd',
|
||||
ticks='outside', ticklen=4, tickcolor='#cccccc',
|
||||
automargin=True
|
||||
)
|
||||
fig.update_traces(
|
||||
hovertemplate="<b>%{y}</b><br>Count: %{x:,}<br>Share: %{customdata[0]}%<extra></extra>",
|
||||
marker_line_width=0,
|
||||
texttemplate='%{x:,}',
|
||||
textposition='outside',
|
||||
textfont=dict(size=10, color='#666666'),
|
||||
cliponaxis=False,
|
||||
selected=dict(marker=dict(opacity=0.6)),
|
||||
unselected=dict(marker=dict(opacity=0.2))
|
||||
)
|
||||
return fig
|
||||
|
||||
if __name__ == "__main__":
|
||||
parser = argparse.ArgumentParser(description="Analyze CAN bus message frequency")
|
||||
parser.add_argument("input", type=Path, help="Path to the input CAN log file")
|
||||
parser.add_argument("output", type=Path, nargs="?", default=Path("freq_report.html"), help="Path to the output HTML report")
|
||||
parser.add_argument("title", nargs="?", default="CAN Bus Message Frequency", help="Title for the HTML report")
|
||||
args = parser.parse_args()
|
||||
|
||||
df = load_data(args.input)
|
||||
stats = calc_freq(df)
|
||||
fig = plot_freq(stats, title=args.title)
|
||||
config = {
|
||||
'responsive': True,
|
||||
'displaylogo': False,
|
||||
'scrollZoom': True,
|
||||
'modeBarButtonsToAdd': ['toggleSpikelines'],
|
||||
'toImageButtonOptions': {'format': 'png', 'scale': 2}
|
||||
}
|
||||
fig.write_html(str(args.output), include_plotlyjs='cdn', config=config)
|
||||
@@ -1,36 +0,0 @@
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
def to_int(x):
|
||||
"""Convert a hex string or integer to int, returning NaN on failure."""
|
||||
if isinstance(x, (int, np.integer)):
|
||||
return int(x)
|
||||
if isinstance(x, str):
|
||||
try:
|
||||
return int(x, 16)
|
||||
except ValueError:
|
||||
return np.nan
|
||||
return np.nan
|
||||
|
||||
def extract_id(row: pd.Series) -> str:
|
||||
"""Extracts PGN from metadata or falls back to CAN ID."""
|
||||
meta = row.get('j1939_metadata')
|
||||
if pd.isna(meta):
|
||||
return f"ID: {row['ID']}"
|
||||
if isinstance(meta, str):
|
||||
try:
|
||||
meta = json.loads(meta)
|
||||
except json.JSONDecodeError:
|
||||
return f"ID: {row['ID']}"
|
||||
if isinstance(meta, dict) and 'PGN' in meta:
|
||||
return f"PGN: {meta['PGN']}"
|
||||
return f"ID: {row['ID']}"
|
||||
|
||||
def load_data(file_path: Path) -> pd.DataFrame:
|
||||
"""Loads Parquet file and adds an Identifier column."""
|
||||
df = pd.read_parquet(file_path)
|
||||
df['Identifier'] = df.apply(extract_id, axis=1)
|
||||
return df
|
||||
Reference in New Issue
Block a user