10 Commits

Author SHA1 Message Date
eeeck 118f1567a2 Merge pull request 'Add header with copyright and SPDX licensing information' (#1) from code-header into dev-entropy-heatmap
Reviewed-on: erickahmed/CANveyor#1
2026-07-13 19:54:10 +02:00
eeeck cbba800536 Merge branch 'dev-entropy-heatmap' into code-header 2026-07-13 19:53:59 +02:00
eeeck 17340746ce Move data extraction functions to utility library 2026-07-13 19:51:22 +02:00
eeeck ad8334c915 Add header with copyright and SPDX licensing information 2026-07-13 19:48:46 +02:00
eeeck 9a5a231841 Add simple CAN message frequency analyzer 2026-07-13 19:28:52 +02:00
eeeck 74b393934c Fix NoneType error 2026-07-13 18:53:27 +02:00
eeeck 4a701e34c5 Remove unused variable 2026-07-13 18:48:45 +02:00
eeeck c42afb34b3 Specify that code is licensed under AGPLv3-or-later 2026-07-13 15:24:52 +02:00
eeeck 70abf7b5ce Use last pre-release tag version 2026-07-13 14:35:52 +02:00
eeeck f975cba4f8 Directory restruture
- Essential files in .
- Analysis or utilities related scripts on respective directories
2026-07-13 14:31:27 +02:00
7 changed files with 131 additions and 26 deletions
+9
View File
@@ -0,0 +1,9 @@
# Copyright
Copyright © 2026 Erick Ahmed
The source code in this repository is licensed under the **GNU Affero General Public License v3.0 or later (AGPL-3.0-or-later)**.
A copy of the license is provided in the `LICENSE` file. If any discrepancy exists between this notice and the `LICENSE` file, the `LICENSE` file shall prevail.
Any third-party components included in this repository at any point during developement remain the property of their respective copyright holders and are subject to their own license terms.
+36 -25
View File
@@ -1,3 +1,7 @@
# File: decoder.py
# Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later
import argparse
import polars as pl
@@ -6,11 +10,7 @@ def get_j1939_mask() -> pl.Expr:
Returns a Polars expression representing the strict J1939 filtering rules.
"""
id_int = pl.col("ID").str.to_integer(base=16).cast(pl.UInt32)
return (
(id_int > 0x7FF) &
((id_int % 33554432 // 16777216) == 0) &
(pl.col("DLC") <= 8)
)
return id_int > 0x7FF
def decode_j1939_metadata(lf: pl.LazyFrame) -> pl.LazyFrame:
"""
@@ -18,17 +18,18 @@ def decode_j1939_metadata(lf: pl.LazyFrame) -> pl.LazyFrame:
"""
id_int = pl.col("ID").str.to_integer(base=16).cast(pl.UInt32)
id_shifted_8 = id_int // 256
id_shifted_16 = id_int // 65536
priority = ((id_int // 67108864) % 8).cast(pl.UInt8)
pf = (id_shifted_16 % 256).cast(pl.UInt8)
ps = (id_shifted_8 % 256).cast(pl.UInt8)
pf = ((id_int // 65536) % 256).cast(pl.UInt8)
ps = ((id_int // 256) % 256).cast(pl.UInt8)
sa = (id_int % 256).cast(pl.UInt8)
da = pl.when(pf < 240).then(ps).otherwise(pl.lit(255, dtype=pl.UInt8))
da = pl.when(pf < 240).then(ps).otherwise(pl.lit(255, dtype=pl.UInt8)).cast(pl.UInt8)
pgn = pl.when(pf < 240).then(id_shifted_8 % 65536).otherwise(id_shifted_8 % 262144).cast(pl.UInt32)
pgn = pl.when(pf < 240).then(
((id_int // 256) & 0x3FF00)
).otherwise(
((id_int // 256) & 0x3FFFF)
).cast(pl.UInt32)
return lf.with_columns(
pl.struct([
@@ -44,26 +45,35 @@ def decode_j1939_metadata(lf: pl.LazyFrame) -> pl.LazyFrame:
def decode_j1939_frames(df: pl.DataFrame) -> pl.DataFrame:
id_int = pl.col("ID").str.to_integer(base=16).cast(pl.UInt32)
is_j1939 = (id_int > 0x7FF) & ((id_int % 33554432 // 16777216) == 0) & (pl.col("DLC") <= 8)
is_j1939 = id_int > 0x7FF
priority = (id_int // 67108864) % 8
pf = (id_int // 65536) % 256
ps = (id_int // 256) % 256
sa = id_int % 256
da = pl.when(pf < 240).then(ps).otherwise(255)
pgn = pl.when(pf < 240).then((id_int // 256) % 65536).otherwise((id_int // 256) % 262144)
priority = ((id_int // 67108864) % 8).cast(pl.UInt8)
pf = ((id_int // 65536) % 256).cast(pl.UInt8)
ps = ((id_int // 256) % 256).cast(pl.UInt8)
sa = (id_int % 256).cast(pl.UInt8)
da = pl.when(pf < 240).then(ps).otherwise(pl.lit(255, dtype=pl.UInt8)).cast(pl.UInt8)
pgn = pl.when(pf < 240).then(
((id_int // 256) & 0x3FF00)
).otherwise(
((id_int // 256) & 0x3FFFF)
).cast(pl.UInt32)
j1939_meta = pl.when(is_j1939).then(
pl.struct([
priority.cast(pl.UInt8).alias("Priority"),
pf.cast(pl.UInt8).alias("PF"),
ps.cast(pl.UInt8).alias("PS"),
sa.cast(pl.UInt8).alias("SA"),
da.cast(pl.UInt8).alias("DA"),
pgn.cast(pl.UInt32).alias("PGN")
priority.alias("Priority"),
pf.alias("PF"),
ps.alias("PS"),
sa.alias("SA"),
da.alias("DA"),
pgn.alias("PGN")
])
).otherwise(None)
return df.with_columns(j1939_meta.alias("j1939_metadata"))
if __name__ == "__main__":
parser = argparse.ArgumentParser(description="J1939 decoder")
parser.add_argument("input_parquet", help="Path to the raw .parquet file")
@@ -72,4 +82,5 @@ if __name__ == "__main__":
df = pl.scan_parquet(args.input_parquet).collect()
decoded_df = decode_j1939_frames(df)
decoded_df.write_parquet(args.output_parquet)
+3
View File
@@ -0,0 +1,3 @@
# File: main.py
# Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later
+4
View File
@@ -1,3 +1,7 @@
# File: parser.py
# Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later
import re
import csv
import polars as pl
+1 -1
View File
@@ -1,6 +1,6 @@
[project]
name = "CANveyor"
version = "0.1.0"
version = "0.0.1"
description = "J1939 CAN bus parser that works in pair with CANdigger"
readme = "README.md"
requires-python = ">=3.14"
+51
View File
@@ -0,0 +1,51 @@
# File: frequency.py
# Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later
import argparse
from pathlib import Path
import pandas as pd
import plotly.express as px
import plotly.graph_objects as go
from utils.extractor import load_data
def calc_freq(df: pd.DataFrame) -> pd.DataFrame:
"""Calculates frequency counts and percentages for identifiers."""
freq_df = df['Identifier'].value_counts().reset_index()
freq_df.columns = ['Identifier', 'Count']
total = freq_df['Count'].sum()
freq_df['Percentage'] = (freq_df['Count'] / total * 100).round(2)
return freq_df.sort_values('Count', ascending=True)
def plot_freq(stats_df: pd.DataFrame, title: str) -> go.Figure:
"""Generates interactive horizontal bar chart with log x-axis."""
fig = px.bar(
stats_df, y='Identifier', x='Count', orientation='h', title=title, log_x=True,
labels={'Identifier': 'PGN / CAN ID', 'Count': 'Message Count'},
color='Count', color_continuous_scale='Turbo',
hover_data={'Percentage': ':.2f', 'Count': True, 'Identifier': True}
)
fig.update_layout(
height=max(600, len(stats_df) * 18), width=1000,
xaxis_title='Total Message Count (Log Scale)', yaxis_title='PGN or CAN ID',
yaxis={'categoryorder': 'total ascending'},
plot_bgcolor='rgba(0,0,0,0)', paper_bgcolor='white',
font=dict(family="Segoe UI, Arial, sans-serif", size=12),
hoverlabel=dict(bgcolor="white", font_size=13, font_family="Segoe UI"),
margin=dict(l=250, r=50, t=80, b=50),
title=dict(font=dict(size=20), x=0.5)
)
fig.update_traces(hovertemplate="<b>%{y}</b><br>Count: %{x:,}<br>Share: %{customdata[0]}%<extra></extra>")
return fig
if __name__ == "__main__":
parser = argparse.ArgumentParser(description="Analyze CAN bus frequency.")
parser.add_argument("-i", "--input", type=Path, required=True)
parser.add_argument("-o", "--output", type=Path, default=Path("freq_report.html"))
parser.add_argument("-t", "--title", type=str, default="CAN Bus Message Frequency")
args = parser.parse_args()
df = load_data(args.input)
stats = calc_freq(df)
fig = plot_freq(stats, title=args.title)
fig.write_html(str(args.output), include_plotlyjs='cdn')
+27
View File
@@ -0,0 +1,27 @@
# File: extractor.py
# Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later
import json
from pathlib import Path
import pandas as pd
def extract_id(row: pd.Series) -> str:
"""Extracts PGN from metadata or falls back to CAN ID."""
meta = row.get('j1939_metadata')
if pd.isna(meta):
return f"ID: {row['ID']}"
if isinstance(meta, str):
try:
meta = json.loads(meta)
except json.JSONDecodeError:
return f"ID: {row['ID']}"
if isinstance(meta, dict) and 'PGN' in meta:
return f"PGN: {meta['PGN']}"
return f"ID: {row['ID']}"
def load_data(file_path: Path) -> pd.DataFrame:
"""Loads Parquet file and adds an Identifier column."""
df = pd.read_parquet(file_path)
df['Identifier'] = df.apply(extract_id, axis=1)
return df