39 Commits

Author SHA1 Message Date
eeeck 2769939f3d Merge pull request 'Implement multi-vehicle support and CAN log viewing options' (#7) from dev-dash into main
Reviewed-on: erickahmed/CANveyor#7
2026-07-22 23:12:53 +02:00
eeeck fab448785b Replace infinite scroll with paginated log view 2026-07-22 23:09:47 +02:00
eeeck 3b703e845f Increase chunk size from 1000 to 50000 2026-07-22 22:00:39 +02:00
eeeck 54fd8ce74c Implement infinite scroll for logs table 2026-07-22 21:39:47 +02:00
eeeck e0a4d098d9 Replace Plotly graph-based tables with Dash DataTable 2026-07-22 21:23:58 +02:00
eeeck 42f8b844d9 Add log visualization tab to dashboard 2026-07-22 21:01:19 +02:00
eeeck 27998ff879 Add multi-vehicle support to dashboard and pipeline 2026-07-22 20:30:57 +02:00
eeeck f5450da96d Merge pull request 'Implement Plotly Dash app with efficient dynamic resampler' (#6) from dev-dash into main
Reviewed-on: erickahmed/CANveyor#6
2026-07-22 20:18:19 +02:00
eeeck d8ca263c0d Bump version and update project dependencies 2026-07-22 20:16:52 +02:00
eeeck 1463fa12ff Parallelize data processing tasks with ThreadPoolExecutor 2026-07-22 20:13:19 +02:00
eeeck 6c198d83c5 Remove debug flag 2026-07-22 20:06:50 +02:00
eeeck 57505074cd Change title to project name 2026-07-22 20:01:58 +02:00
eeeck af8e916116 Create an Overview menu
- To use as a sort of main menu
2026-07-22 20:00:59 +02:00
eeeck d9262e365a Put all CAN bus statistics submenus under a Statistics menu 2026-07-22 20:00:13 +02:00
eeeck 24ce8dad60 Explicitly specify grouping column in dataframe iteration 2026-07-22 19:47:11 +02:00
eeeck 22d4af292c Refactor CSV to Parquet conversion logic 2026-07-22 19:47:05 +02:00
eeeck 5e01c3bb44 Ensure data directories exist before pipeline execution 2026-07-22 19:46:59 +02:00
eeeck d6baaaa1fa Remove redundant Formatted_ID column in frequency calculation 2026-07-22 19:46:51 +02:00
eeeck 9965bc761f Simplify byte column selection in correlation calculation 2026-07-22 19:46:45 +02:00
eeeck 0a4dc5801e Merge pull request 'Implement plotly resamper and precompute data' (#5) from dev-plotly-resampler into dev-dash
Reviewed-on: erickahmed/CANveyor#5
2026-07-22 18:40:44 +02:00
eeeck c06d813c26 Remove resampling information on legend 2026-07-22 18:39:03 +02:00
eeeck 2da646fa80 Precompute CAN data
- Slower startup
- Much faster visualization (from O(n) to O(1))
2026-07-22 18:20:18 +02:00
eeeck 93e0e3f648 Suppress callback exceptions 2026-07-22 18:16:43 +02:00
eeeck 02e46ddf0b Implement plotly-resampler 2026-07-22 18:11:35 +02:00
eeeck d274897cf3 Implement lttbc 2026-07-22 18:07:33 +02:00
eeeck 65591bbc6b Refactor main application to use Polars pipeline
- replaced the caching layer with a pre-processing pipeline that parses
  raw logs into decoded Parquet files
2026-07-15 00:48:51 +02:00
eeeck c98563f541 Fix schema check and update import paths
- Use `collect_schema` for accurate column validation in Polars and
  correct
  relative import paths for statistical modules.
2026-07-15 00:48:30 +02:00
eeeck 3df42fb497 Make subdirectories Python packages 2026-07-15 00:37:19 +02:00
eeeck 73e76d3adc Rename to avoid conflict with Python stat module 2026-07-15 00:31:27 +02:00
eeeck 7a0e2efba1 Integrate Dash background callbacks to handle computations in async 2026-07-15 00:18:27 +02:00
eeeck a2ac49c79f Refactor statistical analysis modules for performance
- Optimize data processing pipelines across files by replacing iterative
  pandas operations with vectorized NumPy routines
2026-07-15 00:17:59 +02:00
eeeck 0780a61d78 Improve bit selection 2026-07-14 23:46:09 +02:00
eeeck fae85d1af9 Change title 2026-07-14 14:28:15 +02:00
eeeck c9461868cc Implement CAN ID visualization at the bit level 2026-07-14 14:14:09 +02:00
eeeck 2408c7a963 Use better function names 2026-07-14 00:37:47 +02:00
eeeck 179ec6e56b Add header 2026-07-14 00:30:05 +02:00
eeeck d8105e2da3 Normalize CAN IDs number of bits 2026-07-14 00:21:03 +02:00
eeeck 9ddc6ea0d5 Use utility function instead of internal 2026-07-13 22:36:31 +02:00
eeeck df3e7b1f0e Update software version 2026-07-13 22:12:44 +02:00
13 changed files with 993 additions and 316 deletions
+83
View File
@@ -0,0 +1,83 @@
# File: logs/view.py
# Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later
import pandas as pd
from dash import html, dash_table, dcc
import dash_bootstrap_components as dbc
PAGE_SIZE = 25000
def prepare_logs_data(df: pd.DataFrame) -> pd.DataFrame:
if df is None or df.empty:
return pd.DataFrame()
df = df.copy()
if 'j1939_metadata' in df.columns:
df['Priority'] = df['j1939_metadata'].apply(lambda x: x.get('Priority') if isinstance(x, dict) else None)
df['PF'] = df['j1939_metadata'].apply(lambda x: x.get('PF') if isinstance(x, dict) else None)
df['PS'] = df['j1939_metadata'].apply(lambda x: x.get('PS') if isinstance(x, dict) else None)
df['SA'] = df['j1939_metadata'].apply(lambda x: x.get('SA') if isinstance(x, dict) else None)
df['DA'] = df['j1939_metadata'].apply(lambda x: x.get('DA') if isinstance(x, dict) else None)
df['PGN'] = df['j1939_metadata'].apply(lambda x: x.get('PGN') if isinstance(x, dict) else None)
else:
for col in ['Priority', 'PF', 'PS', 'SA', 'DA', 'PGN']:
df[col] = None
for i in range(8):
col = f'b{i}'
if col in df.columns:
df[col] = df[col].apply(lambda x: f"{int(x):02X}" if pd.notna(x) else "")
else:
df[col] = ""
if 'ID' in df.columns:
df['ID'] = df['ID'].astype(str)
display_cols = ['Timestamp', 'ID', 'DLC', 'b0', 'b1', 'b2', 'b3', 'b4', 'b5', 'b6', 'b7', 'Priority', 'PF', 'PS', 'SA', 'DA', 'PGN']
display_df = df[[c for c in display_cols if c in df.columns]]
return display_df.fillna("")
def get_logs_table_component():
return html.Div([
html.Div(id='logs-info-text', className="text-muted mb-2"),
dash_table.DataTable(
id='logs-table',
virtualization=True,
page_action='none',
style_table={'overflowX': 'auto', 'height': '70vh', 'overflowY': 'auto'},
style_header={
'backgroundColor': '#1a1a1a',
'color': 'white',
'fontWeight': 'bold',
'textAlign': 'center',
'position': 'sticky',
'top': 0
},
style_data={
'backgroundColor': '#f8f9fa',
'color': '#2a2a2a',
'textAlign': 'center'
},
style_data_conditional=[
{
'if': {'row_index': 'odd'},
'backgroundColor': 'rgb(240, 240, 240)'
}
],
style_cell={
'minWidth': '80px',
'padding': '5px',
'textAlign': 'center',
'fontFamily': 'Segoe UI, Arial, sans-serif'
}
),
html.Div([
dbc.Button("Prev", id='logs-prev-btn', color="secondary", outline=True, size="sm", className="me-2"),
html.Div(id='logs-page-nav', className="d-inline-block", style={'verticalAlign': 'middle'}),
dbc.Button("Next", id='logs-next-btn', color="secondary", outline=True, size="sm", className="ms-2"),
], className="d-flex justify-content-center align-items-center mt-3"),
dcc.Store(id='logs-current-page', data=0),
])
+392
View File
@@ -1,3 +1,395 @@
# File: main.py
# Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later
import os
from pathlib import Path
from concurrent.futures import ThreadPoolExecutor
import polars as pl
import dash
from dash import dcc, html, Input, Output, State
import dash_bootstrap_components as dbc
import numpy as np
from parser import parse_log, parse_csv
from decoder import decode_j1939_frames
from stats.utils.extractor import load_data
from stats.id_viewer import _format_can_id_vec, plot_bits
from stats.frequency import calculate_frequency, plot_frequency
from stats.correlation import calculate_correlation, plot_correlation_heatmap
from stats.entropy import calculate_byte_entropy, plot_entropy_heatmap
from logs.view import get_logs_table_component, prepare_logs_data
RAW_LOG_DIR = "data/logs"
PAGE_SIZE = 25000
def parse_vehicle_from_filename(filename: str):
stem = Path(filename).stem
if '-' in stem:
brand, model_part = stem.split('-', 1)
else:
brand, model_part = stem, "Unknown"
model = model_part.replace('_', ' ')
vehicle = f"{brand} {model}".strip()
return vehicle, brand, model
def run_pipeline():
os.makedirs(RAW_LOG_DIR, exist_ok=True)
os.makedirs("data/csv", exist_ok=True)
os.makedirs("data/parquet", exist_ok=True)
for log_file in Path(RAW_LOG_DIR).glob("*.txt"):
vehicle, brand, model = parse_vehicle_from_filename(log_file.name)
bus1_csv = f"data/csv/{vehicle}_bus1.csv"
bus2_csv = f"data/csv/{vehicle}_bus2.csv"
bus1_parquet = f"data/parquet/{vehicle}_bus1.parquet"
bus2_parquet = f"data/parquet/{vehicle}_bus2.parquet"
bus1_decoded = f"data/parquet/{vehicle}_bus1_decoded.parquet"
bus2_decoded = f"data/parquet/{vehicle}_bus2_decoded.parquet"
if not Path(bus1_decoded).exists() or not Path(bus2_decoded).exists():
print(f"Parsing raw log: {log_file.name}...")
parse_log(str(log_file), bus1_csv, bus2_csv)
print("Converting to parquet...")
parse_csv(bus1_csv).sink_parquet(bus1_parquet)
parse_csv(bus2_csv).sink_parquet(bus2_parquet)
print("Decoding J1939...")
df1 = pl.read_parquet(bus1_parquet)
df2 = pl.read_parquet(bus2_parquet)
dec1 = decode_j1939_frames(df1)
dec2 = decode_j1939_frames(df2)
dec1.write_parquet(bus1_decoded)
dec2.write_parquet(bus2_decoded)
run_pipeline()
print("Loading data into memory...")
DATA = {}
VEHICLE_META = {}
for log_file in Path(RAW_LOG_DIR).glob("*.txt"):
vehicle, brand, model = parse_vehicle_from_filename(log_file.name)
VEHICLE_META[vehicle] = {"brand": brand, "model": model}
bus1_decoded = f"data/parquet/{vehicle}_bus1_decoded.parquet"
bus2_decoded = f"data/parquet/{vehicle}_bus2_decoded.parquet"
if Path(bus1_decoded).exists() and Path(bus2_decoded).exists():
DATA[vehicle] = {
"Bus 1": load_data(bus1_decoded),
"Bus 2": load_data(bus2_decoded)
}
PRECOMPUTED_FIGURES = {}
DATA_BY_ID = {}
CORR_CACHE = {}
PREPARED_LOGS_CACHE = {}
def process_bus_data(vehicle, bus, df):
precomp = {}
precomp[f"{vehicle}_{bus}_freq"] = plot_frequency(calculate_frequency(df), title=f"{vehicle} {bus} Frequency")
precomp[f"{vehicle}_{bus}_entropy"] = plot_entropy_heatmap(calculate_byte_entropy(df), title=f"{vehicle} {bus} Byte-Level Entropy")
can_id_col = 'ID' if 'ID' in df.columns else 'Identifier'
formatted = _format_can_id_vec(df[can_id_col])
df = df.assign(Formatted_ID=formatted)
df = df.sort_values(['Formatted_ID', 'Timestamp'], kind='stable')
grouped = {}
for can_id, group in df.groupby(by='Formatted_ID'):
byte_cols = [f"b{i}" for i in range(8) if f"b{i}" in group.columns]
if not group.empty and len(byte_cols) > 0:
arr = group[byte_cols].to_numpy(dtype=np.float32, copy=False)
if len(arr) > 1:
changed = np.any(arr[1:] != arr[:-1], axis=1)
keep = np.concatenate(([True], changed))
group = group.iloc[keep]
grouped[can_id] = (group, byte_cols)
return vehicle, bus, precomp, grouped
with ThreadPoolExecutor() as executor:
futures = []
for vehicle, buses in DATA.items():
for bus, df in buses.items():
futures.append(executor.submit(process_bus_data, vehicle, bus, df))
for future in futures:
v, b, precomp, grouped = future.result()
PRECOMPUTED_FIGURES.update(precomp)
DATA_BY_ID[(v, b)] = grouped
app = dash.Dash(__name__, external_stylesheets=[dbc.themes.BOOTSTRAP])
app.config.suppress_callback_exceptions = True
app.layout = dbc.Container([
html.H1("CANveyor", className="my-4"),
dbc.Tabs([
dbc.Tab(label="Overview", tab_id="overview", children=[
html.Div(id="overview-content")
]),
dbc.Tab(label="Logs", tab_id="logs", children=[
dbc.Row([
dbc.Col(html.Label("Select Vehicle:", className="mt-2"), width="auto"),
dbc.Col(dcc.Dropdown(
id='logs-vehicle-selector',
options=[{'label': v, 'value': v} for v in DATA.keys()],
value=list(DATA.keys())[0] if DATA else None,
clearable=False
), width=3, className="me-4"),
dbc.Col(html.Label("Select Bus:", className="mt-2"), width="auto"),
dbc.Col(dcc.Dropdown(
id='logs-bus-selector',
options=[{'label': 'Bus 1', 'value': 'Bus 1'}, {'label': 'Bus 2', 'value': 'Bus 2'}],
value='Bus 1',
clearable=False
), width=2),
], className="mb-3 mt-3", align="end"),
get_logs_table_component()
]),
dbc.Tab(label="Statistics", tab_id="statistics", children=[
dbc.Row([
dbc.Col(html.Label("Select Vehicle:", className="mt-2"), width="auto"),
dbc.Col(dcc.Dropdown(
id='vehicle-selector',
options=[{'label': v, 'value': v} for v in DATA.keys()],
value=list(DATA.keys())[0] if DATA else None,
clearable=False
), width=3, className="me-4"),
dbc.Col(html.Label("Select Bus:", className="mt-2"), width="auto"),
dbc.Col(dcc.Dropdown(
id='bus-selector',
options=[{'label': 'Bus 1', 'value': 'Bus 1'}, {'label': 'Bus 2', 'value': 'Bus 2'}],
value='Bus 1',
clearable=False
), width=2),
], className="mb-3 mt-3", align="end"),
dbc.Tabs([
dbc.Tab(label="Frequency", tab_id="freq"),
dbc.Tab(label="ID Viewer", tab_id="id_viewer"),
dbc.Tab(label="Correlation", tab_id="corr"),
dbc.Tab(label="Entropy", tab_id="entropy"),
], id="tabs", active_tab="freq"),
html.Div(id="tab-content", className="mt-3")
])
], id="main-tabs", active_tab="statistics")
], fluid=True)
def get_prepared_logs(vehicle, bus):
cache_key = (vehicle, bus)
if cache_key not in PREPARED_LOGS_CACHE:
df = DATA[vehicle][bus]
PREPARED_LOGS_CACHE[cache_key] = prepare_logs_data(df)
return PREPARED_LOGS_CACHE[cache_key]
def build_page_buttons(current_page: int, total_pages: int, max_buttons: int = 15) -> list:
buttons: list = []
if total_pages <= 1:
return buttons
half = max_buttons // 2
start = max(0, current_page - half)
end = min(total_pages, start + max_buttons)
if end - start < max_buttons:
start = max(0, end - max_buttons)
if start > 0:
buttons.append(
dbc.Button("1", id={'type': 'page-btn', 'index': 0}, color="secondary", outline=True, size="sm", className="me-1")
)
if start > 1:
buttons.append(html.Span("", className="mx-1 align-middle"))
for i in range(start, end):
is_current = (i == current_page)
buttons.append(
dbc.Button(
str(i + 1),
id={'type': 'page-btn', 'index': i},
size="sm",
color="primary" if is_current else "secondary",
outline=not is_current,
className="me-1",
disabled=is_current,
)
)
if end < total_pages:
if end < total_pages - 1:
buttons.append(html.Span("", className="mx-1 align-middle"))
buttons.append(
dbc.Button(
str(total_pages),
id={'type': 'page-btn', 'index': total_pages - 1},
color="secondary", outline=True, size="sm", className="me-1",
)
)
return buttons
@app.callback(
Output('logs-table', 'data'),
Output('logs-table', 'columns'),
Output('logs-info-text', 'children'),
Output('logs-page-nav', 'children'),
Output('logs-current-page', 'data'),
Input('logs-vehicle-selector', 'value'),
Input('logs-bus-selector', 'value'),
Input('logs-prev-btn', 'n_clicks'),
Input('logs-next-btn', 'n_clicks'),
Input({'type': 'page-btn', 'index': dash.ALL}, 'n_clicks'),
State('logs-current-page', 'data'),
)
def update_logs_table(vehicle, bus, prev_clicks, next_clicks, page_btn_clicks, current_page):
if (not vehicle or not bus or vehicle not in DATA or bus not in DATA[vehicle]):
return [], [], "No data available", [], 0
prepared_df = get_prepared_logs(vehicle, bus)
total_rows = len(prepared_df)
if total_rows == 0:
return [], [], "No data available", [], 0
total_pages = max(1, (total_rows + PAGE_SIZE - 1) // PAGE_SIZE)
ctx = dash.callback_context
triggered_id = ctx.triggered_id
current_page = current_page if current_page is not None else 0
if triggered_id in ('logs-vehicle-selector', 'logs-bus-selector'):
current_page = 0
elif triggered_id == 'logs-prev-btn':
current_page = max(0, current_page - 1)
elif triggered_id == 'logs-next-btn':
current_page = current_page + 1
elif isinstance(triggered_id, dict) and triggered_id.get('type') == 'page-btn':
if ctx.triggered and ctx.triggered[0]['value']:
current_page = triggered_id['index']
current_page = max(0, min(current_page, total_pages - 1))
start_idx = current_page * PAGE_SIZE
end_idx = min(start_idx + PAGE_SIZE, total_rows)
page_data = prepared_df.iloc[start_idx:end_idx].to_dict('records')
columns = [{"name": i, "id": i} for i in prepared_df.columns]
info_text = (f"Page {current_page + 1} of {total_pages} | "
f"Showing rows {start_idx + 1:,}{end_idx:,} "
f"of {total_rows:,} total frames")
page_buttons = build_page_buttons(current_page, total_pages)
return page_data, columns, info_text, page_buttons, current_page
@app.callback(
Output('tab-content', 'children'),
Input('tabs', 'active_tab'),
Input('vehicle-selector', 'value'),
Input('bus-selector', 'value')
)
def render_content(tab, vehicle, bus):
if not vehicle or not bus or vehicle not in DATA or bus not in DATA[vehicle]:
return html.Div("No data available")
df = DATA[vehicle][bus]
if tab == 'freq':
return dcc.Graph(figure=PRECOMPUTED_FIGURES[f"{vehicle}_{bus}_freq"], style={'height': '80vh'})
elif tab == 'id_viewer':
ids = sorted(DATA_BY_ID.get((vehicle, bus), {}).keys())
return html.Div([
html.Label("Select CAN ID:"),
dcc.Dropdown(
id='id-selector',
options=[{'label': i, 'value': i} for i in ids],
value=ids[0] if ids else None,
clearable=False,
style={'width': '50%', 'marginBottom': '10px'}
),
dcc.Graph(id='id-viewer-graph', style={'height': '70vh'})
])
elif tab == 'corr':
ids = sorted(DATA_BY_ID.get((vehicle, bus), {}).keys())
return html.Div([
dbc.Row([
dbc.Col(html.Label("Method:"), width=1, className="mt-2"),
dbc.Col(dcc.Dropdown(
id='corr-method',
options=[{'label': 'Pearson', 'value': 'pearson'}, {'label': 'Spearman', 'value': 'spearman'}],
value='pearson',
clearable=False
), width=2),
dbc.Col(html.Label("Target ID:"), width=1, className="mt-2"),
dbc.Col(dcc.Dropdown(
id='corr-target',
options=[{'label': 'All IDs (Max Corr)', 'value': 'all'}] + [{'label': i, 'value': i} for i in ids],
value='all',
clearable=True
), width=4),
], className="mb-3"),
dcc.Graph(id='corr-graph', style={'height': '80vh'})
])
elif tab == 'entropy':
return dcc.Graph(figure=PRECOMPUTED_FIGURES[f"{vehicle}_{bus}_entropy"], style={'height': '80vh'})
return html.Div("Tab not found")
@app.callback(
Output('id-viewer-graph', 'figure'),
Input('id-selector', 'value'),
Input('vehicle-selector', 'value'),
Input('bus-selector', 'value'),
Input('tabs', 'active_tab'),
)
def update_id_viewer(selected_id, vehicle, bus, tab):
if tab != 'id_viewer' or not selected_id or not vehicle or not bus:
return dash.no_update
grouped_data = DATA_BY_ID.get((vehicle, bus), {})
if selected_id not in grouped_data:
return dash.no_update
filtered_df, byte_cols = grouped_data[selected_id]
return plot_bits(filtered_df, byte_cols, selected_id, title=f"{vehicle} {bus} Byte Visualization")
@app.callback(
Output('corr-graph', 'figure'),
Input('corr-method', 'value'),
Input('corr-target', 'value'),
Input('vehicle-selector', 'value'),
Input('bus-selector', 'value'),
Input('tabs', 'active_tab'),
)
def update_corr(method, target, vehicle, bus, tab):
if tab != 'corr' or not vehicle or not bus:
return dash.no_update
target_id = None if target == 'all' or not target else target
cache_key = (vehicle, bus, method, target_id)
if cache_key not in CORR_CACHE:
df = DATA[vehicle][bus]
corr_df = calculate_correlation(df, method=method, target_id=target_id)
CORR_CACHE[cache_key] = corr_df
else:
corr_df = CORR_CACHE[cache_key]
title = f"{vehicle} {bus} Correlation"
if target_id:
title += f" ({target_id})"
return plot_correlation_heatmap(corr_df, target_id=target_id, title=title)
if __name__ == '__main__':
app.run(debug=False)
+8 -4
View File
@@ -19,7 +19,7 @@ def parse_log(input_path: PathLike, out_bus1: PathLike, out_bus2: PathLike) -> N
out1_file = Path(out_bus1)
out2_file = Path(out_bus2)
start_pattern = re.compile(r'(C[12]):([0-9A-Fa-f]{1,8})\s+([0-9A-Fa-f]{1,2})\s+')
start_pattern = re.compile(r'(C[12]):([0-9A-Fa-f]{7,8})\s+([0-9A-Fa-f]{1,2})\s+')
byte_pattern = re.compile(r'^[0-9A-Fa-f]{2}$')
with input_file.open('r', encoding='utf-8') as f_in, \
@@ -35,7 +35,12 @@ def parse_log(input_path: PathLike, out_bus1: PathLike, out_bus2: PathLike) -> N
for line in f_in:
for match in start_pattern.finditer(line):
bus = match.group(1)
can_id = match.group(2).upper()
can_id = match.group(2).upper().zfill(8)
if int(can_id, 16) > 0x1FFFFFFF:
continue
dlc_str = match.group(3)
try:
@@ -72,7 +77,7 @@ def parse_csv(csv_path: PathLike) -> pl.LazyFrame:
"""
lf = pl.scan_csv(csv_path, schema_overrides={"ID": pl.String, "Data": pl.String})
if "Timestamp" not in lf.columns:
if "Timestamp" not in lf.collect_schema().names():
lf = lf.with_row_index("Timestamp")
byte_exprs = []
@@ -116,4 +121,3 @@ if __name__ == '__main__':
print(f"[*] Processing {args.input_csv}...")
lf = parse_csv(args.input_csv)
lf.sink_parquet(args.output_parquet)
print(f"[+] Saved parquet file to {args.output_parquet}")
+11 -3
View File
@@ -1,7 +1,15 @@
[project]
name = "CANveyor"
version = "0.0.1"
version = "0.1.0"
description = "J1939 CAN bus parser that works in pair with CANdigger"
readme = "README.md"
requires-python = ">=3.14"
dependencies = ["polars", "pathlib", "typing"]
requires-python = ">=3.10"
dependencies = [
"polars",
"dash",
"dash-bootstrap-components",
"numpy",
"pandas",
"plotly",
"plotly-resampler"
]
-180
View File
@@ -1,180 +0,0 @@
# File: entropy.py
# Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later
import argparse
from pathlib import Path
import numpy as np
import pandas as pd
import plotly.graph_objects as go
from utils.extractor import load_data
def _to_int(x):
"""Convert a hex string or integer to int, returning NaN on failure."""
if isinstance(x, (int, np.integer)):
return int(x)
if isinstance(x, str):
try:
return int(x, 16)
except ValueError:
return np.nan
return np.nan
def calculate_byte_entropy(df: pd.DataFrame) -> pd.DataFrame:
"""Calculates Shannon entropy per byte position for each identifier."""
byte_cols = [f"b{i}" for i in range(8)]
available_cols = [col for col in byte_cols if col in df.columns]
if not available_cols:
raise ValueError("No byte columns (b0-b7) found in the DataFrame")
df_bytes = df[available_cols].copy()
for col in available_cols:
df_bytes[col] = df_bytes[col].apply(_to_int)
def entropy(s: pd.Series) -> float:
s = s.dropna()
if s.empty:
return 0.0
p = s.value_counts(normalize=True)
return -np.sum(p * np.log2(p))
return df.groupby("Identifier")[available_cols].agg(entropy)
def plot_entropy_heatmap(entropy_df: pd.DataFrame, title: str) -> go.Figure:
"""Generates an interactive heatmap of byte-level Shannon entropy."""
x = entropy_df.columns.tolist()
y = entropy_df.index.tolist()
z = entropy_df.values
fig = go.Figure(
data=go.Heatmap(
z=z,
x=x,
y=y,
colorscale=[
[0.0, "#ffffff"],
[0.15, "#fff7ec"],
[0.35, "#fee8c8"],
[0.55, "#fdd49e"],
[0.75, "#fdbb84"],
[1.0, "#ef6548"],
],
xgap=3,
ygap=3,
text=np.round(z, 2),
texttemplate="%{text}",
textfont={
"size": 11,
"color": "#2a2a2a",
"family": "Segoe UI, Arial, sans-serif",
},
hoverongaps=False,
hovertemplate=(
"<b>%{y}</b><br>"
"Byte %{x}: %{z:.2f} bits<extra></extra>"
),
colorbar=dict(
title=dict(
text="Entropy (bits)",
side="top",
font=dict(size=13, color="#1a1a1a"),
),
orientation="h",
thickness=15,
len=0.35,
x=1.0,
xanchor="right",
y=1.02,
yanchor="bottom",
tickfont=dict(size=11, color="#2a2a2a"),
tickformat=".1f",
outlinewidth=0.5,
outlinecolor="#cccccc",
),
)
)
fig.update_layout(
title=dict(
text=title,
font=dict(size=20, color="#1a1a1a"),
x=0.5,
xanchor="center",
pad=dict(b=20),
),
height=max(600, len(y) * 28 + 150),
autosize=True,
template="plotly_white",
xaxis=dict(
title=dict(text="Byte Position", font=dict(size=13, color="#1a1a1a")),
side="top",
dtick=1,
showgrid=False,
linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"),
ticks="outside",
ticklen=4,
tickcolor="#cccccc",
),
yaxis=dict(
title=dict(text="PGN or CAN ID", font=dict(size=13, color="#1a1a1a")),
autorange="reversed",
showgrid=False,
linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"),
ticks="outside",
ticklen=4,
tickcolor="#cccccc",
automargin=True,
),
font=dict(family="Segoe UI, Arial, sans-serif", size=12, color="#2a2a2a"),
hoverlabel=dict(
bgcolor="white",
font_size=13,
font_family="Segoe UI",
bordercolor="#cccccc",
),
margin=dict(l=200, r=40, t=120, b=60),
)
return fig
if __name__ == "__main__":
parser = argparse.ArgumentParser(
description="Analyze CAN bus byte-level entropy"
)
parser.add_argument(
"input", type=Path, help="Path to the input CAN log file"
)
parser.add_argument(
"output",
type=Path,
nargs="?",
default=Path("entropy_report.html"),
help="Path to the output HTML report",
)
parser.add_argument(
"title",
nargs="?",
default="CAN Bus Byte-Level Entropy",
help="Title for the HTML report",
)
args = parser.parse_args()
df = load_data(args.input)
entropy_df = calculate_byte_entropy(df)
fig = plot_entropy_heatmap(entropy_df, title=args.title)
config = {
"responsive": True,
"displaylogo": False,
"scrollZoom": True,
"modeBarButtonsToAdd": ["toggleSpikelines"],
"toImageButtonOptions": {"format": "png", "scale": 2},
}
fig.write_html(str(args.output), include_plotlyjs="cdn", config=config)
-36
View File
@@ -1,36 +0,0 @@
import json
from pathlib import Path
import numpy as np
import pandas as pd
def to_int(x):
"""Convert a hex string or integer to int, returning NaN on failure."""
if isinstance(x, (int, np.integer)):
return int(x)
if isinstance(x, str):
try:
return int(x, 16)
except ValueError:
return np.nan
return np.nan
def extract_id(row: pd.Series) -> str:
"""Extracts PGN from metadata or falls back to CAN ID."""
meta = row.get('j1939_metadata')
if pd.isna(meta):
return f"ID: {row['ID']}"
if isinstance(meta, str):
try:
meta = json.loads(meta)
except json.JSONDecodeError:
return f"ID: {row['ID']}"
if isinstance(meta, dict) and 'PGN' in meta:
return f"PGN: {meta['PGN']}"
return f"ID: {row['ID']}"
def load_data(file_path: Path) -> pd.DataFrame:
"""Loads Parquet file and adds an Identifier column."""
df = pd.read_parquet(file_path)
df['Identifier'] = df.apply(extract_id, axis=1)
return df
View File
+84 -50
View File
@@ -4,75 +4,111 @@
import argparse
from pathlib import Path
from concurrent.futures import ThreadPoolExecutor
import numpy as np
import pandas as pd
import plotly.graph_objects as go
from utils.extractor import load_data
from stats.utils.extractor import load_data
from stats.utils.extractor import to_int
def _format_can_id_vec(s: pd.Series) -> pd.Series:
s = s.astype('string').str.strip()
s = s.str.replace(r'^0x', '', case=False, regex=True)
s = s.str.upper()
return s.fillna('UNKNOWN').replace('', 'UNKNOWN')
def _to_int(x):
"""Convert a hex string or integer to int, returning NaN on failure."""
if isinstance(x, (int, np.integer)):
return int(x)
if isinstance(x, str):
try:
return int(x, 16)
except ValueError:
return np.nan
return np.nan
def _ensure_int_bytes(df: pd.DataFrame, cols: list) -> pd.DataFrame:
needs = [c for c in cols if not pd.api.types.is_numeric_dtype(df[c])]
if needs:
df = df.copy()
for c in needs:
df[c] = df[c].apply(to_int)
return df
def calculate_correlation(df: pd.DataFrame, method: str, target_id: str | None = None) -> pd.DataFrame:
"""Calculates inter-byte correlation grouped by identifier."""
byte_cols = [f"b{i}" for i in range(8)]
available_cols = [col for col in byte_cols if col in df.columns]
available_cols = [f"b{i}" for i in range(8) if f"b{i}" in df.columns]
if not available_cols:
raise ValueError("No byte columns (b0-b7) found in the DataFrame")
df_bytes = df[["Identifier"] + available_cols].copy()
for col in available_cols:
df_bytes[col] = df_bytes[col].apply(_to_int)
can_id_col = 'ID' if 'ID' in df.columns else 'Identifier'
identifiers = _format_can_id_vec(df[can_id_col]).to_numpy()
if target_id:
group = df_bytes[df_bytes["Identifier"] == target_id]
if group.empty:
df_bytes = _ensure_int_bytes(df, available_cols)[available_cols]
data = df_bytes.to_numpy(dtype=np.float64, copy=False)
if target_id is not None:
target_id = _format_can_id_vec(pd.Series([target_id])).iloc[0]
mask = identifiers == target_id
if not mask.any():
raise ValueError(f"Identifier '{target_id}' not found in data")
return group[available_cols].corr(method=method).fillna(0.0)
sub = data[mask]
mask = ~np.isnan(sub).any(axis=1)
sub = sub[mask]
if method == 'spearman' and sub.shape[0] > 1:
sub = pd.DataFrame(sub).rank().to_numpy()
if sub.shape[0] > 1:
with np.errstate(divide='ignore', invalid='ignore'):
c = np.corrcoef(sub, rowvar=False)
np.nan_to_num(c, copy=False, nan=0.0)
else:
c = np.zeros((len(available_cols), len(available_cols)))
return pd.DataFrame(c, index=available_cols, columns=available_cols)
def max_abs_corr(group: pd.DataFrame) -> pd.Series:
corr_arr = np.abs(group.corr(method=method).to_numpy().copy())
np.fill_diagonal(corr_arr, 0.0)
return pd.Series(corr_arr.max(axis=0), index=group.columns).fillna(0.0)
unique_ids, inverse = np.unique(identifiers, return_inverse=True)
n_cols = len(available_cols)
return df_bytes.groupby("Identifier")[available_cols].apply(max_abs_corr)
sort_idx = np.argsort(inverse, kind='stable')
data_sorted = data[sort_idx]
inverse_sorted = inverse[sort_idx]
if len(inverse_sorted) > 0:
split_points = np.flatnonzero(np.diff(inverse_sorted)) + 1
groups = np.split(data_sorted, split_points)
else:
groups = []
def _process_group(sub):
mask = ~np.isnan(sub).any(axis=1)
sub = sub[mask]
if len(sub) > 1:
if method == 'spearman':
sub = pd.DataFrame(sub).rank().to_numpy()
with np.errstate(divide='ignore', invalid='ignore'):
c = np.abs(np.corrcoef(sub, rowvar=False))
np.nan_to_num(c, copy=False, nan=0.0)
np.fill_diagonal(c, 0.0)
return c.max(axis=0)
return np.zeros(n_cols, dtype=np.float64)
out = np.zeros((len(unique_ids), n_cols), dtype=np.float64)
if len(groups) > 0:
with ThreadPoolExecutor() as executor:
results = list(executor.map(_process_group, groups))
for i, res in enumerate(results):
out[i] = res
result = pd.DataFrame(out, index=unique_ids, columns=available_cols)
result.index.name = 'Identifier'
return result
def plot_correlation_heatmap(corr_df: pd.DataFrame, target_id: str | None, title: str) -> go.Figure:
"""Generates an interactive heatmap of inter-byte correlation."""
is_8x8 = target_id is not None
x = corr_df.columns.tolist()
y = corr_df.index.tolist()
z = corr_df.values
if is_8x8:
x = corr_df.columns.tolist()
y = corr_df.index.tolist()
z = corr_df.values
z_min, z_max = -1.0, 1.0
colorscale = [
[0.0, "#2c7bb6"], [0.25, "#abd9e9"], [0.5, "#ffffff"],
[0.75, "#fdae61"], [1.0, "#d7191c"]
]
colorscale = [[0.0, "#2c7bb6"], [0.25, "#abd9e9"], [0.5, "#ffffff"], [0.75, "#fdae61"], [1.0, "#d7191c"]]
hover_template = "<b>%{y}</b> vs <b>%{x}</b><br>Correlation: %{z:.2f}<extra></extra>"
else:
x = corr_df.columns.tolist()
y = corr_df.index.tolist()
z = corr_df.values
z_min, z_max = 0.0, 1.0
colorscale = [
[0.0, "#ffffff"], [0.2, "#fff5f0"], [0.4, "#fecc5c"],
[0.6, "#fd8d3c"], [0.8, "#e31a1c"], [1.0, "#800026"]
]
colorscale = [[0.0, "#ffffff"], [0.2, "#fff5f0"], [0.4, "#fecc5c"], [0.6, "#fd8d3c"], [0.8, "#e31a1c"], [1.0, "#800026"]]
hover_template = "<b>%{y}</b><br>Byte %{x} max correlation: %{z:.2f}<extra></extra>"
fig = go.Figure(
@@ -98,17 +134,17 @@ def plot_correlation_heatmap(corr_df: pd.DataFrame, target_id: str | None, title
fig.update_layout(
title=dict(text=title, font=dict(size=20, color="#1a1a1a"), x=0.5, xanchor="center", pad=dict(b=20)),
height=max(600, len(y) * 28 + 150) if not is_8x8 else 600,
height=600 if is_8x8 else max(600, len(y) * 28 + 150),
autosize=True,
template="plotly_white",
xaxis=dict(
title=dict(text="Byte Position", font=dict(size=13, color="#1a1a1a")),
side="top" if not is_8x8 else "bottom",
side="bottom" if is_8x8 else "top",
dtick=1, showgrid=False, linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"), ticks="outside", ticklen=4, tickcolor="#cccccc",
),
yaxis=dict(
title=dict(text="PGN or CAN ID" if not is_8x8 else "Byte Position", font=dict(size=13, color="#1a1a1a")),
title=dict(text="Byte Position" if is_8x8 else "PGN or CAN ID", font=dict(size=13, color="#1a1a1a")),
autorange="reversed", showgrid=False, linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"), ticks="outside", ticklen=4, tickcolor="#cccccc", automargin=True,
),
@@ -123,17 +159,15 @@ if __name__ == "__main__":
parser = argparse.ArgumentParser(description="Analyze CAN bus inter-byte correlation")
parser.add_argument("method", choices=["pearson", "spearman"], help="Correlation method to use")
parser.add_argument("input", type=Path, help="Path to the input CAN log file")
parser.add_argument("output", type=Path, nargs="?", default=Path("correlation_report.html"), help="Path to the output HTML report")
parser.add_argument("title", nargs="?", default="CAN Bus Inter-Byte Correlation", help="Title for the HTML report")
parser.add_argument("--identifier", type=str, default=None, help="Specific PGN/CAN ID to analyze (e.g., 'PGN: 65331'). If omitted, shows max correlation per byte for all IDs.")
parser.add_argument("output", type=Path, nargs="?", default=Path("correlation_report.html"))
parser.add_argument("title", nargs="?", default="CAN Bus Inter-Byte Correlation")
parser.add_argument("--identifier", type=str, default=None)
args = parser.parse_args()
df = load_data(args.input)
corr_df = calculate_correlation(df, method=args.method, target_id=args.identifier)
display_title = f"{args.title} ({args.identifier})" if args.identifier else args.title
fig = plot_correlation_heatmap(corr_df, target_id=args.identifier, title=display_title)
config = {
"responsive": True,
"displaylogo": False,
+156
View File
@@ -0,0 +1,156 @@
# File: entropy.py
# Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later
import argparse
from pathlib import Path
from concurrent.futures import ThreadPoolExecutor
import numpy as np
import pandas as pd
import plotly.graph_objects as go
from stats.utils.extractor import load_data
from stats.utils.extractor import to_int
def _format_can_id_vec(s: pd.Series) -> pd.Series:
s = s.astype('string').str.strip()
s = s.str.replace(r'^0x', '', case=False, regex=True)
s = s.str.upper()
return s.fillna('UNKNOWN').replace('', 'UNKNOWN')
def _entropy_col(a: np.ndarray) -> float:
a = a[~np.isnan(a)]
if a.size == 0:
return 0.0
a = a.astype(np.int64)
lo, hi = a.min(), a.max()
span = hi - lo + 1
if span <= 0:
return 0.0
if span > 1 << 20:
_, counts = np.unique(a, return_counts=True)
else:
counts = np.bincount(a - lo, minlength=span)
counts = counts[counts > 0]
p = counts / counts.sum()
return float(-np.sum(p * np.log2(p)))
def calculate_byte_entropy(df: pd.DataFrame) -> pd.DataFrame:
available_cols = [f"b{i}" for i in range(8) if f"b{i}" in df.columns]
if not available_cols:
raise ValueError("No byte columns (b0-b7) found in the DataFrame")
can_id_col = 'ID' if 'ID' in df.columns else 'Identifier'
identifiers = _format_can_id_vec(df[can_id_col]).to_numpy()
needs = [c for c in available_cols if not pd.api.types.is_numeric_dtype(df[c])]
if needs:
df = df.copy()
for c in needs:
df[c] = df[c].apply(to_int)
data = df[available_cols].to_numpy(dtype=np.float64, copy=False)
unique_ids, inverse = np.unique(identifiers, return_inverse=True)
n_cols = len(available_cols)
sort_idx = np.argsort(inverse, kind='stable')
data_sorted = data[sort_idx]
inverse_sorted = inverse[sort_idx]
if len(inverse_sorted) > 0:
split_points = np.flatnonzero(np.diff(inverse_sorted)) + 1
groups = np.split(data_sorted, split_points)
else:
groups = []
def _process_group(sub):
res = np.zeros(n_cols, dtype=np.float64)
for ci in range(n_cols):
res[ci] = _entropy_col(sub[:, ci])
return res
out = np.zeros((len(unique_ids), n_cols), dtype=np.float64)
if len(groups) > 0:
with ThreadPoolExecutor() as executor:
results = list(executor.map(_process_group, groups))
for i, res in enumerate(results):
out[i] = res
result = pd.DataFrame(out, index=unique_ids, columns=available_cols)
result.index.name = 'Identifier'
return result
def plot_entropy_heatmap(entropy_df: pd.DataFrame, title: str) -> go.Figure:
x = entropy_df.columns.tolist()
y = entropy_df.index.tolist()
z = entropy_df.values
fig = go.Figure(
data=go.Heatmap(
z=z, x=x, y=y,
colorscale=[
[0.0, "#ffffff"],
[0.15, "#fff7ec"],
[0.35, "#fee8c8"],
[0.55, "#fdd49e"],
[0.75, "#fdbb84"],
[1.0, "#ef6548"],
],
xgap=3, ygap=3,
text=np.round(z, 2),
texttemplate="%{text}",
textfont={"size": 11, "color": "#2a2a2a", "family": "Segoe UI, Arial, sans-serif"},
hoverongaps=False,
hovertemplate="<b>%{y}</b><br>Byte %{x}: %{z:.2f} bits<extra></extra>",
colorbar=dict(
title=dict(text="Entropy (bits)", side="top", font=dict(size=13, color="#1a1a1a")),
orientation="h", thickness=15, len=0.35,
x=1.0, xanchor="right", y=1.02, yanchor="bottom",
tickfont=dict(size=11, color="#2a2a2a"),
tickformat=".1f", outlinewidth=0.5, outlinecolor="#cccccc",
),
)
)
fig.update_layout(
title=dict(text=title, font=dict(size=20, color="#1a1a1a"), x=0.5, xanchor="center", pad=dict(b=20)),
height=max(600, len(y) * 28 + 150),
autosize=True,
template="plotly_white",
xaxis=dict(
title=dict(text="Byte Position", font=dict(size=13, color="#1a1a1a")),
side="top", dtick=1, showgrid=False, linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"), ticks="outside", ticklen=4, tickcolor="#cccccc",
),
yaxis=dict(
title=dict(text="PGN or CAN ID", font=dict(size=13, color="#1a1a1a")),
autorange="reversed", showgrid=False, linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"), ticks="outside", ticklen=4, tickcolor="#cccccc", automargin=True,
),
font=dict(family="Segoe UI, Arial, sans-serif", size=12, color="#2a2a2a"),
hoverlabel=dict(bgcolor="white", font_size=13, font_family="Segoe UI", bordercolor="#cccccc"),
margin=dict(l=200, r=40, t=120, b=60),
)
return fig
if __name__ == "__main__":
parser = argparse.ArgumentParser(description="Analyze CAN bus byte-level entropy")
parser.add_argument("input", type=Path, help="Path to the input CAN log file")
parser.add_argument("output", type=Path, nargs="?", default=Path("entropy_report.html"))
parser.add_argument("title", nargs="?", default="CAN Bus Byte-Level Entropy")
args = parser.parse_args()
df = load_data(args.input)
entropy_df = calculate_byte_entropy(df)
fig = plot_entropy_heatmap(entropy_df, title=args.title)
config = {
"responsive": True,
"displaylogo": False,
"scrollZoom": True,
"modeBarButtonsToAdd": ["toggleSpikelines"],
"toImageButtonOptions": {"format": "png", "scale": 2},
}
fig.write_html(str(args.output), include_plotlyjs="cdn", config=config)
+53 -43
View File
@@ -4,33 +4,56 @@
import argparse
from pathlib import Path
import numpy as np
import pandas as pd
import plotly.express as px
import plotly.graph_objects as go
from utils.extractor import load_data
from stats.utils.extractor import load_data
def calc_freq(df: pd.DataFrame) -> pd.DataFrame:
"""Calculates frequency counts and percentages for identifiers."""
freq_df = df['Identifier'].value_counts().reset_index()
freq_df.columns = ['Identifier', 'Count']
total = freq_df['Count'].sum()
freq_df['Percentage'] = (freq_df['Count'] / total * 100).round(2)
return freq_df.sort_values('Count', ascending=True)
def _format_can_id_vec(s: pd.Series) -> pd.Series:
s = s.astype('string').str.strip()
s = s.str.replace(r'^0x', '', case=False, regex=True)
s = s.str.upper()
return s.fillna('UNKNOWN').replace('', 'UNKNOWN')
def calculate_frequency(df: pd.DataFrame) -> pd.DataFrame:
can_id_col = 'ID' if 'ID' in df.columns else 'Identifier'
formatted = _format_can_id_vec(df[can_id_col])
counts = formatted.value_counts()
freq_df = pd.DataFrame({
'Identifier': counts.index,
'Count': counts.to_numpy(),
})
total = counts.sum()
freq_df['Percentage'] = np.round(freq_df['Count'] / total * 100, 2) if total else 0.0
return freq_df.sort_values('Count', ascending=True).reset_index(drop=True)
def plot_frequency(stats_df: pd.DataFrame, title: str) -> go.Figure:
n = len(stats_df)
fig = go.Figure(go.Bar(
y=stats_df['Identifier'],
x=stats_df['Count'],
orientation='h',
marker=dict(
color=stats_df['Count'],
colorscale='Turbo',
cmin=int(stats_df['Count'].min()) if n else 0,
cmax=int(stats_df['Count'].max()) if n else 1,
line_width=0,
),
customdata=stats_df[['Percentage']].to_numpy(),
hovertemplate="<b>%{y}</b><br>Count: %{x:,}<br>Share: %{customdata[0]}%<extra></extra>",
texttemplate='%{x:,}',
textposition='outside',
cliponaxis=False,
))
def plot_freq(stats_df: pd.DataFrame, title: str) -> go.Figure:
"""Generates interactive horizontal bar chart with log x-axis."""
fig = px.bar(
stats_df, y='Identifier', x='Count', orientation='h', title=title, log_x=True,
labels={'Identifier': 'PGN / CAN ID', 'Count': 'Message Count'},
color='Count', color_continuous_scale='Turbo',
range_color=(stats_df['Count'].min(), stats_df['Count'].max()),
hover_data={'Percentage': ':.2f', 'Count': ':,', 'Identifier': True}
)
fig.update_layout(
height=max(600, len(stats_df) * 18),
height=max(600, n * 18),
autosize=True,
template='plotly_white',
xaxis=dict(
type='log',
title=dict(text="Message count [log scale]", font=dict(size=13, color="#1a1a1a")),
side="top",
dtick=1,
@@ -43,7 +66,6 @@ def plot_freq(stats_df: pd.DataFrame, title: str) -> go.Figure:
),
yaxis=dict(
title=dict(text="PGN or CAN ID", font=dict(size=13, color="#1a1a1a")),
#autorange="",
showgrid=False,
linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"),
@@ -51,10 +73,10 @@ def plot_freq(stats_df: pd.DataFrame, title: str) -> go.Figure:
ticklen=4,
tickcolor="#cccccc",
automargin=True,
type='category',
),
font=dict(family="Segoe UI, Arial, sans-serif", size=12, color='#2a2a2a'),
hoverlabel=dict(bgcolor="white", font_size=13, font_family="Segoe UI",
bordercolor='#cccccc'),
hoverlabel=dict(bgcolor="white", font_size=13, font_family="Segoe UI", bordercolor='#cccccc'),
margin=dict(l=200, r=40, t=120, b=60),
bargap=0.35,
coloraxis_colorbar=dict(
@@ -68,49 +90,37 @@ def plot_freq(stats_df: pd.DataFrame, title: str) -> go.Figure:
yanchor='bottom',
tickformat=',',
outlinecolor='#cccccc',
outlinewidth=0.5
outlinewidth=0.5,
),
title=dict(font=dict(size=20, color='#1a1a1a'), x=0.5, xanchor='center',
pad=dict(b=20))
title=dict(text=title, font=dict(size=20, color='#1a1a1a'), x=0.5, xanchor='center', pad=dict(b=20)),
)
fig.update_xaxes(
showgrid=True, gridwidth=0.5, gridcolor='#e8e8e8',
zeroline=False, linecolor='#bdbdbd', mirror=False,
tickformat=',',
minor=dict(showgrid=True, gridcolor='#f4f4f4', gridwidth=0.5)
minor=dict(showgrid=True, gridcolor='#f4f4f4', gridwidth=0.5),
)
fig.update_yaxes(
showgrid=False, zeroline=False, linecolor='#bdbdbd',
ticks='outside', ticklen=4, tickcolor='#cccccc',
automargin=True
)
fig.update_traces(
hovertemplate="<b>%{y}</b><br>Count: %{x:,}<br>Share: %{customdata[0]}%<extra></extra>",
marker_line_width=0,
texttemplate='%{x:,}',
textposition='outside',
textfont=dict(size=10, color='#666666'),
cliponaxis=False,
selected=dict(marker=dict(opacity=0.6)),
unselected=dict(marker=dict(opacity=0.2))
ticks='outside', ticklen=4, tickcolor='#cccccc', automargin=True,
)
return fig
if __name__ == "__main__":
parser = argparse.ArgumentParser(description="Analyze CAN bus message frequency")
parser.add_argument("input", type=Path, help="Path to the input CAN log file")
parser.add_argument("output", type=Path, nargs="?", default=Path("freq_report.html"), help="Path to the output HTML report")
parser.add_argument("title", nargs="?", default="CAN Bus Message Frequency", help="Title for the HTML report")
parser.add_argument("output", type=Path, nargs="?", default=Path("freq_report.html"))
parser.add_argument("title", nargs="?", default="Frequency")
args = parser.parse_args()
df = load_data(args.input)
stats = calc_freq(df)
fig = plot_freq(stats, title=args.title)
stats = calculate_frequency(df)
fig = plot_frequency(stats, title=args.title)
config = {
'responsive': True,
'displaylogo': False,
'scrollZoom': True,
'modeBarButtonsToAdd': ['toggleSpikelines'],
'toImageButtonOptions': {'format': 'png', 'scale': 2}
'toImageButtonOptions': {'format': 'png', 'scale': 2},
}
fig.write_html(str(args.output), include_plotlyjs='cdn', config=config)
+143
View File
@@ -0,0 +1,143 @@
# File: id_viewer.py
# Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later
import argparse
from pathlib import Path
import numpy as np
import pandas as pd
import plotly.graph_objects as go
from plotly_resampler import FigureResampler
from stats.utils.extractor import load_data
def _format_can_id_vec(s: pd.Series) -> pd.Series:
s = s.astype('string').str.strip()
s = s.str.replace(r'^0x', '', case=False, regex=True)
s = s.str.upper()
return s.fillna('UNKNOWN').replace('', 'UNKNOWN')
def prepare_data(df, target_id):
can_id_col = 'ID' if 'ID' in df.columns else 'Identifier'
formatted = _format_can_id_vec(df[can_id_col])
df = df.assign(Formatted_ID=formatted)
target_id_clean = _format_can_id_vec(pd.Series([target_id])).iloc[0]
filtered = df[df['Formatted_ID'] == target_id_clean]
byte_cols = [f"b{i}" for i in range(8) if f"b{i}" in filtered.columns]
if filtered.empty:
return filtered, byte_cols
for col in byte_cols:
if not pd.api.types.is_numeric_dtype(filtered[col]):
filtered = filtered.assign(**{col: pd.to_numeric(filtered[col], errors='coerce').astype('float32')})
filtered = filtered.sort_values('Timestamp', kind='stable')
arr = filtered[byte_cols].to_numpy(dtype=np.float32, copy=False)
if len(arr) > 1:
changed = np.any(arr[1:] != arr[:-1], axis=1)
keep = np.concatenate(([True], changed))
filtered = filtered.iloc[keep]
return filtered, byte_cols
def plot_bits(df, byte_cols, can_id, title):
fig = FigureResampler(
resampled_trace_prefix_suffix=("", ""),
show_mean_aggregation_size=False
)
colors = ['#e41a1c', '#377eb8', '#4daf4a', '#984ea3', '#ff7f00', '#ffff33', '#a65628', '#f781bf']
n = len(byte_cols)
x = df['Timestamp'].to_numpy() if not df.empty else np.array([])
for i, col in enumerate(byte_cols):
y = df[col].to_numpy(dtype=np.float32, copy=False) if not df.empty else np.array([])
fig.add_trace(go.Scatter(
mode='lines',
line=dict(shape='hv', width=2, color=colors[i % len(colors)]),
name=col.upper(),
legendgroup=col.upper(),
hovertemplate=f"<b>{col.upper()}</b><br>Time: %{{x}}<br>Value: %{{y}}<extra></extra>",
), hf_x=x, hf_y=y)
all_button = dict(label='ALL', method='restyle', args=[{'visible': [True] * n}])
none_button = dict(label='NONE', method='restyle', args=[{'visible': ['legendonly'] * n}])
fig.update_layout(
height=600,
autosize=True,
template='plotly_white',
title=dict(
text=f"{title} - ID: {can_id}",
font=dict(size=20, color='#1a1a1a'),
x=0.5, xanchor='center',
pad=dict(b=20),
),
font=dict(family="Segoe UI, Arial, sans-serif", size=12, color='#2a2a2a'),
hoverlabel=dict(bgcolor="white", font_size=13, font_family="Segoe UI", bordercolor='#cccccc'),
margin=dict(l=60, r=40, t=120, b=140),
legend=dict(
orientation='h',
x=0.5, xanchor='center',
y=-0.18, yanchor='top',
title=None,
bgcolor='white',
bordercolor='#cccccc',
borderwidth=1,
font=dict(size=12, color="#2a2a2a"),
itemsizing='constant',
itemclick='toggle',
itemdoubleclick='toggleothers',
),
xaxis=dict(
title=dict(text="Timestamp", font=dict(size=13, color="#1a1a1a")),
showgrid=True, gridwidth=0.5, gridcolor='#e8e8e8',
zeroline=False, linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"),
ticks="outside", ticklen=4, tickcolor="#cccccc",
minor=dict(showgrid=True, gridcolor='#f4f4f4', gridwidth=0.5),
),
yaxis=dict(
title=dict(text="Byte Value", font=dict(size=13, color="#1a1a1a")),
showgrid=True, gridwidth=0.5, gridcolor='#e8e8e8',
zeroline=False, linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"),
ticks="outside", ticklen=4, tickcolor="#cccccc",
),
updatemenus=[
dict(
type='buttons',
direction='right',
x=0.5, xanchor='center',
y=-0.06, yanchor='top',
buttons=[all_button, none_button],
bgcolor='white',
bordercolor='#cccccc',
borderwidth=1,
font=dict(size=11, color='#2a2a2a'),
pad=dict(l=5, r=5, t=5, b=5),
)
],
)
return fig
if __name__ == "__main__":
parser = argparse.ArgumentParser(description="Visualize CAN bus byte changes over time")
parser.add_argument("input", type=Path, help="Path to the input CAN log file")
parser.add_argument("can_id", type=str, help="CAN ID to visualize")
parser.add_argument("output", type=Path, nargs="?", default=Path("bits_report.html"))
parser.add_argument("title", nargs="?", default="Byte Visualization")
args = parser.parse_args()
df = load_data(args.input)
filtered_df, byte_cols = prepare_data(df, args.can_id)
fig = plot_bits(filtered_df, byte_cols, args.can_id, title=args.title)
config = {
'responsive': True,
'displaylogo': False,
'scrollZoom': True,
'modeBarButtonsToAdd': ['toggleSpikelines'],
'toImageButtonOptions': {'format': 'png', 'scale': 2},
}
fig.write_html(str(args.output), include_plotlyjs='cdn', config=config)
View File
+63
View File
@@ -0,0 +1,63 @@
# File: extractor.py
# Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later
import json
from pathlib import Path
import numpy as np
import pandas as pd
import polars as pl
def to_int(x):
if isinstance(x, (int, np.integer)):
return int(x)
if isinstance(x, str):
try:
return int(x, 16)
except ValueError:
return np.nan
return np.nan
def extract_id(row: pd.Series) -> str:
meta = row.get('j1939_metadata')
if pd.isna(meta):
return f"ID: {row['ID']}"
if isinstance(meta, str):
try:
meta = json.loads(meta)
except json.JSONDecodeError:
return f"ID: {row['ID']}"
if isinstance(meta, dict) and 'PGN' in meta:
return f"PGN: {meta['PGN']}"
return f"ID: {row['ID']}"
def load_data(file_path: Path) -> pd.DataFrame:
lf = pl.scan_parquet(file_path)
schema = lf.collect_schema()
names = schema.names()
byte_cols = [f"b{i}" for i in range(8) if f"b{i}" in names]
if byte_cols:
lf = lf.with_columns([
pl.col(c).str.to_integer(base=16, strict=False).cast(pl.Int16).alias(c)
for c in byte_cols
])
id_col = 'ID' if 'ID' in names else 'Identifier'
id_expr = pl.col(id_col).cast(pl.Utf8)
if 'j1939_metadata' in names:
try:
lf = lf.with_columns(
pl.when(pl.col('j1939_metadata').is_not_null())
.then(pl.lit('PGN: ') + pl.col('j1939_metadata').struct.field('PGN').cast(pl.Utf8))
.otherwise(pl.lit('ID: ') + id_expr)
.alias('Identifier')
)
except Exception:
lf = lf.with_columns((pl.lit('ID: ') + id_expr).alias('Identifier'))
else:
lf = lf.with_columns((pl.lit('ID: ') + id_expr).alias('Identifier'))
return lf.collect().to_pandas()