From c9461868cc8e679cd1799126dba4e1080a64854b Mon Sep 17 00:00:00 2001 From: Erick Ahmed Date: Tue, 14 Jul 2026 14:14:09 +0200 Subject: [PATCH 01/25] Implement CAN ID visualization at the bit level --- stat/id_viewer.py | 111 ++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 111 insertions(+) create mode 100644 stat/id_viewer.py diff --git a/stat/id_viewer.py b/stat/id_viewer.py new file mode 100644 index 0000000..f657354 --- /dev/null +++ b/stat/id_viewer.py @@ -0,0 +1,111 @@ +import argparse +from pathlib import Path +import pandas as pd +import plotly.graph_objects as go +from utils.extractor import load_data + +def _format_can_id(x): + if pd.isna(x): + return "UNKNOWN" + s = str(x).strip() + if not s: + return "UNKNOWN" + if s.lower().startswith('0x'): + s = s[2:] + return s.upper() + +def prepare_data(df, target_id): + can_id_col = 'ID' if 'ID' in df.columns else 'Identifier' + df['Formatted_ID'] = df[can_id_col].apply(_format_can_id) + target_id_clean = _format_can_id(target_id) + filtered = df[df['Formatted_ID'] == target_id_clean].copy() + + byte_cols = [f"b{i}" for i in range(8)] + for col in byte_cols: + filtered[col] = pd.to_numeric( + filtered[col].apply(lambda x: int(x, 16) if pd.notna(x) else None), + errors='coerce' + ) + + filtered = filtered.sort_values('Timestamp') + mask = (filtered[byte_cols] != filtered[byte_cols].shift()).any(axis=1) + filtered = filtered[mask] + + return filtered, byte_cols + +def plot_bits(df, byte_cols, can_id, title): + fig = go.Figure() + + colors = ['#e41a1c', '#377eb8', '#4daf4a', '#984ea3', '#ff7f00', '#ffff33', '#a65628', '#f781bf'] + + for i, col in enumerate(byte_cols): + fig.add_trace(go.Scatter( + x=df['Timestamp'], + y=df[col], + mode='lines', + line=dict(shape='hv', width=2, color=colors[i]), + name=col.upper(), + hovertemplate=f"{col.upper()}
Time: %{{x}}
Value: %{{y}}" + )) + + fig.update_layout( + height=600, + autosize=True, + template='plotly_white', + title=dict( + text=f"{title} - ID: {can_id}", + font=dict(size=20, color='#1a1a1a'), + x=0.5, xanchor='center', + pad=dict(b=20) + ), + font=dict(family="Segoe UI, Arial, sans-serif", size=12, color='#2a2a2a'), + hoverlabel=dict( + bgcolor="white", font_size=13, font_family="Segoe UI", bordercolor='#cccccc' + ), + margin=dict(l=60, r=40, t=120, b=60), + legend=dict( + title=dict(text="Bytes"), + bgcolor="rgba(255,255,255,0.8)", + bordercolor="#cccccc", + borderwidth=1, + font=dict(size=12, color="#2a2a2a") + ), + xaxis=dict( + title=dict(text="Timestamp", font=dict(size=13, color="#1a1a1a")), + showgrid=True, gridwidth=0.5, gridcolor='#e8e8e8', + zeroline=False, linecolor="#bdbdbd", + tickfont=dict(size=12, color="#2a2a2a"), + ticks="outside", ticklen=4, tickcolor="#cccccc", + minor=dict(showgrid=True, gridcolor='#f4f4f4', gridwidth=0.5) + ), + yaxis=dict( + title=dict(text="Byte Value", font=dict(size=13, color="#1a1a1a")), + showgrid=True, gridwidth=0.5, gridcolor='#e8e8e8', + zeroline=False, linecolor="#bdbdbd", + tickfont=dict(size=12, color="#2a2a2a"), + ticks="outside", ticklen=4, tickcolor="#cccccc" + ) + ) + + return fig + +if __name__ == "__main__": + parser = argparse.ArgumentParser(description="Visualize CAN bus byte changes over time") + parser.add_argument("input", type=Path, help="Path to the input CAN log file") + parser.add_argument("can_id", type=str, help="CAN ID to visualize") + parser.add_argument("output", type=Path, nargs="?", default=Path("bits_report.html"), help="Path to the output HTML report") + parser.add_argument("title", nargs="?", default="Byte Visualization", help="Title for the HTML report") + args = parser.parse_args() + + df = load_data(args.input) + filtered_df, byte_cols = prepare_data(df, args.can_id) + fig = plot_bits(filtered_df, byte_cols, args.can_id, title=args.title) + + config = { + 'responsive': True, + 'displaylogo': False, + 'scrollZoom': True, + 'modeBarButtonsToAdd': ['toggleSpikelines'], + 'toImageButtonOptions': {'format': 'png', 'scale': 2} + } + fig.write_html(str(args.output), include_plotlyjs='cdn', config=config) -- 2.52.0 From fae85d1af9bad5329d50200550de342fcceeb8a4 Mon Sep 17 00:00:00 2001 From: Erick Ahmed Date: Tue, 14 Jul 2026 14:28:15 +0200 Subject: [PATCH 02/25] Change title --- stat/frequency.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/stat/frequency.py b/stat/frequency.py index 70d8002..8c7598f 100644 --- a/stat/frequency.py +++ b/stat/frequency.py @@ -119,7 +119,7 @@ if __name__ == "__main__": parser = argparse.ArgumentParser(description="Analyze CAN bus message frequency") parser.add_argument("input", type=Path, help="Path to the input CAN log file") parser.add_argument("output", type=Path, nargs="?", default=Path("freq_report.html"), help="Path to the output HTML report") - parser.add_argument("title", nargs="?", default="CAN Bus Message Frequency", help="Title for the HTML report") + parser.add_argument("title", nargs="?", default="Frequency", help="Title for the HTML report") args = parser.parse_args() df = load_data(args.input) -- 2.52.0 From 0780a61d78051e589c54d1a9aa63464ad147ae83 Mon Sep 17 00:00:00 2001 From: Erick Ahmed Date: Tue, 14 Jul 2026 23:46:09 +0200 Subject: [PATCH 03/25] Improve bit selection --- stat/id_viewer.py | 59 ++++++++++++++++++++++++++++++++++++++++++----- 1 file changed, 53 insertions(+), 6 deletions(-) diff --git a/stat/id_viewer.py b/stat/id_viewer.py index f657354..8c89267 100644 --- a/stat/id_viewer.py +++ b/stat/id_viewer.py @@ -39,15 +39,38 @@ def plot_bits(df, byte_cols, can_id, title): colors = ['#e41a1c', '#377eb8', '#4daf4a', '#984ea3', '#ff7f00', '#ffff33', '#a65628', '#f781bf'] for i, col in enumerate(byte_cols): + fig.add_trace(go.Scatter( + x=[None], y=[None], + mode='markers', + marker=dict(symbol='square', size=10, color=colors[i]), + name=col.upper(), + showlegend=True, + legendgroup=col.upper(), + hoverinfo='skip' + )) fig.add_trace(go.Scatter( x=df['Timestamp'], y=df[col], mode='lines', line=dict(shape='hv', width=2, color=colors[i]), name=col.upper(), + showlegend=False, + legendgroup=col.upper(), hovertemplate=f"{col.upper()}
Time: %{{x}}
Value: %{{y}}" )) + all_button = dict( + label='ALL', + method='restyle', + args=[{'visible': [True] * 16}] + ) + + none_button = dict( + label='NONE', + method='restyle', + args=[{'visible': ['legendonly'] * 16}] + ) + fig.update_layout( height=600, autosize=True, @@ -62,13 +85,21 @@ def plot_bits(df, byte_cols, can_id, title): hoverlabel=dict( bgcolor="white", font_size=13, font_family="Segoe UI", bordercolor='#cccccc' ), - margin=dict(l=60, r=40, t=120, b=60), + margin=dict(l=60, r=40, t=120, b=140), legend=dict( - title=dict(text="Bytes"), - bgcolor="rgba(255,255,255,0.8)", - bordercolor="#cccccc", + orientation='h', + x=0.5, + xanchor='center', + y=-0.18, + yanchor='top', + title=None, + bgcolor='white', + bordercolor='#cccccc', borderwidth=1, - font=dict(size=12, color="#2a2a2a") + font=dict(size=12, color="#2a2a2a"), + itemsizing='constant', + itemclick='toggle', + itemdoubleclick='toggleothers' ), xaxis=dict( title=dict(text="Timestamp", font=dict(size=13, color="#1a1a1a")), @@ -84,7 +115,23 @@ def plot_bits(df, byte_cols, can_id, title): zeroline=False, linecolor="#bdbdbd", tickfont=dict(size=12, color="#2a2a2a"), ticks="outside", ticklen=4, tickcolor="#cccccc" - ) + ), + updatemenus=[ + dict( + type='buttons', + direction='right', + x=0.5, + xanchor='center', + y=-0.06, + yanchor='top', + buttons=[all_button, none_button], + bgcolor='white', + bordercolor='#cccccc', + borderwidth=1, + font=dict(size=11, color='#2a2a2a'), + pad=dict(l=5, r=5, t=5, b=5) + ) + ] ) return fig -- 2.52.0 From a2ac49c79f14ec78df196793ccf2bdbfedba3702 Mon Sep 17 00:00:00 2001 From: Erick Ahmed Date: Wed, 15 Jul 2026 00:17:59 +0200 Subject: [PATCH 04/25] Refactor statistical analysis modules for performance - Optimize data processing pipelines across files by replacing iterative pandas operations with vectorized NumPy routines --- stat/correlation.py | 119 ++++++++++++++++------------ stat/entropy.py | 171 ++++++++++++++++------------------------ stat/frequency.py | 100 +++++++++++------------ stat/id_viewer.py | 118 ++++++++++++--------------- stat/utils/extractor.py | 35 ++++++-- 5 files changed, 260 insertions(+), 283 deletions(-) diff --git a/stat/correlation.py b/stat/correlation.py index 06aa856..1c232d7 100644 --- a/stat/correlation.py +++ b/stat/correlation.py @@ -12,75 +12,96 @@ import plotly.graph_objects as go from utils.extractor import load_data from utils.extractor import to_int -def _format_can_id(x): - """Safely cleans CAN ID strings without altering their length or value.""" - if pd.isna(x): - return "UNKNOWN" +def _format_can_id_vec(s: pd.Series) -> pd.Series: + s = s.astype('string').str.strip() + s = s.str.replace(r'^0x', '', case=False, regex=True) + s = s.str.upper() + return s.fillna('UNKNOWN').replace('', 'UNKNOWN') - s = str(x).strip() - if not s: - return "UNKNOWN" - - if s.lower().startswith('0x'): - s = s[2:] - - return s.upper() +def _ensure_int_bytes(df: pd.DataFrame, cols: list) -> pd.DataFrame: + needs = [c for c in cols if not pd.api.types.is_numeric_dtype(df[c])] + if needs: + df = df.copy() + for c in needs: + df[c] = df[c].apply(to_int) + return df def calculate_correlation(df: pd.DataFrame, method: str, target_id: str | None = None) -> pd.DataFrame: - """Calculates inter-byte correlation grouped by identifier.""" - byte_cols = [f"b{i}" for i in range(8)] + byte_cols = [f"b{i}" for i in range(8) if f"b{i}" in df.columns] available_cols = [col for col in byte_cols if col in df.columns] if not available_cols: raise ValueError("No byte columns (b0-b7) found in the DataFrame") can_id_col = 'ID' if 'ID' in df.columns else 'Identifier' - identifiers = df[can_id_col].apply(_format_can_id) + identifiers = _format_can_id_vec(df[can_id_col]).to_numpy() - df_bytes = df[available_cols].copy() - for col in available_cols: - df_bytes[col] = df_bytes[col].apply(to_int) + df_bytes = _ensure_int_bytes(df, available_cols)[available_cols] + data = df_bytes.to_numpy(dtype=np.float64, copy=False) if target_id is not None: - target_id = _format_can_id(target_id) - group = df_bytes[identifiers == target_id] - if group.empty: + target_id = _format_can_id_vec(pd.Series([target_id])).iloc[0] + mask = identifiers == target_id + if not mask.any(): raise ValueError(f"Identifier '{target_id}' not found in data") - return group[available_cols].corr(method=method).fillna(0.0) + sub = data[mask] + mask = ~np.isnan(sub).any(axis=1) + sub = sub[mask] + if method == 'spearman' and sub.shape[0] > 1: + sub = pd.DataFrame(sub).rank().to_numpy() + if sub.shape[0] > 1: + with np.errstate(divide='ignore', invalid='ignore'): + c = np.corrcoef(sub, rowvar=False) + np.nan_to_num(c, copy=False, nan=0.0) + else: + c = np.zeros((len(available_cols), len(available_cols))) + return pd.DataFrame(c, index=available_cols, columns=available_cols) - def max_abs_corr(group: pd.DataFrame) -> pd.Series: - corr_arr = np.abs(group.corr(method=method).to_numpy().copy()) - np.fill_diagonal(corr_arr, 0.0) - return pd.Series(corr_arr.max(axis=0), index=group.columns).fillna(0.0) + unique_ids, inverse = np.unique(identifiers, return_inverse=True) + n_cols = len(available_cols) - result = df_bytes.groupby(identifiers)[available_cols].apply(max_abs_corr) + sort_idx = np.argsort(inverse, kind='stable') + data_sorted = data[sort_idx] + inverse_sorted = inverse[sort_idx] + + if len(inverse_sorted) > 0: + split_points = np.flatnonzero(np.diff(inverse_sorted)) + 1 + groups = np.split(data_sorted, split_points) + else: + groups = [] + + out = np.zeros((len(unique_ids), n_cols), dtype=np.float64) + for gi, sub in enumerate(groups): + mask = ~np.isnan(sub).any(axis=1) + sub = sub[mask] + if len(sub) > 1: + if method == 'spearman': + sub = pd.DataFrame(sub).rank().to_numpy() + with np.errstate(divide='ignore', invalid='ignore'): + c = np.abs(np.corrcoef(sub, rowvar=False)) + np.nan_to_num(c, copy=False, nan=0.0) + np.fill_diagonal(c, 0.0) + out[gi] = c.max(axis=0) + + result = pd.DataFrame(out, index=unique_ids, columns=available_cols) result.index.name = 'Identifier' return result def plot_correlation_heatmap(corr_df: pd.DataFrame, target_id: str | None, title: str) -> go.Figure: - """Generates an interactive heatmap of inter-byte correlation.""" is_8x8 = target_id is not None + x = corr_df.columns.tolist() + y = corr_df.index.tolist() + z = corr_df.values + if is_8x8: - x = corr_df.columns.tolist() - y = corr_df.index.tolist() - z = corr_df.values z_min, z_max = -1.0, 1.0 - colorscale = [ - [0.0, "#2c7bb6"], [0.25, "#abd9e9"], [0.5, "#ffffff"], - [0.75, "#fdae61"], [1.0, "#d7191c"] - ] + colorscale = [[0.0, "#2c7bb6"], [0.25, "#abd9e9"], [0.5, "#ffffff"], [0.75, "#fdae61"], [1.0, "#d7191c"]] hover_template = "%{y} vs %{x}
Correlation: %{z:.2f}" else: - x = corr_df.columns.tolist() - y = corr_df.index.tolist() - z = corr_df.values z_min, z_max = 0.0, 1.0 - colorscale = [ - [0.0, "#ffffff"], [0.2, "#fff5f0"], [0.4, "#fecc5c"], - [0.6, "#fd8d3c"], [0.8, "#e31a1c"], [1.0, "#800026"] - ] + colorscale = [[0.0, "#ffffff"], [0.2, "#fff5f0"], [0.4, "#fecc5c"], [0.6, "#fd8d3c"], [0.8, "#e31a1c"], [1.0, "#800026"]] hover_template = "%{y}
Byte %{x} max correlation: %{z:.2f}" fig = go.Figure( @@ -106,17 +127,17 @@ def plot_correlation_heatmap(corr_df: pd.DataFrame, target_id: str | None, title fig.update_layout( title=dict(text=title, font=dict(size=20, color="#1a1a1a"), x=0.5, xanchor="center", pad=dict(b=20)), - height=max(600, len(y) * 28 + 150) if not is_8x8 else 600, + height=600 if is_8x8 else max(600, len(y) * 28 + 150), autosize=True, template="plotly_white", xaxis=dict( title=dict(text="Byte Position", font=dict(size=13, color="#1a1a1a")), - side="top" if not is_8x8 else "bottom", + side="bottom" if is_8x8 else "top", dtick=1, showgrid=False, linecolor="#bdbdbd", tickfont=dict(size=12, color="#2a2a2a"), ticks="outside", ticklen=4, tickcolor="#cccccc", ), yaxis=dict( - title=dict(text="PGN or CAN ID" if not is_8x8 else "Byte Position", font=dict(size=13, color="#1a1a1a")), + title=dict(text="Byte Position" if is_8x8 else "PGN or CAN ID", font=dict(size=13, color="#1a1a1a")), autorange="reversed", showgrid=False, linecolor="#bdbdbd", tickfont=dict(size=12, color="#2a2a2a"), ticks="outside", ticklen=4, tickcolor="#cccccc", automargin=True, ), @@ -131,17 +152,15 @@ if __name__ == "__main__": parser = argparse.ArgumentParser(description="Analyze CAN bus inter-byte correlation") parser.add_argument("method", choices=["pearson", "spearman"], help="Correlation method to use") parser.add_argument("input", type=Path, help="Path to the input CAN log file") - parser.add_argument("output", type=Path, nargs="?", default=Path("correlation_report.html"), help="Path to the output HTML report") - parser.add_argument("title", nargs="?", default="CAN Bus Inter-Byte Correlation", help="Title for the HTML report") - parser.add_argument("--identifier", type=str, default=None, help="Specific PGN/CAN ID to analyze (e.g., 'PGN: 65331'). If omitted, shows max correlation per byte for all IDs.") + parser.add_argument("output", type=Path, nargs="?", default=Path("correlation_report.html")) + parser.add_argument("title", nargs="?", default="CAN Bus Inter-Byte Correlation") + parser.add_argument("--identifier", type=str, default=None) args = parser.parse_args() df = load_data(args.input) corr_df = calculate_correlation(df, method=args.method, target_id=args.identifier) - display_title = f"{args.title} ({args.identifier})" if args.identifier else args.title fig = plot_correlation_heatmap(corr_df, target_id=args.identifier, title=display_title) - config = { "responsive": True, "displaylogo": False, diff --git a/stat/entropy.py b/stat/entropy.py index fc010bc..51240e2 100644 --- a/stat/entropy.py +++ b/stat/entropy.py @@ -12,57 +12,76 @@ import plotly.graph_objects as go from utils.extractor import load_data from utils.extractor import to_int -def _format_can_id(x): - """Safely cleans CAN ID strings without altering their length or value.""" - if pd.isna(x): - return "UNKNOWN" +def _format_can_id_vec(s: pd.Series) -> pd.Series: + s = s.astype('string').str.strip() + s = s.str.replace(r'^0x', '', case=False, regex=True) + s = s.str.upper() + return s.fillna('UNKNOWN').replace('', 'UNKNOWN') - s = str(x).strip() - if not s: - return "UNKNOWN" - - if s.lower().startswith('0x'): - s = s[2:] - - return s.upper() +def _entropy_col(a: np.ndarray) -> float: + a = a[~np.isnan(a)] + if a.size == 0: + return 0.0 + a = a.astype(np.int64) + lo, hi = a.min(), a.max() + span = hi - lo + 1 + if span <= 0: + return 0.0 + if span > 1 << 20: + _, counts = np.unique(a, return_counts=True) + else: + counts = np.bincount(a - lo, minlength=span) + counts = counts[counts > 0] + p = counts / counts.sum() + return float(-np.sum(p * np.log2(p))) def calculate_byte_entropy(df: pd.DataFrame) -> pd.DataFrame: - """Calculates Shannon entropy per byte position for each identifier.""" - byte_cols = [f"b{i}" for i in range(8)] - available_cols = [col for col in byte_cols if col in df.columns] + byte_cols = [f"b{i}" for i in range(8) if f"b{i}" in df.columns] + available_cols = byte_cols if not available_cols: raise ValueError("No byte columns (b0-b7) found in the DataFrame") can_id_col = 'ID' if 'ID' in df.columns else 'Identifier' - identifiers = df[can_id_col].apply(_format_can_id) + identifiers = _format_can_id_vec(df[can_id_col]).to_numpy() - df_bytes = df[available_cols].copy() - for col in available_cols: - df_bytes[col] = df_bytes[col].apply(to_int) + needs = [c for c in available_cols if not pd.api.types.is_numeric_dtype(df[c])] + if needs: + df = df.copy() + for c in needs: + df[c] = df[c].apply(to_int) - def entropy(s: pd.Series) -> float: - s = s.dropna() - if s.empty: - return 0.0 - p = s.value_counts(normalize=True) - return -np.sum(p * np.log2(p)) + data = df[available_cols].to_numpy(dtype=np.float64, copy=False) + unique_ids, inverse = np.unique(identifiers, return_inverse=True) + n_cols = len(available_cols) - result = df_bytes.groupby(identifiers)[available_cols].agg(entropy) + sort_idx = np.argsort(inverse, kind='stable') + data_sorted = data[sort_idx] + inverse_sorted = inverse[sort_idx] + + if len(inverse_sorted) > 0: + split_points = np.flatnonzero(np.diff(inverse_sorted)) + 1 + groups = np.split(data_sorted, split_points) + else: + groups = [] + + out = np.zeros((len(unique_ids), n_cols), dtype=np.float64) + for gi, sub in enumerate(groups): + for ci in range(n_cols): + out[gi, ci] = _entropy_col(sub[:, ci]) + + result = pd.DataFrame(out, index=unique_ids, columns=available_cols) result.index.name = 'Identifier' return result def plot_entropy_heatmap(entropy_df: pd.DataFrame, title: str) -> go.Figure: - """Generates an interactive heatmap of byte-level Shannon entropy.""" x = entropy_df.columns.tolist() y = entropy_df.index.tolist() z = entropy_df.values fig = go.Figure( data=go.Heatmap( - z=z, - x=x, - y=y, + z=z, x=x, y=y, colorscale=[ [0.0, "#ffffff"], [0.15, "#fff7ec"], @@ -71,112 +90,54 @@ def plot_entropy_heatmap(entropy_df: pd.DataFrame, title: str) -> go.Figure: [0.75, "#fdbb84"], [1.0, "#ef6548"], ], - xgap=3, - ygap=3, + xgap=3, ygap=3, text=np.round(z, 2), texttemplate="%{text}", - textfont={ - "size": 11, - "color": "#2a2a2a", - "family": "Segoe UI, Arial, sans-serif", - }, + textfont={"size": 11, "color": "#2a2a2a", "family": "Segoe UI, Arial, sans-serif"}, hoverongaps=False, - hovertemplate=( - "%{y}
" - "Byte %{x}: %{z:.2f} bits" - ), + hovertemplate="%{y}
Byte %{x}: %{z:.2f} bits", colorbar=dict( - title=dict( - text="Entropy (bits)", - side="top", - font=dict(size=13, color="#1a1a1a"), - ), - orientation="h", - thickness=15, - len=0.35, - x=1.0, - xanchor="right", - y=1.02, - yanchor="bottom", + title=dict(text="Entropy (bits)", side="top", font=dict(size=13, color="#1a1a1a")), + orientation="h", thickness=15, len=0.35, + x=1.0, xanchor="right", y=1.02, yanchor="bottom", tickfont=dict(size=11, color="#2a2a2a"), - tickformat=".1f", - outlinewidth=0.5, - outlinecolor="#cccccc", + tickformat=".1f", outlinewidth=0.5, outlinecolor="#cccccc", ), ) ) fig.update_layout( - title=dict( - text=title, - font=dict(size=20, color="#1a1a1a"), - x=0.5, - xanchor="center", - pad=dict(b=20), - ), + title=dict(text=title, font=dict(size=20, color="#1a1a1a"), x=0.5, xanchor="center", pad=dict(b=20)), height=max(600, len(y) * 28 + 150), autosize=True, template="plotly_white", xaxis=dict( title=dict(text="Byte Position", font=dict(size=13, color="#1a1a1a")), - side="top", - dtick=1, - showgrid=False, - linecolor="#bdbdbd", - tickfont=dict(size=12, color="#2a2a2a"), - ticks="outside", - ticklen=4, - tickcolor="#cccccc", + side="top", dtick=1, showgrid=False, linecolor="#bdbdbd", + tickfont=dict(size=12, color="#2a2a2a"), ticks="outside", ticklen=4, tickcolor="#cccccc", ), yaxis=dict( title=dict(text="PGN or CAN ID", font=dict(size=13, color="#1a1a1a")), - autorange="reversed", - showgrid=False, - linecolor="#bdbdbd", - tickfont=dict(size=12, color="#2a2a2a"), - ticks="outside", - ticklen=4, - tickcolor="#cccccc", - automargin=True, + autorange="reversed", showgrid=False, linecolor="#bdbdbd", + tickfont=dict(size=12, color="#2a2a2a"), ticks="outside", ticklen=4, tickcolor="#cccccc", automargin=True, ), font=dict(family="Segoe UI, Arial, sans-serif", size=12, color="#2a2a2a"), - hoverlabel=dict( - bgcolor="white", - font_size=13, - font_family="Segoe UI", - bordercolor="#cccccc", - ), + hoverlabel=dict(bgcolor="white", font_size=13, font_family="Segoe UI", bordercolor="#cccccc"), margin=dict(l=200, r=40, t=120, b=60), ) return fig if __name__ == "__main__": - parser = argparse.ArgumentParser( - description="Analyze CAN bus byte-level entropy" - ) - parser.add_argument( - "input", type=Path, help="Path to the input CAN log file" - ) - parser.add_argument( - "output", - type=Path, - nargs="?", - default=Path("entropy_report.html"), - help="Path to the output HTML report", - ) - parser.add_argument( - "title", - nargs="?", - default="CAN Bus Byte-Level Entropy", - help="Title for the HTML report", - ) + parser = argparse.ArgumentParser(description="Analyze CAN bus byte-level entropy") + parser.add_argument("input", type=Path, help="Path to the input CAN log file") + parser.add_argument("output", type=Path, nargs="?", default=Path("entropy_report.html")) + parser.add_argument("title", nargs="?", default="CAN Bus Byte-Level Entropy") args = parser.parse_args() df = load_data(args.input) entropy_df = calculate_byte_entropy(df) fig = plot_entropy_heatmap(entropy_df, title=args.title) - config = { "responsive": True, "displaylogo": False, diff --git a/stat/frequency.py b/stat/frequency.py index 8c7598f..1e8981b 100644 --- a/stat/frequency.py +++ b/stat/frequency.py @@ -4,52 +4,57 @@ import argparse from pathlib import Path +import numpy as np import pandas as pd -import plotly.express as px import plotly.graph_objects as go from utils.extractor import load_data -def _format_can_id(x): - """Safely cleans CAN ID strings without altering their length or value.""" - if pd.isna(x): - return "UNKNOWN" - - s = str(x).strip() - if not s: - return "UNKNOWN" - - if s.lower().startswith('0x'): - s = s[2:] - - return s.upper() +def _format_can_id_vec(s: pd.Series) -> pd.Series: + s = s.astype('string').str.strip() + s = s.str.replace(r'^0x', '', case=False, regex=True) + s = s.str.upper() + return s.fillna('UNKNOWN').replace('', 'UNKNOWN') def calculate_frequency(df: pd.DataFrame) -> pd.DataFrame: - """Calculates frequency counts and percentages for identifiers.""" can_id_col = 'ID' if 'ID' in df.columns else 'Identifier' + formatted = _format_can_id_vec(df[can_id_col]) + df['Formatted_ID'] = formatted - df['Formatted_ID'] = df[can_id_col].apply(_format_can_id) - - freq_df = df['Formatted_ID'].value_counts().reset_index() - freq_df.columns = ['Identifier', 'Count'] - - total = freq_df['Count'].sum() - freq_df['Percentage'] = (freq_df['Count'] / total * 100).round(2) - return freq_df.sort_values('Count', ascending=True) + counts = formatted.value_counts() + freq_df = pd.DataFrame({ + 'Identifier': counts.index, + 'Count': counts.to_numpy(), + }) + total = counts.sum() + freq_df['Percentage'] = np.round(freq_df['Count'] / total * 100, 2) if total else 0.0 + return freq_df.sort_values('Count', ascending=True).reset_index(drop=True) def plot_frequency(stats_df: pd.DataFrame, title: str) -> go.Figure: - """Generates interactive horizontal bar chart with log x-axis.""" - fig = px.bar( - stats_df, y='Identifier', x='Count', orientation='h', title=title, log_x=True, - labels={'Identifier': 'PGN / CAN ID', 'Count': 'Message Count'}, - color='Count', color_continuous_scale='Turbo', - range_color=(stats_df['Count'].min(), stats_df['Count'].max()), - hover_data={'Percentage': ':.2f', 'Count': ':,', 'Identifier': True} - ) + n = len(stats_df) + fig = go.Figure(go.Bar( + y=stats_df['Identifier'], + x=stats_df['Count'], + orientation='h', + marker=dict( + color=stats_df['Count'], + colorscale='Turbo', + cmin=int(stats_df['Count'].min()) if n else 0, + cmax=int(stats_df['Count'].max()) if n else 1, + line_width=0, + ), + customdata=stats_df[['Percentage']].to_numpy(), + hovertemplate="%{y}
Count: %{x:,}
Share: %{customdata[0]}%", + texttemplate='%{x:,}', + textposition='outside', + cliponaxis=False, + )) + fig.update_layout( - height=max(600, len(stats_df) * 18), + height=max(600, n * 18), autosize=True, template='plotly_white', xaxis=dict( + type='log', title=dict(text="Message count [log scale]", font=dict(size=13, color="#1a1a1a")), side="top", dtick=1, @@ -69,11 +74,10 @@ def plot_frequency(stats_df: pd.DataFrame, title: str) -> go.Figure: ticklen=4, tickcolor="#cccccc", automargin=True, - type='category' + type='category', ), font=dict(family="Segoe UI, Arial, sans-serif", size=12, color='#2a2a2a'), - hoverlabel=dict(bgcolor="white", font_size=13, font_family="Segoe UI", - bordercolor='#cccccc'), + hoverlabel=dict(bgcolor="white", font_size=13, font_family="Segoe UI", bordercolor='#cccccc'), margin=dict(l=200, r=40, t=120, b=60), bargap=0.35, coloraxis_colorbar=dict( @@ -87,39 +91,27 @@ def plot_frequency(stats_df: pd.DataFrame, title: str) -> go.Figure: yanchor='bottom', tickformat=',', outlinecolor='#cccccc', - outlinewidth=0.5 + outlinewidth=0.5, ), - title=dict(font=dict(size=20, color='#1a1a1a'), x=0.5, xanchor='center', - pad=dict(b=20)) + title=dict(text=title, font=dict(size=20, color='#1a1a1a'), x=0.5, xanchor='center', pad=dict(b=20)), ) fig.update_xaxes( showgrid=True, gridwidth=0.5, gridcolor='#e8e8e8', zeroline=False, linecolor='#bdbdbd', mirror=False, tickformat=',', - minor=dict(showgrid=True, gridcolor='#f4f4f4', gridwidth=0.5) + minor=dict(showgrid=True, gridcolor='#f4f4f4', gridwidth=0.5), ) fig.update_yaxes( showgrid=False, zeroline=False, linecolor='#bdbdbd', - ticks='outside', ticklen=4, tickcolor='#cccccc', - automargin=True - ) - fig.update_traces( - hovertemplate="%{y}
Count: %{x:,}
Share: %{customdata[0]}%", - marker_line_width=0, - texttemplate='%{x:,}', - textposition='outside', - textfont=dict(size=10, color='#666666'), - cliponaxis=False, - selected=dict(marker=dict(opacity=0.6)), - unselected=dict(marker=dict(opacity=0.2)) + ticks='outside', ticklen=4, tickcolor='#cccccc', automargin=True, ) return fig if __name__ == "__main__": parser = argparse.ArgumentParser(description="Analyze CAN bus message frequency") parser.add_argument("input", type=Path, help="Path to the input CAN log file") - parser.add_argument("output", type=Path, nargs="?", default=Path("freq_report.html"), help="Path to the output HTML report") - parser.add_argument("title", nargs="?", default="Frequency", help="Title for the HTML report") + parser.add_argument("output", type=Path, nargs="?", default=Path("freq_report.html")) + parser.add_argument("title", nargs="?", default="Frequency") args = parser.parse_args() df = load_data(args.input) @@ -130,6 +122,6 @@ if __name__ == "__main__": 'displaylogo': False, 'scrollZoom': True, 'modeBarButtonsToAdd': ['toggleSpikelines'], - 'toImageButtonOptions': {'format': 'png', 'scale': 2} + 'toImageButtonOptions': {'format': 'png', 'scale': 2}, } fig.write_html(str(args.output), include_plotlyjs='cdn', config=config) diff --git a/stat/id_viewer.py b/stat/id_viewer.py index 8c89267..273a041 100644 --- a/stat/id_viewer.py +++ b/stat/id_viewer.py @@ -1,75 +1,64 @@ +# File: id_viewer.py +# Copyright (C) 2026 Erick Ahmed +# SPDX-License-Identifier: AGPL-3.0-or-later + import argparse from pathlib import Path +import numpy as np import pandas as pd import plotly.graph_objects as go from utils.extractor import load_data -def _format_can_id(x): - if pd.isna(x): - return "UNKNOWN" - s = str(x).strip() - if not s: - return "UNKNOWN" - if s.lower().startswith('0x'): - s = s[2:] - return s.upper() +def _format_can_id_vec(s: pd.Series) -> pd.Series: + s = s.astype('string').str.strip() + s = s.str.replace(r'^0x', '', case=False, regex=True) + s = s.str.upper() + return s.fillna('UNKNOWN').replace('', 'UNKNOWN') def prepare_data(df, target_id): can_id_col = 'ID' if 'ID' in df.columns else 'Identifier' - df['Formatted_ID'] = df[can_id_col].apply(_format_can_id) - target_id_clean = _format_can_id(target_id) - filtered = df[df['Formatted_ID'] == target_id_clean].copy() + formatted = _format_can_id_vec(df[can_id_col]) + df = df.assign(Formatted_ID=formatted) + target_id_clean = _format_can_id_vec(pd.Series([target_id])).iloc[0] + filtered = df[df['Formatted_ID'] == target_id_clean] + + byte_cols = [f"b{i}" for i in range(8) if f"b{i}" in filtered.columns] + if filtered.empty: + return filtered, byte_cols - byte_cols = [f"b{i}" for i in range(8)] for col in byte_cols: - filtered[col] = pd.to_numeric( - filtered[col].apply(lambda x: int(x, 16) if pd.notna(x) else None), - errors='coerce' - ) + if not pd.api.types.is_numeric_dtype(filtered[col]): + filtered = filtered.assign(**{col: pd.to_numeric(filtered[col], errors='coerce').astype('float32')}) - filtered = filtered.sort_values('Timestamp') - mask = (filtered[byte_cols] != filtered[byte_cols].shift()).any(axis=1) - filtered = filtered[mask] + filtered = filtered.sort_values('Timestamp', kind='stable') + arr = filtered[byte_cols].to_numpy(dtype=np.float32, copy=False) + if len(arr) > 1: + changed = np.any(arr[1:] != arr[:-1], axis=1) + keep = np.concatenate(([True], changed)) + filtered = filtered.iloc[keep] return filtered, byte_cols def plot_bits(df, byte_cols, can_id, title): fig = go.Figure() - colors = ['#e41a1c', '#377eb8', '#4daf4a', '#984ea3', '#ff7f00', '#ffff33', '#a65628', '#f781bf'] + n = len(byte_cols) + x = df['Timestamp'].to_numpy() if not df.empty else np.array([]) for i, col in enumerate(byte_cols): - fig.add_trace(go.Scatter( - x=[None], y=[None], - mode='markers', - marker=dict(symbol='square', size=10, color=colors[i]), - name=col.upper(), - showlegend=True, - legendgroup=col.upper(), - hoverinfo='skip' - )) - fig.add_trace(go.Scatter( - x=df['Timestamp'], - y=df[col], + y = df[col].to_numpy(dtype=np.float32, copy=False) if not df.empty else np.array([]) + fig.add_trace(go.Scattergl( + x=x, + y=y, mode='lines', - line=dict(shape='hv', width=2, color=colors[i]), + line=dict(shape='hv', width=2, color=colors[i % len(colors)]), name=col.upper(), - showlegend=False, legendgroup=col.upper(), - hovertemplate=f"{col.upper()}
Time: %{{x}}
Value: %{{y}}" + hovertemplate=f"{col.upper()}
Time: %{{x}}
Value: %{{y}}", )) - all_button = dict( - label='ALL', - method='restyle', - args=[{'visible': [True] * 16}] - ) - - none_button = dict( - label='NONE', - method='restyle', - args=[{'visible': ['legendonly'] * 16}] - ) + all_button = dict(label='ALL', method='restyle', args=[{'visible': [True] * n}]) + none_button = dict(label='NONE', method='restyle', args=[{'visible': ['legendonly'] * n}]) fig.update_layout( height=600, @@ -79,19 +68,15 @@ def plot_bits(df, byte_cols, can_id, title): text=f"{title} - ID: {can_id}", font=dict(size=20, color='#1a1a1a'), x=0.5, xanchor='center', - pad=dict(b=20) + pad=dict(b=20), ), font=dict(family="Segoe UI, Arial, sans-serif", size=12, color='#2a2a2a'), - hoverlabel=dict( - bgcolor="white", font_size=13, font_family="Segoe UI", bordercolor='#cccccc' - ), + hoverlabel=dict(bgcolor="white", font_size=13, font_family="Segoe UI", bordercolor='#cccccc'), margin=dict(l=60, r=40, t=120, b=140), legend=dict( orientation='h', - x=0.5, - xanchor='center', - y=-0.18, - yanchor='top', + x=0.5, xanchor='center', + y=-0.18, yanchor='top', title=None, bgcolor='white', bordercolor='#cccccc', @@ -99,7 +84,7 @@ def plot_bits(df, byte_cols, can_id, title): font=dict(size=12, color="#2a2a2a"), itemsizing='constant', itemclick='toggle', - itemdoubleclick='toggleothers' + itemdoubleclick='toggleothers', ), xaxis=dict( title=dict(text="Timestamp", font=dict(size=13, color="#1a1a1a")), @@ -107,41 +92,38 @@ def plot_bits(df, byte_cols, can_id, title): zeroline=False, linecolor="#bdbdbd", tickfont=dict(size=12, color="#2a2a2a"), ticks="outside", ticklen=4, tickcolor="#cccccc", - minor=dict(showgrid=True, gridcolor='#f4f4f4', gridwidth=0.5) + minor=dict(showgrid=True, gridcolor='#f4f4f4', gridwidth=0.5), ), yaxis=dict( title=dict(text="Byte Value", font=dict(size=13, color="#1a1a1a")), showgrid=True, gridwidth=0.5, gridcolor='#e8e8e8', zeroline=False, linecolor="#bdbdbd", tickfont=dict(size=12, color="#2a2a2a"), - ticks="outside", ticklen=4, tickcolor="#cccccc" + ticks="outside", ticklen=4, tickcolor="#cccccc", ), updatemenus=[ dict( type='buttons', direction='right', - x=0.5, - xanchor='center', - y=-0.06, - yanchor='top', + x=0.5, xanchor='center', + y=-0.06, yanchor='top', buttons=[all_button, none_button], bgcolor='white', bordercolor='#cccccc', borderwidth=1, font=dict(size=11, color='#2a2a2a'), - pad=dict(l=5, r=5, t=5, b=5) + pad=dict(l=5, r=5, t=5, b=5), ) - ] + ], ) - return fig if __name__ == "__main__": parser = argparse.ArgumentParser(description="Visualize CAN bus byte changes over time") parser.add_argument("input", type=Path, help="Path to the input CAN log file") parser.add_argument("can_id", type=str, help="CAN ID to visualize") - parser.add_argument("output", type=Path, nargs="?", default=Path("bits_report.html"), help="Path to the output HTML report") - parser.add_argument("title", nargs="?", default="Byte Visualization", help="Title for the HTML report") + parser.add_argument("output", type=Path, nargs="?", default=Path("bits_report.html")) + parser.add_argument("title", nargs="?", default="Byte Visualization") args = parser.parse_args() df = load_data(args.input) @@ -153,6 +135,6 @@ if __name__ == "__main__": 'displaylogo': False, 'scrollZoom': True, 'modeBarButtonsToAdd': ['toggleSpikelines'], - 'toImageButtonOptions': {'format': 'png', 'scale': 2} + 'toImageButtonOptions': {'format': 'png', 'scale': 2}, } fig.write_html(str(args.output), include_plotlyjs='cdn', config=config) diff --git a/stat/utils/extractor.py b/stat/utils/extractor.py index c2d63e7..3472f07 100644 --- a/stat/utils/extractor.py +++ b/stat/utils/extractor.py @@ -7,9 +7,9 @@ from pathlib import Path import numpy as np import pandas as pd +import polars as pl def to_int(x): - """Convert a hex string or integer to int, returning NaN on failure.""" if isinstance(x, (int, np.integer)): return int(x) if isinstance(x, str): @@ -20,7 +20,6 @@ def to_int(x): return np.nan def extract_id(row: pd.Series) -> str: - """Extracts PGN from metadata or falls back to CAN ID.""" meta = row.get('j1939_metadata') if pd.isna(meta): return f"ID: {row['ID']}" @@ -34,7 +33,31 @@ def extract_id(row: pd.Series) -> str: return f"ID: {row['ID']}" def load_data(file_path: Path) -> pd.DataFrame: - """Loads Parquet file and adds an Identifier column.""" - df = pd.read_parquet(file_path) - df['Identifier'] = df.apply(extract_id, axis=1) - return df + lf = pl.scan_parquet(file_path) + schema = lf.collect_schema() + names = schema.names() + + byte_cols = [f"b{i}" for i in range(8) if f"b{i}" in names] + if byte_cols: + lf = lf.with_columns([ + pl.col(c).str.to_integer(base=16, strict=False).cast(pl.Int16).alias(c) + for c in byte_cols + ]) + + id_col = 'ID' if 'ID' in names else 'Identifier' + id_expr = pl.col(id_col).cast(pl.Utf8) + + if 'j1939_metadata' in names: + try: + lf = lf.with_columns( + pl.when(pl.col('j1939_metadata').is_not_null()) + .then(pl.lit('PGN: ') + pl.col('j1939_metadata').struct.field('PGN').cast(pl.Utf8)) + .otherwise(pl.lit('ID: ') + id_expr) + .alias('Identifier') + ) + except Exception: + lf = lf.with_columns((pl.lit('ID: ') + id_expr).alias('Identifier')) + else: + lf = lf.with_columns((pl.lit('ID: ') + id_expr).alias('Identifier')) + + return lf.collect().to_pandas() -- 2.52.0 From 7a0e2efba175c9ab3035edcd4ac208a14d388ea1 Mon Sep 17 00:00:00 2001 From: Erick Ahmed Date: Wed, 15 Jul 2026 00:18:27 +0200 Subject: [PATCH 05/25] Integrate Dash background callbacks to handle computations in async --- main.py | 48 ++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 48 insertions(+) diff --git a/main.py b/main.py index 3735861..2249de6 100644 --- a/main.py +++ b/main.py @@ -1,3 +1,51 @@ # File: main.py # Copyright (C) 2026 Erick Ahmed # SPDX-License-Identifier: AGPL-3.0-or-later + +import diskcache +import flask_caching +import dash +from dash import Input, Output, State, dcc, html, no_update +import dash_bootstrap_components as dbc + +CACHE_DATA_DIR = ".cache_data" +background_callback_manager = dash.DiskcacheManager(cache_dir=CACHE_DATA_DIR) +data_cache = flask_caching.Cache(config={'CACHE_TYPE': 'FileSystemCache', 'CACHE_DIR': CACHE_DATA_DIR}) + +app = dash.Dash( + __name__, + external_stylesheets=[dbc.themes.BOOTSTRAP], + background_callback_manager=background_callback_manager +) +data_cache.init_app(app.server) + +def _get_df(session_data): + if not session_data or "token" not in session_data: + return None + return data_cache.get(session_data["token"]) + +@app.callback( + Output("graph-correlation", "figure"), + Input("corr-method", "value"), + Input("corr-target", "value"), + Input("session-store", "data"), + + background=True, + prevent_initial_call=True, +) +def update_correlation(method, target, session_data): + df = _get_df(session_data) + if df is None or not method: + return no_update + + target_id = None if target == "all" else target + try: + corr_df = calculate_correlation(df, method=method, target_id=target_id) + title = f"Inter-Byte Correlation ({method.capitalize()})" + if target_id: + title += f" - {target_id}" + return plot_correlation_heatmap(corr_df, target_id=target_id, title=title) + except Exception as exc: + fig = dash.go.Figure() + fig.update_layout(title=f"Error: {exc}") + return fig -- 2.52.0 From 73e76d3adc52067853a7ce9c792558650676741b Mon Sep 17 00:00:00 2001 From: Erick Ahmed Date: Wed, 15 Jul 2026 00:31:27 +0200 Subject: [PATCH 06/25] Rename to avoid conflict with Python stat module --- {stat => stats}/correlation.py | 0 {stat => stats}/entropy.py | 0 {stat => stats}/frequency.py | 0 {stat => stats}/id_viewer.py | 0 {stat => stats}/utils/extractor.py | 0 5 files changed, 0 insertions(+), 0 deletions(-) rename {stat => stats}/correlation.py (100%) rename {stat => stats}/entropy.py (100%) rename {stat => stats}/frequency.py (100%) rename {stat => stats}/id_viewer.py (100%) rename {stat => stats}/utils/extractor.py (100%) diff --git a/stat/correlation.py b/stats/correlation.py similarity index 100% rename from stat/correlation.py rename to stats/correlation.py diff --git a/stat/entropy.py b/stats/entropy.py similarity index 100% rename from stat/entropy.py rename to stats/entropy.py diff --git a/stat/frequency.py b/stats/frequency.py similarity index 100% rename from stat/frequency.py rename to stats/frequency.py diff --git a/stat/id_viewer.py b/stats/id_viewer.py similarity index 100% rename from stat/id_viewer.py rename to stats/id_viewer.py diff --git a/stat/utils/extractor.py b/stats/utils/extractor.py similarity index 100% rename from stat/utils/extractor.py rename to stats/utils/extractor.py -- 2.52.0 From 3df42fb497c2cb0f51daefd18f1f926f738e33e8 Mon Sep 17 00:00:00 2001 From: Erick Ahmed Date: Wed, 15 Jul 2026 00:37:19 +0200 Subject: [PATCH 07/25] Make subdirectories Python packages --- stats/__init__.py | 0 stats/utils/__init__.py | 0 2 files changed, 0 insertions(+), 0 deletions(-) create mode 100644 stats/__init__.py create mode 100644 stats/utils/__init__.py diff --git a/stats/__init__.py b/stats/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/stats/utils/__init__.py b/stats/utils/__init__.py new file mode 100644 index 0000000..e69de29 -- 2.52.0 From c98563f5418f43a57c58a4619d2de700986d3f19 Mon Sep 17 00:00:00 2001 From: Erick Ahmed Date: Wed, 15 Jul 2026 00:48:30 +0200 Subject: [PATCH 08/25] Fix schema check and update import paths - Use `collect_schema` for accurate column validation in Polars and correct relative import paths for statistical modules. --- parser.py | 2 +- stats/correlation.py | 4 ++-- stats/entropy.py | 4 ++-- stats/frequency.py | 2 +- stats/id_viewer.py | 2 +- 5 files changed, 7 insertions(+), 7 deletions(-) diff --git a/parser.py b/parser.py index ad7fc32..1d84f6e 100644 --- a/parser.py +++ b/parser.py @@ -77,7 +77,7 @@ def parse_csv(csv_path: PathLike) -> pl.LazyFrame: """ lf = pl.scan_csv(csv_path, schema_overrides={"ID": pl.String, "Data": pl.String}) - if "Timestamp" not in lf.columns: + if "Timestamp" not in lf.collect_schema().names(): lf = lf.with_row_index("Timestamp") byte_exprs = [] diff --git a/stats/correlation.py b/stats/correlation.py index 1c232d7..d8a9811 100644 --- a/stats/correlation.py +++ b/stats/correlation.py @@ -9,8 +9,8 @@ import numpy as np import pandas as pd import plotly.graph_objects as go -from utils.extractor import load_data -from utils.extractor import to_int +from stats.utils.extractor import load_data +from stats.utils.extractor import to_int def _format_can_id_vec(s: pd.Series) -> pd.Series: s = s.astype('string').str.strip() diff --git a/stats/entropy.py b/stats/entropy.py index 51240e2..11d053b 100644 --- a/stats/entropy.py +++ b/stats/entropy.py @@ -9,8 +9,8 @@ import numpy as np import pandas as pd import plotly.graph_objects as go -from utils.extractor import load_data -from utils.extractor import to_int +from stats.utils.extractor import load_data +from stats.utils.extractor import to_int def _format_can_id_vec(s: pd.Series) -> pd.Series: s = s.astype('string').str.strip() diff --git a/stats/frequency.py b/stats/frequency.py index 1e8981b..0db6b30 100644 --- a/stats/frequency.py +++ b/stats/frequency.py @@ -7,7 +7,7 @@ from pathlib import Path import numpy as np import pandas as pd import plotly.graph_objects as go -from utils.extractor import load_data +from stats.utils.extractor import load_data def _format_can_id_vec(s: pd.Series) -> pd.Series: s = s.astype('string').str.strip() diff --git a/stats/id_viewer.py b/stats/id_viewer.py index 273a041..59f00ac 100644 --- a/stats/id_viewer.py +++ b/stats/id_viewer.py @@ -7,7 +7,7 @@ from pathlib import Path import numpy as np import pandas as pd import plotly.graph_objects as go -from utils.extractor import load_data +from stats.utils.extractor import load_data def _format_can_id_vec(s: pd.Series) -> pd.Series: s = s.astype('string').str.strip() -- 2.52.0 From 65591bbc6b571ff1d59dcc2f0b0bb39b965e6bf3 Mon Sep 17 00:00:00 2001 From: Erick Ahmed Date: Wed, 15 Jul 2026 00:48:51 +0200 Subject: [PATCH 09/25] Refactor main application to use Polars pipeline - replaced the caching layer with a pre-processing pipeline that parses raw logs into decoded Parquet files --- main.py | 193 +++++++++++++++++++++++++++++++++++++++++++++----------- 1 file changed, 155 insertions(+), 38 deletions(-) diff --git a/main.py b/main.py index 2249de6..9d16086 100644 --- a/main.py +++ b/main.py @@ -2,50 +2,167 @@ # Copyright (C) 2026 Erick Ahmed # SPDX-License-Identifier: AGPL-3.0-or-later -import diskcache -import flask_caching +import os +from pathlib import Path + +import polars as pl import dash -from dash import Input, Output, State, dcc, html, no_update +from dash import dcc, html, Input, Output import dash_bootstrap_components as dbc -CACHE_DATA_DIR = ".cache_data" -background_callback_manager = dash.DiskcacheManager(cache_dir=CACHE_DATA_DIR) -data_cache = flask_caching.Cache(config={'CACHE_TYPE': 'FileSystemCache', 'CACHE_DIR': CACHE_DATA_DIR}) +from parser import parse_log, parse_csv +from decoder import decode_j1939_frames +from stats.utils.extractor import load_data -app = dash.Dash( - __name__, - external_stylesheets=[dbc.themes.BOOTSTRAP], - background_callback_manager=background_callback_manager -) -data_cache.init_app(app.server) +from stats.frequency import calculate_frequency, plot_frequency +from stats.id_viewer import prepare_data, plot_bits +from stats.correlation import calculate_correlation, plot_correlation_heatmap +from stats.entropy import calculate_byte_entropy, plot_entropy_heatmap -def _get_df(session_data): - if not session_data or "token" not in session_data: - return None - return data_cache.get(session_data["token"]) +RAW_LOG = "data/logs/rawlog.txt" +BUS1_CSV = "data/csv/bus1.csv" +BUS2_CSV = "data/csv/bus2.csv" +BUS1_PARQUET = "data/parquet/bus1.parquet" +BUS2_PARQUET = "data/parquet/bus2.parquet" +BUS1_DECODED = "data/parquet/bus1_decoded.parquet" +BUS2_DECODED = "data/parquet/bus2_decoded.parquet" + +def run_pipeline(): + os.makedirs("data/logs", exist_ok=True) + if not Path(BUS1_DECODED).exists() or not Path(BUS2_DECODED).exists(): + print("Parsing raw log...") + parse_log(RAW_LOG, BUS1_CSV, BUS2_CSV) + + print("Converting to parquet...") + lf1 = parse_csv(BUS1_CSV) + lf1.sink_parquet(BUS1_PARQUET) + lf2 = parse_csv(BUS2_CSV) + lf2.sink_parquet(BUS2_PARQUET) + + print("Decoding J1939...") + df1 = pl.read_parquet(BUS1_PARQUET) + df2 = pl.read_parquet(BUS2_PARQUET) + dec1 = decode_j1939_frames(df1) + dec2 = decode_j1939_frames(df2) + dec1.write_parquet(BUS1_DECODED) + dec2.write_parquet(BUS2_DECODED) + +run_pipeline() + +print("Loading data into memory...") +DATA = { + "Bus 1": load_data(BUS1_DECODED), + "Bus 2": load_data(BUS2_DECODED) +} + +app = dash.Dash(__name__, external_stylesheets=[dbc.themes.BOOTSTRAP]) + +app.layout = dbc.Container([ + html.H1("CAN Bus Analyzer", className="my-4"), + dbc.Row([ + dbc.Col(html.Label("Select Bus:"), width=1, className="mt-2"), + dbc.Col(dcc.Dropdown( + id='bus-selector', + options=[{'label': k, 'value': k} for k in DATA.keys()], + value='Bus 1', + clearable=False + ), width=2), + ], className="mb-3"), + dbc.Tabs([ + dbc.Tab(label="Frequency", tab_id="freq"), + dbc.Tab(label="ID Viewer", tab_id="id_viewer"), + dbc.Tab(label="Correlation", tab_id="corr"), + dbc.Tab(label="Entropy", tab_id="entropy"), + ], id="tabs", active_tab="freq"), + html.Div(id="tab-content", className="mt-3") +], fluid=True) @app.callback( - Output("graph-correlation", "figure"), - Input("corr-method", "value"), - Input("corr-target", "value"), - Input("session-store", "data"), - - background=True, - prevent_initial_call=True, + Output('tab-content', 'children'), + Input('tabs', 'active_tab'), + Input('bus-selector', 'value') ) -def update_correlation(method, target, session_data): - df = _get_df(session_data) - if df is None or not method: - return no_update +def render_content(tab, bus): + df = DATA[bus] - target_id = None if target == "all" else target - try: - corr_df = calculate_correlation(df, method=method, target_id=target_id) - title = f"Inter-Byte Correlation ({method.capitalize()})" - if target_id: - title += f" - {target_id}" - return plot_correlation_heatmap(corr_df, target_id=target_id, title=title) - except Exception as exc: - fig = dash.go.Figure() - fig.update_layout(title=f"Error: {exc}") - return fig + if tab == 'freq': + stats = calculate_frequency(df) + fig = plot_frequency(stats, title=f"{bus} Frequency") + return dcc.Graph(figure=fig, style={'height': '80vh'}) + + elif tab == 'id_viewer': + ids = sorted(df['ID'].unique().tolist()) + return html.Div([ + html.Label("Select CAN ID:"), + dcc.Dropdown( + id='id-selector', + options=[{'label': i, 'value': i} for i in ids], + value=ids[0] if ids else None, + clearable=False, + style={'width': '50%', 'marginBottom': '10px'} + ), + dcc.Graph(id='id-viewer-graph', style={'height': '70vh'}) + ]) + + elif tab == 'corr': + ids = sorted(df['ID'].unique().tolist()) + return html.Div([ + dbc.Row([ + dbc.Col(html.Label("Method:"), width=1, className="mt-2"), + dbc.Col(dcc.Dropdown( + id='corr-method', + options=[{'label': 'Pearson', 'value': 'pearson'}, {'label': 'Spearman', 'value': 'spearman'}], + value='pearson', + clearable=False + ), width=2), + dbc.Col(html.Label("Target ID:"), width=1, className="mt-2"), + dbc.Col(dcc.Dropdown( + id='corr-target', + options=[{'label': 'All IDs (Max Corr)', 'value': 'all'}] + [{'label': i, 'value': i} for i in ids], + value='all', + clearable=True + ), width=4), + ], className="mb-3"), + dcc.Graph(id='corr-graph', style={'height': '80vh'}) + ]) + + elif tab == 'entropy': + entropy_df = calculate_byte_entropy(df) + fig = plot_entropy_heatmap(entropy_df, title=f"{bus} Byte-Level Entropy") + return dcc.Graph(figure=fig, style={'height': '80vh'}) + + return html.Div("Tab not found") + +@app.callback( + Output('id-viewer-graph', 'figure'), + Input('id-selector', 'value'), + Input('bus-selector', 'value'), + Input('tabs', 'active_tab'), +) +def update_id_viewer(selected_id, bus, tab): + if tab != 'id_viewer' or not selected_id: + return dash.no_update + df = DATA[bus] + filtered_df, byte_cols = prepare_data(df, selected_id) + return plot_bits(filtered_df, byte_cols, selected_id, title=f"{bus} Byte Visualization") + +@app.callback( + Output('corr-graph', 'figure'), + Input('corr-method', 'value'), + Input('corr-target', 'value'), + Input('bus-selector', 'value'), + Input('tabs', 'active_tab'), +) +def update_corr(method, target, bus, tab): + if tab != 'corr': + return dash.no_update + df = DATA[bus] + target_id = None if target == 'all' or not target else target + corr_df = calculate_correlation(df, method=method, target_id=target_id) + title = f"{bus} Correlation" + if target_id: + title += f" ({target_id})" + return plot_correlation_heatmap(corr_df, target_id=target_id, title=title) + +if __name__ == '__main__': + app.run(debug=True) -- 2.52.0 From d274897cf353a7533ac3b462e720d329487d879a Mon Sep 17 00:00:00 2001 From: Erick Ahmed Date: Wed, 22 Jul 2026 18:07:33 +0200 Subject: [PATCH 10/25] Implement lttbc --- stats/id_viewer.py | 12 ++++++++++-- 1 file changed, 10 insertions(+), 2 deletions(-) diff --git a/stats/id_viewer.py b/stats/id_viewer.py index 59f00ac..4c9ed00 100644 --- a/stats/id_viewer.py +++ b/stats/id_viewer.py @@ -7,6 +7,7 @@ from pathlib import Path import numpy as np import pandas as pd import plotly.graph_objects as go +import lttbc from stats.utils.extractor import load_data def _format_can_id_vec(s: pd.Series) -> pd.Series: @@ -43,13 +44,20 @@ def plot_bits(df, byte_cols, can_id, title): fig = go.Figure() colors = ['#e41a1c', '#377eb8', '#4daf4a', '#984ea3', '#ff7f00', '#ffff33', '#a65628', '#f781bf'] n = len(byte_cols) + max_points = 2000 x = df['Timestamp'].to_numpy() if not df.empty else np.array([]) for i, col in enumerate(byte_cols): y = df[col].to_numpy(dtype=np.float32, copy=False) if not df.empty else np.array([]) + + if len(x) > max_points and len(x) == len(y): + x_plot, y_plot = lttbc.downsample(x, y, max_points) + else: + x_plot, y_plot = x, y + fig.add_trace(go.Scattergl( - x=x, - y=y, + x=x_plot, + y=y_plot, mode='lines', line=dict(shape='hv', width=2, color=colors[i % len(colors)]), name=col.upper(), -- 2.52.0 From 02e46ddf0b8286356d9af72e8f28bc63233a91c5 Mon Sep 17 00:00:00 2001 From: Erick Ahmed Date: Wed, 22 Jul 2026 18:11:35 +0200 Subject: [PATCH 11/25] Implement plotly-resampler --- stats/id_viewer.py | 16 ++++------------ 1 file changed, 4 insertions(+), 12 deletions(-) diff --git a/stats/id_viewer.py b/stats/id_viewer.py index 4c9ed00..104f162 100644 --- a/stats/id_viewer.py +++ b/stats/id_viewer.py @@ -7,7 +7,7 @@ from pathlib import Path import numpy as np import pandas as pd import plotly.graph_objects as go -import lttbc +from plotly_resampler import FigureResampler from stats.utils.extractor import load_data def _format_can_id_vec(s: pd.Series) -> pd.Series: @@ -41,29 +41,21 @@ def prepare_data(df, target_id): return filtered, byte_cols def plot_bits(df, byte_cols, can_id, title): - fig = go.Figure() + fig = FigureResampler() colors = ['#e41a1c', '#377eb8', '#4daf4a', '#984ea3', '#ff7f00', '#ffff33', '#a65628', '#f781bf'] n = len(byte_cols) - max_points = 2000 x = df['Timestamp'].to_numpy() if not df.empty else np.array([]) for i, col in enumerate(byte_cols): y = df[col].to_numpy(dtype=np.float32, copy=False) if not df.empty else np.array([]) - if len(x) > max_points and len(x) == len(y): - x_plot, y_plot = lttbc.downsample(x, y, max_points) - else: - x_plot, y_plot = x, y - - fig.add_trace(go.Scattergl( - x=x_plot, - y=y_plot, + fig.add_trace(go.Scatter( mode='lines', line=dict(shape='hv', width=2, color=colors[i % len(colors)]), name=col.upper(), legendgroup=col.upper(), hovertemplate=f"{col.upper()}
Time: %{{x}}
Value: %{{y}}", - )) + ), hf_x=x, hf_y=y) all_button = dict(label='ALL', method='restyle', args=[{'visible': [True] * n}]) none_button = dict(label='NONE', method='restyle', args=[{'visible': ['legendonly'] * n}]) -- 2.52.0 From 93e0e3f64828fa0bfc4200c686cb1ce4e0018722 Mon Sep 17 00:00:00 2001 From: Erick Ahmed Date: Wed, 22 Jul 2026 18:16:43 +0200 Subject: [PATCH 12/25] Suppress callback exceptions --- main.py | 1 + 1 file changed, 1 insertion(+) diff --git a/main.py b/main.py index 9d16086..d5c8d84 100644 --- a/main.py +++ b/main.py @@ -56,6 +56,7 @@ DATA = { } app = dash.Dash(__name__, external_stylesheets=[dbc.themes.BOOTSTRAP]) +app.config.suppress_callback_exceptions = True app.layout = dbc.Container([ html.H1("CAN Bus Analyzer", className="my-4"), -- 2.52.0 From 2da646fa8008c11106f55032b62d89734ac11de3 Mon Sep 17 00:00:00 2001 From: Erick Ahmed Date: Wed, 22 Jul 2026 18:20:18 +0200 Subject: [PATCH 13/25] Precompute CAN data - Slower startup - Much faster visualization (from O(n) to O(1)) --- main.py | 64 ++++++++++++++++++++++++++++++++++++++++++++------------- 1 file changed, 50 insertions(+), 14 deletions(-) diff --git a/main.py b/main.py index d5c8d84..ed45176 100644 --- a/main.py +++ b/main.py @@ -9,13 +9,13 @@ import polars as pl import dash from dash import dcc, html, Input, Output import dash_bootstrap_components as dbc +import numpy as np from parser import parse_log, parse_csv from decoder import decode_j1939_frames from stats.utils.extractor import load_data - +from stats.id_viewer import _format_can_id_vec, plot_bits from stats.frequency import calculate_frequency, plot_frequency -from stats.id_viewer import prepare_data, plot_bits from stats.correlation import calculate_correlation, plot_correlation_heatmap from stats.entropy import calculate_byte_entropy, plot_entropy_heatmap @@ -55,6 +55,33 @@ DATA = { "Bus 2": load_data(BUS2_DECODED) } +PRECOMPUTED_FIGURES = {} +DATA_BY_ID = {} +CORR_CACHE = {} + +for bus, df in DATA.items(): + PRECOMPUTED_FIGURES[f"{bus}_freq"] = plot_frequency(calculate_frequency(df), title=f"{bus} Frequency") + PRECOMPUTED_FIGURES[f"{bus}_entropy"] = plot_entropy_heatmap(calculate_byte_entropy(df), title=f"{bus} Byte-Level Entropy") + + can_id_col = 'ID' if 'ID' in df.columns else 'Identifier' + formatted = _format_can_id_vec(df[can_id_col]) + df = df.assign(Formatted_ID=formatted) + + df = df.sort_values(['Formatted_ID', 'Timestamp'], kind='stable') + + grouped = {} + for can_id, group in df.groupby('Formatted_ID'): + byte_cols = [f"b{i}" for i in range(8) if f"b{i}" in group.columns] + if not group.empty and len(byte_cols) > 0: + arr = group[byte_cols].to_numpy(dtype=np.float32, copy=False) + if len(arr) > 1: + changed = np.any(arr[1:] != arr[:-1], axis=1) + keep = np.concatenate(([True], changed)) + group = group.iloc[keep] + grouped[can_id] = (group, byte_cols) + + DATA_BY_ID[bus] = grouped + app = dash.Dash(__name__, external_stylesheets=[dbc.themes.BOOTSTRAP]) app.config.suppress_callback_exceptions = True @@ -87,12 +114,10 @@ def render_content(tab, bus): df = DATA[bus] if tab == 'freq': - stats = calculate_frequency(df) - fig = plot_frequency(stats, title=f"{bus} Frequency") - return dcc.Graph(figure=fig, style={'height': '80vh'}) + return dcc.Graph(figure=PRECOMPUTED_FIGURES[f"{bus}_freq"], style={'height': '80vh'}) elif tab == 'id_viewer': - ids = sorted(df['ID'].unique().tolist()) + ids = sorted(DATA_BY_ID[bus].keys()) return html.Div([ html.Label("Select CAN ID:"), dcc.Dropdown( @@ -106,7 +131,7 @@ def render_content(tab, bus): ]) elif tab == 'corr': - ids = sorted(df['ID'].unique().tolist()) + ids = sorted(DATA_BY_ID[bus].keys()) return html.Div([ dbc.Row([ dbc.Col(html.Label("Method:"), width=1, className="mt-2"), @@ -128,9 +153,7 @@ def render_content(tab, bus): ]) elif tab == 'entropy': - entropy_df = calculate_byte_entropy(df) - fig = plot_entropy_heatmap(entropy_df, title=f"{bus} Byte-Level Entropy") - return dcc.Graph(figure=fig, style={'height': '80vh'}) + return dcc.Graph(figure=PRECOMPUTED_FIGURES[f"{bus}_entropy"], style={'height': '80vh'}) return html.Div("Tab not found") @@ -143,8 +166,12 @@ def render_content(tab, bus): def update_id_viewer(selected_id, bus, tab): if tab != 'id_viewer' or not selected_id: return dash.no_update - df = DATA[bus] - filtered_df, byte_cols = prepare_data(df, selected_id) + + grouped_data = DATA_BY_ID.get(bus, {}) + if selected_id not in grouped_data: + return dash.no_update + + filtered_df, byte_cols = grouped_data[selected_id] return plot_bits(filtered_df, byte_cols, selected_id, title=f"{bus} Byte Visualization") @app.callback( @@ -157,12 +184,21 @@ def update_id_viewer(selected_id, bus, tab): def update_corr(method, target, bus, tab): if tab != 'corr': return dash.no_update - df = DATA[bus] + target_id = None if target == 'all' or not target else target - corr_df = calculate_correlation(df, method=method, target_id=target_id) + cache_key = (bus, method, target_id) + + if cache_key not in CORR_CACHE: + df = DATA[bus] + corr_df = calculate_correlation(df, method=method, target_id=target_id) + CORR_CACHE[cache_key] = corr_df + else: + corr_df = CORR_CACHE[cache_key] + title = f"{bus} Correlation" if target_id: title += f" ({target_id})" + return plot_correlation_heatmap(corr_df, target_id=target_id, title=title) if __name__ == '__main__': -- 2.52.0 From c06d813c26caf4ca289084c743fdc1677e2086f4 Mon Sep 17 00:00:00 2001 From: Erick Ahmed Date: Wed, 22 Jul 2026 18:21:32 +0200 Subject: [PATCH 14/25] Remove resampling information on legend --- stats/id_viewer.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/stats/id_viewer.py b/stats/id_viewer.py index 104f162..f3e30c8 100644 --- a/stats/id_viewer.py +++ b/stats/id_viewer.py @@ -41,7 +41,10 @@ def prepare_data(df, target_id): return filtered, byte_cols def plot_bits(df, byte_cols, can_id, title): - fig = FigureResampler() + fig = FigureResampler( + resampled_trace_prefix_suffix=("", ""), + show_mean_aggregation_size=False + ) colors = ['#e41a1c', '#377eb8', '#4daf4a', '#984ea3', '#ff7f00', '#ffff33', '#a65628', '#f781bf'] n = len(byte_cols) -- 2.52.0 From 9965bc761fe64b17e6c4d1e5ad894d5b553cfc9d Mon Sep 17 00:00:00 2001 From: Erick Ahmed Date: Wed, 22 Jul 2026 19:46:32 +0200 Subject: [PATCH 15/25] Simplify byte column selection in correlation calculation --- stats/correlation.py | 3 +-- stats/entropy.py | 3 +-- 2 files changed, 2 insertions(+), 4 deletions(-) diff --git a/stats/correlation.py b/stats/correlation.py index d8a9811..f6003d6 100644 --- a/stats/correlation.py +++ b/stats/correlation.py @@ -27,8 +27,7 @@ def _ensure_int_bytes(df: pd.DataFrame, cols: list) -> pd.DataFrame: return df def calculate_correlation(df: pd.DataFrame, method: str, target_id: str | None = None) -> pd.DataFrame: - byte_cols = [f"b{i}" for i in range(8) if f"b{i}" in df.columns] - available_cols = [col for col in byte_cols if col in df.columns] + available_cols = [f"b{i}" for i in range(8) if f"b{i}" in df.columns] if not available_cols: raise ValueError("No byte columns (b0-b7) found in the DataFrame") diff --git a/stats/entropy.py b/stats/entropy.py index 11d053b..df42d15 100644 --- a/stats/entropy.py +++ b/stats/entropy.py @@ -36,8 +36,7 @@ def _entropy_col(a: np.ndarray) -> float: return float(-np.sum(p * np.log2(p))) def calculate_byte_entropy(df: pd.DataFrame) -> pd.DataFrame: - byte_cols = [f"b{i}" for i in range(8) if f"b{i}" in df.columns] - available_cols = byte_cols + available_cols = [f"b{i}" for i in range(8) if f"b{i}" in df.columns] if not available_cols: raise ValueError("No byte columns (b0-b7) found in the DataFrame") -- 2.52.0 From d6baaaa1fa96119d02560d58620dea879d81208b Mon Sep 17 00:00:00 2001 From: Erick Ahmed Date: Wed, 22 Jul 2026 19:46:51 +0200 Subject: [PATCH 16/25] Remove redundant Formatted_ID column in frequency calculation --- stats/frequency.py | 1 - 1 file changed, 1 deletion(-) diff --git a/stats/frequency.py b/stats/frequency.py index 0db6b30..57b1f5f 100644 --- a/stats/frequency.py +++ b/stats/frequency.py @@ -18,7 +18,6 @@ def _format_can_id_vec(s: pd.Series) -> pd.Series: def calculate_frequency(df: pd.DataFrame) -> pd.DataFrame: can_id_col = 'ID' if 'ID' in df.columns else 'Identifier' formatted = _format_can_id_vec(df[can_id_col]) - df['Formatted_ID'] = formatted counts = formatted.value_counts() freq_df = pd.DataFrame({ -- 2.52.0 From 5e01c3bb44cb038967047db6568f3289f859ac0f Mon Sep 17 00:00:00 2001 From: Erick Ahmed Date: Wed, 22 Jul 2026 19:46:59 +0200 Subject: [PATCH 17/25] Ensure data directories exist before pipeline execution --- main.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/main.py b/main.py index ed45176..990e675 100644 --- a/main.py +++ b/main.py @@ -29,6 +29,8 @@ BUS2_DECODED = "data/parquet/bus2_decoded.parquet" def run_pipeline(): os.makedirs("data/logs", exist_ok=True) + os.makedirs("data/csv", exist_ok=True) + os.makedirs("data/parquet", exist_ok=True) if not Path(BUS1_DECODED).exists() or not Path(BUS2_DECODED).exists(): print("Parsing raw log...") parse_log(RAW_LOG, BUS1_CSV, BUS2_CSV) -- 2.52.0 From 22d4af292ccb3a99b6740a18dfe0cafcc54981be Mon Sep 17 00:00:00 2001 From: Erick Ahmed Date: Wed, 22 Jul 2026 19:47:05 +0200 Subject: [PATCH 18/25] Refactor CSV to Parquet conversion logic --- main.py | 6 ++---- 1 file changed, 2 insertions(+), 4 deletions(-) diff --git a/main.py b/main.py index 990e675..cc525d7 100644 --- a/main.py +++ b/main.py @@ -36,10 +36,8 @@ def run_pipeline(): parse_log(RAW_LOG, BUS1_CSV, BUS2_CSV) print("Converting to parquet...") - lf1 = parse_csv(BUS1_CSV) - lf1.sink_parquet(BUS1_PARQUET) - lf2 = parse_csv(BUS2_CSV) - lf2.sink_parquet(BUS2_PARQUET) + parse_csv(BUS1_CSV).sink_parquet(BUS1_PARQUET) + parse_csv(BUS2_CSV).sink_parquet(BUS2_PARQUET) print("Decoding J1939...") df1 = pl.read_parquet(BUS1_PARQUET) -- 2.52.0 From 24ce8dad60f5c1941c2b81ba30bd216846b0cb82 Mon Sep 17 00:00:00 2001 From: Erick Ahmed Date: Wed, 22 Jul 2026 19:47:11 +0200 Subject: [PATCH 19/25] Explicitly specify grouping column in dataframe iteration --- main.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/main.py b/main.py index cc525d7..3483b8c 100644 --- a/main.py +++ b/main.py @@ -70,7 +70,7 @@ for bus, df in DATA.items(): df = df.sort_values(['Formatted_ID', 'Timestamp'], kind='stable') grouped = {} - for can_id, group in df.groupby('Formatted_ID'): + for can_id, group in df.groupby(by='Formatted_ID'): byte_cols = [f"b{i}" for i in range(8) if f"b{i}" in group.columns] if not group.empty and len(byte_cols) > 0: arr = group[byte_cols].to_numpy(dtype=np.float32, copy=False) -- 2.52.0 From d9262e365a9d39544abdd0c2b43b57339fda790a Mon Sep 17 00:00:00 2001 From: Erick Ahmed Date: Wed, 22 Jul 2026 20:00:13 +0200 Subject: [PATCH 20/25] Put all CAN bus statistics submenus under a Statistics menu --- main.py | 16 ++++++++++------ 1 file changed, 10 insertions(+), 6 deletions(-) diff --git a/main.py b/main.py index 3483b8c..6e59539 100644 --- a/main.py +++ b/main.py @@ -97,12 +97,16 @@ app.layout = dbc.Container([ ), width=2), ], className="mb-3"), dbc.Tabs([ - dbc.Tab(label="Frequency", tab_id="freq"), - dbc.Tab(label="ID Viewer", tab_id="id_viewer"), - dbc.Tab(label="Correlation", tab_id="corr"), - dbc.Tab(label="Entropy", tab_id="entropy"), - ], id="tabs", active_tab="freq"), - html.Div(id="tab-content", className="mt-3") + dbc.Tab(label="Statistics", tab_id="statistics", children=[ + dbc.Tabs([ + dbc.Tab(label="Frequency", tab_id="freq"), + dbc.Tab(label="ID Viewer", tab_id="id_viewer"), + dbc.Tab(label="Correlation", tab_id="corr"), + dbc.Tab(label="Entropy", tab_id="entropy"), + ], id="tabs", active_tab="freq", className="mt-3"), + html.Div(id="tab-content", className="mt-3") + ]) + ], id="main-tabs", active_tab="statistics") ], fluid=True) @app.callback( -- 2.52.0 From af8e91611619f376090220c1fcbf49c82b94bf33 Mon Sep 17 00:00:00 2001 From: Erick Ahmed Date: Wed, 22 Jul 2026 20:00:59 +0200 Subject: [PATCH 21/25] Create an Overview menu - To use as a sort of main menu --- main.py | 23 +++++++++++++---------- 1 file changed, 13 insertions(+), 10 deletions(-) diff --git a/main.py b/main.py index 6e59539..8c9fe43 100644 --- a/main.py +++ b/main.py @@ -87,23 +87,26 @@ app.config.suppress_callback_exceptions = True app.layout = dbc.Container([ html.H1("CAN Bus Analyzer", className="my-4"), - dbc.Row([ - dbc.Col(html.Label("Select Bus:"), width=1, className="mt-2"), - dbc.Col(dcc.Dropdown( - id='bus-selector', - options=[{'label': k, 'value': k} for k in DATA.keys()], - value='Bus 1', - clearable=False - ), width=2), - ], className="mb-3"), dbc.Tabs([ + dbc.Tab(label="Overview", tab_id="overview", children=[ + html.Div(id="overview-content") + ]), dbc.Tab(label="Statistics", tab_id="statistics", children=[ + dbc.Row([ + dbc.Col(html.Label("Select Bus:"), width=1, className="mt-2"), + dbc.Col(dcc.Dropdown( + id='bus-selector', + options=[{'label': k, 'value': k} for k in DATA.keys()], + value='Bus 1', + clearable=False + ), width=2), + ], className="mb-3 mt-3"), dbc.Tabs([ dbc.Tab(label="Frequency", tab_id="freq"), dbc.Tab(label="ID Viewer", tab_id="id_viewer"), dbc.Tab(label="Correlation", tab_id="corr"), dbc.Tab(label="Entropy", tab_id="entropy"), - ], id="tabs", active_tab="freq", className="mt-3"), + ], id="tabs", active_tab="freq"), html.Div(id="tab-content", className="mt-3") ]) ], id="main-tabs", active_tab="statistics") -- 2.52.0 From 57505074cd366fd2023cf5c8852f52bf356f69ed Mon Sep 17 00:00:00 2001 From: Erick Ahmed Date: Wed, 22 Jul 2026 20:01:58 +0200 Subject: [PATCH 22/25] Change title to project name --- main.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/main.py b/main.py index 8c9fe43..96ae4b7 100644 --- a/main.py +++ b/main.py @@ -86,7 +86,7 @@ app = dash.Dash(__name__, external_stylesheets=[dbc.themes.BOOTSTRAP]) app.config.suppress_callback_exceptions = True app.layout = dbc.Container([ - html.H1("CAN Bus Analyzer", className="my-4"), + html.H1("CANveyor", className="my-4"), dbc.Tabs([ dbc.Tab(label="Overview", tab_id="overview", children=[ html.Div(id="overview-content") -- 2.52.0 From 6c198d83c54b481c92ef31859b2470530c66a227 Mon Sep 17 00:00:00 2001 From: Erick Ahmed Date: Wed, 22 Jul 2026 20:06:50 +0200 Subject: [PATCH 23/25] Remove debug flag --- main.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/main.py b/main.py index 96ae4b7..70bf667 100644 --- a/main.py +++ b/main.py @@ -209,4 +209,4 @@ def update_corr(method, target, bus, tab): return plot_correlation_heatmap(corr_df, target_id=target_id, title=title) if __name__ == '__main__': - app.run(debug=True) + app.run(debug=False) -- 2.52.0 From 1463fa12ff838d05cbd28645cc6df44358f3602a Mon Sep 17 00:00:00 2001 From: Erick Ahmed Date: Wed, 22 Jul 2026 20:09:08 +0200 Subject: [PATCH 24/25] Parallelize data processing tasks with ThreadPoolExecutor --- main.py | 20 +++++++++++++++----- stats/correlation.py | 10 +++++++--- stats/entropy.py | 11 ++++++++--- 3 files changed, 30 insertions(+), 11 deletions(-) diff --git a/main.py b/main.py index 70bf667..68cd788 100644 --- a/main.py +++ b/main.py @@ -4,6 +4,7 @@ import os from pathlib import Path +from concurrent.futures import ThreadPoolExecutor import polars as pl import dash @@ -59,9 +60,10 @@ PRECOMPUTED_FIGURES = {} DATA_BY_ID = {} CORR_CACHE = {} -for bus, df in DATA.items(): - PRECOMPUTED_FIGURES[f"{bus}_freq"] = plot_frequency(calculate_frequency(df), title=f"{bus} Frequency") - PRECOMPUTED_FIGURES[f"{bus}_entropy"] = plot_entropy_heatmap(calculate_byte_entropy(df), title=f"{bus} Byte-Level Entropy") +def process_bus_data(bus, df): + precomp = {} + precomp[f"{bus}_freq"] = plot_frequency(calculate_frequency(df), title=f"{bus} Frequency") + precomp[f"{bus}_entropy"] = plot_entropy_heatmap(calculate_byte_entropy(df), title=f"{bus} Byte-Level Entropy") can_id_col = 'ID' if 'ID' in df.columns else 'Identifier' formatted = _format_can_id_vec(df[can_id_col]) @@ -80,7 +82,15 @@ for bus, df in DATA.items(): group = group.iloc[keep] grouped[can_id] = (group, byte_cols) - DATA_BY_ID[bus] = grouped + return precomp, grouped + +with ThreadPoolExecutor() as executor: + futures = {executor.submit(process_bus_data, bus, df): bus for bus, df in DATA.items()} + for future in futures: + bus = futures[future] + precomp, grouped = future.result() + PRECOMPUTED_FIGURES.update(precomp) + DATA_BY_ID[bus] = grouped app = dash.Dash(__name__, external_stylesheets=[dbc.themes.BOOTSTRAP]) app.config.suppress_callback_exceptions = True @@ -209,4 +219,4 @@ def update_corr(method, target, bus, tab): return plot_correlation_heatmap(corr_df, target_id=target_id, title=title) if __name__ == '__main__': - app.run(debug=False) + app.run(debug=True) diff --git a/stats/correlation.py b/stats/correlation.py index f6003d6..3035f39 100644 --- a/stats/correlation.py +++ b/stats/correlation.py @@ -4,6 +4,7 @@ import argparse from pathlib import Path +from concurrent.futures import ThreadPoolExecutor import numpy as np import pandas as pd @@ -69,8 +70,7 @@ def calculate_correlation(df: pd.DataFrame, method: str, target_id: str | None = else: groups = [] - out = np.zeros((len(unique_ids), n_cols), dtype=np.float64) - for gi, sub in enumerate(groups): + def _process_group(sub): mask = ~np.isnan(sub).any(axis=1) sub = sub[mask] if len(sub) > 1: @@ -80,7 +80,11 @@ def calculate_correlation(df: pd.DataFrame, method: str, target_id: str | None = c = np.abs(np.corrcoef(sub, rowvar=False)) np.nan_to_num(c, copy=False, nan=0.0) np.fill_diagonal(c, 0.0) - out[gi] = c.max(axis=0) + return c.max(axis=0) + return np.zeros(n_cols, dtype=np.float64) + + with ThreadPoolExecutor() as executor: + out = np.array(list(executor.map(_process_group, groups))) result = pd.DataFrame(out, index=unique_ids, columns=available_cols) result.index.name = 'Identifier' diff --git a/stats/entropy.py b/stats/entropy.py index df42d15..ac3295e 100644 --- a/stats/entropy.py +++ b/stats/entropy.py @@ -4,6 +4,7 @@ import argparse from pathlib import Path +from concurrent.futures import ThreadPoolExecutor import numpy as np import pandas as pd @@ -63,10 +64,14 @@ def calculate_byte_entropy(df: pd.DataFrame) -> pd.DataFrame: else: groups = [] - out = np.zeros((len(unique_ids), n_cols), dtype=np.float64) - for gi, sub in enumerate(groups): + def _process_group(sub): + res = np.zeros(n_cols, dtype=np.float64) for ci in range(n_cols): - out[gi, ci] = _entropy_col(sub[:, ci]) + res[ci] = _entropy_col(sub[:, ci]) + return res + + with ThreadPoolExecutor() as executor: + out = np.array(list(executor.map(_process_group, groups))) result = pd.DataFrame(out, index=unique_ids, columns=available_cols) result.index.name = 'Identifier' -- 2.52.0 From d8ca263c0da241c15dcf1d00fdfd02b32cfb70e4 Mon Sep 17 00:00:00 2001 From: Erick Ahmed Date: Wed, 22 Jul 2026 20:16:52 +0200 Subject: [PATCH 25/25] Bump version and update project dependencies --- pyproject.toml | 14 +++++++++++--- 1 file changed, 11 insertions(+), 3 deletions(-) diff --git a/pyproject.toml b/pyproject.toml index b14b118..2b78937 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,7 +1,15 @@ [project] name = "CANveyor" -version = "0.0.4" +version = "0.1.0" description = "J1939 CAN bus parser that works in pair with CANdigger" readme = "README.md" -requires-python = ">=3.14" -dependencies = ["polars", "pathlib", "typing"] +requires-python = ">=3.10" +dependencies = [ + "polars", + "dash", + "dash-bootstrap-components", + "numpy", + "pandas", + "plotly", + "plotly-resampler" +] -- 2.52.0