diff --git a/stat/correlation.py b/stat/correlation.py new file mode 100644 index 0000000..dc81132 --- /dev/null +++ b/stat/correlation.py @@ -0,0 +1,144 @@ +# File: correlation.py +# Copyright (C) 2026 Erick Ahmed +# SPDX-License-Identifier: AGPL-3.0-or-later + +import argparse +from pathlib import Path + +import numpy as np +import pandas as pd +import plotly.graph_objects as go + +from utils.extractor import load_data + + +def _to_int(x): + """Convert a hex string or integer to int, returning NaN on failure.""" + if isinstance(x, (int, np.integer)): + return int(x) + if isinstance(x, str): + try: + return int(x, 16) + except ValueError: + return np.nan + return np.nan + + +def calculate_correlation(df: pd.DataFrame, method: str, target_id: str | None = None) -> pd.DataFrame: + """Calculates inter-byte correlation grouped by identifier.""" + byte_cols = [f"b{i}" for i in range(8)] + available_cols = [col for col in byte_cols if col in df.columns] + + if not available_cols: + raise ValueError("No byte columns (b0-b7) found in the DataFrame") + + df_bytes = df[["Identifier"] + available_cols].copy() + for col in available_cols: + df_bytes[col] = df_bytes[col].apply(_to_int) + + if target_id: + group = df_bytes[df_bytes["Identifier"] == target_id] + if group.empty: + raise ValueError(f"Identifier '{target_id}' not found in data") + return group[available_cols].corr(method=method).fillna(0.0) + + def max_abs_corr(group: pd.DataFrame) -> pd.Series: + corr_arr = np.abs(group.corr(method=method).to_numpy().copy()) + np.fill_diagonal(corr_arr, 0.0) + return pd.Series(corr_arr.max(axis=0), index=group.columns).fillna(0.0) + + return df_bytes.groupby("Identifier")[available_cols].apply(max_abs_corr) + + +def plot_correlation_heatmap(corr_df: pd.DataFrame, target_id: str | None, title: str) -> go.Figure: + """Generates an interactive heatmap of inter-byte correlation.""" + is_8x8 = target_id is not None + + if is_8x8: + x = corr_df.columns.tolist() + y = corr_df.index.tolist() + z = corr_df.values + z_min, z_max = -1.0, 1.0 + colorscale = [ + [0.0, "#2c7bb6"], [0.25, "#abd9e9"], [0.5, "#ffffff"], + [0.75, "#fdae61"], [1.0, "#d7191c"] + ] + hover_template = "%{y} vs %{x}
Correlation: %{z:.2f}" + else: + x = corr_df.columns.tolist() + y = corr_df.index.tolist() + z = corr_df.values + z_min, z_max = 0.0, 1.0 + colorscale = [ + [0.0, "#ffffff"], [0.2, "#fff5f0"], [0.4, "#fecc5c"], + [0.6, "#fd8d3c"], [0.8, "#e31a1c"], [1.0, "#800026"] + ] + hover_template = "%{y}
Byte %{x} max correlation: %{z:.2f}" + + fig = go.Figure( + data=go.Heatmap( + z=z, x=x, y=y, + zmin=z_min, zmax=z_max, + colorscale=colorscale, + xgap=3, ygap=3, + text=np.round(z, 2), + texttemplate="%{text}", + textfont={"size": 11, "color": "#2a2a2a", "family": "Segoe UI, Arial, sans-serif"}, + hoverongaps=False, + hovertemplate=hover_template, + colorbar=dict( + title=dict(text="Correlation", side="top", font=dict(size=13, color="#1a1a1a")), + orientation="h", thickness=15, len=0.35, + x=1.0, xanchor="right", y=1.02, yanchor="bottom", + tickfont=dict(size=11, color="#2a2a2a"), + tickformat=".1f", outlinewidth=0.5, outlinecolor="#cccccc", + ), + ) + ) + + fig.update_layout( + title=dict(text=title, font=dict(size=20, color="#1a1a1a"), x=0.5, xanchor="center", pad=dict(b=20)), + height=max(600, len(y) * 28 + 150) if not is_8x8 else 600, + autosize=True, + template="plotly_white", + xaxis=dict( + title=dict(text="Byte Position", font=dict(size=13, color="#1a1a1a")), + side="top" if not is_8x8 else "bottom", + dtick=1, showgrid=False, linecolor="#bdbdbd", + tickfont=dict(size=12, color="#2a2a2a"), ticks="outside", ticklen=4, tickcolor="#cccccc", + ), + yaxis=dict( + title=dict(text="PGN or CAN ID" if not is_8x8 else "Byte Position", font=dict(size=13, color="#1a1a1a")), + autorange="reversed", showgrid=False, linecolor="#bdbdbd", + tickfont=dict(size=12, color="#2a2a2a"), ticks="outside", ticklen=4, tickcolor="#cccccc", automargin=True, + ), + font=dict(family="Segoe UI, Arial, sans-serif", size=12, color="#2a2a2a"), + hoverlabel=dict(bgcolor="white", font_size=13, font_family="Segoe UI", bordercolor="#cccccc"), + margin=dict(l=200, r=40, t=120, b=60), + ) + return fig + + +if __name__ == "__main__": + parser = argparse.ArgumentParser(description="Analyze CAN bus inter-byte correlation") + parser.add_argument("method", choices=["pearson", "spearman"], help="Correlation method to use") + parser.add_argument("input", type=Path, help="Path to the input CAN log file") + parser.add_argument("output", type=Path, nargs="?", default=Path("correlation_report.html"), help="Path to the output HTML report") + parser.add_argument("title", nargs="?", default="CAN Bus Inter-Byte Correlation", help="Title for the HTML report") + parser.add_argument("--identifier", type=str, default=None, help="Specific PGN/CAN ID to analyze (e.g., 'PGN: 65331'). If omitted, shows max correlation per byte for all IDs.") + args = parser.parse_args() + + df = load_data(args.input) + corr_df = calculate_correlation(df, method=args.method, target_id=args.identifier) + + display_title = f"{args.title} ({args.identifier})" if args.identifier else args.title + fig = plot_correlation_heatmap(corr_df, target_id=args.identifier, title=display_title) + + config = { + "responsive": True, + "displaylogo": False, + "scrollZoom": True, + "modeBarButtonsToAdd": ["toggleSpikelines"], + "toImageButtonOptions": {"format": "png", "scale": 2}, + } + fig.write_html(str(args.output), include_plotlyjs="cdn", config=config) diff --git a/stat/utils/extractor.py b/stat/utils/extractor.py index 8cdd07e..807835b 100644 --- a/stat/utils/extractor.py +++ b/stat/utils/extractor.py @@ -1,11 +1,20 @@ -# File: extractor.py -# Copyright (C) 2026 Erick Ahmed -# SPDX-License-Identifier: AGPL-3.0-or-later - import json from pathlib import Path + +import numpy as np import pandas as pd +def to_int(x): + """Convert a hex string or integer to int, returning NaN on failure.""" + if isinstance(x, (int, np.integer)): + return int(x) + if isinstance(x, str): + try: + return int(x, 16) + except ValueError: + return np.nan + return np.nan + def extract_id(row: pd.Series) -> str: """Extracts PGN from metadata or falls back to CAN ID.""" meta = row.get('j1939_metadata')