Refactor statistical analysis modules for performance
- Optimize data processing pipelines across files by replacing iterative pandas operations with vectorized NumPy routines
This commit is contained in:
+69
-50
@@ -12,75 +12,96 @@ import plotly.graph_objects as go
|
||||
from utils.extractor import load_data
|
||||
from utils.extractor import to_int
|
||||
|
||||
def _format_can_id(x):
|
||||
"""Safely cleans CAN ID strings without altering their length or value."""
|
||||
if pd.isna(x):
|
||||
return "UNKNOWN"
|
||||
def _format_can_id_vec(s: pd.Series) -> pd.Series:
|
||||
s = s.astype('string').str.strip()
|
||||
s = s.str.replace(r'^0x', '', case=False, regex=True)
|
||||
s = s.str.upper()
|
||||
return s.fillna('UNKNOWN').replace('', 'UNKNOWN')
|
||||
|
||||
s = str(x).strip()
|
||||
if not s:
|
||||
return "UNKNOWN"
|
||||
|
||||
if s.lower().startswith('0x'):
|
||||
s = s[2:]
|
||||
|
||||
return s.upper()
|
||||
def _ensure_int_bytes(df: pd.DataFrame, cols: list) -> pd.DataFrame:
|
||||
needs = [c for c in cols if not pd.api.types.is_numeric_dtype(df[c])]
|
||||
if needs:
|
||||
df = df.copy()
|
||||
for c in needs:
|
||||
df[c] = df[c].apply(to_int)
|
||||
return df
|
||||
|
||||
def calculate_correlation(df: pd.DataFrame, method: str, target_id: str | None = None) -> pd.DataFrame:
|
||||
"""Calculates inter-byte correlation grouped by identifier."""
|
||||
byte_cols = [f"b{i}" for i in range(8)]
|
||||
byte_cols = [f"b{i}" for i in range(8) if f"b{i}" in df.columns]
|
||||
available_cols = [col for col in byte_cols if col in df.columns]
|
||||
|
||||
if not available_cols:
|
||||
raise ValueError("No byte columns (b0-b7) found in the DataFrame")
|
||||
|
||||
can_id_col = 'ID' if 'ID' in df.columns else 'Identifier'
|
||||
identifiers = df[can_id_col].apply(_format_can_id)
|
||||
identifiers = _format_can_id_vec(df[can_id_col]).to_numpy()
|
||||
|
||||
df_bytes = df[available_cols].copy()
|
||||
for col in available_cols:
|
||||
df_bytes[col] = df_bytes[col].apply(to_int)
|
||||
df_bytes = _ensure_int_bytes(df, available_cols)[available_cols]
|
||||
data = df_bytes.to_numpy(dtype=np.float64, copy=False)
|
||||
|
||||
if target_id is not None:
|
||||
target_id = _format_can_id(target_id)
|
||||
group = df_bytes[identifiers == target_id]
|
||||
if group.empty:
|
||||
target_id = _format_can_id_vec(pd.Series([target_id])).iloc[0]
|
||||
mask = identifiers == target_id
|
||||
if not mask.any():
|
||||
raise ValueError(f"Identifier '{target_id}' not found in data")
|
||||
return group[available_cols].corr(method=method).fillna(0.0)
|
||||
sub = data[mask]
|
||||
mask = ~np.isnan(sub).any(axis=1)
|
||||
sub = sub[mask]
|
||||
if method == 'spearman' and sub.shape[0] > 1:
|
||||
sub = pd.DataFrame(sub).rank().to_numpy()
|
||||
if sub.shape[0] > 1:
|
||||
with np.errstate(divide='ignore', invalid='ignore'):
|
||||
c = np.corrcoef(sub, rowvar=False)
|
||||
np.nan_to_num(c, copy=False, nan=0.0)
|
||||
else:
|
||||
c = np.zeros((len(available_cols), len(available_cols)))
|
||||
return pd.DataFrame(c, index=available_cols, columns=available_cols)
|
||||
|
||||
def max_abs_corr(group: pd.DataFrame) -> pd.Series:
|
||||
corr_arr = np.abs(group.corr(method=method).to_numpy().copy())
|
||||
np.fill_diagonal(corr_arr, 0.0)
|
||||
return pd.Series(corr_arr.max(axis=0), index=group.columns).fillna(0.0)
|
||||
unique_ids, inverse = np.unique(identifiers, return_inverse=True)
|
||||
n_cols = len(available_cols)
|
||||
|
||||
result = df_bytes.groupby(identifiers)[available_cols].apply(max_abs_corr)
|
||||
sort_idx = np.argsort(inverse, kind='stable')
|
||||
data_sorted = data[sort_idx]
|
||||
inverse_sorted = inverse[sort_idx]
|
||||
|
||||
if len(inverse_sorted) > 0:
|
||||
split_points = np.flatnonzero(np.diff(inverse_sorted)) + 1
|
||||
groups = np.split(data_sorted, split_points)
|
||||
else:
|
||||
groups = []
|
||||
|
||||
out = np.zeros((len(unique_ids), n_cols), dtype=np.float64)
|
||||
for gi, sub in enumerate(groups):
|
||||
mask = ~np.isnan(sub).any(axis=1)
|
||||
sub = sub[mask]
|
||||
if len(sub) > 1:
|
||||
if method == 'spearman':
|
||||
sub = pd.DataFrame(sub).rank().to_numpy()
|
||||
with np.errstate(divide='ignore', invalid='ignore'):
|
||||
c = np.abs(np.corrcoef(sub, rowvar=False))
|
||||
np.nan_to_num(c, copy=False, nan=0.0)
|
||||
np.fill_diagonal(c, 0.0)
|
||||
out[gi] = c.max(axis=0)
|
||||
|
||||
result = pd.DataFrame(out, index=unique_ids, columns=available_cols)
|
||||
result.index.name = 'Identifier'
|
||||
return result
|
||||
|
||||
|
||||
def plot_correlation_heatmap(corr_df: pd.DataFrame, target_id: str | None, title: str) -> go.Figure:
|
||||
"""Generates an interactive heatmap of inter-byte correlation."""
|
||||
is_8x8 = target_id is not None
|
||||
|
||||
x = corr_df.columns.tolist()
|
||||
y = corr_df.index.tolist()
|
||||
z = corr_df.values
|
||||
|
||||
if is_8x8:
|
||||
x = corr_df.columns.tolist()
|
||||
y = corr_df.index.tolist()
|
||||
z = corr_df.values
|
||||
z_min, z_max = -1.0, 1.0
|
||||
colorscale = [
|
||||
[0.0, "#2c7bb6"], [0.25, "#abd9e9"], [0.5, "#ffffff"],
|
||||
[0.75, "#fdae61"], [1.0, "#d7191c"]
|
||||
]
|
||||
colorscale = [[0.0, "#2c7bb6"], [0.25, "#abd9e9"], [0.5, "#ffffff"], [0.75, "#fdae61"], [1.0, "#d7191c"]]
|
||||
hover_template = "<b>%{y}</b> vs <b>%{x}</b><br>Correlation: %{z:.2f}<extra></extra>"
|
||||
else:
|
||||
x = corr_df.columns.tolist()
|
||||
y = corr_df.index.tolist()
|
||||
z = corr_df.values
|
||||
z_min, z_max = 0.0, 1.0
|
||||
colorscale = [
|
||||
[0.0, "#ffffff"], [0.2, "#fff5f0"], [0.4, "#fecc5c"],
|
||||
[0.6, "#fd8d3c"], [0.8, "#e31a1c"], [1.0, "#800026"]
|
||||
]
|
||||
colorscale = [[0.0, "#ffffff"], [0.2, "#fff5f0"], [0.4, "#fecc5c"], [0.6, "#fd8d3c"], [0.8, "#e31a1c"], [1.0, "#800026"]]
|
||||
hover_template = "<b>%{y}</b><br>Byte %{x} max correlation: %{z:.2f}<extra></extra>"
|
||||
|
||||
fig = go.Figure(
|
||||
@@ -106,17 +127,17 @@ def plot_correlation_heatmap(corr_df: pd.DataFrame, target_id: str | None, title
|
||||
|
||||
fig.update_layout(
|
||||
title=dict(text=title, font=dict(size=20, color="#1a1a1a"), x=0.5, xanchor="center", pad=dict(b=20)),
|
||||
height=max(600, len(y) * 28 + 150) if not is_8x8 else 600,
|
||||
height=600 if is_8x8 else max(600, len(y) * 28 + 150),
|
||||
autosize=True,
|
||||
template="plotly_white",
|
||||
xaxis=dict(
|
||||
title=dict(text="Byte Position", font=dict(size=13, color="#1a1a1a")),
|
||||
side="top" if not is_8x8 else "bottom",
|
||||
side="bottom" if is_8x8 else "top",
|
||||
dtick=1, showgrid=False, linecolor="#bdbdbd",
|
||||
tickfont=dict(size=12, color="#2a2a2a"), ticks="outside", ticklen=4, tickcolor="#cccccc",
|
||||
),
|
||||
yaxis=dict(
|
||||
title=dict(text="PGN or CAN ID" if not is_8x8 else "Byte Position", font=dict(size=13, color="#1a1a1a")),
|
||||
title=dict(text="Byte Position" if is_8x8 else "PGN or CAN ID", font=dict(size=13, color="#1a1a1a")),
|
||||
autorange="reversed", showgrid=False, linecolor="#bdbdbd",
|
||||
tickfont=dict(size=12, color="#2a2a2a"), ticks="outside", ticklen=4, tickcolor="#cccccc", automargin=True,
|
||||
),
|
||||
@@ -131,17 +152,15 @@ if __name__ == "__main__":
|
||||
parser = argparse.ArgumentParser(description="Analyze CAN bus inter-byte correlation")
|
||||
parser.add_argument("method", choices=["pearson", "spearman"], help="Correlation method to use")
|
||||
parser.add_argument("input", type=Path, help="Path to the input CAN log file")
|
||||
parser.add_argument("output", type=Path, nargs="?", default=Path("correlation_report.html"), help="Path to the output HTML report")
|
||||
parser.add_argument("title", nargs="?", default="CAN Bus Inter-Byte Correlation", help="Title for the HTML report")
|
||||
parser.add_argument("--identifier", type=str, default=None, help="Specific PGN/CAN ID to analyze (e.g., 'PGN: 65331'). If omitted, shows max correlation per byte for all IDs.")
|
||||
parser.add_argument("output", type=Path, nargs="?", default=Path("correlation_report.html"))
|
||||
parser.add_argument("title", nargs="?", default="CAN Bus Inter-Byte Correlation")
|
||||
parser.add_argument("--identifier", type=str, default=None)
|
||||
args = parser.parse_args()
|
||||
|
||||
df = load_data(args.input)
|
||||
corr_df = calculate_correlation(df, method=args.method, target_id=args.identifier)
|
||||
|
||||
display_title = f"{args.title} ({args.identifier})" if args.identifier else args.title
|
||||
fig = plot_correlation_heatmap(corr_df, target_id=args.identifier, title=display_title)
|
||||
|
||||
config = {
|
||||
"responsive": True,
|
||||
"displaylogo": False,
|
||||
|
||||
Reference in New Issue
Block a user