diff --git a/main.py b/main.py
index 3735861..68cd788 100644
--- a/main.py
+++ b/main.py
@@ -1,3 +1,222 @@
# File: main.py
# Copyright (C) 2026 Erick Ahmed
# SPDX-License-Identifier: AGPL-3.0-or-later
+
+import os
+from pathlib import Path
+from concurrent.futures import ThreadPoolExecutor
+
+import polars as pl
+import dash
+from dash import dcc, html, Input, Output
+import dash_bootstrap_components as dbc
+import numpy as np
+
+from parser import parse_log, parse_csv
+from decoder import decode_j1939_frames
+from stats.utils.extractor import load_data
+from stats.id_viewer import _format_can_id_vec, plot_bits
+from stats.frequency import calculate_frequency, plot_frequency
+from stats.correlation import calculate_correlation, plot_correlation_heatmap
+from stats.entropy import calculate_byte_entropy, plot_entropy_heatmap
+
+RAW_LOG = "data/logs/rawlog.txt"
+BUS1_CSV = "data/csv/bus1.csv"
+BUS2_CSV = "data/csv/bus2.csv"
+BUS1_PARQUET = "data/parquet/bus1.parquet"
+BUS2_PARQUET = "data/parquet/bus2.parquet"
+BUS1_DECODED = "data/parquet/bus1_decoded.parquet"
+BUS2_DECODED = "data/parquet/bus2_decoded.parquet"
+
+def run_pipeline():
+ os.makedirs("data/logs", exist_ok=True)
+ os.makedirs("data/csv", exist_ok=True)
+ os.makedirs("data/parquet", exist_ok=True)
+ if not Path(BUS1_DECODED).exists() or not Path(BUS2_DECODED).exists():
+ print("Parsing raw log...")
+ parse_log(RAW_LOG, BUS1_CSV, BUS2_CSV)
+
+ print("Converting to parquet...")
+ parse_csv(BUS1_CSV).sink_parquet(BUS1_PARQUET)
+ parse_csv(BUS2_CSV).sink_parquet(BUS2_PARQUET)
+
+ print("Decoding J1939...")
+ df1 = pl.read_parquet(BUS1_PARQUET)
+ df2 = pl.read_parquet(BUS2_PARQUET)
+ dec1 = decode_j1939_frames(df1)
+ dec2 = decode_j1939_frames(df2)
+ dec1.write_parquet(BUS1_DECODED)
+ dec2.write_parquet(BUS2_DECODED)
+
+run_pipeline()
+
+print("Loading data into memory...")
+DATA = {
+ "Bus 1": load_data(BUS1_DECODED),
+ "Bus 2": load_data(BUS2_DECODED)
+}
+
+PRECOMPUTED_FIGURES = {}
+DATA_BY_ID = {}
+CORR_CACHE = {}
+
+def process_bus_data(bus, df):
+ precomp = {}
+ precomp[f"{bus}_freq"] = plot_frequency(calculate_frequency(df), title=f"{bus} Frequency")
+ precomp[f"{bus}_entropy"] = plot_entropy_heatmap(calculate_byte_entropy(df), title=f"{bus} Byte-Level Entropy")
+
+ can_id_col = 'ID' if 'ID' in df.columns else 'Identifier'
+ formatted = _format_can_id_vec(df[can_id_col])
+ df = df.assign(Formatted_ID=formatted)
+
+ df = df.sort_values(['Formatted_ID', 'Timestamp'], kind='stable')
+
+ grouped = {}
+ for can_id, group in df.groupby(by='Formatted_ID'):
+ byte_cols = [f"b{i}" for i in range(8) if f"b{i}" in group.columns]
+ if not group.empty and len(byte_cols) > 0:
+ arr = group[byte_cols].to_numpy(dtype=np.float32, copy=False)
+ if len(arr) > 1:
+ changed = np.any(arr[1:] != arr[:-1], axis=1)
+ keep = np.concatenate(([True], changed))
+ group = group.iloc[keep]
+ grouped[can_id] = (group, byte_cols)
+
+ return precomp, grouped
+
+with ThreadPoolExecutor() as executor:
+ futures = {executor.submit(process_bus_data, bus, df): bus for bus, df in DATA.items()}
+ for future in futures:
+ bus = futures[future]
+ precomp, grouped = future.result()
+ PRECOMPUTED_FIGURES.update(precomp)
+ DATA_BY_ID[bus] = grouped
+
+app = dash.Dash(__name__, external_stylesheets=[dbc.themes.BOOTSTRAP])
+app.config.suppress_callback_exceptions = True
+
+app.layout = dbc.Container([
+ html.H1("CANveyor", className="my-4"),
+ dbc.Tabs([
+ dbc.Tab(label="Overview", tab_id="overview", children=[
+ html.Div(id="overview-content")
+ ]),
+ dbc.Tab(label="Statistics", tab_id="statistics", children=[
+ dbc.Row([
+ dbc.Col(html.Label("Select Bus:"), width=1, className="mt-2"),
+ dbc.Col(dcc.Dropdown(
+ id='bus-selector',
+ options=[{'label': k, 'value': k} for k in DATA.keys()],
+ value='Bus 1',
+ clearable=False
+ ), width=2),
+ ], className="mb-3 mt-3"),
+ dbc.Tabs([
+ dbc.Tab(label="Frequency", tab_id="freq"),
+ dbc.Tab(label="ID Viewer", tab_id="id_viewer"),
+ dbc.Tab(label="Correlation", tab_id="corr"),
+ dbc.Tab(label="Entropy", tab_id="entropy"),
+ ], id="tabs", active_tab="freq"),
+ html.Div(id="tab-content", className="mt-3")
+ ])
+ ], id="main-tabs", active_tab="statistics")
+], fluid=True)
+
+@app.callback(
+ Output('tab-content', 'children'),
+ Input('tabs', 'active_tab'),
+ Input('bus-selector', 'value')
+)
+def render_content(tab, bus):
+ df = DATA[bus]
+
+ if tab == 'freq':
+ return dcc.Graph(figure=PRECOMPUTED_FIGURES[f"{bus}_freq"], style={'height': '80vh'})
+
+ elif tab == 'id_viewer':
+ ids = sorted(DATA_BY_ID[bus].keys())
+ return html.Div([
+ html.Label("Select CAN ID:"),
+ dcc.Dropdown(
+ id='id-selector',
+ options=[{'label': i, 'value': i} for i in ids],
+ value=ids[0] if ids else None,
+ clearable=False,
+ style={'width': '50%', 'marginBottom': '10px'}
+ ),
+ dcc.Graph(id='id-viewer-graph', style={'height': '70vh'})
+ ])
+
+ elif tab == 'corr':
+ ids = sorted(DATA_BY_ID[bus].keys())
+ return html.Div([
+ dbc.Row([
+ dbc.Col(html.Label("Method:"), width=1, className="mt-2"),
+ dbc.Col(dcc.Dropdown(
+ id='corr-method',
+ options=[{'label': 'Pearson', 'value': 'pearson'}, {'label': 'Spearman', 'value': 'spearman'}],
+ value='pearson',
+ clearable=False
+ ), width=2),
+ dbc.Col(html.Label("Target ID:"), width=1, className="mt-2"),
+ dbc.Col(dcc.Dropdown(
+ id='corr-target',
+ options=[{'label': 'All IDs (Max Corr)', 'value': 'all'}] + [{'label': i, 'value': i} for i in ids],
+ value='all',
+ clearable=True
+ ), width=4),
+ ], className="mb-3"),
+ dcc.Graph(id='corr-graph', style={'height': '80vh'})
+ ])
+
+ elif tab == 'entropy':
+ return dcc.Graph(figure=PRECOMPUTED_FIGURES[f"{bus}_entropy"], style={'height': '80vh'})
+
+ return html.Div("Tab not found")
+
+@app.callback(
+ Output('id-viewer-graph', 'figure'),
+ Input('id-selector', 'value'),
+ Input('bus-selector', 'value'),
+ Input('tabs', 'active_tab'),
+)
+def update_id_viewer(selected_id, bus, tab):
+ if tab != 'id_viewer' or not selected_id:
+ return dash.no_update
+
+ grouped_data = DATA_BY_ID.get(bus, {})
+ if selected_id not in grouped_data:
+ return dash.no_update
+
+ filtered_df, byte_cols = grouped_data[selected_id]
+ return plot_bits(filtered_df, byte_cols, selected_id, title=f"{bus} Byte Visualization")
+
+@app.callback(
+ Output('corr-graph', 'figure'),
+ Input('corr-method', 'value'),
+ Input('corr-target', 'value'),
+ Input('bus-selector', 'value'),
+ Input('tabs', 'active_tab'),
+)
+def update_corr(method, target, bus, tab):
+ if tab != 'corr':
+ return dash.no_update
+
+ target_id = None if target == 'all' or not target else target
+ cache_key = (bus, method, target_id)
+
+ if cache_key not in CORR_CACHE:
+ df = DATA[bus]
+ corr_df = calculate_correlation(df, method=method, target_id=target_id)
+ CORR_CACHE[cache_key] = corr_df
+ else:
+ corr_df = CORR_CACHE[cache_key]
+
+ title = f"{bus} Correlation"
+ if target_id:
+ title += f" ({target_id})"
+
+ return plot_correlation_heatmap(corr_df, target_id=target_id, title=title)
+
+if __name__ == '__main__':
+ app.run(debug=True)
diff --git a/parser.py b/parser.py
index ad7fc32..1d84f6e 100644
--- a/parser.py
+++ b/parser.py
@@ -77,7 +77,7 @@ def parse_csv(csv_path: PathLike) -> pl.LazyFrame:
"""
lf = pl.scan_csv(csv_path, schema_overrides={"ID": pl.String, "Data": pl.String})
- if "Timestamp" not in lf.columns:
+ if "Timestamp" not in lf.collect_schema().names():
lf = lf.with_row_index("Timestamp")
byte_exprs = []
diff --git a/pyproject.toml b/pyproject.toml
index b14b118..2b78937 100644
--- a/pyproject.toml
+++ b/pyproject.toml
@@ -1,7 +1,15 @@
[project]
name = "CANveyor"
-version = "0.0.4"
+version = "0.1.0"
description = "J1939 CAN bus parser that works in pair with CANdigger"
readme = "README.md"
-requires-python = ">=3.14"
-dependencies = ["polars", "pathlib", "typing"]
+requires-python = ">=3.10"
+dependencies = [
+ "polars",
+ "dash",
+ "dash-bootstrap-components",
+ "numpy",
+ "pandas",
+ "plotly",
+ "plotly-resampler"
+]
diff --git a/stat/entropy.py b/stat/entropy.py
deleted file mode 100644
index fc010bc..0000000
--- a/stat/entropy.py
+++ /dev/null
@@ -1,187 +0,0 @@
-# File: entropy.py
-# Copyright (C) 2026 Erick Ahmed
-# SPDX-License-Identifier: AGPL-3.0-or-later
-
-import argparse
-from pathlib import Path
-
-import numpy as np
-import pandas as pd
-import plotly.graph_objects as go
-
-from utils.extractor import load_data
-from utils.extractor import to_int
-
-def _format_can_id(x):
- """Safely cleans CAN ID strings without altering their length or value."""
- if pd.isna(x):
- return "UNKNOWN"
-
- s = str(x).strip()
- if not s:
- return "UNKNOWN"
-
- if s.lower().startswith('0x'):
- s = s[2:]
-
- return s.upper()
-
-def calculate_byte_entropy(df: pd.DataFrame) -> pd.DataFrame:
- """Calculates Shannon entropy per byte position for each identifier."""
- byte_cols = [f"b{i}" for i in range(8)]
- available_cols = [col for col in byte_cols if col in df.columns]
- if not available_cols:
- raise ValueError("No byte columns (b0-b7) found in the DataFrame")
-
- can_id_col = 'ID' if 'ID' in df.columns else 'Identifier'
- identifiers = df[can_id_col].apply(_format_can_id)
-
- df_bytes = df[available_cols].copy()
- for col in available_cols:
- df_bytes[col] = df_bytes[col].apply(to_int)
-
- def entropy(s: pd.Series) -> float:
- s = s.dropna()
- if s.empty:
- return 0.0
- p = s.value_counts(normalize=True)
- return -np.sum(p * np.log2(p))
-
- result = df_bytes.groupby(identifiers)[available_cols].agg(entropy)
- result.index.name = 'Identifier'
- return result
-
-
-def plot_entropy_heatmap(entropy_df: pd.DataFrame, title: str) -> go.Figure:
- """Generates an interactive heatmap of byte-level Shannon entropy."""
- x = entropy_df.columns.tolist()
- y = entropy_df.index.tolist()
- z = entropy_df.values
-
- fig = go.Figure(
- data=go.Heatmap(
- z=z,
- x=x,
- y=y,
- colorscale=[
- [0.0, "#ffffff"],
- [0.15, "#fff7ec"],
- [0.35, "#fee8c8"],
- [0.55, "#fdd49e"],
- [0.75, "#fdbb84"],
- [1.0, "#ef6548"],
- ],
- xgap=3,
- ygap=3,
- text=np.round(z, 2),
- texttemplate="%{text}",
- textfont={
- "size": 11,
- "color": "#2a2a2a",
- "family": "Segoe UI, Arial, sans-serif",
- },
- hoverongaps=False,
- hovertemplate=(
- "%{y}
"
- "Byte %{x}: %{z:.2f} bits"
- ),
- colorbar=dict(
- title=dict(
- text="Entropy (bits)",
- side="top",
- font=dict(size=13, color="#1a1a1a"),
- ),
- orientation="h",
- thickness=15,
- len=0.35,
- x=1.0,
- xanchor="right",
- y=1.02,
- yanchor="bottom",
- tickfont=dict(size=11, color="#2a2a2a"),
- tickformat=".1f",
- outlinewidth=0.5,
- outlinecolor="#cccccc",
- ),
- )
- )
-
- fig.update_layout(
- title=dict(
- text=title,
- font=dict(size=20, color="#1a1a1a"),
- x=0.5,
- xanchor="center",
- pad=dict(b=20),
- ),
- height=max(600, len(y) * 28 + 150),
- autosize=True,
- template="plotly_white",
- xaxis=dict(
- title=dict(text="Byte Position", font=dict(size=13, color="#1a1a1a")),
- side="top",
- dtick=1,
- showgrid=False,
- linecolor="#bdbdbd",
- tickfont=dict(size=12, color="#2a2a2a"),
- ticks="outside",
- ticklen=4,
- tickcolor="#cccccc",
- ),
- yaxis=dict(
- title=dict(text="PGN or CAN ID", font=dict(size=13, color="#1a1a1a")),
- autorange="reversed",
- showgrid=False,
- linecolor="#bdbdbd",
- tickfont=dict(size=12, color="#2a2a2a"),
- ticks="outside",
- ticklen=4,
- tickcolor="#cccccc",
- automargin=True,
- ),
- font=dict(family="Segoe UI, Arial, sans-serif", size=12, color="#2a2a2a"),
- hoverlabel=dict(
- bgcolor="white",
- font_size=13,
- font_family="Segoe UI",
- bordercolor="#cccccc",
- ),
- margin=dict(l=200, r=40, t=120, b=60),
- )
- return fig
-
-
-if __name__ == "__main__":
- parser = argparse.ArgumentParser(
- description="Analyze CAN bus byte-level entropy"
- )
- parser.add_argument(
- "input", type=Path, help="Path to the input CAN log file"
- )
- parser.add_argument(
- "output",
- type=Path,
- nargs="?",
- default=Path("entropy_report.html"),
- help="Path to the output HTML report",
- )
- parser.add_argument(
- "title",
- nargs="?",
- default="CAN Bus Byte-Level Entropy",
- help="Title for the HTML report",
- )
- args = parser.parse_args()
-
- df = load_data(args.input)
- entropy_df = calculate_byte_entropy(df)
- fig = plot_entropy_heatmap(entropy_df, title=args.title)
-
- config = {
- "responsive": True,
- "displaylogo": False,
- "scrollZoom": True,
- "modeBarButtonsToAdd": ["toggleSpikelines"],
- "toImageButtonOptions": {"format": "png", "scale": 2},
- }
- fig.write_html(str(args.output), include_plotlyjs="cdn", config=config)
diff --git a/stat/utils/extractor.py b/stat/utils/extractor.py
deleted file mode 100644
index c2d63e7..0000000
--- a/stat/utils/extractor.py
+++ /dev/null
@@ -1,40 +0,0 @@
-# File: extractor.py
-# Copyright (C) 2026 Erick Ahmed
-# SPDX-License-Identifier: AGPL-3.0-or-later
-
-import json
-from pathlib import Path
-
-import numpy as np
-import pandas as pd
-
-def to_int(x):
- """Convert a hex string or integer to int, returning NaN on failure."""
- if isinstance(x, (int, np.integer)):
- return int(x)
- if isinstance(x, str):
- try:
- return int(x, 16)
- except ValueError:
- return np.nan
- return np.nan
-
-def extract_id(row: pd.Series) -> str:
- """Extracts PGN from metadata or falls back to CAN ID."""
- meta = row.get('j1939_metadata')
- if pd.isna(meta):
- return f"ID: {row['ID']}"
- if isinstance(meta, str):
- try:
- meta = json.loads(meta)
- except json.JSONDecodeError:
- return f"ID: {row['ID']}"
- if isinstance(meta, dict) and 'PGN' in meta:
- return f"PGN: {meta['PGN']}"
- return f"ID: {row['ID']}"
-
-def load_data(file_path: Path) -> pd.DataFrame:
- """Loads Parquet file and adds an Identifier column."""
- df = pd.read_parquet(file_path)
- df['Identifier'] = df.apply(extract_id, axis=1)
- return df
diff --git a/stats/__init__.py b/stats/__init__.py
new file mode 100644
index 0000000..e69de29
diff --git a/stat/correlation.py b/stats/correlation.py
similarity index 53%
rename from stat/correlation.py
rename to stats/correlation.py
index 06aa856..3035f39 100644
--- a/stat/correlation.py
+++ b/stats/correlation.py
@@ -4,83 +4,107 @@
import argparse
from pathlib import Path
+from concurrent.futures import ThreadPoolExecutor
import numpy as np
import pandas as pd
import plotly.graph_objects as go
-from utils.extractor import load_data
-from utils.extractor import to_int
+from stats.utils.extractor import load_data
+from stats.utils.extractor import to_int
-def _format_can_id(x):
- """Safely cleans CAN ID strings without altering their length or value."""
- if pd.isna(x):
- return "UNKNOWN"
+def _format_can_id_vec(s: pd.Series) -> pd.Series:
+ s = s.astype('string').str.strip()
+ s = s.str.replace(r'^0x', '', case=False, regex=True)
+ s = s.str.upper()
+ return s.fillna('UNKNOWN').replace('', 'UNKNOWN')
- s = str(x).strip()
- if not s:
- return "UNKNOWN"
-
- if s.lower().startswith('0x'):
- s = s[2:]
-
- return s.upper()
+def _ensure_int_bytes(df: pd.DataFrame, cols: list) -> pd.DataFrame:
+ needs = [c for c in cols if not pd.api.types.is_numeric_dtype(df[c])]
+ if needs:
+ df = df.copy()
+ for c in needs:
+ df[c] = df[c].apply(to_int)
+ return df
def calculate_correlation(df: pd.DataFrame, method: str, target_id: str | None = None) -> pd.DataFrame:
- """Calculates inter-byte correlation grouped by identifier."""
- byte_cols = [f"b{i}" for i in range(8)]
- available_cols = [col for col in byte_cols if col in df.columns]
+ available_cols = [f"b{i}" for i in range(8) if f"b{i}" in df.columns]
if not available_cols:
raise ValueError("No byte columns (b0-b7) found in the DataFrame")
can_id_col = 'ID' if 'ID' in df.columns else 'Identifier'
- identifiers = df[can_id_col].apply(_format_can_id)
+ identifiers = _format_can_id_vec(df[can_id_col]).to_numpy()
- df_bytes = df[available_cols].copy()
- for col in available_cols:
- df_bytes[col] = df_bytes[col].apply(to_int)
+ df_bytes = _ensure_int_bytes(df, available_cols)[available_cols]
+ data = df_bytes.to_numpy(dtype=np.float64, copy=False)
if target_id is not None:
- target_id = _format_can_id(target_id)
- group = df_bytes[identifiers == target_id]
- if group.empty:
+ target_id = _format_can_id_vec(pd.Series([target_id])).iloc[0]
+ mask = identifiers == target_id
+ if not mask.any():
raise ValueError(f"Identifier '{target_id}' not found in data")
- return group[available_cols].corr(method=method).fillna(0.0)
+ sub = data[mask]
+ mask = ~np.isnan(sub).any(axis=1)
+ sub = sub[mask]
+ if method == 'spearman' and sub.shape[0] > 1:
+ sub = pd.DataFrame(sub).rank().to_numpy()
+ if sub.shape[0] > 1:
+ with np.errstate(divide='ignore', invalid='ignore'):
+ c = np.corrcoef(sub, rowvar=False)
+ np.nan_to_num(c, copy=False, nan=0.0)
+ else:
+ c = np.zeros((len(available_cols), len(available_cols)))
+ return pd.DataFrame(c, index=available_cols, columns=available_cols)
- def max_abs_corr(group: pd.DataFrame) -> pd.Series:
- corr_arr = np.abs(group.corr(method=method).to_numpy().copy())
- np.fill_diagonal(corr_arr, 0.0)
- return pd.Series(corr_arr.max(axis=0), index=group.columns).fillna(0.0)
+ unique_ids, inverse = np.unique(identifiers, return_inverse=True)
+ n_cols = len(available_cols)
- result = df_bytes.groupby(identifiers)[available_cols].apply(max_abs_corr)
+ sort_idx = np.argsort(inverse, kind='stable')
+ data_sorted = data[sort_idx]
+ inverse_sorted = inverse[sort_idx]
+
+ if len(inverse_sorted) > 0:
+ split_points = np.flatnonzero(np.diff(inverse_sorted)) + 1
+ groups = np.split(data_sorted, split_points)
+ else:
+ groups = []
+
+ def _process_group(sub):
+ mask = ~np.isnan(sub).any(axis=1)
+ sub = sub[mask]
+ if len(sub) > 1:
+ if method == 'spearman':
+ sub = pd.DataFrame(sub).rank().to_numpy()
+ with np.errstate(divide='ignore', invalid='ignore'):
+ c = np.abs(np.corrcoef(sub, rowvar=False))
+ np.nan_to_num(c, copy=False, nan=0.0)
+ np.fill_diagonal(c, 0.0)
+ return c.max(axis=0)
+ return np.zeros(n_cols, dtype=np.float64)
+
+ with ThreadPoolExecutor() as executor:
+ out = np.array(list(executor.map(_process_group, groups)))
+
+ result = pd.DataFrame(out, index=unique_ids, columns=available_cols)
result.index.name = 'Identifier'
return result
def plot_correlation_heatmap(corr_df: pd.DataFrame, target_id: str | None, title: str) -> go.Figure:
- """Generates an interactive heatmap of inter-byte correlation."""
is_8x8 = target_id is not None
+ x = corr_df.columns.tolist()
+ y = corr_df.index.tolist()
+ z = corr_df.values
+
if is_8x8:
- x = corr_df.columns.tolist()
- y = corr_df.index.tolist()
- z = corr_df.values
z_min, z_max = -1.0, 1.0
- colorscale = [
- [0.0, "#2c7bb6"], [0.25, "#abd9e9"], [0.5, "#ffffff"],
- [0.75, "#fdae61"], [1.0, "#d7191c"]
- ]
+ colorscale = [[0.0, "#2c7bb6"], [0.25, "#abd9e9"], [0.5, "#ffffff"], [0.75, "#fdae61"], [1.0, "#d7191c"]]
hover_template = "%{y} vs %{x}
Correlation: %{z:.2f}"
else:
- x = corr_df.columns.tolist()
- y = corr_df.index.tolist()
- z = corr_df.values
z_min, z_max = 0.0, 1.0
- colorscale = [
- [0.0, "#ffffff"], [0.2, "#fff5f0"], [0.4, "#fecc5c"],
- [0.6, "#fd8d3c"], [0.8, "#e31a1c"], [1.0, "#800026"]
- ]
+ colorscale = [[0.0, "#ffffff"], [0.2, "#fff5f0"], [0.4, "#fecc5c"], [0.6, "#fd8d3c"], [0.8, "#e31a1c"], [1.0, "#800026"]]
hover_template = "%{y}
Byte %{x} max correlation: %{z:.2f}"
fig = go.Figure(
@@ -106,17 +130,17 @@ def plot_correlation_heatmap(corr_df: pd.DataFrame, target_id: str | None, title
fig.update_layout(
title=dict(text=title, font=dict(size=20, color="#1a1a1a"), x=0.5, xanchor="center", pad=dict(b=20)),
- height=max(600, len(y) * 28 + 150) if not is_8x8 else 600,
+ height=600 if is_8x8 else max(600, len(y) * 28 + 150),
autosize=True,
template="plotly_white",
xaxis=dict(
title=dict(text="Byte Position", font=dict(size=13, color="#1a1a1a")),
- side="top" if not is_8x8 else "bottom",
+ side="bottom" if is_8x8 else "top",
dtick=1, showgrid=False, linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"), ticks="outside", ticklen=4, tickcolor="#cccccc",
),
yaxis=dict(
- title=dict(text="PGN or CAN ID" if not is_8x8 else "Byte Position", font=dict(size=13, color="#1a1a1a")),
+ title=dict(text="Byte Position" if is_8x8 else "PGN or CAN ID", font=dict(size=13, color="#1a1a1a")),
autorange="reversed", showgrid=False, linecolor="#bdbdbd",
tickfont=dict(size=12, color="#2a2a2a"), ticks="outside", ticklen=4, tickcolor="#cccccc", automargin=True,
),
@@ -131,17 +155,15 @@ if __name__ == "__main__":
parser = argparse.ArgumentParser(description="Analyze CAN bus inter-byte correlation")
parser.add_argument("method", choices=["pearson", "spearman"], help="Correlation method to use")
parser.add_argument("input", type=Path, help="Path to the input CAN log file")
- parser.add_argument("output", type=Path, nargs="?", default=Path("correlation_report.html"), help="Path to the output HTML report")
- parser.add_argument("title", nargs="?", default="CAN Bus Inter-Byte Correlation", help="Title for the HTML report")
- parser.add_argument("--identifier", type=str, default=None, help="Specific PGN/CAN ID to analyze (e.g., 'PGN: 65331'). If omitted, shows max correlation per byte for all IDs.")
+ parser.add_argument("output", type=Path, nargs="?", default=Path("correlation_report.html"))
+ parser.add_argument("title", nargs="?", default="CAN Bus Inter-Byte Correlation")
+ parser.add_argument("--identifier", type=str, default=None)
args = parser.parse_args()
df = load_data(args.input)
corr_df = calculate_correlation(df, method=args.method, target_id=args.identifier)
-
display_title = f"{args.title} ({args.identifier})" if args.identifier else args.title
fig = plot_correlation_heatmap(corr_df, target_id=args.identifier, title=display_title)
-
config = {
"responsive": True,
"displaylogo": False,
diff --git a/stats/entropy.py b/stats/entropy.py
new file mode 100644
index 0000000..ac3295e
--- /dev/null
+++ b/stats/entropy.py
@@ -0,0 +1,152 @@
+# File: entropy.py
+# Copyright (C) 2026 Erick Ahmed
+# SPDX-License-Identifier: AGPL-3.0-or-later
+
+import argparse
+from pathlib import Path
+from concurrent.futures import ThreadPoolExecutor
+
+import numpy as np
+import pandas as pd
+import plotly.graph_objects as go
+
+from stats.utils.extractor import load_data
+from stats.utils.extractor import to_int
+
+def _format_can_id_vec(s: pd.Series) -> pd.Series:
+ s = s.astype('string').str.strip()
+ s = s.str.replace(r'^0x', '', case=False, regex=True)
+ s = s.str.upper()
+ return s.fillna('UNKNOWN').replace('', 'UNKNOWN')
+
+def _entropy_col(a: np.ndarray) -> float:
+ a = a[~np.isnan(a)]
+ if a.size == 0:
+ return 0.0
+ a = a.astype(np.int64)
+ lo, hi = a.min(), a.max()
+ span = hi - lo + 1
+ if span <= 0:
+ return 0.0
+ if span > 1 << 20:
+ _, counts = np.unique(a, return_counts=True)
+ else:
+ counts = np.bincount(a - lo, minlength=span)
+ counts = counts[counts > 0]
+ p = counts / counts.sum()
+ return float(-np.sum(p * np.log2(p)))
+
+def calculate_byte_entropy(df: pd.DataFrame) -> pd.DataFrame:
+ available_cols = [f"b{i}" for i in range(8) if f"b{i}" in df.columns]
+ if not available_cols:
+ raise ValueError("No byte columns (b0-b7) found in the DataFrame")
+
+ can_id_col = 'ID' if 'ID' in df.columns else 'Identifier'
+ identifiers = _format_can_id_vec(df[can_id_col]).to_numpy()
+
+ needs = [c for c in available_cols if not pd.api.types.is_numeric_dtype(df[c])]
+ if needs:
+ df = df.copy()
+ for c in needs:
+ df[c] = df[c].apply(to_int)
+
+ data = df[available_cols].to_numpy(dtype=np.float64, copy=False)
+ unique_ids, inverse = np.unique(identifiers, return_inverse=True)
+ n_cols = len(available_cols)
+
+ sort_idx = np.argsort(inverse, kind='stable')
+ data_sorted = data[sort_idx]
+ inverse_sorted = inverse[sort_idx]
+
+ if len(inverse_sorted) > 0:
+ split_points = np.flatnonzero(np.diff(inverse_sorted)) + 1
+ groups = np.split(data_sorted, split_points)
+ else:
+ groups = []
+
+ def _process_group(sub):
+ res = np.zeros(n_cols, dtype=np.float64)
+ for ci in range(n_cols):
+ res[ci] = _entropy_col(sub[:, ci])
+ return res
+
+ with ThreadPoolExecutor() as executor:
+ out = np.array(list(executor.map(_process_group, groups)))
+
+ result = pd.DataFrame(out, index=unique_ids, columns=available_cols)
+ result.index.name = 'Identifier'
+ return result
+
+
+def plot_entropy_heatmap(entropy_df: pd.DataFrame, title: str) -> go.Figure:
+ x = entropy_df.columns.tolist()
+ y = entropy_df.index.tolist()
+ z = entropy_df.values
+
+ fig = go.Figure(
+ data=go.Heatmap(
+ z=z, x=x, y=y,
+ colorscale=[
+ [0.0, "#ffffff"],
+ [0.15, "#fff7ec"],
+ [0.35, "#fee8c8"],
+ [0.55, "#fdd49e"],
+ [0.75, "#fdbb84"],
+ [1.0, "#ef6548"],
+ ],
+ xgap=3, ygap=3,
+ text=np.round(z, 2),
+ texttemplate="%{text}",
+ textfont={"size": 11, "color": "#2a2a2a", "family": "Segoe UI, Arial, sans-serif"},
+ hoverongaps=False,
+ hovertemplate="%{y}
Byte %{x}: %{z:.2f} bits",
+ colorbar=dict(
+ title=dict(text="Entropy (bits)", side="top", font=dict(size=13, color="#1a1a1a")),
+ orientation="h", thickness=15, len=0.35,
+ x=1.0, xanchor="right", y=1.02, yanchor="bottom",
+ tickfont=dict(size=11, color="#2a2a2a"),
+ tickformat=".1f", outlinewidth=0.5, outlinecolor="#cccccc",
+ ),
+ )
+ )
+
+ fig.update_layout(
+ title=dict(text=title, font=dict(size=20, color="#1a1a1a"), x=0.5, xanchor="center", pad=dict(b=20)),
+ height=max(600, len(y) * 28 + 150),
+ autosize=True,
+ template="plotly_white",
+ xaxis=dict(
+ title=dict(text="Byte Position", font=dict(size=13, color="#1a1a1a")),
+ side="top", dtick=1, showgrid=False, linecolor="#bdbdbd",
+ tickfont=dict(size=12, color="#2a2a2a"), ticks="outside", ticklen=4, tickcolor="#cccccc",
+ ),
+ yaxis=dict(
+ title=dict(text="PGN or CAN ID", font=dict(size=13, color="#1a1a1a")),
+ autorange="reversed", showgrid=False, linecolor="#bdbdbd",
+ tickfont=dict(size=12, color="#2a2a2a"), ticks="outside", ticklen=4, tickcolor="#cccccc", automargin=True,
+ ),
+ font=dict(family="Segoe UI, Arial, sans-serif", size=12, color="#2a2a2a"),
+ hoverlabel=dict(bgcolor="white", font_size=13, font_family="Segoe UI", bordercolor="#cccccc"),
+ margin=dict(l=200, r=40, t=120, b=60),
+ )
+ return fig
+
+
+if __name__ == "__main__":
+ parser = argparse.ArgumentParser(description="Analyze CAN bus byte-level entropy")
+ parser.add_argument("input", type=Path, help="Path to the input CAN log file")
+ parser.add_argument("output", type=Path, nargs="?", default=Path("entropy_report.html"))
+ parser.add_argument("title", nargs="?", default="CAN Bus Byte-Level Entropy")
+ args = parser.parse_args()
+
+ df = load_data(args.input)
+ entropy_df = calculate_byte_entropy(df)
+ fig = plot_entropy_heatmap(entropy_df, title=args.title)
+ config = {
+ "responsive": True,
+ "displaylogo": False,
+ "scrollZoom": True,
+ "modeBarButtonsToAdd": ["toggleSpikelines"],
+ "toImageButtonOptions": {"format": "png", "scale": 2},
+ }
+ fig.write_html(str(args.output), include_plotlyjs="cdn", config=config)
diff --git a/stat/frequency.py b/stats/frequency.py
similarity index 58%
rename from stat/frequency.py
rename to stats/frequency.py
index 70d8002..57b1f5f 100644
--- a/stat/frequency.py
+++ b/stats/frequency.py
@@ -4,52 +4,56 @@
import argparse
from pathlib import Path
+import numpy as np
import pandas as pd
-import plotly.express as px
import plotly.graph_objects as go
-from utils.extractor import load_data
+from stats.utils.extractor import load_data
-def _format_can_id(x):
- """Safely cleans CAN ID strings without altering their length or value."""
- if pd.isna(x):
- return "UNKNOWN"
-
- s = str(x).strip()
- if not s:
- return "UNKNOWN"
-
- if s.lower().startswith('0x'):
- s = s[2:]
-
- return s.upper()
+def _format_can_id_vec(s: pd.Series) -> pd.Series:
+ s = s.astype('string').str.strip()
+ s = s.str.replace(r'^0x', '', case=False, regex=True)
+ s = s.str.upper()
+ return s.fillna('UNKNOWN').replace('', 'UNKNOWN')
def calculate_frequency(df: pd.DataFrame) -> pd.DataFrame:
- """Calculates frequency counts and percentages for identifiers."""
can_id_col = 'ID' if 'ID' in df.columns else 'Identifier'
+ formatted = _format_can_id_vec(df[can_id_col])
- df['Formatted_ID'] = df[can_id_col].apply(_format_can_id)
-
- freq_df = df['Formatted_ID'].value_counts().reset_index()
- freq_df.columns = ['Identifier', 'Count']
-
- total = freq_df['Count'].sum()
- freq_df['Percentage'] = (freq_df['Count'] / total * 100).round(2)
- return freq_df.sort_values('Count', ascending=True)
+ counts = formatted.value_counts()
+ freq_df = pd.DataFrame({
+ 'Identifier': counts.index,
+ 'Count': counts.to_numpy(),
+ })
+ total = counts.sum()
+ freq_df['Percentage'] = np.round(freq_df['Count'] / total * 100, 2) if total else 0.0
+ return freq_df.sort_values('Count', ascending=True).reset_index(drop=True)
def plot_frequency(stats_df: pd.DataFrame, title: str) -> go.Figure:
- """Generates interactive horizontal bar chart with log x-axis."""
- fig = px.bar(
- stats_df, y='Identifier', x='Count', orientation='h', title=title, log_x=True,
- labels={'Identifier': 'PGN / CAN ID', 'Count': 'Message Count'},
- color='Count', color_continuous_scale='Turbo',
- range_color=(stats_df['Count'].min(), stats_df['Count'].max()),
- hover_data={'Percentage': ':.2f', 'Count': ':,', 'Identifier': True}
- )
+ n = len(stats_df)
+ fig = go.Figure(go.Bar(
+ y=stats_df['Identifier'],
+ x=stats_df['Count'],
+ orientation='h',
+ marker=dict(
+ color=stats_df['Count'],
+ colorscale='Turbo',
+ cmin=int(stats_df['Count'].min()) if n else 0,
+ cmax=int(stats_df['Count'].max()) if n else 1,
+ line_width=0,
+ ),
+ customdata=stats_df[['Percentage']].to_numpy(),
+ hovertemplate="%{y}
Count: %{x:,}
Share: %{customdata[0]}%",
+ texttemplate='%{x:,}',
+ textposition='outside',
+ cliponaxis=False,
+ ))
+
fig.update_layout(
- height=max(600, len(stats_df) * 18),
+ height=max(600, n * 18),
autosize=True,
template='plotly_white',
xaxis=dict(
+ type='log',
title=dict(text="Message count [log scale]", font=dict(size=13, color="#1a1a1a")),
side="top",
dtick=1,
@@ -69,11 +73,10 @@ def plot_frequency(stats_df: pd.DataFrame, title: str) -> go.Figure:
ticklen=4,
tickcolor="#cccccc",
automargin=True,
- type='category'
+ type='category',
),
font=dict(family="Segoe UI, Arial, sans-serif", size=12, color='#2a2a2a'),
- hoverlabel=dict(bgcolor="white", font_size=13, font_family="Segoe UI",
- bordercolor='#cccccc'),
+ hoverlabel=dict(bgcolor="white", font_size=13, font_family="Segoe UI", bordercolor='#cccccc'),
margin=dict(l=200, r=40, t=120, b=60),
bargap=0.35,
coloraxis_colorbar=dict(
@@ -87,39 +90,27 @@ def plot_frequency(stats_df: pd.DataFrame, title: str) -> go.Figure:
yanchor='bottom',
tickformat=',',
outlinecolor='#cccccc',
- outlinewidth=0.5
+ outlinewidth=0.5,
),
- title=dict(font=dict(size=20, color='#1a1a1a'), x=0.5, xanchor='center',
- pad=dict(b=20))
+ title=dict(text=title, font=dict(size=20, color='#1a1a1a'), x=0.5, xanchor='center', pad=dict(b=20)),
)
fig.update_xaxes(
showgrid=True, gridwidth=0.5, gridcolor='#e8e8e8',
zeroline=False, linecolor='#bdbdbd', mirror=False,
tickformat=',',
- minor=dict(showgrid=True, gridcolor='#f4f4f4', gridwidth=0.5)
+ minor=dict(showgrid=True, gridcolor='#f4f4f4', gridwidth=0.5),
)
fig.update_yaxes(
showgrid=False, zeroline=False, linecolor='#bdbdbd',
- ticks='outside', ticklen=4, tickcolor='#cccccc',
- automargin=True
- )
- fig.update_traces(
- hovertemplate="%{y}
Count: %{x:,}
Share: %{customdata[0]}%",
- marker_line_width=0,
- texttemplate='%{x:,}',
- textposition='outside',
- textfont=dict(size=10, color='#666666'),
- cliponaxis=False,
- selected=dict(marker=dict(opacity=0.6)),
- unselected=dict(marker=dict(opacity=0.2))
+ ticks='outside', ticklen=4, tickcolor='#cccccc', automargin=True,
)
return fig
if __name__ == "__main__":
parser = argparse.ArgumentParser(description="Analyze CAN bus message frequency")
parser.add_argument("input", type=Path, help="Path to the input CAN log file")
- parser.add_argument("output", type=Path, nargs="?", default=Path("freq_report.html"), help="Path to the output HTML report")
- parser.add_argument("title", nargs="?", default="CAN Bus Message Frequency", help="Title for the HTML report")
+ parser.add_argument("output", type=Path, nargs="?", default=Path("freq_report.html"))
+ parser.add_argument("title", nargs="?", default="Frequency")
args = parser.parse_args()
df = load_data(args.input)
@@ -130,6 +121,6 @@ if __name__ == "__main__":
'displaylogo': False,
'scrollZoom': True,
'modeBarButtonsToAdd': ['toggleSpikelines'],
- 'toImageButtonOptions': {'format': 'png', 'scale': 2}
+ 'toImageButtonOptions': {'format': 'png', 'scale': 2},
}
fig.write_html(str(args.output), include_plotlyjs='cdn', config=config)
diff --git a/stats/id_viewer.py b/stats/id_viewer.py
new file mode 100644
index 0000000..f3e30c8
--- /dev/null
+++ b/stats/id_viewer.py
@@ -0,0 +1,143 @@
+# File: id_viewer.py
+# Copyright (C) 2026 Erick Ahmed
+# SPDX-License-Identifier: AGPL-3.0-or-later
+
+import argparse
+from pathlib import Path
+import numpy as np
+import pandas as pd
+import plotly.graph_objects as go
+from plotly_resampler import FigureResampler
+from stats.utils.extractor import load_data
+
+def _format_can_id_vec(s: pd.Series) -> pd.Series:
+ s = s.astype('string').str.strip()
+ s = s.str.replace(r'^0x', '', case=False, regex=True)
+ s = s.str.upper()
+ return s.fillna('UNKNOWN').replace('', 'UNKNOWN')
+
+def prepare_data(df, target_id):
+ can_id_col = 'ID' if 'ID' in df.columns else 'Identifier'
+ formatted = _format_can_id_vec(df[can_id_col])
+ df = df.assign(Formatted_ID=formatted)
+ target_id_clean = _format_can_id_vec(pd.Series([target_id])).iloc[0]
+ filtered = df[df['Formatted_ID'] == target_id_clean]
+
+ byte_cols = [f"b{i}" for i in range(8) if f"b{i}" in filtered.columns]
+ if filtered.empty:
+ return filtered, byte_cols
+
+ for col in byte_cols:
+ if not pd.api.types.is_numeric_dtype(filtered[col]):
+ filtered = filtered.assign(**{col: pd.to_numeric(filtered[col], errors='coerce').astype('float32')})
+
+ filtered = filtered.sort_values('Timestamp', kind='stable')
+ arr = filtered[byte_cols].to_numpy(dtype=np.float32, copy=False)
+ if len(arr) > 1:
+ changed = np.any(arr[1:] != arr[:-1], axis=1)
+ keep = np.concatenate(([True], changed))
+ filtered = filtered.iloc[keep]
+
+ return filtered, byte_cols
+
+def plot_bits(df, byte_cols, can_id, title):
+ fig = FigureResampler(
+ resampled_trace_prefix_suffix=("", ""),
+ show_mean_aggregation_size=False
+ )
+ colors = ['#e41a1c', '#377eb8', '#4daf4a', '#984ea3', '#ff7f00', '#ffff33', '#a65628', '#f781bf']
+ n = len(byte_cols)
+
+ x = df['Timestamp'].to_numpy() if not df.empty else np.array([])
+ for i, col in enumerate(byte_cols):
+ y = df[col].to_numpy(dtype=np.float32, copy=False) if not df.empty else np.array([])
+
+ fig.add_trace(go.Scatter(
+ mode='lines',
+ line=dict(shape='hv', width=2, color=colors[i % len(colors)]),
+ name=col.upper(),
+ legendgroup=col.upper(),
+ hovertemplate=f"{col.upper()}
Time: %{{x}}
Value: %{{y}}",
+ ), hf_x=x, hf_y=y)
+
+ all_button = dict(label='ALL', method='restyle', args=[{'visible': [True] * n}])
+ none_button = dict(label='NONE', method='restyle', args=[{'visible': ['legendonly'] * n}])
+
+ fig.update_layout(
+ height=600,
+ autosize=True,
+ template='plotly_white',
+ title=dict(
+ text=f"{title} - ID: {can_id}",
+ font=dict(size=20, color='#1a1a1a'),
+ x=0.5, xanchor='center',
+ pad=dict(b=20),
+ ),
+ font=dict(family="Segoe UI, Arial, sans-serif", size=12, color='#2a2a2a'),
+ hoverlabel=dict(bgcolor="white", font_size=13, font_family="Segoe UI", bordercolor='#cccccc'),
+ margin=dict(l=60, r=40, t=120, b=140),
+ legend=dict(
+ orientation='h',
+ x=0.5, xanchor='center',
+ y=-0.18, yanchor='top',
+ title=None,
+ bgcolor='white',
+ bordercolor='#cccccc',
+ borderwidth=1,
+ font=dict(size=12, color="#2a2a2a"),
+ itemsizing='constant',
+ itemclick='toggle',
+ itemdoubleclick='toggleothers',
+ ),
+ xaxis=dict(
+ title=dict(text="Timestamp", font=dict(size=13, color="#1a1a1a")),
+ showgrid=True, gridwidth=0.5, gridcolor='#e8e8e8',
+ zeroline=False, linecolor="#bdbdbd",
+ tickfont=dict(size=12, color="#2a2a2a"),
+ ticks="outside", ticklen=4, tickcolor="#cccccc",
+ minor=dict(showgrid=True, gridcolor='#f4f4f4', gridwidth=0.5),
+ ),
+ yaxis=dict(
+ title=dict(text="Byte Value", font=dict(size=13, color="#1a1a1a")),
+ showgrid=True, gridwidth=0.5, gridcolor='#e8e8e8',
+ zeroline=False, linecolor="#bdbdbd",
+ tickfont=dict(size=12, color="#2a2a2a"),
+ ticks="outside", ticklen=4, tickcolor="#cccccc",
+ ),
+ updatemenus=[
+ dict(
+ type='buttons',
+ direction='right',
+ x=0.5, xanchor='center',
+ y=-0.06, yanchor='top',
+ buttons=[all_button, none_button],
+ bgcolor='white',
+ bordercolor='#cccccc',
+ borderwidth=1,
+ font=dict(size=11, color='#2a2a2a'),
+ pad=dict(l=5, r=5, t=5, b=5),
+ )
+ ],
+ )
+ return fig
+
+if __name__ == "__main__":
+ parser = argparse.ArgumentParser(description="Visualize CAN bus byte changes over time")
+ parser.add_argument("input", type=Path, help="Path to the input CAN log file")
+ parser.add_argument("can_id", type=str, help="CAN ID to visualize")
+ parser.add_argument("output", type=Path, nargs="?", default=Path("bits_report.html"))
+ parser.add_argument("title", nargs="?", default="Byte Visualization")
+ args = parser.parse_args()
+
+ df = load_data(args.input)
+ filtered_df, byte_cols = prepare_data(df, args.can_id)
+ fig = plot_bits(filtered_df, byte_cols, args.can_id, title=args.title)
+
+ config = {
+ 'responsive': True,
+ 'displaylogo': False,
+ 'scrollZoom': True,
+ 'modeBarButtonsToAdd': ['toggleSpikelines'],
+ 'toImageButtonOptions': {'format': 'png', 'scale': 2},
+ }
+ fig.write_html(str(args.output), include_plotlyjs='cdn', config=config)
diff --git a/stats/utils/__init__.py b/stats/utils/__init__.py
new file mode 100644
index 0000000..e69de29
diff --git a/stats/utils/extractor.py b/stats/utils/extractor.py
new file mode 100644
index 0000000..3472f07
--- /dev/null
+++ b/stats/utils/extractor.py
@@ -0,0 +1,63 @@
+# File: extractor.py
+# Copyright (C) 2026 Erick Ahmed
+# SPDX-License-Identifier: AGPL-3.0-or-later
+
+import json
+from pathlib import Path
+
+import numpy as np
+import pandas as pd
+import polars as pl
+
+def to_int(x):
+ if isinstance(x, (int, np.integer)):
+ return int(x)
+ if isinstance(x, str):
+ try:
+ return int(x, 16)
+ except ValueError:
+ return np.nan
+ return np.nan
+
+def extract_id(row: pd.Series) -> str:
+ meta = row.get('j1939_metadata')
+ if pd.isna(meta):
+ return f"ID: {row['ID']}"
+ if isinstance(meta, str):
+ try:
+ meta = json.loads(meta)
+ except json.JSONDecodeError:
+ return f"ID: {row['ID']}"
+ if isinstance(meta, dict) and 'PGN' in meta:
+ return f"PGN: {meta['PGN']}"
+ return f"ID: {row['ID']}"
+
+def load_data(file_path: Path) -> pd.DataFrame:
+ lf = pl.scan_parquet(file_path)
+ schema = lf.collect_schema()
+ names = schema.names()
+
+ byte_cols = [f"b{i}" for i in range(8) if f"b{i}" in names]
+ if byte_cols:
+ lf = lf.with_columns([
+ pl.col(c).str.to_integer(base=16, strict=False).cast(pl.Int16).alias(c)
+ for c in byte_cols
+ ])
+
+ id_col = 'ID' if 'ID' in names else 'Identifier'
+ id_expr = pl.col(id_col).cast(pl.Utf8)
+
+ if 'j1939_metadata' in names:
+ try:
+ lf = lf.with_columns(
+ pl.when(pl.col('j1939_metadata').is_not_null())
+ .then(pl.lit('PGN: ') + pl.col('j1939_metadata').struct.field('PGN').cast(pl.Utf8))
+ .otherwise(pl.lit('ID: ') + id_expr)
+ .alias('Identifier')
+ )
+ except Exception:
+ lf = lf.with_columns((pl.lit('ID: ') + id_expr).alias('Identifier'))
+ else:
+ lf = lf.with_columns((pl.lit('ID: ') + id_expr).alias('Identifier'))
+
+ return lf.collect().to_pandas()