From 17340746ce1a392c10b29dd9c94e235b8ca0658f Mon Sep 17 00:00:00 2001 From: Erick Ahmed Date: Mon, 13 Jul 2026 19:51:22 +0200 Subject: [PATCH] Move data extraction functions to utility library --- stat/frequency.py | 53 +++++++++++----------------------------------- utils/extractor.py | 27 +++++++++++++++++++++++ 2 files changed, 39 insertions(+), 41 deletions(-) create mode 100644 utils/extractor.py diff --git a/stat/frequency.py b/stat/frequency.py index 48f5cb3..c4fd929 100644 --- a/stat/frequency.py +++ b/stat/frequency.py @@ -1,52 +1,26 @@ import argparse -import json from pathlib import Path -import fastparquet import pandas as pd import plotly.express as px import plotly.graph_objects as go +from utils.extractor import load_data - -def _extract_identifier(row: pd.Series) -> str: - """Extracts PGN from metadata or falls back to CAN ID.""" - meta = row.get('j1939_metadata') - if pd.isna(meta): - return f"{row['ID']} " - if isinstance(meta, str): - try: - meta = json.loads(meta) - except json.JSONDecodeError: - return f"{row['ID']}" - if isinstance(meta, dict) and 'PGN' in meta: - return f"PGN: {meta['PGN']} " - return f"{row['ID']}" - - -def load_data(file_path: Path) -> pd.DataFrame: - """Loads Parquet file and adds an Identifier column.""" - df = pd.read_parquet(file_path) - df['Identifier'] = df.apply(_extract_identifier, axis=1) - return df - - -def calculate_frequency(df: pd.DataFrame) -> pd.DataFrame: +def calc_freq(df: pd.DataFrame) -> pd.DataFrame: """Calculates frequency counts and percentages for identifiers.""" freq_df = df['Identifier'].value_counts().reset_index() freq_df.columns = ['Identifier', 'Count'] - total_messages = freq_df['Count'].sum() - freq_df['Percentage'] = (freq_df['Count'] / total_messages * 100).round(2) + total = freq_df['Count'].sum() + freq_df['Percentage'] = (freq_df['Count'] / total * 100).round(2) return freq_df.sort_values('Count', ascending=True) - -def visualize_frequency(stats_df: pd.DataFrame, title: str = "CAN Bus Message Frequency") -> go.Figure: - """Generates an interactive Plotly horizontal bar chart with a logarithmic x-axis.""" +def plot_freq(stats_df: pd.DataFrame, title: str) -> go.Figure: + """Generates interactive horizontal bar chart with log x-axis.""" fig = px.bar( stats_df, y='Identifier', x='Count', orientation='h', title=title, log_x=True, labels={'Identifier': 'PGN / CAN ID', 'Count': 'Message Count'}, color='Count', color_continuous_scale='Turbo', hover_data={'Percentage': ':.2f', 'Count': True, 'Identifier': True} ) - fig.update_layout( height=max(600, len(stats_df) * 18), width=1000, xaxis_title='Total Message Count (Log Scale)', yaxis_title='PGN or CAN ID', @@ -57,20 +31,17 @@ def visualize_frequency(stats_df: pd.DataFrame, title: str = "CAN Bus Message Fr margin=dict(l=250, r=50, t=80, b=50), title=dict(font=dict(size=20), x=0.5) ) - - fig.update_traces( - hovertemplate="%{y}
Count: %{x:,}
Share: %{customdata[0]}%" - ) + fig.update_traces(hovertemplate="%{y}
Count: %{x:,}
Share: %{customdata[0]}%") return fig if __name__ == "__main__": - parser = argparse.ArgumentParser(description="Analyze CAN bus Parquet data.") + parser = argparse.ArgumentParser(description="Analyze CAN bus frequency.") parser.add_argument("-i", "--input", type=Path, required=True) - parser.add_argument("-o", "--output", type=Path, default=Path("can_analysis_report.html")) - parser.add_argument("-t", "--title", type=str, default="CAN Bus Message Frequency by PGN / ID") + parser.add_argument("-o", "--output", type=Path, default=Path("freq_report.html")) + parser.add_argument("-t", "--title", type=str, default="CAN Bus Message Frequency") args = parser.parse_args() df = load_data(args.input) - stats_df = calculate_frequency(df) - fig = visualize_frequency(stats_df, title=args.title) + stats = calc_freq(df) + fig = plot_freq(stats, title=args.title) fig.write_html(str(args.output), include_plotlyjs='cdn') diff --git a/utils/extractor.py b/utils/extractor.py new file mode 100644 index 0000000..8cdd07e --- /dev/null +++ b/utils/extractor.py @@ -0,0 +1,27 @@ +# File: extractor.py +# Copyright (C) 2026 Erick Ahmed +# SPDX-License-Identifier: AGPL-3.0-or-later + +import json +from pathlib import Path +import pandas as pd + +def extract_id(row: pd.Series) -> str: + """Extracts PGN from metadata or falls back to CAN ID.""" + meta = row.get('j1939_metadata') + if pd.isna(meta): + return f"ID: {row['ID']}" + if isinstance(meta, str): + try: + meta = json.loads(meta) + except json.JSONDecodeError: + return f"ID: {row['ID']}" + if isinstance(meta, dict) and 'PGN' in meta: + return f"PGN: {meta['PGN']}" + return f"ID: {row['ID']}" + +def load_data(file_path: Path) -> pd.DataFrame: + """Loads Parquet file and adds an Identifier column.""" + df = pd.read_parquet(file_path) + df['Identifier'] = df.apply(extract_id, axis=1) + return df