From 1b29ef4fa68bfc6f98c8eea4f9664c99e2acdeb3 Mon Sep 17 00:00:00 2001 From: Rameen Mahmood Date: Sat, 28 Mar 2026 20:32:57 -0400 Subject: [PATCH 01/11] add rich to dependencies --- pyproject.toml | 1 + 1 file changed, 1 insertion(+) diff --git a/pyproject.toml b/pyproject.toml index 650da0b..4923574 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -28,6 +28,7 @@ classifiers = [ dependencies = [ "pandas", "tldextract", + "rich", ] [project.urls] From d266f13e011e678a9852d57202ab3e86ecdea6ad Mon Sep 17 00:00:00 2001 From: Rameen Mahmood Date: Sat, 28 Mar 2026 20:33:37 -0400 Subject: [PATCH 02/11] convert pcap-devices to rich table output --- pcap_parser/devices.py | 91 ++++++++++++++++++++++++------------------ 1 file changed, 53 insertions(+), 38 deletions(-) diff --git a/pcap_parser/devices.py b/pcap_parser/devices.py index f192217..293574d 100644 --- a/pcap_parser/devices.py +++ b/pcap_parser/devices.py @@ -17,18 +17,24 @@ import urllib.error import pandas as pd +from rich.console import Console +from rich.table import Table +from rich.panel import Panel +from rich.text import Text DEVID_API_URL = "https://rameen-mahmood--dev-id-predict.modal.run" DEVID_API_KEY = "momo" +console = Console() + def aggregate_devices(input_csv): """Group parsed packet CSV by source MAC and return per-device summaries.""" df = pd.read_csv(input_csv) if "eth.src" not in df.columns: - print("[!] No eth.src column found. Is this a pcap-parse output file?") + console.print("[bold red][!] No eth.src column found. Is this a pcap-parse output file?[/]") sys.exit(1) df = df[df["eth.src"].notna()] @@ -116,52 +122,61 @@ def _format_bytes(n): def print_devices(devices, identify=False): - """Print device table to stdout.""" + """Print device table to stdout using rich.""" if not devices: - print("[!] No devices found.") + console.print("[bold yellow][!] No devices found.[/]") return if identify: - print(f"[+] Identifying {len(devices)} devices via dev-id API...") + console.print(f"\n[bold cyan][+] Identifying {len(devices)} devices via device ID API...[/]\n") + + table = Table( + title="Devices", + show_header=True, + header_style="bold cyan", + border_style="dim", + title_style="bold white", + ) + + table.add_column("#", style="dim", width=3) + table.add_column("MAC Address", style="bold") + table.add_column("Vendor", style="green") + table.add_column("Hostname", style="yellow") + table.add_column("IPs", style="white") + table.add_column("Packets", justify="right", style="white") + table.add_column("Traffic", justify="right", style="magenta") + table.add_column("Top Destinations", style="cyan") + if identify: + table.add_column("Identified As", style="bold green") for i, device in enumerate(devices): - identification = None + ips = ", ".join(device["ips"][:3]) + if len(device["ips"]) > 3: + ips += f" (+{len(device['ips']) - 3})" + + destinations = ", ".join(device["top_destinations"][:3]) + + row = [ + str(i + 1), + device["mac"], + device["oui_vendor"] or "-", + device["dhcp_hostname"] or "-", + ips or "-", + f"{device['packet_count']:,}", + _format_bytes(device["byte_count"]), + destinations or "-", + ] + if identify: identification = identify_device(device) - - print() - print(f" Device {i + 1}") - print(f" {'=' * 50}") - print(f" MAC: {device['mac']}") - if device["oui_vendor"]: - print(f" OUI Vendor: {device['oui_vendor']}") - if device["dhcp_hostname"]: - print(f" DHCP Hostname: {device['dhcp_hostname']}") - if device["ips"]: - print(f" IPs: {', '.join(device['ips'][:5])}") - if device["user_agent"]: - ua = device["user_agent"] - if len(ua) > 80: - ua = ua[:77] + "..." - print(f" User-Agent: {ua}") - print(f" Packets: {device['packet_count']:,}") - print(f" Traffic: {_format_bytes(device['byte_count'])}") - if device["top_destinations"]: - print(f" Top Hosts: {', '.join(device['top_destinations'][:3])}") - - if identification: - vendor = identification["vendor"] + vendor = identification["vendor"].strip() source = identification["source"] - explanation = identification["explanation"] - print(f" Identified As: {vendor.strip()} (via {source})") - if explanation: - if len(explanation) > 100: - explanation = explanation[:97] + "..." - print(f" Explanation: {explanation}") + row.append(f"{vendor} ({source})") + + table.add_row(*row) - print() - print(f" Total: {len(devices)} devices") - print() + console.print(table) + console.print(f"\n [bold]{len(devices)}[/] devices found\n") def main(): @@ -170,7 +185,7 @@ def main(): ) parser.add_argument("input", help="path to the parsed csv file (output of pcap-parse)") parser.add_argument("--identify", action="store_true", - help="identify devices using the dev-id LLM API") + help="identify devices using the device ID LLM API") parser.add_argument("--json", action="store_true", dest="output_json", help="output as json instead of formatted table") args = parser.parse_args() From ddcb485de3d8b42a810aead1bded635cf9de885e Mon Sep 17 00:00:00 2001 From: Rameen Mahmood Date: Sat, 28 Mar 2026 20:34:17 -0400 Subject: [PATCH 03/11] add pcap-summary command with rich output --- pcap_parser/summary.py | 196 +++++++++++++++++++++++++++++++++++++++++ 1 file changed, 196 insertions(+) create mode 100644 pcap_parser/summary.py diff --git a/pcap_parser/summary.py b/pcap_parser/summary.py new file mode 100644 index 0000000..679cdfa --- /dev/null +++ b/pcap_parser/summary.py @@ -0,0 +1,196 @@ +""" +Print a quick overview of parsed pcap data. + +Shows device count, protocol breakdown, top talkers, top destinations, +and capture time range. + +Usage: + pcap-summary parsed_packets.csv +""" + +import argparse +import sys +from datetime import datetime + +import pandas as pd +from rich.console import Console +from rich.table import Table +from rich.panel import Panel +from rich.columns import Columns +from rich.text import Text + + +console = Console() + + +def _format_bytes(n): + """Format byte count as human-readable string.""" + for unit in ["B", "KB", "MB", "GB"]: + if n < 1024: + return f"{n:.1f} {unit}" + n /= 1024 + return f"{n:.1f} TB" + + +def _format_timestamp(epoch): + """Format epoch timestamp as readable datetime.""" + try: + return datetime.fromtimestamp(epoch).strftime("%Y-%m-%d %H:%M:%S") + except (ValueError, OSError, TypeError): + return str(epoch) + + +def _format_duration(seconds): + """Format duration in seconds as human-readable string.""" + if seconds < 60: + return f"{seconds:.1f}s" + elif seconds < 3600: + return f"{seconds / 60:.1f}m" + elif seconds < 86400: + return f"{seconds / 3600:.1f}h" + return f"{seconds / 86400:.1f}d" + + +def summarize(input_csv): + """Generate summary statistics from parsed pcap CSV.""" + df = pd.read_csv(input_csv) + + total_packets = len(df) + total_bytes = int(df["frame.len"].sum()) if "frame.len" in df.columns else 0 + + # time range + start_ts = None + end_ts = None + duration = 0 + if "frame.time_epoch" in df.columns: + epochs = df["frame.time_epoch"].dropna() + if len(epochs) > 0: + start_ts = epochs.min() + end_ts = epochs.max() + duration = end_ts - start_ts + + # device count + device_count = 0 + if "eth.src" in df.columns: + device_count = df["eth.src"].nunique() + + # protocol breakdown + protocols = {} + if "_ws.col.Protocol" in df.columns: + protocols = df["_ws.col.Protocol"].value_counts().head(8).to_dict() + + # top talkers (by bytes sent) + top_talkers = {} + if "ip.src" in df.columns and "frame.len" in df.columns: + top_talkers = df.groupby("ip.src")["frame.len"].sum().sort_values( + ascending=False + ).head(5).to_dict() + + # top destinations + top_destinations = {} + if "dst_hostname" in df.columns: + dst = df["dst_hostname"].dropna() + dst = dst[dst.astype(str).str.strip() != ""] + if len(dst) > 0: + top_destinations = dst.value_counts().head(8).to_dict() + + # top destination IPs (fallback if no hostnames) + top_dst_ips = {} + if not top_destinations and "ip.dst" in df.columns: + top_dst_ips = df["ip.dst"].value_counts().head(8).to_dict() + + return { + "total_packets": total_packets, + "total_bytes": total_bytes, + "start_ts": start_ts, + "end_ts": end_ts, + "duration": duration, + "device_count": device_count, + "protocols": protocols, + "top_talkers": top_talkers, + "top_destinations": top_destinations, + "top_dst_ips": top_dst_ips, + } + + +def print_summary(stats): + """Print summary using rich panels and tables.""" + # overview panel + overview = Text() + overview.append(f" Packets: ", style="dim") + overview.append(f"{stats['total_packets']:,}\n", style="bold white") + overview.append(f" Traffic: ", style="dim") + overview.append(f"{_format_bytes(stats['total_bytes'])}\n", style="bold magenta") + overview.append(f" Devices: ", style="dim") + overview.append(f"{stats['device_count']}\n", style="bold cyan") + if stats["start_ts"] is not None: + overview.append(f" Time Range: ", style="dim") + overview.append( + f"{_format_timestamp(stats['start_ts'])} to {_format_timestamp(stats['end_ts'])}", + style="white", + ) + overview.append(f" ({_format_duration(stats['duration'])})\n", style="dim") + + console.print() + console.print(Panel(overview, title="[bold]Capture Overview[/]", border_style="cyan")) + + # protocol breakdown + if stats["protocols"]: + proto_table = Table( + show_header=True, + header_style="bold cyan", + border_style="dim", + ) + proto_table.add_column("Protocol", style="bold") + proto_table.add_column("Packets", justify="right", style="white") + proto_table.add_column("Share", justify="right", style="dim") + total = stats["total_packets"] + for proto, count in stats["protocols"].items(): + pct = f"{count / total * 100:.1f}%" + proto_table.add_row(str(proto), f"{count:,}", pct) + console.print(Panel(proto_table, title="[bold]Protocols[/]", border_style="green")) + + # top talkers + if stats["top_talkers"]: + talker_table = Table( + show_header=True, + header_style="bold cyan", + border_style="dim", + ) + talker_table.add_column("Source IP", style="bold") + talker_table.add_column("Traffic", justify="right", style="magenta") + for ip, bytes_sent in stats["top_talkers"].items(): + talker_table.add_row(str(ip), _format_bytes(bytes_sent)) + console.print(Panel(talker_table, title="[bold]Top Talkers[/]", border_style="yellow")) + + # top destinations + destinations = stats["top_destinations"] or stats["top_dst_ips"] + if destinations: + dest_table = Table( + show_header=True, + header_style="bold cyan", + border_style="dim", + ) + label = "Destination" if stats["top_destinations"] else "Destination IP" + dest_table.add_column(label, style="bold") + dest_table.add_column("Packets", justify="right", style="white") + for dest, count in destinations.items(): + dest_table.add_row(str(dest), f"{count:,}") + console.print(Panel(dest_table, title="[bold]Top Destinations[/]", border_style="magenta")) + + console.print() + + +def main(): + parser = argparse.ArgumentParser( + description="print a quick summary of parsed pcap data" + ) + parser.add_argument("input", help="path to the parsed csv file (output of pcap-parse)") + args = parser.parse_args() + + stats = summarize(args.input) + print_summary(stats) + + +if __name__ == "__main__": + main() From 85ccb5a3aa4c908f79139ae8fd4af47b786221f9 Mon Sep 17 00:00:00 2001 From: Rameen Mahmood Date: Sat, 28 Mar 2026 20:34:28 -0400 Subject: [PATCH 04/11] add pcap-summary entry point --- pyproject.toml | 1 + 1 file changed, 1 insertion(+) diff --git a/pyproject.toml b/pyproject.toml index 4923574..ddcd95b 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -40,6 +40,7 @@ Issues = "https://github.com/nyu-mlab/pcap-parser/issues" pcap-parse = "pcap_parser.parse:main" pcap-flow = "pcap_parser.flow:main" pcap-devices = "pcap_parser.devices:main" +pcap-summary = "pcap_parser.summary:main" [project.optional-dependencies] dev = [ From f45cea1e84da9d0db245e5fd37a59236b5183808 Mon Sep 17 00:00:00 2001 From: Rameen Mahmood Date: Sat, 28 Mar 2026 20:35:30 -0400 Subject: [PATCH 05/11] fix unused imports flagged by ruff --- pcap_parser/devices.py | 2 -- pcap_parser/summary.py | 10 ++++------ 2 files changed, 4 insertions(+), 8 deletions(-) diff --git a/pcap_parser/devices.py b/pcap_parser/devices.py index 293574d..2f48b7a 100644 --- a/pcap_parser/devices.py +++ b/pcap_parser/devices.py @@ -19,8 +19,6 @@ import pandas as pd from rich.console import Console from rich.table import Table -from rich.panel import Panel -from rich.text import Text DEVID_API_URL = "https://rameen-mahmood--dev-id-predict.modal.run" diff --git a/pcap_parser/summary.py b/pcap_parser/summary.py index 679cdfa..354bfae 100644 --- a/pcap_parser/summary.py +++ b/pcap_parser/summary.py @@ -9,14 +9,12 @@ """ import argparse -import sys from datetime import datetime import pandas as pd from rich.console import Console from rich.table import Table from rich.panel import Panel -from rich.columns import Columns from rich.text import Text @@ -117,14 +115,14 @@ def print_summary(stats): """Print summary using rich panels and tables.""" # overview panel overview = Text() - overview.append(f" Packets: ", style="dim") + overview.append(" Packets: ", style="dim") overview.append(f"{stats['total_packets']:,}\n", style="bold white") - overview.append(f" Traffic: ", style="dim") + overview.append(" Traffic: ", style="dim") overview.append(f"{_format_bytes(stats['total_bytes'])}\n", style="bold magenta") - overview.append(f" Devices: ", style="dim") + overview.append(" Devices: ", style="dim") overview.append(f"{stats['device_count']}\n", style="bold cyan") if stats["start_ts"] is not None: - overview.append(f" Time Range: ", style="dim") + overview.append(" Time Range: ", style="dim") overview.append( f"{_format_timestamp(stats['start_ts'])} to {_format_timestamp(stats['end_ts'])}", style="white", From cb9e83cc901ee1b7672e6002e9b0bc764b137cac Mon Sep 17 00:00:00 2001 From: Rameen Mahmood Date: Sat, 28 Mar 2026 20:35:34 -0400 Subject: [PATCH 06/11] add tests for summary module --- tests/test_summary.py | 82 +++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 82 insertions(+) create mode 100644 tests/test_summary.py diff --git a/tests/test_summary.py b/tests/test_summary.py new file mode 100644 index 0000000..1d3d9bd --- /dev/null +++ b/tests/test_summary.py @@ -0,0 +1,82 @@ +"""Tests for pcap_parser.summary module.""" + +import pandas as pd +import pytest + +from pcap_parser.summary import summarize, _format_bytes, _format_duration + + +class TestSummarize: + @pytest.fixture + def parsed_csv(self, tmp_path): + """Create a CSV that mimics pcap-parse output.""" + data = { + "frame.time_epoch": [1000.0, 1000.5, 1001.0, 1002.0, 1003.0], + "eth.src": ["AA:BB:CC:DD:EE:01"] * 3 + ["AA:BB:CC:DD:EE:02"] * 2, + "ip.src": ["192.168.1.10"] * 3 + ["192.168.1.20"] * 2, + "ip.dst": ["93.184.216.34"] * 3 + ["142.250.80.46"] * 2, + "_ws.col.Protocol": ["TCP", "TCP", "DNS", "TLS", "TCP"], + "frame.len": [100, 200, 50, 300, 150], + "dst_hostname": ["example.com", "example.com", "", "google.com", "google.com"], + } + df = pd.DataFrame(data) + csv_path = str(tmp_path / "parsed.csv") + df.to_csv(csv_path, index=False) + return csv_path + + def test_total_packets(self, parsed_csv): + stats = summarize(parsed_csv) + assert stats["total_packets"] == 5 + + def test_total_bytes(self, parsed_csv): + stats = summarize(parsed_csv) + assert stats["total_bytes"] == 800 + + def test_device_count(self, parsed_csv): + stats = summarize(parsed_csv) + assert stats["device_count"] == 2 + + def test_time_range(self, parsed_csv): + stats = summarize(parsed_csv) + assert stats["start_ts"] == 1000.0 + assert stats["end_ts"] == 1003.0 + assert stats["duration"] == 3.0 + + def test_protocols(self, parsed_csv): + stats = summarize(parsed_csv) + assert "TCP" in stats["protocols"] + assert stats["protocols"]["TCP"] == 3 + + def test_top_destinations(self, parsed_csv): + stats = summarize(parsed_csv) + assert "example.com" in stats["top_destinations"] + assert "google.com" in stats["top_destinations"] + + def test_top_talkers(self, parsed_csv): + stats = summarize(parsed_csv) + assert "192.168.1.10" in stats["top_talkers"] + + +class TestFormatBytes: + def test_bytes(self): + assert _format_bytes(500) == "500.0 B" + + def test_kilobytes(self): + assert _format_bytes(2048) == "2.0 KB" + + def test_megabytes(self): + assert _format_bytes(1048576) == "1.0 MB" + + +class TestFormatDuration: + def test_seconds(self): + assert _format_duration(30) == "30.0s" + + def test_minutes(self): + assert _format_duration(120) == "2.0m" + + def test_hours(self): + assert _format_duration(7200) == "2.0h" + + def test_days(self): + assert _format_duration(172800) == "2.0d" From fc53119311b1cebb325733bcb633e01f506464a9 Mon Sep 17 00:00:00 2001 From: Rameen Mahmood Date: Sat, 28 Mar 2026 20:36:38 -0400 Subject: [PATCH 07/11] rewrite readme with terminal output examples --- README.md | 101 +++++++++++++++++++++++++++++++++++++----------------- 1 file changed, 69 insertions(+), 32 deletions(-) diff --git a/README.md b/README.md index 951eb19..c30a462 100644 --- a/README.md +++ b/README.md @@ -4,18 +4,72 @@ [![CI](https://github.com/nyu-mlab/pcap-parser/actions/workflows/ci-parse.yml/badge.svg)](https://github.com/nyu-mlab/pcap-parser/actions/workflows/ci-parse.yml) [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT) -Python tool to parse pcap files and extract flow-related network traffic information. Extracts hostnames from DNS, TLS SNI, DHCP, and reverse DNS lookups, then aggregates packets into flows with statistics. +Extract devices, flows, and hostnames from pcap files. + +## Quick Start + +```bash +pip install pcap-extract +``` + +Parse a capture and get an instant overview: + +```bash +pcap-parse output.csv capture.pcap +pcap-summary output.csv +``` + +``` +╭──────────────────────────── Capture Overview ────────────────────────────╮ +│ Packets: 12,847 │ +│ Traffic: 8.3 MB │ +│ Devices: 14 │ +│ Time Range: 2025-01-15 09:00:12 to 2025-01-15 09:30:45 (30.6m) │ +╰──────────────────────────────────────────────────────────────────────────╯ +╭──────────────────────────────── Protocols ────────────────────────────────╮ +│ ┏━━━━━━━━━━┳━━━━━━━━━┳━━━━━━━┓ │ +│ ┃ Protocol ┃ Packets ┃ Share ┃ │ +│ ┡━━━━━━━━━━╇━━━━━━━━━╇━━━━━━━┩ │ +│ │ TCP │ 8,421 │ 65.5% │ │ +│ │ TLS │ 2,103 │ 16.4% │ │ +│ │ DNS │ 1,547 │ 12.0% │ │ +│ │ UDP │ 776 │ 6.0% │ │ +│ └──────────┴─────────┴───────┘ │ +╰──────────────────────────────────────────────────────────────────────────╯ +``` + +List devices on the network: + +```bash +pcap-devices output.csv +``` + +``` + Devices +┏━━━┳━━━━━━━━━━━━━━━━━━━┳━━━━━━━━━━━━━━┳━━━━━━━━━━━━━┳━━━━━━━━━┳━━━━━━━━━┓ +┃ # ┃ MAC Address ┃ Vendor ┃ Hostname ┃ Packets ┃ Traffic ┃ +┡━━━╇━━━━━━━━━━━━━━━━━━━╇━━━━━━━━━━━━━━╇━━━━━━━━━━━━━╇━━━━━━━━━╇━━━━━━━━━┩ +│ 1 │ AA:BB:CC:11:22:33 │ Apple, Inc. │ macbook-pro │ 4,521 │ 3.2 MB │ +│ 2 │ AA:BB:CC:44:55:66 │ Google, Inc. │ pixel-6 │ 2,847 │ 1.8 MB │ +│ 3 │ AA:BB:CC:77:88:99 │ Amazon.com │ echo-dot │ 892 │ 412.0 KB│ +└───┴───────────────────┴──────────────┴─────────────┴─────────┴─────────┘ +``` + +Identify unknown devices with a fine-tuned LLM: + +```bash +pcap-devices output.csv --identify +``` ## Features - Parse `.pcap` and `.pcapng` files using tshark -- Hostname enrichment from DNS queries, TLS SNI, DHCP, and reverse DNS -- Device metadata extraction (OUI vendor, HTTP user-agent) -- Flow aggregation with packet counts, byte counts, and inter-arrival times -- Device discovery with per-device traffic summaries +- Instant capture summaries with protocol breakdown, top talkers, and top destinations +- Per-device traffic profiles with OUI vendor, DHCP hostname, and traffic volume - LLM-powered device identification via [IoT Inspector](https://github.com/nyu-mlab/iot-inspector-client) -- Domain extraction from hostnames -- Persistent IP-to-hostname cache across runs +- Hostname enrichment from DNS, TLS SNI, DHCP, and reverse DNS +- Flow aggregation with packet counts, byte counts, and inter-arrival times +- JSON output for all commands (`--json`) ## Requirements @@ -40,47 +94,32 @@ pip install -e ".[dev]" ### Parse pcap files -Parse a single file: - -```bash -pcap-parse output.csv /path/to/capture.pcap -``` - -Parse all pcap files in a directory: - ```bash +pcap-parse output.csv capture.pcap pcap-parse output.csv /path/to/pcap_directory/ ``` -### Aggregate into flows - -After parsing, aggregate packets into flows: +### Get a quick summary ```bash -pcap-flow output.csv aggregated_flows.csv +pcap-summary output.csv ``` ### List devices -List all devices found in the capture: - ```bash pcap-devices output.csv +pcap-devices output.csv --identify # LLM-powered device identification +pcap-devices output.csv --json # machine-readable output ``` -Identify devices using a fine-tuned LLM: - -```bash -pcap-devices output.csv --identify -``` - -Output as JSON: +### Aggregate into flows ```bash -pcap-devices output.csv --json +pcap-flow output.csv flows.csv ``` -### Output +### Output columns `pcap-parse` produces a CSV with columns including: @@ -97,8 +136,6 @@ pcap-devices output.csv --json | `eth.src.oui_resolved` | Device vendor from MAC OUI | | `http.user_agent` | HTTP user-agent string | -`pcap-flow` aggregates these into flows with start/end timestamps, byte counts, packet counts, and average inter-arrival times. - ## Running tests ```bash From d4e745d35ae2fa30c2213b726e26ef942b70216b Mon Sep 17 00:00:00 2001 From: Rameen Mahmood Date: Sat, 28 Mar 2026 20:41:06 -0400 Subject: [PATCH 08/11] handle both protocol column name variants across tshark versions --- pcap_parser/flow.py | 11 +++++++++++ pcap_parser/summary.py | 12 +++++++++--- 2 files changed, 20 insertions(+), 3 deletions(-) diff --git a/pcap_parser/flow.py b/pcap_parser/flow.py index fdfa52d..74fc84d 100644 --- a/pcap_parser/flow.py +++ b/pcap_parser/flow.py @@ -26,10 +26,21 @@ def get_src_port(row): def get_dst_port(row): return row['tcp.dstport'] if pd.notna(row['tcp.dstport']) else row['udp.dstport'] +def _normalize_protocol_col(df): + """Handle both _ws.col.Protocol and _ws.col.protocol from different tshark versions.""" + if '_ws.col.Protocol' in df.columns: + return '_ws.col.Protocol' + if '_ws.col.protocol' in df.columns: + df.rename(columns={'_ws.col.protocol': '_ws.col.Protocol'}, inplace=True) + return '_ws.col.Protocol' + raise KeyError("No protocol column found. Expected _ws.col.Protocol or _ws.col.protocol.") + + def process_pcap_data(input_csv, output_csv): df = pd.read_csv(input_csv) df['frame.time_epoch'] = pd.to_datetime(df['frame.time_epoch'], unit='s', errors='coerce') + _normalize_protocol_col(df) df = df[df['_ws.col.Protocol'].isin(['TCP', 'TLSv1.2', 'UDP', 'TLS', 'DNS'])] # Combine ports diff --git a/pcap_parser/summary.py b/pcap_parser/summary.py index 354bfae..5ca8ddd 100644 --- a/pcap_parser/summary.py +++ b/pcap_parser/summary.py @@ -72,10 +72,16 @@ def summarize(input_csv): if "eth.src" in df.columns: device_count = df["eth.src"].nunique() - # protocol breakdown - protocols = {} + # protocol breakdown - handle both column name variants + proto_col = None if "_ws.col.Protocol" in df.columns: - protocols = df["_ws.col.Protocol"].value_counts().head(8).to_dict() + proto_col = "_ws.col.Protocol" + elif "_ws.col.protocol" in df.columns: + proto_col = "_ws.col.protocol" + + protocols = {} + if proto_col: + protocols = df[proto_col].value_counts().head(8).to_dict() # top talkers (by bytes sent) top_talkers = {} From d8c7940718d4faccfbe99904156b0ea241c3c437 Mon Sep 17 00:00:00 2001 From: Rameen Mahmood Date: Sat, 28 Mar 2026 20:49:03 -0400 Subject: [PATCH 09/11] add svg screenshots of summary and devices output --- docs/images/pcap-devices.svg | 137 +++++++++++++++++++++ docs/images/pcap-summary.svg | 228 +++++++++++++++++++++++++++++++++++ 2 files changed, 365 insertions(+) create mode 100644 docs/images/pcap-devices.svg create mode 100644 docs/images/pcap-summary.svg diff --git a/docs/images/pcap-devices.svg b/docs/images/pcap-devices.svg new file mode 100644 index 0000000..6f3fb81 --- /dev/null +++ b/docs/images/pcap-devices.svg @@ -0,0 +1,137 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + pcap-devices + + + + + + + + + +                                          Devices                                           +┏━━━━━┳━━━━━━━━━━━━┳━━━━━━━━━━━┳━━━━━━━━━━━━┳━━━━━━━━━━━┳━━━━━━━━━┳━━━━━━━━━┳━━━━━━━━━━━━┓ +┃┃MAC       ┃┃┃┃┃┃Top       ┃ +┃#  ┃Address   ┃Vendor   ┃Hostname  ┃IPs      ┃Packets┃Traffic┃Destinati…┃ +┡━━━━━╇━━━━━━━━━━━━╇━━━━━━━━━━━╇━━━━━━━━━━━━╇━━━━━━━━━━━╇━━━━━━━━━╇━━━━━━━━━╇━━━━━━━━━━━━┩ +│1  │AA:BB:CC:…│Apple,   │macbook-p…│192.168.…│     20│12.0 KB│google.co…│ +│││Inc.     │││││icloud.co…│ +││││││││example.c…│ +│2  │DD:EE:FF:…│Google,  │pixel-7   │192.168.…│     15│ 8.1 KB│youtube.c…│ +│││Inc.     │││││amazon.com│ +│3  │11:22:33:…│Amazon   │echo-dot  │192.168.…│     10│ 6.2 KB│amazonaws…│ +│││Technolo…││││││ +│││Inc.     ││││││ +│4  │44:55:66:…│Samsung  │galaxy-s23│192.168.…│      5│ 3.4 KB│cloudflar…│ +│││Electron…││││││ +└─────┴────────────┴───────────┴────────────┴───────────┴─────────┴─────────┴────────────┘ + +4 devices found + + + + + diff --git a/docs/images/pcap-summary.svg b/docs/images/pcap-summary.svg new file mode 100644 index 0000000..fd61983 --- /dev/null +++ b/docs/images/pcap-summary.svg @@ -0,0 +1,228 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + pcap-summary + + + + + + + + + + +╭──────────────────────────────Capture Overview──────────────────────────────╮ +│  Packets:    50│ +│  Traffic:    29.6 KB│ +│  Devices:    4│ +│  Time Range: 2025-01-17 03:00:12 to 2025-01-17 03:00:36 (24.5s)│ +││ +╰──────────────────────────────────────────────────────────────────────────────╯ +╭─────────────────────────────────Protocols──────────────────────────────────╮ +│┏━━━━━━━━━━┳━━━━━━━━━┳━━━━━━━┓│ +│┃Protocol┃Packets┃Share┃│ +│┡━━━━━━━━━━╇━━━━━━━━━╇━━━━━━━┩│ +││TLS     │     15│30.0%││ +││TCP     │     12│24.0%││ +││DNS     │     10│20.0%││ +││UDP     │      8│16.0%││ +││HTTP    │      5│10.0%││ +│└──────────┴─────────┴───────┘│ +╰──────────────────────────────────────────────────────────────────────────────╯ +╭────────────────────────────────Top Talkers─────────────────────────────────╮ +│┏━━━━━━━━━━━━━━┳━━━━━━━━━┓│ +│┃Source IP   ┃Traffic┃│ +│┡━━━━━━━━━━━━━━╇━━━━━━━━━┩│ +││192.168.1.10│12.0 KB││ +││192.168.1.20│ 8.1 KB││ +││192.168.1.30│ 6.2 KB││ +││192.168.1.40│ 3.4 KB││ +│└──────────────┴─────────┘│ +╰──────────────────────────────────────────────────────────────────────────────╯ +╭──────────────────────────────Top Destinations──────────────────────────────╮ +│┏━━━━━━━━━━━━━━━┳━━━━━━━━━┓│ +│┃Destination  ┃Packets┃│ +│┡━━━━━━━━━━━━━━━╇━━━━━━━━━┩│ +││google.com   │     10││ +││amazonaws.com│     10││ +││youtube.com  │      8││ +││amazon.com   │      7││ +││icloud.com   │      5││ +│└───────────────┴─────────┘│ +╰──────────────────────────────────────────────────────────────────────────────╯ + + + + + From 799686631cb0622a36efa4b0adafd3cf803e453b Mon Sep 17 00:00:00 2001 From: Rameen Mahmood Date: Sat, 28 Mar 2026 20:49:13 -0400 Subject: [PATCH 10/11] add script to regenerate readme screenshots --- docs/generate_screenshots.py | 186 +++++++++++++++++++++++++++++++++++ 1 file changed, 186 insertions(+) create mode 100644 docs/generate_screenshots.py diff --git a/docs/generate_screenshots.py b/docs/generate_screenshots.py new file mode 100644 index 0000000..83e7cf3 --- /dev/null +++ b/docs/generate_screenshots.py @@ -0,0 +1,186 @@ +"""Generate SVG screenshots of CLI output for the README.""" + +import sys +sys.path.insert(0, ".") + +from rich.console import Console + +# Generate pcap-summary screenshot +from pcap_parser.summary import summarize, print_summary + +console = Console(record=True, width=80) + +# Use a richer fake dataset for the screenshot +import pandas as pd +import tempfile, os + +# Create realistic-looking parsed CSV +data = { + "frame.time_epoch": [1737100812.0 + i * 0.5 for i in range(50)], + "eth.src": ( + ["AA:BB:CC:11:22:33"] * 20 + + ["DD:EE:FF:44:55:66"] * 15 + + ["11:22:33:AA:BB:CC"] * 10 + + ["44:55:66:DD:EE:FF"] * 5 + ), + "eth.src.oui_resolved": ( + ["Apple, Inc."] * 20 + + ["Google, Inc."] * 15 + + ["Amazon Technologies Inc."] * 10 + + ["Samsung Electronics"] * 5 + ), + "ip.src": ( + ["192.168.1.10"] * 20 + + ["192.168.1.20"] * 15 + + ["192.168.1.30"] * 10 + + ["192.168.1.40"] * 5 + ), + "ip.dst": ( + ["17.253.144.10"] * 5 + + ["142.250.80.46"] * 10 + + ["93.184.216.34"] * 5 + + ["142.250.80.46"] * 8 + + ["54.239.28.85"] * 7 + + ["54.239.28.85"] * 10 + + ["104.16.132.229"] * 5 + ), + "_ws.col.Protocol": ( + ["TLS"] * 15 + + ["TCP"] * 12 + + ["DNS"] * 10 + + ["UDP"] * 8 + + ["HTTP"] * 5 + ), + "frame.len": [ + 150, 1200, 800, 54, 300, 1400, 600, 200, 54, 1000, + 800, 1200, 150, 300, 54, 600, 1400, 200, 800, 1000, + 54, 300, 1200, 150, 800, 600, 1400, 200, 54, 1000, + 300, 800, 1200, 150, 54, 600, 1400, 200, 1000, 800, + 300, 150, 54, 1200, 600, 1400, 200, 800, 1000, 54, + ], + "dst_hostname": ( + ["icloud.com"] * 5 + + ["google.com"] * 10 + + ["example.com"] * 5 + + ["youtube.com"] * 8 + + ["amazon.com"] * 7 + + ["amazonaws.com"] * 10 + + ["cloudflare.com"] * 5 + ), + "dhcp_hostname": ( + ["macbook-pro"] * 20 + + ["pixel-7"] * 15 + + ["echo-dot"] * 10 + + ["galaxy-s23"] * 5 + ), + "http.user_agent": [None] * 50, +} + +df = pd.DataFrame(data) +with tempfile.NamedTemporaryFile(suffix=".csv", delete=False) as f: + df.to_csv(f.name, index=False) + tmp_csv = f.name + +# Capture summary +from pcap_parser.summary import summarize as _summarize +from pcap_parser.summary import ( + _format_bytes, _format_timestamp, _format_duration, +) +from rich.table import Table +from rich.panel import Panel +from rich.text import Text + +stats = _summarize(tmp_csv) + +overview = Text() +overview.append(" Packets: ", style="dim") +overview.append(f"{stats['total_packets']:,}\n", style="bold white") +overview.append(" Traffic: ", style="dim") +overview.append(f"{_format_bytes(stats['total_bytes'])}\n", style="bold magenta") +overview.append(" Devices: ", style="dim") +overview.append(f"{stats['device_count']}\n", style="bold cyan") +overview.append(" Time Range: ", style="dim") +overview.append( + f"{_format_timestamp(stats['start_ts'])} to {_format_timestamp(stats['end_ts'])}", + style="white", +) +overview.append(f" ({_format_duration(stats['duration'])})\n", style="dim") + +console.print() +console.print(Panel(overview, title="[bold]Capture Overview[/]", border_style="cyan")) + +if stats["protocols"]: + proto_table = Table(show_header=True, header_style="bold cyan", border_style="dim") + proto_table.add_column("Protocol", style="bold") + proto_table.add_column("Packets", justify="right", style="white") + proto_table.add_column("Share", justify="right", style="dim") + total = stats["total_packets"] + for proto, count in stats["protocols"].items(): + pct = f"{count / total * 100:.1f}%" + proto_table.add_row(str(proto), f"{count:,}", pct) + console.print(Panel(proto_table, title="[bold]Protocols[/]", border_style="green")) + +if stats["top_talkers"]: + talker_table = Table(show_header=True, header_style="bold cyan", border_style="dim") + talker_table.add_column("Source IP", style="bold") + talker_table.add_column("Traffic", justify="right", style="magenta") + for ip, bytes_sent in stats["top_talkers"].items(): + talker_table.add_row(str(ip), _format_bytes(bytes_sent)) + console.print(Panel(talker_table, title="[bold]Top Talkers[/]", border_style="yellow")) + +destinations = stats["top_destinations"] or stats["top_dst_ips"] +if destinations: + dest_table = Table(show_header=True, header_style="bold cyan", border_style="dim") + dest_table.add_column("Destination", style="bold") + dest_table.add_column("Packets", justify="right", style="white") + for dest, count in list(destinations.items())[:5]: + dest_table.add_row(str(dest), f"{count:,}") + console.print(Panel(dest_table, title="[bold]Top Destinations[/]", border_style="magenta")) + +console.print() +console.save_svg("docs/images/pcap-summary.svg", title="pcap-summary") +print("Saved docs/images/pcap-summary.svg") + +# Generate pcap-devices screenshot +console2 = Console(record=True, width=90) + +from pcap_parser.devices import aggregate_devices, _format_bytes as _fmt + +devices = aggregate_devices(tmp_csv) + +table = Table( + title="Devices", + show_header=True, + header_style="bold cyan", + border_style="dim", + title_style="bold white", +) +table.add_column("#", style="dim", width=3) +table.add_column("MAC Address", style="bold") +table.add_column("Vendor", style="green") +table.add_column("Hostname", style="yellow") +table.add_column("IPs", style="white") +table.add_column("Packets", justify="right", style="white") +table.add_column("Traffic", justify="right", style="magenta") +table.add_column("Top Destinations", style="cyan") + +for i, device in enumerate(devices): + ips = ", ".join(device["ips"][:3]) + destinations = ", ".join(device["top_destinations"][:3]) + table.add_row( + str(i + 1), + device["mac"], + device["oui_vendor"] or "-", + device["dhcp_hostname"] or "-", + ips or "-", + f"{device['packet_count']:,}", + _fmt(device["byte_count"]), + destinations or "-", + ) + +console2.print(table) +console2.print(f"\n [bold]{len(devices)}[/] devices found\n") +console2.save_svg("docs/images/pcap-devices.svg", title="pcap-devices") +print("Saved docs/images/pcap-devices.svg") + +os.unlink(tmp_csv) From 81373988515f800d5fc8addf19d57d1e23d11ac4 Mon Sep 17 00:00:00 2001 From: Rameen Mahmood Date: Sat, 28 Mar 2026 20:49:26 -0400 Subject: [PATCH 11/11] update readme with svg screenshots --- README.md | 34 ++++++---------------------------- 1 file changed, 6 insertions(+), 28 deletions(-) diff --git a/README.md b/README.md index c30a462..e182b8b 100644 --- a/README.md +++ b/README.md @@ -19,24 +19,9 @@ pcap-parse output.csv capture.pcap pcap-summary output.csv ``` -``` -╭──────────────────────────── Capture Overview ────────────────────────────╮ -│ Packets: 12,847 │ -│ Traffic: 8.3 MB │ -│ Devices: 14 │ -│ Time Range: 2025-01-15 09:00:12 to 2025-01-15 09:30:45 (30.6m) │ -╰──────────────────────────────────────────────────────────────────────────╯ -╭──────────────────────────────── Protocols ────────────────────────────────╮ -│ ┏━━━━━━━━━━┳━━━━━━━━━┳━━━━━━━┓ │ -│ ┃ Protocol ┃ Packets ┃ Share ┃ │ -│ ┡━━━━━━━━━━╇━━━━━━━━━╇━━━━━━━┩ │ -│ │ TCP │ 8,421 │ 65.5% │ │ -│ │ TLS │ 2,103 │ 16.4% │ │ -│ │ DNS │ 1,547 │ 12.0% │ │ -│ │ UDP │ 776 │ 6.0% │ │ -│ └──────────┴─────────┴───────┘ │ -╰──────────────────────────────────────────────────────────────────────────╯ -``` +

+ pcap-summary output +

List devices on the network: @@ -44,16 +29,9 @@ List devices on the network: pcap-devices output.csv ``` -``` - Devices -┏━━━┳━━━━━━━━━━━━━━━━━━━┳━━━━━━━━━━━━━━┳━━━━━━━━━━━━━┳━━━━━━━━━┳━━━━━━━━━┓ -┃ # ┃ MAC Address ┃ Vendor ┃ Hostname ┃ Packets ┃ Traffic ┃ -┡━━━╇━━━━━━━━━━━━━━━━━━━╇━━━━━━━━━━━━━━╇━━━━━━━━━━━━━╇━━━━━━━━━╇━━━━━━━━━┩ -│ 1 │ AA:BB:CC:11:22:33 │ Apple, Inc. │ macbook-pro │ 4,521 │ 3.2 MB │ -│ 2 │ AA:BB:CC:44:55:66 │ Google, Inc. │ pixel-6 │ 2,847 │ 1.8 MB │ -│ 3 │ AA:BB:CC:77:88:99 │ Amazon.com │ echo-dot │ 892 │ 412.0 KB│ -└───┴───────────────────┴──────────────┴─────────────┴─────────┴─────────┘ -``` +

+ pcap-devices output +

Identify unknown devices with a fine-tuned LLM: