summaryrefslogtreecommitdiff
path: root/tools/perf/python
diff options
context:
space:
mode:
authorIan Rogers <irogers@google.com>2026-09-25 23:19:54 -0700
committerArnaldo Carvalho de Melo <acme@redhat.com>2026-09-26 20:49:29 +0200
commit47b00c38780d27bedb91b8090f74e0523ccbc95e (patch)
tree3c864f89bb2809a398e79114400d0bb35ccef3f5 /tools/perf/python
parentb83f0bacf5e4f9369257f5c1191df02ab1be5602 (diff)
downloadlinux-next-47b00c38780d27bedb91b8090f74e0523ccbc95e.tar.gz
linux-next-47b00c38780d27bedb91b8090f74e0523ccbc95e.zip
perf python: Port rw-by-file from Perl to perf module
Replace the legacy Perl script rw-by-file.pl with a standalone Python script in tools/perf/python/rw-by-file.py using the perf Python module. Improvements compared to the legacy Perl script: - Remove the dependency on libperl and Perf::Trace::Util. - Encapsulate per-file-descriptor read/write byte and call count aggregation in an RwByFile class using perf.session and resolve thread command names via session.find_thread(pid, sample_tid). - Add argparse CLI support (-i/--input and target program filter) and full type annotations. Add a shell test (test_rw_by_file_python.sh) to verify the standalone script. Assisted-by: Antigravity:gemini-3.1-pro Signed-off-by: Ian Rogers <irogers@google.com> Signed-off-by: Arnaldo Carvalho de Melo <acme@redhat.com>
Diffstat (limited to 'tools/perf/python')
-rwxr-xr-xtools/perf/python/rw-by-file.py113
1 files changed, 113 insertions, 0 deletions
diff --git a/tools/perf/python/rw-by-file.py b/tools/perf/python/rw-by-file.py
new file mode 100755
index 000000000000..b33330261606
--- /dev/null
+++ b/tools/perf/python/rw-by-file.py
@@ -0,0 +1,113 @@
+#!/usr/bin/env python3
+# SPDX-License-Identifier: GPL-2.0-only
+"""Display r/w activity for files read/written to for a given program."""
+from __future__ import annotations
+
+import argparse
+from collections import defaultdict
+import sys
+from typing import Optional, Dict
+import perf
+
+class RwByFile:
+ """Tracks and displays read/write activity by file descriptor."""
+ def __init__(self, comm: str) -> None:
+ self.for_comm = comm
+ self.reads: Dict[int, Dict[str, int]] = defaultdict(
+ lambda: {"bytes_requested": 0, "total_reads": 0}
+ )
+ self.writes: Dict[int, Dict[str, int]] = defaultdict(
+ lambda: {"bytes_written": 0, "total_writes": 0}
+ )
+ self.unhandled: Dict[str, int] = defaultdict(int)
+ self.session: Optional[perf.session] = None
+
+ def process_event(self, sample: perf.sample_event) -> None:
+ """Process events."""
+ event_name = str(sample.evsel)
+ if event_name.startswith("evsel(") and event_name.endswith(")"):
+ event_name = event_name[6:-1]
+ event_name = "".join(c if c.isprintable() else "?" for c in event_name)
+
+ pid = sample.sample_pid
+ assert self.session is not None
+ try:
+ thread = self.session.find_thread(pid, sample.sample_tid)
+ comm = (thread.comm() if thread else None) or "unknown"
+ except (TypeError, AttributeError):
+ comm = "unknown"
+
+ if comm != self.for_comm:
+ return
+
+ if event_name == "syscalls:sys_enter_read":
+ try:
+ fd = sample.fd
+ count = sample.count
+ self.reads[fd]["bytes_requested"] += count
+ self.reads[fd]["total_reads"] += 1
+ except AttributeError:
+ self.unhandled[event_name] += 1
+ elif event_name == "syscalls:sys_enter_write":
+ try:
+ fd = sample.fd
+ count = sample.count
+ self.writes[fd]["bytes_written"] += count
+ self.writes[fd]["total_writes"] += 1
+ except AttributeError:
+ self.unhandled[event_name] += 1
+ else:
+ self.unhandled[event_name] += 1
+
+ def print_totals(self) -> None:
+ """Print summary tables."""
+ print(f"file read counts for {self.for_comm}:\n")
+ print(f"{'fd':>6s} {'# reads':>10s} {'bytes_requested':>15s}")
+ print(f"{'-'*6} {'-'*10} {'-'*15}")
+
+ for fd, data in sorted(self.reads.items(),
+ key=lambda kv: kv[1]["bytes_requested"], reverse=True):
+ print(f"{fd:6d} {data['total_reads']:10d} {data['bytes_requested']:15d}")
+
+ print(f"\nfile write counts for {self.for_comm}:\n")
+ print(f"{'fd':>6s} {'# writes':>10s} {'bytes_written':>15s}")
+ print(f"{'-'*6} {'-'*10} {'-'*15}")
+
+ for fd, data in sorted(self.writes.items(),
+ key=lambda kv: kv[1]["bytes_written"], reverse=True):
+ print(f"{fd:6d} {data['total_writes']:10d} {data['bytes_written']:15d}")
+
+ if self.unhandled:
+ print("\nunhandled events:\n")
+ print(f"{'event':<40s} {'count':>10s}")
+ print(f"{'-'*40} {'-'*10}")
+ for event_name, count in self.unhandled.items():
+ print(f"{event_name:<40s} {count:10d}")
+
+ def run(self, input_file: str) -> None:
+ """Run the session."""
+ self.session = perf.session(perf.data(input_file), sample=self.process_event)
+ try:
+ self.session.process_events()
+ finally:
+ # Break the reference cycle between perf.session and self.process_event
+ # because perf.session lacks cyclic GC support (tp_traverse).
+ self.session = None
+ self.print_totals()
+
+def main() -> None:
+ """Main function."""
+ parser = argparse.ArgumentParser(description="Trace r/w activity by file")
+ parser.add_argument("comm", help="Filter by command name")
+ parser.add_argument("-i", "--input", default="perf.data", help="Input file")
+ args = parser.parse_args()
+
+ analyzer = RwByFile(args.comm)
+ try:
+ analyzer.run(args.input)
+ except IOError as e:
+ print(e, file=sys.stderr)
+ sys.exit(1)
+
+if __name__ == "__main__":
+ main()