diff options
| author | Ian Rogers <irogers@google.com> | 2026-09-25 23:19:54 -0700 |
|---|---|---|
| committer | Arnaldo Carvalho de Melo <acme@redhat.com> | 2026-09-26 20:49:29 +0200 |
| commit | 47b00c38780d27bedb91b8090f74e0523ccbc95e (patch) | |
| tree | 3c864f89bb2809a398e79114400d0bb35ccef3f5 /tools/perf/python | |
| parent | b83f0bacf5e4f9369257f5c1191df02ab1be5602 (diff) | |
| download | linux-next-47b00c38780d27bedb91b8090f74e0523ccbc95e.tar.gz linux-next-47b00c38780d27bedb91b8090f74e0523ccbc95e.zip | |
perf python: Port rw-by-file from Perl to perf module
Replace the legacy Perl script rw-by-file.pl with a standalone Python
script in tools/perf/python/rw-by-file.py using the perf Python module.
Improvements compared to the legacy Perl script:
- Remove the dependency on libperl and Perf::Trace::Util.
- Encapsulate per-file-descriptor read/write byte and call count
aggregation in an RwByFile class using perf.session and resolve thread
command names via session.find_thread(pid, sample_tid).
- Add argparse CLI support (-i/--input and target program filter) and
full type annotations.
Add a shell test
(test_rw_by_file_python.sh) to verify the standalone script.
Assisted-by: Antigravity:gemini-3.1-pro
Signed-off-by: Ian Rogers <irogers@google.com>
Signed-off-by: Arnaldo Carvalho de Melo <acme@redhat.com>
Diffstat (limited to 'tools/perf/python')
| -rwxr-xr-x | tools/perf/python/rw-by-file.py | 113 |
1 files changed, 113 insertions, 0 deletions
diff --git a/tools/perf/python/rw-by-file.py b/tools/perf/python/rw-by-file.py new file mode 100755 index 000000000000..b33330261606 --- /dev/null +++ b/tools/perf/python/rw-by-file.py @@ -0,0 +1,113 @@ +#!/usr/bin/env python3 +# SPDX-License-Identifier: GPL-2.0-only +"""Display r/w activity for files read/written to for a given program.""" +from __future__ import annotations + +import argparse +from collections import defaultdict +import sys +from typing import Optional, Dict +import perf + +class RwByFile: + """Tracks and displays read/write activity by file descriptor.""" + def __init__(self, comm: str) -> None: + self.for_comm = comm + self.reads: Dict[int, Dict[str, int]] = defaultdict( + lambda: {"bytes_requested": 0, "total_reads": 0} + ) + self.writes: Dict[int, Dict[str, int]] = defaultdict( + lambda: {"bytes_written": 0, "total_writes": 0} + ) + self.unhandled: Dict[str, int] = defaultdict(int) + self.session: Optional[perf.session] = None + + def process_event(self, sample: perf.sample_event) -> None: + """Process events.""" + event_name = str(sample.evsel) + if event_name.startswith("evsel(") and event_name.endswith(")"): + event_name = event_name[6:-1] + event_name = "".join(c if c.isprintable() else "?" for c in event_name) + + pid = sample.sample_pid + assert self.session is not None + try: + thread = self.session.find_thread(pid, sample.sample_tid) + comm = (thread.comm() if thread else None) or "unknown" + except (TypeError, AttributeError): + comm = "unknown" + + if comm != self.for_comm: + return + + if event_name == "syscalls:sys_enter_read": + try: + fd = sample.fd + count = sample.count + self.reads[fd]["bytes_requested"] += count + self.reads[fd]["total_reads"] += 1 + except AttributeError: + self.unhandled[event_name] += 1 + elif event_name == "syscalls:sys_enter_write": + try: + fd = sample.fd + count = sample.count + self.writes[fd]["bytes_written"] += count + self.writes[fd]["total_writes"] += 1 + except AttributeError: + self.unhandled[event_name] += 1 + else: + self.unhandled[event_name] += 1 + + def print_totals(self) -> None: + """Print summary tables.""" + print(f"file read counts for {self.for_comm}:\n") + print(f"{'fd':>6s} {'# reads':>10s} {'bytes_requested':>15s}") + print(f"{'-'*6} {'-'*10} {'-'*15}") + + for fd, data in sorted(self.reads.items(), + key=lambda kv: kv[1]["bytes_requested"], reverse=True): + print(f"{fd:6d} {data['total_reads']:10d} {data['bytes_requested']:15d}") + + print(f"\nfile write counts for {self.for_comm}:\n") + print(f"{'fd':>6s} {'# writes':>10s} {'bytes_written':>15s}") + print(f"{'-'*6} {'-'*10} {'-'*15}") + + for fd, data in sorted(self.writes.items(), + key=lambda kv: kv[1]["bytes_written"], reverse=True): + print(f"{fd:6d} {data['total_writes']:10d} {data['bytes_written']:15d}") + + if self.unhandled: + print("\nunhandled events:\n") + print(f"{'event':<40s} {'count':>10s}") + print(f"{'-'*40} {'-'*10}") + for event_name, count in self.unhandled.items(): + print(f"{event_name:<40s} {count:10d}") + + def run(self, input_file: str) -> None: + """Run the session.""" + self.session = perf.session(perf.data(input_file), sample=self.process_event) + try: + self.session.process_events() + finally: + # Break the reference cycle between perf.session and self.process_event + # because perf.session lacks cyclic GC support (tp_traverse). + self.session = None + self.print_totals() + +def main() -> None: + """Main function.""" + parser = argparse.ArgumentParser(description="Trace r/w activity by file") + parser.add_argument("comm", help="Filter by command name") + parser.add_argument("-i", "--input", default="perf.data", help="Input file") + args = parser.parse_args() + + analyzer = RwByFile(args.comm) + try: + analyzer.run(args.input) + except IOError as e: + print(e, file=sys.stderr) + sys.exit(1) + +if __name__ == "__main__": + main() |
