codekingpro/portable-devtools
114k
1"""Print a summary of specialization stats for all files in the2default stats folders.3"""4 5from __future__ import annotations6 7# NOTE: Bytecode introspection modules (opcode, dis, etc.) should only8# be imported when loading a single dataset. When comparing datasets, it9# could get it wrong, leading to subtle errors.10 11import argparse12import collections13from collections.abc import KeysView14from dataclasses import dataclass15from datetime import date16import enum17import functools18import itertools19import json20from operator import itemgetter21import os22from pathlib import Path23import re24import sys25import textwrap26from typing import Any, Callable, TextIO, TypeAlias27 28 29RawData: TypeAlias = dict[str, Any]30Rows: TypeAlias = list[tuple]31Columns: TypeAlias = tuple[str, ...]32RowCalculator: TypeAlias = Callable[["Stats"], Rows]33 34 35# TODO: Check for parity36 37 38if os.name == "nt":39 DEFAULT_DIR = "c:\\temp\\py_stats\\"40else:41 DEFAULT_DIR = "/tmp/py_stats/"42 43 44SOURCE_DIR = Path(__file__).parents[2]45 46 47TOTAL = "specialization.hit", "specialization.miss", "execution_count"48 49 50def pretty(name: str) -> str:51 return name.replace("_", " ").lower()52 53 54def _load_metadata_from_source():55 def get_defines(filepath: Path, prefix: str = "SPEC_FAIL"):56 with open(SOURCE_DIR / filepath) as spec_src:57 defines = collections.defaultdict(list)58 start = "#define " + prefix + "_"59 for line in spec_src:60 line = line.strip()61 if not line.startswith(start):62 continue63 line = line[len(start) :]64 name, val = line.split()65 defines[int(val.strip())].append(name.strip())66 return defines67 68 import opcode69 70 return {71 "_specialized_instructions": [72 op for op in opcode._specialized_opmap.keys() if "__" not in op # type: ignore73 ],74 "_stats_defines": get_defines(75 Path("Include") / "cpython" / "pystats.h", "EVAL_CALL"76 ),77 "_defines": get_defines(Path("Python") / "specialize.c"),78 }79 80 81def load_raw_data(input: Path) -> RawData:82 if input.is_file():83 with open(input, "r") as fd:84 data = json.load(fd)85 86 data["_stats_defines"] = {int(k): v for k, v in data["_stats_defines"].items()}87 data["_defines"] = {int(k): v for k, v in data["_defines"].items()}88 89 return data90 91 elif input.is_dir():92 stats = collections.Counter[str]()93 94 for filename in input.iterdir():95 with open(filename) as fd:96 for line in fd:97 try:98 key, value = line.split(":")99 except ValueError:100 print(101 f"Unparsable line: '{line.strip()}' in {filename}",102 file=sys.stderr,103 )104 continue105 # Hack to handle older data files where some uops106 # are missing an underscore prefix in their name107 if key.startswith("uops[") and key[5:6] != "_":108 key = "uops[_" + key[5:]109 stats[key.strip()] += int(value)110 stats["__nfiles__"] += 1111 112 data = dict(stats)113 data.update(_load_metadata_from_source())114 return data115 116 else:117 raise ValueError(f"{input} is not a file or directory path")118 119 120def save_raw_data(data: RawData, json_output: TextIO):121 json.dump(data, json_output)122 123 124@dataclass(frozen=True)125class Doc:126 text: str127 doc: str128 129 def markdown(self) -> str:130 return textwrap.dedent(131 f"""132 {self.text}133 <details>134 <summary>ⓘ</summary>135 136 {self.doc}137 </details>138 """139 )140 141 142class Count(int):143 def markdown(self) -> str:144 return format(self, ",d")145 146 147@dataclass(frozen=True)148class Ratio:149 num: int150 den: int | None = None151 percentage: bool = True152 153 def __float__(self):154 if self.den == 0:155 return 0.0156 elif self.den is None:157 return self.num158 else:159 return self.num / self.den160 161 def markdown(self) -> str:162 if self.den is None:163 return ""164 elif self.den == 0:165 if self.num != 0:166 return f"{self.num:,} / 0 !!"167 return ""168 elif self.percentage:169 return f"{self.num / self.den:,.01%}"170 else:171 return f"{self.num / self.den:,.02f}"172 173 174class DiffRatio(Ratio):175 def __init__(self, base: int | str, head: int | str):176 if isinstance(base, str) or isinstance(head, str):177 super().__init__(0, 0)178 else:179 super().__init__(head - base, base)180 181 182class OpcodeStats:183 """184 Manages the data related to specific set of opcodes, e.g. tier1 (with prefix185 "opcode") or tier2 (with prefix "uops").186 """187 188 def __init__(self, data: dict[str, Any], defines, specialized_instructions):189 self._data = data190 self._defines = defines191 self._specialized_instructions = specialized_instructions192 193 def get_opcode_names(self) -> KeysView[str]:194 return self._data.keys()195 196 def get_pair_counts(self) -> dict[tuple[str, str], int]:197 pair_counts = {}198 for name_i, opcode_stat in self._data.items():199 for key, value in opcode_stat.items():200 if value and key.startswith("pair_count"):201 name_j, _, _ = key[len("pair_count") + 1 :].partition("]")202 pair_counts[(name_i, name_j)] = value203 return pair_counts204 205 def get_total_execution_count(self) -> int:206 return sum(x.get("execution_count", 0) for x in self._data.values())207 208 def get_execution_counts(self) -> dict[str, tuple[int, int]]:209 counts = {}210 for name, opcode_stat in self._data.items():211 if "execution_count" in opcode_stat:212 count = opcode_stat["execution_count"]213 miss = 0214 if "specializable" not in opcode_stat:215 miss = opcode_stat.get("specialization.miss", 0)216 counts[name] = (count, miss)217 return counts218 219 @functools.cache220 def _get_pred_succ(221 self,222 ) -> tuple[dict[str, collections.Counter], dict[str, collections.Counter]]:223 pair_counts = self.get_pair_counts()224 225 predecessors: dict[str, collections.Counter] = collections.defaultdict(226 collections.Counter227 )228 successors: dict[str, collections.Counter] = collections.defaultdict(229 collections.Counter230 )231 for (first, second), count in pair_counts.items():232 if count:233 predecessors[second][first] = count234 successors[first][second] = count235 236 return predecessors, successors237 238 def get_predecessors(self, opcode: str) -> collections.Counter[str]:239 return self._get_pred_succ()[0][opcode]240 241 def get_successors(self, opcode: str) -> collections.Counter[str]:242 return self._get_pred_succ()[1][opcode]243 244 def _get_stats_for_opcode(self, opcode: str) -> dict[str, int]:245 return self._data[opcode]246 247 def get_specialization_total(self, opcode: str) -> int:248 family_stats = self._get_stats_for_opcode(opcode)249 return sum(family_stats.get(kind, 0) for kind in TOTAL)250 251 def get_specialization_counts(self, opcode: str) -> dict[str, int]:252 family_stats = self._get_stats_for_opcode(opcode)253 254 result = {}255 for key, value in sorted(family_stats.items()):256 if key.startswith("specialization."):257 label = key[len("specialization.") :]258 if label in ("success", "failure") or label.startswith("failure_kinds"):259 continue260 elif key in (261 "execution_count",262 "specializable",263 ) or key.startswith("pair"):264 continue265 else:266 label = key267 result[label] = value268 269 return result270 271 def get_specialization_success_failure(self, opcode: str) -> dict[str, int]:272 family_stats = self._get_stats_for_opcode(opcode)273 result = {}274 for key in ("specialization.success", "specialization.failure"):275 label = key[len("specialization.") :]276 val = family_stats.get(key, 0)277 result[label] = val278 return result279 280 def get_specialization_failure_total(self, opcode: str) -> int:281 return self._get_stats_for_opcode(opcode).get("specialization.failure", 0)282 283 def get_specialization_failure_kinds(self, opcode: str) -> dict[str, int]:284 def kind_to_text(kind: int, opcode: str):285 if kind <= 8:286 return pretty(self._defines[kind][0])287 if opcode == "LOAD_SUPER_ATTR":288 opcode = "SUPER"289 elif opcode.endswith("ATTR"):290 opcode = "ATTR"291 elif opcode in ("FOR_ITER", "GET_ITER", "SEND"):292 opcode = "ITER"293 elif opcode.endswith("SUBSCR"):294 opcode = "SUBSCR"295 for name in self._defines[kind]:296 if name.startswith(opcode):297 return pretty(name[len(opcode) + 1 :])298 return "kind " + str(kind)299 300 family_stats = self._get_stats_for_opcode(opcode)301 302 def key_to_index(key):303 return int(key[:-1].split("[")[1])304 305 max_index = 0306 for key in family_stats:307 if key.startswith("specialization.failure_kind"):308 max_index = max(max_index, key_to_index(key))309 310 failure_kinds = [0] * (max_index + 1)311 for key in family_stats:312 if not key.startswith("specialization.failure_kind"):313 continue314 failure_kinds[key_to_index(key)] = family_stats[key]315 return {316 kind_to_text(index, opcode): value317 for (index, value) in enumerate(failure_kinds)318 if value319 }320 321 def is_specializable(self, opcode: str) -> bool:322 return "specializable" in self._get_stats_for_opcode(opcode)323 324 def get_specialized_total_counts(self) -> tuple[int, int, int]:325 basic = 0326 specialized_hits = 0327 specialized_misses = 0328 not_specialized = 0329 for opcode, opcode_stat in self._data.items():330 if "execution_count" not in opcode_stat:331 continue332 count = opcode_stat["execution_count"]333 if "specializable" in opcode_stat:334 not_specialized += count335 elif opcode in self._specialized_instructions:336 miss = opcode_stat.get("specialization.miss", 0)337 specialized_hits += count - miss338 specialized_misses += miss339 else:340 basic += count341 return basic, specialized_hits, specialized_misses, not_specialized342 343 def get_deferred_counts(self) -> dict[str, int]:344 return {345 opcode: opcode_stat.get("specialization.deferred", 0)346 for opcode, opcode_stat in self._data.items()347 if opcode != "RESUME"348 }349 350 def get_misses_counts(self) -> dict[str, int]:351 return {352 opcode: opcode_stat.get("specialization.miss", 0)353 for opcode, opcode_stat in self._data.items()354 if not self.is_specializable(opcode)355 }356 357 def get_opcode_counts(self) -> dict[str, int]:358 counts = {}359 for opcode, entry in self._data.items():360 count = entry.get("count", 0)361 if count:362 counts[opcode] = count363 return counts364 365 366class Stats:367 def __init__(self, data: RawData):368 self._data = data369 370 def get(self, key: str) -> int:371 return self._data.get(key, 0)372 373 @functools.cache374 def get_opcode_stats(self, prefix: str) -> OpcodeStats:375 opcode_stats = collections.defaultdict[str, dict](dict)376 for key, value in self._data.items():377 if not key.startswith(prefix):378 continue379 name, _, rest = key[len(prefix) + 1 :].partition("]")380 opcode_stats[name][rest.strip(".")] = value381 return OpcodeStats(382 opcode_stats,383 self._data["_defines"],384 self._data["_specialized_instructions"],385 )386 387 def get_call_stats(self) -> dict[str, int]:388 defines = self._data["_stats_defines"]389 result = {}390 for key, value in sorted(self._data.items()):391 if "Calls to" in key:392 result[key] = value393 elif key.startswith("Calls "):394 name, index = key[:-1].split("[")395 label = f"{name} ({pretty(defines[int(index)][0])})"396 result[label] = value397 398 for key, value in sorted(self._data.items()):399 if key.startswith("Frame"):400 result[key] = value401 402 return result403 404 def get_object_stats(self) -> dict[str, tuple[int, int]]:405 total_materializations = self._data.get("Object inline values", 0)406 total_allocations = self._data.get("Object allocations", 0) + self._data.get(407 "Object allocations from freelist", 0408 )409 total_increfs = (410 self._data.get("Object interpreter mortal increfs", 0) +411 self._data.get("Object mortal increfs", 0) +412 self._data.get("Object interpreter immortal increfs", 0) +413 self._data.get("Object immortal increfs", 0)414 )415 total_decrefs = (416 self._data.get("Object interpreter mortal decrefs", 0) +417 self._data.get("Object mortal decrefs", 0) +418 self._data.get("Object interpreter immortal decrefs", 0) +419 self._data.get("Object immortal decrefs", 0)420 )421 422 result = {}423 for key, value in self._data.items():424 if key.startswith("Object"):425 if "materialize" in key:426 den = total_materializations427 elif "allocations" in key:428 den = total_allocations429 elif "increfs" in key:430 den = total_increfs431 elif "decrefs" in key:432 den = total_decrefs433 else:434 den = None435 label = key[6:].strip()436 label = label[0].upper() + label[1:]437 result[label] = (value, den)438 return result439 440 def get_gc_stats(self) -> list[dict[str, int]]:441 gc_stats: list[dict[str, int]] = []442 for key, value in self._data.items():443 if not key.startswith("GC"):444 continue445 n, _, rest = key[3:].partition("]")446 name = rest.strip()447 gen_n = int(n)448 while len(gc_stats) <= gen_n:449 gc_stats.append({})450 gc_stats[gen_n][name] = value451 return gc_stats452 453 def get_optimization_stats(self) -> dict[str, tuple[int, int | None]]:454 if "Optimization attempts" not in self._data:455 return {}456 457 attempts = self._data["Optimization attempts"]458 created = self._data["Optimization traces created"]459 executed = self._data["Optimization traces executed"]460 uops = self._data["Optimization uops executed"]461 trace_stack_overflow = self._data["Optimization trace stack overflow"]462 trace_stack_underflow = self._data["Optimization trace stack underflow"]463 trace_too_long = self._data["Optimization trace too long"]464 trace_too_short = self._data["Optimization trace too short"]465 inner_loop = self._data["Optimization inner loop"]466 recursive_call = self._data["Optimization recursive call"]467 low_confidence = self._data["Optimization low confidence"]468 unknown_callee = self._data["Optimization unknown callee"]469 executors_invalidated = self._data["Executors invalidated"]470 471 return {472 Doc(473 "Optimization attempts",474 "The number of times a potential trace is identified. Specifically, this "475 "occurs in the JUMP BACKWARD instruction when the counter reaches a "476 "threshold.",477 ): (attempts, None),478 Doc(479 "Traces created", "The number of traces that were successfully created."480 ): (created, attempts),481 Doc(482 "Trace stack overflow",483 "A trace is truncated because it would require more than 5 stack frames.",484 ): (trace_stack_overflow, attempts),485 Doc(486 "Trace stack underflow",487 "A potential trace is abandoned because it pops more frames than it pushes.",488 ): (trace_stack_underflow, attempts),489 Doc(490 "Trace too long",491 "A trace is truncated because it is longer than the instruction buffer.",492 ): (trace_too_long, attempts),493 Doc(494 "Trace too short",495 "A potential trace is abandoned because it is too short.",496 ): (trace_too_short, attempts),497 Doc(498 "Inner loop found", "A trace is truncated because it has an inner loop"499 ): (inner_loop, attempts),500 Doc(501 "Recursive call",502 "A trace is truncated because it has a recursive call.",503 ): (recursive_call, attempts),504 Doc(505 "Low confidence",506 "A trace is abandoned because the likelihood of the jump to top being taken "507 "is too low.",508 ): (low_confidence, attempts),509 Doc(510 "Unknown callee",511 "A trace is abandoned because the target of a call is unknown.",512 ): (unknown_callee, attempts),513 Doc(514 "Executors invalidated",515 "The number of executors that were invalidated due to watched "516 "dictionary changes.",517 ): (executors_invalidated, created),518 Doc("Traces executed", "The number of traces that were executed"): (519 executed,520 None,521 ),522 Doc(523 "Uops executed",524 "The total number of uops (micro-operations) that were executed",525 ): (526 uops,527 executed,528 ),529 }530 531 def get_optimizer_stats(self) -> dict[str, tuple[int, int | None]]:532 attempts = self._data["Optimization optimizer attempts"]533 successes = self._data["Optimization optimizer successes"]534 no_memory = self._data["Optimization optimizer failure no memory"]535 builtins_changed = self._data["Optimizer remove globals builtins changed"]536 incorrect_keys = self._data["Optimizer remove globals incorrect keys"]537 538 return {539 Doc(540 "Optimizer attempts",541 "The number of times the trace optimizer (_Py_uop_analyze_and_optimize) was run.",542 ): (attempts, None),543 Doc(544 "Optimizer successes",545 "The number of traces that were successfully optimized.",546 ): (successes, attempts),547 Doc(548 "Optimizer no memory",549 "The number of optimizations that failed due to no memory.",550 ): (no_memory, attempts),551 Doc(552 "Remove globals builtins changed",553 "The builtins changed during optimization",554 ): (builtins_changed, attempts),555 Doc(556 "Remove globals incorrect keys",557 "The keys in the globals dictionary aren't what was expected",558 ): (incorrect_keys, attempts),559 }560 561 def get_jit_memory_stats(self) -> dict[Doc, tuple[int, int | None]]:562 jit_total_memory_size = self._data["JIT total memory size"]563 jit_code_size = self._data["JIT code size"]564 jit_trampoline_size = self._data["JIT trampoline size"]565 jit_data_size = self._data["JIT data size"]566 jit_padding_size = self._data["JIT padding size"]567 jit_freed_memory_size = self._data["JIT freed memory size"]568 569 return {570 Doc(571 "Total memory size",572 "The total size of the memory allocated for the JIT traces",573 ): (jit_total_memory_size, None),574 Doc(575 "Code size",576 "The size of the memory allocated for the code of the JIT traces",577 ): (jit_code_size, jit_total_memory_size),578 Doc(579 "Trampoline size",580 "The size of the memory allocated for the trampolines of the JIT traces",581 ): (jit_trampoline_size, jit_total_memory_size),582 Doc(583 "Data size",584 "The size of the memory allocated for the data of the JIT traces",585 ): (jit_data_size, jit_total_memory_size),586 Doc(587 "Padding size",588 "The size of the memory allocated for the padding of the JIT traces",589 ): (jit_padding_size, jit_total_memory_size),590 Doc(591 "Freed memory size",592 "The size of the memory freed from the JIT traces",593 ): (jit_freed_memory_size, jit_total_memory_size),594 }595 596 def get_histogram(self, prefix: str) -> list[tuple[int, int]]:597 rows = []598 for k, v in self._data.items():599 match = re.match(f"{prefix}\\[([0-9]+)\\]", k)600 if match is not None:601 entry = int(match.groups()[0])602 rows.append((entry, v))603 rows.sort()604 return rows605 606 def get_rare_events(self) -> list[tuple[str, int]]:607 prefix = "Rare event "608 return [609 (key[len(prefix) + 1 : -1].replace("_", " "), val)610 for key, val in self._data.items()611 if key.startswith(prefix)612 ]613 614 615class JoinMode(enum.Enum):616 # Join using the first column as a key617 SIMPLE = 0618 # Join using the first column as a key, and indicate the change in the619 # second column of each input table as a new column620 CHANGE = 1621 # Join using the first column as a key, indicating the change in the second622 # column of each input table as a new column, and omit all other columns623 CHANGE_ONE_COLUMN = 2624 # Join using the first column as a key, and indicate the change as a new625 # column, but don't sort by the amount of change.626 CHANGE_NO_SORT = 3627 628 629class Table:630 """631 A Table defines how to convert a set of Stats into a specific set of rows632 displaying some aspect of the data.633 """634 635 def __init__(636 self,637 column_names: Columns,638 calc_rows: RowCalculator,639 join_mode: JoinMode = JoinMode.SIMPLE,640 ):641 self.columns = column_names642 self.calc_rows = calc_rows643 self.join_mode = join_mode644 645 def join_row(self, key: str, row_a: tuple, row_b: tuple) -> tuple:646 match self.join_mode:647 case JoinMode.SIMPLE:648 return (key, *row_a, *row_b)649 case JoinMode.CHANGE | JoinMode.CHANGE_NO_SORT:650 return (key, *row_a, *row_b, DiffRatio(row_a[0], row_b[0]))651 case JoinMode.CHANGE_ONE_COLUMN:652 return (key, row_a[0], row_b[0], DiffRatio(row_a[0], row_b[0]))653 654 def join_columns(self, columns: Columns) -> Columns:655 match self.join_mode:656 case JoinMode.SIMPLE:657 return (658 columns[0],659 *("Base " + x for x in columns[1:]),660 *("Head " + x for x in columns[1:]),661 )662 case JoinMode.CHANGE | JoinMode.CHANGE_NO_SORT:663 return (664 columns[0],665 *("Base " + x for x in columns[1:]),666 *("Head " + x for x in columns[1:]),667 ) + ("Change:",)668 case JoinMode.CHANGE_ONE_COLUMN:669 return (670 columns[0],671 "Base " + columns[1],672 "Head " + columns[1],673 "Change:",674 )675 676 def join_tables(self, rows_a: Rows, rows_b: Rows) -> tuple[Columns, Rows]:677 ncols = len(self.columns)678 679 default = ("",) * (ncols - 1)680 data_a = {x[0]: x[1:] for x in rows_a}681 data_b = {x[0]: x[1:] for x in rows_b}682 683 if len(data_a) != len(rows_a) or len(data_b) != len(rows_b):684 raise ValueError("Duplicate keys")685 686 # To preserve ordering, use A's keys as is and then add any in B that687 # aren't in A688 keys = list(data_a.keys()) + [k for k in data_b.keys() if k not in data_a]689 rows = [690 self.join_row(k, data_a.get(k, default), data_b.get(k, default))691 for k in keys692 ]693 if self.join_mode in (JoinMode.CHANGE, JoinMode.CHANGE_ONE_COLUMN):694 rows.sort(key=lambda row: abs(float(row[-1])), reverse=True)695 696 columns = self.join_columns(self.columns)697 return columns, rows698 699 def get_table(700 self, base_stats: Stats, head_stats: Stats | None = None701 ) -> tuple[Columns, Rows]:702 if head_stats is None:703 rows = self.calc_rows(base_stats)704 return self.columns, rows705 else:706 rows_a = self.calc_rows(base_stats)707 rows_b = self.calc_rows(head_stats)708 cols, rows = self.join_tables(rows_a, rows_b)709 return cols, rows710 711 712class Section:713 """714 A Section defines a section of the output document.715 """716 717 def __init__(718 self,719 title: str = "",720 summary: str = "",721 part_iter=None,722 *,723 comparative: bool = True,724 doc: str = "",725 ):726 self.title = title727 if not summary:728 self.summary = title.lower()729 else:730 self.summary = summary731 self.doc = textwrap.dedent(doc)732 if part_iter is None:733 part_iter = []734 if isinstance(part_iter, list):735 736 def iter_parts(base_stats: Stats, head_stats: Stats | None):737 yield from part_iter738 739 self.part_iter = iter_parts740 else:741 self.part_iter = part_iter742 self.comparative = comparative743 744 745def calc_execution_count_table(prefix: str) -> RowCalculator:746 def calc(stats: Stats) -> Rows:747 opcode_stats = stats.get_opcode_stats(prefix)748 counts = opcode_stats.get_execution_counts()749 total = opcode_stats.get_total_execution_count()750 cumulative = 0751 rows: Rows = []752 for opcode, (count, miss) in sorted(753 counts.items(), key=itemgetter(1), reverse=True754 ):755 cumulative += count756 if miss:757 miss_val = Ratio(miss, count)758 else:759 miss_val = None760 rows.append(761 (762 opcode,763 Count(count),764 Ratio(count, total),765 Ratio(cumulative, total),766 miss_val,767 )768 )769 return rows770 771 return calc772 773 774def execution_count_section() -> Section:775 return Section(776 "Execution counts",777 "Execution counts for Tier 1 instructions.",778 [779 Table(780 ("Name", "Count:", "Self:", "Cumulative:", "Miss ratio:"),781 calc_execution_count_table("opcode"),782 join_mode=JoinMode.CHANGE_ONE_COLUMN,783 )784 ],785 doc="""786 The "miss ratio" column shows the percentage of times the instruction787 executed that it deoptimized. When this happens, the base unspecialized788 instruction is not counted.789 """,790 )791 792 793def pair_count_section(prefix: str, title=None) -> Section:794 def calc_pair_count_table(stats: Stats) -> Rows:795 opcode_stats = stats.get_opcode_stats(prefix)796 pair_counts = opcode_stats.get_pair_counts()797 total = opcode_stats.get_total_execution_count()798 799 cumulative = 0800 rows: Rows = []801 for (opcode_i, opcode_j), count in itertools.islice(802 sorted(pair_counts.items(), key=itemgetter(1), reverse=True), 100803 ):804 cumulative += count805 rows.append(806 (807 f"{opcode_i} {opcode_j}",808 Count(count),809 Ratio(count, total),810 Ratio(cumulative, total),811 )812 )813 return rows814 815 return Section(816 "Pair counts",817 f"Pair counts for top 100 {title if title else prefix} pairs",818 [819 Table(820 ("Pair", "Count:", "Self:", "Cumulative:"),821 calc_pair_count_table,822 )823 ],824 comparative=False,825 doc="""826 Pairs of specialized operations that deoptimize and are then followed by827 the corresponding unspecialized instruction are not counted as pairs.828 """,829 )830 831 832def pre_succ_pairs_section() -> Section:833 def iter_pre_succ_pairs_tables(base_stats: Stats, head_stats: Stats | None = None):834 assert head_stats is None835 836 opcode_stats = base_stats.get_opcode_stats("opcode")837 838 for opcode in opcode_stats.get_opcode_names():839 predecessors = opcode_stats.get_predecessors(opcode)840 successors = opcode_stats.get_successors(opcode)841 predecessors_total = predecessors.total()842 successors_total = successors.total()843 if predecessors_total == 0 and successors_total == 0:844 continue845 pred_rows = [846 (pred, Count(count), Ratio(count, predecessors_total))847 for (pred, count) in predecessors.most_common(5)848 ]849 succ_rows = [850 (succ, Count(count), Ratio(count, successors_total))851 for (succ, count) in successors.most_common(5)852 ]853 854 yield Section(855 opcode,856 f"Successors and predecessors for {opcode}",857 [858 Table(859 ("Predecessors", "Count:", "Percentage:"),860 lambda *_: pred_rows, # type: ignore861 ),862 Table(863 ("Successors", "Count:", "Percentage:"),864 lambda *_: succ_rows, # type: ignore865 ),866 ],867 )868 869 return Section(870 "Predecessor/Successor Pairs",871 "Top 5 predecessors and successors of each Tier 1 opcode.",872 iter_pre_succ_pairs_tables,873 comparative=False,874 doc="""875 This does not include the unspecialized instructions that occur after a876 specialized instruction deoptimizes.877 """,878 )879 880 881def specialization_section() -> Section:882 def calc_specialization_table(opcode: str) -> RowCalculator:883 def calc(stats: Stats) -> Rows:884 DOCS = {885 "deferred": 'Lists the number of "deferred" (i.e. not specialized) instructions executed.',886 "hit": "Specialized instructions that complete.",887 "miss": "Specialized instructions that deopt.",888 "deopt": "Specialized instructions that deopt.",889 }890 891 opcode_stats = stats.get_opcode_stats("opcode")892 total = opcode_stats.get_specialization_total(opcode)893 specialization_counts = opcode_stats.get_specialization_counts(opcode)894 895 return [896 (897 Doc(label, DOCS[label]),898 Count(count),899 Ratio(count, total),900 )901 for label, count in specialization_counts.items()902 ]903 904 return calc905 906 def calc_specialization_success_failure_table(name: str) -> RowCalculator:907 def calc(stats: Stats) -> Rows:908 values = stats.get_opcode_stats(909 "opcode"910 ).get_specialization_success_failure(name)911 total = sum(values.values())912 if total:913 return [914 (label.capitalize(), Count(val), Ratio(val, total))915 for label, val in values.items()916 ]917 else:918 return []919 920 return calc921 922 def calc_specialization_failure_kind_table(name: str) -> RowCalculator:923 def calc(stats: Stats) -> Rows:924 opcode_stats = stats.get_opcode_stats("opcode")925 failures = opcode_stats.get_specialization_failure_kinds(name)926 total = opcode_stats.get_specialization_failure_total(name)927 928 return sorted(929 [930 (label, Count(value), Ratio(value, total))931 for label, value in failures.items()932 if value933 ],934 key=itemgetter(1),935 reverse=True,936 )937 938 return calc939 940 def iter_specialization_tables(base_stats: Stats, head_stats: Stats | None = None):941 opcode_base_stats = base_stats.get_opcode_stats("opcode")942 names = opcode_base_stats.get_opcode_names()943 if head_stats is not None:944 opcode_head_stats = head_stats.get_opcode_stats("opcode")945 names &= opcode_head_stats.get_opcode_names() # type: ignore946 else:947 opcode_head_stats = None948 949 for opcode in sorted(names):950 if not opcode_base_stats.is_specializable(opcode):951 continue952 if opcode_base_stats.get_specialization_total(opcode) == 0 and (953 opcode_head_stats is None954 or opcode_head_stats.get_specialization_total(opcode) == 0955 ):956 continue957 yield Section(958 opcode,959 f"specialization stats for {opcode} family",960 [961 Table(962 ("Kind", "Count:", "Ratio:"),963 calc_specialization_table(opcode),964 JoinMode.CHANGE,965 ),966 Table(967 ("Success", "Count:", "Ratio:"),968 calc_specialization_success_failure_table(opcode),969 JoinMode.CHANGE,970 ),971 Table(972 ("Failure kind", "Count:", "Ratio:"),973 calc_specialization_failure_kind_table(opcode),974 JoinMode.CHANGE,975 ),976 ],977 )978 979 return Section(980 "Specialization stats",981 "Specialization stats by family",982 iter_specialization_tables,983 )984 985 986def specialization_effectiveness_section() -> Section:987 def calc_specialization_effectiveness_table(stats: Stats) -> Rows:988 opcode_stats = stats.get_opcode_stats("opcode")989 total = opcode_stats.get_total_execution_count()990 991 (992 basic,993 specialized_hits,994 specialized_misses,995 not_specialized,996 ) = opcode_stats.get_specialized_total_counts()997 998 return [999 (1000 Doc(1001 "Basic",1002 "Instructions that are not and cannot be specialized, e.g. `LOAD_FAST`.",1003 ),1004 Count(basic),1005 Ratio(basic, total),1006 ),1007 (1008 Doc(1009 "Not specialized",1010 "Instructions that could be specialized but aren't, e.g. `LOAD_ATTR`, `BINARY_SLICE`.",1011 ),1012 Count(not_specialized),1013 Ratio(not_specialized, total),1014 ),1015 (1016 Doc(1017 "Specialized hits",1018 "Specialized instructions, e.g. `LOAD_ATTR_MODULE` that complete.",1019 ),1020 Count(specialized_hits),1021 Ratio(specialized_hits, total),1022 ),1023 (1024 Doc(1025 "Specialized misses",1026 "Specialized instructions, e.g. `LOAD_ATTR_MODULE` that deopt.",1027 ),1028 Count(specialized_misses),1029 Ratio(specialized_misses, total),1030 ),1031 ]1032 1033 def calc_deferred_by_table(stats: Stats) -> Rows:1034 opcode_stats = stats.get_opcode_stats("opcode")1035 deferred_counts = opcode_stats.get_deferred_counts()1036 total = sum(deferred_counts.values())1037 if total == 0:1038 return []1039 1040 return [1041 (name, Count(value), Ratio(value, total))1042 for name, value in sorted(1043 deferred_counts.items(), key=itemgetter(1), reverse=True1044 )[:10]1045 ]1046 1047 def calc_misses_by_table(stats: Stats) -> Rows:1048 opcode_stats = stats.get_opcode_stats("opcode")1049 misses_counts = opcode_stats.get_misses_counts()1050 total = sum(misses_counts.values())1051 if total == 0:1052 return []1053 1054 return [1055 (name, Count(value), Ratio(value, total))1056 for name, value in sorted(1057 misses_counts.items(), key=itemgetter(1), reverse=True1058 )[:10]1059 ]1060 1061 return Section(1062 "Specialization effectiveness",1063 "",1064 [1065 Table(1066 ("Instructions", "Count:", "Ratio:"),1067 calc_specialization_effectiveness_table,1068 JoinMode.CHANGE,1069 ),1070 Section(1071 "Deferred by instruction",1072 "Breakdown of deferred (not specialized) instruction counts by family",1073 [1074 Table(1075 ("Name", "Count:", "Ratio:"),1076 calc_deferred_by_table,1077 JoinMode.CHANGE,1078 )1079 ],1080 ),1081 Section(1082 "Misses by instruction",1083 "Breakdown of misses (specialized deopts) instruction counts by family",1084 [1085 Table(1086 ("Name", "Count:", "Ratio:"),1087 calc_misses_by_table,1088 JoinMode.CHANGE,1089 )1090 ],1091 ),1092 ],1093 doc="""1094 All entries are execution counts. Should add up to the total number of1095 Tier 1 instructions executed.1096 """,1097 )1098 1099 1100def call_stats_section() -> Section:1101 def calc_call_stats_table(stats: Stats) -> Rows:1102 call_stats = stats.get_call_stats()1103 total = sum(v for k, v in call_stats.items() if "Calls to" in k)1104 return [1105 (key, Count(value), Ratio(value, total))1106 for key, value in call_stats.items()1107 ]1108 1109 return Section(1110 "Call stats",1111 "Inlined calls and frame stats",1112 [1113 Table(1114 ("", "Count:", "Ratio:"),1115 calc_call_stats_table,1116 JoinMode.CHANGE,1117 )1118 ],1119 doc="""1120 This shows what fraction of calls to Python functions are inlined (i.e.1121 not having a call at the C level) and for those that are not, where the1122 call comes from. The various categories overlap.1123 1124 Also includes the count of frame objects created.1125 """,1126 )1127 1128 1129def object_stats_section() -> Section:1130 def calc_object_stats_table(stats: Stats) -> Rows:1131 object_stats = stats.get_object_stats()1132 return [1133 (label, Count(value), Ratio(value, den))1134 for label, (value, den) in object_stats.items()1135 ]1136 1137 return Section(1138 "Object stats",1139 "Allocations, frees and dict materializatons",1140 [1141 Table(1142 ("", "Count:", "Ratio:"),1143 calc_object_stats_table,1144 JoinMode.CHANGE,1145 )1146 ],1147 doc="""1148 Below, "allocations" means "allocations that are not from a freelist".1149 Total allocations = "Allocations from freelist" + "Allocations".1150 1151 "Inline values" is the number of values arrays inlined into objects.1152 1153 The cache hit/miss numbers are for the MRO cache, split into dunder and1154 other names.1155 """,1156 )1157 1158 1159def gc_stats_section() -> Section:1160 def calc_gc_stats(stats: Stats) -> Rows:1161 gc_stats = stats.get_gc_stats()1162 1163 return [1164 (1165 Count(i),1166 Count(gen["collections"]),1167 Count(gen["objects collected"]),1168 Count(gen["object visits"]),1169 Count(gen["objects reachable from roots"]),1170 Count(gen["objects not reachable from roots"]),1171 )1172 for (i, gen) in enumerate(gc_stats)1173 ]1174 1175 return Section(1176 "GC stats",1177 "GC collections and effectiveness",1178 [1179 Table(1180 ("Generation:", "Collections:", "Objects collected:", "Object visits:",1181 "Reachable from roots:", "Not reachable from roots:"),1182 calc_gc_stats,1183 )1184 ],1185 doc="""1186 Collected/visits gives some measure of efficiency.1187 """,1188 )1189 1190 1191def optimization_section() -> Section:1192 def calc_optimization_table(stats: Stats) -> Rows:1193 optimization_stats = stats.get_optimization_stats()1194 1195 return [1196 (1197 label,1198 Count(value),1199 Ratio(value, den, percentage=label != "Uops executed"),1200 )