From d91336ece6d53fe1a1e92a51fbcfce576ff5be1b Mon Sep 17 00:00:00 2001 From: Tanner Andrulis Date: Wed, 29 Jul 2026 11:22:38 -0400 Subject: [PATCH 1/8] component_latency columns --- accelforge/mapper/FFM/mappings.py | 6 +++--- accelforge/model/run_model.py | 5 ++--- accelforge/plotting/mappings.py | 4 ++-- tests/network/test_network.py | 12 ++++++++---- 4 files changed, 15 insertions(+), 12 deletions(-) diff --git a/accelforge/mapper/FFM/mappings.py b/accelforge/mapper/FFM/mappings.py index 518ec8db..2fbdb7c2 100755 --- a/accelforge/mapper/FFM/mappings.py +++ b/accelforge/mapper/FFM/mappings.py @@ -445,7 +445,7 @@ def should_keep(col): if len(parts) >= 3 and (parts[1] == "energy" or parts[1] == "action"): comp = parts[2] return comp in keep - if len(parts) == 3 and parts[1] == "latency": + if len(parts) == 3 and parts[1] == "component_latency": comp = parts[2] return comp in keep return True # Keep non-component columns (Total, mapping, etc.) @@ -482,7 +482,7 @@ def should_keep(col): comp = None if len(parts) >= 3 and parts[1] in ("energy", "action"): comp = parts[2] - elif len(parts) == 3 and parts[1] == "latency": + elif len(parts) == 3 and parts[1] == "component_latency": comp = parts[2] if comp is None: return True # Non-component columns always kept @@ -658,7 +658,7 @@ def latency( parameters are set to True, a float or a list of floats is returned. """ - energy = self.access("latency") + energy = self.access("component_latency", col_idx=1) result = {} for einsum in self.einsum_names: diff --git a/accelforge/model/run_model.py b/accelforge/model/run_model.py index 667e8ad1..4b7840d2 100644 --- a/accelforge/model/run_model.py +++ b/accelforge/model/run_model.py @@ -195,8 +195,6 @@ def run_model( ) for key, energy_val in detailed_energy.items(): df[energy2col(key)] = energy_val * n_instances - for component, cur_latency in latency.items(): - df[f"latency{component}"] = cur_latency * n_instances actions_df = {} simple_actions = gather_actions( @@ -206,8 +204,9 @@ def run_model( actions_df[action2col(key)] = count.total * n_instances if metrics.includes_latency(): + for component, cur_latency in latency.items(): + df[f"component_latency{component}"] = cur_latency * n_instances df["Totallatency"] = overall_latency * n_instances - # df[f"latencycompute"] = comp_latency * n_instances # For first latency, we'll follow the convention of treating compute # as a component, similarly to memory (see below). for compute_level, stats in reuse.compute_stats.items(): # FIRST LATENCY diff --git a/accelforge/plotting/mappings.py b/accelforge/plotting/mappings.py index 66fbb15d..f4789ceb 100644 --- a/accelforge/plotting/mappings.py +++ b/accelforge/plotting/mappings.py @@ -23,9 +23,9 @@ def _col2latency(colname: str): - """Parse latency columns: einsumlatencycomponent -> VerboseActionKey.""" + """Parse latency columns: einsumcomponent_latencyname -> VerboseActionKey.""" parts = colname.split("") - if len(parts) == 3 and parts[1] == "latency": + if len(parts) == 3 and parts[1] == "component_latency": return VerboseActionKey( level=parts[2], action="latency", tensor="None", einsum=parts[0] ) diff --git a/tests/network/test_network.py b/tests/network/test_network.py index 06434cac..000861f7 100644 --- a/tests/network/test_network.py +++ b/tests/network/test_network.py @@ -272,7 +272,7 @@ def test_flat(self): (M / M_TILE * KN // MAC_TILE * M_TILE * MAC_TILE * BITS_PER_VALUE), ) self.assertEqual( - result.data["Matmul0latencyRowBuffer"].iloc[0], + result.data["Matmul0component_latencyRowBuffer"].iloc[0], ( M / M_TILE @@ -285,7 +285,7 @@ def test_flat(self): ), ) self.assertEqual( - result.data["Matmul0latencyDistributedBuffer"].iloc[0], + result.data["Matmul0component_latencyDistributedBuffer"].iloc[0], ( # Reads from child M / M_TILE @@ -385,8 +385,12 @@ def test_hierarchical_1d_all_to_all(self): # --- Latency ------------------------------------------------------ # The switch's uniform single-hop routing gives MacArray a constant # latency of 1, versus the mesh PeArray's 2. - self.assertEqual(result.data["Matmul0latencyMacArray"].iloc[0], 1) - self.assertEqual(result.data["Matmul0latencyPeArray"].iloc[0], 2) + self.assertEqual( + result.data["Matmul0component_latencyMacArray"].iloc[0], 1 + ) + self.assertEqual( + result.data["Matmul0component_latencyPeArray"].iloc[0], 2 + ) self.assertEqual(result.data["Totallatency"].iloc[0], 2) From 75935d028c9c30a08194c5d767403dff4a13d4a6 Mon Sep 17 00:00:00 2001 From: Tanner Andrulis Date: Wed, 29 Jul 2026 14:31:19 -0400 Subject: [PATCH 2/8] Single-Einsum wind-up wind-down latency --- accelforge/frontend/arch/components.py | 159 ++++++------ accelforge/frontend/spec.py | 10 +- accelforge/model/_looptree/energy.py | 19 +- accelforge/model/_looptree/latency/memory.py | 83 +++++- .../model/_looptree/reuse/symbolic/_stats.py | 82 +++--- .../_looptree/reuse/symbolic/_symbolic.py | 47 ++-- accelforge/model/run_model.py | 49 +++- accelforge/util/_migration.py | 7 - .../colonnade_jssc_2021.yaml | 2 +- .../compute_in_memory/jia_jssc_2020.yaml | 2 +- .../compute_in_memory/sinangil_jssc_2021.yaml | 2 +- .../compute_in_memory/wang_vlsi_2022.yaml | 4 +- examples/arches/fanout_variations/at_glb.yaml | 10 +- .../at_glb_with_fanout_node.yaml | 10 +- examples/arches/fanout_variations/at_mac.yaml | 10 +- .../at_mac_with_constraints.yaml | 10 +- .../at_mac_with_fanout_node.yaml | 10 +- examples/arches/nvdla.yaml | 10 +- examples/arches/simple.yaml | 10 +- examples/arches/tpu_v4i.yaml | 20 +- tests/input_files/toll.arch.yaml | 2 +- tests/input_files/toll_no_outer.arch.yaml | 2 +- tests/network/input_files/networked/flat.yaml | 26 +- .../input_files/networked/hierarchical.yaml | 14 +- .../networked/hierarchical_1d.yaml | 15 +- .../networked/hierarchical_1d_all_to_all.yaml | 15 +- .../networked/hierarchical_switched.yaml | 18 +- tests/network/test_network.py | 18 +- .../test_api_gaps.py | 104 +++++--- .../test_arch_flattening.py | 243 +++++++++++++++--- .../test_component_fields.py | 74 ++++-- .../test_evaluation.py | 24 +- .../test_parsed_values.py | 12 +- .../test_rendering.py | 61 ++++- .../test_yaml_and_expressions.py | 46 ++-- .../vpa_precedence_action_bpa.arch.yaml | 2 +- .../vpa_precedence_action_wins.arch.yaml | 2 +- .../vpa_precedence_component_bpa.arch.yaml | 2 +- .../vpa_precedence_component_wins.arch.yaml | 2 +- .../vpa_precedence_default.arch.yaml | 2 +- .../test_values_per_action_precedence.py | 18 +- 41 files changed, 850 insertions(+), 408 deletions(-) delete mode 100644 accelforge/util/_migration.py diff --git a/accelforge/frontend/arch/components.py b/accelforge/frontend/arch/components.py index 6ad4b423..34579586 100644 --- a/accelforge/frontend/arch/components.py +++ b/accelforge/frontend/arch/components.py @@ -11,7 +11,6 @@ Self, TypeVar, ) -import warnings from pydantic import ConfigDict, PrivateAttr, field_validator, model_validator from hwcomponents import ( ComponentModel, @@ -19,8 +18,6 @@ get_model, ) -from accelforge.util._migration import LATENCY_TO_THROUGHPUT_MIGRATION - # Verify the installed `hwcomponents` exposes the new `get_action_cost` API. If it # can't be imported, the user is on an old hwcomponents that pre-dates the latency -> @@ -87,14 +84,14 @@ class Action(EvalableModel): latency: EvalsTo[int | float | None] = None """ - Deprecated; use `throughput` instead. Setting this emits a deprecation warning and - auto-converts to `throughput = 1 / latency`. + Latency of one call of this action in seconds. Per-action latency is multiplied by + the component's latency_scale and the action's latency_scale. """ latency_scale: EvalsTo[int | float] = 1 """ - Deprecated; use `throughput_scale` instead. Setting this to a non-default value - emits a deprecation warning and auto-converts to `throughput_scale = 1 / latency_scale`. + The scale factor for latency of this action. Multiplies this action's latency by + this value. """ throughput: EvalsTo[int | float | None] = None @@ -145,38 +142,6 @@ def _set_n_calls(self, value: int | float) -> None: self._n_calls = value self._model_running = True - @model_validator(mode="before") - @classmethod - def _deprecate_latency_fields(cls, data): - if isinstance(data, dict): - if "latency" in data and not "throughput" in data: - l = data.pop("latency") - warnings.warn( - f"Setting `latency` on `{cls.__name__}` is deprecated; use " - f"`throughput` instead. Auto-converting `latency={l!r}` to " - f"`throughput`. Please update your input.\n\n" - f"{LATENCY_TO_THROUGHPUT_MIGRATION}", - DeprecationWarning, - stacklevel=2, - ) - l = str(l).strip() - data["throughput"] = f"1 / ({l}) if ({l}) != 0 else float('inf')" - if "latency_scale" in data and not "throughput_scale" in data: - ls = data.pop("latency_scale") - warnings.warn( - f"Setting `latency_scale` on `{cls.__name__}` is deprecated; use " - f"`throughput_scale` instead (with the reciprocal). Auto-converting " - f"`latency_scale={ls!r}` to `throughput_scale`. Please update your " - f"input.\n\n{LATENCY_TO_THROUGHPUT_MIGRATION}", - DeprecationWarning, - stacklevel=2, - ) - ls = str(ls).strip() - data["throughput_scale"] = ( - f"1 / ({ls}) if ({ls}) != 0 else float('inf')" - ) - return data - def _attributes_for_component_model(self) -> dict[str, Any]: return { **self.shallow_model_dump(), @@ -352,8 +317,8 @@ class Component(Spatialable): latency_scale: EvalsTo[int | float] = 1 """ - Deprecated; use `throughput_scale` instead. Setting this to a non-default value - emits a deprecation warning and auto-converts to `throughput_scale = 1 / latency_scale`. + The scale factor for the latency of this component. Multiplies the calculated + latency of each action. """ throughput_scale: EvalsTo[int | float] = 1 @@ -363,28 +328,6 @@ class Component(Spatialable): the scale factor is 2, then the final throughput is 2M actions/second. """ - @model_validator(mode="before") - @classmethod - def _deprecate_latency_scale_on_component(cls, data): - if isinstance(data, dict) and "latency_scale" in data: - ls = data.pop("latency_scale") - warnings.warn( - f"Setting `latency_scale` on `{cls.__name__}` is deprecated; use " - f"`throughput_scale` instead (with the reciprocal). Auto-converting " - f"`latency_scale={ls!r}` to `throughput_scale`. Please update your " - f"input.\n\n{LATENCY_TO_THROUGHPUT_MIGRATION}", - DeprecationWarning, - stacklevel=2, - ) - if "throughput_scale" in data: - raise ValueError( - f"Cannot specify both `latency_scale` and `throughput_scale` " - f"on `{cls.__name__}`. Drop the deprecated `latency_scale`." - ) - ls = str(ls).strip() - data["throughput_scale"] = f"1 / ({ls}) if ({ls}) != 0 else float('inf')" - return data - n_parallel_instances: EvalsTo[int | float] = 1 """ The number of parallel instances of this component. Increasing parallel instances @@ -431,7 +374,7 @@ def get_component_class(self, trying_to_calculate: str = None) -> str: "talk to hwcomponents, but was missing necessary attributes. If you do " "not want to use hwcomponents component_models, ensure that area and " "leak_power are set, as well as, for each action, " - f"energy and throughput are set.{extra_info}", + f"energy, throughput, and latency are set.{extra_info}", source_field=f"{self.name}.component_class", ) return self.component_class @@ -783,6 +726,77 @@ def calculate_action_throughput( ) return self + def calculate_action_latency( + self, + component_models: list[ComponentModel] | None = None, + in_place: bool = False, + ) -> Self: + """ + Calculates the latency for each action by this component. Populates the + ``.latency`` field. Extends the ``component_modeling_log`` field with + log messages. + + Parameters + ---------- + component_models : list[hwcomponents.ComponentModel] | None + The models to use for latency calculation. If not provided, the models + will be found with `hwcomponents.get_models()`. + in_place : bool + If True, the component will be modified in place. Otherwise, a copy will be + returned. + + Returns + ------- + Self + A copy of the component with the calculated latency for each action. + """ + if not in_place: + self: Self = self._copy_for_component_modeling() + + messages = self.component_modeling_log + + for action in self.actions: + messages.append( + f"Calculating latency for {self.name} action {action.name}." + ) + if action.latency is not None: + latency = action.latency + messages.append(f"Setting {self.name} latency to {action.latency=}") + elif self._is_dummy(): + latency = 0 + messages.append("Component is dummy. Setting latency to 0.") + else: + self.populate_component_model( + component_models, + in_place=True, + trying_to_calculate=f"latency for action {action.name}", + ) + latency = self.component_model.try_call_arbitrary_action( + action_name=action.name, + _return_estimation_object=True, + **{ + **self._attributes_for_component_model(), + **action._attributes_for_component_model(), + }, + ) + messages.extend(latency.messages) + latency = latency.value.latency + if self.latency_scale != 1: + latency *= self.latency_scale + messages.append(f"Scaling {self.name} latency by {self.latency_scale=}") + if action.latency_scale != 1: + latency *= action.latency_scale + messages.append( + f"Scaling {self.name} latency by {action.latency_scale=}" + ) + action.latency = latency + if action.latency < 0: + logging.warning( + f"Component {self.name} action {action.name} has negative latency: " + f"{action.latency}" + ) + return self + def calculate_component_costs( self, component_models: list[ComponentModel] | None = None, @@ -842,6 +856,7 @@ def calculate_component_costs( self.calculate_area(component_models, in_place=True) self.calculate_action_energy(component_models, in_place=True) self.calculate_action_throughput(component_models, in_place=True) + self.calculate_action_latency(component_models, in_place=True) self.calculate_leak_power(component_models, in_place=True) if _use_cache: _set_component_model_cache(cachekey, self) @@ -1326,18 +1341,16 @@ class Network(Component, Leaf): actions: EvalableList[Action] = NETWORK_ACTIONS - total_latency: str | int | float = ( - "max(max_hops*actions['hop'].latency, max_link_traffic/actions['hop'].throughput)" - ) + total_latency: str | int | float | None = "max_link_traffic/actions['hop'].throughput" """ - Models latency as either: - - *Latency-bound*, which means that the latency of the route with the most number of - hops dominate the overall communication latency. - - *Bandwidth-bound*, which means that the traffic over the most congested link - dominates the overall communication latency. + Models latency as bandwidth-bound, which means that the traffic over the most + congested link dominates the overall communication latency. Note that max_hops * + actions['hop'].latency will already be included in wind-up and wind-down of the + network. Keywords: - - `max_hops` returns the number of hops in the longest route. + + - `max_hops` returns the number of hops in the longest route. - `max_link_traffic` returns the amount of traffic (in bits) over the most congested link. """ diff --git a/accelforge/frontend/spec.py b/accelforge/frontend/spec.py index c9059c68..1850e075 100755 --- a/accelforge/frontend/spec.py +++ b/accelforge/frontend/spec.py @@ -191,6 +191,7 @@ def calculate_component_costs( area: bool = True, energy: bool = True, throughput: bool = True, + latency: bool = True, leak: bool = True, ) -> "Spec": """ @@ -218,10 +219,12 @@ def calculate_component_costs( Whether to compute and populate energy entries. throughput : bool, optional Whether to compute and populate throughput entries. + latency : bool, optional + Whether to compute and populate latency entries. leak : bool, optional Whether to compute and populate leak power entries. """ - if not area and not energy and not throughput and not leak: + if not area and not energy and not throughput and not latency and not leak: return self models = hwcomponents.get_models( @@ -275,6 +278,11 @@ def calculate_component_costs( for a in c.actions: orig_action = orig.actions[a.name] orig_action.throughput = a.throughput + if latency: + c = c.calculate_action_latency(models) + for a in c.actions: + orig_action = orig.actions[a.name] + orig_action.latency = a.latency if leak: c = c.calculate_leak_power(models) orig.leak_power = c.leak_power diff --git a/accelforge/model/_looptree/energy.py b/accelforge/model/_looptree/energy.py index 3b2382a1..44f5bd6a 100755 --- a/accelforge/model/_looptree/energy.py +++ b/accelforge/model/_looptree/energy.py @@ -40,17 +40,14 @@ def gather_actions( else: level = bindings[level] - key = buffet_keyer(buffet, "read") - if key not in actions: - actions[key] = ActionCount.default() - actions[key].total += accesses.net_total_read_actions() - actions[key].max_per_unit += accesses.net_max_per_unit_read_actions() - - key = buffet_keyer(buffet, "write") - if key not in actions: - actions[key] = ActionCount.default() - actions[key].total += accesses.net_total_write_actions() - actions[key].max_per_unit += accesses.net_max_per_unit_write_actions() + net_total = accesses.net_total_actions() + net_max_per_unit = accesses.net_max_per_unit_actions() + for action_name in net_total: + key = buffet_keyer(buffet, action_name) + if key not in actions: + actions[key] = ActionCount.default() + actions[key].total += net_total[action_name] + actions[key].max_per_unit += net_max_per_unit[action_name] for compute, ops in looptree_results.compute_stats.items(): key = compute_keyer(compute, "compute") diff --git a/accelforge/model/_looptree/latency/memory.py b/accelforge/model/_looptree/latency/memory.py index ee97fbff..61f5587d 100755 --- a/accelforge/model/_looptree/latency/memory.py +++ b/accelforge/model/_looptree/latency/memory.py @@ -1,4 +1,5 @@ from collections import defaultdict +from numbers import Number from accelforge.frontend import arch from accelforge.frontend.arch import Leaf, Memory, TensorHolder, Component @@ -12,9 +13,10 @@ from accelforge.model._looptree.reuse import SymbolicAnalysisOutput from accelforge.model._looptree.types import Buffet -from accelforge.model._looptree.reuse.symbolic import BuffetStats +from accelforge.model._looptree.reuse.symbolic import BuffetStats, NetworkStats from accelforge.util._eval_expressions import MATH_FUNCS, eval_expression -from accelforge.util._sympy.broadcast_max import Max, Min, MaxGeqZero +from accelforge.util._frozenset import oset +from accelforge.util._sympy.broadcast_max import Max, Min, MaxGeqZero, max_nonzero from accelforge.util._basetypes import EvalableList import symengine as se @@ -62,6 +64,73 @@ def _min(*args): return Min(*args) +def communication_latency( + reuse: SymbolicAnalysisOutput, + flattened_arch: FlattenedArch, + tensor_to_backing: dict[str, str], + output_tensors, +) -> dict[str, object]: + """ + Communication latency per component: the time to move one tile of each tensor to + that level. Inputs travel from their backing storage down; outputs are produced + after the worst input reaches compute, then travel from compute up. Each level on + the path charges one call of each action it performs for the tensor; each network + on the path charges its longest route. The whole path repeats once per temporal + iteration above the backing storage. Returns, for each component, the worst + communication latency over all tensors. + """ + name2component = {n.name: n for n in flattened_arch} + name2index = {n.name: i for i, n in enumerate(flattened_arch)} + + tensor_stats = {} + for b, s in reuse.buffet_stats.items(): + tensor_stats.setdefault(b.tensor, []).append((b.level, s)) + for n, s in reuse.network_stats.items(): + tensor_stats.setdefault(n.tensor, []).append((n.component, s)) + + communication = defaultdict(list) + worst_input_to_compute = 0 + # Inputs first: outputs build on the worst input's arrival at compute. + tensor_order = sorted(tensor_stats, key=lambda t: t in output_tensors) + for tensor in tensor_order: + stats = tensor_stats[tensor] + is_output = tensor in output_tensors + stats.sort(key=lambda c: name2index[c[0]], reverse=is_output) + backing = tensor_to_backing[tensor] + + cur_latency = 0 + iterations = None + tensor_latency = {} + for level, stats in stats: + if isinstance(stats, BuffetStats): + if level == backing: + iterations = stats.iterations_above + cur_latency += sum( + name2component[level].actions[action].latency + for action, count in stats.total_actions.items() + if not isinstance(count, Number) or count != 0 + ) + tensor_latency[level] = cur_latency + elif isinstance(stats, NetworkStats): + cur_latency += ( + stats.max_hops * name2component[level].actions["hop"].latency + ) + tensor_latency[level] = cur_latency + else: + raise ValueError(f"Unknown stats type: {type(stats)}") + + assert iterations is not None, f"Tensor {tensor} has no backing storage" + start = worst_input_to_compute if is_output else 0 + for level in tensor_latency: + communication[level].append(start + tensor_latency[level] * iterations) + if not is_output: + worst_input_to_compute = max_nonzero( + worst_input_to_compute, cur_latency * iterations + ) + + return {level: max_nonzero(*vals) for level, vals in communication.items()} + + def component_latency( looptree_results: SymbolicAnalysisOutput, flattened_arch: FlattenedArch, @@ -91,15 +160,9 @@ def component_latency( actions[action.name] += 0 if isinstance(name2component[component], TensorHolder): - actions["read"] += ( - buffet_stats.max_per_unit_read_actions - - buffet_stats.min_per_unit_skipped_first_read_actions - ) + actions["read"] += buffet_stats.net_max_per_unit_actions("read") if not isinstance(name2component[component], arch.Toll): - actions["write"] += ( - buffet_stats.max_per_unit_write_actions - - buffet_stats.min_per_unit_skipped_first_write_actions - ) + actions["write"] += buffet_stats.net_max_per_unit_actions("write") elif isinstance(name2component[component], arch.Compute): pass else: diff --git a/accelforge/model/_looptree/reuse/symbolic/_stats.py b/accelforge/model/_looptree/reuse/symbolic/_stats.py index 096ff2a3..c591a869 100644 --- a/accelforge/model/_looptree/reuse/symbolic/_stats.py +++ b/accelforge/model/_looptree/reuse/symbolic/_stats.py @@ -1,4 +1,5 @@ import copy +import operator from dataclasses import dataclass, field from typing import Any @@ -37,6 +38,25 @@ def repeat(self, n_repeats): return new +class ActionCounts(dict): + """Per-action-name counts. Missing actions count as 0 so `d[action] += x` + works for actions a component lacks.""" + + def __missing__(self, key): + return 0 + +def _scale(value: Any, factor: Any) -> Any: + if isinstance(value, ActionCounts): + return ActionCounts({k: v * factor for k, v in value.items()}) + return value * factor + + +def _combine(a: Any, b: Any, op) -> Any: + if isinstance(a, ActionCounts) or isinstance(b, ActionCounts): + return ActionCounts({k: op(a[k], b[k]) for k in oset(a) | oset(b)}) + return op(a, b) + + @dataclass class BuffetStats: total_reads_to_parent: Any = field(default=0) @@ -58,21 +78,21 @@ class BuffetStats: max_occupancy: Any = field(default=0) _n_loops_above: int = field(default=0) - # These are used to calculate energy and latency - total_write_actions: Any = field(default=0) - max_per_unit_write_actions: Any = field(default=0) - total_read_actions: Any = field(default=0) - max_per_unit_read_actions: Any = field(default=0) - - total_skipped_first_write_actions: Any = field(default=0) - min_per_unit_skipped_first_write_actions: Any = field(default=0) - total_skipped_first_read_actions: Any = field(default=0) - min_per_unit_skipped_first_read_actions: Any = field(default=0) + # These are used to calculate energy and latency. Keyed by action name. + total_actions: ActionCounts = field(default_factory=ActionCounts) + max_per_unit_actions: ActionCounts = field(default_factory=ActionCounts) + total_skipped_first_actions: ActionCounts = field(default_factory=ActionCounts) + min_per_unit_skipped_first_actions: ActionCounts = field( + default_factory=ActionCounts + ) # NOTE: anything other than min_, max_, or total_ must default to # None. There are asserts that check this. persistent: bool = field(default=None) + # Number of temporal iterations above this buffet's storage node. + iterations_above: Any = field(default=1) + @property def n_loops_above(self) -> int: if self.persistent: @@ -96,7 +116,7 @@ def repeat_temporal(self, factor: int, is_fully_relevant: bool) -> "BuffetStats" continue # First actions occur once per relevant iteration. if k == "max_occupancy": continue # Max occupancy is not affected by temporal loops above - new.__dict__[k] = v * factor + new.__dict__[k] = _scale(v, factor) return new def repeat_spatial(self, factor: int, reuse_parent_accesses: bool) -> "BuffetStats": @@ -120,7 +140,7 @@ def repeat_spatial(self, factor: int, reuse_parent_accesses: bool) -> "BuffetSta continue # Spatial fanout doesn't affect per-unit stats if k == "max_occupancy": continue # Max occupancy is not affected by temporal loops above - new.__dict__[k] = v * factor + new.__dict__[k] = _scale(v, factor) return new def max(self, **kwargs: Any): @@ -136,11 +156,13 @@ def __add__(self, other: "BuffetStats") -> "BuffetStats": for k, v in self.__dict__.items(): other_v = other.__dict__[k] if k.startswith("min_"): - new.__dict__[k] = min_nonzero(v, other_v) + new.__dict__[k] = _combine(v, other_v, min_nonzero) elif k.startswith("max_"): - new.__dict__[k] = max_nonzero(v, other_v) + new.__dict__[k] = _combine(v, other_v, max_nonzero) elif k.startswith("total_"): - new.__dict__[k] = v + other_v + new.__dict__[k] = _combine(v, other_v, operator.add) + elif k == "iterations_above" and v is not None and other_v is not None: + new.__dict__[k] = max_nonzero(v, other_v) elif v is None: new.__dict__[k] = other_v else: @@ -158,28 +180,26 @@ def __iadd__(self, other: "BuffetStats") -> "BuffetStats": setattr(self, key, value) return self - def net_total_read_actions(self) -> Any: - return self.total_read_actions - self.total_skipped_first_read_actions - - def net_total_write_actions(self) -> Any: - return self.total_write_actions - self.total_skipped_first_write_actions - - def net_max_per_unit_read_actions(self) -> Any: - return ( - self.max_per_unit_read_actions - - self.min_per_unit_skipped_first_read_actions - ) - - def net_max_per_unit_write_actions(self) -> Any: - return ( - self.max_per_unit_write_actions - - self.min_per_unit_skipped_first_write_actions + def net_total_actions(self, action: str | None = None) -> Any: + if action is not None: + return self.total_actions[action] - self.total_skipped_first_actions[action] + return ActionCounts({a: self.net_total_actions(a) for a in self.total_actions}) + + def net_max_per_unit_actions(self, action: str | None = None) -> Any: + if action is not None: + return ( + self.max_per_unit_actions[action] + - self.min_per_unit_skipped_first_actions[action] + ) + return ActionCounts( + {a: self.net_max_per_unit_actions(a) for a in self.max_per_unit_actions} ) @classmethod def blank(cls): stats = cls() stats.n_loops_above = None # Inherit from whoever is added to this + stats.iterations_above = None return stats diff --git a/accelforge/model/_looptree/reuse/symbolic/_symbolic.py b/accelforge/model/_looptree/reuse/symbolic/_symbolic.py index 93cf79e2..895d523f 100755 --- a/accelforge/model/_looptree/reuse/symbolic/_symbolic.py +++ b/accelforge/model/_looptree/reuse/symbolic/_symbolic.py @@ -503,8 +503,7 @@ def label_fused_loops(mapping: List[MappingNode], fusable_tensors: set[TensorNam def label_shared_tensor_binding_loops( - mapping: List[MappingNode], - fusable_tensors: set[TensorName] + mapping: List[MappingNode], fusable_tensors: set[TensorName] ): found_distributed = False for node in mapping: @@ -512,10 +511,8 @@ def label_shared_tensor_binding_loops( component = node.component_object found_distributed = ( len(node._backing & fusable_tensors) > 0 - and - isinstance(component, arch.Memory) - and - component._is_distributed() + and isinstance(component, arch.Memory) + and component._is_distributed() ) continue if isinstance(node, Spatial) and found_distributed: @@ -601,8 +598,13 @@ def handle_repeated_value(repeated_shape): for repeated_shape in shape.sequence: assert isinstance(repeated_shape, RepeatedValue) handle_repeated_value(repeated_shape) + total_iterations = sum(r.repeats for r in shape.sequence) elif isinstance(shape, RepeatedValue): handle_repeated_value(shape) + total_iterations = shape.repeats + + for stats in result_accumulator.buffet_stats.values(): + stats.iterations_above *= total_iterations if node_idx in info.idxs_to_track_first_latency: for compute_stat in result_accumulator.compute_stats.values(): @@ -863,14 +865,14 @@ def inherit_add(attr: str, default_value: Any = fills) -> Any: # ========================== # Data exchanges with parent if count_downward_movement[tensor]: # Parent -> Me - stats.total_write_actions += stats.total_reads_to_parent * write_scale - stats.max_per_unit_write_actions += ( + stats.total_actions["write"] += stats.total_reads_to_parent * write_scale + stats.max_per_unit_actions["write"] += ( stats.total_reads_to_parent * write_scale / n_active_physical_units ) - stats.total_skipped_first_write_actions += ( + stats.total_skipped_first_actions["write"] += ( stats.total_skipped_first_reads_to_parent * write_scale ) - stats.min_per_unit_skipped_first_write_actions += ( + stats.min_per_unit_skipped_first_actions["write"] += ( stats.min_per_parent_skipped_first_reads_to_parent * write_scale / n_active_physical_units @@ -879,40 +881,42 @@ def inherit_add(attr: str, default_value: Any = fills) -> Any: if count_upward_movement[tensor]: # Me -> Parent # Comment this to have the final writeback to a buffer hit both that buffer and # go directly to the parent without incurring another read from the buffer. - stats.total_read_actions += stats.total_writes_to_parent * read_scale - stats.max_per_unit_read_actions += ( + stats.total_actions["read"] += stats.total_writes_to_parent * read_scale + stats.max_per_unit_actions["read"] += ( stats.total_writes_to_parent * read_scale / n_active_physical_units ) # ======================== # Data exchanges with peer - stats.total_read_actions += stats.total_reads_to_peer * read_scale - stats.total_write_actions += stats.total_reads_to_peer * write_scale + stats.total_actions["read"] += stats.total_reads_to_peer * read_scale + stats.total_actions["write"] += stats.total_reads_to_peer * write_scale # ========================= # Data exchanges with child if child is not None: if count_downward_movement[tensor]: # Me -> Child - stats.total_read_actions += child.total_reads_to_parent * read_scale - stats.max_per_unit_read_actions += ( + stats.total_actions["read"] += child.total_reads_to_parent * read_scale + stats.max_per_unit_actions["read"] += ( child.max_per_parent_reads_to_parent * read_scale / n_active_physical_units ) # Skip first read if skip_initial: - stats.total_skipped_first_read_actions += ( + stats.total_skipped_first_actions["read"] += ( child.total_skipped_first_reads_to_parent * read_scale ) - stats.min_per_unit_skipped_first_read_actions += ( + stats.min_per_unit_skipped_first_actions["read"] += ( child.min_per_parent_skipped_first_reads_to_parent * read_scale / n_active_physical_units ) if count_upward_movement[tensor]: # Child -> Me - stats.total_write_actions += child.total_writes_to_parent * write_scale - stats.max_per_unit_write_actions += ( + stats.total_actions["write"] += ( + child.total_writes_to_parent * write_scale + ) + stats.max_per_unit_actions["write"] += ( child.max_per_parent_writes_to_parent * write_scale / n_active_physical_units @@ -942,7 +946,7 @@ def analyze_toll(node_idx, current_shape, info: AnalysisInfo): buffet = Buffet(tensor, einsum_name, node.component) stats = storage_result.buffet_stats[buffet] stats.max_occupancy = 0 - assert stats.total_write_actions == 0 + assert stats.total_actions["write"] == 0 return storage_result @@ -1033,6 +1037,7 @@ def analyze_compute( if tensor in info.workload.einsums[einsum].output_tensor_names: stats.total_writes_to_parent = 1 stats.max_per_parent_writes_to_parent = 1 + stats.total_actions["compute"] = computes if skip_initial: stats.total_skipped_first_reads_to_parent = 1 stats.min_per_parent_skipped_first_reads_to_parent = 1 diff --git a/accelforge/model/run_model.py b/accelforge/model/run_model.py index 4b7840d2..48c7c8fa 100644 --- a/accelforge/model/run_model.py +++ b/accelforge/model/run_model.py @@ -12,7 +12,10 @@ compute_energy_from_actions, gather_actions, ) -from accelforge.model._looptree.latency.memory import component_latency +from accelforge.model._looptree.latency.memory import ( + communication_latency, + component_latency, +) from accelforge.mapper.FFM._join_pmappings.pmapping_dataframe import ( memory_usage2col, reservation2col, @@ -120,11 +123,10 @@ def run_model( for node in pmapping.nodes: if isinstance(node, TensorHolder): for tensor in node.tensors: - if tensor not in tensor_to_backing and tensor in job.fusable_tensors: - tensor_to_backing[tensor] = node.component + tensor_to_backing.setdefault(tensor, node.component) - # A Toll is a pass-through and must never be the outermost level backing - # a fusable tensor — that would leave no real Memory holding it. + # A Toll is a pass-through and must never be the outermost level backing a tensor — + # that would leave no real Memory holding it. for node in pmapping.nodes: if isinstance(node, Toll): for tensor in node.tensors: @@ -134,7 +136,7 @@ def run_model( ): raise ValueError( f"Toll '{node.component}' is the outermost level holding " - f"fusable tensor '{tensor}' in einsum '{job.einsum_name}'. " + f"tensor '{tensor}' in einsum '{job.einsum_name}'. " f"A Toll cannot be the outermost holder of a tensor — a " f"Memory above the Toll must also keep it." ) @@ -144,6 +146,9 @@ def run_model( n_instances = workload.n_instances * workload.einsums[job.einsum_name].n_instances + # ================================================================================= + # Tensor sizes + # ================================================================================= n_loop_options = oset() for buffet, stats in reuse.buffet_stats.items(): if buffet.level == compute_unit: @@ -157,6 +162,8 @@ def run_model( occupancy *= n_instances for tensor, backing in tensor_to_backing.items(): + if tensor not in job.fusable_tensors: + continue if (is_copy_op or buffet.tensor == tensor) and buffet.level == backing: df[tensor2col(tensor)] = occupancy / memory_to_size[buffet.level] @@ -168,6 +175,9 @@ def run_model( key = memory_usage2col(buffet.level, buffet.tensor) df[key] = occupancy / memory_to_size[buffet.level] + # ================================================================================= + # Reservations + # ================================================================================= for memory, occupancies in total_occupancy.items(): if memory not in job.memories_track_all: continue @@ -184,6 +194,9 @@ def run_model( f"storage nodes of {memory}." ) + # ================================================================================= + # Actions + # ================================================================================= if metrics & Metrics.ACTIONS: detailed_actions = gather_actions( reuse, None, verbose=True, use_name=True, spec=spec @@ -203,10 +216,26 @@ def run_model( for key, count in simple_actions.items(): actions_df[action2col(key)] = count.total * n_instances + # ================================================================================= + # Latency + # ================================================================================= if metrics.includes_latency(): for component, cur_latency in latency.items(): df[f"component_latency{component}"] = cur_latency * n_instances - df["Totallatency"] = overall_latency * n_instances + + # Total latency of each component is its own total_latency plus the worst + # communication latency to reach it. + comm_latency = communication_latency( + reuse, + job.flattened_arch, + tensor_to_backing, + workload.einsums[job.einsum_name].output_tensor_names + ) + per_component_total = [ + latency.get(component, 0) + comm_latency.get(component, 0) + for component in oset(latency) | oset(comm_latency) + ] + df["Totallatency"] = max_nonzero(*per_component_total) * n_instances # For first latency, we'll follow the convention of treating compute # as a component, similarly to memory (see below). for compute_level, stats in reuse.compute_stats.items(): # FIRST LATENCY @@ -215,6 +244,9 @@ def run_model( max_first_latency * n_instances ) + # ================================================================================= + # Energy + # ================================================================================= if metrics.includes_dynamic_energy(): dynamic_energy = [e for k, e in energy.items() if k.action != "leak"] df["Totaldynamic_energy"] = sum(dynamic_energy) * n_instances @@ -223,6 +255,9 @@ def run_model( leak_energy = [e for k, e in energy.items() if k.action == "leak"] df["Totalleak_energy"] = sum(leak_energy) * n_instances + # ================================================================================= + # Memory usage + # ================================================================================= per_memory_spatial_usage_df = {} for memory, occupancies in total_occupancy.items(): ignored = job.ignored_resources is not None and memory in job.ignored_resources diff --git a/accelforge/util/_migration.py b/accelforge/util/_migration.py deleted file mode 100644 index 04dc0983..00000000 --- a/accelforge/util/_migration.py +++ /dev/null @@ -1,7 +0,0 @@ -LATENCY_TO_THROUGHPUT_MIGRATION = """ -AccelForge has migrated from `latency` to `throughput` to measure action timing. To -migrate, please change all `latency` expressions to `throughput` value being the -reciprocal of the latency. Additionally, `latency_scale` should be changed to -`throughput_scale`. Finally, `total_latency` is still present, but supported expressions -have changed; please refer to its docstring for more information. -""" diff --git a/examples/arches/compute_in_memory/colonnade_jssc_2021.yaml b/examples/arches/compute_in_memory/colonnade_jssc_2021.yaml index 360d503d..a76e2bc9 100644 --- a/examples/arches/compute_in_memory/colonnade_jssc_2021.yaml +++ b/examples/arches/compute_in_memory/colonnade_jssc_2021.yaml @@ -164,7 +164,7 @@ arch: bits_per_value: weight: 1 output: n_input_slices - actions: [{name: read, throughput: 1 / cycle_period}] + actions: [{name: read, throughput: 1 / cycle_period, latency: 0}] # Each group of rows has a register for pipelining. Rows share outputs. - !Container diff --git a/examples/arches/compute_in_memory/jia_jssc_2020.yaml b/examples/arches/compute_in_memory/jia_jssc_2020.yaml index 9deac072..71854b95 100644 --- a/examples/arches/compute_in_memory/jia_jssc_2020.yaml +++ b/examples/arches/compute_in_memory/jia_jssc_2020.yaml @@ -185,7 +185,7 @@ arch: weight: weight.bits_per_value * (20 / 50) * max_weight_bits_per_slice output: output.bits_per_value / encoded_output_bits input: 0 - actions: [{name: read, latency: cycle_period}] + actions: [{name: read, throughput: 1 / cycle_period, latency: 0}] # Each row receives a different input slice. Rows share outputs. - !Container diff --git a/examples/arches/compute_in_memory/sinangil_jssc_2021.yaml b/examples/arches/compute_in_memory/sinangil_jssc_2021.yaml index 5b03eb21..d767df83 100644 --- a/examples/arches/compute_in_memory/sinangil_jssc_2021.yaml +++ b/examples/arches/compute_in_memory/sinangil_jssc_2021.yaml @@ -159,7 +159,7 @@ arch: bits_per_value: weight: 0.75 output: n_input_slices - actions: [{name: read, latency: cycle_period}] + actions: [{name: read, throughput: 1 / cycle_period, latency: 0}] # Binary-weighted capacitors sum outputs across columns. - !Toll diff --git a/examples/arches/compute_in_memory/wang_vlsi_2022.yaml b/examples/arches/compute_in_memory/wang_vlsi_2022.yaml index 3ee87a93..cf00a8c8 100644 --- a/examples/arches/compute_in_memory/wang_vlsi_2022.yaml +++ b/examples/arches/compute_in_memory/wang_vlsi_2022.yaml @@ -163,7 +163,7 @@ arch: tensors: {keep: output} direction: up component_class: ArrayColumnDrivers - actions: [{name: read, throughput: cols_active_at_once / cycle_period}] + actions: [{name: read, throughput: cols_active_at_once / cycle_period, latency: 0}] # Each column stores a different weight slice. Columns share inputs. - !Container @@ -191,7 +191,7 @@ arch: # time a sliced psum is read from the array, consume 1 "bit". bits_per_value: {weight: 0.5, output: 1} # One cycle period per "bit" - actions: [{name: read, throughput: 1 / cycle_period}] + actions: [{name: read, throughput: 1 / cycle_period, latency: 0}] # Each row receives a different input slice. Rows share outputs. - !Container diff --git a/examples/arches/fanout_variations/at_glb.yaml b/examples/arches/fanout_variations/at_glb.yaml index 010b9dbf..72ff534a 100755 --- a/examples/arches/fanout_variations/at_glb.yaml +++ b/examples/arches/fanout_variations/at_glb.yaml @@ -7,8 +7,8 @@ arch: area: 0 tensors: {keep: ~Intermediates, may_keep: All} actions: - - {name: read, energy: 1, throughput: inf} - - {name: write, energy: 1, throughput: inf} + - {name: read, energy: 1, throughput: inf, latency: 0} + - {name: write, energy: 1, throughput: inf, latency: 0} - !Memory name: GlobalBuffer @@ -20,12 +20,12 @@ arch: - name: X fanout: 4 actions: - - {name: read, energy: 0, throughput: inf} - - {name: write, energy: 0, throughput: inf} + - {name: read, energy: 0, throughput: inf, latency: 0} + - {name: write, energy: 0, throughput: inf, latency: 0} - !Compute name: MAC leak_power: 0 area: 0 actions: - - {name: compute, energy: 0, throughput: 1} + - {name: compute, energy: 0, throughput: 1, latency: 0} diff --git a/examples/arches/fanout_variations/at_glb_with_fanout_node.yaml b/examples/arches/fanout_variations/at_glb_with_fanout_node.yaml index 84b2a4fe..2b44c58d 100755 --- a/examples/arches/fanout_variations/at_glb_with_fanout_node.yaml +++ b/examples/arches/fanout_variations/at_glb_with_fanout_node.yaml @@ -7,8 +7,8 @@ arch: area: 0 tensors: {keep: ~Intermediates, may_keep: All} actions: - - {name: read, energy: 1, throughput: inf} - - {name: write, energy: 1, throughput: inf} + - {name: read, energy: 1, throughput: inf, latency: 0} + - {name: write, energy: 1, throughput: inf, latency: 0} - !Container name: GlobalBufferArray @@ -23,12 +23,12 @@ arch: area: 0 tensors: {keep: All} actions: - - {name: read, energy: 0, throughput: inf} - - {name: write, energy: 0, throughput: inf} + - {name: read, energy: 0, throughput: inf, latency: 0} + - {name: write, energy: 0, throughput: inf, latency: 0} - !Compute name: MAC leak_power: 0 area: 0 actions: - - {name: compute, energy: 0, throughput: 1} + - {name: compute, energy: 0, throughput: 1, latency: 0} diff --git a/examples/arches/fanout_variations/at_mac.yaml b/examples/arches/fanout_variations/at_mac.yaml index ba2bbe7c..824ac3ea 100755 --- a/examples/arches/fanout_variations/at_mac.yaml +++ b/examples/arches/fanout_variations/at_mac.yaml @@ -7,8 +7,8 @@ arch: area: 0 tensors: {keep: ~Intermediates, may_keep: All} actions: - - {name: read, energy: 1, throughput: inf} - - {name: write, energy: 1, throughput: inf} + - {name: read, energy: 1, throughput: inf, latency: 0} + - {name: write, energy: 1, throughput: inf, latency: 0} - !Memory name: GlobalBuffer @@ -17,8 +17,8 @@ arch: area: 0 tensors: {keep: All} actions: - - {name: read, energy: 0, throughput: inf} - - {name: write, energy: 0, throughput: inf} + - {name: read, energy: 0, throughput: inf, latency: 0} + - {name: write, energy: 0, throughput: inf, latency: 0} - !Compute name: MAC @@ -28,4 +28,4 @@ arch: - name: X fanout: 4 actions: - - {name: compute, energy: 0, throughput: 1 / (1)} + - {name: compute, energy: 0, throughput: 1 / (1), latency: 0} diff --git a/examples/arches/fanout_variations/at_mac_with_constraints.yaml b/examples/arches/fanout_variations/at_mac_with_constraints.yaml index 714cbc16..60c83fbd 100755 --- a/examples/arches/fanout_variations/at_mac_with_constraints.yaml +++ b/examples/arches/fanout_variations/at_mac_with_constraints.yaml @@ -7,8 +7,8 @@ arch: area: 0 tensors: {keep: ~Intermediates, may_keep: All} actions: - - {name: read, energy: 1, throughput: inf} - - {name: write, energy: 1, throughput: inf} + - {name: read, energy: 1, throughput: inf, latency: 0} + - {name: write, energy: 1, throughput: inf, latency: 0} - !Memory name: GlobalBuffer @@ -17,8 +17,8 @@ arch: area: 0 tensors: {keep: All} actions: - - {name: read, energy: 0, throughput: inf} - - {name: write, energy: 0, throughput: inf} + - {name: read, energy: 0, throughput: inf, latency: 0} + - {name: write, energy: 0, throughput: inf, latency: 0} - !Container name: MACArray @@ -35,4 +35,4 @@ arch: leak_power: 0 area: 0 actions: - - {name: compute, energy: 0, throughput: 1} + - {name: compute, energy: 0, throughput: 1, latency: 0} diff --git a/examples/arches/fanout_variations/at_mac_with_fanout_node.yaml b/examples/arches/fanout_variations/at_mac_with_fanout_node.yaml index ddd64279..1943050a 100755 --- a/examples/arches/fanout_variations/at_mac_with_fanout_node.yaml +++ b/examples/arches/fanout_variations/at_mac_with_fanout_node.yaml @@ -7,8 +7,8 @@ arch: area: 0 tensors: {keep: ~Intermediates, may_keep: All} actions: - - {name: read, energy: 1, throughput: inf} - - {name: write, energy: 1, throughput: inf} + - {name: read, energy: 1, throughput: inf, latency: 0} + - {name: write, energy: 1, throughput: inf, latency: 0} - !Memory name: GlobalBuffer @@ -17,8 +17,8 @@ arch: area: 0 tensors: {keep: All} actions: - - {name: read, energy: 0, throughput: inf} - - {name: write, energy: 0, throughput: inf} + - {name: read, energy: 0, throughput: inf, latency: 0} + - {name: write, energy: 0, throughput: inf, latency: 0} - !Container name: MACArray @@ -31,4 +31,4 @@ arch: leak_power: 0 area: 0 actions: - - {name: compute, energy: 0, throughput: 1 / (1)} + - {name: compute, energy: 0, throughput: 1 / (1), latency: 0} diff --git a/examples/arches/nvdla.yaml b/examples/arches/nvdla.yaml index 6ac4092a..88290464 100755 --- a/examples/arches/nvdla.yaml +++ b/examples/arches/nvdla.yaml @@ -9,8 +9,8 @@ arch: # their reference, and they said it left out some things. Latency is 38.4 GB/s. # DDR5-4800. Chip runs at at 1GHz, so divide to get per-cycle bandwidth. # https://www.jedec.org/news/pressreleases/jedec-updates-standard-low-power-memory-devices-lpddr5 - - {name: read, energy: 7.03e-12, throughput: 8 * 38.4e9} - - {name: write, energy: 7.03e-12, throughput: 8 * 38.4e9} + - {name: read, energy: 7.03e-12, throughput: 8 * 38.4e9, latency: 0} + - {name: write, energy: 7.03e-12, throughput: 8 * 38.4e9, latency: 0} tensors: {keep: ~Intermediates, may_keep: All} - !Memory @@ -20,8 +20,8 @@ arch: leak_power: 0 actions: # 512 GB/s read, 128 GB/s write - - {name: read, energy: 0.249e-12, throughput: 512e9 * 8} - - {name: write, energy: 0.293e-12, throughput: 128e9 * 8} + - {name: read, energy: 0.249e-12, throughput: 512e9 * 8, latency: 0} + - {name: write, energy: 0.293e-12, throughput: 128e9 * 8, latency: 0} tensors: {keep: All} - !Container @@ -40,4 +40,4 @@ arch: name: MAC leak_power: 0 actions: - - {name: compute, energy: 0.084e-12, throughput: 1e9} + - {name: compute, energy: 0.084e-12, throughput: 1e9, latency: 0} diff --git a/examples/arches/simple.yaml b/examples/arches/simple.yaml index 162178a5..21a2c0d5 100755 --- a/examples/arches/simple.yaml +++ b/examples/arches/simple.yaml @@ -11,8 +11,8 @@ arch: area: 0 tensors: {keep: ~Intermediates, may_keep: All} actions: - - {name: read, energy: {{MainMemoryEnergy}}, throughput: inf} - - {name: write, energy: {{MainMemoryEnergy}}, throughput: inf} + - {name: read, energy: {{MainMemoryEnergy}}, throughput: inf, latency: 0} + - {name: write, energy: {{MainMemoryEnergy}}, throughput: inf, latency: 0} - !Memory name: GlobalBuffer @@ -21,12 +21,12 @@ arch: area: 0 tensors: {keep: ~MainMemory, may_keep: All} actions: - - {name: read, energy: 1, throughput: {{GlobalBufferThroughput}}} - - {name: write, energy: 1, throughput: {{GlobalBufferThroughput}}} + - {name: read, energy: 1, throughput: {{GlobalBufferThroughput}}, latency: 0} + - {name: write, energy: 1, throughput: {{GlobalBufferThroughput}}, latency: 0} - !Compute name: MAC leak_power: 0 area: 0 actions: - - {name: compute, energy: 1, throughput: 1} + - {name: compute, energy: 1, throughput: 1, latency: 0} diff --git a/examples/arches/tpu_v4i.yaml b/examples/arches/tpu_v4i.yaml index a07bc815..c95c83c6 100755 --- a/examples/arches/tpu_v4i.yaml +++ b/examples/arches/tpu_v4i.yaml @@ -16,8 +16,8 @@ arch: actions: # Upper end of the range from the TPU paper. The lower end came from their # reference, and they said it left out some things. - - {name: read, energy: 7.03e-12, throughput: 8 * 614e9} - - {name: write, energy: 7.03e-12, throughput: 8 * 614e9} + - {name: read, energy: 7.03e-12, throughput: 8 * 614e9, latency: 0} + - {name: write, energy: 7.03e-12, throughput: 8 * 614e9, latency: 0} tensors: {keep: ~Intermediates, may_keep: All} - !Memory @@ -27,8 +27,8 @@ arch: leak_power: 0 area: 112e-6 # From paper fig. 6 actions: - - {name: read, energy: 1.88e-12, throughput: 8 * 2048e9} - - {name: write, energy: 2.36e-12, throughput: 8 * 1024e9} + - {name: read, energy: 1.88e-12, throughput: 8 * 2048e9, latency: 0} + - {name: write, energy: 2.36e-12, throughput: 8 * 1024e9, latency: 0} tensors: {keep: ~MainMemory.tensors, may_keep: All} - !Memory @@ -38,8 +38,8 @@ arch: leak_power: 0 area: 51e-6 # From paper fig. 6 actions: - - {name: read, energy: 0.249e-12, throughput: inf} - - {name: write, energy: 0.293e-12, throughput: inf} + - {name: read, energy: 0.249e-12, throughput: inf, latency: 0} + - {name: write, energy: 0.293e-12, throughput: inf, latency: 0} tensors: {keep: input | output} - !Compute @@ -47,7 +47,7 @@ arch: area: 8.7e-6 # From die photo. leak_power: 0 actions: - - {name: compute, energy: 0, throughput: 1.05e9 * 128} + - {name: compute, energy: 0, throughput: 1.05e9 * 128, latency: 0} enabled: len(All) == 2 - !Container @@ -62,8 +62,8 @@ arch: area: 8.5e-11 # Fig. 6 has 14mm^2 for a 128x128. Assume regs are 10%. leak_power: 0 actions: - - {name: read, energy: 0, throughput: inf} - - {name: write, energy: 0, throughput: inf} + - {name: read, energy: 0, throughput: inf, latency: 0} + - {name: write, energy: 0, throughput: inf, latency: 0} tensors: {keep: weight} - !Compute @@ -71,5 +71,5 @@ arch: leak_power: 0 area: 8.5e-10 # Fig. 6 has 14mm^2 for a 128x128. Assume MAC is 90%. actions: - - {name: compute, energy: 0.084e-12, throughput: 1.05e9} + - {name: compute, energy: 0.084e-12, throughput: 1.05e9, latency: 0} enabled: len(All) == 3 diff --git a/tests/input_files/toll.arch.yaml b/tests/input_files/toll.arch.yaml index beaaef28..d660bded 100644 --- a/tests/input_files/toll.arch.yaml +++ b/tests/input_files/toll.arch.yaml @@ -13,7 +13,7 @@ arch: area: 0 tensors: {keep: All} actions: - - {name: read, energy: 100, latency: 0} + - {name: read, energy: 100, throughput: inf, latency: 0} - !Memory name: LocalBuffer diff --git a/tests/input_files/toll_no_outer.arch.yaml b/tests/input_files/toll_no_outer.arch.yaml index 2b8cdb39..4d9ba9a9 100644 --- a/tests/input_files/toll_no_outer.arch.yaml +++ b/tests/input_files/toll_no_outer.arch.yaml @@ -16,7 +16,7 @@ arch: area: 0 tensors: {keep: All} actions: - - {name: read, energy: 100, latency: 0} + - {name: read, energy: 100, throughput: inf, latency: 0} - !Memory name: LocalBuffer diff --git a/tests/network/input_files/networked/flat.yaml b/tests/network/input_files/networked/flat.yaml index e748ab0a..8ff9e085 100644 --- a/tests/network/input_files/networked/flat.yaml +++ b/tests/network/input_files/networked/flat.yaml @@ -7,8 +7,8 @@ arch: leak_power: 0 tensors: {keep: All} actions: - - {name: read, energy: 0, throughput: inf} - - {name: write, energy: 0, throughput: inf} + - {name: read, energy: 0, throughput: inf, latency: 0} + - {name: write, energy: 0, throughput: inf, latency: 0} - !Array name: Array @@ -23,8 +23,8 @@ arch: leak_power: 0 tensors: {keep: ~MainMemory, may_keep: All} actions: - - {name: read, energy: 0, throughput: inf} - - {name: write, energy: 0, throughput: inf} + - {name: read, energy: 0, throughput: inf, latency: 0} + - {name: write, energy: 0, throughput: inf, latency: 0} - !Memory name: RowBuffer @@ -33,8 +33,8 @@ arch: leak_power: 0 tensors: {keep: input, may_keep: input} actions: - - {name: read, energy: 5, throughput: 1} - - {name: write, energy: 5, throughput: inf} + - {name: read, energy: 5, throughput: 1, latency: 0} + - {name: write, energy: 5, throughput: inf, latency: 0} spatial: - {name: X, fanout: 4} @@ -45,8 +45,8 @@ arch: leak_power: 0 tensors: {keep: output, may_keep: output} actions: - - {name: read, energy: 5, throughput: inf} - - {name: write, energy: 5, throughput: inf} + - {name: read, energy: 5, throughput: inf, latency: 0} + - {name: write, energy: 5, throughput: inf, latency: 0} spatial: - {name: Y, fanout: 4} @@ -57,8 +57,8 @@ arch: leak_power: 0 tensors: {keep: weight, may_keep: weight} actions: - - {name: read, energy: 5, throughput: 1} - - {name: write, energy: 5, throughput: 1} + - {name: read, energy: 5, throughput: 1, latency: 0} + - {name: write, energy: 5, throughput: 1, latency: 0} spatial: - {name: X, fanout: 2} - {name: Y, fanout: 2} @@ -77,12 +77,12 @@ arch: leak_power: 0 tensors: {keep: weight, may_keep: weight} actions: - - {name: read, energy: 1, throughput: inf} - - {name: write, energy: 1, throughput: inf} + - {name: read, energy: 1, throughput: inf, latency: 0} + - {name: write, energy: 1, throughput: inf, latency: 0} - !Compute name: MAC area: 0 leak_power: 0 actions: - - {name: compute, energy: 1, throughput: inf} + - {name: compute, energy: 1, throughput: inf, latency: 0} diff --git a/tests/network/input_files/networked/hierarchical.yaml b/tests/network/input_files/networked/hierarchical.yaml index a6327b76..0d7805aa 100644 --- a/tests/network/input_files/networked/hierarchical.yaml +++ b/tests/network/input_files/networked/hierarchical.yaml @@ -7,8 +7,8 @@ arch: leak_power: 0 tensors: {keep: All} actions: - - {name: read, energy: 0, throughput: 1e9} - - {name: write, energy: 0, throughput: 1e9} + - {name: read, energy: 0, throughput: 1e9, latency: 0} + - {name: write, energy: 0, throughput: 1e9, latency: 0} - !Memory name: GlobalBuffer @@ -17,8 +17,8 @@ arch: leak_power: 0 tensors: {keep: ~MainMemory, may_keep: All} actions: - - {name: read, energy: 0, throughput: 4e9} - - {name: write, energy: 0, throughput: 4e9} + - {name: read, energy: 0, throughput: 4e9, latency: 0} + - {name: write, energy: 0, throughput: 4e9, latency: 0} - !Network name: PeArray @@ -34,8 +34,8 @@ arch: leak_power: 0 tensors: {keep: All} actions: - - {name: read, energy: 0, throughput: 16e9} - - {name: write, energy: 0, throughput: 16e9} + - {name: read, energy: 0, throughput: 16e9, latency: 0} + - {name: write, energy: 0, throughput: 16e9, latency: 0} spatial: - {name: X, fanout: 2} - {name: Y, fanout: 2} @@ -52,7 +52,7 @@ arch: area: 0 leak_power: 0 actions: - - {name: compute, energy: 0, throughput: 1e9} + - {name: compute, energy: 0, throughput: 1e9, latency: 0} spatial: - {name: X, fanout: 2} - {name: Y, fanout: 2} \ No newline at end of file diff --git a/tests/network/input_files/networked/hierarchical_1d.yaml b/tests/network/input_files/networked/hierarchical_1d.yaml index e9a3ae49..55b7b555 100644 --- a/tests/network/input_files/networked/hierarchical_1d.yaml +++ b/tests/network/input_files/networked/hierarchical_1d.yaml @@ -7,8 +7,8 @@ arch: leak_power: 0 tensors: {keep: All} actions: - - {name: read, energy: 0, throughput: inf} - - {name: write, energy: 0, throughput: inf} + - {name: read, energy: 0, throughput: inf, latency: 0} + - {name: write, energy: 0, throughput: inf, latency: 0} - !Memory name: GlobalBuffer @@ -17,14 +17,13 @@ arch: leak_power: 0 tensors: {keep: ~MainMemory, may_keep: All} actions: - - {name: read, energy: 0, throughput: inf} - - {name: write, energy: 0, throughput: inf} + - {name: read, energy: 0, throughput: inf, latency: 0} + - {name: write, energy: 0, throughput: inf, latency: 0} - !Network name: PeArray area: 0 leak_power: 0 - total_latency: "max_hops" actions: - {name: hop, energy: 1, latency: 0, throughput: 1} @@ -35,8 +34,8 @@ arch: leak_power: 0 tensors: {keep: All} actions: - - {name: read, energy: 0, throughput: inf} - - {name: write, energy: 0, throughput: inf} + - {name: read, energy: 0, throughput: inf, latency: 0} + - {name: write, energy: 0, throughput: inf, latency: 0} spatial: - {name: X, fanout: 4} @@ -52,6 +51,6 @@ arch: area: 0 leak_power: 0 actions: - - {name: compute, energy: 0, throughput: inf} + - {name: compute, energy: 0, throughput: inf, latency: 0} spatial: - {name: X, fanout: 2} \ No newline at end of file diff --git a/tests/network/input_files/networked/hierarchical_1d_all_to_all.yaml b/tests/network/input_files/networked/hierarchical_1d_all_to_all.yaml index 3ecf4892..59a8e7a3 100644 --- a/tests/network/input_files/networked/hierarchical_1d_all_to_all.yaml +++ b/tests/network/input_files/networked/hierarchical_1d_all_to_all.yaml @@ -7,8 +7,8 @@ arch: leak_power: 0 tensors: {keep: All} actions: - - {name: read, energy: 0, throughput: inf} - - {name: write, energy: 0, throughput: inf} + - {name: read, energy: 0, throughput: inf, latency: 0} + - {name: write, energy: 0, throughput: inf, latency: 0} - !Memory name: GlobalBuffer @@ -17,14 +17,13 @@ arch: leak_power: 0 tensors: {keep: ~MainMemory, may_keep: All} actions: - - {name: read, energy: 0, throughput: inf} - - {name: write, energy: 0, throughput: inf} + - {name: read, energy: 0, throughput: inf, latency: 0} + - {name: write, energy: 0, throughput: inf, latency: 0} - !Network name: PeArray area: 0 leak_power: 0 - total_latency: "max_hops" actions: - {name: hop, energy: 1, latency: 1, throughput: inf} @@ -35,8 +34,8 @@ arch: leak_power: 0 tensors: {keep: All} actions: - - {name: read, energy: 0, throughput: inf} - - {name: write, energy: 0, throughput: inf} + - {name: read, energy: 0, throughput: inf, latency: 0} + - {name: write, energy: 0, throughput: inf, latency: 0} spatial: - {name: X, fanout: 4} @@ -54,6 +53,6 @@ arch: area: 0 leak_power: 0 actions: - - {name: compute, energy: 0, throughput: inf} + - {name: compute, energy: 0, throughput: inf, latency: 0} spatial: - {name: X, fanout: 4} diff --git a/tests/network/input_files/networked/hierarchical_switched.yaml b/tests/network/input_files/networked/hierarchical_switched.yaml index 69752bf3..c7fcfc94 100644 --- a/tests/network/input_files/networked/hierarchical_switched.yaml +++ b/tests/network/input_files/networked/hierarchical_switched.yaml @@ -7,8 +7,8 @@ arch: leak_power: 0 tensors: {keep: All} actions: - - {name: read, energy: 100, latency: 1e-9} - - {name: write, energy: 100, latency: 1e-9} + - {name: read, energy: 100, throughput: 1/(1e-9), latency: 0} + - {name: write, energy: 100, throughput: 1/(1e-9), latency: 0} - !Memory name: GlobalBuffer @@ -17,15 +17,15 @@ arch: leak_power: 0 tensors: {keep: ~MainMemory, may_keep: All} actions: - - {name: read, energy: 10, latency: 1e-9/4} - - {name: write, energy: 10, latency: 1e-9/4} + - {name: read, energy: 10, throughput: 1/(1e-9/4), latency: 0} + - {name: write, energy: 10, throughput: 1/(1e-9/4), latency: 0} - !Network name: PeArray area: 0 leak_power: 0 actions: - - {name: hop, energy: 5, latency: 1e-9/4} + - {name: hop, energy: 5, throughput: 1/(1e-9/4), latency: 0} - !Memory name: Scratchpad @@ -34,8 +34,8 @@ arch: leak_power: 0 tensors: {keep: All} actions: - - {name: read, energy: 2, latency: 1e-9/16} - - {name: write, energy: 2, latency: 1e-9/16} + - {name: read, energy: 2, throughput: 1/(1e-9/16), latency: 0} + - {name: write, energy: 2, throughput: 1/(1e-9/16), latency: 0} spatial: - {name: X, fanout: 2} - {name: Y, fanout: 2} @@ -45,14 +45,14 @@ arch: area: 0 leak_power: 0 actions: - - {name: hop, energy: 1, latency: 1e-9/16} + - {name: hop, energy: 1, throughput: 1/(1e-9/16), latency: 0} - !Compute name: MAC area: 0 leak_power: 0 actions: - - {name: compute, energy: 1, latency: 1e-9} + - {name: compute, energy: 1, throughput: 1/(1e-9), latency: 0} spatial: - {name: X, fanout: 2} - {name: Y, fanout: 2} \ No newline at end of file diff --git a/tests/network/test_network.py b/tests/network/test_network.py index 000861f7..ed3b7829 100644 --- a/tests/network/test_network.py +++ b/tests/network/test_network.py @@ -111,7 +111,10 @@ def test_hierarchical_1d(self): * KN * BITS_PER_VALUE, ) - self.assertEqual(result.data["Totallatency"].iloc[0], 4) + # PeArray is bandwidth-bound at 1 bit/s: its most congested link carries + # T0 (3 * 64) + W0 (3 * 128) + T1 (256) = 832 bits. Wind-up/down adds the + # 2 MacArray hops each way (PeArray hops have zero latency). + self.assertEqual(result.data["Totallatency"].iloc[0], 832 + 4) def test_hierarchical(self): M = 8 @@ -383,15 +386,10 @@ def test_hierarchical_1d_all_to_all(self): ) # --- Latency ------------------------------------------------------ - # The switch's uniform single-hop routing gives MacArray a constant - # latency of 1, versus the mesh PeArray's 2. - self.assertEqual( - result.data["Matmul0component_latencyMacArray"].iloc[0], 1 - ) - self.assertEqual( - result.data["Matmul0component_latencyPeArray"].iloc[0], 2 - ) - self.assertEqual(result.data["Totallatency"].iloc[0], 2) + # Network hops surface as communication latency: the worst input winds + # down to compute (2 mesh + 1 all-to-all hops), then the output winds + # back up to MainMemory (1 + 2 hops) = 6. + self.assertEqual(result.data["Totallatency"].iloc[0], 6) class TestMapper(TestCase): diff --git a/tests/vibe_see_readme_in_this_dir/test_api_gaps.py b/tests/vibe_see_readme_in_this_dir/test_api_gaps.py index 4373b1fc..4f5564fe 100644 --- a/tests/vibe_see_readme_in_this_dir/test_api_gaps.py +++ b/tests/vibe_see_readme_in_this_dir/test_api_gaps.py @@ -306,14 +306,14 @@ def test_on_memory_from_yaml(self): tile_shape: - {expression: "~m", operator: "<=", value: 64} actions: - - {name: read, energy: 1, latency: 0} - - {name: write, energy: 1, latency: 0} + - {name: read, energy: 1, throughput: .inf, latency: 0} + - {name: write, energy: 1, throughput: .inf, latency: 0} - !Compute name: MAC leak_power: 0 area: 0 actions: - - {name: compute, energy: 1, latency: 1} + - {name: compute, energy: 1, throughput: 1, latency: 0} """ ) mem = spec.arch.find("Mem") @@ -345,14 +345,14 @@ def _make_arch_with_totals(self): area: 100 tensors: {keep: All} actions: - - {name: read, energy: 1, latency: 0} - - {name: write, energy: 1, latency: 0} + - {name: read, energy: 1, throughput: .inf, latency: 0} + - {name: write, energy: 1, throughput: .inf, latency: 0} - !Compute name: MAC leak_power: 0.1 area: 10 actions: - - {name: compute, energy: 1, latency: 1} + - {name: compute, energy: 1, throughput: 1, latency: 0} workload: rank_sizes: {M: 16} @@ -402,14 +402,14 @@ def test_raises_without_calculation(self): leak_power: 0 area: 0 actions: - - {name: read, energy: 1, latency: 0} - - {name: write, energy: 1, latency: 0} + - {name: read, energy: 1, throughput: .inf, latency: 0} + - {name: write, energy: 1, throughput: .inf, latency: 0} - !Compute name: MAC leak_power: 0 area: 0 actions: - - {name: compute, energy: 1, latency: 1} + - {name: compute, energy: 1, throughput: 1, latency: 0} """ ) # total_area depends on total_area being set per-component @@ -438,14 +438,14 @@ def _make_spec(self): area: 100 tensors: {keep: All} actions: - - {name: read, energy: 2.0, latency: 10} - - {name: write, energy: 3.0, latency: 15} + - {name: read, energy: 2.0, throughput: 1/10, latency: 0} + - {name: write, energy: 3.0, throughput: 1/15, latency: 0} - !Compute name: MAC leak_power: 0.1 area: 10 actions: - - {name: compute, energy: 0.5, latency: 1} + - {name: compute, energy: 0.5, throughput: 1, latency: 0} workload: rank_sizes: {M: 16} @@ -501,7 +501,7 @@ def test_selective_area_only(self): def test_noop_if_all_false(self): spec = self._make_spec() result = spec.calculate_component_costs( - area=False, energy=False, throughput=False, leak=False + area=False, energy=False, throughput=False, latency=False, leak=False ) self.assertIsInstance(result, Spec) @@ -518,8 +518,8 @@ def test_with_fanout_multiplies_area(self): area: 0 tensors: {keep: All} actions: - - {name: read, energy: 1, latency: 0} - - {name: write, energy: 1, latency: 0} + - {name: read, energy: 1, throughput: .inf, latency: 0} + - {name: write, energy: 1, throughput: .inf, latency: 0} - !Container name: Array spatial: @@ -529,7 +529,7 @@ def test_with_fanout_multiplies_area(self): leak_power: 0.1 area: 10 actions: - - {name: compute, energy: 1, latency: 1} + - {name: compute, energy: 1, throughput: 1, latency: 0} workload: rank_sizes: {M: 16} @@ -568,7 +568,7 @@ def test_component_latency_scale_default(self): name="MAC", leak_power=0, area=0, - actions=[{"name": "compute", "energy": 1, "latency": 1}], + actions=[{"name": "compute", "energy": 1, "throughput": 1, "latency": 0}], ) self.assertEqual(c.latency_scale, 1) @@ -587,7 +587,7 @@ def test_component_total_latency_default(self): name="MAC", leak_power=0, area=0, - actions=[{"name": "compute", "energy": 1, "latency": 1}], + actions=[{"name": "compute", "energy": 1, "throughput": 1, "latency": 0}], ) # Default total_latency is an expression self.assertIsInstance(c.total_latency, str) @@ -609,8 +609,13 @@ def test_bits_per_value_default(self): leak_power=0, area=0, actions=[ - {"name": "read", "energy": 1, "latency": 0}, - {"name": "write", "energy": 1, "latency": 0}, + {"name": "read", "energy": 1, "throughput": float("inf"), "latency": 0}, + { + "name": "write", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, ], ) self.assertEqual(m.bits_per_value, {}) @@ -623,8 +628,13 @@ def test_bits_per_value_custom(self): area=0, bits_per_value={"All": 16}, actions=[ - {"name": "read", "energy": 1, "latency": 0}, - {"name": "write", "energy": 1, "latency": 0}, + {"name": "read", "energy": 1, "throughput": float("inf"), "latency": 0}, + { + "name": "write", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, ], ) self.assertEqual(m.bits_per_value, {"All": 16}) @@ -637,8 +647,13 @@ def test_bits_per_action_default_none(self): leak_power=0, area=0, actions=[ - {"name": "read", "energy": 1, "latency": 0}, - {"name": "write", "energy": 1, "latency": 0}, + {"name": "read", "energy": 1, "throughput": float("inf"), "latency": 0}, + { + "name": "write", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, ], ) self.assertIsNone(m.bits_per_action) @@ -651,8 +666,13 @@ def test_bits_per_action_custom(self): area=0, bits_per_action=32, actions=[ - {"name": "read", "energy": 1, "latency": 0}, - {"name": "write", "energy": 1, "latency": 0}, + {"name": "read", "energy": 1, "throughput": float("inf"), "latency": 0}, + { + "name": "write", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, ], ) self.assertEqual(m.bits_per_action, 32) @@ -818,7 +838,7 @@ def test_excludes_none_values(self): name="MAC", leak_power=0, area=0, - actions=[{"name": "compute", "energy": 1, "latency": 1}], + actions=[{"name": "compute", "energy": 1, "throughput": 1, "latency": 0}], ) dump = c.model_dump_non_none() for k, v in dump.items(): @@ -829,7 +849,7 @@ def test_includes_non_none_values(self): name="MAC", leak_power=0, area=0, - actions=[{"name": "compute", "energy": 1, "latency": 1}], + actions=[{"name": "compute", "energy": 1, "throughput": 1, "latency": 0}], ) dump = c.model_dump_non_none() self.assertEqual(dump["name"], "MAC") @@ -844,7 +864,7 @@ def test_returns_dict(self): name="MAC", leak_power=0, area=0, - actions=[{"name": "compute", "energy": 1, "latency": 1}], + actions=[{"name": "compute", "energy": 1, "throughput": 1, "latency": 0}], ) dump = c.shallow_model_dump() self.assertIsInstance(dump, dict) @@ -855,7 +875,7 @@ def test_excludes_none_by_default(self): name="MAC", leak_power=0, area=0, - actions=[{"name": "compute", "energy": 1, "latency": 1}], + actions=[{"name": "compute", "energy": 1, "throughput": 1, "latency": 0}], ) dump = c.shallow_model_dump() for k, v in dump.items(): @@ -866,7 +886,7 @@ def test_include_none(self): name="MAC", leak_power=0, area=0, - actions=[{"name": "compute", "energy": 1, "latency": 1}], + actions=[{"name": "compute", "energy": 1, "throughput": 1, "latency": 0}], ) dump = c.shallow_model_dump(include_None=True) # Should include fields that are None @@ -997,7 +1017,7 @@ def test_default_none(self): name="MAC", leak_power=0, area=0, - actions=[{"name": "compute", "energy": 1, "latency": 1}], + actions=[{"name": "compute", "energy": 1, "throughput": 1, "latency": 0}], ) self.assertIsNone(c.component_class) @@ -1007,7 +1027,7 @@ def test_custom_class(self): leak_power=0, area=0, component_class="MyCustomMAC", - actions=[{"name": "compute", "energy": 1, "latency": 1}], + actions=[{"name": "compute", "energy": 1, "throughput": 1, "latency": 0}], ) self.assertEqual(c.component_class, "MyCustomMAC") @@ -1021,8 +1041,13 @@ def test_get_component_class_raises_when_none(self): leak_power=0, area=0, actions=[ - {"name": "read", "energy": 1, "latency": 0}, - {"name": "write", "energy": 1, "latency": 0}, + {"name": "read", "energy": 1, "throughput": float("inf"), "latency": 0}, + { + "name": "write", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, ], ) with self.assertRaises(EvaluationError): @@ -1037,8 +1062,13 @@ def test_get_component_class_when_set(self): area=0, component_class="smartbuffer_SRAM", actions=[ - {"name": "read", "energy": 1, "latency": 0}, - {"name": "write", "energy": 1, "latency": 0}, + {"name": "read", "energy": 1, "throughput": float("inf"), "latency": 0}, + { + "name": "write", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, ], ) cc = m.get_component_class() diff --git a/tests/vibe_see_readme_in_this_dir/test_arch_flattening.py b/tests/vibe_see_readme_in_this_dir/test_arch_flattening.py index d02980c9..accd6ce2 100644 --- a/tests/vibe_see_readme_in_this_dir/test_arch_flattening.py +++ b/tests/vibe_see_readme_in_this_dir/test_arch_flattening.py @@ -39,8 +39,18 @@ def _simple_arch(): name="MainMemory", size=1_000_000, actions=[ - {"name": "read", "energy": 1, "latency": 0}, - {"name": "write", "energy": 1, "latency": 0}, + { + "name": "read", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, + { + "name": "write", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, ], leak_power=0, area=0, @@ -49,15 +59,27 @@ def _simple_arch(): name="GlobalBuffer", size=100_000, actions=[ - {"name": "read", "energy": 0.5, "latency": 0}, - {"name": "write", "energy": 0.5, "latency": 0}, + { + "name": "read", + "energy": 0.5, + "throughput": float("inf"), + "latency": 0, + }, + { + "name": "write", + "energy": 0.5, + "throughput": float("inf"), + "latency": 0, + }, ], leak_power=0, area=0, ), Compute( name="MAC", - actions=[{"name": "compute", "energy": 1, "latency": 1}], + actions=[ + {"name": "compute", "energy": 1, "throughput": 1, "latency": 0} + ], leak_power=0, area=0, ), @@ -91,8 +113,18 @@ def _fork_arch(): name="MainMemory", size=1_000_000, actions=[ - {"name": "read", "energy": 1, "latency": 0}, - {"name": "write", "energy": 1, "latency": 0}, + { + "name": "read", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, + { + "name": "write", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, ], leak_power=0, area=0, @@ -101,8 +133,18 @@ def _fork_arch(): name="GlobalBuffer", size=100_000, actions=[ - {"name": "read", "energy": 0.5, "latency": 0}, - {"name": "write", "energy": 0.5, "latency": 0}, + { + "name": "read", + "energy": 0.5, + "throughput": float("inf"), + "latency": 0, + }, + { + "name": "write", + "energy": 0.5, + "throughput": float("inf"), + "latency": 0, + }, ], leak_power=0, area=0, @@ -111,7 +153,14 @@ def _fork_arch(): nodes=[ Compute( name="ScalarUnit", - actions=[{"name": "compute", "energy": 0, "latency": 1}], + actions=[ + { + "name": "compute", + "energy": 0, + "throughput": 1, + "latency": 0, + } + ], leak_power=0, area=0, ), @@ -130,15 +179,27 @@ def _fork_arch(): name="Register", size=8, actions=[ - {"name": "read", "energy": 0, "latency": 0}, - {"name": "write", "energy": 0, "latency": 0}, + { + "name": "read", + "energy": 0, + "throughput": float("inf"), + "latency": 0, + }, + { + "name": "write", + "energy": 0, + "throughput": float("inf"), + "latency": 0, + }, ], leak_power=0, area=0, ), Compute( name="MAC", - actions=[{"name": "compute", "energy": 1, "latency": 1}], + actions=[ + {"name": "compute", "energy": 1, "throughput": 1, "latency": 0} + ], leak_power=0, area=0, ), @@ -426,8 +487,18 @@ def test_duplicate_leaf_name_rejected_by_pydantic(self): name="Mem", size=100, actions=[ - {"name": "read", "energy": 1, "latency": 0}, - {"name": "write", "energy": 1, "latency": 0}, + { + "name": "read", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, + { + "name": "write", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, ], leak_power=0, area=0, @@ -436,15 +507,32 @@ def test_duplicate_leaf_name_rejected_by_pydantic(self): name="Mem", size=200, actions=[ - {"name": "read", "energy": 1, "latency": 0}, - {"name": "write", "energy": 1, "latency": 0}, + { + "name": "read", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, + { + "name": "write", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, ], leak_power=0, area=0, ), Compute( name="MAC", - actions=[{"name": "compute", "energy": 1, "latency": 1}], + actions=[ + { + "name": "compute", + "energy": 1, + "throughput": 1, + "latency": 0, + } + ], leak_power=0, area=0, ), @@ -518,8 +606,18 @@ def _spatial_arch(): name="MainMemory", size=1_000_000, actions=[ - {"name": "read", "energy": 1, "latency": 0}, - {"name": "write", "energy": 1, "latency": 0}, + { + "name": "read", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, + { + "name": "write", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, ], leak_power=0, area=0, @@ -532,15 +630,27 @@ def _spatial_arch(): name="Register", size=8, actions=[ - {"name": "read", "energy": 0, "latency": 0}, - {"name": "write", "energy": 0, "latency": 0}, + { + "name": "read", + "energy": 0, + "throughput": float("inf"), + "latency": 0, + }, + { + "name": "write", + "energy": 0, + "throughput": float("inf"), + "latency": 0, + }, ], leak_power=0, area=0, ), Compute( name="MAC", - actions=[{"name": "compute", "energy": 1, "latency": 1}], + actions=[ + {"name": "compute", "energy": 1, "throughput": 1, "latency": 0} + ], leak_power=0, area=0, ), @@ -556,8 +666,18 @@ def _multi_spatial_arch(): name="MainMemory", size=1_000_000, actions=[ - {"name": "read", "energy": 1, "latency": 0}, - {"name": "write", "energy": 1, "latency": 0}, + { + "name": "read", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, + { + "name": "write", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, ], leak_power=0, area=0, @@ -570,8 +690,18 @@ def _multi_spatial_arch(): name="Register", size=8, actions=[ - {"name": "read", "energy": 0, "latency": 0}, - {"name": "write", "energy": 0, "latency": 0}, + { + "name": "read", + "energy": 0, + "throughput": float("inf"), + "latency": 0, + }, + { + "name": "write", + "energy": 0, + "throughput": float("inf"), + "latency": 0, + }, ], spatial=[{"name": "Y", "fanout": 8}], leak_power=0, @@ -579,7 +709,9 @@ def _multi_spatial_arch(): ), Compute( name="MAC", - actions=[{"name": "compute", "energy": 1, "latency": 1}], + actions=[ + {"name": "compute", "energy": 1, "throughput": 1, "latency": 0} + ], leak_power=0, area=0, ), @@ -595,8 +727,18 @@ def _duplicate_spatial_arch(): name="MainMemory", size=1_000_000, actions=[ - {"name": "read", "energy": 1, "latency": 0}, - {"name": "write", "energy": 1, "latency": 0}, + { + "name": "read", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, + { + "name": "write", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, ], leak_power=0, area=0, @@ -611,7 +753,9 @@ def _duplicate_spatial_arch(): ), Compute( name="MAC", - actions=[{"name": "compute", "energy": 1, "latency": 1}], + actions=[ + {"name": "compute", "energy": 1, "throughput": 1, "latency": 0} + ], leak_power=0, area=0, ), @@ -627,8 +771,18 @@ def _array_spatial_arch(): name="MainMemory", size=1_000_000, actions=[ - {"name": "read", "energy": 1, "latency": 0}, - {"name": "write", "energy": 1, "latency": 0}, + { + "name": "read", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, + { + "name": "write", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, ], leak_power=0, area=0, @@ -641,15 +795,32 @@ def _array_spatial_arch(): name="Register", size=8, actions=[ - {"name": "read", "energy": 0, "latency": 0}, - {"name": "write", "energy": 0, "latency": 0}, + { + "name": "read", + "energy": 0, + "throughput": float("inf"), + "latency": 0, + }, + { + "name": "write", + "energy": 0, + "throughput": float("inf"), + "latency": 0, + }, ], leak_power=0, area=0, ), Compute( name="MAC", - actions=[{"name": "compute", "energy": 1, "latency": 1}], + actions=[ + { + "name": "compute", + "energy": 1, + "throughput": 1, + "latency": 0, + } + ], leak_power=0, area=0, ), diff --git a/tests/vibe_see_readme_in_this_dir/test_component_fields.py b/tests/vibe_see_readme_in_this_dir/test_component_fields.py index f0981c09..299cbcd3 100644 --- a/tests/vibe_see_readme_in_this_dir/test_component_fields.py +++ b/tests/vibe_see_readme_in_this_dir/test_component_fields.py @@ -96,7 +96,7 @@ def test_energy_scale_default(self): name="MAC", leak_power=0, area=0, - actions=[{"name": "compute", "energy": 1, "latency": 1}], + actions=[{"name": "compute", "energy": 1, "throughput": 1, "latency": 0}], ) self.assertEqual(c.energy_scale, 1) @@ -105,7 +105,7 @@ def test_area_scale_default(self): name="MAC", leak_power=0, area=0, - actions=[{"name": "compute", "energy": 1, "latency": 1}], + actions=[{"name": "compute", "energy": 1, "throughput": 1, "latency": 0}], ) self.assertEqual(c.area_scale, 1) @@ -114,7 +114,7 @@ def test_leak_power_scale_default(self): name="MAC", leak_power=0, area=0, - actions=[{"name": "compute", "energy": 1, "latency": 1}], + actions=[{"name": "compute", "energy": 1, "throughput": 1, "latency": 0}], ) self.assertEqual(c.leak_power_scale, 1) @@ -123,7 +123,7 @@ def test_n_parallel_instances_default(self): name="MAC", leak_power=0, area=0, - actions=[{"name": "compute", "energy": 1, "latency": 1}], + actions=[{"name": "compute", "energy": 1, "throughput": 1, "latency": 0}], ) self.assertEqual(c.n_parallel_instances, 1) @@ -132,7 +132,7 @@ def test_custom_scales(self): name="MAC", leak_power=0, area=100, - actions=[{"name": "compute", "energy": 1, "latency": 1}], + actions=[{"name": "compute", "energy": 1, "throughput": 1, "latency": 0}], energy_scale=0.5, area_scale=2, leak_power_scale=0.1, @@ -146,7 +146,7 @@ def test_n_parallel_instances_custom(self): name="MAC", leak_power=0, area=0, - actions=[{"name": "compute", "energy": 1, "latency": 1}], + actions=[{"name": "compute", "energy": 1, "throughput": 1, "latency": 0}], n_parallel_instances=4, ) self.assertEqual(c.n_parallel_instances, 4) @@ -162,8 +162,13 @@ def test_memory_size(self): leak_power=0, area=0, actions=[ - {"name": "read", "energy": 1, "latency": 0}, - {"name": "write", "energy": 1, "latency": 0}, + {"name": "read", "energy": 1, "throughput": float("inf"), "latency": 0}, + { + "name": "write", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, ], ) self.assertEqual(m.size, 1024) @@ -175,8 +180,13 @@ def test_memory_size_inf(self): leak_power=0, area=0, actions=[ - {"name": "read", "energy": 1, "latency": 0}, - {"name": "write", "energy": 1, "latency": 0}, + {"name": "read", "energy": 1, "throughput": float("inf"), "latency": 0}, + { + "name": "write", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, ], ) self.assertEqual(m.size, "inf") @@ -188,8 +198,13 @@ def test_memory_total_latency_default(self): leak_power=0, area=0, actions=[ - {"name": "read", "energy": 1, "latency": 0}, - {"name": "write", "energy": 1, "latency": 0}, + {"name": "read", "energy": 1, "throughput": float("inf"), "latency": 0}, + { + "name": "write", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, ], ) self.assertIsNotNone(m.total_latency) @@ -201,8 +216,8 @@ def test_memory_total_latency_expression(self): leak_power=0, area=0, actions=[ - {"name": "read", "energy": 1, "latency": 5}, - {"name": "write", "energy": 1, "latency": 3}, + {"name": "read", "energy": 1, "throughput": 1 / 5, "latency": 0}, + {"name": "write", "energy": 1, "throughput": 1 / 3, "latency": 0}, ], total_latency="max(read_latency, write_latency)", ) @@ -215,8 +230,13 @@ def test_memory_energy_scale_custom(self): leak_power=0, area=0, actions=[ - {"name": "read", "energy": 1, "latency": 0}, - {"name": "write", "energy": 1, "latency": 0}, + {"name": "read", "energy": 1, "throughput": float("inf"), "latency": 0}, + { + "name": "write", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, ], energy_scale=0.5, ) @@ -229,8 +249,13 @@ def test_memory_area_and_leak(self): leak_power=0.01, area=100, actions=[ - {"name": "read", "energy": 1, "latency": 0}, - {"name": "write", "energy": 1, "latency": 0}, + {"name": "read", "energy": 1, "throughput": float("inf"), "latency": 0}, + { + "name": "write", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, ], ) self.assertEqual(m.area, 100) @@ -261,8 +286,13 @@ def test_tensors_on_memory(self): area=0, tensors={"keep": "~Intermediates", "may_keep": "All"}, actions=[ - {"name": "read", "energy": 1, "latency": 0}, - {"name": "write", "energy": 1, "latency": 0}, + {"name": "read", "energy": 1, "throughput": float("inf"), "latency": 0}, + { + "name": "write", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, ], ) self.assertEqual(m.tensors.keep, "~Intermediates") @@ -430,8 +460,8 @@ def test_toll_with_actions(self): leak_power=0, area=0, actions=[ - {"name": "read", "energy": 0.5, "latency": 1}, - {"name": "write", "energy": 0.3, "latency": 1}, + {"name": "read", "energy": 0.5, "throughput": 1, "latency": 0}, + {"name": "write", "energy": 0.3, "throughput": 1, "latency": 0}, ], ) self.assertEqual(len(t.actions), 2) diff --git a/tests/vibe_see_readme_in_this_dir/test_evaluation.py b/tests/vibe_see_readme_in_this_dir/test_evaluation.py index 534cfc6b..53a4ff40 100644 --- a/tests/vibe_see_readme_in_this_dir/test_evaluation.py +++ b/tests/vibe_see_readme_in_this_dir/test_evaluation.py @@ -55,15 +55,32 @@ def _simple_spec(self, **overrides): name="MainMemory", size=1_000_000, actions=[ - {"name": "read", "energy": 1, "latency": 0}, - {"name": "write", "energy": 1, "latency": 0}, + { + "name": "read", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, + { + "name": "write", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, ], leak_power=0, area=0, ), Compute( name="MAC", - actions=[{"name": "compute", "energy": 1, "latency": 1}], + actions=[ + { + "name": "compute", + "energy": 1, + "throughput": 1, + "latency": 0, + } + ], leak_power=0, area=0, ), @@ -476,5 +493,6 @@ def test_same_tensor_names_per_einsum(self): f"Tensor names differ for {c_einsum.name}", ) + if __name__ == "__main__": unittest.main() diff --git a/tests/vibe_see_readme_in_this_dir/test_parsed_values.py b/tests/vibe_see_readme_in_this_dir/test_parsed_values.py index 2cace143..59a0d840 100644 --- a/tests/vibe_see_readme_in_this_dir/test_parsed_values.py +++ b/tests/vibe_see_readme_in_this_dir/test_parsed_values.py @@ -493,6 +493,7 @@ def test_AV_dict_projection_V(self): def test_persistent_tensors_field(self): self.assertEqual(self.wl.persistent_tensors, "weight - Intermediates") + # ============================================================================ # simple.yaml arch -- golden values # ============================================================================ @@ -1083,8 +1084,13 @@ def test_memory_defaults(self): name="TestMem", size=1000, actions=[ - {"name": "read", "energy": 1, "latency": 0}, - {"name": "write", "energy": 1, "latency": 0}, + {"name": "read", "energy": 1, "throughput": float("inf"), "latency": 0}, + { + "name": "write", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, ], leak_power=0, area=0, @@ -1097,7 +1103,7 @@ def test_memory_defaults(self): def test_compute_defaults(self): comp = Compute( name="MAC", - actions=[{"name": "compute", "energy": 1, "latency": 1}], + actions=[{"name": "compute", "energy": 1, "throughput": 1, "latency": 0}], leak_power=0, area=0, ) diff --git a/tests/vibe_see_readme_in_this_dir/test_rendering.py b/tests/vibe_see_readme_in_this_dir/test_rendering.py index 0c9cbaac..39a399aa 100644 --- a/tests/vibe_see_readme_in_this_dir/test_rendering.py +++ b/tests/vibe_see_readme_in_this_dir/test_rendering.py @@ -138,8 +138,18 @@ def _make_arch(self): name="MainMemory", size=1_000_000, actions=[ - {"name": "read", "energy": 1, "latency": 0}, - {"name": "write", "energy": 1, "latency": 0}, + { + "name": "read", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, + { + "name": "write", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, ], leak_power=0, area=0, @@ -148,15 +158,27 @@ def _make_arch(self): name="GlobalBuffer", size=100_000, actions=[ - {"name": "read", "energy": 0.5, "latency": 0}, - {"name": "write", "energy": 0.5, "latency": 0}, + { + "name": "read", + "energy": 0.5, + "throughput": float("inf"), + "latency": 0, + }, + { + "name": "write", + "energy": 0.5, + "throughput": float("inf"), + "latency": 0, + }, ], leak_power=0, area=0, ), Compute( name="MAC", - actions=[{"name": "compute", "energy": 1, "latency": 1}], + actions=[ + {"name": "compute", "energy": 1, "throughput": 1, "latency": 0} + ], leak_power=0, area=0, ), @@ -216,8 +238,13 @@ def test_render_node_name_memory(self): name="TestMem", size=100, actions=[ - {"name": "read", "energy": 1, "latency": 0}, - {"name": "write", "energy": 1, "latency": 0}, + {"name": "read", "energy": 1, "throughput": float("inf"), "latency": 0}, + { + "name": "write", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, ], leak_power=0, area=0, @@ -231,8 +258,13 @@ def test_render_node_label_memory(self): name="TestMem", size=100, actions=[ - {"name": "read", "energy": 1, "latency": 0}, - {"name": "write", "energy": 1, "latency": 0}, + {"name": "read", "energy": 1, "throughput": float("inf"), "latency": 0}, + { + "name": "write", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, ], leak_power=0, area=0, @@ -246,8 +278,13 @@ def test_render_node_shape_memory(self): name="TestMem", size=100, actions=[ - {"name": "read", "energy": 1, "latency": 0}, - {"name": "write", "energy": 1, "latency": 0}, + {"name": "read", "energy": 1, "throughput": float("inf"), "latency": 0}, + { + "name": "write", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, ], leak_power=0, area=0, @@ -258,7 +295,7 @@ def test_render_node_shape_memory(self): def test_render_node_compute(self): comp = Compute( name="MAC", - actions=[{"name": "compute", "energy": 1, "latency": 1}], + actions=[{"name": "compute", "energy": 1, "throughput": 1, "latency": 0}], leak_power=0, area=0, ) diff --git a/tests/vibe_see_readme_in_this_dir/test_yaml_and_expressions.py b/tests/vibe_see_readme_in_this_dir/test_yaml_and_expressions.py index 1a0621b7..0ba415ea 100644 --- a/tests/vibe_see_readme_in_this_dir/test_yaml_and_expressions.py +++ b/tests/vibe_see_readme_in_this_dir/test_yaml_and_expressions.py @@ -341,14 +341,14 @@ def test_arch_expressions_reference_variables(self): leak_power: 0 area: 0 actions: - - {name: read, energy: 1, latency: 0} - - {name: write, energy: 1, latency: 0} + - {name: read, energy: 1, throughput: .inf, latency: 0} + - {name: write, energy: 1, throughput: .inf, latency: 0} - !Compute name: MAC leak_power: 0 area: 0 actions: - - {name: compute, energy: 1, latency: 1} + - {name: compute, energy: 1, throughput: 1, latency: 0} workload: rank_sizes: {M: 16} @@ -386,14 +386,14 @@ def test_complement(self): area: 0 tensors: {keep: ~Intermediates, may_keep: All} actions: - - {name: read, energy: 1, latency: 0} - - {name: write, energy: 1, latency: 0} + - {name: read, energy: 1, throughput: .inf, latency: 0} + - {name: write, energy: 1, throughput: .inf, latency: 0} - !Compute name: MAC leak_power: 0 area: 0 actions: - - {name: compute, energy: 1, latency: 1} + - {name: compute, energy: 1, throughput: 1, latency: 0} """ ) mem = spec.arch.find("Mem") @@ -411,14 +411,14 @@ def test_union_expression(self): area: 0 tensors: {keep: input | output} actions: - - {name: read, energy: 1, latency: 0} - - {name: write, energy: 1, latency: 0} + - {name: read, energy: 1, throughput: .inf, latency: 0} + - {name: write, energy: 1, throughput: .inf, latency: 0} - !Compute name: MAC leak_power: 0 area: 0 actions: - - {name: compute, energy: 1, latency: 1} + - {name: compute, energy: 1, throughput: 1, latency: 0} """ ) mem = spec.arch.find("Mem") @@ -437,8 +437,8 @@ def test_reference_above_memory(self): area: 0 tensors: {keep: ~Intermediates, may_keep: All} actions: - - {name: read, energy: 1, latency: 0} - - {name: write, energy: 1, latency: 0} + - {name: read, energy: 1, throughput: .inf, latency: 0} + - {name: write, energy: 1, throughput: .inf, latency: 0} - !Memory name: Buffer size: inf @@ -446,14 +446,14 @@ def test_reference_above_memory(self): area: 0 tensors: {keep: ~MainMemory.tensors, may_keep: All} actions: - - {name: read, energy: 1, latency: 0} - - {name: write, energy: 1, latency: 0} + - {name: read, energy: 1, throughput: .inf, latency: 0} + - {name: write, energy: 1, throughput: .inf, latency: 0} - !Compute name: MAC leak_power: 0 area: 0 actions: - - {name: compute, energy: 1, latency: 1} + - {name: compute, energy: 1, throughput: 1, latency: 0} """ ) buf = spec.arch.find("Buffer") @@ -608,7 +608,9 @@ def test_toll_direction(self): direction="up", leak_power=0, area=0, - actions=[{"name": "read", "energy": 1, "latency": 0}], + actions=[ + {"name": "read", "energy": 1, "throughput": float("inf"), "latency": 0} + ], ) self.assertEqual(t.direction, "up") @@ -659,7 +661,7 @@ def test_enabled_default_true(self): name="MAC", leak_power=0, area=0, - actions=[{"name": "compute", "energy": 1, "latency": 1}], + actions=[{"name": "compute", "energy": 1, "throughput": 1, "latency": 0}], ) self.assertTrue(c.enabled) @@ -668,7 +670,7 @@ def test_enabled_expression_string(self): name="MAC", leak_power=0, area=0, - actions=[{"name": "compute", "energy": 1, "latency": 1}], + actions=[{"name": "compute", "energy": 1, "throughput": 1, "latency": 0}], enabled="len(All) == 3", ) self.assertEqual(c.enabled, "len(All) == 3") @@ -751,8 +753,13 @@ def test_get_fanout_no_spatial(self): leak_power=0, area=0, actions=[ - {"name": "read", "energy": 1, "latency": 0}, - {"name": "write", "energy": 1, "latency": 0}, + {"name": "read", "energy": 1, "throughput": float("inf"), "latency": 0}, + { + "name": "write", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + }, ], ) self.assertEqual(mem.get_fanout(), 1) @@ -993,5 +1000,6 @@ def test_same_projection_values(self): f"{c_ta.name} in {c_e.name}", ) + if __name__ == "__main__": unittest.main() diff --git a/tests/vibe_see_readme_in_this_dir/values_per_action/input_files/vpa_precedence_action_bpa.arch.yaml b/tests/vibe_see_readme_in_this_dir/values_per_action/input_files/vpa_precedence_action_bpa.arch.yaml index 7b88c8b1..04582686 100644 --- a/tests/vibe_see_readme_in_this_dir/values_per_action/input_files/vpa_precedence_action_bpa.arch.yaml +++ b/tests/vibe_see_readme_in_this_dir/values_per_action/input_files/vpa_precedence_action_bpa.arch.yaml @@ -18,7 +18,7 @@ arch: bits_per_value: {All: 4} bits_per_action: 32 actions: - - {name: read, energy: 100, latency: 0, bits_per_action: 8} + - {name: read, energy: 100, throughput: inf, latency: 0, bits_per_action: 8} - !Memory name: LocalBuffer diff --git a/tests/vibe_see_readme_in_this_dir/values_per_action/input_files/vpa_precedence_action_wins.arch.yaml b/tests/vibe_see_readme_in_this_dir/values_per_action/input_files/vpa_precedence_action_wins.arch.yaml index 4de7d308..4d8542fa 100644 --- a/tests/vibe_see_readme_in_this_dir/values_per_action/input_files/vpa_precedence_action_wins.arch.yaml +++ b/tests/vibe_see_readme_in_this_dir/values_per_action/input_files/vpa_precedence_action_wins.arch.yaml @@ -16,7 +16,7 @@ arch: tensors: {keep: All} values_per_action: {T0: 5, W0: 5, T1: 5} actions: - - {name: read, energy: 100, latency: 0, values_per_action: {T0: 2, W0: 2, T1: 2}} + - {name: read, energy: 100, throughput: inf, latency: 0, values_per_action: {T0: 2, W0: 2, T1: 2}} - !Memory name: LocalBuffer diff --git a/tests/vibe_see_readme_in_this_dir/values_per_action/input_files/vpa_precedence_component_bpa.arch.yaml b/tests/vibe_see_readme_in_this_dir/values_per_action/input_files/vpa_precedence_component_bpa.arch.yaml index d7135322..b81fd75c 100644 --- a/tests/vibe_see_readme_in_this_dir/values_per_action/input_files/vpa_precedence_component_bpa.arch.yaml +++ b/tests/vibe_see_readme_in_this_dir/values_per_action/input_files/vpa_precedence_component_bpa.arch.yaml @@ -18,7 +18,7 @@ arch: bits_per_value: {All: 4} bits_per_action: 16 actions: - - {name: read, energy: 100, latency: 0} + - {name: read, energy: 100, throughput: inf, latency: 0} - !Memory name: LocalBuffer diff --git a/tests/vibe_see_readme_in_this_dir/values_per_action/input_files/vpa_precedence_component_wins.arch.yaml b/tests/vibe_see_readme_in_this_dir/values_per_action/input_files/vpa_precedence_component_wins.arch.yaml index 94c3b661..dd140cfa 100644 --- a/tests/vibe_see_readme_in_this_dir/values_per_action/input_files/vpa_precedence_component_wins.arch.yaml +++ b/tests/vibe_see_readme_in_this_dir/values_per_action/input_files/vpa_precedence_component_wins.arch.yaml @@ -15,7 +15,7 @@ arch: tensors: {keep: All} values_per_action: {T0: 5, W0: 5, T1: 5} actions: - - {name: read, energy: 100, latency: 0} + - {name: read, energy: 100, throughput: inf, latency: 0} - !Memory name: LocalBuffer diff --git a/tests/vibe_see_readme_in_this_dir/values_per_action/input_files/vpa_precedence_default.arch.yaml b/tests/vibe_see_readme_in_this_dir/values_per_action/input_files/vpa_precedence_default.arch.yaml index c45ba53d..bff4e355 100644 --- a/tests/vibe_see_readme_in_this_dir/values_per_action/input_files/vpa_precedence_default.arch.yaml +++ b/tests/vibe_see_readme_in_this_dir/values_per_action/input_files/vpa_precedence_default.arch.yaml @@ -15,7 +15,7 @@ arch: area: 0 tensors: {keep: All} actions: - - {name: read, energy: 100, latency: 0} + - {name: read, energy: 100, throughput: inf, latency: 0} - !Memory name: LocalBuffer diff --git a/tests/vibe_see_readme_in_this_dir/values_per_action/test_values_per_action_precedence.py b/tests/vibe_see_readme_in_this_dir/values_per_action/test_values_per_action_precedence.py index e5fed552..b1f2f3e3 100644 --- a/tests/vibe_see_readme_in_this_dir/values_per_action/test_values_per_action_precedence.py +++ b/tests/vibe_see_readme_in_this_dir/values_per_action/test_values_per_action_precedence.py @@ -32,13 +32,23 @@ def _spec( if comp_bpv is not None: comp_kwargs["bits_per_value"] = comp_bpv - read_action = {"name": "read", "energy": 1, "latency": 0} + read_action = { + "name": "read", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + } if read_bpa is not None: read_action["bits_per_action"] = read_bpa if read_vpa is not None: read_action["values_per_action"] = read_vpa - write_action = {"name": "write", "energy": 1, "latency": 0} + write_action = { + "name": "write", + "energy": 1, + "throughput": float("inf"), + "latency": 0, + } if write_bpa is not None: write_action["bits_per_action"] = write_bpa if write_vpa is not None: @@ -70,7 +80,9 @@ def _spec( ), Compute( name="MAC", - actions=[{"name": "compute", "energy": 1, "latency": 1}], + actions=[ + {"name": "compute", "energy": 1, "throughput": 1, "latency": 0} + ], leak_power=0, area=0, ), From 9c4c11b39949a2ab5f6ddbf1fb408c89bd1e7130 Mon Sep 17 00:00:00 2001 From: Tanner Andrulis Date: Fri, 31 Jul 2026 11:20:27 -0400 Subject: [PATCH 3/8] [mapper, model] Sharing of component latency across Einsums --- accelforge/frontend/arch/components.py | 2 +- .../FFM/_join_pmappings/join_pmappings.py | 12 +- .../FFM/_join_pmappings/pmapping_dataframe.py | 192 ++++++++++++++---- .../FFM/_make_pmappings/make_pmappings.py | 78 ++++--- .../make_pmappings_from_templates.py | 31 ++- .../mapper/FFM/_make_pmappings/pmapper_job.py | 1 + .../mapper/FFM/_pareto_df/df_convention.py | 26 ++- accelforge/model/_looptree/latency/memory.py | 79 +++++-- accelforge/model/main.py | 8 +- accelforge/model/run_model.py | 33 ++- .../fused_matmuls_to_networked.mapping.yaml | 61 ++++++ tests/input_files/latency.arch.yaml | 43 ++++ tests/input_files/networked_latency.arch.yaml | 58 ++++++ tests/test_latency.py | 135 ++++++++++++ 14 files changed, 643 insertions(+), 116 deletions(-) create mode 100644 tests/input_files/fused_matmuls_to_networked.mapping.yaml create mode 100644 tests/input_files/latency.arch.yaml create mode 100644 tests/input_files/networked_latency.arch.yaml create mode 100644 tests/test_latency.py diff --git a/accelforge/frontend/arch/components.py b/accelforge/frontend/arch/components.py index 34579586..41cff91c 100644 --- a/accelforge/frontend/arch/components.py +++ b/accelforge/frontend/arch/components.py @@ -1341,7 +1341,7 @@ class Network(Component, Leaf): actions: EvalableList[Action] = NETWORK_ACTIONS - total_latency: str | int | float | None = "max_link_traffic/actions['hop'].throughput" + total_latency: str | int | float = "max_link_traffic/actions['hop'].throughput" """ Models latency as bandwidth-bound, which means that the traffic over the most congested link dominates the overall communication latency. Note that max_hops * diff --git a/accelforge/mapper/FFM/_join_pmappings/join_pmappings.py b/accelforge/mapper/FFM/_join_pmappings/join_pmappings.py index 3f4dc2fa..7313c879 100755 --- a/accelforge/mapper/FFM/_join_pmappings/join_pmappings.py +++ b/accelforge/mapper/FFM/_join_pmappings/join_pmappings.py @@ -419,7 +419,6 @@ def get_memories_to_track( always_below.add(reservation_key.name) total_sizes = {} - ignored_resources = oset() for _, einsum_pmapping_groups in pmapping_groups.items(): max_sizes = {} @@ -441,10 +440,6 @@ def get_memories_to_track( size = s.mappings.data[col].max() max_sizes[name] = max(max_sizes.get(name, 0), size) - # nloops < 0 means that the reservation will live through all Einsums - if nloops < 0: - ignored_resources.add(name) - for name, size in max_sizes.items(): total_sizes[name] = total_sizes.get(name, 0) + size @@ -636,7 +631,8 @@ def join_pmappings( print_progress=print_progress, ) einsum_pmappings.pmapping_groups = PmappingGroup.group( - einsum_pmappings.pmapping_groups, left_tensors, + einsum_pmappings.pmapping_groups, + left_tensors, ) einsum, prev_einsum = einsum_pmappings.einsum_name, pmgroups[i - 1].einsum_name step_time = time.time() - t0 @@ -997,6 +993,10 @@ def no_match_lookahead_error( mappings = s_final[0].mappings mappings.limit_capacity(next_shared_loop_index=-1, finished=True) mappings.free_to_loop_index(-2) + assert not mappings._make_latencies(), ( + f"Component latency columns were not folded into the total latency: " + f"{mappings._make_latencies()}" + ) mappings.make_pareto() timer.log_total_time() diff --git a/accelforge/mapper/FFM/_join_pmappings/pmapping_dataframe.py b/accelforge/mapper/FFM/_join_pmappings/pmapping_dataframe.py index 2cfcad08..94c366d3 100755 --- a/accelforge/mapper/FFM/_join_pmappings/pmapping_dataframe.py +++ b/accelforge/mapper/FFM/_join_pmappings/pmapping_dataframe.py @@ -98,20 +98,36 @@ def get_reservation_or_parent( return_name_level_left: bool = False, ) -> str | tuple[str, int, bool] | None: reservations = l_reservations if left else r_reservations - if (reservations := reservations.get(name, None)) is not None: - while level >= -1: - if level in reservations: - if return_name_level_left: - return name, level, left - return reservation2col(name, level, left) - # The parent of left nodes are right nodes, so if we don't find a - # left node immediately then we're back on the right nodes - reservations = r_reservations.get(name, oset()) - left = False - level -= 1 + # Previously we had the below here, but I think that's bugged because it would + # return None if we checked for left, it didn't exist on the left, and did exist on + # the right. + # if (levels := reservations.get(name, None)) is not None: + + levels = reservations.get(name, oset()) + while level >= -1: + if level in levels: + if return_name_level_left: + return name, level, left + return reservation2col(name, level, left) + # The parent of left nodes are right nodes, so if we don't find a + # left node immediately then we're back on the right nodes + levels = r_reservations.get(name, oset()) + left = False + level -= 1 return None +def get_latency_or_below( + name: str, level: int, latencies: dict[str, oset] +) -> str | None: + """Latency columns include all levels below them, so the value at a level with + no column is the value at the closest level below it.""" + levels = [l for l in latencies.get(name, oset()) if l >= level] + if not levels: + return None + return complatency2col(name, min(levels)) + + class PmappingDataframe: def __init__( self, @@ -173,7 +189,9 @@ def rename(self, renames: dict[str, str]) -> "PmappingDataframe": @error_check_wrapper def fill_reservation_cols(self, columns: set | str): l_reservations, r_reservations = self._make_reservations() + latencies = self._make_latencies() targets = [] + latency_targets = [] if columns == "auto": for left, reservations_dict in [ (True, l_reservations), @@ -187,20 +205,36 @@ def fill_reservation_cols(self, columns: set | str): if above is not None: below = reservation2col(resource, r, left=left) targets.append((r, above, below)) + for component, levels in latencies.items(): + for r in sorted(levels): + source = get_latency_or_below(component, r + 1, latencies) + if source is not None: + latency_targets.append( + (r, source, complatency2col(component, r)) + ) else: for below in columns: - if (name_nloops := col2reservation(below)) is None: + if (name_nloops := col2reservation(below)) is not None: + name, nloops = name_nloops.name, name_nloops.nloops + above = get_reservation_or_parent( + name, nloops - 1, l_reservations, r_reservations + ) + if above is not None: + targets.append((nloops, above, below)) + elif (name_nloops := col2complatency(below)) is not None: + name, nloops = name_nloops.name, name_nloops.nloops + source = get_latency_or_below(name, nloops + 1, latencies) + if source is not None: + latency_targets.append((nloops, source, below)) + else: raise ValueError(f"{below} is not a valid reservation column") - name, nloops = name_nloops.name, name_nloops.nloops - above = get_reservation_or_parent( - name, nloops - 1, l_reservations, r_reservations - ) - if above is not None: - targets.append((nloops, above, below)) - # Sort so we go from top to bottom. Needed in case we have to max 0->1 - # then 1->2 - for _, above, below in sorted(targets, key=lambda x: x[0]): + # Sort so maxes propagate: reservations from top to bottom (e.g. 0->1 then + # 1->2), latencies from bottom to top. + ordered = sorted(targets, key=lambda x: x[0]) + sorted( + latency_targets, key=lambda x: x[0], reverse=True + ) + for _, above, below in ordered: assert ( above in self.data.columns ), f"Missing column {above}. Have columns:\n\t" + "\n\t".join( @@ -230,9 +264,9 @@ def data(self) -> pd.DataFrame: @error_check_wrapper def _make_reservations(self) -> tuple[dict[str, set[int]], dict[str, set[int]]]: """ - Create a dictionary of reservations for each resource. - The dictionary keys are the resource names and the values are lists - of column names for each loop index. + Create a dictionary of reservations for each resource. The dictionary keys are + the resource names and the values are lists of above_loop_index values for that + resource. """ l_reservations, r_reservations = {}, {} for c in self.data.columns: @@ -243,6 +277,17 @@ def _make_reservations(self) -> tuple[dict[str, set[int]], dict[str, set[int]]]: return l_reservations, r_reservations + def _make_latencies(self) -> dict[str, oset]: + """ + Keys component name, values are sets of column indices for that component. + """ + latencies = {} + for c in self.data.columns: + if (key := col2complatency(c)) is not None: + latencies.setdefault(key.name, oset()).add(key.nloops) + assert key.nloops >= -1 + return latencies + def clear_fused_loop_symbols(self): dropcols = [c for c in self.data.columns if is_fused_loop_col(c)] if not dropcols: @@ -253,10 +298,11 @@ def clear_fused_loop_symbols(self): @error_check_wrapper def free_to_loop_index(self, loop_index: int) -> bool: """ + Reservations follow: A B / | --- 0 C D - / | --- 1 < Shared Loop Index + / | --- 1 <- Shared Loop Index E F / | --- 2 G H @@ -264,10 +310,30 @@ def free_to_loop_index(self, loop_index: int) -> bool: A B / | --- 0 C D - | --- 1 < Shared Loop Index + | --- 1 <- Shared Loop Index max(E,G,H) We skip incorporating E into the max because its reservations are already incorporated into F and G. + + For latencies, we do: + A + | --- 0 + B + | --- 1 <- Shared Loop Index + C + | --- 2 + D + -> + A + | --- 0 + B + | --- 1 <- Shared Loop Index + + Skip incorporating D,C into B because levels already include sub-levels. Also we + do: + Totallatency = max(Totallatency, C) + + since C must complete before exiting this fused loop block. """ if loop_index == self._prev_free_to_loop_index: return False @@ -279,9 +345,9 @@ def free_to_loop_index(self, loop_index: int) -> bool: max_columns = [] cur_l_reservations = l_reservations.get(resource, oset()) cur_r_reservations = r_reservations.get(resource, oset()) - left_big_enough = [l for l in cur_l_reservations if l >= loop_index + 1] + left_big_enough = [l for l in cur_l_reservations if l > loop_index] right_big_enough = [ - r for r in cur_r_reservations if r >= loop_index + 2 + r for r in cur_r_reservations if r > loop_index + 1 ] # + 1 is target if len(right_big_enough) > 1: # All ones above the last are subsets @@ -307,6 +373,22 @@ def free_to_loop_index(self, loop_index: int) -> bool: for c in max_columns: max_to_col(self.data, target, c) drop_columns += [m for m in max_columns if m != target] + + for component, levels in self._make_latencies().items(): + done = [l for l in levels if l > loop_index] + if not done: + continue + # Latency can only be shared between Einsums whose reservations live at the + # same time (a transfer needs its space to already be reserved), so levels + # whose reservations future Einsums would max rather than sum are finished. + # Columns include all levels below, so the shallowest done column has all of + # the finished latency + assert "Totallatency" in self.data + max_to_col( + self.data, "Totallatency", complatency2col(component, min(done)) + ) + drop_columns += [complatency2col(component, l) for l in done] + self._data = self.data.drop(columns=drop_columns) return len(drop_columns) != 0 @@ -451,6 +533,8 @@ def check_match(la: Loop, lb: Loop, param: str): assert not right_df_l_reservations, f"{right_df_l_reservations} is not None" l_reservations, r_reservations = self._make_reservations() + l_latencies = self._make_latencies() + r_latencies = right._make_latencies() for resource, reservations in r_reservations.items(): n_reservations = max(reservations, default=-1) @@ -503,6 +587,12 @@ def check_match(la: Loop, lb: Loop, param: str): n_total_pmappings *= scale_by n_valid_pmappings *= scale_by + def from_right(c: str) -> str: + # If a column is in both DFs, do the appropriate rename to grab the one from + # the right side + right_merge = c + "_RIGHT_MERGE" + return right_merge if right_merge in df else c + # Make sure everything is done in increasing loop order so we don't have # read-after-write hazards for nloops in range(max_nloops, min_nloops - 1, -1): @@ -540,14 +630,9 @@ def iter_reservations(reservations_dict): ) ) is None: continue - right_merge_source = source + "_RIGHT_MERGE" target = reservation2col(resource, nloops, left=True) if source is not None: - add_to_col( - df, - target, - right_merge_source if right_merge_source in df else source, - ) + add_to_col(df, target, from_right(source)) # For LEFT tree, RIGHT reservations: Add the same-level reservation from the # right tree. This will double-count reservations that are in both branches, # so we remove them later. @@ -561,14 +646,38 @@ def iter_reservations(reservations_dict): ) ) is None: continue - right_merge_source = source + "_RIGHT_MERGE" target = reservation2col(resource, nloops) if source is not None: - add_to_col( - df, - target, - right_merge_source if right_merge_source in df else source, - ) + add_to_col(df, target, from_right(source)) + + # PIPELINE: The following would need some maxes + + # Component latencies are always summed. Some extra logic needed in case there's + # missing columns that make us need to grab from a column below. A side with no + # column at or below a level has no latency there and contributes nothing. + for component in oset(l_latencies) | oset(r_latencies): + l_levels = l_latencies.get(component, oset()) + r_levels = r_latencies.get(component, oset()) + # max_nloops/min_nloops only cover reservation levels, so iterate the + # latency levels directly (ascending so the deeper columns we read are + # still unmodified). + for nloops in sorted(l_levels | r_levels): + target = complatency2col(component, nloops) + # In both: Simple add + if nloops in l_levels and nloops in r_levels: + add_to_col(df, target, from_right(target)) + + # In left only: Add the closest-below one from the right + elif nloops in l_levels: + source = get_latency_or_below(component, nloops, r_latencies) + if source is not None: + add_to_col(df, target, from_right(source)) + + # In right only: Same thing + elif nloops in r_levels: + source = get_latency_or_below(component, nloops, l_latencies) + if source is not None: + add_to_col(df, target, source) # For everything else: Simple add dropcols = [c for c in df.columns if c.endswith("_RIGHT_MERGE")] @@ -576,12 +685,15 @@ def iter_reservations(reservations_dict): target = source[: -len("_RIGHT_MERGE")] if is_tensor_col(target): continue + if col2complatency(target) is not None: + continue if not col_used_in_pareto(target): raise ValueError(f"{target} is not used in pareto") if col2reservation(target) is None: add_to_col(df, target, source) df = df.drop(columns=dropcols) + result = PmappingDataframe( df, skip_pareto=True, diff --git a/accelforge/mapper/FFM/_make_pmappings/make_pmappings.py b/accelforge/mapper/FFM/_make_pmappings/make_pmappings.py index e36fa08c..83788c2a 100755 --- a/accelforge/mapper/FFM/_make_pmappings/make_pmappings.py +++ b/accelforge/mapper/FFM/_make_pmappings/make_pmappings.py @@ -232,20 +232,24 @@ def make_jobs_for_einsum( return einsum2jobs -def get_memories_to_track( +def get_components_to_track( spec: Spec, einsum2jobs: dict[EinsumName, list[Job]], metrics: Metrics, can_combine_multiple_runs: bool, print_progress: bool = True, -) -> tuple[list[str], list[str]]: +) -> tuple[list[str], list[str], set[str], set[str]]: memories_track_all = oset() + components_track_latency = oset() for einsum, jobs in einsum2jobs.items(): for job in jobs: memories_track_all.update( m.name for m in job.flattened_arch if isinstance(m, arch.Memory) ) + components_track_latency.update( + m.name for m in job.flattened_arch if isinstance(m, arch.TensorHolder) + ) memories_track_pmappings_only = [] ignored_resources = oset() @@ -258,6 +262,7 @@ def get_memories_to_track( memories_track_all, memories_track_pmappings_only, ignored_resources, + components_track_latency, ) tensor_sizes = {} @@ -300,24 +305,23 @@ def get_memories_to_track( ) memories_track_all.remove(memory) + # A component is shared across Einsums if it's above any fused loops or it's backing + shared_components = oset() + for jobs in einsum2jobs.values(): + for job in jobs: + seen = oset() + for node in job.mapping.nodes: + if isinstance(node, TensorHolder): + seen.add(node.component) + if node._backing or node.persistent: + shared_components.add(node.component) + if isinstance(node, Loop) and node._fused: + shared_components.update(seen) + # If the memory is below every backing tensor holder node, then we need it for the # pmapping exploration but can drop it immediately for m in list(memories_track_all): - must_track = False - for _, jobs in einsum2jobs.items(): - for job in jobs: - seen = False - for node in job.mapping.nodes: - if isinstance(node, TensorHolder) and node.component == m: - seen = True - if node.persistent: - ignored_resources.add(m) - if node._backing: - must_track = True - if isinstance(node, Loop) and node._fused and seen: - must_track = True - - if not must_track: + if m not in shared_components: memories_track_all.remove(m) memories_track_pmappings_only.append(m) if print_progress: @@ -326,7 +330,25 @@ def get_memories_to_track( f"reserved across fused loop iterations." ) - return memories_track_all, memories_track_pmappings_only, ignored_resources + # A component's latency can only be shared with other Einsums if the component's + # reservations may be shared. + # PIPELINE: This would have to change + for m in list(components_track_latency): + if m not in shared_components: + components_track_latency.remove(m) + if print_progress: + print( + f"Not tracking latency for component {m}. It never holds tensors " + f"at or above a backing storage node, so its latency is private " + f"to each Einsum." + ) + + return ( + memories_track_all, + memories_track_pmappings_only, + ignored_resources, + components_track_latency, + ) def make_pmappings( @@ -477,17 +499,21 @@ def _fill_jobs_with_memories_to_track( e: [j for jobs in v.values() for j in jobs] for e, v in einsum2jobs.items() } - memories_track_all, memories_track_pmappings_only, ignored_resources = ( - get_memories_to_track( - spec, - einsum2jobs_flattened, - metrics, - can_combine_multiple_runs, - print_progress, - ) + ( + memories_track_all, + memories_track_pmappings_only, + ignored_resources, + components_track_latency, + ) = get_components_to_track( + spec, + einsum2jobs_flattened, + metrics, + can_combine_multiple_runs, + print_progress, ) for jobs in einsum2jobs_flattened.values(): for j in jobs: j.memories_track_all = memories_track_all j.memories_track_pmappings_only = memories_track_pmappings_only j.ignored_resources = ignored_resources + j.components_track_latency = components_track_latency diff --git a/accelforge/mapper/FFM/_make_pmappings/make_pmappings_from_templates/make_pmappings_from_templates.py b/accelforge/mapper/FFM/_make_pmappings/make_pmappings_from_templates/make_pmappings_from_templates.py index eb59f204..d70a88cd 100755 --- a/accelforge/mapper/FFM/_make_pmappings/make_pmappings_from_templates/make_pmappings_from_templates.py +++ b/accelforge/mapper/FFM/_make_pmappings/make_pmappings_from_templates/make_pmappings_from_templates.py @@ -14,8 +14,10 @@ from accelforge.mapper.FFM._join_pmappings.pmapping_dataframe import ( MAPPING_COLUMN, PmappingDataframe, + col2complatency, col2reservation, col_used_in_pareto, + complatency2col, is_reservation_col, makepareto, tensor2col, @@ -56,28 +58,35 @@ def shift_reservations_by_null_loop_indices( target2newabovename = {} dropcols = [] for c in mappings.columns: - if not is_reservation_col(c): + if is_reservation_col(c): + key, make_col = col2reservation(c), reservation2col + elif (key := col2complatency(c)) is not None: + make_col = complatency2col + else: continue - reservation = col2reservation(c) - name = reservation.name - above = reservation.nloops + above = key.nloops new_above = above - sum(above > i for i in null_loop_indices) - target = reservation2col(name, new_above) + target = make_col(key.name, new_above) if target in target2newabovename: - if above > target2newabovename[target][1]: - dropcols.append(reservation2col(*target2newabovename[target])) - target2newabovename[target] = (name, above) + # On a collision, keep the column that includes the other: the deeper one + # for reservations, the shallower one for latencies. + keep_new = above > target2newabovename[target][1] + if not is_reservation_col(c): + keep_new = not keep_new + if keep_new: + dropcols.append(target2newabovename[target][0]) + target2newabovename[target] = (c, above) else: dropcols.append(c) else: - target2newabovename[target] = (name, above) + target2newabovename[target] = (c, above) if dropcols: drop_set = set(dropcols) mappings = mappings[[c for c in mappings.columns if c not in drop_set]] renames = {} - for target, (name, above) in target2newabovename.items(): - renames[reservation2col(name, above)] = target + for target, (source, _) in target2newabovename.items(): + renames[source] = target mappings = mappings.rename(columns=renames) if len(mappings.columns) != len(mappings.columns.unique()): raise ValueError(f"Duplicate columns: {mappings.columns}") diff --git a/accelforge/mapper/FFM/_make_pmappings/pmapper_job.py b/accelforge/mapper/FFM/_make_pmappings/pmapper_job.py index 7fdf11b4..9e1daba5 100755 --- a/accelforge/mapper/FFM/_make_pmappings/pmapper_job.py +++ b/accelforge/mapper/FFM/_make_pmappings/pmapper_job.py @@ -67,6 +67,7 @@ class Job: memories_track_all: list[str] | None = None memories_track_pmappings_only: list[str] | None = None ignored_resources: set[str] | None = None + components_track_latency: set[str] | None = None time_limit: float | int = float("inf") memory_limit: float | int = float("inf") messages: list[str] = field(default_factory=list) diff --git a/accelforge/mapper/FFM/_pareto_df/df_convention.py b/accelforge/mapper/FFM/_pareto_df/df_convention.py index 059ff91f..dbfa435c 100755 --- a/accelforge/mapper/FFM/_pareto_df/df_convention.py +++ b/accelforge/mapper/FFM/_pareto_df/df_convention.py @@ -132,6 +132,26 @@ def col2energy(colname: str) -> ActionKey | VerboseActionKey: ReservationKey = namedtuple("ReservationKey", ["name", "nloops"]) +ComponentLatencyKey = namedtuple("ComponentLatencyKey", ["name", "nloops"]) + + +@dict_cached +def col2complatency(x: str) -> ComponentLatencyKey | None: + """ + Format: component_latency name nloops. Distinct from the 2-part + component_latencyname columns, these are specific to a loop level and may be + pushed to work during earlier Einsums + """ + parts = x.split(SEP) + if len(parts) != 3 or parts[0] != "component_latency": + return None + return ComponentLatencyKey(parts[1], int(parts[2])) + + +@dict_cached +def complatency2col(name: str, nloops: int) -> str: + """Format: component_latency name nloops""" + return f"component_latency{name}{nloops}" @dict_cached @@ -295,7 +315,11 @@ def is_objective_col(c): def col_used_in_pareto(c): - return col2reservation(c) is not None or is_objective_col(c) + return ( + col2reservation(c) is not None + or col2complatency(c) is not None + or is_objective_col(c) + ) def col_used_in_joining(c): diff --git a/accelforge/model/_looptree/latency/memory.py b/accelforge/model/_looptree/latency/memory.py index 61f5587d..0d37d9d7 100755 --- a/accelforge/model/_looptree/latency/memory.py +++ b/accelforge/model/_looptree/latency/memory.py @@ -136,10 +136,24 @@ def component_latency( flattened_arch: FlattenedArch, mapping: Mapping, spec: Spec, + per_level_components: oset = oset(), ): + """ + Returns (component latency, per-level component latency). + + The per-level dict covers components in per_level_components, mapping each to + {n_loops_above: latency of the actions at that many loops above plus the latency of + actions lower in the tree}. Each level includes the levels below it, so the + shallowest level is the component's whole latency. Levels must be completed before + moving to other branches of the LoopTree, but may be overlapped with the latencies + of other branches below the current level. + """ component_to_actions: dict[str, dict[str, float]] = defaultdict( lambda: defaultdict(lambda: 0) ) + component_to_level_actions: dict[str, dict[int, dict[str, float]]] = defaultdict( + lambda: defaultdict(lambda: defaultdict(lambda: 0)) + ) # Holds ``keywords" that do not map neatly to actions, e.g., max_hops for network component_to_keywords: dict[str, dict[str, float]] = defaultdict( lambda: defaultdict(lambda: 0) @@ -160,9 +174,18 @@ def component_latency( actions[action.name] += 0 if isinstance(name2component[component], TensorHolder): - actions["read"] += buffet_stats.net_max_per_unit_actions("read") - if not isinstance(name2component[component], arch.Toll): - actions["write"] += buffet_stats.net_max_per_unit_actions("write") + reads = buffet_stats.net_max_per_unit_actions("read") + writes = buffet_stats.net_max_per_unit_actions("write") + actions["read"] += reads + include_writes = not isinstance(name2component[component], arch.Toll) + n_loops_above = buffet_stats.n_loops_above + if include_writes: + actions["write"] += writes + if component in per_level_components: + level_actions = component_to_level_actions[component][n_loops_above] + level_actions["read"] += reads + if include_writes: + level_actions["write"] += writes elif isinstance(name2component[component], arch.Compute): pass else: @@ -191,30 +214,25 @@ def component_latency( *network_to_max_link_traffic[component].values() ) keywords["max_hops"] = MaxGeqZero(*network_to_max_hops[component]) + actions = component_to_actions[component] + for action in name2component[component].actions: + actions[action.name] = 0 longest_compute_latency = Max( 0, *[s.max_latency for s in looptree_results.compute_stats.values()] ) component_to_actions[compute_obj.name]["compute"] = longest_compute_latency - new_component_to_actions: dict[str, list] = {} for component, action_counts in component_to_actions.items(): component_obj = name2component[component] - scale = getattr(component_obj, "actions_scale", 1) for action_name in action_counts: if action_name not in component_obj.actions: raise ValueError( f"Action {action_name} not found in component {component}" ) - cur_actions = EvalableList() - for a in component_obj.actions: - a = a.model_copy() - a._set_n_calls(action_counts.get(a.name, 0) * scale) - cur_actions.append(a) - new_component_to_actions[component] = cur_actions - component_to_actions = new_component_to_actions component_latency = {} + component_level_latency = {} arch_vars = dict(spec.arch.variables) if spec.arch.variables else {} symbol_table_base = { # TODO: Make a global symbol table initialization function @@ -235,20 +253,39 @@ def component_latency( continue component_obj = name2component[component] dump = component_obj.shallow_model_dump(include_None=True) - # Replace serialized `actions` dump with local Action copies that carry - # the correct n_calls for this job, so formulas can access `a.n_calls`, - # `a.throughput`, etc. without mutating the shared spec state. - if component in component_to_actions: - dump["actions"] = component_to_actions[component] if component in component_to_keywords: dump |= component_to_keywords[component] - symbol_table = {**symbol_table_base, **dump} - if component_obj.total_latency is not None: - component_latency[component] = eval_expression( + + def eval_latency(action_counts): + symbol_table = {**symbol_table_base, **dump} + cur_actions = EvalableList() + for action, count in action_counts.items(): + a = component_obj.actions[action].model_copy() + a._set_n_calls(count * component_obj.actions_scale) + cur_actions.append(a) + symbol_table["actions"] = cur_actions + + return eval_expression( component_obj.total_latency, symbol_table, attr_name="latency", location=component, ) - return component_latency + counts = component_to_actions[component] + if component in component_to_level_actions: + actions = component_to_level_actions[component] + # Evaluate on running action totals from the deepest level up so each + # level's latency includes the levels below it and the shallowest level + # is the whole latency. + cumulative = defaultdict(lambda: 0) + per_level = component_level_latency[component] = {} + for level, cur_actions in sorted(actions.items(), reverse=True): + for action, count in cur_actions.items(): + cumulative[action] += count + per_level[level] = eval_latency(cumulative) + component_latency[component] = per_level[min(per_level)] + else: + component_latency[component] = eval_latency(counts) + + return component_latency, component_level_latency diff --git a/accelforge/model/main.py b/accelforge/model/main.py index 15647ea7..0ee22e34 100644 --- a/accelforge/model/main.py +++ b/accelforge/model/main.py @@ -175,15 +175,17 @@ def evaluate_mapping( job.memories_track_all = [ m.name for m in flattened_arch if isinstance(m, Memory) ] + job.components_track_latency = oset( + m.name for m in flattened_arch if isinstance(m, arch.TensorHolder) + ) + job.ignored_resources = oset() job.fusable_tensors = fusable_tensors & oset(job.tensor_to_relevancy) einsum = cur_spec.workload.einsums[job.einsum_name] infer_default_binding(job.mapping, job) - _, df, _, _, tensor2mapping, _ = run_model( - job, add_reservations=True - ) + _, df, _, _, tensor2mapping, _ = run_model(job, add_reservations=True) # Calculate iteration counts and rank columns _clean_energy_columns(df, job.metrics) diff --git a/accelforge/model/run_model.py b/accelforge/model/run_model.py index 48c7c8fa..1f1058e0 100644 --- a/accelforge/model/run_model.py +++ b/accelforge/model/run_model.py @@ -19,6 +19,7 @@ from accelforge.mapper.FFM._join_pmappings.pmapping_dataframe import ( memory_usage2col, reservation2col, + complatency2col, tensor2col, firstlatency2col, action2col, @@ -47,7 +48,13 @@ def run_model( job, add_reservations=add_reservations ) - latency = component_latency(reuse, job.flattened_arch, pmapping, spec) + latency, latency_per_level = component_latency( + reuse, + job.flattened_arch, + pmapping, + spec, + per_level_components=oset(job.components_track_latency), + ) overall_latency = max_nonzero(*latency.values()) # try: @@ -223,18 +230,30 @@ def run_model( for component, cur_latency in latency.items(): df[f"component_latency{component}"] = cur_latency * n_instances + # Components shared across Einsums get per-level latency columns so joining can + # sum their busy time across Einsums and let it overlap with the other Einsums' + # latency. Their latency is folded into Totallatency at joining time + # instead of here. + for component, level_latency in latency_per_level.items(): + for level, cur_latency in level_latency.items(): + df[complatency2col(component, level)] = cur_latency * n_instances + # Total latency of each component is its own total_latency plus the worst # communication latency to reach it. comm_latency = communication_latency( reuse, job.flattened_arch, tensor_to_backing, - workload.einsums[job.einsum_name].output_tensor_names + workload.einsums[job.einsum_name].output_tensor_names, ) - per_component_total = [ - latency.get(component, 0) + comm_latency.get(component, 0) - for component in oset(latency) | oset(comm_latency) - ] + per_component_total = [] + for component in oset(latency) | oset(comm_latency): + l = comm_latency.get(component, 0) + if component not in latency_per_level: + l += latency.get(component, 0) + if not isinstance(l, Number) or l != 0: + per_component_total.append(l) + df["Totallatency"] = max_nonzero(*per_component_total) * n_instances # For first latency, we'll follow the convention of treating compute # as a component, similarly to memory (see below). @@ -260,7 +279,7 @@ def run_model( # ================================================================================= per_memory_spatial_usage_df = {} for memory, occupancies in total_occupancy.items(): - ignored = job.ignored_resources is not None and memory in job.ignored_resources + ignored = memory in job.ignored_resources key = f"usagememory{memory}" if not ignored: per_memory_spatial_usage_df[key] = ( diff --git a/tests/input_files/fused_matmuls_to_networked.mapping.yaml b/tests/input_files/fused_matmuls_to_networked.mapping.yaml new file mode 100644 index 00000000..6e915ab0 --- /dev/null +++ b/tests/input_files/fused_matmuls_to_networked.mapping.yaml @@ -0,0 +1,61 @@ +mapping: + nodes: + - !Storage + component: MainMemory + tensors: [T0, T2, W0, W1] + - !Storage + component: GlobalBuffer + tensors: [W0, W1] + - !Temporal + rank_variable: m + tile_shape: 1 + - !Sequential + nodes: + - !Nested + nodes: + - !Storage + component: GlobalBuffer + tensors: [T0, T1] + - !Spatial + rank_variable: n0 + tile_shape: {{ MAC_TILE }} + component: Scratchpad + name: X + - !Storage + component: Scratchpad + tensors: [T0, T1, W0] + - !Temporal + rank_variable: n1 + tile_shape: 1 + - !Spatial + rank_variable: n0 + tile_shape: 1 + component: MAC + name: X + - !Compute + einsum: Matmul0 + component: MAC + - !Nested + nodes: + - !Storage + component: GlobalBuffer + tensors: [T1, T2] + - !Spatial + rank_variable: n1 + tile_shape: {{ MAC_TILE }} + component: Scratchpad + name: X + - !Storage + component: Scratchpad + tensors: [T1, T2, W1] + - !Temporal + rank_variable: n2 + tile_shape: 1 + - !Spatial + rank_variable: n1 + tile_shape: 1 + component: MAC + name: X + - !Compute + einsum: Matmul1 + component: MAC diff --git a/tests/input_files/latency.arch.yaml b/tests/input_files/latency.arch.yaml new file mode 100644 index 00000000..c81af779 --- /dev/null +++ b/tests/input_files/latency.arch.yaml @@ -0,0 +1,43 @@ +arch: + nodes: + - !Memory + name: MainMemory + size: inf + leak_power: 0 + area: 0 + tensors: {keep: All} + actions: + - name: read + energy: 1 + throughput: {{ MM_READ_THROUGHPUT | default("inf") }} + latency: {{ MM_READ_LATENCY | default(0) }} + - name: write + energy: 1 + throughput: {{ MM_WRITE_THROUGHPUT | default("inf") }} + latency: {{ MM_WRITE_LATENCY | default(0) }} + + - !Memory + name: GlobalBuffer + size: inf + leak_power: 0 + area: 0 + tensors: {keep: All} + actions: + - name: read + energy: 1 + throughput: {{ GB_READ_THROUGHPUT | default("inf") }} + latency: {{ GB_READ_LATENCY | default(0) }} + - name: write + energy: 1 + throughput: {{ GB_WRITE_THROUGHPUT | default("inf") }} + latency: {{ GB_WRITE_LATENCY | default(0) }} + + - !Compute + name: MAC + leak_power: 0 + area: 0 + actions: + - name: compute + energy: 1 + throughput: {{ COMPUTE_THROUGHPUT | default(1) }} + latency: {{ COMPUTE_LATENCY | default(0) }} diff --git a/tests/input_files/networked_latency.arch.yaml b/tests/input_files/networked_latency.arch.yaml new file mode 100644 index 00000000..71bf185a --- /dev/null +++ b/tests/input_files/networked_latency.arch.yaml @@ -0,0 +1,58 @@ +arch: + nodes: + - !Memory + name: MainMemory + size: inf + area: 0 + leak_power: 0 + tensors: {keep: All} + actions: + - {name: read, energy: 0, throughput: inf, latency: 0} + - {name: write, energy: 0, throughput: inf, latency: 0} + + - !Memory + name: GlobalBuffer + size: inf + area: 0 + leak_power: 0 + tensors: {keep: ~MainMemory, may_keep: All} + actions: + - {name: read, energy: 0, throughput: inf, latency: 0} + - {name: write, energy: 0, throughput: inf, latency: 0} + + - !Network + name: PeArray + area: 0 + leak_power: 0 + actions: + - {name: hop, energy: 1, latency: {{ HOP_LATENCY | default(1) }}, throughput: inf} + + - !Memory + name: Scratchpad + size: inf + area: 0 + leak_power: 0 + tensors: {keep: All} + actions: + - {name: read, energy: 0, throughput: inf, latency: 0} + - {name: write, energy: 0, throughput: inf, latency: 0} + spatial: + - {name: X, fanout: 4} + + # All-to-all switch (NVLink-like): every node is one hop from every other + - !Network + name: MacArray + topology: all_to_all + area: 0 + leak_power: 0 + actions: + - {name: hop, energy: 1, latency: {{ HOP_LATENCY | default(1) }}, throughput: inf} + + - !Compute + name: MAC + area: 0 + leak_power: 0 + actions: + - {name: compute, energy: 0, throughput: inf, latency: 0} + spatial: + - {name: X, fanout: 4} diff --git a/tests/test_latency.py b/tests/test_latency.py new file mode 100644 index 00000000..c9790c8f --- /dev/null +++ b/tests/test_latency.py @@ -0,0 +1,135 @@ +""" +Tests for the latency model: communication (wind-up/down) latency, and per-level +component latency that lets a shared memory's busy time overlap with other Einsums' +compute during fusion. + +All tests use the matmul-chain workload with M = KN = 4 and 8 bits per value, so each +Einsum does 64 MACs and each tensor is 128 bits. +""" + +import unittest +from pathlib import Path + +import accelforge as af +from accelforge.frontend.spec import Spec +from accelforge.model.main import evaluate_mapping + +TESTS_DIR = Path(__file__).resolve().parent +INPUT_FILES_DIR = TESTS_DIR / "input_files" +LATENCY_ARCH = INPUT_FILES_DIR / "latency.arch.yaml" +NETWORKED_ARCH = INPUT_FILES_DIR / "networked_latency.arch.yaml" +NETWORKED_MAPPING = INPUT_FILES_DIR / "fused_matmuls_to_networked.mapping.yaml" + + +def total_latency(arch, mapping, n_einsums, **jinja): + spec = Spec.from_yaml( + af.examples.workloads.basic.matmuls, + arch, + mapping, + jinja_parse_data={"N_EINSUMS": n_einsums, "M": 4, "KN": 4, **jinja}, + ) + result = evaluate_mapping(spec) + return result.data.iloc[0] + + +class TestCommunicationLatency(unittest.TestCase): + def test_communication_latency_in_total(self): + """Action latencies add wind-up/down (communication) latency to the total.""" + # With zero action latencies there is no communication latency and the total + # is the compute steady state: 64 MACs / 1 MAC per second. + row = total_latency( + LATENCY_ARCH, af.examples.mappings.fused_matmuls_to_simple, 1 + ) + self.assertEqual(row["Totallatency"], 64) + + row = total_latency( + LATENCY_ARCH, + af.examples.mappings.fused_matmuls_to_simple, + 1, + MM_READ_LATENCY=100, + MM_WRITE_LATENCY=200, + GB_READ_LATENCY=10, + GB_WRITE_LATENCY=20, + COMPUTE_LATENCY=3, + ) + # The worst input reaches compute in one MainMemory read + one GlobalBuffer + # write + read (100 + 20 + 10 = 130). The output follows with one MAC (3), + # winds back up through the GlobalBuffer (20 + 10 = 30), and is read-modify- + # written at MainMemory (100 + 200 = 300). 130 + 3 + 30 + 300 = 463, which + # dominates the 64-cycle compute steady state. + self.assertEqual(row["Totallatency"], 463) + + def test_fused_repays_communication_latency_each_switch(self): + """Fused Einsums exchange their intermediate tensor through the shared buffer + once per shared-loop iteration, and the communication latency of that exchange + is paid on every switch.""" + row = total_latency( + LATENCY_ARCH, + af.examples.mappings.fused_matmuls_to_simple, + 2, + MM_READ_LATENCY=100, + MM_WRITE_LATENCY=200, + GB_READ_LATENCY=10, + GB_WRITE_LATENCY=20, + COMPUTE_LATENCY=3, + ) + # T1 is backed at the GlobalBuffer below the fused m loop, so its wind-down + # repeats for each of the 4 m iterations. Matmul0: worst input 130, then + # (one MAC + GlobalBuffer write + read = 33) x 4 iterations = 262. Matmul1 + # is the unfused 463 from above (T1's inputs wind down 30 x 4 = 120 < 130 + # from MainMemory). 262 + 463 = 725. + self.assertEqual(row["Totallatency"], 725) + + def test_fused_through_slow_interconnect(self): + """Two Einsums fused through networks pay the hop latency of the intermediate + tensor's route on every shared-loop iteration.""" + mesh_hops = 4 # All 4 Scratchpad positions are used: 4 hops on the PeArray + switch_hops = 1 # The all-to-all MacArray is one hop for any route + down = mesh_hops + switch_hops # backing -> compute, and compute -> backing + # Matmul0: inputs wind down from MainMemory once (5 hops), then T1 winds up + # to its GlobalBuffer backing below the fused m loop on every one of the 4 + # iterations: 5 + 5 x 4 = 25. Matmul1 mirrors it: T1 winds down 5 x 4 = 20, + # then T2 winds up to MainMemory once: 20 + 5 = 25. + expected = 2 * (down + down * 4) + for hop_latency in [0, 1, 100]: + row = total_latency( + NETWORKED_ARCH, + NETWORKED_MAPPING, + 2, + MAC_TILE=1, + HOP_LATENCY=hop_latency, + ) + self.assertEqual(row["Totallatency"], expected * hop_latency) + + +class TestSharedMemoryOverlap(unittest.TestCase): + def test_fused_memory_bound_overlaps_compute(self): + """When a compute-bound Einsum is fused with a memory-bound Einsum, the shared + memory's busy time is summed across the Einsums and overlapped with their + compute: the total is the max of the two, not the sum of per-Einsum maxes. + The tensors at MainMemory are backed above the shared m loop, so their + reservations are co-resident and their transfers may fill any slice of the + fused execution.""" + row = total_latency( + LATENCY_ARCH, + af.examples.mappings.fused_matmuls_to_simple, + 2, + MM_READ_THROUGHPUT=1, + MM_WRITE_THROUGHPUT=0.1, + COMPUTE_THROUGHPUT=0.2, + ) + # Matmul0 is compute-bound: 64 MACs / 0.2 = 320 vs reading T0 and W0 from + # MainMemory (256 bits / 1 = 256). Matmul1 is memory-bound: writing T2 back + # (128 bits / 0.1 = 1280) plus reading W1 (128 / 1 = 128) is 1408 vs 320. + self.assertEqual(row["Matmul0component_latencyMAC"], 320) + self.assertEqual(row["Matmul0component_latencyMainMemory"], 256) + self.assertEqual(row["Matmul1component_latencyMAC"], 320) + self.assertEqual(row["Matmul1component_latencyMainMemory"], 1408) + + # max(320 + 320, 256 + 1408) = 1664: Matmul0's MainMemory slack absorbs part + # of Matmul1's traffic. Summing per-Einsum maxes would give 320 + 1408 = 1728. + self.assertEqual(row["Totallatency"], 1664) + + +if __name__ == "__main__": + unittest.main() From deb02937b187ce0686fedf87cb5c797985aaaaaf Mon Sep 17 00:00:00 2001 From: Tanner Andrulis Date: Mon, 3 Aug 2026 12:46:19 -0400 Subject: [PATCH 4/8] [mapper, model] Split latency for parent/child communication, update some DF columns --- accelforge/frontend/arch/components.py | 88 +++--- .../make_pmappings_from_templates.py | 3 +- .../make_tile_shapes.py | 76 +++-- .../mapper/FFM/_make_pmappings/pmapper_job.py | 4 + .../mapper/FFM/_pareto_df/df_convention.py | 50 +++- accelforge/model/_looptree/latency/memory.py | 262 +++++++++--------- .../model/_looptree/reuse/symbolic/_common.py | 10 - .../model/_looptree/reuse/symbolic/_stats.py | 109 ++++++-- .../_looptree/reuse/symbolic/_symbolic.py | 58 ++-- accelforge/model/main.py | 50 ++-- accelforge/model/run_model.py | 46 +-- .../modeling/accelerator_energy_latency.rst | 17 +- examples/arches/eyeriss.yaml | 13 +- examples/arches/nvdla.yaml | 2 +- examples/arches/tpu_v4i.yaml | 2 +- .../fused_matmuls_weights_inside.mapping.yaml | 34 +++ tests/input_files/latency.arch.yaml | 1 + tests/network/test_network.py | 7 +- .../test_api_gaps.py | 18 +- .../test_component_fields.py | 51 ++-- .../test_yaml_and_expressions.py | 21 +- 21 files changed, 525 insertions(+), 397 deletions(-) create mode 100644 tests/input_files/fused_matmuls_weights_inside.mapping.yaml diff --git a/accelforge/frontend/arch/components.py b/accelforge/frontend/arch/components.py index 41cff91c..4841688d 100644 --- a/accelforge/frontend/arch/components.py +++ b/accelforge/frontend/arch/components.py @@ -113,35 +113,6 @@ class Action(EvalableModel): that are known to the component model, but not accelforge, such as clock frequency.""" - _n_calls: int | float = PrivateAttr(default=0) - """ - How many times this action is performed in the current Einsum. DO NOT SET THIS - VALUE. Used by the model. - """ - - _model_running: bool = PrivateAttr(default=False) - """ Guard: True only while the model is actively populating ``_n_calls``. Reading - ``n_calls`` outside of that window raises. Set via ``_set_n_calls``. """ - - @property - def n_calls(self) -> int | float: - """ - The number of times this action is performed in the current Einsum. When - accessed through the total_latency expression, returns the number of calls of - this action in the current Einsum. - """ - if not self._model_running: - raise RuntimeError( - f"Action {self.name!r}.n_calls is only valid while the model is " - f"running; access it from a Component.total_latency expression rather " - f"than directly." - ) - return self._n_calls - - def _set_n_calls(self, value: int | float) -> None: - self._n_calls = value - self._model_running = True - def _attributes_for_component_model(self) -> dict[str, Any]: return { **self.shallow_model_dump(), @@ -295,26 +266,28 @@ class Component(Spatialable): component's energy and latency. """ - total_latency: str | int | float = "sum(a.n_calls / a.throughput for a in actions)" + total_latency: Any = None """ - An expression representing the total latency of this component in seconds. This is - used to calculate the latency of a given Einsum. Special variables available are the - following: - - - `min`: The minimum value of all arguments to the expression. - - `max`: The maximum value of all arguments to the expression. - - `sum`: The sum of all arguments to the expression. - - `actions`: The list of ``Action`` objects for this component. Action attributes - include n_calls as well as all generally-available action attributes. - - Additionally, all component attributes are availble as variables, and all other - functions generally available in parsing. Note this expression is evaluated after - other component attributes are evaluated. - - For example, the following expression takes the max bound across separate ports: - ``max(a.n_calls / a.throughput for a in actions)``. + REMOVED. A component's latency is the sum, over its actions, of the action + count divided by the action's ``throughput``. Set per-action ``throughput`` + to control latency. For a storage with separate read and write ports, set + ``separate_read_write_ports: True`` instead of a max-over-ports expression. """ + @field_validator("total_latency") + @classmethod + def _total_latency_removed(cls, v): + if v is not None: + raise ValueError( + "total_latency was removed. A component's latency is the sum over " + "its actions of the action count divided by the action's " + "throughput, so set per-action `throughput` attributes instead. " + "For a storage with separate read and write ports, set " + "`separate_read_write_ports: True` to model read and write latency " + "independently." + ) + return v + latency_scale: EvalsTo[int | float] = 1 """ The scale factor for the latency of this component. Multiplies the calculated @@ -1199,6 +1172,13 @@ class Memory(TensorHolder): actions: EvalableList[TensorHolderAction] = MEMORY_ACTIONS """ The actions that this `Memory` can perform. """ + separate_read_write_ports: EvalsTo[bool] = False + """ + If True, reads and writes use independent ports: for latency, this memory is + treated as two components, "{name} (read)" and "{name} (write)", which may + overlap in time. If False, reads and writes are serialized. + """ + _n_physical: NoParse[int] = 1 """ Number of physical units bound to this memory level. @@ -1341,18 +1321,12 @@ class Network(Component, Leaf): actions: EvalableList[Action] = NETWORK_ACTIONS - total_latency: str | int | float = "max_link_traffic/actions['hop'].throughput" + total_latency: None = None """ - Models latency as bandwidth-bound, which means that the traffic over the most - congested link dominates the overall communication latency. Note that max_hops * - actions['hop'].latency will already be included in wind-up and wind-down of the - network. - - Keywords: - - - `max_hops` returns the number of hops in the longest route. - - `max_link_traffic` returns the amount of traffic (in bits) over the most congested - link. + REMOVED. A network's latency is bandwidth-bound: the traffic over the most + congested link divided by the hop action's ``throughput``. Wind-up/down uses + max_hops * the hop action's ``latency``. Setting this attribute raises an + error. """ bits_per_value: EvalsTo[dict] = {} diff --git a/accelforge/mapper/FFM/_make_pmappings/make_pmappings_from_templates/make_pmappings_from_templates.py b/accelforge/mapper/FFM/_make_pmappings/make_pmappings_from_templates/make_pmappings_from_templates.py index d70a88cd..08668ef8 100755 --- a/accelforge/mapper/FFM/_make_pmappings/make_pmappings_from_templates/make_pmappings_from_templates.py +++ b/accelforge/mapper/FFM/_make_pmappings/make_pmappings_from_templates/make_pmappings_from_templates.py @@ -46,6 +46,7 @@ def _imperfect_per_loop(rank_var, mapping) -> tuple[bool, ...]: SameTemplateJobs, ) from accelforge.mapper.FFM._pareto_df.df_convention import ( + is_energy_col, is_fused_loop_col, is_n_iterations_col, ) @@ -413,7 +414,7 @@ def make_pmappings_from_templates( mappings[v] = mappings[f"{einsum_name}{k}"] mappings = shift_reservations_by_null_loop_indices(mappings, null_loop_indices) - energy_cols = [c for c in mappings.columns if "Totalenergy" in c] + energy_cols = [c for c in mappings.columns if is_energy_col(c)] if (mappings[energy_cols] < 0).any(axis=None): mapping_with_negative_energy = mappings[ (mappings[energy_cols] < 0).any(axis=1) diff --git a/accelforge/mapper/FFM/_make_pmappings/make_pmappings_from_templates/make_tile_shapes.py b/accelforge/mapper/FFM/_make_pmappings/make_pmappings_from_templates/make_tile_shapes.py index 14085f91..32251dbf 100755 --- a/accelforge/mapper/FFM/_make_pmappings/make_pmappings_from_templates/make_tile_shapes.py +++ b/accelforge/mapper/FFM/_make_pmappings/make_pmappings_from_templates/make_tile_shapes.py @@ -23,6 +23,15 @@ ) from accelforge.mapper.FFM._make_pmappings.pmapper_job import Job from accelforge.mapper.FFM._pareto_df.df_convention import ( + SEP, + col2complatency, + col2reservation, + is_action_col, + is_energy_col, + is_objective_col, + is_usage_col, + is_latency_col, + is_tensor_col, stride2col, initial2col, iterations2col, @@ -2052,9 +2061,8 @@ def makesymbol(name: str): def make_keep_symbols(pmapping: Mapping) -> set[Symbol]: keep_symbols = oset() for node in pmapping.nodes: - if ( - (isinstance(node, Loop) and node._fused) - or (isinstance(node, Spatial) and node._shared_tensor_binding) + if (isinstance(node, Loop) and node._fused) or ( + isinstance(node, Spatial) and node._shared_tensor_binding ): if isinstance(node.initial_tile_shape, Symbol): keep_symbols.add(node.initial_tile_shape) @@ -2327,21 +2335,19 @@ def _to_sp(v): # ================================================================================== # Memory usage and usage constraints. # ================================================================================== + # Each resource's usage is its reservation at the first level other Einsums + # can never share (the reservation with the most loops). Only that one gets + # max_value=1: shallower reservation columns are running subsets of it. for k, v in {**per_memory_usage_df, **usage_df}.items(): # If we only track for pmappings, we only care if it's valid. If we track for # all, we care about the value too. - split = k.split("") - assert split[0] == "usage", f"invalid {split}" - if split[1] == "spatial": - assert len(split) == 4 - elif split[1] == "memory": - assert len(split) == 3 - else: - assert False, f"invalid {split}" + reservation = col2reservation(k) + assert reservation is not None, f"invalid usage column {k}" + resource = reservation.name only_care_if_valid = False - if split[2] in job.memories_track_pmappings_only: + if resource in job.memories_track_pmappings_only: only_care_if_valid = True # TODO: Update check to see if we may be sharing usage with other @@ -2364,7 +2370,7 @@ def _to_sp(v): # with a different number of objectives). absolute_tolerance = tolerance - if split[-1] not in job.memories_track_pmappings_only: + if resource not in job.memories_track_pmappings_only: absolute_tolerance /= job.workload_n_einsums objectives.append( @@ -2388,13 +2394,17 @@ def _to_sp(v): component_name, name, ), constraint in job.constraints.min_usage_constraints.items(): - usage_key = f"usagespatial{component_name}{name}" - if usage_key not in usage_df: - continue + n = f"{component_name} {name}" + rcols = [(k, col2reservation(k)) for k in usage_df] + matches = [r for r in rcols if r[1] is not None and r[1].name == n] + assert matches, ( + f"min_usage constraint on {component_name} {name} matches no usage " + f"column; have {sorted(usage_df)}" + ) objectives.append( Objective( name=f"min_usage_{component_name}_{name}", - formula=usage_df[usage_key], + formula=usage_df[max(matches, key=lambda x: x[1].nloops)[0]], symbols=symbols, only_care_if_valid=True, min_value=constraint.min_usage, @@ -2423,15 +2433,37 @@ def _to_sp(v): # transformation of the values between these steps, so error doesn't stack (it's # just doing the same pruning, perhaps with a different number of objectives). for k, v in symbolic_df.items(): - if "Total" not in k: + # Objectives (total latency, total energy) + if is_objective_col(k): + pass + # Per-component latencies. Include as separate objectives since they may be + # overlapped during joining + elif col2complatency(k) is not None: + pass + # Handled by reservations above + elif col2reservation(k) is not None: continue + # Reservation sizes for allocs&frees in joining. The same for all pmappings with + # a compatibility, so ignore here + elif is_tensor_col(k): + continue + # For model. Skip. + elif ( + is_latency_col(k) or is_action_col(k) or is_energy_col(k) or is_usage_col(k) + ): + continue + else: + raise ValueError( + f"Column {k} is not handled by the tile-shape search objective " + f"selection. Add it above as an objective (pass) or not (continue)." + ) objectives.append( Objective( name=k, formula=v, symbols=symbols, - terms_do_not_cross_zero="energy" in k or "latency" in k, + terms_do_not_cross_zero=is_energy_col(k) or is_latency_col(k), tolerance=job.objective_tolerance, ) ) @@ -2500,11 +2532,11 @@ def _to_sp(v): t0 = time.time() for key in compiled_df: df[key] = call_compiled_objective(compiled_df[key], *choices_float.T) - if "latency" in key and "first_latency" not in key: + if is_latency_col(key): val = [df[key]] if isinstance(df[key], Number) else df[key] if any(l < 0 for l in val): raise ValueError(f"Negative latency for {key}: {val}") - if "energy" in key: + if is_energy_col(key): arr = df[key] if isinstance(arr, Number): if arr < 0: @@ -2531,7 +2563,7 @@ def _to_sp(v): df = pd.DataFrame(df, columns=df.keys(), index=[0]) assert not df.isna().any().any() - energy_cols = [c for c in df.columns if "energy" in c] + energy_cols = [c for c in df.columns if is_energy_col(c)] if (df[energy_cols] < 0).any(axis=None): for col in energy_cols: series = df[col] diff --git a/accelforge/mapper/FFM/_make_pmappings/pmapper_job.py b/accelforge/mapper/FFM/_make_pmappings/pmapper_job.py index 9e1daba5..e0686627 100755 --- a/accelforge/mapper/FFM/_make_pmappings/pmapper_job.py +++ b/accelforge/mapper/FFM/_make_pmappings/pmapper_job.py @@ -65,9 +65,12 @@ class Job: compatibility: Compatibility | None = None memories_track_all: list[str] | None = None + """ Memories to track for reservations that may be shared accross pmappings.""" memories_track_pmappings_only: list[str] | None = None + """ Memories to track for reservations that are never shared accross pmappings.""" ignored_resources: set[str] | None = None components_track_latency: set[str] | None = None + """ Components whose latency may be shared across pmappings """ time_limit: float | int = float("inf") memory_limit: float | int = float("inf") messages: list[str] = field(default_factory=list) @@ -75,6 +78,7 @@ class Job: tensor_to_relevancy: ( dict[TensorName, dict[RankVariable, Relevant | PartiallyRelevant]] | None ) = None + """ Tensor --> dict of rank variables: relevancy""" n_total_pmappings: int = 1 n_valid_pmappings: int = 1 diff --git a/accelforge/mapper/FFM/_pareto_df/df_convention.py b/accelforge/mapper/FFM/_pareto_df/df_convention.py index dbfa435c..36cf9f9b 100755 --- a/accelforge/mapper/FFM/_pareto_df/df_convention.py +++ b/accelforge/mapper/FFM/_pareto_df/df_convention.py @@ -138,20 +138,20 @@ def col2energy(colname: str) -> ActionKey | VerboseActionKey: @dict_cached def col2complatency(x: str) -> ComponentLatencyKey | None: """ - Format: component_latency name nloops. Distinct from the 2-part - component_latencyname columns, these are specific to a loop level and may be - pushed to work during earlier Einsums + Format: level_latency component_name nloops. One column per (component, loop level). + Unlike component_latencycomponent_name, these are used to calculate total + latency because they may be shared across Einsums. """ parts = x.split(SEP) - if len(parts) != 3 or parts[0] != "component_latency": + if len(parts) != 3 or parts[0] != "level_latency": return None return ComponentLatencyKey(parts[1], int(parts[2])) @dict_cached def complatency2col(name: str, nloops: int) -> str: - """Format: component_latency name nloops""" - return f"component_latency{name}{nloops}" + """Format: level_latency name nloops""" + return f"level_latency{name}{nloops}" @dict_cached @@ -208,12 +208,6 @@ def col2iterations(col: str) -> int | None: return x[0] -@dict_cached -def firstlatency2col(name: str, nloops: int) -> str: - """Format: first latency name level""" - return f"first_latency{name}{nloops}" - - @dict_cached def tensor2col(tensor: str) -> str: """Format: tensor tensor_name""" @@ -234,6 +228,38 @@ def is_tensor_col(c: str) -> bool: return c.startswith("tensor") +@dict_cached +def is_action_col(c: str) -> bool: + return c.startswith(f"action{SEP}") + + +@dict_cached +def is_usage_col(c: str) -> bool: + return c.startswith(f"usage{SEP}") + + +@dict_cached +def is_latency_col(c: str) -> bool: + return ( + c == f"Total{SEP}latency" + or c.split(SEP)[0] == "component_latency" + or col2complatency(c) is not None + ) + + +@dict_cached +def is_energy_col(c: str) -> bool: + parts = c.split(SEP) + if parts[0] == "energy": + return True + return parts[0] == "Total" and parts[1] in ( + "energy", + "dynamic_energy", + "leak_energy", + "energy_delay_product", + ) + + @dict_cached def col2nameloopleft(x: str) -> tuple[str, int, bool] | None: """Format: reservation name level left""" diff --git a/accelforge/model/_looptree/latency/memory.py b/accelforge/model/_looptree/latency/memory.py index 0d37d9d7..78898817 100755 --- a/accelforge/model/_looptree/latency/memory.py +++ b/accelforge/model/_looptree/latency/memory.py @@ -2,23 +2,19 @@ from numbers import Number from accelforge.frontend import arch -from accelforge.frontend.arch import Leaf, Memory, TensorHolder, Component +from accelforge.frontend.arch import Memory, TensorHolder, Component from accelforge.frontend.arch._flattened_arch import FlattenedArch from accelforge.frontend.mapping import Compute, Mapping -from accelforge.frontend.spec import Spec +from accelforge.frontend.mapping import TensorHolder as MappingTensorHolder from accelforge.model._looptree.accesses import isl_buffer_accesses_from_buffet_actions -from accelforge.model._looptree.mapping_utilities import get_leaves from accelforge.model._looptree.reuse.isl import IslReuseAnalysisOutput from accelforge.model._looptree.reuse import SymbolicAnalysisOutput from accelforge.model._looptree.types import Buffet from accelforge.model._looptree.reuse.symbolic import BuffetStats, NetworkStats -from accelforge.util._eval_expressions import MATH_FUNCS, eval_expression from accelforge.util._frozenset import oset -from accelforge.util._sympy.broadcast_max import Max, Min, MaxGeqZero, max_nonzero -from accelforge.util._basetypes import EvalableList -import symengine as se +from accelforge.util._sympy.broadcast_max import Max, MaxGeqZero, max_nonzero def isl_to_summarized( @@ -41,29 +37,6 @@ def isl_to_summarized( return SymbolicAnalysisOutput(buffet_stats=buffet_stats) -def _sum(*args): - """Sum that accepts either a single iterable (e.g. a generator) or varargs, so - total_latency expressions like ``sum(x for x in ...)`` and ``sum(*values)`` both - evaluate to a symengine expression.""" - if len(args) == 1 and hasattr(args[0], "__iter__"): - args = tuple(args[0]) - return se.Add(*args) if args else se.Integer(0) - - -def _max(*args): - """Max that accepts either a single iterable (generator) or varargs.""" - if len(args) == 1 and hasattr(args[0], "__iter__"): - args = tuple(args[0]) - return Max(*args) - - -def _min(*args): - """Min that accepts either a single iterable (generator) or varargs.""" - if len(args) == 1 and hasattr(args[0], "__iter__"): - args = tuple(args[0]) - return Min(*args) - - def communication_latency( reuse: SymbolicAnalysisOutput, flattened_arch: FlattenedArch, @@ -135,18 +108,26 @@ def component_latency( looptree_results: SymbolicAnalysisOutput, flattened_arch: FlattenedArch, mapping: Mapping, - spec: Spec, - per_level_components: oset = oset(), + per_level_components: oset, + n_shared_loops: int, ): """ Returns (component latency, per-level component latency). + A component's latency is the sum, over its actions, of the action count (times the + component's actions_scale) divided by the action's throughput. A Memory with + separate_read_write_ports is two components for latency, "{name} (read)" and "{name} + (write)". A network's `hop` count is the traffic over its most congested link. + The per-level dict covers components in per_level_components, mapping each to {n_loops_above: latency of the actions at that many loops above plus the latency of actions lower in the tree}. Each level includes the levels below it, so the - shallowest level is the component's whole latency. Levels must be completed before - moving to other branches of the LoopTree, but may be overlapped with the latencies - of other branches below the current level. + shallowest level is the component's whole latency. + + A transfer can only run while both ends' reservations are alive, so a storage's + actions exchanged with its parent land at the storage's own loop level and actions + exchanged with its child (or peers) at the child's level. Levels deeper than + n_shared_loops cannot be shared with other Einsums and are collapsed into one. """ component_to_actions: dict[str, dict[str, float]] = defaultdict( lambda: defaultdict(lambda: 0) @@ -154,125 +135,144 @@ def component_latency( component_to_level_actions: dict[str, dict[int, dict[str, float]]] = defaultdict( lambda: defaultdict(lambda: defaultdict(lambda: 0)) ) - # Holds ``keywords" that do not map neatly to actions, e.g., max_hops for network - component_to_keywords: dict[str, dict[str, float]] = defaultdict( - lambda: defaultdict(lambda: 0) - ) name2component: dict[str, Component] = {node.name: node for node in flattened_arch} + # May be different than actual name if reads and writes are separated + latency_name2component: dict[str, Component] = {} + compute_obj = flattened_arch[-1] if not isinstance(compute_obj, arch.Compute): raise ValueError("Last node in flattened_arch must be a Compute") - for buffet, buffet_stats in looptree_results.buffet_stats.items(): - component = buffet.level - actions = component_to_actions[component] - if component not in name2component: - raise ValueError(f"Component {component} found in mapping but not arch") - - for action in name2component[component].actions: - actions[action.name] += 0 - - if isinstance(name2component[component], TensorHolder): - reads = buffet_stats.net_max_per_unit_actions("read") - writes = buffet_stats.net_max_per_unit_actions("write") - actions["read"] += reads - include_writes = not isinstance(name2component[component], arch.Toll) - n_loops_above = buffet_stats.n_loops_above - if include_writes: - actions["write"] += writes - if component in per_level_components: - level_actions = component_to_level_actions[component][n_loops_above] - level_actions["read"] += reads - if include_writes: - level_actions["write"] += writes - elif isinstance(name2component[component], arch.Compute): - pass - else: + def add(component_obj: Component, action: str, count, level=None): + if action == "write" and isinstance(component_obj, arch.Toll): + assert count == 0 + return + + name = component_obj.name + if getattr(component_obj, "separate_read_write_ports", False): + assert action in ("read", "write") + name = f"{name} ({'read' if action == 'read' else 'write'})" + + latency_name2component[name] = component_obj + count = 0 if count is None else count + component_to_actions[name][action] += count + if level is not None: + level = min(level, n_shared_loops) + component_to_level_actions[name][level][action] += count + + name2index = {node.name: i for i, node in enumerate(flattened_arch)} + tensor2buffets: dict[str, list] = defaultdict(list) + for buffet in looptree_results.buffet_stats: + component_obj = name2component.get(buffet.level) + if component_obj is None: + raise ValueError(f"Component {buffet.level} found in mapping but not arch") + if isinstance(component_obj, TensorHolder): + tensor2buffets[buffet.tensor].append(buffet) + for action in component_obj.actions: + add(component_obj, action.name, 0) + elif not isinstance(component_obj, arch.Compute): raise NotImplementedError( - f"Component {component} is not a TensorHolder or Compute" + f"Component {buffet.level} is not a TensorHolder or Compute" ) - network_to_max_link_traffic = defaultdict(lambda: defaultdict(lambda: 0)) - network_to_max_hops = defaultdict(lambda: []) - # Aggregates across tensors + mapping_order: dict = {} + for idx, node in enumerate(mapping.nodes): + if isinstance(node, MappingTensorHolder): + for tensor in node.tensors: + mapping_order.setdefault((node.component, tensor), idx) + mapping_order.setdefault(node.component, idx) + + def mapping_order_get(buffet): + # Need this extra logic for copy Einsums, where the mapping has all storage + # nodes changed to be only the copy source tensor + return mapping_order.get( + (buffet.level, buffet.tensor), mapping_order[buffet.level] + ) + + # ================================================================================= + # Buffet actions + # ================================================================================= + bs = looptree_results.buffet_stats + for bufs in tensor2buffets.values(): + bufs = sorted(bufs, key=mapping_order_get) + for i, buffet in enumerate(bufs): + stats = looptree_results.buffet_stats[buffet] + component_obj = name2component[buffet.level] + track = buffet.level in per_level_components + + # = above_loop_index + n_above = min(stats.n_loops_above, n_shared_loops) + n = i + 1 + child_above = bs[bufs[i + 1]].n_loops_above if n < len(bufs) else n_above + + for action, count in stats.net_max_per_unit_actions_to_parent().items(): + add(component_obj, action, count, n_above if track else None) + for action, count in stats.net_max_per_unit_actions_to_child().items(): + add(component_obj, action, count, child_above if track else None) + + # ================================================================================= + # Compute actions + # ================================================================================= + add( + compute_obj, + "compute", + Max(0, *[s.max_latency for s in looptree_results.compute_stats.values()]), + n_shared_loops if compute_obj.name in per_level_components else None, + ) + + # ================================================================================= + # Network actions + # ================================================================================== + network_to_level_dim_traffic = defaultdict( + lambda: defaultdict(lambda: defaultdict(lambda: 0)) + ) for network, network_stats in looptree_results.network_stats.items(): component = network.component if component not in name2component: raise ValueError(f"Component {component} found in mapping but not arch") - dim_traffic = network_to_max_link_traffic[component] + tensor_bufs = tensor2buffets[network.tensor] + n2i = name2index + bufs_after = [b for b in tensor_bufs if n2i[b.level] > n2i[component]] + below = min([bs[b].n_loops_above for b in bufs_after], default=n_shared_loops) + dim_traffic = network_to_level_dim_traffic[component][below] for dim, max_traffic_in_dim in network_stats.max_traffic.items(): dim_traffic[dim] += max_traffic_in_dim - network_to_max_hops[component].append(network_stats.max_hops) - - for network, network_stats in looptree_results.network_stats.items(): - component = network.component - keywords = component_to_keywords[component] - keywords["max_link_traffic"] = MaxGeqZero( - *network_to_max_link_traffic[component].values() - ) - keywords["max_hops"] = MaxGeqZero(*network_to_max_hops[component]) - actions = component_to_actions[component] - for action in name2component[component].actions: - actions[action.name] = 0 - - longest_compute_latency = Max( - 0, *[s.max_latency for s in looptree_results.compute_stats.values()] - ) - component_to_actions[compute_obj.name]["compute"] = longest_compute_latency - - for component, action_counts in component_to_actions.items(): + for component, level_dim_traffic in network_to_level_dim_traffic.items(): component_obj = name2component[component] - for action_name in action_counts: - if action_name not in component_obj.actions: - raise ValueError( - f"Action {action_name} not found in component {component}" - ) + for level, dim_traffic in level_dim_traffic.items(): + add( + component_obj, + "hop", + MaxGeqZero(*dim_traffic.values()), + level if component in per_level_components else None, + ) + # ================================================================================= + # Actions to latency + # ================================================================================= component_latency = {} component_level_latency = {} - arch_vars = dict(spec.arch.variables) if spec.arch.variables else {} - symbol_table_base = { # TODO: Make a global symbol table initialization function - **arch_vars, - **dict(spec.variables), - "variables": spec.variables, - "arch_variables": spec.arch.variables, - "max": _max, - "min": _min, - "sum": _sum, - } - - for component in name2component: - if ( - component not in component_to_actions - and component not in component_to_keywords - ): - continue - component_obj = name2component[component] - dump = component_obj.shallow_model_dump(include_None=True) - if component in component_to_keywords: - dump |= component_to_keywords[component] + for component, counts in component_to_actions.items(): + component_obj = latency_name2component[component] - def eval_latency(action_counts): - symbol_table = {**symbol_table_base, **dump} - cur_actions = EvalableList() + def counts2latency(action_counts): + total = 0 for action, count in action_counts.items(): - a = component_obj.actions[action].model_copy() - a._set_n_calls(count * component_obj.actions_scale) - cur_actions.append(a) - symbol_table["actions"] = cur_actions - - return eval_expression( - component_obj.total_latency, - symbol_table, - attr_name="latency", - location=component, - ) + if action not in component_obj.actions: + raise ValueError( + f"Action {action} not found in component {component}" + ) + total += ( + count + * component_obj.actions_scale + / component_obj.actions[action].throughput + ) + return total - counts = component_to_actions[component] if component in component_to_level_actions: actions = component_to_level_actions[component] # Evaluate on running action totals from the deepest level up so each @@ -283,9 +283,9 @@ def eval_latency(action_counts): for level, cur_actions in sorted(actions.items(), reverse=True): for action, count in cur_actions.items(): cumulative[action] += count - per_level[level] = eval_latency(cumulative) + per_level[level] = counts2latency(cumulative) component_latency[component] = per_level[min(per_level)] else: - component_latency[component] = eval_latency(counts) + component_latency[component] = counts2latency(counts) return component_latency, component_level_latency diff --git a/accelforge/model/_looptree/reuse/symbolic/_common.py b/accelforge/model/_looptree/reuse/symbolic/_common.py index c620c0ee..070e665b 100644 --- a/accelforge/model/_looptree/reuse/symbolic/_common.py +++ b/accelforge/model/_looptree/reuse/symbolic/_common.py @@ -56,16 +56,6 @@ class AnalysisInfo: tensor_rank_variables: set = field(default_factory=set) - # We track first latency for these nodes (should be Temporal) - last_temporal_node_idx: int = None - """ - node idx of the last (above) temporal node - """ - idxs_to_track_first_latency: set[int] = field(default_factory=set) - """ - node idxs for which we track first latency - """ - def reduce_dicts(dict1: dict, dict2: dict, reduce_op): for key in dict1: diff --git a/accelforge/model/_looptree/reuse/symbolic/_stats.py b/accelforge/model/_looptree/reuse/symbolic/_stats.py index c591a869..b20c9fd4 100644 --- a/accelforge/model/_looptree/reuse/symbolic/_stats.py +++ b/accelforge/model/_looptree/reuse/symbolic/_stats.py @@ -44,7 +44,8 @@ class ActionCounts(dict): def __missing__(self, key): return 0 - + + def _scale(value: Any, factor: Any) -> Any: if isinstance(value, ActionCounts): return ActionCounts({k: v * factor for k, v in value.items()}) @@ -78,11 +79,25 @@ class BuffetStats: max_occupancy: Any = field(default=0) _n_loops_above: int = field(default=0) - # These are used to calculate energy and latency. Keyed by action name. - total_actions: ActionCounts = field(default_factory=ActionCounts) - max_per_unit_actions: ActionCounts = field(default_factory=ActionCounts) - total_skipped_first_actions: ActionCounts = field(default_factory=ActionCounts) - min_per_unit_skipped_first_actions: ActionCounts = field( + # These are used to calculate energy and latency. Keyed by action name and + # split by which side of the storage the data moves on: exchanges with the + # parent (fills in, drains out) versus exchanges with the child or peers. + # By construction, skipped-first writes are parent-side and skipped-first + # reads child-side. total_actions and max_per_unit_actions are the sums. + total_actions_to_parent: ActionCounts = field(default_factory=ActionCounts) + total_actions_to_child: ActionCounts = field(default_factory=ActionCounts) + max_per_unit_actions_to_parent: ActionCounts = field(default_factory=ActionCounts) + max_per_unit_actions_to_child: ActionCounts = field(default_factory=ActionCounts) + total_skipped_first_actions_to_parent: ActionCounts = field( + default_factory=ActionCounts + ) + total_skipped_first_actions_to_child: ActionCounts = field( + default_factory=ActionCounts + ) + min_per_unit_skipped_first_actions_to_parent: ActionCounts = field( + default_factory=ActionCounts + ) + min_per_unit_skipped_first_actions_to_child: ActionCounts = field( default_factory=ActionCounts ) @@ -93,6 +108,36 @@ class BuffetStats: # Number of temporal iterations above this buffet's storage node. iterations_above: Any = field(default=1) + @property + def total_actions(self) -> ActionCounts: + return _combine( + self.total_actions_to_parent, self.total_actions_to_child, operator.add + ) + + @property + def max_per_unit_actions(self) -> ActionCounts: + return _combine( + self.max_per_unit_actions_to_parent, + self.max_per_unit_actions_to_child, + operator.add, + ) + + @property + def total_skipped_first_actions(self) -> ActionCounts: + return _combine( + self.total_skipped_first_actions_to_parent, + self.total_skipped_first_actions_to_child, + operator.add, + ) + + @property + def min_per_unit_skipped_first_actions(self) -> ActionCounts: + return _combine( + self.min_per_unit_skipped_first_actions_to_parent, + self.min_per_unit_skipped_first_actions_to_child, + operator.add, + ) + @property def n_loops_above(self) -> int: if self.persistent: @@ -134,8 +179,10 @@ def repeat_spatial(self, factor: int, reuse_parent_accesses: bool) -> "BuffetSta for k, v in new.__dict__.items(): if not k.startswith(("total_", "max_", "min_")): continue - if "parent" in k and reuse_parent_accesses: - continue # If parent accesses are reused, no need to multiply + # If parent accesses are reused, no need to multiply. Action count + # dicts always scale. + if "parent" in k and "actions" not in k and reuse_parent_accesses: + continue if "per_unit" in k: continue # Spatial fanout doesn't affect per-unit stats if k == "max_occupancy": @@ -195,6 +242,32 @@ def net_max_per_unit_actions(self, action: str | None = None) -> Any: {a: self.net_max_per_unit_actions(a) for a in self.max_per_unit_actions} ) + def net_max_per_unit_actions_to_parent(self, action: str | None = None) -> Any: + if action is not None: + return ( + self.max_per_unit_actions_to_parent[action] + - self.min_per_unit_skipped_first_actions_to_parent[action] + ) + return ActionCounts( + { + a: self.net_max_per_unit_actions_to_parent(a) + for a in self.max_per_unit_actions_to_parent + } + ) + + def net_max_per_unit_actions_to_child(self, action: str | None = None) -> Any: + if action is not None: + return ( + self.max_per_unit_actions_to_child[action] + - self.min_per_unit_skipped_first_actions_to_child[action] + ) + return ActionCounts( + { + a: self.net_max_per_unit_actions_to_child(a) + for a in self.max_per_unit_actions_to_child + } + ) + @classmethod def blank(cls): stats = cls() @@ -209,10 +282,6 @@ class ComputeStats: max_per_unit_ops: Any = field(default=0) # "max" below refers to the longest latency of any iteration max_latency: Any = field(default=0) - # Mapping from the loop-index (0 at top) to the latency of the first - # iteration of that loop. "Max" because we may have loops above that and we - # will take the maximum of the firsts. - max_first_latency: dict[int, Any] = field(default_factory=dict) def repeat_temporal(self, factor: int) -> "ComputeStats": new = copy.copy(self) @@ -223,7 +292,6 @@ def repeat_temporal(self, factor: int) -> "ComputeStats": new.total_ops = new.total_ops * factor new.max_per_unit_ops = new.max_per_unit_ops * factor new.max_latency = new.max_latency * factor - # NOTE: max_first_latency does not change return new def repeat_spatial(self, factor: int) -> "ComputeStats": @@ -240,22 +308,12 @@ def __add__(self, other: "ComputeStats") -> "ComputeStats": new.total_ops += other.total_ops new.max_per_unit_ops += other.max_per_unit_ops new.max_latency += other.max_latency - # max_first_latency is only ever updated across loops ABOVE the loop - # for which we calculated that first latency, so we should MAX - new.max_first_latency = max_dict( - self.max_first_latency, other.max_first_latency - ) # FIRST LATENCY return new def combine_temporal(self, other: "ComputeStats"): self.total_ops += other.total_ops self.max_per_unit_ops += other.max_per_unit_ops self.max_latency += other.max_latency - # max_first_latency is only ever updated across loops ABOVE the loop - # for which we calculated that first latency, so we should MAX - self.max_first_latency = max_dict( - self.max_first_latency, other.max_first_latency - ) # FIRST LATENCY def combine_spatial(self, other: "ComputeStats"): self.total_ops += other.total_ops @@ -263,11 +321,6 @@ def combine_spatial(self, other: "ComputeStats"): self.max_per_unit_ops, other.max_per_unit_ops ) self.max_latency = max_nonzero(self.max_latency, other.max_latency) - # max_first_latency is only ever updated across loops ABOVE the loop - # for which we calculated that first latency, so we should MAX - self.max_first_latency = max_dict( - self.max_first_latency, other.max_first_latency - ) # FIRST LATENCY @dataclass diff --git a/accelforge/model/_looptree/reuse/symbolic/_symbolic.py b/accelforge/model/_looptree/reuse/symbolic/_symbolic.py index 895d523f..1ef6faec 100755 --- a/accelforge/model/_looptree/reuse/symbolic/_symbolic.py +++ b/accelforge/model/_looptree/reuse/symbolic/_symbolic.py @@ -545,10 +545,7 @@ def analyze_temporal( result_accumulator = SymbolicAnalysisOutput() - first_latency = None - def handle_repeated_value(repeated_shape): - nonlocal first_latency shape_value = repeated_shape.value shape_repeats = repeated_shape.repeats @@ -580,9 +577,6 @@ def handle_repeated_value(repeated_shape): result_accumulator.max(fanout=child_result.fanout) for key in child_result.compute_stats: - if first_latency is None: - first_latency = child_result.compute_stats[key].max_latency - compute_stats = result_accumulator.compute_stats.setdefault( key, ComputeStats() ) @@ -591,8 +585,6 @@ def handle_repeated_value(repeated_shape): ) result_accumulator.compute_stats[key] = compute_stats - info.last_temporal_node_idx = node_idx - shape = stride_and_shape.shape if isinstance(shape, SequenceOfRepatedvalues): for repeated_shape in shape.sequence: @@ -606,12 +598,6 @@ def handle_repeated_value(repeated_shape): for stats in result_accumulator.buffet_stats.values(): stats.iterations_above *= total_iterations - if node_idx in info.idxs_to_track_first_latency: - for compute_stat in result_accumulator.compute_stats.values(): - # Should be the first time we store this value - assert node_idx not in compute_stat.max_first_latency - compute_stat.max_first_latency[node_idx] = first_latency - return result_accumulator @@ -865,14 +851,16 @@ def inherit_add(attr: str, default_value: Any = fills) -> Any: # ========================== # Data exchanges with parent if count_downward_movement[tensor]: # Parent -> Me - stats.total_actions["write"] += stats.total_reads_to_parent * write_scale - stats.max_per_unit_actions["write"] += ( + stats.total_actions_to_parent["write"] += ( + stats.total_reads_to_parent * write_scale + ) + stats.max_per_unit_actions_to_parent["write"] += ( stats.total_reads_to_parent * write_scale / n_active_physical_units ) - stats.total_skipped_first_actions["write"] += ( + stats.total_skipped_first_actions_to_parent["write"] += ( stats.total_skipped_first_reads_to_parent * write_scale ) - stats.min_per_unit_skipped_first_actions["write"] += ( + stats.min_per_unit_skipped_first_actions_to_parent["write"] += ( stats.min_per_parent_skipped_first_reads_to_parent * write_scale / n_active_physical_units @@ -881,42 +869,46 @@ def inherit_add(attr: str, default_value: Any = fills) -> Any: if count_upward_movement[tensor]: # Me -> Parent # Comment this to have the final writeback to a buffer hit both that buffer and # go directly to the parent without incurring another read from the buffer. - stats.total_actions["read"] += stats.total_writes_to_parent * read_scale - stats.max_per_unit_actions["read"] += ( + stats.total_actions_to_parent["read"] += ( + stats.total_writes_to_parent * read_scale + ) + stats.max_per_unit_actions_to_parent["read"] += ( stats.total_writes_to_parent * read_scale / n_active_physical_units ) # ======================== # Data exchanges with peer - stats.total_actions["read"] += stats.total_reads_to_peer * read_scale - stats.total_actions["write"] += stats.total_reads_to_peer * write_scale + stats.total_actions_to_child["read"] += stats.total_reads_to_peer * read_scale + stats.total_actions_to_child["write"] += stats.total_reads_to_peer * write_scale # ========================= # Data exchanges with child if child is not None: if count_downward_movement[tensor]: # Me -> Child - stats.total_actions["read"] += child.total_reads_to_parent * read_scale - stats.max_per_unit_actions["read"] += ( + stats.total_actions_to_child["read"] += ( + child.total_reads_to_parent * read_scale + ) + stats.max_per_unit_actions_to_child["read"] += ( child.max_per_parent_reads_to_parent * read_scale / n_active_physical_units ) # Skip first read if skip_initial: - stats.total_skipped_first_actions["read"] += ( + stats.total_skipped_first_actions_to_child["read"] += ( child.total_skipped_first_reads_to_parent * read_scale ) - stats.min_per_unit_skipped_first_actions["read"] += ( + stats.min_per_unit_skipped_first_actions_to_child["read"] += ( child.min_per_parent_skipped_first_reads_to_parent * read_scale / n_active_physical_units ) if count_upward_movement[tensor]: # Child -> Me - stats.total_actions["write"] += ( + stats.total_actions_to_child["write"] += ( child.total_writes_to_parent * write_scale ) - stats.max_per_unit_actions["write"] += ( + stats.max_per_unit_actions_to_child["write"] += ( child.max_per_parent_writes_to_parent * write_scale / n_active_physical_units @@ -956,11 +948,6 @@ def analyze_reservation(node_idx, current_shape, info: AnalysisInfo): node = mapping[node_idx] tensor = TensorName(node.purpose) - if info.last_temporal_node_idx is not None and id( - node - ) == info.tensor_to_reservation_backer_id.get(node.purpose, None): - info.idxs_to_track_first_latency.add(info.last_temporal_node_idx) - child_result = analyze_node(node_idx + 1, current_shape, info) buffet = Buffet(tensor, einsum_name, node.resource) @@ -1037,7 +1024,10 @@ def analyze_compute( if tensor in info.workload.einsums[einsum].output_tensor_names: stats.total_writes_to_parent = 1 stats.max_per_parent_writes_to_parent = 1 - stats.total_actions["compute"] = computes + # The to-parent/to-child thing really doesn't matter here since compute is + # the lowest leaf in the tree & any the level at which any latency is paid + # is set by the lower level of the exchange + stats.total_actions_to_parent["compute"] = computes if skip_initial: stats.total_skipped_first_reads_to_parent = 1 stats.min_per_parent_skipped_first_reads_to_parent = 1 diff --git a/accelforge/model/main.py b/accelforge/model/main.py index 0ee22e34..bf9f8e65 100644 --- a/accelforge/model/main.py +++ b/accelforge/model/main.py @@ -60,14 +60,31 @@ def evaluate_mapping( specification for that particular Einsum. If provided, then these will be used instead of re-parsing the specification. """ + from accelforge.mapper.FFM._join_pmappings.join_pmappings import ( + clean_compress_and_join_pmappings, + ) + + return clean_compress_and_join_pmappings( + pmappings=_model_pmappings(spec, flattened_arches, evaluated_specs), + metrics=spec.model.metrics, + print_progress=False, + for_model=True, + ) + + +def _model_pmappings( + spec: Spec, + flattened_arches: dict[(EinsumName, str), list[arch.Leaf]] | None = None, + evaluated_specs: dict[EinsumName, Spec] | None = None, +): + """ + Run the model on each pmapping, returning ready-to-join per-Einsum pmappings. + """ from accelforge.mapper.FFM._join_pmappings.compatibility import Compatibility from accelforge.mapper.FFM._join_pmappings.pmapping_dataframe import ( PmappingDataframe, ) from accelforge.mapper.FFM._join_pmappings.pmapping_group import PmappingGroup - from accelforge.mapper.FFM._join_pmappings.join_pmappings import ( - clean_compress_and_join_pmappings, - ) from accelforge.mapper.FFM.pmappings import MultiEinsumPmappings from accelforge.mapper.FFM._make_pmappings.make_pmappings import ( get_rank_variable_bounds_for_all_einsums, @@ -175,9 +192,7 @@ def evaluate_mapping( job.memories_track_all = [ m.name for m in flattened_arch if isinstance(m, Memory) ] - job.components_track_latency = oset( - m.name for m in flattened_arch if isinstance(m, arch.TensorHolder) - ) + job.components_track_latency = oset(m.name for m in flattened_arch) job.ignored_resources = oset() job.fusable_tensors = fusable_tensors & oset(job.tensor_to_relevancy) @@ -235,20 +250,15 @@ def evaluate_mapping( if einsum_name in einsum2pmappings } - return clean_compress_and_join_pmappings( - pmappings=MultiEinsumPmappings( - spec=spec, - einsum2pmappings=einsum2pmappings, - pmapping_objects=pmapping_objects, - einsum2jobs=einsum2jobs, - can_combine_multiple_runs=False, - einsums_with_pmappings_generated=oset(spec.workload.einsum_names), - flattened_arches=flattened_arches, - evaluated_specs=evaluated_specs, - ), - metrics=spec.model.metrics, - print_progress=False, - for_model=True, + return MultiEinsumPmappings( + spec=spec, + einsum2pmappings=einsum2pmappings, + pmapping_objects=pmapping_objects, + einsum2jobs=einsum2jobs, + can_combine_multiple_runs=False, + einsums_with_pmappings_generated=oset(spec.workload.einsum_names), + flattened_arches=flattened_arches, + evaluated_specs=evaluated_specs, ) diff --git a/accelforge/model/run_model.py b/accelforge/model/run_model.py index 1f1058e0..3a2ef247 100644 --- a/accelforge/model/run_model.py +++ b/accelforge/model/run_model.py @@ -1,7 +1,7 @@ from numbers import Number from sympy import Symbol import accelforge.frontend.arch as arch -from accelforge.frontend.mapping import TensorHolder, Toll +from accelforge.frontend.mapping import Loop, TensorHolder, Toll from accelforge.mapper.FFM._make_pmappings.pmapper_job import Job from accelforge.model._looptree.reuse import symbolic from accelforge.util._frozenset import oset @@ -21,7 +21,6 @@ reservation2col, complatency2col, tensor2col, - firstlatency2col, action2col, energy2col, ) @@ -48,12 +47,27 @@ def run_model( job, add_reservations=add_reservations ) + tensor_to_backing = {} + n_shared_loops = 0 + n_loops = 0 + for node in pmapping.nodes: + if isinstance(node, Loop): + n_loops += 1 + elif isinstance(node, TensorHolder): + for tensor in node.tensors: + if tensor not in tensor_to_backing: + tensor_to_backing[tensor] = node.component + if tensor in job.fusable_tensors: + n_shared_loops = n_loops + + private_level = n_shared_loops + latency, latency_per_level = component_latency( reuse, job.flattened_arch, pmapping, - spec, per_level_components=oset(job.components_track_latency), + n_shared_loops=n_shared_loops, ) overall_latency = max_nonzero(*latency.values()) @@ -110,7 +124,7 @@ def run_model( ) scaled_usage = usage * s.usage_scale spatial_usage[node.name, s.name] = scaled_usage - s = f"usagespatial{node.name}{s.name}" + s = reservation2col(f"{node.name} {s.name}", private_level) spatial_usage_df[s] = scaled_usage component_to_non_power_gated_porp, _ = spec.arch._power_gating( @@ -126,12 +140,6 @@ def run_model( spec, actions, overall_latency, component_to_non_power_gated_porp ) - tensor_to_backing = {} - for node in pmapping.nodes: - if isinstance(node, TensorHolder): - for tensor in node.tensors: - tensor_to_backing.setdefault(tensor, node.component) - # A Toll is a pass-through and must never be the outermost level backing a tensor — # that would leave no real Memory holding it. for node in pmapping.nodes: @@ -190,10 +198,10 @@ def run_model( continue size = memory_to_size[memory] running_total = 0 - for n_loop in sorted(n_loop_options): - if n_loop in occupancies: - running_total += occupancies[n_loop] - df[reservation2col(memory, n_loop)] = running_total / size + for n_loop, occupancy in sorted(occupancies.items()): + running_total += occupancy + col = reservation2col(memory, int(min(n_loop, private_level))) + df[col] = running_total / size if isinstance(running_total, Number) and running_total > size: raise InvalidMappingError( f"The mapping uses {running_total} bits of {memory} but its size is " @@ -246,6 +254,7 @@ def run_model( tensor_to_backing, workload.einsums[job.einsum_name].output_tensor_names, ) + per_component_total = [] for component in oset(latency) | oset(comm_latency): l = comm_latency.get(component, 0) @@ -255,13 +264,6 @@ def run_model( per_component_total.append(l) df["Totallatency"] = max_nonzero(*per_component_total) * n_instances - # For first latency, we'll follow the convention of treating compute - # as a component, similarly to memory (see below). - for compute_level, stats in reuse.compute_stats.items(): # FIRST LATENCY - for idx, max_first_latency in stats.max_first_latency.items(): - df[firstlatency2col(compute_level.level, idx)] = ( - max_first_latency * n_instances - ) # ================================================================================= # Energy @@ -280,7 +282,7 @@ def run_model( per_memory_spatial_usage_df = {} for memory, occupancies in total_occupancy.items(): ignored = memory in job.ignored_resources - key = f"usagememory{memory}" + key = reservation2col(memory, private_level) if not ignored: per_memory_spatial_usage_df[key] = ( sum(occupancies.values()) / memory_to_size[memory] diff --git a/docs/source/guide/modeling/accelerator_energy_latency.rst b/docs/source/guide/modeling/accelerator_energy_latency.rst index f7cdb768..0e2961d0 100755 --- a/docs/source/guide/modeling/accelerator_energy_latency.rst +++ b/docs/source/guide/modeling/accelerator_energy_latency.rst @@ -43,14 +43,15 @@ is set to a different value. Calculating Latency from a Pmapping ----------------------------------- -The total latency of a component, defined in the class's -:py:obj:`~accelforge.frontend.arch.Component.total_latency` field, is a Python -expression that is evaluated using the component's actions. - -The :py:obj:`~accelforge.frontend.arch.Component.total_latency` field is: - -.. include-docstring:: accelforge.frontend.arch.Component.total_latency - :decapitalize: +The latency of a component is the sum, over its actions, of the action count (times the +component's :py:attr:`~accelforge.frontend.arch.Component.actions_scale`) divided by the +action's :py:attr:`~accelforge.frontend.arch.Action.throughput`. Network components have +their latency limited by their most-congested link. + +A :py:class:`~accelforge.frontend.arch.Memory` with +:py:attr:`~accelforge.frontend.arch.Memory.separate_read_write_ports` set is treated as +two components for latency, "{name} (read)" and "{name} (write)", which may overlap in +time; otherwise one shared port serializes reads and writes. Calculating Area and Leak Power diff --git a/examples/arches/eyeriss.yaml b/examples/arches/eyeriss.yaml index acd57210..b7d96eeb 100644 --- a/examples/arches/eyeriss.yaml +++ b/examples/arches/eyeriss.yaml @@ -15,14 +15,12 @@ arch: component_class: SmartBufferSRAM size: 1024 * 1024 # 1Mb # 32 reads and writes per cycle, 200MHz. Note that the bits per read/write is the - # bits per action set below. This is optional, and if it is omitted, then the total - # latency sum(a.n_calls / a.throughput for a in actions) will be used instead, with - # the throughput gotten from component models. - total_latency: sum(a.n_calls for a in actions) / 32 / 200e6 + # bits per action set below. Setting per-action throughput is optional, and if + # it is omitted, the throughput is gotten from component models. extra_attributes_for_component_model: {n_banks: 32} actions: - - {name: read, bits_per_action: 64} - - {name: write, bits_per_action: 64} + - {name: read, bits_per_action: 64, throughput: 32 * 200e6} + - {name: write, bits_per_action: 64, throughput: 32 * 200e6} tensors: {keep: ~MainMemory, may_keep: All} - !Container @@ -52,5 +50,6 @@ arch: - !Compute # MAC unit name: MAC component_class: IntMAC - total_latency: sum(a.n_calls for a in actions) / 200e6 + actions: + - {name: compute, throughput: 200e6} extra_attributes_for_component_model: {multiplier_width: 8, adder_width: 16} diff --git a/examples/arches/nvdla.yaml b/examples/arches/nvdla.yaml index 88290464..eb96c0f6 100755 --- a/examples/arches/nvdla.yaml +++ b/examples/arches/nvdla.yaml @@ -16,7 +16,7 @@ arch: - !Memory name: GlobalBuffer size: 1024*64*8 # 64 kB - total_latency: max(a.n_calls / a.throughput for a in actions) # Separate ports + separate_read_write_ports: True leak_power: 0 actions: # 512 GB/s read, 128 GB/s write diff --git a/examples/arches/tpu_v4i.yaml b/examples/arches/tpu_v4i.yaml index c95c83c6..ed240fb7 100755 --- a/examples/arches/tpu_v4i.yaml +++ b/examples/arches/tpu_v4i.yaml @@ -23,7 +23,7 @@ arch: - !Memory name: GlobalBuffer size: 1024*1024*128*8 # 128MB - total_latency: max(a.n_calls / a.throughput for a in actions) # Separate ports + separate_read_write_ports: True leak_power: 0 area: 112e-6 # From paper fig. 6 actions: diff --git a/tests/input_files/fused_matmuls_weights_inside.mapping.yaml b/tests/input_files/fused_matmuls_weights_inside.mapping.yaml new file mode 100644 index 00000000..0ccf7a85 --- /dev/null +++ b/tests/input_files/fused_matmuls_weights_inside.mapping.yaml @@ -0,0 +1,34 @@ +mapping: + nodes: + - !Storage + tensors: [T0, T{{N_EINSUMS}}] + component: MainMemory + {% for i in range(N_EINSUMS) %} + - !Storage + tensors: [W{{i}}] + component: MainMemory + {% endfor %} + - !Temporal + rank_variable: m + tile_shape: 1 + - !Sequential + nodes: + {% for i in range(N_EINSUMS) %} + - !Nested + nodes: + - !Storage + tensors: [W{{i}}] + component: GlobalBuffer + - !Storage + tensors: [T{{i}}, T{{i+1}}] + component: GlobalBuffer + - !Temporal + rank_variable: n{{i}} + tile_shape: 1 + - !Temporal + rank_variable: n{{i+1}} + tile_shape: 1 + - !Compute + einsum: Matmul{{i}} + component: MAC + {% endfor %} diff --git a/tests/input_files/latency.arch.yaml b/tests/input_files/latency.arch.yaml index c81af779..1684a0ef 100644 --- a/tests/input_files/latency.arch.yaml +++ b/tests/input_files/latency.arch.yaml @@ -5,6 +5,7 @@ arch: size: inf leak_power: 0 area: 0 + separate_read_write_ports: {{ MM_SEPARATE_PORTS | default(False) }} tensors: {keep: All} actions: - name: read diff --git a/tests/network/test_network.py b/tests/network/test_network.py index ed3b7829..69518159 100644 --- a/tests/network/test_network.py +++ b/tests/network/test_network.py @@ -111,10 +111,9 @@ def test_hierarchical_1d(self): * KN * BITS_PER_VALUE, ) - # PeArray is bandwidth-bound at 1 bit/s: its most congested link carries - # T0 (3 * 64) + W0 (3 * 128) + T1 (256) = 832 bits. Wind-up/down adds the - # 2 MacArray hops each way (PeArray hops have zero latency). - self.assertEqual(result.data["Totallatency"].iloc[0], 832 + 4) + # PeArray is bandwidth-bound at 1 bit/s: its most congested link carries T0 (3 * + # 64) + W0 (3 * 128) + T1 (256) = 832 bits. + self.assertEqual(result.data["Totallatency"].iloc[0], 832) def test_hierarchical(self): M = 8 diff --git a/tests/vibe_see_readme_in_this_dir/test_api_gaps.py b/tests/vibe_see_readme_in_this_dir/test_api_gaps.py index 4f5564fe..ccaa2758 100644 --- a/tests/vibe_see_readme_in_this_dir/test_api_gaps.py +++ b/tests/vibe_see_readme_in_this_dir/test_api_gaps.py @@ -582,16 +582,26 @@ def test_component_throughput_scale_custom(self): ) self.assertEqual(c.throughput_scale, 0.5) - def test_component_total_latency_default(self): + def test_component_total_latency_removed(self): c = Compute( name="MAC", leak_power=0, area=0, actions=[{"name": "compute", "energy": 1, "throughput": 1, "latency": 0}], ) - # Default total_latency is an expression - self.assertIsInstance(c.total_latency, str) - self.assertIn("sum", c.total_latency) + # total_latency was removed: latency is the inverse-throughput-weighted + # sum of action counts, and setting the old attribute errors. + self.assertIsNone(c.total_latency) + with self.assertRaisesRegex(Exception, "total_latency was removed"): + Compute( + name="MAC", + leak_power=0, + area=0, + actions=[ + {"name": "compute", "energy": 1, "throughput": 1, "latency": 0} + ], + total_latency="sum(a.n_calls / a.throughput for a in actions)", + ) # ============================================================================ diff --git a/tests/vibe_see_readme_in_this_dir/test_component_fields.py b/tests/vibe_see_readme_in_this_dir/test_component_fields.py index 299cbcd3..31ef153d 100644 --- a/tests/vibe_see_readme_in_this_dir/test_component_fields.py +++ b/tests/vibe_see_readme_in_this_dir/test_component_fields.py @@ -6,7 +6,7 @@ - Component.n_parallel_instances - Component.total_area, total_leak_power - Action.energy_scale, Action.bits_per_action - - Component.total_latency expression + - Memory.separate_read_write_ports - TensorHolder.tensors.keep / may_keep - Memory.size and related fields """ @@ -191,25 +191,21 @@ def test_memory_size_inf(self): ) self.assertEqual(m.size, "inf") - def test_memory_total_latency_default(self): - m = Memory( - name="Buf", - size=1024, - leak_power=0, - area=0, - actions=[ - {"name": "read", "energy": 1, "throughput": float("inf"), "latency": 0}, - { - "name": "write", - "energy": 1, - "throughput": float("inf"), - "latency": 0, - }, - ], - ) - self.assertIsNotNone(m.total_latency) - - def test_memory_total_latency_expression(self): + def test_memory_total_latency_removed(self): + with self.assertRaisesRegex(Exception, "total_latency was removed"): + Memory( + name="Buf", + size=1024, + leak_power=0, + area=0, + actions=[ + {"name": "read", "energy": 1, "throughput": 1 / 5, "latency": 0}, + {"name": "write", "energy": 1, "throughput": 1 / 3, "latency": 0}, + ], + total_latency="max(read_latency, write_latency)", + ) + + def test_memory_separate_read_write_ports(self): m = Memory( name="Buf", size=1024, @@ -219,9 +215,9 @@ def test_memory_total_latency_expression(self): {"name": "read", "energy": 1, "throughput": 1 / 5, "latency": 0}, {"name": "write", "energy": 1, "throughput": 1 / 3, "latency": 0}, ], - total_latency="max(read_latency, write_latency)", + separate_read_write_ports=True, ) - self.assertEqual(m.total_latency, "max(read_latency, write_latency)") + self.assertTrue(m.separate_read_write_ports) def test_memory_energy_scale_custom(self): m = Memory( @@ -331,10 +327,9 @@ def test_global_buffer_has_read_write_actions(self): self.assertIn("read", action_names) self.assertIn("write", action_names) - def test_global_buffer_total_latency_is_expression(self): + def test_global_buffer_separate_ports(self): gb = self.spec.arch.find("GlobalBuffer") - self.assertIsInstance(gb.total_latency, str) - self.assertIn("max", gb.total_latency) + self.assertTrue(gb.separate_read_write_ports) def test_array_fanout_spatial(self): """ProcessingElement has spatial fanouts for reuse.""" @@ -378,11 +373,9 @@ def test_main_memory_size_evaluates_to_inf(self): mm = self.spec.arch.find("MainMemory") self.assertEqual(mm.size, math.inf) - def test_global_buffer_total_latency_stays_expression(self): - """total_latency is NOT evaluated by _spec_eval_expressions; it stays as a string.""" + def test_global_buffer_separate_ports_evaluated(self): gb = self.spec.arch.find("GlobalBuffer") - self.assertIsInstance(gb.total_latency, str) - self.assertIn("max", gb.total_latency) + self.assertTrue(gb.separate_read_write_ports) def test_global_buffer_size_evaluated(self): gb = self.spec.arch.find("GlobalBuffer") diff --git a/tests/vibe_see_readme_in_this_dir/test_yaml_and_expressions.py b/tests/vibe_see_readme_in_this_dir/test_yaml_and_expressions.py index 0ba415ea..227f2a1c 100644 --- a/tests/vibe_see_readme_in_this_dir/test_yaml_and_expressions.py +++ b/tests/vibe_see_readme_in_this_dir/test_yaml_and_expressions.py @@ -638,19 +638,28 @@ def test_toll_name(self): class TestMemoryTotalLatency(unittest.TestCase): - """Test that total_latency expression is stored on Memory.""" + """total_latency was removed: setting it errors, and separate read/write + ports are expressed with separate_read_write_ports.""" + + def test_total_latency_raises(self): + with self.assertRaisesRegex(Exception, "total_latency was removed"): + Memory( + name="GB", + size=1024, + leak_power=0, + area=0, + total_latency="max(a.n_calls / a.throughput for a in actions)", + ) - def test_total_latency_expression(self): - """The TPU GlobalBuffer uses a throughput-based total_latency expression.""" + def test_separate_read_write_ports(self): + """The TPU GlobalBuffer models its separate ports with the attribute.""" arch_path = EXAMPLES_DIR / "arches" / "tpu_v4i.yaml" wl_path = EXAMPLES_DIR / "workloads" / "basic" / "three_matmuls_annotated.yaml" if not arch_path.exists() or not wl_path.exists(): self.skipTest("YAML not found") spec = Spec.from_yaml(arch_path, wl_path) gb = spec.arch.find("GlobalBuffer") - self.assertEqual( - gb.total_latency, "max(a.n_calls / a.throughput for a in actions)" - ) + self.assertTrue(gb.separate_read_write_ports) class TestEnabledField(unittest.TestCase): From 01f87eb676b0b7404fd6e009f5d5800548daadf0 Mon Sep 17 00:00:00 2001 From: Tanner Andrulis Date: Mon, 3 Aug 2026 12:48:23 -0400 Subject: [PATCH 5/8] Add latency plotting --- accelforge/plotting/__init__.py | 2 + accelforge/plotting/latency.py | 288 +++++++++++ tests/hwcomponents_expected.json | 17 +- tests/regression_reference.json | 848 ++++++++++++++++--------------- tests/test_latency.py | 182 ++++++- 5 files changed, 925 insertions(+), 412 deletions(-) create mode 100644 accelforge/plotting/latency.py diff --git a/accelforge/plotting/__init__.py b/accelforge/plotting/__init__.py index 14d405db..c7223042 100644 --- a/accelforge/plotting/__init__.py +++ b/accelforge/plotting/__init__.py @@ -1,7 +1,9 @@ +from . import latency from . import mappings from . import specs __all__ = [ + "latency", "mappings", "specs", "roofline", diff --git a/accelforge/plotting/latency.py b/accelforge/plotting/latency.py new file mode 100644 index 00000000..cc1a9b15 --- /dev/null +++ b/accelforge/plotting/latency.py @@ -0,0 +1,288 @@ +"""Latency timelines showing how per-component latency overlaps across Einsums.""" + +import math +from bisect import insort +from collections import defaultdict, namedtuple +from copy import deepcopy + +import matplotlib.pyplot as plt + +from accelforge.mapper.FFM import Mappings +from accelforge.mapper.FFM._pareto_df.df_convention import col2complatency +from accelforge.util import oset + +_Block = namedtuple("_Block", ["einsum", "start", "end"]) +_Bar = namedtuple("_Bar", ["component", "level", "einsum", "start", "end", "shared"]) + + +def _latency_timeline( + spec, mappings: Mappings = None +) -> tuple[list[_Block], list[_Bar], float]: + """ + Lay the latencies of a mapping's Einsums out on a shared timeline. X axis is time, + and Y axis is component and loop level. + + Uses spec.mapping, or the single mapping in a `Mappings` object returned by the + mapper if `mappings` is given. + + Returns (blocks, bars, total). Bars with a None component are the per-Einsum latency + that is not tracked per-component (compute, networks, and communication). The total + equals the model's Totallatency. + """ + from accelforge.mapper.FFM._join_pmappings.join_pmappings import ( + clean_compress_and_join_pmappings, + ) + from accelforge.model.main import _model_pmappings + + flattened_arches = evaluated_specs = None + if mappings is not None: + if len(mappings.data) != 1: + raise ValueError( + f"mappings holds {len(mappings.data)} mappings; pass exactly one" + ) + spec = deepcopy(spec) + spec.model.metrics = spec.mapper.info_metrics + spec.mapping = mappings.data.iloc[0]["Totalmapping"](_for_model=True) + flattened_arches = mappings.flattened_arches + evaluated_specs = mappings.evaluated_specs + + pmappings = _model_pmappings(spec, flattened_arches, evaluated_specs) + einsums = list(pmappings.einsum2pmappings) + groups = [pmappings.einsum2pmappings[e][0] for e in einsums] + n = len(groups) + tensors = [g.compatibility.tensor_names for g in groups] + + # Per-Einsum latency outside the per-level columns, and per-(component, level) + # latency. Columns are upward-inclusive, so the latency at a level alone is the + # column minus the next-deeper column. + privates = [] + inclusives = [] + excls = [] + for group in groups: + row = group.mappings.data.iloc[0] + privates.append(row["Totallatency"]) + inclusive = defaultdict(dict) + for col in group.mappings.data.columns: + if (key := col2complatency(col)) is not None: + inclusive[key.name][key.nloops] = row[col] + inclusives.append(inclusive) + excl = {} + for component, by_level in inclusive.items(): + levels = sorted(by_level) + excl[component] = { + l: by_level[l] - by_level[deeper] + for l, deeper in zip(levels, levels[1:]) + } + excl[component][levels[-1]] = by_level[levels[-1]] + excls.append(excl) + # Outermost-first architecture order, so lanes read top-down like the arch. + arch_order = list( + dict.fromkeys( + m.name for e in einsums for m in pmappings.einsum2jobs[e].flattened_arch + ) + ) + # Port-split names like "GlobalBuffer (read)" sort by their base component. + components = sorted( + oset(c for e in excls for c in e), + key=lambda c: (arch_order.index(c.split(" (")[0]), c), + ) + + # keep[i]: deepest loop index at which Einsum i's reservations are co-resident + # with another Einsum's. Deeper latency is private to block i. settle[i]: + # deepest loop index still co-resident after Einsum i. Deeper latency must + # finish by the end of block i; shallower latency may spill into later blocks. + keeps, settles = [], [] + backings = {} + for i, group in enumerate(groups): + live = oset.union(oset(), *tensors[i + 1 :]) + past = oset.union(oset(), *tensors[:i]) + keeps.append(group.compatibility.shared_loop_index(live | (past & tensors[i]))) + for t, l in group.compatibility.get_backing_levels().items(): + backings[t] = min(backings.get(t, l), l) + settles.append( + max((backings[t] for t in live if t in backings), default=-1) - 1 + ) + settles[-1] = -2 + + # Exact block ends, mirroring the join: each Einsum folds levels deeper than + # keep into its own width by max; merged pool columns sum across Einsums, a + # missing level falling back to the next-deeper column; levels deeper than + # settle fold into the running total by max. + def at_or_below(cols, level): + deeper = [l for l in cols if l >= level] + return cols[min(deeper)] if deeper else 0 + + total = 0.0 + ends = [] + pool = defaultdict(dict) # component -> {level: latency summed across Einsums} + for i in range(n): + width = privates[i] + for cols in inclusives[i].values(): + done = [l for l in cols if l > keeps[i]] + if done: + width = max(width, cols[min(done)]) + total += width + for component, cols in inclusives[i].items(): + kept = {l: v for l, v in cols.items() if l <= keeps[i]} + mine = pool[component] + pool[component] = { + l: at_or_below(mine, l) + at_or_below(kept, l) + for l in set(mine) | set(kept) + } + for cols in pool.values(): + done = [l for l in cols if l > settles[i]] + if done: + total = max(total, cols[min(done)]) + for l in done: + del cols[l] + ends.append(total) + + def floor(level, i): + return max((ends[j] for j in range(i) if settles[j] < level), default=0.0) + + def deadline(level, i): + return next((ends[j] for j in range(i, n) if settles[j] < level), ends[-1]) + + busy = defaultdict(list) # component -> sorted (start, end) of placed bars + + def place(component, lo, amount, due): + # Earliest gap in the component's schedule that fits, from lo onward. + start = lo + for s, e in busy[component]: + if s >= start + amount: + break + start = max(start, e) + if start + amount > due: + # The model lets busy time fill slack anywhere before the deadline, + # even when this component was already busy then. + start = due - amount + insort(busy[component], (start, start + amount)) + return start + + blocks, bars = [], [] + for i, einsum in enumerate(einsums): + block_start = ends[i - 1] if i else 0.0 + blocks.append(_Block(einsum, block_start, ends[i])) + if privates[i] > 0: + bars.append( + _Bar(None, None, einsum, block_start, block_start + privates[i], False) + ) + + # Latency deeper than keep is private to its Einsum's block; co-resident + # latency may fill slack anywhere in its co-residency window (down to the + # window's start, until it folds into the total at its deadline block). + # Private latency has a fixed home, so place it first and let co-resident + # latency fill the gaps that remain. + for shared in [False, True]: + for i, einsum in enumerate(einsums): + for component in components: + for level in sorted(excls[i].get(component, {}), reverse=True): + amount = excls[i][component][level] + if (level <= keeps[i]) != shared or amount <= 0: + continue + if shared: + start = place( + component, floor(level, i), amount, deadline(level, i) + ) + else: + start = place(component, blocks[i].start, amount, ends[i]) + bars.append( + _Bar(component, level, einsum, start, start + amount, shared) + ) + + result = clean_compress_and_join_pmappings( + pmappings=pmappings, + metrics=spec.model.metrics, + print_progress=False, + for_model=True, + ) + expected = result.data["Totallatency"].iloc[0] + assert math.isclose(total, expected, rel_tol=1e-6), ( + f"Latency timeline total {total} does not match the model's " + f"Totallatency {expected}" + ) + return blocks, bars, total + + +def plot_latency( + spec, mappings: Mappings = None, ax: plt.Axes = None +) -> tuple[plt.Figure, plt.Axes]: + """ + Plots a latency timeline. Time is on the X axis, and component on the Y axis. + Component latencies may be divided across multiple shared loop levels, and shared + latencies may overlap across Einsums. Private latencies do not get a loop level. + + Parameters + ---------- + spec: + The spec to plot. + mappings: + The mapping to plot. Must hold exactly one mapping. If not given, uses + spec.mapping. + ax: + The axes to plot on. If not given, creates a new figure and axes. + """ + blocks, bars, total = _latency_timeline(spec, mappings) + + # Components from top to bottom with each component's shared levels deepest + # at the bottom, so bars within a block stair upward toward shared levels. + # Private latency is lumped into one lane per component below its shared + # levels; the Other lane is at the very bottom. + def lane(bar): + return (bar.component, bar.level if bar.shared else "private") + + lanes = defaultdict(oset) + for bar in bars: + if bar.component is not None: + lanes[bar.component].add(lane(bar)[1]) + lane_order = [ + (c, l) + for c in lanes + for l in sorted(lanes[c], key=lambda l: (l == "private", l != "private" and l)) + ] + if any(bar.component is None for bar in bars): + lane_order.append((None, "private")) + y = {lane: len(lane_order) - 1 - i for i, lane in enumerate(lane_order)} + + colors = {} + cycle = plt.rcParams["axes.prop_cycle"].by_key()["color"] + for component, _ in lane_order: + colors.setdefault(component, cycle[len(colors) % len(cycle)]) + + if ax is None: + fig, ax = plt.subplots( + figsize=(max(6, 2 * len(blocks)), 1.5 + 0.4 * len(lane_order)) + ) + else: + fig = ax.get_figure() + + for bar in bars: + ax.barh( + y[lane(bar)], + bar.end - bar.start, + left=bar.start, + height=0.8, + color=colors[bar.component], + edgecolor="black", + linewidth=0.5, + ) + for block in blocks: + ax.axvline(block.end, linestyle=":", color="gray", linewidth=1) + # Skip labels of blocks too narrow to hold them. + if block.end - block.start >= 0.01 * len(block.einsum) * total: + ax.text((block.start + block.end) / 2, -1, block.einsum, ha="center") + + def label(component, level): + if component is None: + return "Other" + if len(lanes[component]) == 1: + return component + if level == "private": + return f"{component} (private)" + return f"{component} (above loop {level})" + + ax.set_yticks([y[l] for l in lane_order], [label(*l) for l in lane_order]) + ax.set_ylim(-1.5, len(lane_order) - 0.5) + ax.set_xlim(0, total * 1.02) + ax.set_xlabel("Time") + return fig, ax diff --git a/tests/hwcomponents_expected.json b/tests/hwcomponents_expected.json index 7b0016df..e563c7f7 100644 --- a/tests/hwcomponents_expected.json +++ b/tests/hwcomponents_expected.json @@ -6,10 +6,12 @@ "actions": { "read": { "energy": 8e-12, + "latency": 1.9073486328125e-08, "throughput": 3355443200.0 }, "write": { "energy": 8e-12, + "latency": 1.9073486328125e-08, "throughput": 3355443200.0 } } @@ -20,11 +22,13 @@ "actions": { "read": { "energy": 9.056430298994264e-11, - "throughput": 9846153846.153845 + "latency": 6.4376400000000005e-09, + "throughput": 6400000000.0 }, "write": { "energy": 9.122346940403464e-11, - "throughput": 9846153846.153845 + "latency": 8.06264e-09, + "throughput": 6400000000.0 } } }, @@ -34,10 +38,12 @@ "actions": { "read": { "energy": 2.2733893884485344e-14, + "latency": 2.6455e-09, "throughput": 9846153846.153845 }, "write": { "energy": 3.1554801000964905e-14, + "latency": 4.2705e-09, "throughput": 9846153846.153845 } } @@ -48,10 +54,12 @@ "actions": { "read": { "energy": 4.0152965248454046e-14, + "latency": 2.69905e-09, "throughput": 34461538461.53846 }, "write": { "energy": 6.235570339922515e-14, + "latency": 4.324050000000001e-09, "throughput": 34461538461.53846 } } @@ -62,10 +70,12 @@ "actions": { "read": { "energy": 2.6247402328781873e-14, + "latency": 2.6455e-09, "throughput": 9846153846.153845 }, "write": { "energy": 3.728731056542402e-14, + "latency": 4.2705e-09, "throughput": 9846153846.153845 } } @@ -76,7 +86,8 @@ "actions": { "compute": { "energy": 1.2813796807700769e-12, - "throughput": 615384615.3846153 + "latency": 3.25e-09, + "throughput": 200000000.0 } } } diff --git a/tests/regression_reference.json b/tests/regression_reference.json index 232167a8..a631f76a 100644 --- a/tests/regression_reference.json +++ b/tests/regression_reference.json @@ -1967,7 +1967,7 @@ "n_mappings": 1.0 }, "eyeriss|gpt3_6.7B|BATCH_SIZE=1,DECODE=True,N_CACHED_TOKENS=2047,N_NEW_TOKENS=1|fused": { - "energy": 0.031553364671354084, + "energy": 0.031553459477181967, "latency": 1.0545854568481445, "energy_per_component": { "('I', 'GlobalBuffer', 'read')": 9.273784626170126e-08, @@ -2059,8 +2059,8 @@ "('QK_softmax', 'MAC', 'leak')": 6.088245640967216e-08, "('AV', 'OutputScratchpad', 'read')": 7.040573109406978e-06, "('AV', 'OutputScratchpad', 'write')": 1.0001905138778966e-05, - "('AV', 'GlobalBuffer', 'read')": 0.00019122471788501634, "('AV', 'GlobalBuffer', 'write')": 0.00019121606601402164, + "('AV', 'GlobalBuffer', 'read')": 0.00019122471788501634, "('AV', 'InputScratchpad', 'read')": 3.049801762244897e-06, "('AV', 'InputScratchpad', 'write')": 3.30714513552266e-08, "('AV', 'WeightScratchpad', 'read')": 5.386608303379935e-06, @@ -2091,13 +2091,13 @@ "('Z', 'MAC', 'leak')": 0.00011896914656972513, "('FFA', 'OutputScratchpad', 'read')": 3.165075395372696e-05, "('FFA', 'OutputScratchpad', 'write')": 4.496336623560637e-05, - "('FFA', 'GlobalBuffer', 'read')": 2.346267532657261e-05, + "('FFA', 'GlobalBuffer', 'read')": 2.3555412553832866e-05, "('FFA', 'GlobalBuffer', 'write')": 2.3913684344734065e-05, "('FFA', 'WeightScratchpad', 'read')": 4.311391814488366e-05, "('FFA', 'WeightScratchpad', 'write')": 6.695392670468701e-05, "('FFA', 'MainMemory', 'read')": 0.008589934592, "('FFA', 'InputScratchpad', 'read')": 2.4410332116531208e-05, - "('FFA', 'InputScratchpad', 'write')": 2.0679753465202566e-09, + "('FFA', 'InputScratchpad', 'write')": 4.135950693040513e-09, "('FFA', 'MAC', 'compute')": 8.59919347291625e-05, "('FFA', 'MainMemory', 'leak')": 0.0, "('FFA', 'GlobalBuffer', 'leak')": 8.162572339642793e-05, @@ -2168,8 +2168,8 @@ "('Z', 'OutputScratchpad')": 0.007654400076717138, "('Z', 'MAC')": 0.010485759936273098, "('FFA', 'MainMemory')": 0.3199999928474426, - "('FFA', 'GlobalBuffer')": 8.143999730236828e-05, - "('FFA', 'InputScratchpad')": 0.013632319867610931, + "('FFA', 'GlobalBuffer')": 8.160000288626179e-05, + "('FFA', 'InputScratchpad')": 0.013633151538670063, "('FFA', 'WeightScratchpad')": 0.0077894218266010284, "('FFA', 'OutputScratchpad')": 0.030617600306868553, "('FFA', 'MAC')": 0.04194303974509239, @@ -2183,8 +2183,8 @@ "actions": { "('I', 'GlobalBuffer', 'I_in', 'read')": 1024.0, "('I', 'GlobalBuffer', 'I_in', 'write')": 0.0, - "('I', 'MainMemory', 'I_in', 'read')": 0.0, "('I', 'MainMemory', 'I_in', 'write')": 65536.0, + "('I', 'MainMemory', 'I_in', 'read')": 0.0, "('I', 'MAC', 'None', 'compute')": 0.0, "('V', 'InputScratchpad', 'I', 'read')": 268435456.0, "('V', 'InputScratchpad', 'I', 'write')": 65536.0, @@ -2192,14 +2192,14 @@ "('V', 'GlobalBuffer', 'I', 'write')": 0.0, "('V', 'OutputScratchpad', 'V', 'read')": 301465600.0, "('V', 'OutputScratchpad', 'V', 'write')": 301465600.0, - "('V', 'GlobalBuffer', 'V', 'read')": 65536.0, "('V', 'GlobalBuffer', 'V', 'write')": 65536.0, - "('V', 'MainMemory', 'V', 'read')": 0.0, + "('V', 'GlobalBuffer', 'V', 'read')": 65536.0, "('V', 'MainMemory', 'V', 'write')": 65536.0, + "('V', 'MainMemory', 'V', 'read')": 0.0, "('V', 'WeightScratchpad', 'WV', 'read')": 268435456.0, "('V', 'WeightScratchpad', 'WV', 'write')": 268435456.0, - "('V', 'MainMemory', 'WV', 'read')": 268435456.0, "('V', 'MainMemory', 'WV', 'write')": 0.0, + "('V', 'MainMemory', 'WV', 'read')": 268435456.0, "('V', 'MAC', 'None', 'compute')": 16777216.0, "('K', 'InputScratchpad', 'I', 'read')": 268435456.0, "('K', 'InputScratchpad', 'I', 'write')": 65536.0, @@ -2207,14 +2207,14 @@ "('K', 'GlobalBuffer', 'I', 'write')": 0.0, "('K', 'OutputScratchpad', 'K', 'read')": 301465600.0, "('K', 'OutputScratchpad', 'K', 'write')": 301465600.0, - "('K', 'GlobalBuffer', 'K', 'read')": 65536.0, "('K', 'GlobalBuffer', 'K', 'write')": 65536.0, - "('K', 'MainMemory', 'K', 'read')": 0.0, + "('K', 'GlobalBuffer', 'K', 'read')": 65536.0, "('K', 'MainMemory', 'K', 'write')": 65536.0, + "('K', 'MainMemory', 'K', 'read')": 0.0, "('K', 'WeightScratchpad', 'WK', 'read')": 268435456.0, "('K', 'WeightScratchpad', 'WK', 'write')": 268435456.0, - "('K', 'MainMemory', 'WK', 'read')": 268435456.0, "('K', 'MainMemory', 'WK', 'write')": 0.0, + "('K', 'MainMemory', 'WK', 'read')": 268435456.0, "('K', 'MAC', 'None', 'compute')": 16777216.0, "('Q', 'InputScratchpad', 'I', 'read')": 268435456.0, "('Q', 'InputScratchpad', 'I', 'write')": 65536.0, @@ -2222,21 +2222,21 @@ "('Q', 'GlobalBuffer', 'I', 'write')": 0.0, "('Q', 'OutputScratchpad', 'Q', 'read')": 301465600.0, "('Q', 'OutputScratchpad', 'Q', 'write')": 301465600.0, - "('Q', 'GlobalBuffer', 'Q', 'read')": 64512.0, "('Q', 'GlobalBuffer', 'Q', 'write')": 65536.0, + "('Q', 'GlobalBuffer', 'Q', 'read')": 64512.0, "('Q', 'WeightScratchpad', 'WQ', 'read')": 268435456.0, "('Q', 'WeightScratchpad', 'WQ', 'write')": 268435456.0, - "('Q', 'MainMemory', 'WQ', 'read')": 268435456.0, "('Q', 'MainMemory', 'WQ', 'write')": 0.0, + "('Q', 'MainMemory', 'WQ', 'read')": 268435456.0, "('Q', 'MAC', 'None', 'compute')": 16777216.0, "('QK', 'WeightScratchpad', 'K_cache', 'read')": 134152192.0, "('QK', 'WeightScratchpad', 'K_cache', 'write')": 134152192.0, - "('QK', 'MainMemory', 'K_cache', 'read')": 134152192.0, "('QK', 'MainMemory', 'K_cache', 'write')": 0.0, + "('QK', 'MainMemory', 'K_cache', 'read')": 134152192.0, "('QK', 'InputScratchpad', 'Q', 'read')": 134152192.0, "('QK', 'InputScratchpad', 'Q', 'write')": 65536.0, - "('QK', 'GlobalBuffer', 'Q', 'read')": 1024.0, "('QK', 'GlobalBuffer', 'Q', 'write')": 0.0, + "('QK', 'GlobalBuffer', 'Q', 'read')": 1024.0, "('QK', 'OutputScratchpad', 'QK', 'read')": 142536704.0, "('QK', 'OutputScratchpad', 'QK', 'write')": 142536704.0, "('QK', 'GlobalBuffer', 'QK', 'read')": 16376.0, @@ -2253,29 +2253,29 @@ "('QK_softmax', 'MAC', 'None', 'compute')": 65504.0, "('AV', 'OutputScratchpad', 'AV', 'read')": 268238848.0, "('AV', 'OutputScratchpad', 'AV', 'write')": 268238848.0, - "('AV', 'GlobalBuffer', 'AV', 'read')": 2095104.0, "('AV', 'GlobalBuffer', 'AV', 'write')": 2096128.0, + "('AV', 'GlobalBuffer', 'AV', 'read')": 2095104.0, "('AV', 'InputScratchpad', 'QK_softmax', 'read')": 134152192.0, "('AV', 'InputScratchpad', 'QK_softmax', 'write')": 1048064.0, "('AV', 'GlobalBuffer', 'QK_softmax', 'read')": 16376.0, "('AV', 'GlobalBuffer', 'QK_softmax', 'write')": 0.0, "('AV', 'WeightScratchpad', 'V_cache', 'read')": 134152192.0, "('AV', 'WeightScratchpad', 'V_cache', 'write')": 134152192.0, - "('AV', 'MainMemory', 'V_cache', 'read')": 134152192.0, "('AV', 'MainMemory', 'V_cache', 'write')": 0.0, + "('AV', 'MainMemory', 'V_cache', 'read')": 134152192.0, "('AV', 'MAC', 'None', 'compute')": 8384512.0, "('Z', 'InputScratchpad', 'AV', 'read')": 268435456.0, "('Z', 'InputScratchpad', 'AV', 'write')": 65536.0, - "('Z', 'GlobalBuffer', 'AV', 'read')": 1024.0, "('Z', 'GlobalBuffer', 'AV', 'write')": 0.0, + "('Z', 'GlobalBuffer', 'AV', 'read')": 1024.0, "('Z', 'WeightScratchpad', 'WZ', 'read')": 268435456.0, "('Z', 'WeightScratchpad', 'WZ', 'write')": 268435456.0, - "('Z', 'MainMemory', 'WZ', 'read')": 268435456.0, "('Z', 'MainMemory', 'WZ', 'write')": 0.0, + "('Z', 'MainMemory', 'WZ', 'read')": 268435456.0, "('Z', 'OutputScratchpad', 'Z', 'read')": 301465600.0, "('Z', 'OutputScratchpad', 'Z', 'write')": 301465600.0, - "('Z', 'GlobalBuffer', 'Z', 'read')": 64512.0, "('Z', 'GlobalBuffer', 'Z', 'write')": 65536.0, + "('Z', 'GlobalBuffer', 'Z', 'read')": 64512.0, "('Z', 'MAC', 'None', 'compute')": 16777216.0, "('FFA', 'OutputScratchpad', 'FFA', 'read')": 1205862400.0, "('FFA', 'OutputScratchpad', 'FFA', 'write')": 1205862400.0, @@ -2283,12 +2283,12 @@ "('FFA', 'GlobalBuffer', 'FFA', 'write')": 262144.0, "('FFA', 'WeightScratchpad', 'WFFA', 'read')": 1073741824.0, "('FFA', 'WeightScratchpad', 'WFFA', 'write')": 1073741824.0, - "('FFA', 'MainMemory', 'WFFA', 'read')": 1073741824.0, "('FFA', 'MainMemory', 'WFFA', 'write')": 0.0, + "('FFA', 'MainMemory', 'WFFA', 'read')": 1073741824.0, "('FFA', 'InputScratchpad', 'Z', 'read')": 1073741824.0, - "('FFA', 'InputScratchpad', 'Z', 'write')": 65536.0, - "('FFA', 'GlobalBuffer', 'Z', 'read')": 1024.0, + "('FFA', 'InputScratchpad', 'Z', 'write')": 131072.0, "('FFA', 'GlobalBuffer', 'Z', 'write')": 0.0, + "('FFA', 'GlobalBuffer', 'Z', 'read')": 2048.0, "('FFA', 'MAC', 'None', 'compute')": 67108864.0, "('FFB', 'InputScratchpad', 'FFA', 'read')": 1073741824.0, "('FFB', 'InputScratchpad', 'FFA', 'write')": 262144.0, @@ -2296,14 +2296,14 @@ "('FFB', 'GlobalBuffer', 'FFA', 'write')": 0.0, "('FFB', 'OutputScratchpad', 'FFB', 'read')": 1207435264.0, "('FFB', 'OutputScratchpad', 'FFB', 'write')": 1207435264.0, - "('FFB', 'GlobalBuffer', 'FFB', 'read')": 262144.0, "('FFB', 'GlobalBuffer', 'FFB', 'write')": 262144.0, - "('FFB', 'MainMemory', 'FFB', 'read')": 0.0, + "('FFB', 'GlobalBuffer', 'FFB', 'read')": 262144.0, "('FFB', 'MainMemory', 'FFB', 'write')": 65536.0, + "('FFB', 'MainMemory', 'FFB', 'read')": 0.0, "('FFB', 'WeightScratchpad', 'WFFB', 'read')": 1073741824.0, "('FFB', 'WeightScratchpad', 'WFFB', 'write')": 1073741824.0, - "('FFB', 'MainMemory', 'WFFB', 'read')": 1073741824.0, "('FFB', 'MainMemory', 'WFFB', 'write')": 0.0, + "('FFB', 'MainMemory', 'WFFB', 'read')": 1073741824.0, "('FFB', 'MAC', 'None', 'compute')": 67108864.0 }, "n_mappings": 1.0 @@ -2863,12 +2863,12 @@ }, "simba|three_matmuls_annotated||fused": { "energy": 7.272642342745611e-06, - "latency": 0.00019531248835846782, + "latency": 0.0001953125, "energy_per_component": { "('Matmul1', 'InputBuffer', 'read')": 2.022827239045455e-08, "('Matmul1', 'InputBuffer', 'write')": 2.7116731082799106e-09, - "('Matmul1', 'GlobalBuffer', 'read')": 1.0872260177895587e-08, "('Matmul1', 'GlobalBuffer', 'write')": 2.368311925741953e-08, + "('Matmul1', 'GlobalBuffer', 'read')": 1.0872260177895587e-08, "('Matmul1', 'MainMemory', 'read')": 2.097152e-06, "('Matmul1', 'AccumulationBuffer', 'read')": 4.451062807220296e-08, "('Matmul1', 'AccumulationBuffer', 'write')": 8.699198872363922e-08, @@ -2925,77 +2925,77 @@ "('Matmul3', 'MAC', 'leak')": 9.905858888714647e-08 }, "latency_per_component": { - "('Matmul1', 'MainMemory')": 7.812499825377017e-05, - "('Matmul1', 'GlobalBuffer')": 2.169467592239016e-07, "('Matmul1', 'InputBuffer')": 2.1760000379345001e-07, - "('Matmul1', 'WeightBuffer')": 1.2800000170898329e-08, - "('Matmul1', 'AccumulationBuffer')": 2.0036484329466475e-06, + "('Matmul1', 'GlobalBuffer')": 2.169467586206897e-07, + "('Matmul1', 'MainMemory')": 7.8125e-05, + "('Matmul1', 'AccumulationBuffer')": 2.0036484076433128e-06, "('Matmul1', 'Register')": 0.0, + "('Matmul1', 'WeightBuffer')": 1.2800000000000002e-08, "('Matmul1', 'MAC')": 8.192000109374931e-07, - "('Matmul2', 'MainMemory')": 3.9062499126885086e-05, - "('Matmul2', 'GlobalBuffer')": 1.446311728159344e-07, "('Matmul2', 'InputBuffer')": 2.1760000379345001e-07, - "('Matmul2', 'WeightBuffer')": 1.2800000170898329e-08, - "('Matmul2', 'AccumulationBuffer')": 2.0036484329466475e-06, + "('Matmul2', 'GlobalBuffer')": 1.4463117241379313e-07, + "('Matmul2', 'AccumulationBuffer')": 2.0036484076433128e-06, "('Matmul2', 'Register')": 0.0, + "('Matmul2', 'WeightBuffer')": 1.2800000000000002e-08, + "('Matmul2', 'MainMemory')": 3.90625e-05, "('Matmul2', 'MAC')": 8.192000109374931e-07, - "('Matmul3', 'MainMemory')": 7.812499825377017e-05, - "('Matmul3', 'GlobalBuffer')": 2.169467592239016e-07, "('Matmul3', 'InputBuffer')": 2.1760000379345001e-07, - "('Matmul3', 'WeightBuffer')": 1.2800000170898329e-08, - "('Matmul3', 'AccumulationBuffer')": 2.0036484329466475e-06, + "('Matmul3', 'GlobalBuffer')": 2.169467586206897e-07, + "('Matmul3', 'AccumulationBuffer')": 2.0036484076433128e-06, + "('Matmul3', 'MainMemory')": 7.8125e-05, "('Matmul3', 'Register')": 0.0, + "('Matmul3', 'WeightBuffer')": 1.2800000000000002e-08, "('Matmul3', 'MAC')": 8.192000109374931e-07 }, "actions": { "('Matmul1', 'InputBuffer', 'T0', 'read')": 2097152.0, "('Matmul1', 'InputBuffer', 'T0', 'write')": 131072.0, - "('Matmul1', 'GlobalBuffer', 'T0', 'read')": 131072.0, "('Matmul1', 'GlobalBuffer', 'T0', 'write')": 131072.0, - "('Matmul1', 'MainMemory', 'T0', 'read')": 131072.0, + "('Matmul1', 'GlobalBuffer', 'T0', 'read')": 131072.0, "('Matmul1', 'MainMemory', 'T0', 'write')": 0.0, + "('Matmul1', 'MainMemory', 'T0', 'read')": 131072.0, "('Matmul1', 'AccumulationBuffer', 'T1', 'read')": 6291456.0, "('Matmul1', 'AccumulationBuffer', 'T1', 'write')": 6291456.0, - "('Matmul1', 'GlobalBuffer', 'T1', 'read')": 0.0, "('Matmul1', 'GlobalBuffer', 'T1', 'write')": 131072.0, + "('Matmul1', 'GlobalBuffer', 'T1', 'read')": 0.0, "('Matmul1', 'Register', 'W0', 'read')": 16777216.0, "('Matmul1', 'Register', 'W0', 'write')": 131072.0, "('Matmul1', 'WeightBuffer', 'W0', 'read')": 131072.0, "('Matmul1', 'WeightBuffer', 'W0', 'write')": 131072.0, - "('Matmul1', 'MainMemory', 'W0', 'read')": 131072.0, "('Matmul1', 'MainMemory', 'W0', 'write')": 0.0, + "('Matmul1', 'MainMemory', 'W0', 'read')": 131072.0, "('Matmul1', 'MAC', 'None', 'compute')": 2097152.0, "('Matmul2', 'InputBuffer', 'T1', 'read')": 2097152.0, "('Matmul2', 'InputBuffer', 'T1', 'write')": 131072.0, - "('Matmul2', 'GlobalBuffer', 'T1', 'read')": 131072.0, "('Matmul2', 'GlobalBuffer', 'T1', 'write')": 0.0, + "('Matmul2', 'GlobalBuffer', 'T1', 'read')": 131072.0, "('Matmul2', 'AccumulationBuffer', 'T2', 'read')": 6291456.0, "('Matmul2', 'AccumulationBuffer', 'T2', 'write')": 6291456.0, - "('Matmul2', 'GlobalBuffer', 'T2', 'read')": 0.0, "('Matmul2', 'GlobalBuffer', 'T2', 'write')": 131072.0, + "('Matmul2', 'GlobalBuffer', 'T2', 'read')": 0.0, "('Matmul2', 'Register', 'W1', 'read')": 16777216.0, "('Matmul2', 'Register', 'W1', 'write')": 131072.0, "('Matmul2', 'WeightBuffer', 'W1', 'read')": 131072.0, "('Matmul2', 'WeightBuffer', 'W1', 'write')": 131072.0, - "('Matmul2', 'MainMemory', 'W1', 'read')": 131072.0, "('Matmul2', 'MainMemory', 'W1', 'write')": 0.0, + "('Matmul2', 'MainMemory', 'W1', 'read')": 131072.0, "('Matmul2', 'MAC', 'None', 'compute')": 2097152.0, "('Matmul3', 'InputBuffer', 'T2', 'read')": 2097152.0, "('Matmul3', 'InputBuffer', 'T2', 'write')": 131072.0, - "('Matmul3', 'GlobalBuffer', 'T2', 'read')": 131072.0, "('Matmul3', 'GlobalBuffer', 'T2', 'write')": 0.0, + "('Matmul3', 'GlobalBuffer', 'T2', 'read')": 131072.0, "('Matmul3', 'AccumulationBuffer', 'T3', 'read')": 6291456.0, "('Matmul3', 'AccumulationBuffer', 'T3', 'write')": 6291456.0, - "('Matmul3', 'GlobalBuffer', 'T3', 'read')": 131072.0, "('Matmul3', 'GlobalBuffer', 'T3', 'write')": 131072.0, - "('Matmul3', 'MainMemory', 'T3', 'read')": 0.0, + "('Matmul3', 'GlobalBuffer', 'T3', 'read')": 131072.0, "('Matmul3', 'MainMemory', 'T3', 'write')": 131072.0, + "('Matmul3', 'MainMemory', 'T3', 'read')": 0.0, "('Matmul3', 'Register', 'W2', 'read')": 16777216.0, "('Matmul3', 'Register', 'W2', 'write')": 131072.0, "('Matmul3', 'WeightBuffer', 'W2', 'read')": 131072.0, "('Matmul3', 'WeightBuffer', 'W2', 'write')": 131072.0, - "('Matmul3', 'MainMemory', 'W2', 'read')": 131072.0, "('Matmul3', 'MainMemory', 'W2', 'write')": 0.0, + "('Matmul3', 'MainMemory', 'W2', 'read')": 131072.0, "('Matmul3', 'MAC', 'None', 'compute')": 2097152.0 }, "n_mappings": 1.0 @@ -4000,8 +4000,8 @@ "n_mappings": 1.0 }, "simba|gpt3_6.7B|BATCH_SIZE=1,DECODE=True,N_CACHED_TOKENS=2047,N_NEW_TOKENS=1|fused": { - "energy": 0.030040219986931464, - "latency": 1.0400407314300537, + "energy": 0.03004027084359975, + "latency": 1.0400407001000076, "energy_per_component": { "('I', 'GlobalBuffer', 'read')": 5.436130088947793e-09, "('I', 'MainMemory', 'write')": 5.24288e-07, @@ -4106,8 +4106,8 @@ "('QK_softmax', 'MAC', 'leak')": 2.076394833849804e-09, "('AV', 'AccumulationBuffer', 'read')": 2.8778904379578307e-06, "('AV', 'AccumulationBuffer', 'write')": 5.624575805995846e-06, - "('AV', 'GlobalBuffer', 'read')": 2.065304701177589e-07, "('AV', 'GlobalBuffer', 'write')": 1.3617793115372478e-07, + "('AV', 'GlobalBuffer', 'read')": 2.065304701177589e-07, "('AV', 'InputBuffer', 'read')": 1.6174716677141987e-07, "('AV', 'InputBuffer', 'write')": 2.168279245040594e-08, "('AV', 'Register', 'read')": 0.0, @@ -4151,8 +4151,8 @@ "('FFA', 'WeightBuffer', 'write')": 7.379920310466056e-05, "('FFA', 'MainMemory', 'read')": 0.008589934592, "('FFA', 'InputBuffer', 'read')": 1.2946094329890911e-06, - "('FFA', 'InputBuffer', 'write')": 1.3558365541399553e-09, - "('FFA', 'GlobalBuffer', 'read')": 5.436130088947793e-09, + "('FFA', 'InputBuffer', 'write')": 2.7116731082799106e-09, + "('FFA', 'GlobalBuffer', 'read')": 1.087226042528755e-08, "('FFA', 'MAC', 'compute')": 1.2028869474000728e-05, "('FFA', 'MainMemory', 'leak')": 0.0, "('FFA', 'GlobalBuffer', 'leak')": 3.0427711408265168e-06, @@ -4162,11 +4162,11 @@ "('FFA', 'Register', 'leak')": 0.0, "('FFA', 'MAC', 'leak')": 0.00040574398008175194, "('FFB', 'InputBuffer', 'read')": 1.2946094329890911e-06, - "('FFB', 'InputBuffer', 'write')": 5.423346216559821e-09, - "('FFB', 'GlobalBuffer', 'read')": 2.7180650444738968e-08, - "('FFB', 'AccumulationBuffer', 'read')": 2.8486801966209896e-06, - "('FFB', 'AccumulationBuffer', 'write')": 5.56748727831291e-06, - "('FFB', 'GlobalBuffer', 'write')": 5.920779814354882e-09, + "('FFB', 'InputBuffer', 'write')": 2.1693384866239285e-08, + "('FFB', 'GlobalBuffer', 'read')": 3.261678078107872e-08, + "('FFB', 'AccumulationBuffer', 'read')": 2.8542440304590855e-06, + "('FFB', 'AccumulationBuffer', 'write')": 5.5783611969673075e-06, + "('FFB', 'GlobalBuffer', 'write')": 1.1841559732772566e-08, "('FFB', 'MainMemory', 'write')": 5.24288e-07, "('FFB', 'Register', 'read')": 0.0, "('FFB', 'Register', 'write')": 0.0, @@ -4183,135 +4183,135 @@ "('FFB', 'MAC', 'leak')": 0.0004057687474414706 }, "latency_per_component": { - "('I', 'MainMemory')": 1.9531249563442543e-05, - "('I', 'GlobalBuffer')": 3.61577932039836e-08, + "('I', 'GlobalBuffer')": 3.615779310344828e-08, + "('I', 'MainMemory')": 1.953125e-05, "('I', 'MAC')": 0.0, - "('V', 'MainMemory')": 0.08001953363418579, - "('V', 'GlobalBuffer')": 1.084733796119508e-07, "('V', 'InputBuffer')": 3.283199930592673e-06, - "('V', 'WeightBuffer')": 2.6214400349999778e-05, - "('V', 'AccumulationBuffer')": 3.205837492714636e-05, + "('V', 'GlobalBuffer')": 1.0847337931034484e-07, + "('V', 'AccumulationBuffer')": 3.2058374522293004e-05, + "('V', 'MainMemory')": 0.08001953125, "('V', 'Register')": 0.0, + "('V', 'WeightBuffer')": 2.6214400000000004e-05, "('V', 'MAC')": 6.5536000874999445e-06, - "('K', 'MainMemory')": 0.08001953363418579, - "('K', 'GlobalBuffer')": 1.084733796119508e-07, "('K', 'InputBuffer')": 3.283199930592673e-06, - "('K', 'WeightBuffer')": 2.6214400349999778e-05, - "('K', 'AccumulationBuffer')": 3.205837492714636e-05, + "('K', 'GlobalBuffer')": 1.0847337931034484e-07, + "('K', 'AccumulationBuffer')": 3.2058374522293004e-05, + "('K', 'MainMemory')": 0.08001953125, "('K', 'Register')": 0.0, + "('K', 'WeightBuffer')": 2.6214400000000004e-05, "('K', 'MAC')": 6.5536000874999445e-06, - "('Q', 'MainMemory')": 0.07999999821186066, - "('Q', 'GlobalBuffer')": 7.23155864079672e-08, "('Q', 'InputBuffer')": 3.283199930592673e-06, - "('Q', 'WeightBuffer')": 2.6214400349999778e-05, - "('Q', 'AccumulationBuffer')": 3.205837492714636e-05, + "('Q', 'GlobalBuffer')": 7.231558620689656e-08, + "('Q', 'AccumulationBuffer')": 3.2058374522293004e-05, "('Q', 'Register')": 0.0, + "('Q', 'WeightBuffer')": 2.6214400000000004e-05, + "('Q', 'MainMemory')": 0.08, "('Q', 'MAC')": 6.5536000874999445e-06, - "('QK', 'MainMemory')": 0.03998046740889549, - "('QK', 'GlobalBuffer')": 1.4098714018473402e-06, - "('QK', 'InputBuffer')": 1.3247999959276058e-05, - "('QK', 'WeightBuffer')": 1.3100800060783513e-05, - "('QK', 'AccumulationBuffer')": 1.6021360352169722e-05, "('QK', 'Register')": 0.0, + "('QK', 'WeightBuffer')": 1.3100800000000002e-05, + "('QK', 'MainMemory')": 0.03998046875, + "('QK', 'InputBuffer')": 1.3247999959276058e-05, + "('QK', 'GlobalBuffer')": 1.4098714018473402e-06, + "('QK', 'AccumulationBuffer')": 1.6021360509554146e-05, "('QK', 'MAC')": 2.6201600121567026e-05, - "('QK_softmax', 'GlobalBuffer')": 1.1564843589439988e-06, - "('QK_softmax', 'InputBuffer')": 2.047000009497424e-07, - "('QK_softmax', 'AccumulationBuffer')": 1.0013350220106076e-06, + "('QK_softmax', 'InputBuffer')": 2.0470000000000003e-07, + "('QK_softmax', 'GlobalBuffer')": 1.1564844137931036e-06, + "('QK_softmax', 'AccumulationBuffer')": 1.0013350318471341e-06, "('QK_softmax', 'MAC')": 1.6376000075979391e-06, - "('AV', 'MainMemory')": 0.03998046740889549, + "('AV', 'AccumulationBuffer')": 0.00012954839621670544, "('AV', 'GlobalBuffer')": 2.205342980232672e-06, "('AV', 'InputBuffer')": 1.7399499938619556e-06, - "('AV', 'WeightBuffer')": 1.3100800060783513e-05, - "('AV', 'AccumulationBuffer')": 0.00012954839621670544, "('AV', 'Register')": 0.0, + "('AV', 'WeightBuffer')": 1.3100800000000002e-05, + "('AV', 'MainMemory')": 0.03998046875, "('AV', 'MAC')": 2.6201600121567026e-05, - "('Z', 'MainMemory')": 0.07999999821186066, - "('Z', 'GlobalBuffer')": 7.23155864079672e-08, "('Z', 'InputBuffer')": 3.283199930592673e-06, - "('Z', 'WeightBuffer')": 2.6214400349999778e-05, - "('Z', 'AccumulationBuffer')": 3.205837492714636e-05, + "('Z', 'GlobalBuffer')": 7.231558620689656e-08, "('Z', 'Register')": 0.0, + "('Z', 'WeightBuffer')": 2.6214400000000004e-05, + "('Z', 'MainMemory')": 0.08, + "('Z', 'AccumulationBuffer')": 3.205837492714636e-05, "('Z', 'MAC')": 6.5536000874999445e-06, - "('FFA', 'MainMemory')": 0.3199999928474426, - "('FFA', 'GlobalBuffer')": 1.8078895891449065e-07, - "('FFA', 'InputBuffer')": 1.3113600289216265e-05, - "('FFA', 'WeightBuffer')": 0.00010485760139999911, - "('FFA', 'AccumulationBuffer')": 0.00012823349970858544, + "('FFA', 'AccumulationBuffer')": 0.00012823349808917202, + "('FFA', 'GlobalBuffer')": 2.169467592239016e-07, "('FFA', 'Register')": 0.0, + "('FFA', 'WeightBuffer')": 0.00010485760000000002, + "('FFA', 'MainMemory')": 0.32, + "('FFA', 'InputBuffer')": 1.3120000403432641e-05, "('FFA', 'MAC')": 2.6214400349999778e-05, - "('FFB', 'MainMemory')": 0.32001954317092896, - "('FFB', 'GlobalBuffer')": 2.169467592239016e-07, - "('FFB', 'InputBuffer')": 1.3132799722370692e-05, - "('FFB', 'WeightBuffer')": 0.00010485760139999911, - "('FFB', 'AccumulationBuffer')": 0.00012823349970858544, + "('FFB', 'InputBuffer')": 1.3209600000000001e-05, + "('FFB', 'GlobalBuffer')": 2.892623456318688e-07, + "('FFB', 'AccumulationBuffer')": 0.00012848395272158086, + "('FFB', 'MainMemory')": 0.32001953125, "('FFB', 'Register')": 0.0, + "('FFB', 'WeightBuffer')": 0.00010485760000000002, "('FFB', 'MAC')": 2.6214400349999778e-05 }, "actions": { - "('I', 'GlobalBuffer', 'I_in', 'read')": 65536.0, "('I', 'GlobalBuffer', 'I_in', 'write')": 0.0, - "('I', 'MainMemory', 'I_in', 'read')": 0.0, + "('I', 'GlobalBuffer', 'I_in', 'read')": 65536.0, "('I', 'MainMemory', 'I_in', 'write')": 65536.0, + "('I', 'MainMemory', 'I_in', 'read')": 0.0, "('I', 'MAC', 'None', 'compute')": 0.0, "('V', 'InputBuffer', 'I', 'read')": 33554432.0, "('V', 'InputBuffer', 'I', 'write')": 65536.0, - "('V', 'GlobalBuffer', 'I', 'read')": 65536.0, "('V', 'GlobalBuffer', 'I', 'write')": 0.0, + "('V', 'GlobalBuffer', 'I', 'read')": 65536.0, "('V', 'AccumulationBuffer', 'V', 'read')": 100663296.0, "('V', 'AccumulationBuffer', 'V', 'write')": 100663296.0, - "('V', 'GlobalBuffer', 'V', 'read')": 65536.0, "('V', 'GlobalBuffer', 'V', 'write')": 65536.0, - "('V', 'MainMemory', 'V', 'read')": 0.0, + "('V', 'GlobalBuffer', 'V', 'read')": 65536.0, "('V', 'MainMemory', 'V', 'write')": 65536.0, + "('V', 'MainMemory', 'V', 'read')": 0.0, "('V', 'Register', 'WV', 'read')": 268435456.0, "('V', 'Register', 'WV', 'write')": 268435456.0, "('V', 'WeightBuffer', 'WV', 'read')": 268435456.0, "('V', 'WeightBuffer', 'WV', 'write')": 268435456.0, - "('V', 'MainMemory', 'WV', 'read')": 268435456.0, "('V', 'MainMemory', 'WV', 'write')": 0.0, + "('V', 'MainMemory', 'WV', 'read')": 268435456.0, "('V', 'MAC', 'None', 'compute')": 16777216.0, "('K', 'InputBuffer', 'I', 'read')": 33554432.0, "('K', 'InputBuffer', 'I', 'write')": 65536.0, - "('K', 'GlobalBuffer', 'I', 'read')": 65536.0, "('K', 'GlobalBuffer', 'I', 'write')": 0.0, + "('K', 'GlobalBuffer', 'I', 'read')": 65536.0, "('K', 'AccumulationBuffer', 'K', 'read')": 100663296.0, "('K', 'AccumulationBuffer', 'K', 'write')": 100663296.0, - "('K', 'GlobalBuffer', 'K', 'read')": 65536.0, "('K', 'GlobalBuffer', 'K', 'write')": 65536.0, - "('K', 'MainMemory', 'K', 'read')": 0.0, + "('K', 'GlobalBuffer', 'K', 'read')": 65536.0, "('K', 'MainMemory', 'K', 'write')": 65536.0, + "('K', 'MainMemory', 'K', 'read')": 0.0, "('K', 'Register', 'WK', 'read')": 268435456.0, "('K', 'Register', 'WK', 'write')": 268435456.0, "('K', 'WeightBuffer', 'WK', 'read')": 268435456.0, "('K', 'WeightBuffer', 'WK', 'write')": 268435456.0, - "('K', 'MainMemory', 'WK', 'read')": 268435456.0, "('K', 'MainMemory', 'WK', 'write')": 0.0, + "('K', 'MainMemory', 'WK', 'read')": 268435456.0, "('K', 'MAC', 'None', 'compute')": 16777216.0, "('Q', 'InputBuffer', 'I', 'read')": 33554432.0, "('Q', 'InputBuffer', 'I', 'write')": 65536.0, - "('Q', 'GlobalBuffer', 'I', 'read')": 65536.0, "('Q', 'GlobalBuffer', 'I', 'write')": 0.0, + "('Q', 'GlobalBuffer', 'I', 'read')": 65536.0, "('Q', 'AccumulationBuffer', 'Q', 'read')": 100663296.0, "('Q', 'AccumulationBuffer', 'Q', 'write')": 100663296.0, - "('Q', 'GlobalBuffer', 'Q', 'read')": 0.0, "('Q', 'GlobalBuffer', 'Q', 'write')": 65536.0, + "('Q', 'GlobalBuffer', 'Q', 'read')": 0.0, "('Q', 'Register', 'WQ', 'read')": 268435456.0, "('Q', 'Register', 'WQ', 'write')": 268435456.0, "('Q', 'WeightBuffer', 'WQ', 'read')": 268435456.0, "('Q', 'WeightBuffer', 'WQ', 'write')": 268435456.0, - "('Q', 'MainMemory', 'WQ', 'read')": 268435456.0, "('Q', 'MainMemory', 'WQ', 'write')": 0.0, + "('Q', 'MainMemory', 'WQ', 'read')": 268435456.0, "('Q', 'MAC', 'None', 'compute')": 16777216.0, "('QK', 'Register', 'K_cache', 'read')": 134152192.0, "('QK', 'Register', 'K_cache', 'write')": 134152192.0, "('QK', 'WeightBuffer', 'K_cache', 'read')": 134152192.0, "('QK', 'WeightBuffer', 'K_cache', 'write')": 134152192.0, - "('QK', 'MainMemory', 'K_cache', 'read')": 134152192.0, "('QK', 'MainMemory', 'K_cache', 'write')": 0.0, + "('QK', 'MainMemory', 'K_cache', 'read')": 134152192.0, "('QK', 'InputBuffer', 'Q', 'read')": 134152192.0, "('QK', 'InputBuffer', 'Q', 'write')": 1507328.0, - "('QK', 'GlobalBuffer', 'Q', 'read')": 1507328.0, "('QK', 'GlobalBuffer', 'Q', 'write')": 0.0, + "('QK', 'GlobalBuffer', 'Q', 'read')": 1507328.0, "('QK', 'AccumulationBuffer', 'QK', 'read')": 50307072.0, "('QK', 'AccumulationBuffer', 'QK', 'write')": 50307072.0, "('QK', 'GlobalBuffer', 'QK', 'read')": 0.0, @@ -4328,8 +4328,8 @@ "('QK_softmax', 'MAC', 'None', 'compute')": 65504.0, "('AV', 'AccumulationBuffer', 'AV', 'read')": 406781952.0, "('AV', 'AccumulationBuffer', 'AV', 'write')": 406781952.0, - "('AV', 'GlobalBuffer', 'AV', 'read')": 1441792.0, "('AV', 'GlobalBuffer', 'AV', 'write')": 1507328.0, + "('AV', 'GlobalBuffer', 'AV', 'read')": 1441792.0, "('AV', 'InputBuffer', 'QK_softmax', 'read')": 16769024.0, "('AV', 'InputBuffer', 'QK_softmax', 'write')": 1048064.0, "('AV', 'GlobalBuffer', 'QK_softmax', 'read')": 1048064.0, @@ -4338,23 +4338,23 @@ "('AV', 'Register', 'V_cache', 'write')": 134152192.0, "('AV', 'WeightBuffer', 'V_cache', 'read')": 134152192.0, "('AV', 'WeightBuffer', 'V_cache', 'write')": 134152192.0, - "('AV', 'MainMemory', 'V_cache', 'read')": 134152192.0, "('AV', 'MainMemory', 'V_cache', 'write')": 0.0, + "('AV', 'MainMemory', 'V_cache', 'read')": 134152192.0, "('AV', 'MAC', 'None', 'compute')": 8384512.0, "('Z', 'InputBuffer', 'AV', 'read')": 33554432.0, "('Z', 'InputBuffer', 'AV', 'write')": 65536.0, - "('Z', 'GlobalBuffer', 'AV', 'read')": 65536.0, "('Z', 'GlobalBuffer', 'AV', 'write')": 0.0, + "('Z', 'GlobalBuffer', 'AV', 'read')": 65536.0, "('Z', 'Register', 'WZ', 'read')": 268435456.0, "('Z', 'Register', 'WZ', 'write')": 268435456.0, "('Z', 'WeightBuffer', 'WZ', 'read')": 268435456.0, "('Z', 'WeightBuffer', 'WZ', 'write')": 268435456.0, - "('Z', 'MainMemory', 'WZ', 'read')": 268435456.0, "('Z', 'MainMemory', 'WZ', 'write')": 0.0, + "('Z', 'MainMemory', 'WZ', 'read')": 268435456.0, "('Z', 'AccumulationBuffer', 'Z', 'read')": 100663296.0, "('Z', 'AccumulationBuffer', 'Z', 'write')": 100663296.0, - "('Z', 'GlobalBuffer', 'Z', 'read')": 0.0, "('Z', 'GlobalBuffer', 'Z', 'write')": 65536.0, + "('Z', 'GlobalBuffer', 'Z', 'read')": 0.0, "('Z', 'MAC', 'None', 'compute')": 16777216.0, "('FFA', 'AccumulationBuffer', 'FFA', 'read')": 402653184.0, "('FFA', 'AccumulationBuffer', 'FFA', 'write')": 402653184.0, @@ -4364,29 +4364,29 @@ "('FFA', 'Register', 'WFFA', 'write')": 1073741824.0, "('FFA', 'WeightBuffer', 'WFFA', 'read')": 1073741824.0, "('FFA', 'WeightBuffer', 'WFFA', 'write')": 1073741824.0, - "('FFA', 'MainMemory', 'WFFA', 'read')": 1073741824.0, "('FFA', 'MainMemory', 'WFFA', 'write')": 0.0, + "('FFA', 'MainMemory', 'WFFA', 'read')": 1073741824.0, "('FFA', 'InputBuffer', 'Z', 'read')": 134217728.0, - "('FFA', 'InputBuffer', 'Z', 'write')": 65536.0, - "('FFA', 'GlobalBuffer', 'Z', 'read')": 65536.0, + "('FFA', 'InputBuffer', 'Z', 'write')": 131072.0, "('FFA', 'GlobalBuffer', 'Z', 'write')": 0.0, + "('FFA', 'GlobalBuffer', 'Z', 'read')": 131072.0, "('FFA', 'MAC', 'None', 'compute')": 67108864.0, "('FFB', 'InputBuffer', 'FFA', 'read')": 134217728.0, - "('FFB', 'InputBuffer', 'FFA', 'write')": 262144.0, + "('FFB', 'InputBuffer', 'FFA', 'write')": 1048576.0, "('FFB', 'GlobalBuffer', 'FFA', 'read')": 262144.0, "('FFB', 'GlobalBuffer', 'FFA', 'write')": 0.0, - "('FFB', 'AccumulationBuffer', 'FFB', 'read')": 402653184.0, - "('FFB', 'AccumulationBuffer', 'FFB', 'write')": 402653184.0, - "('FFB', 'GlobalBuffer', 'FFB', 'read')": 65536.0, - "('FFB', 'GlobalBuffer', 'FFB', 'write')": 65536.0, - "('FFB', 'MainMemory', 'FFB', 'read')": 0.0, + "('FFB', 'AccumulationBuffer', 'FFB', 'read')": 403439616.0, + "('FFB', 'AccumulationBuffer', 'FFB', 'write')": 403439616.0, + "('FFB', 'GlobalBuffer', 'FFB', 'write')": 131072.0, + "('FFB', 'GlobalBuffer', 'FFB', 'read')": 131072.0, "('FFB', 'MainMemory', 'FFB', 'write')": 65536.0, + "('FFB', 'MainMemory', 'FFB', 'read')": 0.0, "('FFB', 'Register', 'WFFB', 'read')": 1073741824.0, "('FFB', 'Register', 'WFFB', 'write')": 1073741824.0, "('FFB', 'WeightBuffer', 'WFFB', 'read')": 1073741824.0, "('FFB', 'WeightBuffer', 'WFFB', 'write')": 1073741824.0, - "('FFB', 'MainMemory', 'WFFB', 'read')": 1073741824.0, "('FFB', 'MainMemory', 'WFFB', 'write')": 0.0, + "('FFB', 'MainMemory', 'WFFB', 'read')": 1073741824.0, "('FFB', 'MAC', 'None', 'compute')": 67108864.0 }, "n_mappings": 1.0 @@ -4851,54 +4851,56 @@ "('Matmul1', 'MAC', 'leak')": 0.0 }, "latency_per_component": { - "('Matmul0', 'MainMemory')": 1.3342019933304528e-08, - "('Matmul0', 'GlobalBuffer')": 7.999999773744548e-09, "('Matmul0', 'LocalBuffer')": 0.0, + "('Matmul0', 'MainMemory')": 1.3342019543973941e-08, + "('Matmul0', 'GlobalBuffer (read)')": 7.999999773744548e-09, + "('Matmul0', 'GlobalBuffer (write)')": 8e-09, "('Matmul0', 'Register')": 0.0, "('Matmul0', 'MAC')": 1.5238095230074578e-08, - "('Matmul1', 'MainMemory')": 1.3342019933304528e-08, - "('Matmul1', 'GlobalBuffer')": 9.99999993922529e-09, "('Matmul1', 'LocalBuffer')": 0.0, + "('Matmul1', 'GlobalBuffer (read)')": 9.99999993922529e-09, + "('Matmul1', 'GlobalBuffer (write)')": 4e-09, + "('Matmul1', 'MainMemory')": 1.3342019543973941e-08, "('Matmul1', 'Register')": 0.0, "('Matmul1', 'MAC')": 1.5238095230074578e-08 }, "actions": { "('Matmul0', 'LocalBuffer', 'T0', 'read')": 32768.0, "('Matmul0', 'LocalBuffer', 'T0', 'write')": 32768.0, - "('Matmul0', 'MainMemory', 'T0', 'read')": 32768.0, "('Matmul0', 'MainMemory', 'T0', 'write')": 0.0, + "('Matmul0', 'MainMemory', 'T0', 'read')": 32768.0, "('Matmul0', 'LocalBuffer', 'T1', 'read')": 32768.0, "('Matmul0', 'LocalBuffer', 'T1', 'write')": 32768.0, - "('Matmul0', 'GlobalBuffer', 'T1', 'read')": 0.0, "('Matmul0', 'GlobalBuffer', 'T1', 'write')": 32768.0, + "('Matmul0', 'GlobalBuffer', 'T1', 'read')": 0.0, "('Matmul0', 'Register', 'W0', 'read')": 2097152.0, "('Matmul0', 'Register', 'W0', 'write')": 131072.0, - "('Matmul0', 'GlobalBuffer', 'W0', 'read')": 131072.0, "('Matmul0', 'GlobalBuffer', 'W0', 'write')": 32768.0, - "('Matmul0', 'MainMemory', 'W0', 'read')": 32768.0, + "('Matmul0', 'GlobalBuffer', 'W0', 'read')": 131072.0, "('Matmul0', 'MainMemory', 'W0', 'write')": 0.0, + "('Matmul0', 'MainMemory', 'W0', 'read')": 32768.0, "('Matmul0', 'MAC', 'None', 'compute')": 262144.0, "('Matmul1', 'LocalBuffer', 'T1', 'read')": 32768.0, "('Matmul1', 'LocalBuffer', 'T1', 'write')": 32768.0, - "('Matmul1', 'GlobalBuffer', 'T1', 'read')": 32768.0, "('Matmul1', 'GlobalBuffer', 'T1', 'write')": 0.0, + "('Matmul1', 'GlobalBuffer', 'T1', 'read')": 32768.0, "('Matmul1', 'LocalBuffer', 'T2', 'read')": 32768.0, "('Matmul1', 'LocalBuffer', 'T2', 'write')": 32768.0, - "('Matmul1', 'MainMemory', 'T2', 'read')": 0.0, "('Matmul1', 'MainMemory', 'T2', 'write')": 32768.0, + "('Matmul1', 'MainMemory', 'T2', 'read')": 0.0, "('Matmul1', 'Register', 'W1', 'read')": 2097152.0, "('Matmul1', 'Register', 'W1', 'write')": 131072.0, - "('Matmul1', 'GlobalBuffer', 'W1', 'read')": 131072.0, "('Matmul1', 'GlobalBuffer', 'W1', 'write')": 32768.0, - "('Matmul1', 'MainMemory', 'W1', 'read')": 32768.0, + "('Matmul1', 'GlobalBuffer', 'W1', 'read')": 131072.0, "('Matmul1', 'MainMemory', 'W1', 'write')": 0.0, + "('Matmul1', 'MainMemory', 'W1', 'read')": 32768.0, "('Matmul1', 'MAC', 'None', 'compute')": 262144.0 }, "n_mappings": 1.0 }, "tpu_v4i|matmuls|KN=64,M=64,N_EINSUMS=2|unfused": { "energy": 2.144731152171063e-06, - "latency": 4.002605891173516e-08, + "latency": 4.0026058631921826e-08, "energy_per_component": { "('Matmul0', 'LocalBuffer', 'read')": 1.6318464e-08, "('Matmul0', 'LocalBuffer', 'write')": 1.9202048e-08, @@ -4906,8 +4908,8 @@ "('Matmul0', 'MainMemory', 'write')": 2.3035904e-07, "('Matmul0', 'Register', 'read')": 0.0, "('Matmul0', 'Register', 'write')": 0.0, - "('Matmul0', 'GlobalBuffer', 'read')": 2.4641536811031983e-07, "('Matmul0', 'GlobalBuffer', 'write')": 7.733247997521175e-08, + "('Matmul0', 'GlobalBuffer', 'read')": 2.4641536811031983e-07, "('Matmul0', 'MAC', 'compute')": 2.2020096e-08, "('Matmul0', 'MainMemory', 'leak')": 0.0, "('Matmul0', 'GlobalBuffer', 'leak')": 0.0, @@ -4921,8 +4923,8 @@ "('Matmul1', 'MainMemory', 'write')": 2.3035904e-07, "('Matmul1', 'Register', 'read')": 0.0, "('Matmul1', 'Register', 'write')": 0.0, - "('Matmul1', 'GlobalBuffer', 'read')": 2.4641536811031983e-07, "('Matmul1', 'GlobalBuffer', 'write')": 7.733247997521175e-08, + "('Matmul1', 'GlobalBuffer', 'read')": 2.4641536811031983e-07, "('Matmul1', 'MAC', 'compute')": 2.2020096e-08, "('Matmul1', 'MainMemory', 'leak')": 0.0, "('Matmul1', 'GlobalBuffer', 'leak')": 0.0, @@ -4932,54 +4934,56 @@ "('Matmul1', 'MAC', 'leak')": 0.0 }, "latency_per_component": { - "('Matmul0', 'MainMemory')": 2.001302945586758e-08, - "('Matmul0', 'GlobalBuffer')": 7.999999773744548e-09, "('Matmul0', 'LocalBuffer')": 0.0, + "('Matmul0', 'MainMemory')": 2.0013029315960913e-08, "('Matmul0', 'Register')": 0.0, + "('Matmul0', 'GlobalBuffer (read)')": 7.999999773744548e-09, + "('Matmul0', 'GlobalBuffer (write)')": 4e-09, "('Matmul0', 'MAC')": 1.5238095230074578e-08, - "('Matmul1', 'MainMemory')": 2.001302945586758e-08, - "('Matmul1', 'GlobalBuffer')": 7.999999773744548e-09, "('Matmul1', 'LocalBuffer')": 0.0, + "('Matmul1', 'MainMemory')": 2.0013029315960913e-08, "('Matmul1', 'Register')": 0.0, + "('Matmul1', 'GlobalBuffer (read)')": 7.999999773744548e-09, + "('Matmul1', 'GlobalBuffer (write)')": 4e-09, "('Matmul1', 'MAC')": 1.5238095230074578e-08 }, "actions": { "('Matmul0', 'LocalBuffer', 'T0', 'read')": 32768.0, "('Matmul0', 'LocalBuffer', 'T0', 'write')": 32768.0, - "('Matmul0', 'MainMemory', 'T0', 'read')": 32768.0, "('Matmul0', 'MainMemory', 'T0', 'write')": 0.0, + "('Matmul0', 'MainMemory', 'T0', 'read')": 32768.0, "('Matmul0', 'LocalBuffer', 'T1', 'read')": 32768.0, "('Matmul0', 'LocalBuffer', 'T1', 'write')": 32768.0, - "('Matmul0', 'MainMemory', 'T1', 'read')": 0.0, "('Matmul0', 'MainMemory', 'T1', 'write')": 32768.0, + "('Matmul0', 'MainMemory', 'T1', 'read')": 0.0, "('Matmul0', 'Register', 'W0', 'read')": 2097152.0, "('Matmul0', 'Register', 'W0', 'write')": 131072.0, - "('Matmul0', 'GlobalBuffer', 'W0', 'read')": 131072.0, "('Matmul0', 'GlobalBuffer', 'W0', 'write')": 32768.0, - "('Matmul0', 'MainMemory', 'W0', 'read')": 32768.0, + "('Matmul0', 'GlobalBuffer', 'W0', 'read')": 131072.0, "('Matmul0', 'MainMemory', 'W0', 'write')": 0.0, + "('Matmul0', 'MainMemory', 'W0', 'read')": 32768.0, "('Matmul0', 'MAC', 'None', 'compute')": 262144.0, "('Matmul1', 'LocalBuffer', 'T1', 'read')": 32768.0, "('Matmul1', 'LocalBuffer', 'T1', 'write')": 32768.0, - "('Matmul1', 'MainMemory', 'T1', 'read')": 32768.0, "('Matmul1', 'MainMemory', 'T1', 'write')": 0.0, + "('Matmul1', 'MainMemory', 'T1', 'read')": 32768.0, "('Matmul1', 'LocalBuffer', 'T2', 'read')": 32768.0, "('Matmul1', 'LocalBuffer', 'T2', 'write')": 32768.0, - "('Matmul1', 'MainMemory', 'T2', 'read')": 0.0, "('Matmul1', 'MainMemory', 'T2', 'write')": 32768.0, + "('Matmul1', 'MainMemory', 'T2', 'read')": 0.0, "('Matmul1', 'Register', 'W1', 'read')": 2097152.0, "('Matmul1', 'Register', 'W1', 'write')": 131072.0, - "('Matmul1', 'GlobalBuffer', 'W1', 'read')": 131072.0, "('Matmul1', 'GlobalBuffer', 'W1', 'write')": 32768.0, - "('Matmul1', 'MainMemory', 'W1', 'read')": 32768.0, + "('Matmul1', 'GlobalBuffer', 'W1', 'read')": 131072.0, "('Matmul1', 'MainMemory', 'W1', 'write')": 0.0, + "('Matmul1', 'MainMemory', 'W1', 'read')": 32768.0, "('Matmul1', 'MAC', 'None', 'compute')": 262144.0 }, "n_mappings": 1.0 }, "tpu_v4i|three_matmuls_annotated||fused": { "energy": 1.0558373985026378e-05, - "latency": 1.467361556706237e-07, + "latency": 1.4673615610869268e-07, "energy_per_component": { "('Matmul1', 'LocalBuffer', 'read')": 6.5273856e-08, "('Matmul1', 'LocalBuffer', 'write')": 7.6808192e-08, @@ -5026,74 +5030,77 @@ "('Matmul3', 'MAC', 'leak')": 0.0 }, "latency_per_component": { - "('Matmul1', 'MainMemory')": 5.336807973321811e-08, - "('Matmul1', 'GlobalBuffer')": 3.199999909497819e-08, "('Matmul1', 'LocalBuffer')": 0.0, + "('Matmul1', 'MainMemory')": 5.3368078175895765e-08, + "('Matmul1', 'GlobalBuffer (read)')": 3.199999909497819e-08, + "('Matmul1', 'GlobalBuffer (write)')": 3.2e-08, "('Matmul1', 'Register')": 0.0, "('Matmul1', 'MAC')": 3.0476190460149155e-08, - "('Matmul2', 'MainMemory')": 2.6684039866609055e-08, - "('Matmul2', 'GlobalBuffer')": 3.999999975690116e-08, "('Matmul2', 'LocalBuffer')": 0.0, + "('Matmul2', 'GlobalBuffer (read)')": 3.999999975690116e-08, + "('Matmul2', 'GlobalBuffer (write)')": 3.2e-08, "('Matmul2', 'Register')": 0.0, + "('Matmul2', 'MainMemory')": 2.6684039087947883e-08, "('Matmul2', 'MAC')": 3.0476190460149155e-08, - "('Matmul3', 'MainMemory')": 5.336807973321811e-08, - "('Matmul3', 'GlobalBuffer')": 3.999999975690116e-08, "('Matmul3', 'LocalBuffer')": 0.0, + "('Matmul3', 'GlobalBuffer (read)')": 3.999999975690116e-08, + "('Matmul3', 'GlobalBuffer (write)')": 1.6e-08, + "('Matmul3', 'MainMemory')": 5.3368078175895765e-08, "('Matmul3', 'Register')": 0.0, "('Matmul3', 'MAC')": 3.0476190460149155e-08 }, "actions": { "('Matmul1', 'LocalBuffer', 'T0', 'read')": 131072.0, "('Matmul1', 'LocalBuffer', 'T0', 'write')": 131072.0, - "('Matmul1', 'MainMemory', 'T0', 'read')": 131072.0, "('Matmul1', 'MainMemory', 'T0', 'write')": 0.0, + "('Matmul1', 'MainMemory', 'T0', 'read')": 131072.0, "('Matmul1', 'LocalBuffer', 'T1', 'read')": 131072.0, "('Matmul1', 'LocalBuffer', 'T1', 'write')": 131072.0, - "('Matmul1', 'GlobalBuffer', 'T1', 'read')": 0.0, "('Matmul1', 'GlobalBuffer', 'T1', 'write')": 131072.0, + "('Matmul1', 'GlobalBuffer', 'T1', 'read')": 0.0, "('Matmul1', 'Register', 'W0', 'read')": 16777216.0, "('Matmul1', 'Register', 'W0', 'write')": 524288.0, - "('Matmul1', 'GlobalBuffer', 'W0', 'read')": 524288.0, "('Matmul1', 'GlobalBuffer', 'W0', 'write')": 131072.0, - "('Matmul1', 'MainMemory', 'W0', 'read')": 131072.0, + "('Matmul1', 'GlobalBuffer', 'W0', 'read')": 524288.0, "('Matmul1', 'MainMemory', 'W0', 'write')": 0.0, + "('Matmul1', 'MainMemory', 'W0', 'read')": 131072.0, "('Matmul1', 'MAC', 'None', 'compute')": 2097152.0, "('Matmul2', 'LocalBuffer', 'T1', 'read')": 131072.0, "('Matmul2', 'LocalBuffer', 'T1', 'write')": 131072.0, - "('Matmul2', 'GlobalBuffer', 'T1', 'read')": 131072.0, "('Matmul2', 'GlobalBuffer', 'T1', 'write')": 0.0, + "('Matmul2', 'GlobalBuffer', 'T1', 'read')": 131072.0, "('Matmul2', 'LocalBuffer', 'T2', 'read')": 131072.0, "('Matmul2', 'LocalBuffer', 'T2', 'write')": 131072.0, - "('Matmul2', 'GlobalBuffer', 'T2', 'read')": 0.0, "('Matmul2', 'GlobalBuffer', 'T2', 'write')": 131072.0, + "('Matmul2', 'GlobalBuffer', 'T2', 'read')": 0.0, "('Matmul2', 'Register', 'W1', 'read')": 16777216.0, "('Matmul2', 'Register', 'W1', 'write')": 524288.0, - "('Matmul2', 'GlobalBuffer', 'W1', 'read')": 524288.0, "('Matmul2', 'GlobalBuffer', 'W1', 'write')": 131072.0, - "('Matmul2', 'MainMemory', 'W1', 'read')": 131072.0, + "('Matmul2', 'GlobalBuffer', 'W1', 'read')": 524288.0, "('Matmul2', 'MainMemory', 'W1', 'write')": 0.0, + "('Matmul2', 'MainMemory', 'W1', 'read')": 131072.0, "('Matmul2', 'MAC', 'None', 'compute')": 2097152.0, "('Matmul3', 'LocalBuffer', 'T2', 'read')": 131072.0, "('Matmul3', 'LocalBuffer', 'T2', 'write')": 131072.0, - "('Matmul3', 'GlobalBuffer', 'T2', 'read')": 131072.0, "('Matmul3', 'GlobalBuffer', 'T2', 'write')": 0.0, + "('Matmul3', 'GlobalBuffer', 'T2', 'read')": 131072.0, "('Matmul3', 'LocalBuffer', 'T3', 'read')": 131072.0, "('Matmul3', 'LocalBuffer', 'T3', 'write')": 131072.0, - "('Matmul3', 'MainMemory', 'T3', 'read')": 0.0, "('Matmul3', 'MainMemory', 'T3', 'write')": 131072.0, + "('Matmul3', 'MainMemory', 'T3', 'read')": 0.0, "('Matmul3', 'Register', 'W2', 'read')": 16777216.0, "('Matmul3', 'Register', 'W2', 'write')": 524288.0, - "('Matmul3', 'GlobalBuffer', 'W2', 'read')": 524288.0, "('Matmul3', 'GlobalBuffer', 'W2', 'write')": 131072.0, - "('Matmul3', 'MainMemory', 'W2', 'read')": 131072.0, + "('Matmul3', 'GlobalBuffer', 'W2', 'read')": 524288.0, "('Matmul3', 'MainMemory', 'W2', 'write')": 0.0, + "('Matmul3', 'MainMemory', 'W2', 'read')": 131072.0, "('Matmul3', 'MAC', 'None', 'compute')": 2097152.0 }, "n_mappings": 1.0 }, "tpu_v4i|three_matmuls_annotated||unfused": { "energy": 1.313262806502638e-05, - "latency": 2.4015633925955626e-07, + "latency": 2.4015635179153095e-07, "energy_per_component": { "('Matmul1', 'LocalBuffer', 'read')": 6.5273856e-08, "('Matmul1', 'LocalBuffer', 'write')": 7.6808192e-08, @@ -5101,8 +5108,8 @@ "('Matmul1', 'MainMemory', 'write')": 9.2143616e-07, "('Matmul1', 'Register', 'read')": 0.0, "('Matmul1', 'Register', 'write')": 0.0, - "('Matmul1', 'GlobalBuffer', 'read')": 9.856614724412793e-07, "('Matmul1', 'GlobalBuffer', 'write')": 3.09329919900847e-07, + "('Matmul1', 'GlobalBuffer', 'read')": 9.856614724412793e-07, "('Matmul1', 'MAC', 'compute')": 1.76160768e-07, "('Matmul1', 'MainMemory', 'leak')": 0.0, "('Matmul1', 'GlobalBuffer', 'leak')": 0.0, @@ -5116,8 +5123,8 @@ "('Matmul2', 'MainMemory', 'write')": 9.2143616e-07, "('Matmul2', 'Register', 'read')": 0.0, "('Matmul2', 'Register', 'write')": 0.0, - "('Matmul2', 'GlobalBuffer', 'read')": 9.856614724412793e-07, "('Matmul2', 'GlobalBuffer', 'write')": 3.09329919900847e-07, + "('Matmul2', 'GlobalBuffer', 'read')": 9.856614724412793e-07, "('Matmul2', 'MAC', 'compute')": 1.76160768e-07, "('Matmul2', 'MainMemory', 'leak')": 0.0, "('Matmul2', 'GlobalBuffer', 'leak')": 0.0, @@ -5131,8 +5138,8 @@ "('Matmul3', 'MainMemory', 'write')": 9.2143616e-07, "('Matmul3', 'Register', 'read')": 0.0, "('Matmul3', 'Register', 'write')": 0.0, - "('Matmul3', 'GlobalBuffer', 'read')": 9.856614724412793e-07, "('Matmul3', 'GlobalBuffer', 'write')": 3.09329919900847e-07, + "('Matmul3', 'GlobalBuffer', 'read')": 9.856614724412793e-07, "('Matmul3', 'MAC', 'compute')": 1.76160768e-07, "('Matmul3', 'MainMemory', 'leak')": 0.0, "('Matmul3', 'GlobalBuffer', 'leak')": 0.0, @@ -5142,74 +5149,77 @@ "('Matmul3', 'MAC', 'leak')": 0.0 }, "latency_per_component": { - "('Matmul1', 'MainMemory')": 8.005211782347033e-08, - "('Matmul1', 'GlobalBuffer')": 3.199999909497819e-08, "('Matmul1', 'LocalBuffer')": 0.0, + "('Matmul1', 'MainMemory')": 8.005211726384365e-08, "('Matmul1', 'Register')": 0.0, + "('Matmul1', 'GlobalBuffer (read)')": 3.199999909497819e-08, + "('Matmul1', 'GlobalBuffer (write)')": 1.6e-08, "('Matmul1', 'MAC')": 3.0476190460149155e-08, - "('Matmul2', 'MainMemory')": 8.005211782347033e-08, - "('Matmul2', 'GlobalBuffer')": 3.199999909497819e-08, "('Matmul2', 'LocalBuffer')": 0.0, + "('Matmul2', 'MainMemory')": 8.005211726384365e-08, "('Matmul2', 'Register')": 0.0, + "('Matmul2', 'GlobalBuffer (read)')": 3.199999909497819e-08, + "('Matmul2', 'GlobalBuffer (write)')": 1.6e-08, "('Matmul2', 'MAC')": 3.0476190460149155e-08, - "('Matmul3', 'MainMemory')": 8.005211782347033e-08, - "('Matmul3', 'GlobalBuffer')": 3.199999909497819e-08, "('Matmul3', 'LocalBuffer')": 0.0, + "('Matmul3', 'MainMemory')": 8.005211726384365e-08, "('Matmul3', 'Register')": 0.0, + "('Matmul3', 'GlobalBuffer (read)')": 3.199999909497819e-08, + "('Matmul3', 'GlobalBuffer (write)')": 1.6e-08, "('Matmul3', 'MAC')": 3.0476190460149155e-08 }, "actions": { "('Matmul1', 'LocalBuffer', 'T0', 'read')": 131072.0, "('Matmul1', 'LocalBuffer', 'T0', 'write')": 131072.0, - "('Matmul1', 'MainMemory', 'T0', 'read')": 131072.0, "('Matmul1', 'MainMemory', 'T0', 'write')": 0.0, + "('Matmul1', 'MainMemory', 'T0', 'read')": 131072.0, "('Matmul1', 'LocalBuffer', 'T1', 'read')": 131072.0, "('Matmul1', 'LocalBuffer', 'T1', 'write')": 131072.0, - "('Matmul1', 'MainMemory', 'T1', 'read')": 0.0, "('Matmul1', 'MainMemory', 'T1', 'write')": 131072.0, + "('Matmul1', 'MainMemory', 'T1', 'read')": 0.0, "('Matmul1', 'Register', 'W0', 'read')": 16777216.0, "('Matmul1', 'Register', 'W0', 'write')": 524288.0, - "('Matmul1', 'GlobalBuffer', 'W0', 'read')": 524288.0, "('Matmul1', 'GlobalBuffer', 'W0', 'write')": 131072.0, - "('Matmul1', 'MainMemory', 'W0', 'read')": 131072.0, + "('Matmul1', 'GlobalBuffer', 'W0', 'read')": 524288.0, "('Matmul1', 'MainMemory', 'W0', 'write')": 0.0, + "('Matmul1', 'MainMemory', 'W0', 'read')": 131072.0, "('Matmul1', 'MAC', 'None', 'compute')": 2097152.0, "('Matmul2', 'LocalBuffer', 'T1', 'read')": 131072.0, "('Matmul2', 'LocalBuffer', 'T1', 'write')": 131072.0, - "('Matmul2', 'MainMemory', 'T1', 'read')": 131072.0, "('Matmul2', 'MainMemory', 'T1', 'write')": 0.0, + "('Matmul2', 'MainMemory', 'T1', 'read')": 131072.0, "('Matmul2', 'LocalBuffer', 'T2', 'read')": 131072.0, "('Matmul2', 'LocalBuffer', 'T2', 'write')": 131072.0, - "('Matmul2', 'MainMemory', 'T2', 'read')": 0.0, "('Matmul2', 'MainMemory', 'T2', 'write')": 131072.0, + "('Matmul2', 'MainMemory', 'T2', 'read')": 0.0, "('Matmul2', 'Register', 'W1', 'read')": 16777216.0, "('Matmul2', 'Register', 'W1', 'write')": 524288.0, - "('Matmul2', 'GlobalBuffer', 'W1', 'read')": 524288.0, "('Matmul2', 'GlobalBuffer', 'W1', 'write')": 131072.0, - "('Matmul2', 'MainMemory', 'W1', 'read')": 131072.0, + "('Matmul2', 'GlobalBuffer', 'W1', 'read')": 524288.0, "('Matmul2', 'MainMemory', 'W1', 'write')": 0.0, + "('Matmul2', 'MainMemory', 'W1', 'read')": 131072.0, "('Matmul2', 'MAC', 'None', 'compute')": 2097152.0, "('Matmul3', 'LocalBuffer', 'T2', 'read')": 131072.0, "('Matmul3', 'LocalBuffer', 'T2', 'write')": 131072.0, - "('Matmul3', 'MainMemory', 'T2', 'read')": 131072.0, "('Matmul3', 'MainMemory', 'T2', 'write')": 0.0, + "('Matmul3', 'MainMemory', 'T2', 'read')": 131072.0, "('Matmul3', 'LocalBuffer', 'T3', 'read')": 131072.0, "('Matmul3', 'LocalBuffer', 'T3', 'write')": 131072.0, - "('Matmul3', 'MainMemory', 'T3', 'read')": 0.0, "('Matmul3', 'MainMemory', 'T3', 'write')": 131072.0, + "('Matmul3', 'MainMemory', 'T3', 'read')": 0.0, "('Matmul3', 'Register', 'W2', 'read')": 16777216.0, "('Matmul3', 'Register', 'W2', 'write')": 524288.0, - "('Matmul3', 'GlobalBuffer', 'W2', 'read')": 524288.0, "('Matmul3', 'GlobalBuffer', 'W2', 'write')": 131072.0, - "('Matmul3', 'MainMemory', 'W2', 'read')": 131072.0, + "('Matmul3', 'GlobalBuffer', 'W2', 'read')": 524288.0, "('Matmul3', 'MainMemory', 'W2', 'write')": 0.0, + "('Matmul3', 'MainMemory', 'W2', 'read')": 131072.0, "('Matmul3', 'MAC', 'None', 'compute')": 2097152.0 }, "n_mappings": 1.0 }, "tpu_v4i|gpt3_175B|BATCH_SIZE=1,DECODE=False,N_NEW_TOKENS=2048|fused": { "energy": 1.174936754034471, - "latency": 0.056330591440200806, + "latency": 0.056330588035307466, "energy_per_component": { "('I', 'GlobalBuffer', 'read')": 0.00075698798592, "('I', 'MainMemory', 'write')": 0.00283065188352, @@ -5345,106 +5355,116 @@ "('FFB', 'MAC', 'leak')": 0.0 }, "latency_per_component": { - "('I', 'MainMemory')": 8.197336865123361e-05, - "('I', 'GlobalBuffer')": 2.4576000214437954e-05, + "('I', 'GlobalBuffer (read)')": 2.4576e-05, + "('I', 'GlobalBuffer (write)')": 0.0, + "('I', 'MainMemory')": 8.19733680781759e-05, "('I', 'ScalarUnit')": 0.0, - "('V', 'MainMemory')": 0.0005738135660067201, - "('V', 'GlobalBuffer')": 0.00039321600343100727, "('V', 'LocalBuffer')": 0.0, + "('V', 'GlobalBuffer (read)')": 0.00039321600343100727, + "('V', 'GlobalBuffer (write)')": 0.0, + "('V', 'MainMemory')": 0.0005738135765472312, "('V', 'Register')": 0.0, "('V', 'MAC')": 0.0044938973151147366, - "('K', 'MainMemory')": 0.0004918401828035712, - "('K', 'GlobalBuffer')": 0.00039321600343100727, "('K', 'LocalBuffer')": 0.0, + "('K', 'GlobalBuffer (read)')": 0.00039321600343100727, + "('K', 'GlobalBuffer (write)')": 4.9152e-05, "('K', 'Register')": 0.0, + "('K', 'MainMemory')": 0.0004918402084690554, "('K', 'MAC')": 0.0044938973151147366, - "('Q', 'MainMemory')": 0.0005738135660067201, - "('Q', 'GlobalBuffer')": 0.00039321600343100727, "('Q', 'LocalBuffer')": 0.0, + "('Q', 'GlobalBuffer (read)')": 0.00039321600343100727, + "('Q', 'GlobalBuffer (write)')": 0.0, + "('Q', 'MainMemory')": 0.0005738135765472312, "('Q', 'Register')": 0.0, "('Q', 'MAC')": 0.0044938973151147366, - "('QK', 'MainMemory')": 8.197336865123361e-05, - "('QK', 'GlobalBuffer')": 0.0007864320068620145, - "('QK', 'LocalBuffer')": 0.0, "('QK', 'Register')": 0.0, + "('QK', 'GlobalBuffer (read)')": 4.915200042887591e-05, + "('QK', 'GlobalBuffer (write)')": 0.000786432, + "('QK', 'LocalBuffer')": 0.0, + "('QK', 'MainMemory')": 8.19733680781759e-05, "('QK', 'MAC')": 0.0007489828858524561, - "('QK_softmax', 'GlobalBuffer')": 0.0007864320068620145, "('QK_softmax', 'LocalBuffer')": 0.0, + "('QK_softmax', 'GlobalBuffer (read)')": 0.000393216, + "('QK_softmax', 'GlobalBuffer (write)')": 0.000786432, "('QK_softmax', 'ScalarUnit')": 0.0007489828858524561, - "('AV', 'MainMemory')": 8.197336865123361e-05, - "('AV', 'GlobalBuffer')": 0.00039321600343100727, "('AV', 'LocalBuffer')": 0.0, + "('AV', 'GlobalBuffer (read)')": 0.00039321600343100727, + "('AV', 'GlobalBuffer (write)')": 9.830400085775182e-05, "('AV', 'Register')": 0.0, + "('AV', 'MainMemory')": 8.19733680781759e-05, "('AV', 'MAC')": 0.0007489828858524561, - "('Z', 'MainMemory')": 0.0004918401828035712, - "('Z', 'GlobalBuffer')": 0.00039321600343100727, "('Z', 'LocalBuffer')": 0.0, + "('Z', 'GlobalBuffer (read)')": 0.00039321600343100727, + "('Z', 'GlobalBuffer (write)')": 4.9152e-05, "('Z', 'Register')": 0.0, + "('Z', 'MainMemory')": 0.0004918402084690554, "('Z', 'MAC')": 0.0044938973151147366, - "('FFA', 'MainMemory')": 0.001967360731214285, - "('FFA', 'GlobalBuffer')": 0.001572864013724029, "('FFA', 'LocalBuffer')": 0.0, + "('FFA', 'GlobalBuffer (read)')": 0.001572864013724029, + "('FFA', 'GlobalBuffer (write)')": 0.000196608, "('FFA', 'Register')": 0.0, + "('FFA', 'MainMemory')": 0.0019673608338762216, "('FFA', 'MAC')": 0.017975589260458946, - "('FFB', 'MainMemory')": 0.0020493341144174337, - "('FFB', 'GlobalBuffer')": 0.001769472029991448, "('FFB', 'LocalBuffer')": 0.0, + "('FFB', 'GlobalBuffer (read)')": 0.001769472029991448, + "('FFB', 'GlobalBuffer (write)')": 0.00039321600343100727, + "('FFB', 'MainMemory')": 0.0020493342019543975, "('FFB', 'Register')": 0.0, "('FFB', 'MAC')": 0.017975589260458946 }, "actions": { - "('I', 'GlobalBuffer', 'I_in', 'read')": 402653184.0, "('I', 'GlobalBuffer', 'I_in', 'write')": 0.0, - "('I', 'MainMemory', 'I_in', 'read')": 0.0, + "('I', 'GlobalBuffer', 'I_in', 'read')": 402653184.0, "('I', 'MainMemory', 'I_in', 'write')": 402653184.0, + "('I', 'MainMemory', 'I_in', 'read')": 0.0, "('I', 'ScalarUnit', 'None', 'compute')": 0.0, "('V', 'LocalBuffer', 'I', 'read')": 38654705664.0, "('V', 'LocalBuffer', 'I', 'write')": 6442450944.0, - "('V', 'GlobalBuffer', 'I', 'read')": 6442450944.0, "('V', 'GlobalBuffer', 'I', 'write')": 0.0, + "('V', 'GlobalBuffer', 'I', 'read')": 6442450944.0, "('V', 'LocalBuffer', 'V', 'read')": 38654705664.0, "('V', 'LocalBuffer', 'V', 'write')": 38654705664.0, - "('V', 'MainMemory', 'V', 'read')": 0.0, "('V', 'MainMemory', 'V', 'write')": 402653184.0, + "('V', 'MainMemory', 'V', 'read')": 0.0, "('V', 'Register', 'WV', 'read')": 4947802324992.0, "('V', 'Register', 'WV', 'write')": 2415919104.0, - "('V', 'MainMemory', 'WV', 'read')": 2415919104.0, "('V', 'MainMemory', 'WV', 'write')": 0.0, + "('V', 'MainMemory', 'WV', 'read')": 2415919104.0, "('V', 'MAC', 'None', 'compute')": 309237645312.0, "('K', 'LocalBuffer', 'I', 'read')": 38654705664.0, "('K', 'LocalBuffer', 'I', 'write')": 6442450944.0, - "('K', 'GlobalBuffer', 'I', 'read')": 6442450944.0, "('K', 'GlobalBuffer', 'I', 'write')": 0.0, + "('K', 'GlobalBuffer', 'I', 'read')": 6442450944.0, "('K', 'LocalBuffer', 'K', 'read')": 38654705664.0, "('K', 'LocalBuffer', 'K', 'write')": 38654705664.0, - "('K', 'GlobalBuffer', 'K', 'read')": 0.0, "('K', 'GlobalBuffer', 'K', 'write')": 402653184.0, + "('K', 'GlobalBuffer', 'K', 'read')": 0.0, "('K', 'Register', 'WK', 'read')": 4947802324992.0, "('K', 'Register', 'WK', 'write')": 2415919104.0, - "('K', 'MainMemory', 'WK', 'read')": 2415919104.0, "('K', 'MainMemory', 'WK', 'write')": 0.0, + "('K', 'MainMemory', 'WK', 'read')": 2415919104.0, "('K', 'MAC', 'None', 'compute')": 309237645312.0, "('Q', 'LocalBuffer', 'I', 'read')": 38654705664.0, "('Q', 'LocalBuffer', 'I', 'write')": 6442450944.0, - "('Q', 'GlobalBuffer', 'I', 'read')": 6442450944.0, "('Q', 'GlobalBuffer', 'I', 'write')": 0.0, + "('Q', 'GlobalBuffer', 'I', 'read')": 6442450944.0, "('Q', 'LocalBuffer', 'Q', 'read')": 38654705664.0, "('Q', 'LocalBuffer', 'Q', 'write')": 38654705664.0, - "('Q', 'MainMemory', 'Q', 'read')": 0.0, "('Q', 'MainMemory', 'Q', 'write')": 402653184.0, + "('Q', 'MainMemory', 'Q', 'read')": 0.0, "('Q', 'Register', 'WQ', 'read')": 4947802324992.0, "('Q', 'Register', 'WQ', 'write')": 2415919104.0, - "('Q', 'MainMemory', 'WQ', 'read')": 2415919104.0, "('Q', 'MainMemory', 'WQ', 'write')": 0.0, + "('Q', 'MainMemory', 'WQ', 'read')": 2415919104.0, "('Q', 'MAC', 'None', 'compute')": 309237645312.0, "('QK', 'Register', 'K', 'read')": 824633720832.0, "('QK', 'Register', 'K', 'write')": 805306368.0, - "('QK', 'GlobalBuffer', 'K', 'read')": 805306368.0, "('QK', 'GlobalBuffer', 'K', 'write')": 0.0, + "('QK', 'GlobalBuffer', 'K', 'read')": 805306368.0, "('QK', 'LocalBuffer', 'Q', 'read')": 6442450944.0, "('QK', 'LocalBuffer', 'Q', 'write')": 402653184.0, - "('QK', 'MainMemory', 'Q', 'read')": 402653184.0, "('QK', 'MainMemory', 'Q', 'write')": 0.0, + "('QK', 'MainMemory', 'Q', 'read')": 402653184.0, "('QK', 'LocalBuffer', 'QK', 'read')": 6442450944.0, "('QK', 'LocalBuffer', 'QK', 'write')": 6442450944.0, "('QK', 'GlobalBuffer', 'QK', 'read')": 0.0, @@ -5461,29 +5481,29 @@ "('QK_softmax', 'ScalarUnit', 'None', 'compute')": 402653184.0, "('AV', 'LocalBuffer', 'AV', 'read')": 6442450944.0, "('AV', 'LocalBuffer', 'AV', 'write')": 6442450944.0, - "('AV', 'GlobalBuffer', 'AV', 'read')": 0.0, "('AV', 'GlobalBuffer', 'AV', 'write')": 805306368.0, + "('AV', 'GlobalBuffer', 'AV', 'read')": 0.0, "('AV', 'LocalBuffer', 'QK_softmax', 'read')": 6442450944.0, "('AV', 'LocalBuffer', 'QK_softmax', 'write')": 6442450944.0, "('AV', 'GlobalBuffer', 'QK_softmax', 'read')": 6442450944.0, "('AV', 'GlobalBuffer', 'QK_softmax', 'write')": 0.0, "('AV', 'Register', 'V', 'read')": 824633720832.0, "('AV', 'Register', 'V', 'write')": 402653184.0, - "('AV', 'MainMemory', 'V', 'read')": 402653184.0, "('AV', 'MainMemory', 'V', 'write')": 0.0, + "('AV', 'MainMemory', 'V', 'read')": 402653184.0, "('AV', 'MAC', 'None', 'compute')": 51539607552.0, "('Z', 'LocalBuffer', 'AV', 'read')": 38654705664.0, "('Z', 'LocalBuffer', 'AV', 'write')": 6442450944.0, - "('Z', 'GlobalBuffer', 'AV', 'read')": 6442450944.0, "('Z', 'GlobalBuffer', 'AV', 'write')": 0.0, + "('Z', 'GlobalBuffer', 'AV', 'read')": 6442450944.0, "('Z', 'Register', 'WZ', 'read')": 4947802324992.0, "('Z', 'Register', 'WZ', 'write')": 2415919104.0, - "('Z', 'MainMemory', 'WZ', 'read')": 2415919104.0, "('Z', 'MainMemory', 'WZ', 'write')": 0.0, + "('Z', 'MainMemory', 'WZ', 'read')": 2415919104.0, "('Z', 'LocalBuffer', 'Z', 'read')": 38654705664.0, "('Z', 'LocalBuffer', 'Z', 'write')": 38654705664.0, - "('Z', 'GlobalBuffer', 'Z', 'read')": 0.0, "('Z', 'GlobalBuffer', 'Z', 'write')": 402653184.0, + "('Z', 'GlobalBuffer', 'Z', 'read')": 0.0, "('Z', 'MAC', 'None', 'compute')": 309237645312.0, "('FFA', 'LocalBuffer', 'FFA', 'read')": 154618822656.0, "('FFA', 'LocalBuffer', 'FFA', 'write')": 154618822656.0, @@ -5491,12 +5511,12 @@ "('FFA', 'GlobalBuffer', 'FFA', 'write')": 1610612736.0, "('FFA', 'Register', 'WFFA', 'read')": 19791209299968.0, "('FFA', 'Register', 'WFFA', 'write')": 9663676416.0, - "('FFA', 'MainMemory', 'WFFA', 'read')": 9663676416.0, "('FFA', 'MainMemory', 'WFFA', 'write')": 0.0, + "('FFA', 'MainMemory', 'WFFA', 'read')": 9663676416.0, "('FFA', 'LocalBuffer', 'Z', 'read')": 154618822656.0, "('FFA', 'LocalBuffer', 'Z', 'write')": 25769803776.0, - "('FFA', 'GlobalBuffer', 'Z', 'read')": 25769803776.0, "('FFA', 'GlobalBuffer', 'Z', 'write')": 0.0, + "('FFA', 'GlobalBuffer', 'Z', 'read')": 25769803776.0, "('FFA', 'MAC', 'None', 'compute')": 1236950581248.0, "('FFB', 'LocalBuffer', 'FFA', 'read')": 154618822656.0, "('FFB', 'LocalBuffer', 'FFA', 'write')": 25769803776.0, @@ -5504,21 +5524,21 @@ "('FFB', 'GlobalBuffer', 'FFA', 'write')": 0.0, "('FFB', 'LocalBuffer', 'FFB', 'read')": 157437394944.0, "('FFB', 'LocalBuffer', 'FFB', 'write')": 157437394944.0, - "('FFB', 'GlobalBuffer', 'FFB', 'read')": 3221225472.0, "('FFB', 'GlobalBuffer', 'FFB', 'write')": 3221225472.0, - "('FFB', 'MainMemory', 'FFB', 'read')": 0.0, + "('FFB', 'GlobalBuffer', 'FFB', 'read')": 3221225472.0, "('FFB', 'MainMemory', 'FFB', 'write')": 402653184.0, + "('FFB', 'MainMemory', 'FFB', 'read')": 0.0, "('FFB', 'Register', 'WFFB', 'read')": 19791209299968.0, "('FFB', 'Register', 'WFFB', 'write')": 9663676416.0, - "('FFB', 'MainMemory', 'WFFB', 'read')": 9663676416.0, "('FFB', 'MainMemory', 'WFFB', 'write')": 0.0, + "('FFB', 'MainMemory', 'WFFB', 'read')": 9663676416.0, "('FFB', 'MAC', 'None', 'compute')": 1236950581248.0 }, "n_mappings": 1.0 }, "tpu_v4i|gpt3_175B|BATCH_SIZE=1,DECODE=False,N_NEW_TOKENS=2048|unfused": { "energy": 1.3358087807345111, - "latency": 0.059500956907868385, + "latency": 0.0595009568106928, "energy_per_component": { "('I', 'MainMemory', 'leak')": 0.0, "('I', 'GlobalBuffer', 'leak')": 0.0, @@ -5528,8 +5548,8 @@ "('I', 'MAC', 'leak')": 0.0, "('V', 'LocalBuffer', 'read')": 0.019250042736530304, "('V', 'LocalBuffer', 'write')": 0.013213466852903366, - "('V', 'GlobalBuffer', 'read')": 0.01211180817335844, "('V', 'GlobalBuffer', 'write')": 0.0009502615430392325, + "('V', 'GlobalBuffer', 'read')": 0.01211180817335844, "('V', 'MainMemory', 'read')": 0.019814563184639998, "('V', 'MainMemory', 'write')": 0.00283065188352, "('V', 'Register', 'read')": 0.0, @@ -5543,8 +5563,8 @@ "('V', 'MAC', 'leak')": 0.0, "('K', 'LocalBuffer', 'read')": 0.019250042736530304, "('K', 'LocalBuffer', 'write')": 0.013213466852903366, - "('K', 'GlobalBuffer', 'read')": 0.01211180817335844, "('K', 'GlobalBuffer', 'write')": 0.0009502615430392325, + "('K', 'GlobalBuffer', 'read')": 0.01211180817335844, "('K', 'MainMemory', 'read')": 0.019814563184639998, "('K', 'MainMemory', 'write')": 0.00283065188352, "('K', 'Register', 'read')": 0.0, @@ -5558,8 +5578,8 @@ "('K', 'MAC', 'leak')": 0.0, "('Q', 'LocalBuffer', 'read')": 0.019250042736530304, "('Q', 'LocalBuffer', 'write')": 0.013213466852903366, - "('Q', 'GlobalBuffer', 'read')": 0.01211180817335844, "('Q', 'GlobalBuffer', 'write')": 0.0009502615430392325, + "('Q', 'GlobalBuffer', 'read')": 0.01211180817335844, "('Q', 'MainMemory', 'read')": 0.019814563184639998, "('Q', 'MainMemory', 'write')": 0.00283065188352, "('Q', 'Register', 'read')": 0.0, @@ -5610,8 +5630,8 @@ "('AV', 'MAC', 'leak')": 0.0, "('Z', 'LocalBuffer', 'read')": 0.019250042736530304, "('Z', 'LocalBuffer', 'write')": 0.013213466852903366, - "('Z', 'GlobalBuffer', 'read')": 0.01211180817335844, "('Z', 'GlobalBuffer', 'write')": 0.0009502615430392325, + "('Z', 'GlobalBuffer', 'read')": 0.01211180817335844, "('Z', 'MainMemory', 'read')": 0.019814563184639998, "('Z', 'Register', 'read')": 0.0, "('Z', 'Register', 'write')": 0.0, @@ -5629,8 +5649,8 @@ "('FFA', 'Register', 'read')": 0.0, "('FFA', 'Register', 'write')": 0.0, "('FFA', 'MainMemory', 'read')": 0.07076629708799999, - "('FFA', 'GlobalBuffer', 'read')": 0.04844723269343376, "('FFA', 'GlobalBuffer', 'write')": 0.0009502615430392325, + "('FFA', 'GlobalBuffer', 'read')": 0.04844723269343376, "('FFA', 'MAC', 'compute')": 0.103903848824832, "('FFA', 'MainMemory', 'leak')": 0.0, "('FFA', 'GlobalBuffer', 'leak')": 0.0, @@ -5657,185 +5677,191 @@ "latency_per_component": { "('I', 'MainMemory')": 0.0, "('I', 'ScalarUnit')": 0.0, - "('V', 'MainMemory')": 0.0006557869492098689, - "('V', 'GlobalBuffer')": 0.00039321600343100727, "('V', 'LocalBuffer')": 0.0, + "('V', 'GlobalBuffer (read)')": 0.00039321600343100727, + "('V', 'GlobalBuffer (write)')": 4.9152e-05, + "('V', 'MainMemory')": 0.0006557869446254072, "('V', 'Register')": 0.0, "('V', 'MAC')": 0.0044938973151147366, - "('K', 'MainMemory')": 0.0006557869492098689, - "('K', 'GlobalBuffer')": 0.00039321600343100727, "('K', 'LocalBuffer')": 0.0, + "('K', 'GlobalBuffer (read)')": 0.00039321600343100727, + "('K', 'GlobalBuffer (write)')": 4.9152e-05, + "('K', 'MainMemory')": 0.0006557869446254072, "('K', 'Register')": 0.0, "('K', 'MAC')": 0.0044938973151147366, - "('Q', 'MainMemory')": 0.0006557869492098689, - "('Q', 'GlobalBuffer')": 0.00039321600343100727, "('Q', 'LocalBuffer')": 0.0, + "('Q', 'GlobalBuffer (read)')": 0.00039321600343100727, + "('Q', 'GlobalBuffer (write)')": 4.9152e-05, + "('Q', 'MainMemory')": 0.0006557869446254072, "('Q', 'Register')": 0.0, "('Q', 'MAC')": 0.0044938973151147366, - "('QK', 'MainMemory')": 0.0014755206648260355, - "('QK', 'LocalBuffer')": 0.0, "('QK', 'Register')": 0.0, + "('QK', 'MainMemory')": 0.0014755206254071663, + "('QK', 'LocalBuffer')": 0.0, "('QK', 'MAC')": 0.0007489828858524561, - "('QK_softmax', 'MainMemory')": 0.0026231477968394756, "('QK_softmax', 'LocalBuffer')": 0.0, + "('QK_softmax', 'MainMemory')": 0.0026231477785016288, "('QK_softmax', 'ScalarUnit')": 0.0007489828858524561, - "('AV', 'MainMemory')": 0.0014755206648260355, "('AV', 'LocalBuffer')": 0.0, + "('AV', 'MainMemory')": 0.0014755206254071663, "('AV', 'Register')": 0.0, "('AV', 'MAC')": 0.0007489828858524561, - "('Z', 'MainMemory')": 0.0006557869492098689, - "('Z', 'GlobalBuffer')": 0.00039321600343100727, "('Z', 'LocalBuffer')": 0.0, + "('Z', 'GlobalBuffer (read)')": 0.00039321600343100727, + "('Z', 'GlobalBuffer (write)')": 4.9152e-05, + "('Z', 'MainMemory')": 0.0006557869446254072, "('Z', 'Register')": 0.0, "('Z', 'MAC')": 0.0044938973151147366, - "('FFA', 'MainMemory')": 0.002377227647230029, - "('FFA', 'GlobalBuffer')": 0.001572864013724029, "('FFA', 'LocalBuffer')": 0.0, + "('FFA', 'MainMemory')": 0.0023772276742671013, "('FFA', 'Register')": 0.0, + "('FFA', 'GlobalBuffer (read)')": 0.001572864013724029, + "('FFA', 'GlobalBuffer (write)')": 4.9152e-05, "('FFA', 'MAC')": 0.017975589260458946, - "('FFB', 'MainMemory')": 0.002377227647230029, - "('FFB', 'GlobalBuffer')": 0.0016465920489281416, "('FFB', 'LocalBuffer')": 0.0, + "('FFB', 'GlobalBuffer (read)')": 0.0016465920489281416, + "('FFB', 'GlobalBuffer (write)')": 0.00034406399936415255, + "('FFB', 'MainMemory')": 0.002377227674267101, "('FFB', 'Register')": 0.0, "('FFB', 'MAC')": 0.017975589260458946 }, "actions": { - "('I', 'MainMemory', 'I_in', 'read')": 0.0, "('I', 'MainMemory', 'I_in', 'write')": 0.0, + "('I', 'MainMemory', 'I_in', 'read')": 0.0, "('I', 'ScalarUnit', 'None', 'compute')": 0.0, "('V', 'LocalBuffer', 'I', 'read')": 38654705664.0, "('V', 'LocalBuffer', 'I', 'write')": 6442450944.0, - "('V', 'GlobalBuffer', 'I', 'read')": 6442450944.0, "('V', 'GlobalBuffer', 'I', 'write')": 402653184.0, - "('V', 'MainMemory', 'I', 'read')": 402653184.0, + "('V', 'GlobalBuffer', 'I', 'read')": 6442450944.0, "('V', 'MainMemory', 'I', 'write')": 0.0, + "('V', 'MainMemory', 'I', 'read')": 402653184.0, "('V', 'LocalBuffer', 'V', 'read')": 38654705664.0, "('V', 'LocalBuffer', 'V', 'write')": 38654705664.0, - "('V', 'MainMemory', 'V', 'read')": 0.0, "('V', 'MainMemory', 'V', 'write')": 402653184.0, + "('V', 'MainMemory', 'V', 'read')": 0.0, "('V', 'Register', 'WV', 'read')": 4947802324992.0, "('V', 'Register', 'WV', 'write')": 2415919104.0, - "('V', 'MainMemory', 'WV', 'read')": 2415919104.0, "('V', 'MainMemory', 'WV', 'write')": 0.0, + "('V', 'MainMemory', 'WV', 'read')": 2415919104.0, "('V', 'MAC', 'None', 'compute')": 309237645312.0, "('K', 'LocalBuffer', 'I', 'read')": 38654705664.0, "('K', 'LocalBuffer', 'I', 'write')": 6442450944.0, - "('K', 'GlobalBuffer', 'I', 'read')": 6442450944.0, "('K', 'GlobalBuffer', 'I', 'write')": 402653184.0, - "('K', 'MainMemory', 'I', 'read')": 402653184.0, + "('K', 'GlobalBuffer', 'I', 'read')": 6442450944.0, "('K', 'MainMemory', 'I', 'write')": 0.0, + "('K', 'MainMemory', 'I', 'read')": 402653184.0, "('K', 'LocalBuffer', 'K', 'read')": 38654705664.0, "('K', 'LocalBuffer', 'K', 'write')": 38654705664.0, - "('K', 'MainMemory', 'K', 'read')": 0.0, "('K', 'MainMemory', 'K', 'write')": 402653184.0, + "('K', 'MainMemory', 'K', 'read')": 0.0, "('K', 'Register', 'WK', 'read')": 4947802324992.0, "('K', 'Register', 'WK', 'write')": 2415919104.0, - "('K', 'MainMemory', 'WK', 'read')": 2415919104.0, "('K', 'MainMemory', 'WK', 'write')": 0.0, + "('K', 'MainMemory', 'WK', 'read')": 2415919104.0, "('K', 'MAC', 'None', 'compute')": 309237645312.0, "('Q', 'LocalBuffer', 'I', 'read')": 38654705664.0, "('Q', 'LocalBuffer', 'I', 'write')": 6442450944.0, - "('Q', 'GlobalBuffer', 'I', 'read')": 6442450944.0, "('Q', 'GlobalBuffer', 'I', 'write')": 402653184.0, - "('Q', 'MainMemory', 'I', 'read')": 402653184.0, + "('Q', 'GlobalBuffer', 'I', 'read')": 6442450944.0, "('Q', 'MainMemory', 'I', 'write')": 0.0, + "('Q', 'MainMemory', 'I', 'read')": 402653184.0, "('Q', 'LocalBuffer', 'Q', 'read')": 38654705664.0, "('Q', 'LocalBuffer', 'Q', 'write')": 38654705664.0, - "('Q', 'MainMemory', 'Q', 'read')": 0.0, "('Q', 'MainMemory', 'Q', 'write')": 402653184.0, + "('Q', 'MainMemory', 'Q', 'read')": 0.0, "('Q', 'Register', 'WQ', 'read')": 4947802324992.0, "('Q', 'Register', 'WQ', 'write')": 2415919104.0, - "('Q', 'MainMemory', 'WQ', 'read')": 2415919104.0, "('Q', 'MainMemory', 'WQ', 'write')": 0.0, + "('Q', 'MainMemory', 'WQ', 'read')": 2415919104.0, "('Q', 'MAC', 'None', 'compute')": 309237645312.0, "('QK', 'Register', 'K', 'read')": 824633720832.0, "('QK', 'Register', 'K', 'write')": 402653184.0, - "('QK', 'MainMemory', 'K', 'read')": 402653184.0, "('QK', 'MainMemory', 'K', 'write')": 0.0, + "('QK', 'MainMemory', 'K', 'read')": 402653184.0, "('QK', 'LocalBuffer', 'Q', 'read')": 6442450944.0, "('QK', 'LocalBuffer', 'Q', 'write')": 402653184.0, - "('QK', 'MainMemory', 'Q', 'read')": 402653184.0, "('QK', 'MainMemory', 'Q', 'write')": 0.0, + "('QK', 'MainMemory', 'Q', 'read')": 402653184.0, "('QK', 'LocalBuffer', 'QK', 'read')": 6442450944.0, "('QK', 'LocalBuffer', 'QK', 'write')": 6442450944.0, - "('QK', 'MainMemory', 'QK', 'read')": 0.0, "('QK', 'MainMemory', 'QK', 'write')": 6442450944.0, + "('QK', 'MainMemory', 'QK', 'read')": 0.0, "('QK', 'MAC', 'None', 'compute')": 51539607552.0, "('QK_softmax', 'LocalBuffer', 'QK', 'read')": 6442450944.0, "('QK_softmax', 'LocalBuffer', 'QK', 'write')": 6442450944.0, - "('QK_softmax', 'MainMemory', 'QK', 'read')": 6442450944.0, "('QK_softmax', 'MainMemory', 'QK', 'write')": 0.0, + "('QK_softmax', 'MainMemory', 'QK', 'read')": 6442450944.0, "('QK_softmax', 'LocalBuffer', 'QK_softmax', 'read')": 6442450944.0, "('QK_softmax', 'LocalBuffer', 'QK_softmax', 'write')": 6442450944.0, - "('QK_softmax', 'MainMemory', 'QK_softmax', 'read')": 0.0, "('QK_softmax', 'MainMemory', 'QK_softmax', 'write')": 6442450944.0, + "('QK_softmax', 'MainMemory', 'QK_softmax', 'read')": 0.0, "('QK_softmax', 'ScalarUnit', 'None', 'compute')": 402653184.0, "('AV', 'LocalBuffer', 'AV', 'read')": 6442450944.0, "('AV', 'LocalBuffer', 'AV', 'write')": 6442450944.0, - "('AV', 'MainMemory', 'AV', 'read')": 0.0, "('AV', 'MainMemory', 'AV', 'write')": 402653184.0, + "('AV', 'MainMemory', 'AV', 'read')": 0.0, "('AV', 'LocalBuffer', 'QK_softmax', 'read')": 6442450944.0, "('AV', 'LocalBuffer', 'QK_softmax', 'write')": 6442450944.0, - "('AV', 'MainMemory', 'QK_softmax', 'read')": 6442450944.0, "('AV', 'MainMemory', 'QK_softmax', 'write')": 0.0, + "('AV', 'MainMemory', 'QK_softmax', 'read')": 6442450944.0, "('AV', 'Register', 'V', 'read')": 824633720832.0, "('AV', 'Register', 'V', 'write')": 402653184.0, - "('AV', 'MainMemory', 'V', 'read')": 402653184.0, "('AV', 'MainMemory', 'V', 'write')": 0.0, + "('AV', 'MainMemory', 'V', 'read')": 402653184.0, "('AV', 'MAC', 'None', 'compute')": 51539607552.0, "('Z', 'LocalBuffer', 'AV', 'read')": 38654705664.0, "('Z', 'LocalBuffer', 'AV', 'write')": 6442450944.0, - "('Z', 'GlobalBuffer', 'AV', 'read')": 6442450944.0, "('Z', 'GlobalBuffer', 'AV', 'write')": 402653184.0, - "('Z', 'MainMemory', 'AV', 'read')": 402653184.0, + "('Z', 'GlobalBuffer', 'AV', 'read')": 6442450944.0, "('Z', 'MainMemory', 'AV', 'write')": 0.0, + "('Z', 'MainMemory', 'AV', 'read')": 402653184.0, "('Z', 'Register', 'WZ', 'read')": 4947802324992.0, "('Z', 'Register', 'WZ', 'write')": 2415919104.0, - "('Z', 'MainMemory', 'WZ', 'read')": 2415919104.0, "('Z', 'MainMemory', 'WZ', 'write')": 0.0, + "('Z', 'MainMemory', 'WZ', 'read')": 2415919104.0, "('Z', 'LocalBuffer', 'Z', 'read')": 38654705664.0, "('Z', 'LocalBuffer', 'Z', 'write')": 38654705664.0, - "('Z', 'MainMemory', 'Z', 'read')": 0.0, "('Z', 'MainMemory', 'Z', 'write')": 402653184.0, + "('Z', 'MainMemory', 'Z', 'read')": 0.0, "('Z', 'MAC', 'None', 'compute')": 309237645312.0, "('FFA', 'LocalBuffer', 'FFA', 'read')": 154618822656.0, "('FFA', 'LocalBuffer', 'FFA', 'write')": 154618822656.0, - "('FFA', 'MainMemory', 'FFA', 'read')": 0.0, "('FFA', 'MainMemory', 'FFA', 'write')": 1610612736.0, + "('FFA', 'MainMemory', 'FFA', 'read')": 0.0, "('FFA', 'Register', 'WFFA', 'read')": 19791209299968.0, "('FFA', 'Register', 'WFFA', 'write')": 9663676416.0, - "('FFA', 'MainMemory', 'WFFA', 'read')": 9663676416.0, "('FFA', 'MainMemory', 'WFFA', 'write')": 0.0, + "('FFA', 'MainMemory', 'WFFA', 'read')": 9663676416.0, "('FFA', 'LocalBuffer', 'Z', 'read')": 154618822656.0, "('FFA', 'LocalBuffer', 'Z', 'write')": 25769803776.0, - "('FFA', 'GlobalBuffer', 'Z', 'read')": 25769803776.0, "('FFA', 'GlobalBuffer', 'Z', 'write')": 402653184.0, - "('FFA', 'MainMemory', 'Z', 'read')": 402653184.0, + "('FFA', 'GlobalBuffer', 'Z', 'read')": 25769803776.0, "('FFA', 'MainMemory', 'Z', 'write')": 0.0, + "('FFA', 'MainMemory', 'Z', 'read')": 402653184.0, "('FFA', 'MAC', 'None', 'compute')": 1236950581248.0, "('FFB', 'LocalBuffer', 'FFA', 'read')": 154618822656.0, "('FFB', 'LocalBuffer', 'FFA', 'write')": 25769803776.0, "('FFB', 'GlobalBuffer', 'FFA', 'read')": 25769803776.0, "('FFB', 'GlobalBuffer', 'FFA', 'write')": 1610612736.0, - "('FFB', 'MainMemory', 'FFA', 'read')": 1610612736.0, "('FFB', 'MainMemory', 'FFA', 'write')": 0.0, + "('FFB', 'MainMemory', 'FFA', 'read')": 1610612736.0, "('FFB', 'LocalBuffer', 'FFB', 'read')": 155424129024.0, "('FFB', 'LocalBuffer', 'FFB', 'write')": 155424129024.0, - "('FFB', 'GlobalBuffer', 'FFB', 'read')": 1207959552.0, "('FFB', 'GlobalBuffer', 'FFB', 'write')": 1207959552.0, - "('FFB', 'MainMemory', 'FFB', 'read')": 0.0, + "('FFB', 'GlobalBuffer', 'FFB', 'read')": 1207959552.0, "('FFB', 'MainMemory', 'FFB', 'write')": 402653184.0, + "('FFB', 'MainMemory', 'FFB', 'read')": 0.0, "('FFB', 'Register', 'WFFB', 'read')": 19791209299968.0, "('FFB', 'Register', 'WFFB', 'write')": 9663676416.0, - "('FFB', 'MainMemory', 'WFFB', 'read')": 9663676416.0, "('FFB', 'MainMemory', 'WFFB', 'write')": 0.0, + "('FFB', 'MainMemory', 'WFFB', 'read')": 9663676416.0, "('FFB', 'MAC', 'None', 'compute')": 1236950581248.0 }, "n_mappings": 1.0 }, "tpu_v4i|gpt3_175B|BATCH_SIZE=1,DECODE=True,N_CACHED_TOKENS=2047,N_NEW_TOKENS=1|fused": { "energy": 0.20985938712185517, - "latency": 0.006066492758691311, + "latency": 0.00606649310240228, "energy_per_component": { "('I', 'GlobalBuffer', 'read')": 3.6962304e-07, "('I', 'MainMemory', 'write')": 1.38215424e-06, @@ -5971,106 +5997,116 @@ "('FFB', 'MAC', 'leak')": 0.0 }, "latency_per_component": { - "('I', 'MainMemory')": 4.002605891173516e-08, - "('I', 'GlobalBuffer')": 1.2000000104706032e-08, + "('I', 'GlobalBuffer (read)')": 1.2e-08, + "('I', 'GlobalBuffer (write)')": 0.0, + "('I', 'MainMemory')": 4.0026058631921826e-08, "('I', 'ScalarUnit')": 0.0, - "('V', 'MainMemory')": 0.0004918802296742797, - "('V', 'GlobalBuffer')": 4.800000041882413e-08, "('V', 'LocalBuffer')": 0.0, + "('V', 'GlobalBuffer (read)')": 4.800000041882413e-08, + "('V', 'GlobalBuffer (write)')": 0.0, + "('V', 'MainMemory')": 0.0004918802345276873, "('V', 'Register')": 0.0, "('V', 'MAC')": 2.1942857983958675e-06, - "('K', 'MainMemory')": 0.0004918802296742797, - "('K', 'GlobalBuffer')": 4.800000041882413e-08, "('K', 'LocalBuffer')": 0.0, + "('K', 'GlobalBuffer (read)')": 4.800000041882413e-08, + "('K', 'GlobalBuffer (write)')": 0.0, + "('K', 'MainMemory')": 0.0004918802345276873, "('K', 'Register')": 0.0, "('K', 'MAC')": 2.1942857983958675e-06, - "('Q', 'MainMemory')": 0.0004918401828035712, - "('Q', 'GlobalBuffer')": 4.800000041882413e-08, "('Q', 'LocalBuffer')": 0.0, + "('Q', 'GlobalBuffer (read)')": 2.4000000209412065e-08, + "('Q', 'GlobalBuffer (write)')": 4.800000041882413e-08, "('Q', 'Register')": 0.0, + "('Q', 'MainMemory')": 0.0004918402084690554, "('Q', 'MAC')": 2.1942857983958675e-06, - "('QK', 'MainMemory')": 8.193334360839799e-05, - "('QK', 'GlobalBuffer')": 3.8381250533348066e-07, - "('QK', 'LocalBuffer')": 0.0, "('QK', 'Register')": 0.0, + "('QK', 'MainMemory')": 8.193334201954397e-05, + "('QK', 'LocalBuffer')": 0.0, + "('QK', 'GlobalBuffer (read)')": 1.2e-08, + "('QK', 'GlobalBuffer (write)')": 3.838125e-07, "('QK', 'MAC')": 5.257143129711039e-07, - "('QK_softmax', 'GlobalBuffer')": 3.8381250533348066e-07, "('QK_softmax', 'LocalBuffer')": 0.0, + "('QK_softmax', 'GlobalBuffer (read)')": 1.9190625e-07, + "('QK_softmax', 'GlobalBuffer (write)')": 3.838125e-07, "('QK_softmax', 'ScalarUnit')": 3.6553572613229335e-07, - "('AV', 'MainMemory')": 8.193334360839799e-05, - "('AV', 'GlobalBuffer')": 1.9190625266674033e-07, "('AV', 'LocalBuffer')": 0.0, + "('AV', 'GlobalBuffer (read)')": 1.9190625e-07, + "('AV', 'GlobalBuffer (write)')": 2.4e-08, "('AV', 'Register')": 0.0, + "('AV', 'MainMemory')": 8.193334201954397e-05, "('AV', 'MAC')": 5.257143129711039e-07, - "('Z', 'MainMemory')": 0.0004918401828035712, - "('Z', 'GlobalBuffer')": 4.800000041882413e-08, "('Z', 'LocalBuffer')": 0.0, + "('Z', 'GlobalBuffer (read)')": 2.4000000209412065e-08, + "('Z', 'GlobalBuffer (write)')": 4.800000041882413e-08, "('Z', 'Register')": 0.0, + "('Z', 'MainMemory')": 0.0004918402084690554, "('Z', 'MAC')": 2.1942857983958675e-06, - "('FFA', 'MainMemory')": 0.001967360731214285, - "('FFA', 'GlobalBuffer')": 9.600000083764826e-08, "('FFA', 'LocalBuffer')": 0.0, + "('FFA', 'GlobalBuffer (read)')": 4.800000041882413e-08, + "('FFA', 'GlobalBuffer (write)')": 9.6e-08, "('FFA', 'Register')": 0.0, + "('FFA', 'MainMemory')": 0.0019673608338762216, "('FFA', 'MAC')": 8.77714319358347e-06, - "('FFB', 'MainMemory')": 0.0019674007780849934, - "('FFB', 'GlobalBuffer')": 9.600000083764826e-08, "('FFB', 'LocalBuffer')": 0.0, + "('FFB', 'GlobalBuffer (read)')": 5.99999978589949e-08, + "('FFB', 'GlobalBuffer (write)')": 9.600000083764826e-08, + "('FFB', 'MainMemory')": 0.0019674008599348536, "('FFB', 'Register')": 0.0, "('FFB', 'MAC')": 8.77714319358347e-06 }, "actions": { - "('I', 'GlobalBuffer', 'I_in', 'read')": 196608.0, "('I', 'GlobalBuffer', 'I_in', 'write')": 0.0, - "('I', 'MainMemory', 'I_in', 'read')": 0.0, + "('I', 'GlobalBuffer', 'I_in', 'read')": 196608.0, "('I', 'MainMemory', 'I_in', 'write')": 196608.0, + "('I', 'MainMemory', 'I_in', 'read')": 0.0, "('I', 'ScalarUnit', 'None', 'compute')": 0.0, "('V', 'LocalBuffer', 'I', 'read')": 18874368.0, "('V', 'LocalBuffer', 'I', 'write')": 786432.0, - "('V', 'GlobalBuffer', 'I', 'read')": 786432.0, "('V', 'GlobalBuffer', 'I', 'write')": 0.0, + "('V', 'GlobalBuffer', 'I', 'read')": 786432.0, "('V', 'LocalBuffer', 'V', 'read')": 18874368.0, "('V', 'LocalBuffer', 'V', 'write')": 18874368.0, - "('V', 'MainMemory', 'V', 'read')": 0.0, "('V', 'MainMemory', 'V', 'write')": 196608.0, + "('V', 'MainMemory', 'V', 'read')": 0.0, "('V', 'Register', 'WV', 'read')": 2415919104.0, "('V', 'Register', 'WV', 'write')": 2415919104.0, - "('V', 'MainMemory', 'WV', 'read')": 2415919104.0, "('V', 'MainMemory', 'WV', 'write')": 0.0, + "('V', 'MainMemory', 'WV', 'read')": 2415919104.0, "('V', 'MAC', 'None', 'compute')": 150994944.0, "('K', 'LocalBuffer', 'I', 'read')": 18874368.0, "('K', 'LocalBuffer', 'I', 'write')": 786432.0, - "('K', 'GlobalBuffer', 'I', 'read')": 786432.0, "('K', 'GlobalBuffer', 'I', 'write')": 0.0, + "('K', 'GlobalBuffer', 'I', 'read')": 786432.0, "('K', 'LocalBuffer', 'K', 'read')": 18874368.0, "('K', 'LocalBuffer', 'K', 'write')": 18874368.0, - "('K', 'MainMemory', 'K', 'read')": 0.0, "('K', 'MainMemory', 'K', 'write')": 196608.0, + "('K', 'MainMemory', 'K', 'read')": 0.0, "('K', 'Register', 'WK', 'read')": 2415919104.0, "('K', 'Register', 'WK', 'write')": 2415919104.0, - "('K', 'MainMemory', 'WK', 'read')": 2415919104.0, "('K', 'MainMemory', 'WK', 'write')": 0.0, + "('K', 'MainMemory', 'WK', 'read')": 2415919104.0, "('K', 'MAC', 'None', 'compute')": 150994944.0, "('Q', 'LocalBuffer', 'I', 'read')": 18874368.0, "('Q', 'LocalBuffer', 'I', 'write')": 393216.0, - "('Q', 'GlobalBuffer', 'I', 'read')": 393216.0, "('Q', 'GlobalBuffer', 'I', 'write')": 0.0, + "('Q', 'GlobalBuffer', 'I', 'read')": 393216.0, "('Q', 'LocalBuffer', 'Q', 'read')": 18874368.0, "('Q', 'LocalBuffer', 'Q', 'write')": 18874368.0, - "('Q', 'GlobalBuffer', 'Q', 'read')": 0.0, "('Q', 'GlobalBuffer', 'Q', 'write')": 393216.0, + "('Q', 'GlobalBuffer', 'Q', 'read')": 0.0, "('Q', 'Register', 'WQ', 'read')": 2415919104.0, "('Q', 'Register', 'WQ', 'write')": 2415919104.0, - "('Q', 'MainMemory', 'WQ', 'read')": 2415919104.0, "('Q', 'MainMemory', 'WQ', 'write')": 0.0, + "('Q', 'MainMemory', 'WQ', 'read')": 2415919104.0, "('Q', 'MAC', 'None', 'compute')": 150994944.0, "('QK', 'Register', 'K_cache', 'read')": 402456576.0, "('QK', 'Register', 'K_cache', 'write')": 402456576.0, - "('QK', 'MainMemory', 'K_cache', 'read')": 402456576.0, "('QK', 'MainMemory', 'K_cache', 'write')": 0.0, + "('QK', 'MainMemory', 'K_cache', 'read')": 402456576.0, "('QK', 'LocalBuffer', 'Q', 'read')": 4521984.0, "('QK', 'LocalBuffer', 'Q', 'write')": 196608.0, - "('QK', 'GlobalBuffer', 'Q', 'read')": 196608.0, "('QK', 'GlobalBuffer', 'Q', 'write')": 0.0, + "('QK', 'GlobalBuffer', 'Q', 'read')": 196608.0, "('QK', 'LocalBuffer', 'QK', 'read')": 3144192.0, "('QK', 'LocalBuffer', 'QK', 'write')": 3144192.0, "('QK', 'GlobalBuffer', 'QK', 'read')": 0.0, @@ -6082,69 +6118,69 @@ "('QK_softmax', 'GlobalBuffer', 'QK', 'write')": 0.0, "('QK_softmax', 'LocalBuffer', 'QK_softmax', 'read')": 3144192.0, "('QK_softmax', 'LocalBuffer', 'QK_softmax', 'write')": 3144192.0, - "('QK_softmax', 'GlobalBuffer', 'QK_softmax', 'read')": 0.0, "('QK_softmax', 'GlobalBuffer', 'QK_softmax', 'write')": 3144192.0, + "('QK_softmax', 'GlobalBuffer', 'QK_softmax', 'read')": 0.0, "('QK_softmax', 'ScalarUnit', 'None', 'compute')": 196512.0, "('AV', 'LocalBuffer', 'AV', 'read')": 4521984.0, "('AV', 'LocalBuffer', 'AV', 'write')": 4521984.0, - "('AV', 'GlobalBuffer', 'AV', 'read')": 0.0, "('AV', 'GlobalBuffer', 'AV', 'write')": 196608.0, + "('AV', 'GlobalBuffer', 'AV', 'read')": 0.0, "('AV', 'LocalBuffer', 'QK_softmax', 'read')": 3144192.0, "('AV', 'LocalBuffer', 'QK_softmax', 'write')": 3144192.0, - "('AV', 'GlobalBuffer', 'QK_softmax', 'read')": 3144192.0, "('AV', 'GlobalBuffer', 'QK_softmax', 'write')": 0.0, + "('AV', 'GlobalBuffer', 'QK_softmax', 'read')": 3144192.0, "('AV', 'Register', 'V_cache', 'read')": 402456576.0, "('AV', 'Register', 'V_cache', 'write')": 402456576.0, - "('AV', 'MainMemory', 'V_cache', 'read')": 402456576.0, "('AV', 'MainMemory', 'V_cache', 'write')": 0.0, + "('AV', 'MainMemory', 'V_cache', 'read')": 402456576.0, "('AV', 'MAC', 'None', 'compute')": 25153536.0, "('Z', 'LocalBuffer', 'AV', 'read')": 18874368.0, "('Z', 'LocalBuffer', 'AV', 'write')": 393216.0, - "('Z', 'GlobalBuffer', 'AV', 'read')": 393216.0, "('Z', 'GlobalBuffer', 'AV', 'write')": 0.0, + "('Z', 'GlobalBuffer', 'AV', 'read')": 393216.0, "('Z', 'Register', 'WZ', 'read')": 2415919104.0, "('Z', 'Register', 'WZ', 'write')": 2415919104.0, - "('Z', 'MainMemory', 'WZ', 'read')": 2415919104.0, "('Z', 'MainMemory', 'WZ', 'write')": 0.0, + "('Z', 'MainMemory', 'WZ', 'read')": 2415919104.0, "('Z', 'LocalBuffer', 'Z', 'read')": 18874368.0, "('Z', 'LocalBuffer', 'Z', 'write')": 18874368.0, - "('Z', 'GlobalBuffer', 'Z', 'read')": 0.0, "('Z', 'GlobalBuffer', 'Z', 'write')": 393216.0, + "('Z', 'GlobalBuffer', 'Z', 'read')": 0.0, "('Z', 'MAC', 'None', 'compute')": 150994944.0, "('FFA', 'LocalBuffer', 'FFA', 'read')": 75497472.0, "('FFA', 'LocalBuffer', 'FFA', 'write')": 75497472.0, - "('FFA', 'GlobalBuffer', 'FFA', 'read')": 0.0, "('FFA', 'GlobalBuffer', 'FFA', 'write')": 786432.0, + "('FFA', 'GlobalBuffer', 'FFA', 'read')": 0.0, "('FFA', 'Register', 'WFFA', 'read')": 9663676416.0, "('FFA', 'Register', 'WFFA', 'write')": 9663676416.0, - "('FFA', 'MainMemory', 'WFFA', 'read')": 9663676416.0, "('FFA', 'MainMemory', 'WFFA', 'write')": 0.0, + "('FFA', 'MainMemory', 'WFFA', 'read')": 9663676416.0, "('FFA', 'LocalBuffer', 'Z', 'read')": 75497472.0, "('FFA', 'LocalBuffer', 'Z', 'write')": 786432.0, - "('FFA', 'GlobalBuffer', 'Z', 'read')": 786432.0, "('FFA', 'GlobalBuffer', 'Z', 'write')": 0.0, + "('FFA', 'GlobalBuffer', 'Z', 'read')": 786432.0, "('FFA', 'MAC', 'None', 'compute')": 603979776.0, "('FFB', 'LocalBuffer', 'FFA', 'read')": 75497472.0, "('FFB', 'LocalBuffer', 'FFA', 'write')": 786432.0, - "('FFB', 'GlobalBuffer', 'FFA', 'read')": 786432.0, "('FFB', 'GlobalBuffer', 'FFA', 'write')": 0.0, + "('FFB', 'GlobalBuffer', 'FFA', 'read')": 786432.0, "('FFB', 'LocalBuffer', 'FFB', 'read')": 75497472.0, "('FFB', 'LocalBuffer', 'FFB', 'write')": 75497472.0, - "('FFB', 'GlobalBuffer', 'FFB', 'read')": 196608.0, "('FFB', 'GlobalBuffer', 'FFB', 'write')": 786432.0, - "('FFB', 'MainMemory', 'FFB', 'read')": 0.0, + "('FFB', 'GlobalBuffer', 'FFB', 'read')": 196608.0, "('FFB', 'MainMemory', 'FFB', 'write')": 196608.0, + "('FFB', 'MainMemory', 'FFB', 'read')": 0.0, "('FFB', 'Register', 'WFFB', 'read')": 9663676416.0, "('FFB', 'Register', 'WFFB', 'write')": 9663676416.0, - "('FFB', 'MainMemory', 'WFFB', 'read')": 9663676416.0, "('FFB', 'MainMemory', 'WFFB', 'write')": 0.0, + "('FFB', 'MainMemory', 'WFFB', 'read')": 9663676416.0, "('FFB', 'MAC', 'None', 'compute')": 603979776.0 }, "n_mappings": 1.0 }, "tpu_v4i|gpt3_175B|BATCH_SIZE=1,DECODE=True,N_CACHED_TOKENS=2047,N_NEW_TOKENS=1|unfused": { "energy": 0.20994088871961328, - "latency": 0.006069310187854171, + "latency": 0.006069310123778501, "energy_per_component": { "('I', 'MainMemory', 'leak')": 0.0, "('I', 'GlobalBuffer', 'leak')": 0.0, @@ -6154,8 +6190,8 @@ "('I', 'MAC', 'leak')": 0.0, "('V', 'LocalBuffer', 'read')": 9.399434929946437e-06, "('V', 'LocalBuffer', 'write')": 5.760614385508234e-06, - "('V', 'GlobalBuffer', 'read')": 1.478492208661919e-06, "('V', 'GlobalBuffer', 'write')": 4.6399489406212524e-07, + "('V', 'GlobalBuffer', 'read')": 1.478492208661919e-06, "('V', 'MainMemory', 'read')": 0.016985293455359998, "('V', 'MainMemory', 'write')": 1.38215424e-06, "('V', 'Register', 'read')": 0.0, @@ -6169,8 +6205,8 @@ "('V', 'MAC', 'leak')": 0.0, "('K', 'LocalBuffer', 'read')": 9.399434929946437e-06, "('K', 'LocalBuffer', 'write')": 5.760614385508234e-06, - "('K', 'GlobalBuffer', 'read')": 1.478492208661919e-06, "('K', 'GlobalBuffer', 'write')": 4.6399489406212524e-07, + "('K', 'GlobalBuffer', 'read')": 1.478492208661919e-06, "('K', 'MainMemory', 'read')": 0.016985293455359998, "('K', 'MainMemory', 'write')": 1.38215424e-06, "('K', 'Register', 'read')": 0.0, @@ -6184,8 +6220,8 @@ "('K', 'MAC', 'leak')": 0.0, "('Q', 'LocalBuffer', 'read')": 9.399434929946437e-06, "('Q', 'LocalBuffer', 'write')": 5.760614385508234e-06, - "('Q', 'GlobalBuffer', 'read')": 1.478492208661919e-06, "('Q', 'GlobalBuffer', 'write')": 4.6399489406212524e-07, + "('Q', 'GlobalBuffer', 'read')": 1.478492208661919e-06, "('Q', 'MainMemory', 'read')": 0.016985293455359998, "('Q', 'MainMemory', 'write')": 1.38215424e-06, "('Q', 'Register', 'read')": 0.0, @@ -6236,8 +6272,8 @@ "('AV', 'MAC', 'leak')": 0.0, "('Z', 'LocalBuffer', 'read')": 9.399434929946437e-06, "('Z', 'LocalBuffer', 'write')": 5.760614385508234e-06, - "('Z', 'GlobalBuffer', 'read')": 1.478492208661919e-06, "('Z', 'GlobalBuffer', 'write')": 4.6399489406212524e-07, + "('Z', 'GlobalBuffer', 'read')": 1.478492208661919e-06, "('Z', 'MainMemory', 'read')": 0.016985293455359998, "('Z', 'Register', 'read')": 0.0, "('Z', 'Register', 'write')": 0.0, @@ -6255,8 +6291,8 @@ "('FFA', 'Register', 'read')": 0.0, "('FFA', 'Register', 'write')": 0.0, "('FFA', 'MainMemory', 'read')": 0.06793702735872, - "('FFA', 'GlobalBuffer', 'read')": 1.478492208661919e-06, "('FFA', 'GlobalBuffer', 'write')": 4.6399489406212524e-07, + "('FFA', 'GlobalBuffer', 'read')": 1.478492208661919e-06, "('FFA', 'MAC', 'compute')": 5.0734301184e-05, "('FFA', 'MainMemory', 'leak')": 0.0, "('FFA', 'GlobalBuffer', 'leak')": 0.0, @@ -6267,8 +6303,8 @@ "('FFB', 'LocalBuffer', 'read')": 3.759773971978575e-05, "('FFB', 'LocalBuffer', 'write')": 2.2351183361024596e-05, "('FFB', 'MainMemory', 'read')": 0.06794117382143999, - "('FFB', 'GlobalBuffer', 'read')": 3.6962305216547975e-07, "('FFB', 'GlobalBuffer', 'write')": 1.855979576248501e-06, + "('FFB', 'GlobalBuffer', 'read')": 3.6962305216547975e-07, "('FFB', 'MainMemory', 'write')": 1.38215424e-06, "('FFB', 'Register', 'read')": 0.0, "('FFB', 'Register', 'write')": 0.0, @@ -6283,176 +6319,182 @@ "latency_per_component": { "('I', 'MainMemory')": 0.0, "('I', 'ScalarUnit')": 0.0, - "('V', 'MainMemory')": 0.0004919202765449882, - "('V', 'GlobalBuffer')": 4.800000041882413e-08, "('V', 'LocalBuffer')": 0.0, + "('V', 'GlobalBuffer (read)')": 4.800000041882413e-08, + "('V', 'GlobalBuffer (write)')": 2.4e-08, + "('V', 'MainMemory')": 0.0004919202605863192, "('V', 'Register')": 0.0, "('V', 'MAC')": 2.1942857983958675e-06, - "('K', 'MainMemory')": 0.0004919202765449882, - "('K', 'GlobalBuffer')": 4.800000041882413e-08, "('K', 'LocalBuffer')": 0.0, + "('K', 'GlobalBuffer (read)')": 4.800000041882413e-08, + "('K', 'GlobalBuffer (write)')": 2.4e-08, + "('K', 'MainMemory')": 0.0004919202605863192, "('K', 'Register')": 0.0, "('K', 'MAC')": 2.1942857983958675e-06, - "('Q', 'MainMemory')": 0.0004919202765449882, - "('Q', 'GlobalBuffer')": 4.800000041882413e-08, "('Q', 'LocalBuffer')": 0.0, + "('Q', 'GlobalBuffer (read)')": 4.800000041882413e-08, + "('Q', 'GlobalBuffer (write)')": 2.4e-08, + "('Q', 'MainMemory')": 0.0004919202605863192, "('Q', 'Register')": 0.0, "('Q', 'MAC')": 2.1942857983958675e-06, - "('QK', 'MainMemory')": 8.26134710223414e-05, - "('QK', 'LocalBuffer')": 0.0, "('QK', 'Register')": 0.0, + "('QK', 'MainMemory')": 8.261347231270359e-05, + "('QK', 'LocalBuffer')": 0.0, "('QK', 'MAC')": 5.257143129711039e-07, - "('QK_softmax', 'MainMemory')": 1.2802084938812186e-06, "('QK_softmax', 'LocalBuffer')": 0.0, + "('QK_softmax', 'MainMemory')": 1.2802084690553745e-06, "('QK_softmax', 'ScalarUnit')": 3.6553572613229335e-07, - "('AV', 'MainMemory')": 8.26134710223414e-05, "('AV', 'LocalBuffer')": 0.0, + "('AV', 'MainMemory')": 8.261347231270357e-05, "('AV', 'Register')": 0.0, "('AV', 'MAC')": 5.257143129711039e-07, - "('Z', 'MainMemory')": 0.0004919202765449882, - "('Z', 'GlobalBuffer')": 4.800000041882413e-08, "('Z', 'LocalBuffer')": 0.0, + "('Z', 'GlobalBuffer (read)')": 4.800000041882413e-08, + "('Z', 'GlobalBuffer (write)')": 2.4e-08, + "('Z', 'MainMemory')": 0.0004919202605863192, "('Z', 'Register')": 0.0, "('Z', 'MAC')": 2.1942857983958675e-06, - "('FFA', 'MainMemory')": 0.0019675609655678272, - "('FFA', 'GlobalBuffer')": 4.800000041882413e-08, "('FFA', 'LocalBuffer')": 0.0, + "('FFA', 'MainMemory')": 0.001967560964169381, "('FFA', 'Register')": 0.0, + "('FFA', 'GlobalBuffer (read)')": 4.800000041882413e-08, + "('FFA', 'GlobalBuffer (write)')": 2.4e-08, "('FFA', 'MAC')": 8.77714319358347e-06, - "('FFB', 'MainMemory')": 0.0019675609655678272, - "('FFB', 'GlobalBuffer')": 9.600000083764826e-08, "('FFB', 'LocalBuffer')": 0.0, + "('FFB', 'MainMemory')": 0.001967560964169381, + "('FFB', 'GlobalBuffer (read)')": 1.2000000104706032e-08, + "('FFB', 'GlobalBuffer (write)')": 9.600000083764826e-08, "('FFB', 'Register')": 0.0, "('FFB', 'MAC')": 8.77714319358347e-06 }, "actions": { - "('I', 'MainMemory', 'I_in', 'read')": 0.0, "('I', 'MainMemory', 'I_in', 'write')": 0.0, + "('I', 'MainMemory', 'I_in', 'read')": 0.0, "('I', 'ScalarUnit', 'None', 'compute')": 0.0, "('V', 'LocalBuffer', 'I', 'read')": 18874368.0, "('V', 'LocalBuffer', 'I', 'write')": 786432.0, - "('V', 'GlobalBuffer', 'I', 'read')": 786432.0, "('V', 'GlobalBuffer', 'I', 'write')": 196608.0, - "('V', 'MainMemory', 'I', 'read')": 196608.0, + "('V', 'GlobalBuffer', 'I', 'read')": 786432.0, "('V', 'MainMemory', 'I', 'write')": 0.0, + "('V', 'MainMemory', 'I', 'read')": 196608.0, "('V', 'LocalBuffer', 'V', 'read')": 18874368.0, "('V', 'LocalBuffer', 'V', 'write')": 18874368.0, - "('V', 'MainMemory', 'V', 'read')": 0.0, "('V', 'MainMemory', 'V', 'write')": 196608.0, + "('V', 'MainMemory', 'V', 'read')": 0.0, "('V', 'Register', 'WV', 'read')": 2415919104.0, "('V', 'Register', 'WV', 'write')": 2415919104.0, - "('V', 'MainMemory', 'WV', 'read')": 2415919104.0, "('V', 'MainMemory', 'WV', 'write')": 0.0, + "('V', 'MainMemory', 'WV', 'read')": 2415919104.0, "('V', 'MAC', 'None', 'compute')": 150994944.0, "('K', 'LocalBuffer', 'I', 'read')": 18874368.0, "('K', 'LocalBuffer', 'I', 'write')": 786432.0, - "('K', 'GlobalBuffer', 'I', 'read')": 786432.0, "('K', 'GlobalBuffer', 'I', 'write')": 196608.0, - "('K', 'MainMemory', 'I', 'read')": 196608.0, + "('K', 'GlobalBuffer', 'I', 'read')": 786432.0, "('K', 'MainMemory', 'I', 'write')": 0.0, + "('K', 'MainMemory', 'I', 'read')": 196608.0, "('K', 'LocalBuffer', 'K', 'read')": 18874368.0, "('K', 'LocalBuffer', 'K', 'write')": 18874368.0, - "('K', 'MainMemory', 'K', 'read')": 0.0, "('K', 'MainMemory', 'K', 'write')": 196608.0, + "('K', 'MainMemory', 'K', 'read')": 0.0, "('K', 'Register', 'WK', 'read')": 2415919104.0, "('K', 'Register', 'WK', 'write')": 2415919104.0, - "('K', 'MainMemory', 'WK', 'read')": 2415919104.0, "('K', 'MainMemory', 'WK', 'write')": 0.0, + "('K', 'MainMemory', 'WK', 'read')": 2415919104.0, "('K', 'MAC', 'None', 'compute')": 150994944.0, "('Q', 'LocalBuffer', 'I', 'read')": 18874368.0, "('Q', 'LocalBuffer', 'I', 'write')": 786432.0, - "('Q', 'GlobalBuffer', 'I', 'read')": 786432.0, "('Q', 'GlobalBuffer', 'I', 'write')": 196608.0, - "('Q', 'MainMemory', 'I', 'read')": 196608.0, + "('Q', 'GlobalBuffer', 'I', 'read')": 786432.0, "('Q', 'MainMemory', 'I', 'write')": 0.0, + "('Q', 'MainMemory', 'I', 'read')": 196608.0, "('Q', 'LocalBuffer', 'Q', 'read')": 18874368.0, "('Q', 'LocalBuffer', 'Q', 'write')": 18874368.0, - "('Q', 'MainMemory', 'Q', 'read')": 0.0, "('Q', 'MainMemory', 'Q', 'write')": 196608.0, + "('Q', 'MainMemory', 'Q', 'read')": 0.0, "('Q', 'Register', 'WQ', 'read')": 2415919104.0, "('Q', 'Register', 'WQ', 'write')": 2415919104.0, - "('Q', 'MainMemory', 'WQ', 'read')": 2415919104.0, "('Q', 'MainMemory', 'WQ', 'write')": 0.0, + "('Q', 'MainMemory', 'WQ', 'read')": 2415919104.0, "('Q', 'MAC', 'None', 'compute')": 150994944.0, "('QK', 'Register', 'K_cache', 'read')": 402456576.0, "('QK', 'Register', 'K_cache', 'write')": 402456576.0, - "('QK', 'MainMemory', 'K_cache', 'read')": 402456576.0, "('QK', 'MainMemory', 'K_cache', 'write')": 0.0, + "('QK', 'MainMemory', 'K_cache', 'read')": 402456576.0, "('QK', 'LocalBuffer', 'Q', 'read')": 4521984.0, "('QK', 'LocalBuffer', 'Q', 'write')": 196608.0, - "('QK', 'MainMemory', 'Q', 'read')": 196608.0, "('QK', 'MainMemory', 'Q', 'write')": 0.0, + "('QK', 'MainMemory', 'Q', 'read')": 196608.0, "('QK', 'LocalBuffer', 'QK', 'read')": 3144192.0, "('QK', 'LocalBuffer', 'QK', 'write')": 3144192.0, - "('QK', 'MainMemory', 'QK', 'read')": 0.0, "('QK', 'MainMemory', 'QK', 'write')": 3144192.0, + "('QK', 'MainMemory', 'QK', 'read')": 0.0, "('QK', 'MAC', 'None', 'compute')": 25153536.0, "('QK_softmax', 'LocalBuffer', 'QK', 'read')": 3144192.0, "('QK_softmax', 'LocalBuffer', 'QK', 'write')": 3144192.0, - "('QK_softmax', 'MainMemory', 'QK', 'read')": 3144192.0, "('QK_softmax', 'MainMemory', 'QK', 'write')": 0.0, + "('QK_softmax', 'MainMemory', 'QK', 'read')": 3144192.0, "('QK_softmax', 'LocalBuffer', 'QK_softmax', 'read')": 3144192.0, "('QK_softmax', 'LocalBuffer', 'QK_softmax', 'write')": 3144192.0, - "('QK_softmax', 'MainMemory', 'QK_softmax', 'read')": 0.0, "('QK_softmax', 'MainMemory', 'QK_softmax', 'write')": 3144192.0, + "('QK_softmax', 'MainMemory', 'QK_softmax', 'read')": 0.0, "('QK_softmax', 'ScalarUnit', 'None', 'compute')": 196512.0, "('AV', 'LocalBuffer', 'AV', 'read')": 4521984.0, "('AV', 'LocalBuffer', 'AV', 'write')": 4521984.0, - "('AV', 'MainMemory', 'AV', 'read')": 0.0, "('AV', 'MainMemory', 'AV', 'write')": 196608.0, + "('AV', 'MainMemory', 'AV', 'read')": 0.0, "('AV', 'LocalBuffer', 'QK_softmax', 'read')": 3144192.0, "('AV', 'LocalBuffer', 'QK_softmax', 'write')": 3144192.0, - "('AV', 'MainMemory', 'QK_softmax', 'read')": 3144192.0, "('AV', 'MainMemory', 'QK_softmax', 'write')": 0.0, + "('AV', 'MainMemory', 'QK_softmax', 'read')": 3144192.0, "('AV', 'Register', 'V_cache', 'read')": 402456576.0, "('AV', 'Register', 'V_cache', 'write')": 402456576.0, - "('AV', 'MainMemory', 'V_cache', 'read')": 402456576.0, "('AV', 'MainMemory', 'V_cache', 'write')": 0.0, + "('AV', 'MainMemory', 'V_cache', 'read')": 402456576.0, "('AV', 'MAC', 'None', 'compute')": 25153536.0, "('Z', 'LocalBuffer', 'AV', 'read')": 18874368.0, "('Z', 'LocalBuffer', 'AV', 'write')": 786432.0, - "('Z', 'GlobalBuffer', 'AV', 'read')": 786432.0, "('Z', 'GlobalBuffer', 'AV', 'write')": 196608.0, - "('Z', 'MainMemory', 'AV', 'read')": 196608.0, + "('Z', 'GlobalBuffer', 'AV', 'read')": 786432.0, "('Z', 'MainMemory', 'AV', 'write')": 0.0, + "('Z', 'MainMemory', 'AV', 'read')": 196608.0, "('Z', 'Register', 'WZ', 'read')": 2415919104.0, "('Z', 'Register', 'WZ', 'write')": 2415919104.0, - "('Z', 'MainMemory', 'WZ', 'read')": 2415919104.0, "('Z', 'MainMemory', 'WZ', 'write')": 0.0, + "('Z', 'MainMemory', 'WZ', 'read')": 2415919104.0, "('Z', 'LocalBuffer', 'Z', 'read')": 18874368.0, "('Z', 'LocalBuffer', 'Z', 'write')": 18874368.0, - "('Z', 'MainMemory', 'Z', 'read')": 0.0, "('Z', 'MainMemory', 'Z', 'write')": 196608.0, + "('Z', 'MainMemory', 'Z', 'read')": 0.0, "('Z', 'MAC', 'None', 'compute')": 150994944.0, "('FFA', 'LocalBuffer', 'FFA', 'read')": 75497472.0, "('FFA', 'LocalBuffer', 'FFA', 'write')": 75497472.0, - "('FFA', 'MainMemory', 'FFA', 'read')": 0.0, "('FFA', 'MainMemory', 'FFA', 'write')": 786432.0, + "('FFA', 'MainMemory', 'FFA', 'read')": 0.0, "('FFA', 'Register', 'WFFA', 'read')": 9663676416.0, "('FFA', 'Register', 'WFFA', 'write')": 9663676416.0, - "('FFA', 'MainMemory', 'WFFA', 'read')": 9663676416.0, "('FFA', 'MainMemory', 'WFFA', 'write')": 0.0, + "('FFA', 'MainMemory', 'WFFA', 'read')": 9663676416.0, "('FFA', 'LocalBuffer', 'Z', 'read')": 75497472.0, "('FFA', 'LocalBuffer', 'Z', 'write')": 786432.0, - "('FFA', 'GlobalBuffer', 'Z', 'read')": 786432.0, "('FFA', 'GlobalBuffer', 'Z', 'write')": 196608.0, - "('FFA', 'MainMemory', 'Z', 'read')": 196608.0, + "('FFA', 'GlobalBuffer', 'Z', 'read')": 786432.0, "('FFA', 'MainMemory', 'Z', 'write')": 0.0, + "('FFA', 'MainMemory', 'Z', 'read')": 196608.0, "('FFA', 'MAC', 'None', 'compute')": 603979776.0, "('FFB', 'LocalBuffer', 'FFA', 'read')": 75497472.0, "('FFB', 'LocalBuffer', 'FFA', 'write')": 786432.0, - "('FFB', 'MainMemory', 'FFA', 'read')": 786432.0, "('FFB', 'MainMemory', 'FFA', 'write')": 0.0, + "('FFB', 'MainMemory', 'FFA', 'read')": 786432.0, "('FFB', 'LocalBuffer', 'FFB', 'read')": 75497472.0, "('FFB', 'LocalBuffer', 'FFB', 'write')": 75497472.0, - "('FFB', 'GlobalBuffer', 'FFB', 'read')": 196608.0, "('FFB', 'GlobalBuffer', 'FFB', 'write')": 786432.0, - "('FFB', 'MainMemory', 'FFB', 'read')": 0.0, + "('FFB', 'GlobalBuffer', 'FFB', 'read')": 196608.0, "('FFB', 'MainMemory', 'FFB', 'write')": 196608.0, + "('FFB', 'MainMemory', 'FFB', 'read')": 0.0, "('FFB', 'Register', 'WFFB', 'read')": 9663676416.0, "('FFB', 'Register', 'WFFB', 'write')": 9663676416.0, - "('FFB', 'MainMemory', 'WFFB', 'read')": 9663676416.0, "('FFB', 'MainMemory', 'WFFB', 'write')": 0.0, + "('FFB', 'MainMemory', 'WFFB', 'read')": 9663676416.0, "('FFB', 'MAC', 'None', 'compute')": 603979776.0 }, "n_mappings": 1.0 diff --git a/tests/test_latency.py b/tests/test_latency.py index c9790c8f..dd136124 100644 --- a/tests/test_latency.py +++ b/tests/test_latency.py @@ -7,7 +7,10 @@ Einsum does 64 MACs and each tensor is 128 bits. """ +import copy import unittest + +import pandas as pd from pathlib import Path import accelforge as af @@ -19,16 +22,20 @@ LATENCY_ARCH = INPUT_FILES_DIR / "latency.arch.yaml" NETWORKED_ARCH = INPUT_FILES_DIR / "networked_latency.arch.yaml" NETWORKED_MAPPING = INPUT_FILES_DIR / "fused_matmuls_to_networked.mapping.yaml" +WEIGHTS_INSIDE_MAPPING = INPUT_FILES_DIR / "fused_matmuls_weights_inside.mapping.yaml" -def total_latency(arch, mapping, n_einsums, **jinja): - spec = Spec.from_yaml( +def make_spec(arch, mapping, n_einsums, **jinja): + return Spec.from_yaml( af.examples.workloads.basic.matmuls, arch, mapping, jinja_parse_data={"N_EINSUMS": n_einsums, "M": 4, "KN": 4, **jinja}, ) - result = evaluate_mapping(spec) + + +def total_latency(arch, mapping, n_einsums, **jinja): + result = evaluate_mapping(make_spec(arch, mapping, n_einsums, **jinja)) return result.data.iloc[0] @@ -107,9 +114,11 @@ def test_fused_memory_bound_overlaps_compute(self): """When a compute-bound Einsum is fused with a memory-bound Einsum, the shared memory's busy time is summed across the Einsums and overlapped with their compute: the total is the max of the two, not the sum of per-Einsum maxes. - The tensors at MainMemory are backed above the shared m loop, so their - reservations are co-resident and their transfers may fill any slice of the - fused execution.""" + A MainMemory transfer runs while the GlobalBuffer reservation on its other + end is alive: the weights' GlobalBuffer staging is above the shared m loop, + so their reads may fill any slice of the fused execution, while T0/T2's + GlobalBuffer reservations live below it, confining that traffic to its own + Einsum. MainMemory's total busy time still bounds the total either way.""" row = total_latency( LATENCY_ARCH, af.examples.mappings.fused_matmuls_to_simple, @@ -130,6 +139,167 @@ def test_fused_memory_bound_overlaps_compute(self): # of Matmul1's traffic. Summing per-Einsum maxes would give 320 + 1408 = 1728. self.assertEqual(row["Totallatency"], 1664) + def test_separate_read_write_ports(self): + """With separate MainMemory ports, reads and writes are independent + components that may overlap: W1's reads no longer serialize behind T2's + writeback, saving their 64 bit-times versus the shared port's 1664.""" + row = total_latency( + LATENCY_ARCH, + af.examples.mappings.fused_matmuls_to_simple, + 2, + MM_READ_THROUGHPUT=1, + MM_WRITE_THROUGHPUT=0.1, + COMPUTE_THROUGHPUT=0.2, + MM_SEPARATE_PORTS=True, + ) + self.assertEqual( + row["Matmul0component_latencyMainMemory (read)"], 256 + ) + self.assertEqual( + row["Matmul1component_latencyMainMemory (read)"], 128 + ) + self.assertEqual( + row["Matmul1component_latencyMainMemory (write)"], 1280 + ) + # T2's writes go to its GlobalBuffer staging below the fused m loop, so + # the write port's 1280 is confined to Matmul1's block and sets its + # width; the shareable reads (256 + 128 = 384) and the MACs (320 + 320) + # hide beneath it. 320 + 1280 = 1600. + self.assertEqual(row["Totallatency"], 1600) + + def test_transfer_needs_deeper_reservation(self): + """With every GlobalBuffer staging below the fused m loop, MainMemory + transfers can only run during their own Einsum's per-iteration windows, so + each Einsum's MainMemory traffic is confined to its block and the overlap + credit above is lost. Matmul0 is compute-bound (320 vs reading T0 once and + W0 per m iteration: (128 + 4 x 128) / 4 = 160); Matmul1's writeback still + dominates its block (128 + 1280 = 1408). 320 + 1408 = 1728, though + MainMemory is only busy for 160 + 1408 = 1568 of it.""" + row = total_latency( + LATENCY_ARCH, + WEIGHTS_INSIDE_MAPPING, + 2, + MM_READ_THROUGHPUT=4, + MM_WRITE_THROUGHPUT=0.1, + COMPUTE_THROUGHPUT=0.2, + ) + self.assertEqual(row["Totallatency"], 1728) + + +class TestLatencyTimeline(unittest.TestCase): + """_latency_timeline asserts internally that its total matches the model's + Totallatency, so each call here also cross-checks the layout math.""" + + def test_timeline_totals(self): + from accelforge.plotting.latency import _latency_timeline + + scenarios = [ + (LATENCY_ARCH, af.examples.mappings.fused_matmuls_to_simple, 1, 463), + (LATENCY_ARCH, af.examples.mappings.fused_matmuls_to_simple, 2, 725), + (NETWORKED_ARCH, NETWORKED_MAPPING, 2, 5000), + ] + jinja = { + "MM_READ_LATENCY": 100, + "MM_WRITE_LATENCY": 200, + "GB_READ_LATENCY": 10, + "GB_WRITE_LATENCY": 20, + "COMPUTE_LATENCY": 3, + "MAC_TILE": 1, + "HOP_LATENCY": 100, + } + for arch, mapping, n_einsums, expected in scenarios: + blocks, _, total = _latency_timeline( + make_spec(arch, mapping, n_einsums, **jinja) + ) + self.assertEqual(total, expected) + self.assertEqual(len(blocks), n_einsums) + + def test_timeline_overlap_layout(self): + """The layout from test_fused_memory_bound_overlaps_compute, bar by bar. + T0/T2 traffic sits at level 1 (their GlobalBuffer reservations live below + the fused m loop, confining it to its own Einsum's block); weight reads sit + at level 0 and may fill slack anywhere. Matmul1's T2 writeback fills its + block, and MainMemory's total busy time (256 + 1408 = 1664) sets the end.""" + from accelforge.plotting.latency import _latency_timeline + + blocks, bars, total = _latency_timeline( + make_spec( + LATENCY_ARCH, + af.examples.mappings.fused_matmuls_to_simple, + 2, + MM_READ_THROUGHPUT=1, + MM_WRITE_THROUGHPUT=0.1, + COMPUTE_THROUGHPUT=0.2, + ) + ) + self.assertEqual(total, 1664) + self.assertEqual( + [(b.einsum, b.start, b.end) for b in blocks], + [("Matmul0", 0, 320), ("Matmul1", 320, 1664)], + ) + mm = [ + (b.einsum, b.level, b.start, b.end) + for b in bars + if b.component == "MainMemory" + ] + self.assertEqual( + sorted(mm), + [ + ("Matmul0", 0, 128, 256), # W0 reads, shareable + ("Matmul0", 1, 0, 128), # T0 reads, private + # W1 reads, shareable; right-aligned to the 1664 deadline + ("Matmul1", 0, 1536, 1664), + ("Matmul1", 1, 320, 1600), # T2 writeback, private + ], + ) + # Compute has its own lane now, and nothing else (communication is zero + # here) is left for the Other lane. + mac = [(b.einsum, b.start, b.end) for b in bars if b.component == "MAC"] + self.assertEqual(mac, [("Matmul0", 0, 320), ("Matmul1", 320, 640)]) + self.assertEqual([b for b in bars if b.component is None], []) + + def test_plot_latency(self): + import matplotlib + + matplotlib.use("Agg") + import matplotlib.pyplot as plt + + from accelforge.plotting.latency import plot_latency + + fig, _ = plot_latency( + make_spec( + LATENCY_ARCH, + af.examples.mappings.fused_matmuls_to_simple, + 2, + MM_READ_THROUGHPUT=1, + MM_WRITE_THROUGHPUT=0.1, + COMPUTE_THROUGHPUT=0.2, + ) + ) + plt.close(fig) + + def test_timeline_matches_mapper_output(self): + """Timelines built from mapper results match the mapper-reported latency + (and, via the internal assert, a fresh model evaluation).""" + from accelforge.plotting.latency import _latency_timeline + + spec = Spec.from_yaml( + af.examples.arches.simple, + af.examples.workloads.basic.matmuls, + jinja_parse_data={"N_EINSUMS": 2, "M": 16, "KN": 16}, + ) + spec.mapper.metrics = af.mapper.Metrics.LATENCY + result = spec.map_workload_to_arch(print_progress=False) + two = copy.copy(result) + two.data = pd.concat([result.data, result.data]) + with self.assertRaises(ValueError): + _latency_timeline(spec, two) + for i in range(len(result.data)): + one = copy.copy(result) + one.data = result.data.iloc[[i]] + _, _, total = _latency_timeline(spec, one) + self.assertEqual(total, result.data["Totallatency"].iloc[i]) + if __name__ == "__main__": unittest.main() From 8e59616b88b66a932d76b964060d60636e24d5ab Mon Sep 17 00:00:00 2001 From: Tanner Andrulis Date: Thu, 6 Aug 2026 15:23:14 -0400 Subject: [PATCH 6/8] Communication latency --- .../FFM/_join_pmappings/join_pmappings.py | 6 +- .../FFM/_join_pmappings/pmapping_dataframe.py | 26 +++- .../make_pmappings_from_templates.py | 68 +++++---- .../make_tile_shapes.py | 5 + .../mapper/FFM/_pareto_df/df_convention.py | 21 +++ accelforge/model/_looptree/latency/memory.py | 134 ++++++++++++------ accelforge/model/run_model.py | 24 ++-- accelforge/plotting/latency.py | 45 +++++- tests/regression_reference.json | 11 +- tests/test_latency.py | 52 ++++--- 10 files changed, 279 insertions(+), 113 deletions(-) diff --git a/accelforge/mapper/FFM/_join_pmappings/join_pmappings.py b/accelforge/mapper/FFM/_join_pmappings/join_pmappings.py index 7313c879..e82786cd 100755 --- a/accelforge/mapper/FFM/_join_pmappings/join_pmappings.py +++ b/accelforge/mapper/FFM/_join_pmappings/join_pmappings.py @@ -131,7 +131,7 @@ def __call__(self, mapping: pd.DataFrame) -> bool: if k not in edp_mapping.columns: nondominated |= True else: - nondominated |= edp_mapping[k] <= v + nondominated |= edp_mapping[k] <= v * (1 + 1e-5) nondominated_by_all &= nondominated if self._pmapping_row_filter_function is not None: @@ -997,6 +997,10 @@ def no_match_lookahead_error( f"Component latency columns were not folded into the total latency: " f"{mappings._make_latencies()}" ) + assert not any(mappings._make_commlatencies().values()), ( + f"Descent/ascent latency columns were not folded into the total latency: " + f"{mappings._make_commlatencies()}" + ) mappings.make_pareto() timer.log_total_time() diff --git a/accelforge/mapper/FFM/_join_pmappings/pmapping_dataframe.py b/accelforge/mapper/FFM/_join_pmappings/pmapping_dataframe.py index 94c366d3..62a198c8 100755 --- a/accelforge/mapper/FFM/_join_pmappings/pmapping_dataframe.py +++ b/accelforge/mapper/FFM/_join_pmappings/pmapping_dataframe.py @@ -226,6 +226,8 @@ def fill_reservation_cols(self, columns: set | str): source = get_latency_or_below(name, nloops + 1, latencies) if source is not None: latency_targets.append((nloops, source, below)) + elif col2commlatency(below) is not None: + pass else: raise ValueError(f"{below} is not a valid reservation column") @@ -288,6 +290,14 @@ def _make_latencies(self) -> dict[str, oset]: assert key.nloops >= -1 return latencies + def _make_commlatencies(self) -> dict[str, oset]: + """Fused loop levels with a "descent" or "ascent" column.""" + levels = {"descent": oset(), "ascent": oset()} + for c in self.data.columns: + if (key := col2commlatency(c)) is not None: + levels[key.direction].add(key.nloops) + return levels + def clear_fused_loop_symbols(self): dropcols = [c for c in self.data.columns if is_fused_loop_col(c)] if not dropcols: @@ -389,6 +399,15 @@ def free_to_loop_index(self, loop_index: int) -> bool: ) drop_columns += [complatency2col(component, l) for l in done] + # Wind-up/down times serialize with the freed window's busy time, so add them + # after the maxes above. + for direction, levels in self._make_commlatencies().items(): + for level in levels: + if level > loop_index: + col = commlatency2col(direction, level) + self.data["Totallatency"] += self.data[col] + drop_columns.append(col) + self._data = self.data.drop(columns=drop_columns) return len(drop_columns) != 0 @@ -679,7 +698,9 @@ def iter_reservations(reservations_dict): if source is not None: add_to_col(df, target, source) - # For everything else: Simple add + # Most other things: Simple add. Descent/ascent wind-up latencies are maxed + # because they can be paid in parallel across all Einsums that are + # descending/ascending together. dropcols = [c for c in df.columns if c.endswith("_RIGHT_MERGE")] for source in dropcols: target = source[: -len("_RIGHT_MERGE")] @@ -687,6 +708,9 @@ def iter_reservations(reservations_dict): continue if col2complatency(target) is not None: continue + if col2commlatency(target) is not None: + max_to_col(df, target, source) + continue if not col_used_in_pareto(target): raise ValueError(f"{target} is not used in pareto") if col2reservation(target) is None: diff --git a/accelforge/mapper/FFM/_make_pmappings/make_pmappings_from_templates/make_pmappings_from_templates.py b/accelforge/mapper/FFM/_make_pmappings/make_pmappings_from_templates/make_pmappings_from_templates.py index 08668ef8..e5dcfb3b 100755 --- a/accelforge/mapper/FFM/_make_pmappings/make_pmappings_from_templates/make_pmappings_from_templates.py +++ b/accelforge/mapper/FFM/_make_pmappings/make_pmappings_from_templates/make_pmappings_from_templates.py @@ -46,6 +46,8 @@ def _imperfect_per_loop(rank_var, mapping) -> tuple[bool, ...]: SameTemplateJobs, ) from accelforge.mapper.FFM._pareto_df.df_convention import ( + col2commlatency, + commlatency2col, is_energy_col, is_fused_loop_col, is_n_iterations_col, @@ -54,41 +56,49 @@ def _imperfect_per_loop(rank_var, mapping) -> tuple[bool, ...]: def shift_reservations_by_null_loop_indices( - mappings: pd.DataFrame, null_loop_indices: set[int] + mappings: pd.DataFrame, null_loop_indices: set[int], n_loops: int = 0 ): - target2newabovename = {} - dropcols = [] + def shift(level): + return level - sum(level > i for i in null_loop_indices) + + non_null_loops = [i for i in range(n_loops) if i not in null_loop_indices] + + # Group columns by the column they land on once the null loops are removed. + target2sources = {} for c in mappings.columns: if is_reservation_col(c): - key, make_col = col2reservation(c), reservation2col + key = col2reservation(c) + target = reservation2col(key.name, shift(key.nloops)) elif (key := col2complatency(c)) is not None: - make_col = complatency2col - else: - continue - above = key.nloops - new_above = above - sum(above > i for i in null_loop_indices) - target = make_col(key.name, new_above) - if target in target2newabovename: - # On a collision, keep the column that includes the other: the deeper one - # for reservations, the shallower one for latencies. - keep_new = above > target2newabovename[target][1] - if not is_reservation_col(c): - keep_new = not keep_new - if keep_new: - dropcols.append(target2newabovename[target][0]) - target2newabovename[target] = (c, above) + target = complatency2col(key.name, shift(key.nloops)) + elif (key := col2commlatency(c)) is not None: + # A wind-up pours into the outermost remaining loop at or below its + # own or, with none remaining, is private and pours into the total. + remaining = [i for i in non_null_loops if i >= key.nloops] + if remaining: + target = commlatency2col(key.direction, shift(min(remaining))) else: - dropcols.append(c) + target = "Totallatency" else: - target2newabovename[target] = (c, above) + continue + target2sources.setdefault(target, []).append((key.nloops, c)) - if dropcols: - drop_set = set(dropcols) - mappings = mappings[[c for c in mappings.columns if c not in drop_set]] renames = {} - for target, (source, _) in target2newabovename.items(): - renames[source] = target + for target, sources in target2sources.items(): + # Wind-ups are serialized --> sum + if col2commlatency(target) is not None or target == "Totallatency": + for _, c in sources: + if c != target: + mappings[target] = mappings.get(target, 0) + mappings.pop(c) + else: + # Reservation/latency columns include one another, so keep the one that + # includes the rest: the deepest reservation (cumulative downward) or the + # shallowest latency (cumulative upward). + _, keep = (max if is_reservation_col(target) else min)(sources) + renames[keep] = target + mappings = mappings.drop(columns=[c for _, c in sources if c != keep]) mappings = mappings.rename(columns=renames) + if len(mappings.columns) != len(mappings.columns.unique()): raise ValueError(f"Duplicate columns: {mappings.columns}") return mappings @@ -412,7 +422,11 @@ def make_pmappings_from_templates( ) for k, v in symbol_renames.items(): mappings[v] = mappings[f"{einsum_name}{k}"] - mappings = shift_reservations_by_null_loop_indices(mappings, null_loop_indices) + mappings = shift_reservations_by_null_loop_indices( + mappings, + null_loop_indices, + n_loops=compatibility.n_loops + len(null_loop_indices), + ) energy_cols = [c for c in mappings.columns if is_energy_col(c)] if (mappings[energy_cols] < 0).any(axis=None): diff --git a/accelforge/mapper/FFM/_make_pmappings/make_pmappings_from_templates/make_tile_shapes.py b/accelforge/mapper/FFM/_make_pmappings/make_pmappings_from_templates/make_tile_shapes.py index 32251dbf..905ca244 100755 --- a/accelforge/mapper/FFM/_make_pmappings/make_pmappings_from_templates/make_tile_shapes.py +++ b/accelforge/mapper/FFM/_make_pmappings/make_pmappings_from_templates/make_tile_shapes.py @@ -24,6 +24,7 @@ from accelforge.mapper.FFM._make_pmappings.pmapper_job import Job from accelforge.mapper.FFM._pareto_df.df_convention import ( SEP, + col2commlatency, col2complatency, col2reservation, is_action_col, @@ -2440,6 +2441,10 @@ def _to_sp(v): # overlapped during joining elif col2complatency(k) is not None: pass + # Descent/ascent latency columns are identical for all pmappings within a + # compatibility + elif col2commlatency(k) is not None: + continue # Handled by reservations above elif col2reservation(k) is not None: continue diff --git a/accelforge/mapper/FFM/_pareto_df/df_convention.py b/accelforge/mapper/FFM/_pareto_df/df_convention.py index 36cf9f9b..4edba1c5 100755 --- a/accelforge/mapper/FFM/_pareto_df/df_convention.py +++ b/accelforge/mapper/FFM/_pareto_df/df_convention.py @@ -133,6 +133,7 @@ def col2energy(colname: str) -> ActionKey | VerboseActionKey: ReservationKey = namedtuple("ReservationKey", ["name", "nloops"]) ComponentLatencyKey = namedtuple("ComponentLatencyKey", ["name", "nloops"]) +CommLatencyKey = namedtuple("CommLatencyKey", ["direction", "nloops"]) @dict_cached @@ -154,6 +155,25 @@ def complatency2col(name: str, nloops: int) -> str: return f"level_latency{name}{nloops}" +@dict_cached +def col2commlatency(x: str) -> CommLatencyKey | None: + """Format: (descent_latency | ascent_latency) nloops. The time it takes for + computation to move down (or up) the LoopTree across fused loops.""" + parts = x.split(SEP) + if len(parts) != 2 or parts[0] not in ("descent_latency", "ascent_latency"): + return None + if not parts[1].isdigit(): # e.g. merge-suffixed copies of the column + return None + return CommLatencyKey(parts[0][: -len("_latency")], int(parts[1])) + + +@dict_cached +def commlatency2col(direction: str, nloops: int) -> str: + """Format: (descent_latency | ascent_latency) nloops""" + assert direction in ("descent", "ascent") + return f"{direction}_latency{SEP}{nloops}" + + @dict_cached def col2reservation(x: str) -> ReservationKey | None: """Format: reservation name nloops left""" @@ -344,6 +364,7 @@ def col_used_in_pareto(c): return ( col2reservation(c) is not None or col2complatency(c) is not None + or col2commlatency(c) is not None or is_objective_col(c) ) diff --git a/accelforge/model/_looptree/latency/memory.py b/accelforge/model/_looptree/latency/memory.py index 78898817..aeafabea 100755 --- a/accelforge/model/_looptree/latency/memory.py +++ b/accelforge/model/_looptree/latency/memory.py @@ -42,66 +42,106 @@ def communication_latency( flattened_arch: FlattenedArch, tensor_to_backing: dict[str, str], output_tensors, -) -> dict[str, object]: + n_fused: int = 0, +) -> tuple[object, dict[str, dict[int, object]]]: """ - Communication latency per component: the time to move one tile of each tensor to - that level. Inputs travel from their backing storage down; outputs are produced - after the worst input reaches compute, then travel from compute up. Each level on - the path charges one call of each action it performs for the tensor; each network - on the path charges its longest route. The whole path repeats once per temporal - iteration above the backing storage. Returns, for each component, the worst - communication latency over all tensors. + Communication latency, split at the fused loops. + + Each tensor's path is a chain of connection latencies, one into each storage. + Connections that cross loops are paid for each loop iteration, and are returned as + descent (inputs moving down) or ascent (outputs moving up) latencies. These are + multiplied by the number of iterations above the destination. + + The rest of each path is lumped into the Einsum delay: the slowest input winds down + to compute, a computation runs, and the result winds up to the slowest output, all + repeating once per iteration of the deepest backing storage. + + Returns (Einsum delay, {"descent" | "ascent": {level: latency}}). """ name2component = {n.name: n for n in flattened_arch} name2index = {n.name: i for i, n in enumerate(flattened_arch)} + compute = flattened_arch[-1] + + def connection_latencies(stats): + """ + Pay latency for one action from every storage node. Once a buffet stats + (reservation) is available, it can immediately receive data from above & start + sending data to below. Per-level latency assumes that we can pre-send to a level + below once this level is available, assuming that it arrives at the next level + as soon as that level is available. + """ + stats = sorted(stats, key=lambda c: name2index[c[0]]) + connections = [] + connection_latency = 0 + for i, (level, s) in enumerate(stats): + if isinstance(s, NetworkStats): + connection_latency += ( + s.max_hops * name2component[level].actions["hop"].latency + ) + elif isinstance(s, BuffetStats): + actions = name2component[level].actions + for action, count in s.net_total_actions().items(): + if not isinstance(count, Number) or count != 0: + connection_latency += actions[action].latency + connections.append((level, s, connection_latency)) + connection_latency = 0 + else: + raise ValueError(f"Unknown stats type: {type(s)}") + return connections, connection_latency tensor_stats = {} for b, s in reuse.buffet_stats.items(): - tensor_stats.setdefault(b.tensor, []).append((b.level, s)) + if b.level != compute.name: + tensor_stats.setdefault(b.tensor, []).append((b.level, s)) for n, s in reuse.network_stats.items(): tensor_stats.setdefault(n.tensor, []).append((n.component, s)) - communication = defaultdict(list) - worst_input_to_compute = 0 - # Inputs first: outputs build on the worst input's arrival at compute. - tensor_order = sorted(tensor_stats, key=lambda t: t in output_tensors) - for tensor in tensor_order: - stats = tensor_stats[tensor] + per_level_latencies = {"descent": defaultdict(list), "ascent": defaultdict(list)} + slowest_input, slowest_output = 0, 0 + backing = [] + + for tensor, stats in tensor_stats.items(): is_output = tensor in output_tensors - stats.sort(key=lambda c: name2index[c[0]], reverse=is_output) - backing = tensor_to_backing[tensor] - - cur_latency = 0 - iterations = None - tensor_latency = {} - for level, stats in stats: - if isinstance(stats, BuffetStats): - if level == backing: - iterations = stats.iterations_above - cur_latency += sum( - name2component[level].actions[action].latency - for action, count in stats.total_actions.items() - if not isinstance(count, Number) or count != 0 - ) - tensor_latency[level] = cur_latency - elif isinstance(stats, NetworkStats): - cur_latency += ( - stats.max_hops * name2component[level].actions["hop"].latency - ) - tensor_latency[level] = cur_latency + cur_latencies = per_level_latencies["ascent" if is_output else "descent"] + + connections, into_compute = connection_latencies(stats) + cumulative_latency = into_compute + for level, s, connection_latency in connections: + if level == tensor_to_backing.get(tensor): + backing.append(s) + + # < n_fused -> track each level independently + if s.n_loops_above < n_fused: + nloops = max(0, s.n_loops_above) + cur_latencies[nloops].append(connection_latency * s.iterations_above) + + # >= n_fused -> lump all levels together. We can start once our + # longest-latency input arrives at compute, does a computation, and then + # goes back up to the longest-latency output. else: - raise ValueError(f"Unknown stats type: {type(stats)}") - - assert iterations is not None, f"Tensor {tensor} has no backing storage" - start = worst_input_to_compute if is_output else 0 - for level in tensor_latency: - communication[level].append(start + tensor_latency[level] * iterations) - if not is_output: - worst_input_to_compute = max_nonzero( - worst_input_to_compute, cur_latency * iterations - ) + cumulative_latency += connection_latency - return {level: max_nonzero(*vals) for level, vals in communication.items()} + if is_output: + slowest_output = max_nonzero(slowest_output, cumulative_latency) + else: + slowest_input = max_nonzero(slowest_input, cumulative_latency) + + compute_latency = compute.actions["compute"].latency + if backing: + deepest_backing = max(backing, key=lambda s: s.n_loops_above) + fused_iterations = deepest_backing.iterations_above + else: + fused_iterations = 1 + einsum_delay = slowest_input + slowest_output + compute_latency + einsum_delay *= fused_iterations + + return ( + einsum_delay, + { + direction: {level: max_nonzero(*vals) for level, vals in per.items()} + for direction, per in per_level_latencies.items() + }, + ) def component_latency( diff --git a/accelforge/model/run_model.py b/accelforge/model/run_model.py index 3a2ef247..588da045 100644 --- a/accelforge/model/run_model.py +++ b/accelforge/model/run_model.py @@ -20,6 +20,7 @@ memory_usage2col, reservation2col, complatency2col, + commlatency2col, tensor2col, action2col, energy2col, @@ -246,24 +247,29 @@ def run_model( for level, cur_latency in level_latency.items(): df[complatency2col(component, level)] = cur_latency * n_instances - # Total latency of each component is its own total_latency plus the worst - # communication latency to reach it. - comm_latency = communication_latency( + # The total latency is the sum of the Einsum's wind-up/down delay and the + # slowest component's busy time. Fused-loop-crossing descent and ascent + # latencies get their own columns because they may be overlapped with other + # Einsums. + einsum_delay, ascent_descent_latency = communication_latency( reuse, job.flattened_arch, tensor_to_backing, workload.einsums[job.einsum_name].output_tensor_names, + n_fused=n_shared_loops, ) + for direction, per_level in ascent_descent_latency.items(): + for level, cur_latency in per_level.items(): + df[commlatency2col(direction, level)] = cur_latency * n_instances per_component_total = [] - for component in oset(latency) | oset(comm_latency): - l = comm_latency.get(component, 0) + for component, l in latency.items(): if component not in latency_per_level: - l += latency.get(component, 0) - if not isinstance(l, Number) or l != 0: - per_component_total.append(l) + if not isinstance(l, Number) or l != 0: + per_component_total.append(l) - df["Totallatency"] = max_nonzero(*per_component_total) * n_instances + slowest = max_nonzero(*per_component_total) + df["Totallatency"] = (slowest + einsum_delay) * n_instances # ================================================================================= # Energy diff --git a/accelforge/plotting/latency.py b/accelforge/plotting/latency.py index cc1a9b15..15e0115d 100644 --- a/accelforge/plotting/latency.py +++ b/accelforge/plotting/latency.py @@ -8,7 +8,10 @@ import matplotlib.pyplot as plt from accelforge.mapper.FFM import Mappings -from accelforge.mapper.FFM._pareto_df.df_convention import col2complatency +from accelforge.mapper.FFM._pareto_df.df_convention import ( + col2commlatency, + col2complatency, +) from accelforge.util import oset _Block = namedtuple("_Block", ["einsum", "start", "end"]) @@ -58,14 +61,19 @@ def _latency_timeline( privates = [] inclusives = [] excls = [] + comms = [] # per Einsum: {(direction, level): wind-up/down latency} for group in groups: row = group.mappings.data.iloc[0] privates.append(row["Totallatency"]) inclusive = defaultdict(dict) + comm = {} for col in group.mappings.data.columns: if (key := col2complatency(col)) is not None: inclusive[key.name][key.nloops] = row[col] + elif (key := col2commlatency(col)) is not None: + comm[key] = row[col] inclusives.append(inclusive) + comms.append(comm) excl = {} for component, by_level in inclusive.items(): levels = sorted(by_level) @@ -114,14 +122,18 @@ def at_or_below(cols, level): total = 0.0 ends = [] + winds = [] # per block: wind-up/down added at its end pool = defaultdict(dict) # component -> {level: latency summed across Einsums} + comm_pool = {} # (direction, level) -> wind-up/down maxed across Einsums for i in range(n): width = privates[i] for cols in inclusives[i].values(): done = [l for l in cols if l > keeps[i]] if done: width = max(width, cols[min(done)]) - total += width + # Wind-up/down of unshared fused loops serializes with the busy time. + wind = sum(v for (_, l), v in comms[i].items() if l > keeps[i]) + total += width + wind for component, cols in inclusives[i].items(): kept = {l: v for l, v in cols.items() if l <= keeps[i]} mine = pool[component] @@ -129,19 +141,32 @@ def at_or_below(cols, level): l: at_or_below(mine, l) + at_or_below(kept, l) for l in set(mine) | set(kept) } + # Shared fused loops' wind-up/down is maxed across the Einsums filling + # them, then added when the loop is freed. + for key, v in comms[i].items(): + if key[1] <= keeps[i]: + comm_pool[key] = max(comm_pool.get(key, 0), v) for cols in pool.values(): done = [l for l in cols if l > settles[i]] if done: total = max(total, cols[min(done)]) for l in done: del cols[l] + for key in list(comm_pool): + if key[1] > settles[i]: + folded = comm_pool.pop(key) + total += folded + wind += folded ends.append(total) + winds.append(wind) def floor(level, i): return max((ends[j] for j in range(i) if settles[j] < level), default=0.0) def deadline(level, i): - return next((ends[j] for j in range(i, n) if settles[j] < level), ends[-1]) + # Busy time must finish before the wind-up/down at its deadline block's end. + j = next((j for j in range(i, n) if settles[j] < level), n - 1) + return ends[j] - winds[j] busy = defaultdict(list) # component -> sorted (start, end) of placed bars @@ -167,6 +192,9 @@ def place(component, lo, amount, due): bars.append( _Bar(None, None, einsum, block_start, block_start + privates[i], False) ) + # Wind-up/down folded at this block's end serializes after its busy time. + if winds[i] > 0: + bars.append(_Bar(None, None, einsum, ends[i] - winds[i], ends[i], False)) # Latency deeper than keep is private to its Einsum's block; co-resident # latency may fill slack anywhere in its co-residency window (down to the @@ -185,7 +213,9 @@ def place(component, lo, amount, due): component, floor(level, i), amount, deadline(level, i) ) else: - start = place(component, blocks[i].start, amount, ends[i]) + start = place( + component, blocks[i].start, amount, ends[i] - winds[i] + ) bars.append( _Bar(component, level, einsum, start, start + amount, shared) ) @@ -221,6 +251,13 @@ def plot_latency( spec.mapping. ax: The axes to plot on. If not given, creates a new figure and axes. + + Returns + ------- + fig: + The figure containing the plot. + ax: + The axes containing the plot. """ blocks, bars, total = _latency_timeline(spec, mappings) diff --git a/tests/regression_reference.json b/tests/regression_reference.json index a631f76a..ee98ee7f 100644 --- a/tests/regression_reference.json +++ b/tests/regression_reference.json @@ -536,11 +536,14 @@ "('QK_softmax', 'MAC', 'leak')": 0.0, "('AV', 'MainMemory', 'read')": 1207173120.0, "('AV', 'MainMemory', 'write')": 402456576.0, + "('AV', 'GlobalBuffer', 'read')": 0.0, + "('AV', 'GlobalBuffer', 'write')": 0.0, "('AV', 'MAC', 'compute')": 25153536.0, "('AV', 'MainMemory', 'leak')": 0.0, "('AV', 'GlobalBuffer', 'leak')": 0.0, "('AV', 'MAC', 'leak')": 0.0, "('Z', 'MainMemory', 'read')": 7247560704.0, + "('Z', 'GlobalBuffer', 'read')": 0.0, "('Z', 'MainMemory', 'write')": 2415919104.0, "('Z', 'MAC', 'compute')": 150994944.0, "('Z', 'MainMemory', 'leak')": 0.0, @@ -574,8 +577,10 @@ "('QK_softmax', 'MAC')": 196512.0, "('AV', 'MainMemory')": 0.0, "('AV', 'MAC')": 25153536.0, + "('AV', 'GlobalBuffer')": 0.0, "('Z', 'MainMemory')": 0.0, "('Z', 'MAC')": 150994944.0, + "('Z', 'GlobalBuffer')": 0.0, "('FFA', 'MainMemory')": 0.0, "('FFA', 'MAC')": 603979776.0, "('FFB', 'MainMemory')": 0.0, @@ -620,6 +625,8 @@ "('QK_softmax', 'MAC', 'None', 'compute')": 196512.0, "('AV', 'MainMemory', 'AV', 'read')": 402259968.0, "('AV', 'MainMemory', 'AV', 'write')": 402456576.0, + "('AV', 'GlobalBuffer', 'AV', 'read')": 0.0, + "('AV', 'GlobalBuffer', 'AV', 'write')": 0.0, "('AV', 'MainMemory', 'QK_softmax', 'read')": 402456576.0, "('AV', 'MainMemory', 'QK_softmax', 'write')": 0.0, "('AV', 'MainMemory', 'V_cache', 'read')": 402456576.0, @@ -627,6 +634,8 @@ "('AV', 'MAC', 'None', 'compute')": 25153536.0, "('Z', 'MainMemory', 'AV', 'read')": 2415919104.0, "('Z', 'MainMemory', 'AV', 'write')": 0.0, + "('Z', 'GlobalBuffer', 'AV', 'read')": 0.0, + "('Z', 'GlobalBuffer', 'AV', 'write')": 0.0, "('Z', 'MainMemory', 'WZ', 'read')": 2415919104.0, "('Z', 'MainMemory', 'WZ', 'write')": 0.0, "('Z', 'MainMemory', 'Z', 'read')": 2415722496.0, @@ -647,7 +656,7 @@ "('FFB', 'MainMemory', 'WFFB', 'write')": 0.0, "('FFB', 'MAC', 'None', 'compute')": 603979776.0 }, - "n_mappings": 1.0 + "n_mappings": 2.0 }, "simple|gpt3_175B|BATCH_SIZE=1,DECODE=True,N_CACHED_TOKENS=2047,N_NEW_TOKENS=1|unfused": { "energy": 121047390624.0, diff --git a/tests/test_latency.py b/tests/test_latency.py index dd136124..439cd371 100644 --- a/tests/test_latency.py +++ b/tests/test_latency.py @@ -59,17 +59,20 @@ def test_communication_latency_in_total(self): GB_WRITE_LATENCY=20, COMPUTE_LATENCY=3, ) - # The worst input reaches compute in one MainMemory read + one GlobalBuffer - # write + read (100 + 20 + 10 = 130). The output follows with one MAC (3), - # winds back up through the GlobalBuffer (20 + 10 = 30), and is read-modify- - # written at MainMemory (100 + 200 = 300). 130 + 3 + 30 + 300 = 463, which - # dominates the 64-cycle compute steady state. - self.assertEqual(row["Totallatency"], 463) + # Each storage's connection is one of each action it performs. The worst + # input reaches compute in one MainMemory read + one GlobalBuffer write + + # read (100 + 20 + 10 = 130). The output follows: one MAC (3), the + # GlobalBuffer's write + read (30), and the MainMemory write (200; the + # first tile has nothing to read-modify, so its read is skipped). + # 130 + 3 + 30 + 200 = 363, which dominates the 64-cycle compute steady + # state. + self.assertEqual(row["Totallatency"], 363) def test_fused_repays_communication_latency_each_switch(self): - """Fused Einsums exchange their intermediate tensor through the shared buffer - once per shared-loop iteration, and the communication latency of that exchange - is paid on every switch.""" + """The exchange of the intermediate through the shared buffer happens below + the fused loop and is paid on every switch; the weights' fills and the + output's drain cross the fused loop, so they are paid once for the whole + fused group (descent/ascent latency), bounded by the slowest Einsum.""" row = total_latency( LATENCY_ARCH, af.examples.mappings.fused_matmuls_to_simple, @@ -80,12 +83,16 @@ def test_fused_repays_communication_latency_each_switch(self): GB_WRITE_LATENCY=20, COMPUTE_LATENCY=3, ) - # T1 is backed at the GlobalBuffer below the fused m loop, so its wind-down - # repeats for each of the 4 m iterations. Matmul0: worst input 130, then - # (one MAC + GlobalBuffer write + read = 33) x 4 iterations = 262. Matmul1 - # is the unfused 463 from above (T1's inputs wind down 30 x 4 = 120 < 130 - # from MainMemory). 262 + 463 = 725. - self.assertEqual(row["Totallatency"], 725) + # Storages above the fused m loop are reserved once and can pre-send + # ahead of it, so their connections are paid once (descent/ascent, maxed + # across Einsums); storages below repeat for each of the 4 m iterations + # via the Einsum delay. Matmul0: T0's and T1's GlobalBuffer connections + # (write + read = 30 each) plus one MAC: (30 + 3 + 30) x 4 = 252. + # Matmul1 mirrors it with T1's and T2's GlobalBuffer connections: + # (30 + 3 + 30) x 4 = 252. Paid once: the inputs' MainMemory reads + # (descent, max = 100) and T2's MainMemory write (ascent, 200; the first + # tile skips the read-modify-write read). 252 + 252 + 100 + 200 = 804. + self.assertEqual(row["Totallatency"], 804) def test_fused_through_slow_interconnect(self): """Two Einsums fused through networks pay the hop latency of the intermediate @@ -93,11 +100,10 @@ def test_fused_through_slow_interconnect(self): mesh_hops = 4 # All 4 Scratchpad positions are used: 4 hops on the PeArray switch_hops = 1 # The all-to-all MacArray is one hop for any route down = mesh_hops + switch_hops # backing -> compute, and compute -> backing - # Matmul0: inputs wind down from MainMemory once (5 hops), then T1 winds up - # to its GlobalBuffer backing below the fused m loop on every one of the 4 - # iterations: 5 + 5 x 4 = 25. Matmul1 mirrors it: T1 winds down 5 x 4 = 20, - # then T2 winds up to MainMemory once: 20 + 5 = 25. - expected = 2 * (down + down * 4) + # Each Einsum's delay is its slowest input's wind-down plus its output's + # wind-up (5 hops each way), repeated for each of the 4 iterations of the + # fused loop above the intermediate's GlobalBuffer backing. + expected = 2 * (down + down) * 4 for hop_latency in [0, 1, 100]: row = total_latency( NETWORKED_ARCH, @@ -194,9 +200,9 @@ def test_timeline_totals(self): from accelforge.plotting.latency import _latency_timeline scenarios = [ - (LATENCY_ARCH, af.examples.mappings.fused_matmuls_to_simple, 1, 463), - (LATENCY_ARCH, af.examples.mappings.fused_matmuls_to_simple, 2, 725), - (NETWORKED_ARCH, NETWORKED_MAPPING, 2, 5000), + (LATENCY_ARCH, af.examples.mappings.fused_matmuls_to_simple, 1, 363), + (LATENCY_ARCH, af.examples.mappings.fused_matmuls_to_simple, 2, 804), + (NETWORKED_ARCH, NETWORKED_MAPPING, 2, 8000), ] jinja = { "MM_READ_LATENCY": 100, From 9f72995f69838dca482a51f7832a2d3b943340b4 Mon Sep 17 00:00:00 2001 From: Michael Gilbert Date: Mon, 31 Aug 2026 13:49:30 -0400 Subject: [PATCH 7/8] Update tests to use naming convention --- .../model/_looptree/reuse/symbolic/_stats.py | 5 ++ .../mappings/fused_matmuls_to_simple.yaml | 3 - .../fused_matmuls_to_networked.mapping.yaml | 28 ++++----- .../fused_matmuls_weights_inside.mapping.yaml | 25 +++++--- tests/test_latency.py | 62 +++++++++---------- 5 files changed, 68 insertions(+), 55 deletions(-) diff --git a/accelforge/model/_looptree/reuse/symbolic/_stats.py b/accelforge/model/_looptree/reuse/symbolic/_stats.py index 527f7b1c..7148194e 100644 --- a/accelforge/model/_looptree/reuse/symbolic/_stats.py +++ b/accelforge/model/_looptree/reuse/symbolic/_stats.py @@ -228,6 +228,11 @@ def __iadd__(self, other: "BuffetStats") -> "BuffetStats": setattr(self, key, value) return self + def net_total_actions(self, action: str | None = None) -> Any: + if action is not None: + return self.total_actions[action] - self.total_skipped_first_actions[action] + return ActionCounts({a: self.net_total_actions(a) for a in self.total_actions}) + def min_take_zero(self, other: "BuffetStats") -> "BuffetStats": """ Take the smallest value of each stat, or zero if either is zero """ new = copy.copy(self) diff --git a/examples/mappings/fused_matmuls_to_simple.yaml b/examples/mappings/fused_matmuls_to_simple.yaml index 52497656..8d256d32 100644 --- a/examples/mappings/fused_matmuls_to_simple.yaml +++ b/examples/mappings/fused_matmuls_to_simple.yaml @@ -46,9 +46,6 @@ mapping: component: MAC {% endfor %} {% else %} - - !Storage - tensors: [{{tensor_name(0)}}, {{tensor_name(1)}}] - component: GlobalBuffer - !Temporal rank_variable: n{{tensor_name(0)}} tile_shape: 1 diff --git a/tests/input_files/fused_matmuls_to_networked.mapping.yaml b/tests/input_files/fused_matmuls_to_networked.mapping.yaml index 6e915ab0..8d25cf22 100644 --- a/tests/input_files/fused_matmuls_to_networked.mapping.yaml +++ b/tests/input_files/fused_matmuls_to_networked.mapping.yaml @@ -2,10 +2,10 @@ mapping: nodes: - !Storage component: MainMemory - tensors: [T0, T2, W0, W1] + tensors: [I, B, WA, WB] - !Storage component: GlobalBuffer - tensors: [W0, W1] + tensors: [WA, WB] - !Temporal rank_variable: m tile_shape: 1 @@ -15,47 +15,47 @@ mapping: nodes: - !Storage component: GlobalBuffer - tensors: [T0, T1] + tensors: [I, A] - !Spatial - rank_variable: n0 + rank_variable: nI tile_shape: {{ MAC_TILE }} component: Scratchpad name: X - !Storage component: Scratchpad - tensors: [T0, T1, W0] + tensors: [I, A, WA] - !Temporal - rank_variable: n1 + rank_variable: nA tile_shape: 1 - !Spatial - rank_variable: n0 + rank_variable: nI tile_shape: 1 component: MAC name: X - !Compute - einsum: Matmul0 + einsum: MatmulA component: MAC - !Nested nodes: - !Storage component: GlobalBuffer - tensors: [T1, T2] + tensors: [A, B] - !Spatial - rank_variable: n1 + rank_variable: nA tile_shape: {{ MAC_TILE }} component: Scratchpad name: X - !Storage component: Scratchpad - tensors: [T1, T2, W1] + tensors: [A, B, WB] - !Temporal - rank_variable: n2 + rank_variable: nB tile_shape: 1 - !Spatial - rank_variable: n1 + rank_variable: nA tile_shape: 1 component: MAC name: X - !Compute - einsum: Matmul1 + einsum: MatmulB component: MAC diff --git a/tests/input_files/fused_matmuls_weights_inside.mapping.yaml b/tests/input_files/fused_matmuls_weights_inside.mapping.yaml index 0ccf7a85..dc2cb00c 100644 --- a/tests/input_files/fused_matmuls_weights_inside.mapping.yaml +++ b/tests/input_files/fused_matmuls_weights_inside.mapping.yaml @@ -1,11 +1,22 @@ +{# Same naming convention as examples/mappings/_matmul_convention.yaml, inlined + because Jinja only searches this file's own directory for imports. #} +{% set TENSOR_NAMES = "ABCDEFGHIJKLMNOPQRSTUVWXYZ" %} +{%- macro tensor_name(i) -%} + {% if i == 0 %} + {{- TENSOR_NAMES[8] -}} + {% else %} + {{- TENSOR_NAMES[i-1] -}} + {% endif %} +{%- endmacro -%} + mapping: nodes: - !Storage - tensors: [T0, T{{N_EINSUMS}}] + tensors: [{{tensor_name(0)}}, {{tensor_name(N_EINSUMS)}}] component: MainMemory {% for i in range(N_EINSUMS) %} - !Storage - tensors: [W{{i}}] + tensors: [W{{tensor_name(i+1)}}] component: MainMemory {% endfor %} - !Temporal @@ -17,18 +28,18 @@ mapping: - !Nested nodes: - !Storage - tensors: [W{{i}}] + tensors: [W{{tensor_name(i+1)}}] component: GlobalBuffer - !Storage - tensors: [T{{i}}, T{{i+1}}] + tensors: [{{tensor_name(i)}}, {{tensor_name(i+1)}}] component: GlobalBuffer - !Temporal - rank_variable: n{{i}} + rank_variable: n{{tensor_name(i)}} tile_shape: 1 - !Temporal - rank_variable: n{{i+1}} + rank_variable: n{{tensor_name(i+1)}} tile_shape: 1 - !Compute - einsum: Matmul{{i}} + einsum: Matmul{{tensor_name(i+1)}} component: MAC {% endfor %} diff --git a/tests/test_latency.py b/tests/test_latency.py index 439cd371..86effdfd 100644 --- a/tests/test_latency.py +++ b/tests/test_latency.py @@ -86,11 +86,11 @@ def test_fused_repays_communication_latency_each_switch(self): # Storages above the fused m loop are reserved once and can pre-send # ahead of it, so their connections are paid once (descent/ascent, maxed # across Einsums); storages below repeat for each of the 4 m iterations - # via the Einsum delay. Matmul0: T0's and T1's GlobalBuffer connections + # via the Einsum delay. MatmulA: I's and A's GlobalBuffer connections # (write + read = 30 each) plus one MAC: (30 + 3 + 30) x 4 = 252. - # Matmul1 mirrors it with T1's and T2's GlobalBuffer connections: + # MatmulB mirrors it with A's and B's GlobalBuffer connections: # (30 + 3 + 30) x 4 = 252. Paid once: the inputs' MainMemory reads - # (descent, max = 100) and T2's MainMemory write (ascent, 200; the first + # (descent, max = 100) and B's MainMemory write (ascent, 200; the first # tile skips the read-modify-write read). 252 + 252 + 100 + 200 = 804. self.assertEqual(row["Totallatency"], 804) @@ -122,7 +122,7 @@ def test_fused_memory_bound_overlaps_compute(self): compute: the total is the max of the two, not the sum of per-Einsum maxes. A MainMemory transfer runs while the GlobalBuffer reservation on its other end is alive: the weights' GlobalBuffer staging is above the shared m loop, - so their reads may fill any slice of the fused execution, while T0/T2's + so their reads may fill any slice of the fused execution, while I/B's GlobalBuffer reservations live below it, confining that traffic to its own Einsum. MainMemory's total busy time still bounds the total either way.""" row = total_latency( @@ -133,21 +133,21 @@ def test_fused_memory_bound_overlaps_compute(self): MM_WRITE_THROUGHPUT=0.1, COMPUTE_THROUGHPUT=0.2, ) - # Matmul0 is compute-bound: 64 MACs / 0.2 = 320 vs reading T0 and W0 from - # MainMemory (256 bits / 1 = 256). Matmul1 is memory-bound: writing T2 back - # (128 bits / 0.1 = 1280) plus reading W1 (128 / 1 = 128) is 1408 vs 320. - self.assertEqual(row["Matmul0component_latencyMAC"], 320) - self.assertEqual(row["Matmul0component_latencyMainMemory"], 256) - self.assertEqual(row["Matmul1component_latencyMAC"], 320) - self.assertEqual(row["Matmul1component_latencyMainMemory"], 1408) - - # max(320 + 320, 256 + 1408) = 1664: Matmul0's MainMemory slack absorbs part - # of Matmul1's traffic. Summing per-Einsum maxes would give 320 + 1408 = 1728. + # MatmulA is compute-bound: 64 MACs / 0.2 = 320 vs reading I and WA from + # MainMemory (256 bits / 1 = 256). MatmulB is memory-bound: writing B back + # (128 bits / 0.1 = 1280) plus reading WB (128 / 1 = 128) is 1408 vs 320. + self.assertEqual(row["MatmulAcomponent_latencyMAC"], 320) + self.assertEqual(row["MatmulAcomponent_latencyMainMemory"], 256) + self.assertEqual(row["MatmulBcomponent_latencyMAC"], 320) + self.assertEqual(row["MatmulBcomponent_latencyMainMemory"], 1408) + + # max(320 + 320, 256 + 1408) = 1664: MatmulA's MainMemory slack absorbs part + # of MatmulB's traffic. Summing per-Einsum maxes would give 320 + 1408 = 1728. self.assertEqual(row["Totallatency"], 1664) def test_separate_read_write_ports(self): """With separate MainMemory ports, reads and writes are independent - components that may overlap: W1's reads no longer serialize behind T2's + components that may overlap: WB's reads no longer serialize behind B's writeback, saving their 64 bit-times versus the shared port's 1664.""" row = total_latency( LATENCY_ARCH, @@ -159,16 +159,16 @@ def test_separate_read_write_ports(self): MM_SEPARATE_PORTS=True, ) self.assertEqual( - row["Matmul0component_latencyMainMemory (read)"], 256 + row["MatmulAcomponent_latencyMainMemory (read)"], 256 ) self.assertEqual( - row["Matmul1component_latencyMainMemory (read)"], 128 + row["MatmulBcomponent_latencyMainMemory (read)"], 128 ) self.assertEqual( - row["Matmul1component_latencyMainMemory (write)"], 1280 + row["MatmulBcomponent_latencyMainMemory (write)"], 1280 ) - # T2's writes go to its GlobalBuffer staging below the fused m loop, so - # the write port's 1280 is confined to Matmul1's block and sets its + # B's writes go to its GlobalBuffer staging below the fused m loop, so + # the write port's 1280 is confined to MatmulB's block and sets its # width; the shareable reads (256 + 128 = 384) and the MACs (320 + 320) # hide beneath it. 320 + 1280 = 1600. self.assertEqual(row["Totallatency"], 1600) @@ -177,8 +177,8 @@ def test_transfer_needs_deeper_reservation(self): """With every GlobalBuffer staging below the fused m loop, MainMemory transfers can only run during their own Einsum's per-iteration windows, so each Einsum's MainMemory traffic is confined to its block and the overlap - credit above is lost. Matmul0 is compute-bound (320 vs reading T0 once and - W0 per m iteration: (128 + 4 x 128) / 4 = 160); Matmul1's writeback still + credit above is lost. MatmulA is compute-bound (320 vs reading I once and + WA per m iteration: (128 + 4 x 128) / 4 = 160); MatmulB's writeback still dominates its block (128 + 1280 = 1408). 320 + 1408 = 1728, though MainMemory is only busy for 160 + 1408 = 1568 of it.""" row = total_latency( @@ -222,9 +222,9 @@ def test_timeline_totals(self): def test_timeline_overlap_layout(self): """The layout from test_fused_memory_bound_overlaps_compute, bar by bar. - T0/T2 traffic sits at level 1 (their GlobalBuffer reservations live below + I/B traffic sits at level 1 (their GlobalBuffer reservations live below the fused m loop, confining it to its own Einsum's block); weight reads sit - at level 0 and may fill slack anywhere. Matmul1's T2 writeback fills its + at level 0 and may fill slack anywhere. MatmulB's B writeback fills its block, and MainMemory's total busy time (256 + 1408 = 1664) sets the end.""" from accelforge.plotting.latency import _latency_timeline @@ -241,7 +241,7 @@ def test_timeline_overlap_layout(self): self.assertEqual(total, 1664) self.assertEqual( [(b.einsum, b.start, b.end) for b in blocks], - [("Matmul0", 0, 320), ("Matmul1", 320, 1664)], + [("MatmulA", 0, 320), ("MatmulB", 320, 1664)], ) mm = [ (b.einsum, b.level, b.start, b.end) @@ -251,17 +251,17 @@ def test_timeline_overlap_layout(self): self.assertEqual( sorted(mm), [ - ("Matmul0", 0, 128, 256), # W0 reads, shareable - ("Matmul0", 1, 0, 128), # T0 reads, private - # W1 reads, shareable; right-aligned to the 1664 deadline - ("Matmul1", 0, 1536, 1664), - ("Matmul1", 1, 320, 1600), # T2 writeback, private + ("MatmulA", 0, 128, 256), # WA reads, shareable + ("MatmulA", 1, 0, 128), # I reads, private + # WB reads, shareable; right-aligned to the 1664 deadline + ("MatmulB", 0, 1536, 1664), + ("MatmulB", 1, 320, 1600), # B writeback, private ], ) # Compute has its own lane now, and nothing else (communication is zero # here) is left for the Other lane. mac = [(b.einsum, b.start, b.end) for b in bars if b.component == "MAC"] - self.assertEqual(mac, [("Matmul0", 0, 320), ("Matmul1", 320, 640)]) + self.assertEqual(mac, [("MatmulA", 0, 320), ("MatmulB", 320, 640)]) self.assertEqual([b for b in bars if b.component is None], []) def test_plot_latency(self): From e160c7678586bc0b15c424e41f6e3143320b82c2 Mon Sep 17 00:00:00 2001 From: Tanner Andrulis Date: Sat, 26 Sep 2026 12:50:19 -0400 Subject: [PATCH 8/8] New latency model is optional --- accelforge/frontend/model.py | 7 ++++ accelforge/model/run_model.py | 73 ++++++++++++++++++++--------------- tests/network/test_network.py | 1 + tests/test_latency.py | 5 ++- 4 files changed, 53 insertions(+), 33 deletions(-) diff --git a/accelforge/frontend/model.py b/accelforge/frontend/model.py index ac0077ec..52f6dcd6 100644 --- a/accelforge/frontend/model.py +++ b/accelforge/frontend/model.py @@ -12,3 +12,10 @@ class Model(EvalableModel): If using spec to call mapper, leave this configuration as is. The mapper will make necessary configurations. """ + + _use_new_latency_model: bool = False + + def __init__(self, **kwargs): + use_new_latency_model = kwargs.pop("_use_new_latency_model", False) + super().__init__(**kwargs) + self._use_new_latency_model = use_new_latency_model diff --git a/accelforge/model/run_model.py b/accelforge/model/run_model.py index 4c2ba813..0ad50da6 100644 --- a/accelforge/model/run_model.py +++ b/accelforge/model/run_model.py @@ -68,7 +68,11 @@ def run_model( reuse, job.flattened_arch, pmapping, - per_level_components=oset(job.components_track_latency), + per_level_components=( + oset(job.components_track_latency) + if spec.model._use_new_latency_model + else oset() + ), n_shared_loops=n_shared_loops, ) overall_latency = max_nonzero(*latency.values()) @@ -243,37 +247,42 @@ def run_model( for component, cur_latency in latency.items(): df[f"component_latency{component}"] = cur_latency * n_instances - # Components shared across Einsums get per-level latency columns so joining can - # sum their busy time across Einsums and let it overlap with the other Einsums' - # latency. Their latency is folded into Totallatency at joining time - # instead of here. - for component, level_latency in latency_per_level.items(): - for level, cur_latency in level_latency.items(): - df[complatency2col(component, level)] = cur_latency * n_instances - - # The total latency is the sum of the Einsum's wind-up/down delay and the - # slowest component's busy time. Fused-loop-crossing descent and ascent - # latencies get their own columns because they may be overlapped with other - # Einsums. - einsum_delay, ascent_descent_latency = communication_latency( - reuse, - job.flattened_arch, - tensor_to_backing, - workload.einsums[job.einsum_name].output_tensor_names, - n_fused=n_shared_loops, - ) - for direction, per_level in ascent_descent_latency.items(): - for level, cur_latency in per_level.items(): - df[commlatency2col(direction, level)] = cur_latency * n_instances - - per_component_total = [] - for component, l in latency.items(): - if component not in latency_per_level: - if not isinstance(l, Number) or l != 0: - per_component_total.append(l) - - slowest = max_nonzero(*per_component_total) - df["Totallatency"] = (slowest + einsum_delay) * n_instances + if spec.model._use_new_latency_model: + # Components shared across Einsums get per-level latency columns so + # joining can sum their busy time across Einsums and let it overlap with + # the other Einsums' latency. Their latency is folded into + # Totallatency at joining time instead of here. + for component, level_latency in latency_per_level.items(): + for level, cur_latency in level_latency.items(): + df[complatency2col(component, level)] = cur_latency * n_instances + + # The total latency is the sum of the Einsum's wind-up/down delay and the + # slowest component's busy time. Fused-loop-crossing descent and ascent + # latencies get their own columns because they may be overlapped with + # other Einsums. + einsum_delay, ascent_descent_latency = communication_latency( + reuse, + job.flattened_arch, + tensor_to_backing, + workload.einsums[job.einsum_name].output_tensor_names, + n_fused=n_shared_loops, + ) + for direction, per_level in ascent_descent_latency.items(): + for level, cur_latency in per_level.items(): + df[commlatency2col(direction, level)] = cur_latency * n_instances + + per_component_total = [] + for component, l in latency.items(): + if component not in latency_per_level: + if not isinstance(l, Number) or l != 0: + per_component_total.append(l) + + slowest = max_nonzero(*per_component_total) + df["Totallatency"] = (slowest + einsum_delay) * n_instances + else: + # Each Einsum's latency is the max of its per-component latencies. Einsums + # do not share latency. + df["Totallatency"] = overall_latency * n_instances # ================================================================================= # Energy diff --git a/tests/network/test_network.py b/tests/network/test_network.py index 1e64d693..37d65960 100644 --- a/tests/network/test_network.py +++ b/tests/network/test_network.py @@ -335,6 +335,7 @@ def test_hierarchical_1d_all_to_all(self): "M_TILE": M_TILE, }, ) + spec.model._use_new_latency_model = True result = spec.evaluate_mapping() # --- MacArray: all-to-all switch --------------------------------- diff --git a/tests/test_latency.py b/tests/test_latency.py index 86effdfd..0a56ca23 100644 --- a/tests/test_latency.py +++ b/tests/test_latency.py @@ -26,12 +26,14 @@ def make_spec(arch, mapping, n_einsums, **jinja): - return Spec.from_yaml( + spec = Spec.from_yaml( af.examples.workloads.basic.matmuls, arch, mapping, jinja_parse_data={"N_EINSUMS": n_einsums, "M": 4, "KN": 4, **jinja}, ) + spec.model._use_new_latency_model = True + return spec def total_latency(arch, mapping, n_einsums, **jinja): @@ -294,6 +296,7 @@ def test_timeline_matches_mapper_output(self): af.examples.workloads.basic.matmuls, jinja_parse_data={"N_EINSUMS": 2, "M": 16, "KN": 16}, ) + spec.model._use_new_latency_model = True spec.mapper.metrics = af.mapper.Metrics.LATENCY result = spec.map_workload_to_arch(print_progress=False) two = copy.copy(result)