Optimize and finalize
compile() and trace() return a ModelGraph as built, with every operator as the model states it and nothing optimized, placed or finalized. Four calls take it from there. optimize() runs the graph optimizer when you want it and reports what it did. to() names the device or the stream the graph serves on. finalize() places the graph there, packs the weights into their kernel layouts and plans the execution order, and from then on the graph runs. The calls are available from C++ and Python, under the same names.
A graph as built
The model below computes y = Relu(Neg(Neg(x))). The two Negs cancel each other, which gives the graph optimizer one rewrite to find. Every example on this page traces it. labels lists a graph's nodes in order, each by its input name or its operator, and trace_model traces the model over an x of shape [2, 3]. The input every program runs is input() in C++ and x in Python, and values prints a C++ result on one line.
A graph that is not finalized refuses run() with the code name FAILED_PRECONDITION, and the message names the call to make.
- C++
- Python
#include <algorithm>
#include <cstdio>
#include <string>
#include <vector>
#include <ClikaRT/clika_rt.h>
using ClikaRT::DataType;
using ClikaRT::Error;
using ClikaRT::Tensor;
using ClikaRT::graph::ModelGraph;
using ClikaRT::graph::Node;
using ClikaRT::graph::NodeKind;
using ClikaRT::graph::OptimizeOptions;
using ClikaRT::graph::OptimizeReport;
using ClikaRT::graph::Transform;
using ClikaRT::graph::TransformReport;
namespace ops = ClikaRT::ops;
namespace transforms = ClikaRT::graph::transforms;
namespace {
// y = Relu(Neg(Neg(x))): the two Negs cancel, which the graph optimizer finds.
std::vector<Tensor> model(const std::vector<Tensor>& inputs) {
return {ops::relu(ops::neg(ops::neg(inputs[0])))};
}
// A node's label: an input's name, an operator's code.
std::string label(const Node& node) {
return node.kind() == NodeKind::Input ? node.name() : std::string(ClikaRT::graph::op_code_name(node.op_code()));
}
// The labels of a graph's nodes, in order, separated by spaces.
std::string labels(const ModelGraph& graph) {
std::string out;
for (const Node& node : graph.nodes()) out += (out.empty() ? "" : " ") + label(node);
return out;
}
// A trace returns the graph as built: every operator as written, nothing optimized or finalized.
ModelGraph trace_model() {
const std::vector<ClikaRT::spec::TensorSpec> signature = {{"x", DataType::Float32, {2, 3}}};
const std::vector<std::string> outputs = {"y"};
return ClikaRT::graph::trace(model, signature, "lifecycle", outputs);
}
// The input every program runs, [2, 3].
Tensor input() {
const std::vector<float> x = {1.0F, -2.0F, 3.0F, -4.0F, 5.0F, -6.0F};
return Tensor::from_data(x.data(), {2, 3}, DataType::Float32);
}
// A tensor's values, flattened and separated by spaces.
std::string values(const Tensor& tensor) {
std::string out;
for (const float value : tensor.reshape({-1}).item_as_vec<float>()) {
char text[32];
std::snprintf(text, sizeof(text), "%g", static_cast<double>(value));
out += (out.empty() ? "" : " ") + std::string(text);
}
return out;
}
} // namespace
int main() {
const ModelGraph graph = trace_model();
std::printf("%s | %s\n", labels(graph).c_str(), graph.is_finalized() ? "true" : "false"); // x Neg Neg Relu | false
// Only a finalized graph runs.
try {
graph.run({input()});
} catch (const Error& error) {
std::printf("%s | %s\n", error.code_name().c_str(), error.what());
// FAILED_PRECONDITION | run: the graph is not finalized; call finalize() first
}
return 0;
}
import numpy as np
import clika_runtime as crt
from clika_runtime.graph import DefaultsOptions, NodeKind, transforms
def model(inputs: list[crt.Tensor]) -> list[crt.Tensor]:
return [crt.relu(crt.neg(crt.neg(inputs[0])))] # y = Relu(Neg(Neg(x))): the two Negs cancel
def labels(graph: crt.graph.ModelGraph) -> list[str]:
return [node.name if node.kind == NodeKind.Input else node.op_code.name for node in graph.nodes()]
# A trace returns the graph as built: every operator as written, nothing optimized or finalized.
def trace_model() -> crt.graph.ModelGraph:
return crt.trace(model, [crt.TensorSpec("x", crt.float32, [2, 3])], output_names=["y"]).graph
x = np.array([[1.0, -2.0, 3.0], [-4.0, 5.0, -6.0]], np.float32) # the input every program runs (numpy as the data entry)
graph = trace_model()
print(labels(graph), graph.is_finalized()) # ['x', 'Neg', 'Neg', 'Relu'] False
# Only a finalized graph runs.
try:
graph.run([crt.tensor(x)])
except crt.InvalidArgumentError as error:
print(error.code_name, "|", error)
# FAILED_PRECONDITION | run: the graph is not finalized; call finalize() first
Each program on this page is complete and runs on its own. From here on, a block shows the part of its program that follows the opening lines the first block shows (the includes or imports, model, the helpers, the input and, in Python, the trace).
Run the graph optimizer
optimize() runs the graph's default list of transforms. Each iteration runs the whole list once, and the run stops at the first iteration that changes nothing. It returns an OptimizeReport. iterations counts the iterations that ran, converged says the run stopped on its own, and changed says it modified the graph. transforms holds one TransformReport per list entry, in list order, and each row's applications counts the iterations in which its transform changed the graph. Here remove_double_neg removes both Negs in the first iteration, and the second iteration changes nothing.
- C++
- Python
int main() {
ModelGraph graph = trace_model();
const OptimizeReport report = graph.optimize();
std::printf("%s\n", labels(graph).c_str()); // x Relu
std::printf("%lld %s %s\n", static_cast<long long>(report.iterations), report.converged ? "true" : "false",
report.changed ? "true" : "false"); // 2 true true
for (const TransformReport& row : report.transforms) {
if (row.applications > 0) {
std::printf("%s %lld\n", row.transform.name().c_str(), static_cast<long long>(row.applications));
// remove_double_neg 1
}
}
return 0;
}
report = graph.optimize()
print(labels(graph)) # ['x', 'Relu']
print(report.iterations, report.converged, report.changed) # 2 True True
print([(row.transform.name, row.applications) for row in report.transforms if row.applications])
# [('remove_double_neg', 1)]
A view you took before optimize() finds its node again by its key, or refuses once the optimizer removed that node, as Views across edits on the Query a graph page shows.
The default list
default_transforms() returns the list optimize() runs on this graph, in the order it runs them, as a starting point to edit and pass back. transforms::defaults() (in Python, transforms.defaults()) returns the default list for a set of choices without a graph. A graph's own list follows how the graph was built, so the two can differ. For a traced graph with no choices set, they are the same list.
Each entry is a Transform handle named as its factory is, so transforms::remove_double_neg() names remove_double_neg (name() in C++, the name property in Python). Compare and store transforms by name, since the values behind kind() move when the runtime's transform list changes.
- C++
- Python
int main() {
const ModelGraph graph = trace_model();
const std::vector<Transform> listed = graph.default_transforms();
const auto holds = [&listed](const Transform& transform) {
return std::find(listed.begin(), listed.end(), transform) != listed.end() ? "true" : "false";
};
std::printf("%s\n", listed.front().name().c_str()); // cleanup_graph
std::printf("%s %s\n", holds(transforms::remove_double_neg()),
holds(transforms::fixate_dyn_shape_queues())); // true false
std::printf("%s\n", listed == transforms::defaults() ? "true" : "false"); // true
return 0;
}
listed = graph.default_transforms()
print(listed[0].name) # cleanup_graph
print(transforms.remove_double_neg() in listed,
transforms.fixate_dyn_shape_queues() in listed) # True False
print(listed == transforms.defaults()) # True
Run a list of your own
OptimizeOptions::transforms runs the list you give it instead, in order, each iteration. In Python, pass the list as the first argument of optimize(). A list may hold a transform more than once. An empty list runs none of them, though the run still drops the nodes nothing reads.
The list is checked before anything runs. A transform placed ahead of one it must follow is refused with the code name INVALID_ARGUMENT, and the message names both entries and the order to use. The refused call leaves the graph as it was.
- C++
- Python
int main() {
ModelGraph graph = trace_model();
OptimizeOptions options;
options.transforms = std::vector<Transform>{transforms::remove_double_neg()};
const OptimizeReport report = graph.optimize(options);
const TransformReport& row = report.transforms.front();
std::printf("%s | %zu %s %lld\n", labels(graph).c_str(), report.transforms.size(), row.transform.name().c_str(),
static_cast<long long>(row.applications)); // x Relu | 1 remove_double_neg 1
// A list is checked before anything runs: matmul_absorb_transpose must run after remove_double_permute.
ModelGraph other = trace_model();
options.transforms = std::vector<Transform>{transforms::matmul_absorb_transpose(), transforms::remove_double_permute()};
try {
other.optimize(options);
} catch (const Error& error) {
std::printf("%s | %s\n", error.code_name().c_str(), error.what());
// INVALID_ARGUMENT | optimize: transforms[0] (matmul_absorb_transpose) must run after remove_double_permute, which the list places later, at transforms[1]; list remove_double_permute ahead of it
}
std::printf("%s\n", labels(other).c_str()); // x Neg Neg Relu
return 0;
}
report = graph.optimize([transforms.remove_double_neg()])
print(labels(graph), [(row.transform.name, row.applications) for row in report.transforms])
# ['x', 'Relu'] [('remove_double_neg', 1)]
# A list is checked before anything runs: matmul_absorb_transpose must run after remove_double_permute.
other = trace_model()
try:
other.optimize([transforms.matmul_absorb_transpose(), transforms.remove_double_permute()])
except crt.InvalidArgumentError as error:
print(error.code_name, "|", error)
# INVALID_ARGUMENT | optimize: transforms[0] (matmul_absorb_transpose) must run after remove_double_permute, which the list places later, at transforms[1]; list remove_double_permute ahead of it
print(labels(other)) # ['x', 'Neg', 'Neg', 'Relu']
A list can also hold a transform you write, a function that edits the graph and runs beside the runtime's transforms (Transform::from_function in C++, Transform(name, fn) in Python). Write a transform covers them.
Cap the iterations
max_num_iterations caps one run at that many iterations. It is 100 unless you set it, and a value outside 1 to 2147483647 is refused with INVALID_ARGUMENT. A run that reaches the cap still leaves a correct graph. Its report reads converged as false, firing_at_cap marks the transforms that changed the graph in the last iteration, and the runtime logs a warning that names them. Here the single iteration applies remove_double_neg, and the cap ends the run before a second iteration can show that nothing else changes.
- C++
- Python
int main() {
ModelGraph graph = trace_model();
OptimizeOptions options;
options.max_num_iterations = 1; // one pass over the list
const OptimizeReport report = graph.optimize(options);
std::printf("%lld %s %s\n", static_cast<long long>(report.iterations), report.converged ? "true" : "false",
report.changed ? "true" : "false"); // 1 false true
for (const TransformReport& row : report.transforms) {
if (row.firing_at_cap) std::printf("%s\n", row.transform.name().c_str()); // remove_double_neg
}
std::printf("%s\n", labels(graph).c_str()); // x Relu
options.max_num_iterations = 0;
try {
graph.optimize(options);
} catch (const Error& error) {
std::printf("%s | %s\n", error.code_name().c_str(), error.what());
// INVALID_ARGUMENT | optimize: max_num_iterations must be between 1 and 2147483647, got 0
}
return 0;
}
report = graph.optimize(max_num_iterations=1) # one pass over the list
print(report.iterations, report.converged, report.changed) # 1 False True
print([row.transform.name for row in report.transforms if row.firing_at_cap]) # ['remove_double_neg']
print(labels(graph)) # ['x', 'Relu']
try:
graph.optimize(max_num_iterations=0)
except crt.InvalidArgumentError as error:
print(error.code_name, "|", error)
# INVALID_ARGUMENT | optimize: max_num_iterations must be between 1 and 2147483647, got 0
A transform of the runtime's that fails is undone and skipped for the rest of the run, which goes on. Its row reads failed, with the status name in failure_code and the message in failure_message. The graph stays correct, only less optimized.
Fix the shapes
DefaultsOptions holds the choices the transforms read, each off by default. With every choice off, the graph keeps its input and output contract and serves every shape it admits. Two choices fix the graph to its shapes instead, collapsing shape arithmetic such as shape cones, masks and reshape templates into constants. Each adds the shape-fixation transforms, fixate_dyn_shape_queues among them, to the default list.
A traced graph follows shape_fixation, which fixes it to the shapes it was traced with. A graph compiled from ONNX follows specialize, which fixes it to the input shapes the compile pinned (CompileOptions::specs or sample_inputs). Without pinned shapes, the arithmetic the model's own static shapes fix collapses either way. The graph-free list takes the choices as stated.
Pass the choices to optimize() in OptimizeOptions::defaults (in Python, the defaults= keyword), and to default_transforms() to see the list they give. A list of your own that holds a transform these choices switch off is refused, and the message names the choice to set.
- C++
- Python
int main() {
const ModelGraph graph = trace_model();
const Transform fixate = transforms::fixate_dyn_shape_queues();
const auto holds = [&fixate](const std::vector<Transform>& listed) {
return std::find(listed.begin(), listed.end(), fixate) != listed.end() ? "true" : "false";
};
// A traced graph follows shape_fixation: its list gains the transforms that fix it to the traced shapes.
ClikaRT::graph::DefaultsOptions fixation;
fixation.shape_fixation = true;
std::printf("%s %s\n", holds(graph.default_transforms()), holds(graph.default_transforms(fixation))); // false true
// A graph compiled from ONNX follows specialize; the graph-free list takes the choice as stated.
ClikaRT::graph::DefaultsOptions specialize;
specialize.specialize = true;
std::printf("%s %s\n", holds(transforms::defaults()), holds(transforms::defaults(specialize))); // false true
return 0;
}
fixate = transforms.fixate_dyn_shape_queues()
# A traced graph follows shape_fixation: its list gains the transforms that fix it to the traced shapes.
print(fixate in graph.default_transforms(),
fixate in graph.default_transforms(DefaultsOptions(shape_fixation=True))) # False True
# A graph compiled from ONNX follows specialize; the graph-free list takes the choice as stated.
print(fixate in transforms.defaults(),
fixate in transforms.defaults(DefaultsOptions(specialize=True))) # False True
Channels-last inputs and outputs
A model exported channels-first reaches the runtime's channels-last operators through a conversion at each input a spatial operator, such as a convolution, reads. channels_last_inputs omits that conversion, so the input is supplied channels-last ([N, spatial..., C]) from then on. channels_last_outputs does the same for an output that leaves through a conversion. Both change the graph's input and output contract, and inputs() and outputs() report the layout to supply and the layout delivered.
The program builds its ONNX model with a helper outside the block, an X of shape [1, 2, 4, 4] that a 1x1 Conv reads, followed by a Relu, and layout prints an input's name and dims. drop_boundary_permutes, the transform that omits the conversions, runs only under one of the two choices.
- C++
- Python
int main() {
ModelGraph graph = compile_conv_model();
std::printf("%s\n", layout(graph.inputs().front()).c_str()); // X [1, 2, 4, 4]
// drop_boundary_permutes runs only under a channels-last choice.
OptimizeOptions listed;
listed.transforms = std::vector<Transform>{transforms::drop_boundary_permutes()};
try {
graph.optimize(listed);
} catch (const Error& error) {
std::printf("%s | %s\n", error.code_name().c_str(), error.what());
// INVALID_ARGUMENT | optimize: transforms[0] (drop_boundary_permutes) changes the layout of the graph's inputs and outputs; set DefaultsOptions::channels_last_inputs or channels_last_outputs to run it
}
OptimizeOptions options;
options.defaults.channels_last_inputs = true; // X is supplied channels-last from here on
graph.optimize(options);
std::printf("%s\n", layout(graph.inputs().front()).c_str()); // X [1, 4, 4, 2]
return 0;
}
graph = compile_conv_model()
print(layout(graph)) # X [1, 2, 4, 4]
# drop_boundary_permutes runs only under a channels-last choice.
try:
graph.optimize([transforms.drop_boundary_permutes()])
except crt.InvalidArgumentError as error:
print(error.code_name, "|", error)
# INVALID_ARGUMENT | optimize: transforms[0] (drop_boundary_permutes) changes the layout of the graph's inputs and outputs; set DefaultsOptions::channels_last_inputs or channels_last_outputs to run it
graph.optimize(defaults=DefaultsOptions(channels_last_inputs=True)) # X is supplied channels-last from here on
print(layout(graph)) # X [1, 4, 4, 2]
Place the graph with to()
to() takes a device or a stream. Before finalize(), it records where finalizing places the graph and moves nothing, and device() reports the new device at once. After finalize(), it moves the finalized graph. An operator that changes device re-packs its weights there once, and no copy of a moved weight stays behind. A graph no to() moved serves where its operators sit, the CPU for this trace, and a graph compiled from ONNX serves on the device CompileOptions::weights names until to() names another. In Python, device is a property.
To compute on a stream of your own, create it and pass it to to() once, and every run computes there. On a host with an accelerator, to(Device::cuda(0)) places the operators and their weights on it; the program stays on the CPU, so it runs on any host. to() refuses while a run is in flight, while a KV cache is attached (call it before attach_kv_cache()), and while optimize(), finalize() or an edit runs on the graph.
- C++
- Python
int main() {
ModelGraph graph = trace_model();
std::printf("%s\n", ClikaRT::device::compute_api_name(graph.device().api)); // CPU: where its operators sit
// Before finalize(), to() records where finalize() places the graph; nothing moves yet.
graph.to(ClikaRT::Stream::create(ClikaRT::Device::cpu()));
std::printf("%s %s\n", ClikaRT::device::compute_api_name(graph.device().api),
graph.is_finalized() ? "true" : "false"); // CPU false
graph.optimize();
graph.finalize(); // places the operators on that stream and packs their weights there
std::printf("%s\n", values(graph.run({input()}).front()).c_str()); // 1 0 3 0 5 0
// After finalize(), to() moves the finalized graph.
graph.to(ClikaRT::Device::cpu());
std::printf("%s\n", values(graph.run({input()}).front()).c_str()); // 1 0 3 0 5 0
return 0;
}
print(graph.device) # cpu: where its operators sit
# Before finalize(), to() records where finalize() places the graph; nothing moves yet.
graph.to(crt.Stream.create(crt.Device.cpu()))
print(graph.device, graph.is_finalized()) # cpu False
graph.optimize()
graph.finalize() # places the operators on that stream and packs their weights there
(y,) = graph.run([crt.tensor(x)])
print(y.numpy().tolist()) # [[1.0, 0.0, 3.0], [0.0, 5.0, 0.0]]
# After finalize(), to() moves the finalized graph.
graph.to(crt.Device.cpu())
(again,) = graph.run([crt.tensor(x)])
print(again.numpy().tolist()) # [[1.0, 0.0, 3.0], [0.0, 5.0, 0.0]]
Finalize and run
finalize() places the graph on its device, packs the weights into their kernel layouts, absorbs the constants the operators hold and plans the execution order. A second call succeeds and changes nothing. A finalized graph runs, benches and takes a KV cache. It refuses optimize() and every edit with the code name FAILED_PRECONDITION, and the message says to compile or trace the model again, while run() and to() keep working. Each program checks the run against a reference it computes itself, the Python one with numpy.
- C++
- Python
int main() {
ModelGraph graph = trace_model();
graph.optimize();
graph.finalize();
graph.finalize(); // a second call succeeds and changes nothing
std::printf("%s\n", graph.is_finalized() ? "true" : "false"); // true
const Tensor y = graph.run({input()}).front();
const std::vector<float> got = y.reshape({-1}).item_as_vec<float>();
const std::vector<float> x = {1.0F, -2.0F, 3.0F, -4.0F, 5.0F, -6.0F};
bool matches = got.size() == x.size();
for (std::size_t i = 0; matches && i < x.size(); ++i) {
matches = got[i] == std::max(x[i], 0.0F); // the reference, Relu(x), by hand
}
std::printf("%s\n", values(y).c_str()); // 1 0 3 0 5 0
std::printf("%s\n", matches ? "true" : "false"); // true
// A finalized graph refuses optimize() and every edit; run() and to() keep working.
try {
graph.optimize();
} catch (const Error& error) {
std::printf("%s | %s\n", error.code_name().c_str(), error.what());
// FAILED_PRECONDITION | optimize: the graph is finalized; optimize() runs before finalize(), so compile or trace the model again to optimize it
}
try {
graph.rename_output("y", "out");
} catch (const Error& error) {
std::printf("%s\n", error.code_name().c_str()); // FAILED_PRECONDITION
}
return 0;
}
graph.optimize()
graph.finalize()
graph.finalize() # a second call succeeds and changes nothing
print(graph.is_finalized()) # True
(y,) = graph.run([crt.tensor(x)])
print(y.numpy().tolist()) # [[1.0, 0.0, 3.0], [0.0, 5.0, 0.0]]
print(np.array_equal(y.numpy(), np.maximum(x, 0))) # True (numpy states the reference)
# A finalized graph refuses optimize() and every edit; run() and to() keep working.
try:
graph.optimize()
except crt.InvalidArgumentError as error:
print(error.code_name, "|", error)
# FAILED_PRECONDITION | optimize: the graph is finalized; optimize() runs before finalize(), so compile or trace the model again to optimize it
try:
graph.rename_output("y", "out")
except crt.InvalidArgumentError as error:
print(error.code_name) # FAILED_PRECONDITION
Optimize and edit before you finalize. Edit a graph covers the edits, and Query a graph reads the graph at any point in its lifecycle.