From 3e6d61b4fec1f395a6d43f67d655ddff14935b77 Mon Sep 17 00:00:00 2001 From: pnikolic-amd Date: Wed, 2 Sep 2026 16:16:35 +0200 Subject: [PATCH 01/14] Sql backend implmementation for binary cache; Added common interface to support mutiple backend in binary cache; Added functionalities to sql wrapper to support blobs --- src/include/migraphx/sqlite.hpp | 65 ++- src/sqlite.cpp | 128 +++++- src/targets/gpu/CMakeLists.txt | 2 + src/targets/gpu/binary_cache.cpp | 123 +++--- src/targets/gpu/file_binary_cache.cpp | 112 +++++ .../gpu/include/migraphx/gpu/binary_cache.hpp | 35 +- .../migraphx/gpu/binary_cache_backend.hpp | 399 ++++++++++++++++++ .../migraphx/gpu/binary_cache_entry.hpp | 68 +++ .../migraphx/gpu/file_binary_cache.hpp | 60 +++ .../migraphx/gpu/sqlite_binary_cache.hpp | 81 ++++ src/targets/gpu/sqlite_binary_cache.cpp | 210 +++++++++ test/gpu/binary_cache.cpp | 333 +++++++++++++-- test/sqlite.cpp | 79 +++- 13 files changed, 1575 insertions(+), 120 deletions(-) create mode 100644 src/targets/gpu/file_binary_cache.cpp create mode 100644 src/targets/gpu/include/migraphx/gpu/binary_cache_backend.hpp create mode 100644 src/targets/gpu/include/migraphx/gpu/binary_cache_entry.hpp create mode 100644 src/targets/gpu/include/migraphx/gpu/file_binary_cache.hpp create mode 100644 src/targets/gpu/include/migraphx/gpu/sqlite_binary_cache.hpp create mode 100644 src/targets/gpu/sqlite_binary_cache.cpp diff --git a/src/include/migraphx/sqlite.hpp b/src/include/migraphx/sqlite.hpp index 97f3c0405f7..b11bb6e000b 100644 --- a/src/include/migraphx/sqlite.hpp +++ b/src/include/migraphx/sqlite.hpp @@ -1,7 +1,7 @@ /* * The MIT License (MIT) * - * Copyright (c) 2015-2023 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2015-2026 Advanced Micro Devices, Inc. All rights reserved. * * Permission is hereby granted, free of charge, to any person obtaining a copy * of this software and associated documentation files (the "Software"), to deal @@ -26,7 +26,11 @@ #include #include +#include +#include #include +#include +#include #include #include @@ -34,14 +38,73 @@ namespace migraphx { inline namespace MIGRAPHX_INLINE_NS { struct sqlite_impl; +struct sqlite_stmt_impl; + +/// A prepared statement, holding a reference to the connection it was prepared on so it can +/// never outlive it. +/// +/// Not thread safe: one statement may be used by one thread at a time, even though the +/// connection itself is serialized. Reuse is the point of preparing -- prepare once, then +/// reset/bind/step per operation. +struct MIGRAPHX_EXPORT sqlite_stmt +{ + sqlite_stmt() = default; + + /// Bind a parameter. Indices are 1-based, matching sqlite's own convention. + sqlite_stmt& bind(int i, std::string_view s); + sqlite_stmt& bind(int i, std::int64_t x); + sqlite_stmt& bind(int i, const std::vector& blob); + + /// Step once. True when a row is available, false when the statement is done. + bool step(); + + /// Clear bindings and rewind, so the statement can be used again. Safe at any point, + /// including after step() has thrown. + void reset() noexcept; + + /// Read a column of the current row. Indices here are 0-based, again matching sqlite. + std::string column_text(int i) const; + std::vector column_blob(int i) const; + + bool valid() const { return impl != nullptr; } + + private: + friend struct sqlite; + std::shared_ptr impl; +}; + +/// Resets a statement on scope exit, so an early return or a thrown exception cannot leave +/// bindings or a half-consumed result set behind for whoever uses the statement next. +struct sqlite_stmt_reset +{ + explicit sqlite_stmt_reset(sqlite_stmt& s) : stmt(&s) {} + sqlite_stmt_reset(const sqlite_stmt_reset&) = delete; + sqlite_stmt_reset& operator=(const sqlite_stmt_reset&) = delete; + ~sqlite_stmt_reset() { stmt->reset(); } + + private: + sqlite_stmt* stmt; +}; struct MIGRAPHX_EXPORT sqlite { sqlite() = default; static sqlite read(const fs::path& p); static sqlite write(const fs::path& p); + + /// Open for writing, or nullopt if the file cannot be opened or created. For callers + /// that treat an unusable database as "no cache" rather than as an error. + static optional try_write(const fs::path& p); + std::vector> execute(const std::string& s); + sqlite_stmt prepare(const std::string& sql); + + /// How long to wait for a lock held by another connection before failing. + void set_busy_timeout(int ms); + + bool valid() const { return impl != nullptr; } + private: std::shared_ptr impl; }; diff --git a/src/sqlite.cpp b/src/sqlite.cpp index 823d74c3a77..5dcdd247150 100644 --- a/src/sqlite.cpp +++ b/src/sqlite.cpp @@ -1,7 +1,7 @@ /* * The MIT License (MIT) * - * Copyright (c) 2015-2023 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2015-2026 Advanced Micro Devices, Inc. All rights reserved. * * Permission is hereby granted, free of charge, to any person obtaining a copy * of this software and associated documentation files (the "Software"), to deal @@ -36,12 +36,21 @@ using sqlite3_ptr = MIGRAPHX_MANAGE_PTR(sqlite3*, sqlite3_close); struct sqlite_impl { sqlite3* get() const { return ptr.get(); } - void open(const fs::path& p, int flags) + + // Returns false rather than throwing, so callers that treat an unusable database as + // "no cache" do not have to catch. sqlite3_open_v2 hands back a handle even on failure + // (that is where the error message lives), so `ptr` takes ownership either way. + bool try_open(const fs::path& p, int flags) { sqlite3* ptr_tmp = nullptr; int rc = sqlite3_open_v2(p.string().c_str(), &ptr_tmp, flags, nullptr); ptr = sqlite3_ptr{ptr_tmp}; - if(rc != 0) + return rc == 0; + } + + void open(const fs::path& p, int flags) + { + if(not try_open(p, flags)) MIGRAPHX_THROW("error opening " + p.string() + ": " + error_message()); } @@ -74,6 +83,19 @@ struct sqlite_impl sqlite3_ptr ptr; }; +using sqlite3_stmt_ptr = MIGRAPHX_MANAGE_PTR(sqlite3_stmt*, sqlite3_finalize); + +struct sqlite_stmt_impl +{ + sqlite3_stmt* get() const { return ptr.get(); } + std::string error_message() const { return db->error_message(); } + + // Holding the connection keeps it alive for as long as any statement prepared on it, + // since finalizing after the connection is closed is undefined. + std::shared_ptr db; + sqlite3_stmt_ptr ptr; +}; + sqlite sqlite::read(const fs::path& p) { sqlite r; @@ -91,6 +113,16 @@ sqlite sqlite::write(const fs::path& p) return r; } +optional sqlite::try_write(const fs::path& p) +{ + sqlite r; + r.impl = std::make_shared(); + // Using '+' instead of bitwise '|' to avoid compilation warning + if(not r.impl->try_open(p, SQLITE_OPEN_READWRITE + SQLITE_OPEN_CREATE)) + return nullopt; + return r; +} + std::vector> sqlite::execute(const std::string& s) { std::vector> result; @@ -108,5 +140,95 @@ std::vector> sqlite::execute(const return result; } +sqlite_stmt sqlite::prepare(const std::string& s) +{ + sqlite3_stmt* stmt_tmp = nullptr; + int rc = sqlite3_prepare_v2(impl->get(), s.c_str(), -1, &stmt_tmp, nullptr); + sqlite_stmt result; + result.impl = std::make_shared(); + result.impl->db = impl; + result.impl->ptr = sqlite3_stmt_ptr{stmt_tmp}; + if(rc != SQLITE_OK) + MIGRAPHX_THROW("error preparing '" + s + "': " + impl->error_message()); + return result; +} + +void sqlite::set_busy_timeout(int ms) { sqlite3_busy_timeout(impl->get(), ms); } + +sqlite_stmt& sqlite_stmt::bind(int i, std::string_view s) +{ + // A null pointer binds SQL NULL rather than an empty string, and a default-constructed + // string_view has null data(), so empty input needs its own case. + int rc = s.empty() ? sqlite3_bind_text64(impl->get(), i, "", 0, SQLITE_STATIC, SQLITE_UTF8) + : sqlite3_bind_text64( + impl->get(), i, s.data(), s.size(), SQLITE_TRANSIENT, SQLITE_UTF8); + if(rc != SQLITE_OK) + MIGRAPHX_THROW(impl->error_message()); + return *this; +} + +sqlite_stmt& sqlite_stmt::bind(int i, std::int64_t x) +{ + int rc = sqlite3_bind_int64(impl->get(), i, x); + if(rc != SQLITE_OK) + MIGRAPHX_THROW(impl->error_message()); + return *this; +} + +sqlite_stmt& sqlite_stmt::bind(int i, const std::vector& blob) +{ + // As with text, an empty vector's data() may be null, which would bind SQL NULL; a + // zero-length zeroblob is an empty BLOB instead. bind_blob64 is used because the + // non-64 form takes the size as an int. + int rc = blob.empty() + ? sqlite3_bind_zeroblob(impl->get(), i, 0) + : sqlite3_bind_blob64( + impl->get(), i, blob.data(), blob.size(), SQLITE_TRANSIENT); + if(rc != SQLITE_OK) + MIGRAPHX_THROW(impl->error_message()); + return *this; +} + +bool sqlite_stmt::step() +{ + int rc = sqlite3_step(impl->get()); + if(rc == SQLITE_ROW) + return true; + if(rc == SQLITE_DONE) + return false; + MIGRAPHX_THROW(impl->error_message()); +} + +void sqlite_stmt::reset() noexcept +{ + if(impl == nullptr) + return; + // The return of sqlite3_reset is the error from the preceding step(), which the caller + // has already seen as a throw. There is nothing new to report, and this must not throw. + (void)sqlite3_reset(impl->get()); + (void)sqlite3_clear_bindings(impl->get()); +} + +std::string sqlite_stmt::column_text(int i) const +{ + const auto* text = sqlite3_column_text(impl->get(), i); + int n = sqlite3_column_bytes(impl->get(), i); + if(text == nullptr or n <= 0) + return {}; + // NOLINTNEXTLINE(cppcoreguidelines-pro-type-reinterpret-cast) + return {reinterpret_cast(text), static_cast(n)}; +} + +std::vector sqlite_stmt::column_blob(int i) const +{ + // sqlite3_column_blob must be called before sqlite3_column_bytes: the other order can + // force a type conversion that invalidates the pointer. + const auto* data = static_cast(sqlite3_column_blob(impl->get(), i)); + int n = sqlite3_column_bytes(impl->get(), i); + if(data == nullptr or n <= 0) + return {}; + return {data, data + n}; +} + } // namespace MIGRAPHX_INLINE_NS } // namespace migraphx diff --git a/src/targets/gpu/CMakeLists.txt b/src/targets/gpu/CMakeLists.txt index 5924f0e3097..0d60bef9584 100644 --- a/src/targets/gpu/CMakeLists.txt +++ b/src/targets/gpu/CMakeLists.txt @@ -235,6 +235,7 @@ add_library(migraphx_gpu device_description.cpp device_name.cpp eliminate_data_type_for_gpu.cpp + file_binary_cache.cpp fixed_pad.cpp fuse_ck.cpp fuse_mlir.cpp @@ -262,6 +263,7 @@ add_library(migraphx_gpu problem_cache.cpp rocblas.cpp schedule_model.cpp + sqlite_binary_cache.cpp sqlite_problem_cache.cpp sync_device.cpp target.cpp diff --git a/src/targets/gpu/binary_cache.cpp b/src/targets/gpu/binary_cache.cpp index f07862c6b27..c896c3ecb9b 100644 --- a/src/targets/gpu/binary_cache.cpp +++ b/src/targets/gpu/binary_cache.cpp @@ -22,16 +22,15 @@ * THE SOFTWARE. */ #include +#include +#include #include #include -#include -#include #include #include #include #include #include -#include #include #include @@ -95,69 +94,65 @@ static std::string device_dir(const context& ctx) "_wf" + std::to_string(device.get_wavefront_size()); } -/// Where an entry lives, or an empty path when the toolchain cannot be identified and entries -/// from different toolchains would be indistinguishable. -static fs::path entry_path(const fs::path& root, const context& ctx, const std::string& key) +const std::string& binary_cache::version_stamp() { - const auto& version = binary_cache::version_dir(); - if(version.empty()) - return {}; - return root / version / device_dir(ctx) / (md5(key) + ".mxr"); -} - -/// Publish by rename so a reader never sees a half-written file. The temporary stays beside -/// the destination since the rename is only atomic within one filesystem. -static void write_atomically(const fs::path& dest, const std::vector& content) -{ - tmp_dir td{"cache", dest.parent_path()}; - auto tmp = td.path / dest.filename(); - write_buffer(tmp, content); - fs::rename(tmp, dest); -} - -/// Record what this build is, so a directory full of hashes can be identified later. -static void write_stamp(const fs::path& dir) -{ - auto stamp = dir / "cache.info"; - if(fs::exists(stamp)) - return; - std::stringstream ss; - ss << "format: " << binary_cache_format << "\n"; - ss << "hip: " << hip_compiler_version().version << "\n"; - ss << "kernels: " << kernels_digest() << "\n"; - ss << "rocmlir: " << rocmlir_id << "\n"; - auto s = ss.str(); - write_atomically(stamp, std::vector(s.begin(), s.end())); + static const std::string stamp = [] { + std::stringstream ss; + ss << "format: " << binary_cache_format << "\n"; + ss << "hip: " << hip_compiler_version().version << "\n"; + ss << "kernels: " << kernels_digest() << "\n"; + ss << "rocmlir: " << rocmlir_id << "\n"; + return ss.str(); + }(); + return stamp; } -/// Read the entry for a key off disk. Any failure is just a miss, so a damaged entry costs a -/// recompile and is written over. -static optional -read_entry(const fs::path& root, const context& ctx, const std::string& key) +/// Turn a stored blob back into an entry. Any failure is just a miss, so a damaged entry costs +/// a recompile and is written over. Shared by every backend, so they all tolerate corruption +/// and survive a hash collision the same way. +static optional decode_entry(const std::vector& blob, + const std::string& key) { - if(root.empty()) - return nullopt; - auto path = entry_path(root, ctx, key); - if(path.empty() or not fs::exists(path)) - return nullopt; binary_cache::entry e; try { - migraphx::from_value(from_msgpack(read_buffer(path)), e); + migraphx::from_value(from_msgpack(blob), e); } catch(const std::exception& ex) { - log::warn() << "Ignoring unreadable binary cache entry " << path << ": " << ex.what(); + log::warn() << "Ignoring unreadable binary cache entry " << md5(key) << ": " << ex.what(); return nullopt; } + // Entries are addressed by a hash of the key, so the full key is checked here to make a + // collision a miss rather than a wrong kernel. if(e.key != key) { - log::warn() << "Ignoring binary cache entry with mismatched key: " << path; + log::warn() << "Ignoring binary cache entry with mismatched key: " << md5(key); return nullopt; } return e; } +// Select the storage backend by file type, matching make_problem_cache_backend: a ".db"/".sqlite" +// path is a SQLite database, anything else is a directory of entries. +static optional make_binary_cache_backend(const std::string& path) +{ + if(path.empty()) + return nullopt; + if(ends_with(path, ".db") or ends_with(path, ".sqlite")) + return sqlite_binary_cache::open(path); // nullopt when the database is unusable + return binary_cache_backend{file_binary_cache{path}}; +} + +// Nothing can be persisted safely when the compiler cannot be identified, since entries from +// different toolchains would be indistinguishable. That is a property of the cache rather than +// of the storage medium, so it is checked here instead of in each backend. +binary_cache::binary_cache(binary_cache_settings s) : settings(std::move(s)) +{ + if(not version_dir().empty()) + backend = make_binary_cache_backend(settings.path); +} + optional binary_cache::get(const context& ctx, const std::string& key) { if(key.empty()) @@ -168,14 +163,21 @@ optional binary_cache::get(const context& ctx, const std::string& counters.reused++; return it->second; } - auto e = read_entry(settings.path, ctx, key); - if(not e.has_value()) + if(backend.has_value()) { - counters.misses++; - return nullopt; + auto blob = backend->load(version_dir(), device_dir(ctx), md5(key)); + if(blob.has_value()) + { + auto e = decode_entry(*blob, key); + if(e.has_value()) + { + counters.hits++; + return memo.emplace(key, std::move(e->code)).first->second; + } + } } - counters.hits++; - return memo.emplace(key, std::move(e->code)).first->second; + counters.misses++; + return nullopt; } void binary_cache::insert(const context& ctx, entry e) @@ -183,21 +185,18 @@ void binary_cache::insert(const context& ctx, entry e) if(e.key.empty()) return; counters.compiled++; - const auto& root = settings.path; - auto path = root.empty() ? fs::path{} : entry_path(root, ctx, e.key); - if(not path.empty()) + if(backend.has_value()) { - // The content is decided entirely by the key, so a writer that loses the publish race - // replaces the file with the same bytes and no locking is needed. try { - fs::create_directories(path.parent_path()); - write_stamp(fs::path(root) / version_dir()); - write_atomically(path, to_msgpack(migraphx::to_value(e))); + // Serializing inside the guard, rather than in the argument list, keeps a failure + // here a warning like any other storage failure instead of escaping insert(). + auto blob = to_msgpack(migraphx::to_value(e)); + backend->store(version_dir(), device_dir(ctx), md5(e.key), e, blob); } catch(const std::exception& ex) { - log::warn() << "Failed to write binary cache entry " << path << ": " << ex.what(); + log::warn() << "Failed to store binary cache entry " << md5(e.key) << ": " << ex.what(); } } memo[std::move(e.key)] = std::move(e.code); diff --git a/src/targets/gpu/file_binary_cache.cpp b/src/targets/gpu/file_binary_cache.cpp new file mode 100644 index 00000000000..92adb413c62 --- /dev/null +++ b/src/targets/gpu/file_binary_cache.cpp @@ -0,0 +1,112 @@ +/* + * The MIT License (MIT) + * + * Copyright (c) 2015-2026 Advanced Micro Devices, Inc. All rights reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy + * of this software and associated documentation files (the "Software"), to deal + * in the Software without restriction, including without limitation the rights + * to use, copy, modify, merge, publish, distribute, sublicense, and/or sell + * copies of the Software, and to permit persons to whom the Software is + * furnished to do so, subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in + * all copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN + * THE SOFTWARE. + * + */ +#include +#include +#include +#include +#include +#include +#include + +namespace migraphx { +inline namespace MIGRAPHX_INLINE_NS { +namespace gpu { + +static_assert(std::is_constructible{}, + "file_binary_cache must satisfy the binary_cache_backend concept"); + +/// Where an entry lives. The caller guarantees a non-empty version, so entries compiled by +/// different toolchains can never land on the same path. +static fs::path entry_path(const fs::path& root, + const std::string& version, + const std::string& device, + const std::string& key_hash) +{ + return root / version / device / (key_hash + ".mxr"); +} + +/// Publish by rename so a reader never sees a half-written file. The temporary stays beside +/// the destination since the rename is only atomic within one filesystem. +static void write_atomically(const fs::path& dest, const std::vector& content) +{ + tmp_dir td{"cache", dest.parent_path()}; + auto tmp = td.path / dest.filename(); + write_buffer(tmp, content); + fs::rename(tmp, dest); +} + +/// Record what this build is, so a directory full of hashes can be identified later. +static void write_stamp(const fs::path& dir) +{ + auto stamp = dir / "cache.info"; + if(fs::exists(stamp)) + return; + const auto& s = binary_cache::version_stamp(); + write_atomically(stamp, std::vector(s.begin(), s.end())); +} + +optional> file_binary_cache::load(const std::string& version, + const std::string& device, + const std::string& key_hash) +{ + auto path = entry_path(root, version, device, key_hash); + try + { + if(not fs::exists(path)) + return nullopt; + return read_buffer(path); + } + catch(const std::exception& ex) + { + // An unreadable entry is a miss, which costs a recompile and nothing else. + log::warn() << "Failed to read binary cache entry " << path << ": " << ex.what(); + return nullopt; + } +} + +void file_binary_cache::store(const std::string& version, + const std::string& device, + const std::string& key_hash, + const binary_cache_entry&, + const std::vector& blob) +{ + auto path = entry_path(root, version, device, key_hash); + // The content is decided entirely by the key, so a writer that loses the publish race + // replaces the file with the same bytes and no locking is needed. + try + { + fs::create_directories(path.parent_path()); + write_stamp(root / version); + write_atomically(path, blob); + } + catch(const std::exception& ex) + { + log::warn() << "Failed to write binary cache entry " << path << ": " << ex.what(); + } +} + +} // namespace gpu +} // namespace MIGRAPHX_INLINE_NS +} // namespace migraphx diff --git a/src/targets/gpu/include/migraphx/gpu/binary_cache.hpp b/src/targets/gpu/include/migraphx/gpu/binary_cache.hpp index 21e3b4464ac..f928d091aee 100644 --- a/src/targets/gpu/include/migraphx/gpu/binary_cache.hpp +++ b/src/targets/gpu/include/migraphx/gpu/binary_cache.hpp @@ -26,6 +26,8 @@ #include #include +#include +#include #include #include #include @@ -66,33 +68,16 @@ struct binary_cache_settings */ struct MIGRAPHX_GPU_EXPORT binary_cache { - /// What gets written to disk for one compiled kernel. The op name, problem and solution are - /// stored for offline inspection; only the key is checked when an entry is loaded. - struct entry - { - std::string key = {}; - std::string op_name = {}; - value problem = {}; - value solution = {}; - compiled_code code = {}; - - template - static auto reflect(Self& self, F f) - { - return pack(f(self.key, "key"), - f(self.op_name, "op_name"), - f(self.problem, "problem"), - f(self.solution, "solution"), - f(self.code, "code")); - } - }; + /// What gets stored for one compiled kernel. Defined in binary_cache_entry.hpp so the + /// storage backends can name it; the alias keeps binary_cache::entry working. + using entry = binary_cache_entry; /// Counts of what the cache did. struct stats { /// Served from memory, from an earlier compile or disk read in this process. std::size_t reused = 0; - /// Served from the cache directory. + /// Served from the storage backend. std::size_t hits = 0; /// Not found, so the caller had to compile. std::size_t misses = 0; @@ -100,7 +85,7 @@ struct MIGRAPHX_GPU_EXPORT binary_cache std::size_t compiled = 0; }; - explicit binary_cache(binary_cache_settings s = {}) : settings(std::move(s)) {} + explicit binary_cache(binary_cache_settings s = {}); /// Look up a key, consulting memory first and then the cache directory. optional get(const context& ctx, const std::string& key); @@ -119,9 +104,15 @@ struct MIGRAPHX_GPU_EXPORT binary_cache /// since entries from different compilers could not be told apart. static const std::string& version_dir(); + /// A human-readable description of what version_dir() encodes, for backends that can store + /// a self-describing marker alongside their entries. + static const std::string& version_stamp(); + private: std::unordered_map memo; binary_cache_settings settings; + /// Where entries are persisted, or empty for a memory-only cache. + optional backend; stats counters; }; diff --git a/src/targets/gpu/include/migraphx/gpu/binary_cache_backend.hpp b/src/targets/gpu/include/migraphx/gpu/binary_cache_backend.hpp new file mode 100644 index 00000000000..f4a9938df17 --- /dev/null +++ b/src/targets/gpu/include/migraphx/gpu/binary_cache_backend.hpp @@ -0,0 +1,399 @@ +/* + * The MIT License (MIT) + * + * Copyright (c) 2015-2026 Advanced Micro Devices, Inc. All rights reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy + * of this software and associated documentation files (the "Software"), to deal + * in the Software without restriction, including without limitation the rights + * to use, copy, modify, merge, publish, distribute, sublicense, and/or sell + * copies of the Software, and to permit persons to whom the Software is + * furnished to do so, subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in + * all copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN + * THE SOFTWARE. + */ +// +// te.py DSL for migraphx::gpu::binary_cache_backend. +// +// The generated header lives at +// src/targets/gpu/include/migraphx/gpu/binary_cache_backend.hpp; regenerate it +// with `cd tools && python generate.py` (generate_all routes include/gpu/ inputs +// into the gpu target tree). Do not edit the generated header by hand. +// +// Any type T satisfies the binary_cache_backend concept if it provides the +// member functions listed below. The wrapper holds T by shared_ptr and forwards +// each call through a virtual dispatch, matching problem_cache_backend. +// +// Notes: +// * binary_cache_entry is defined in ; +// the include below pulls in its full definition. +// * Backends typically own non-trivial resources (a cache directory, a SQLite +// connection) and are not meaningfully copyable beyond shared ownership. +// * Both members are non-const: binary_cache::get and insert are themselves +// non-const, so nothing forces a const qualifier here, and a backend holding +// prepared statements needs the mutability. +// +#ifndef MIGRAPHX_GUARD_GPU_BINARY_CACHE_BACKEND_HPP +#define MIGRAPHX_GUARD_GPU_BINARY_CACHE_BACKEND_HPP + +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include + +namespace migraphx { +inline namespace MIGRAPHX_INLINE_NS { +namespace gpu { + +#ifdef DOXYGEN + +/// Type-erased interface for binary-cache storage backends. +/// +/// A backend persists serialized binary_cache_entry blobs to some medium (a +/// directory of files, a SQLite database, an in-memory map for tests). Entries +/// are addressed by three strings the caller has already computed: +/// +/// * `version` -- binary_cache::version_dir(), identifying the toolchain and +/// the embedded kernel sources that produced the entry. Never empty; the +/// caller skips persistence entirely when it is. +/// * `device` -- the GPU the entry was compiled for. +/// * `key_hash` -- md5 of the compile key. A hash rather than the key itself +/// because a file backend needs a short name; a collision is harmless, +/// since the caller re-checks the full key against the decoded entry. +/// +/// Together these three form the identity of an entry. A backend must keep +/// entries with different scopes distinct rather than overwriting across them. +struct binary_cache_backend +{ + /// Return the serialized entry for this key, or nullopt for a miss. + /// + /// nullopt also covers every failure: a missing file, an unreadable + /// database, a permissions problem. A cache that cannot be read is not an + /// error, it is a cache miss, and the caller recompiles. + /// + /// Must not throw. + optional> + load(const std::string& version, const std::string& device, const std::string& key_hash); + + /// Persist `blob`, the msgpack encoding of `e`, under this key. + /// + /// `e` is passed alongside `blob` so a backend may denormalize op_name, + /// problem and solution into queryable columns. Those fields are also + /// inside `blob`, which stays the authoritative record -- a backend that + /// stores them separately must still be able to answer a load() with the + /// blob alone. + /// + /// Overwriting an existing entry is expected and safe: the content is + /// decided entirely by the key, so a writer that loses a race replaces the + /// entry with equivalent bytes. + /// + /// Must not throw. A failure to store costs a recompile next run, nothing + /// more, and the caller still keeps the result in memory. + void store(const std::string& version, + const std::string& device, + const std::string& key_hash, + const binary_cache_entry& e, + const std::vector& blob); +}; + +#else + +#ifdef TYPE_ERASED_DECLARATION + +// Type-erased interface for: +struct MIGRAPHX_EXPORT binary_cache_backend +{ + // + optional> + load(const std::string& version, const std::string& device, const std::string& key_hash); + // + void store(const std::string& version, + const std::string& device, + const std::string& key_hash, + const binary_cache_entry& e, + const std::vector& blob); +}; + +#else +// NOLINTBEGIN(performance-unnecessary-value-param) +struct binary_cache_backend +{ + private: + template + struct private_te_unwrap_reference + { + using type = PrivateDetailTypeErasedT; + }; + template + struct private_te_unwrap_reference> + { + using type = PrivateDetailTypeErasedT; + }; + template + using private_te_pure = typename std::remove_cv< + typename std::remove_reference::type>::type; + + template + using private_te_constraints_impl = + decltype(std::declval().load(std::declval(), + std::declval(), + std::declval()), + std::declval().store( + std::declval(), + std::declval(), + std::declval(), + std::declval(), + std::declval&>()), + void()); + + template + using private_te_constraints = private_te_constraints_impl< + typename private_te_unwrap_reference>::type>; + + public: + // Constructors + binary_cache_backend() = default; + + template , + typename = typename std::enable_if< + not std::is_same, + binary_cache_backend>{}>::type> + binary_cache_backend(PrivateDetailTypeErasedT&& value) + : private_detail_te_handle_mem_var( + std::make_shared< + private_detail_te_handle_type>>( + std::forward(value))) + { + } + + // Assignment + template , + typename = typename std::enable_if< + not std::is_same, + binary_cache_backend>{}>::type> + binary_cache_backend& operator=(PrivateDetailTypeErasedT && value) + { + using std::swap; + auto* derived = this->any_cast>(); + if(derived and private_detail_te_handle_mem_var.use_count() == 1) + { + *derived = std::forward(value); + } + else + { + binary_cache_backend rhs(value); + swap(private_detail_te_handle_mem_var, rhs.private_detail_te_handle_mem_var); + } + return *this; + } + + // Cast + template + PrivateDetailTypeErasedT* any_cast() + { + return this->type_id() == typeid(PrivateDetailTypeErasedT) + ? std::addressof(static_cast::type>&>( + private_detail_te_get_handle()) + .private_detail_te_value) + : nullptr; + } + + template + const typename std::remove_cv::type* any_cast() const + { + return this->type_id() == typeid(PrivateDetailTypeErasedT) + ? std::addressof(static_cast::type>&>( + private_detail_te_get_handle()) + .private_detail_te_value) + : nullptr; + } + + const std::type_info& type_id() const + { + if(private_detail_te_handle_empty()) + return typeid(std::nullptr_t); + else + return private_detail_te_get_handle().type(); + } + + optional> + load(const std::string& version, const std::string& device, const std::string& key_hash) + { + assert((*this).private_detail_te_handle_mem_var); + return (*this).private_detail_te_get_handle().load(version, device, key_hash); + } + + void store(const std::string& version, + const std::string& device, + const std::string& key_hash, + const binary_cache_entry& e, + const std::vector& blob) + { + assert((*this).private_detail_te_handle_mem_var); + (*this).private_detail_te_get_handle().store(version, device, key_hash, e, blob); + } + + friend bool is_shared(const binary_cache_backend& private_detail_x, + const binary_cache_backend& private_detail_y) + { + return private_detail_x.private_detail_te_handle_mem_var == + private_detail_y.private_detail_te_handle_mem_var; + } + + private: + struct private_detail_te_handle_base_type + { + virtual ~private_detail_te_handle_base_type() {} + virtual std::shared_ptr clone() const = 0; + virtual const std::type_info& type() const = 0; + + virtual optional> load(const std::string& version, + const std::string& device, + const std::string& key_hash) = 0; + virtual void store(const std::string& version, + const std::string& device, + const std::string& key_hash, + const binary_cache_entry& e, + const std::vector& blob) = 0; + }; + + template + struct private_detail_te_handle_type : private_detail_te_handle_base_type + { + template + private_detail_te_handle_type( + PrivateDetailTypeErasedT value, + typename std::enable_if{}>::type* = nullptr) + : private_detail_te_value(value) + { + } + + template + private_detail_te_handle_type( + PrivateDetailTypeErasedT value, + typename std::enable_if{}, int>::type* = + nullptr) noexcept + : private_detail_te_value(std::move(value)) + { + } + + std::shared_ptr clone() const override + { + return std::make_shared(private_detail_te_value); + } + + const std::type_info& type() const override { return typeid(private_detail_te_value); } + + optional> load(const std::string& version, + const std::string& device, + const std::string& key_hash) override + { + + return private_detail_te_value.load(version, device, key_hash); + } + + void store(const std::string& version, + const std::string& device, + const std::string& key_hash, + const binary_cache_entry& e, + const std::vector& blob) override + { + + private_detail_te_value.store(version, device, key_hash, e, blob); + } + + PrivateDetailTypeErasedT private_detail_te_value; + }; + + template + struct private_detail_te_handle_type> + : private_detail_te_handle_type + { + private_detail_te_handle_type(std::reference_wrapper ref) + : private_detail_te_handle_type(ref.get()) + { + } + }; + + bool private_detail_te_handle_empty() const + { + return private_detail_te_handle_mem_var == nullptr; + } + + const private_detail_te_handle_base_type& private_detail_te_get_handle() const + { + assert(private_detail_te_handle_mem_var != nullptr); + return *private_detail_te_handle_mem_var; + } + + private_detail_te_handle_base_type& private_detail_te_get_handle() + { + assert(private_detail_te_handle_mem_var != nullptr); + if(private_detail_te_handle_mem_var.use_count() > 1) + private_detail_te_handle_mem_var = private_detail_te_handle_mem_var->clone(); + return *private_detail_te_handle_mem_var; + } + + std::shared_ptr private_detail_te_handle_mem_var; +}; + +template +inline const ValueType* any_cast(const binary_cache_backend* x) +{ + return x->any_cast(); +} + +template +inline ValueType* any_cast(binary_cache_backend* x) +{ + return x->any_cast(); +} + +template +inline ValueType& any_cast(binary_cache_backend& x) +{ + auto* y = x.any_cast::type>(); + if(y == nullptr) + throw std::bad_cast(); + return *y; +} + +template +inline const ValueType& any_cast(const binary_cache_backend& x) +{ + const auto* y = x.any_cast::type>(); + if(y == nullptr) + throw std::bad_cast(); + return *y; +} +// NOLINTEND(performance-unnecessary-value-param) +#endif + +#endif + +} // namespace gpu +} // namespace MIGRAPHX_INLINE_NS +} // namespace migraphx + +#endif // MIGRAPHX_GUARD_GPU_BINARY_CACHE_BACKEND_HPP diff --git a/src/targets/gpu/include/migraphx/gpu/binary_cache_entry.hpp b/src/targets/gpu/include/migraphx/gpu/binary_cache_entry.hpp new file mode 100644 index 00000000000..9cbf68a59fa --- /dev/null +++ b/src/targets/gpu/include/migraphx/gpu/binary_cache_entry.hpp @@ -0,0 +1,68 @@ +/* + * The MIT License (MIT) + * + * Copyright (c) 2015-2026 Advanced Micro Devices, Inc. All rights reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy + * of this software and associated documentation files (the "Software"), to deal + * in the Software without restriction, including without limitation the rights + * to use, copy, modify, merge, publish, distribute, sublicense, and/or sell + * copies of the Software, and to permit persons to whom the Software is + * furnished to do so, subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in + * all copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN + * THE SOFTWARE. + * + */ +#ifndef MIGRAPHX_GUARD_GPU_BINARY_CACHE_ENTRY_HPP +#define MIGRAPHX_GUARD_GPU_BINARY_CACHE_ENTRY_HPP + +#include +#include +#include +#include +#include + +namespace migraphx { +inline namespace MIGRAPHX_INLINE_NS { +namespace gpu { + +/// What gets stored for one compiled kernel. The op name, problem and solution are stored for +/// offline inspection; only the key is checked when an entry is loaded. +/// +/// This lives in its own header, rather than nested inside binary_cache, so that the +/// binary_cache_backend interface can name it without including binary_cache.hpp -- which in +/// turn includes the backend header. Same reason cache_device_key.hpp exists for the problem +/// cache. binary_cache::entry remains an alias for it. +struct binary_cache_entry +{ + std::string key = {}; + std::string op_name = {}; + value problem = {}; + value solution = {}; + compiled_code code = {}; + + template + static auto reflect(Self& self, F f) + { + return pack(f(self.key, "key"), + f(self.op_name, "op_name"), + f(self.problem, "problem"), + f(self.solution, "solution"), + f(self.code, "code")); + } +}; + +} // namespace gpu +} // namespace MIGRAPHX_INLINE_NS +} // namespace migraphx + +#endif // MIGRAPHX_GUARD_GPU_BINARY_CACHE_ENTRY_HPP diff --git a/src/targets/gpu/include/migraphx/gpu/file_binary_cache.hpp b/src/targets/gpu/include/migraphx/gpu/file_binary_cache.hpp new file mode 100644 index 00000000000..8e44c807570 --- /dev/null +++ b/src/targets/gpu/include/migraphx/gpu/file_binary_cache.hpp @@ -0,0 +1,60 @@ +/* + * The MIT License (MIT) + * + * Copyright (c) 2015-2026 Advanced Micro Devices, Inc. All rights reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy + * of this software and associated documentation files (the "Software"), to deal + * in the Software without restriction, including without limitation the rights + * to use, copy, modify, merge, publish, distribute, sublicense, and/or sell + * copies of the Software, and to permit persons to whom the Software is + * furnished to do so, subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in + * all copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN + * THE SOFTWARE. + * + */ +#ifndef MIGRAPHX_GUARD_GPU_FILE_BINARY_CACHE_HPP +#define MIGRAPHX_GUARD_GPU_FILE_BINARY_CACHE_HPP + +#include +#include +#include +#include +#include +#include + +namespace migraphx { +inline namespace MIGRAPHX_INLINE_NS { +namespace gpu { + +// A binary_cache_backend that keeps entries as files under a root directory, laid out +// ///.mxr, with a cache.info stamp beside each version +// directory describing the build that produced it. Path resolution is the caller's job. +struct MIGRAPHX_GPU_EXPORT file_binary_cache +{ + // binary_cache_backend concept members: + optional> + load(const std::string& version, const std::string& device, const std::string& key_hash); + void store(const std::string& version, + const std::string& device, + const std::string& key_hash, + const binary_cache_entry& e, + const std::vector& blob); + + fs::path root = {}; +}; + +} // namespace gpu +} // namespace MIGRAPHX_INLINE_NS +} // namespace migraphx + +#endif // MIGRAPHX_GUARD_GPU_FILE_BINARY_CACHE_HPP diff --git a/src/targets/gpu/include/migraphx/gpu/sqlite_binary_cache.hpp b/src/targets/gpu/include/migraphx/gpu/sqlite_binary_cache.hpp new file mode 100644 index 00000000000..9cad6f58b7b --- /dev/null +++ b/src/targets/gpu/include/migraphx/gpu/sqlite_binary_cache.hpp @@ -0,0 +1,81 @@ +/* + * The MIT License (MIT) + * + * Copyright (c) 2015-2026 Advanced Micro Devices, Inc. All rights reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy + * of this software and associated documentation files (the "Software"), to deal + * in the Software without restriction, including without limitation the rights + * to use, copy, modify, merge, publish, distribute, sublicense, and/or sell + * copies of the Software, and to permit persons to whom the Software is + * furnished to do so, subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in + * all copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN + * THE SOFTWARE. + * + */ +#ifndef MIGRAPHX_GUARD_GPU_SQLITE_BINARY_CACHE_HPP +#define MIGRAPHX_GUARD_GPU_SQLITE_BINARY_CACHE_HPP + +#include +#include +#include +#include +#include +#include +#include + +namespace migraphx { +inline namespace MIGRAPHX_INLINE_NS { +namespace gpu { + +// A binary_cache_backend that keeps entries as rows in a SQLite database, one row per +// (version, device, key_hash). The stored blob is byte-identical to what the file backend +// writes into a .mxr file, so the two are interchangeable payloads; op_name, problem and +// solution are additionally denormalized into columns so a cache can be inspected with SQL. +// +// Holding sqlite and sqlite_stmt by value does not leak the SQLite dependency into this +// target: migraphx/sqlite.hpp forward-declares both impl types and never includes sqlite3.h. +struct MIGRAPHX_GPU_EXPORT sqlite_binary_cache +{ + /// Open the database, create the schema and prepare the statements. Returns nullopt when + /// any of that fails, so an unusable database leaves the cache memory-only rather than + /// raising an error. Returns the wrapper so the caller can hand the result straight back. + static optional open(const std::string& path); + + // binary_cache_backend concept members: + optional> + load(const std::string& version, const std::string& device, const std::string& key_hash); + void store(const std::string& version, + const std::string& device, + const std::string& key_hash, + const binary_cache_entry& e, + const std::vector& blob); + + private: + /// Record what this build is, once per version per process, so a table full of hashes can + /// be identified later. The analogue of the file backend's cache.info. + void stamp_version(const std::string& version); + + sqlite db = {}; + sqlite_stmt get_stmt = {}; + sqlite_stmt store_stmt = {}; + sqlite_stmt info_stmt = {}; + /// The last version stamped, so the stamp costs one statement per process rather than one + /// per stored kernel. + std::string info_written = {}; +}; + +} // namespace gpu +} // namespace MIGRAPHX_INLINE_NS +} // namespace migraphx + +#endif // MIGRAPHX_GUARD_GPU_SQLITE_BINARY_CACHE_HPP diff --git a/src/targets/gpu/sqlite_binary_cache.cpp b/src/targets/gpu/sqlite_binary_cache.cpp new file mode 100644 index 00000000000..79662addfeb --- /dev/null +++ b/src/targets/gpu/sqlite_binary_cache.cpp @@ -0,0 +1,210 @@ +/* + * The MIT License (MIT) + * + * Copyright (c) 2015-2026 Advanced Micro Devices, Inc. All rights reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy + * of this software and associated documentation files (the "Software"), to deal + * in the Software without restriction, including without limitation the rights + * to use, copy, modify, merge, publish, distribute, sublicense, and/or sell + * copies of the Software, and to permit persons to whom the Software is + * furnished to do so, subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in + * all copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN + * THE SOFTWARE. + * + */ +#include +#include +#include +#include +#include +#include +#include + +namespace migraphx { +inline namespace MIGRAPHX_INLINE_NS { +namespace gpu { + +// Compile-time confirmation that sqlite_binary_cache satisfies the backend concept. If a method +// signature drifts, this assertion fires at the definition site rather than at some far-away +// usage. +static_assert(std::is_constructible{}, + "sqlite_binary_cache must satisfy the binary_cache_backend concept"); + +namespace { + +// How long to wait for a lock held by another process before giving up, matching rocFFT. This +// is the entire cross-process strategy: whatever still fails degrades to a recompile. +constexpr int busy_timeout_ms = 5000; + +// The table name carries the schema version, so an incompatible change is a new table that old +// binaries ignore rather than a migration. This is orthogonal to binary_cache_format, which +// versions the entry payload and reaches the row through the version column. +// +// Deliberately not WITHOUT ROWID, unlike the sibling table in sqlite_problem_cache: that clause +// stores the payload inside the index B-tree, which suits short JSON but not a whole serialized +// program fragment, which would spill into overflow chains hanging off the index. +// +// The primary key leads with version so that dropping everything belonging to a superseded +// toolchain is a range scan rather than a full table scan. Point lookups bind all three and do +// not care about the order. +constexpr const char* schema_sql = R"__migraphx__( +CREATE TABLE IF NOT EXISTS cache_v1 ( + version TEXT NOT NULL, + device TEXT NOT NULL, + key_hash TEXT NOT NULL, + op_name TEXT NOT NULL, + problem TEXT NOT NULL, + solution TEXT NOT NULL, + entry BLOB NOT NULL, + timestamp INTEGER NOT NULL, + PRIMARY KEY (version, device, key_hash) +); +CREATE TABLE IF NOT EXISTS cache_info_v1 ( + version TEXT PRIMARY KEY, + stamp TEXT NOT NULL +); +)__migraphx__"; + +constexpr const char* get_sql = + "SELECT entry FROM cache_v1 WHERE version = ?1 AND device = ?2 AND key_hash = ?3;"; + +// INSERT OR REPLACE is the analogue of the file backend's publish-by-rename: the content is +// decided entirely by the key, so two processes compiling the same kernel is benign and the +// last writer wins with equivalent bytes. The timestamp is computed by the database so it is +// consistent across writers. +constexpr const char* store_sql = + "INSERT OR REPLACE INTO cache_v1" + " (version, device, key_hash, op_name, problem, solution, entry, timestamp)" + " VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7, CAST(STRFTIME('%s','now') AS INTEGER));"; + +constexpr const char* info_sql = + "INSERT OR IGNORE INTO cache_info_v1 (version, stamp) VALUES (?1, ?2);"; + +} // namespace + +optional sqlite_binary_cache::open(const std::string& path) +{ + sqlite_binary_cache r; + try + { + // sqlite will not create missing directories, but the file backend does, so without + // this a fresh machine would silently get no cache from a path that would have worked + // had it named a directory instead of a database. + auto parent = fs::path{path}.parent_path(); + if(not parent.empty()) + fs::create_directories(parent); + auto db = sqlite::try_write(path); + if(not db.has_value()) + { + log::warn() << "Disabling the binary cache: cannot open " << path; + return nullopt; + } + r.db = std::move(*db); + r.db.set_busy_timeout(busy_timeout_ms); + (void)r.db.execute(schema_sql); + // Without a working lookup there is no cache, so this failure disables the backend. + r.get_stmt = r.db.prepare(get_sql); + } + catch(const std::exception& ex) + { + log::warn() << "Disabling the binary cache at " << path << ": " << ex.what(); + return nullopt; + } + try + { + r.store_stmt = r.db.prepare(store_sql); + r.info_stmt = r.db.prepare(info_sql); + } + catch(const std::exception& ex) + { + // A database that can be read but not written to is still worth having: reads serve + // hits and stores quietly do nothing. + log::warn() << "Binary cache at " << path << " is read-only: " << ex.what(); + } + return binary_cache_backend{std::move(r)}; +} + +void sqlite_binary_cache::stamp_version(const std::string& version) +{ + if(info_written == version or not info_stmt.valid()) + return; + try + { + sqlite_stmt_reset guard{info_stmt}; + info_stmt.bind(1, version).bind(2, binary_cache::version_stamp()); + info_stmt.step(); + info_written = version; + } + catch(const std::exception& ex) + { + // The stamp is provenance for a human reading the database later, so failing to write + // it must not stop entries being stored. Remember the version anyway so a database + // that rejects this never retries it once per kernel. + log::warn() << "Failed to stamp the binary cache: " << ex.what(); + info_written = version; + } +} + +optional> sqlite_binary_cache::load(const std::string& version, + const std::string& device, + const std::string& key_hash) +{ + if(not get_stmt.valid()) + return nullopt; + try + { + sqlite_stmt_reset guard{get_stmt}; + get_stmt.bind(1, version).bind(2, device).bind(3, key_hash); + if(not get_stmt.step()) + return nullopt; + return get_stmt.column_blob(0); + } + catch(const std::exception& ex) + { + // A cache that cannot be read is a miss, which costs a recompile and nothing else. + log::warn() << "Failed to read binary cache entry " << key_hash << ": " << ex.what(); + return nullopt; + } +} + +void sqlite_binary_cache::store(const std::string& version, + const std::string& device, + const std::string& key_hash, + const binary_cache_entry& e, + const std::vector& blob) +{ + if(not store_stmt.valid()) + return; + stamp_version(version); + try + { + // The json strings are temporaries, which is safe because binding copies immediately. + sqlite_stmt_reset guard{store_stmt}; + store_stmt.bind(1, version) + .bind(2, device) + .bind(3, key_hash) + .bind(4, e.op_name) + .bind(5, to_json_string(e.problem)) + .bind(6, to_json_string(e.solution)) + .bind(7, blob); + store_stmt.step(); + } + catch(const std::exception& ex) + { + log::warn() << "Failed to write binary cache entry " << key_hash << ": " << ex.what(); + } +} + +} // namespace gpu +} // namespace MIGRAPHX_INLINE_NS +} // namespace migraphx diff --git a/test/gpu/binary_cache.cpp b/test/gpu/binary_cache.cpp index 27cda498189..012c2f5a617 100644 --- a/test/gpu/binary_cache.cpp +++ b/test/gpu/binary_cache.cpp @@ -31,10 +31,15 @@ #include #include #include +#include +#include #include #include #include #include +#include +#include +#include #include #include #include @@ -45,6 +50,9 @@ #include #include #include +#include +#include +#include static migraphx::program pointwise_program() { @@ -60,11 +68,11 @@ static migraphx::program pointwise_program() return p; } -static migraphx::compile_options cache_options(const migraphx::fs::path& dir, bool verify = false) +static migraphx::compile_options cache_options(const std::string& path, bool verify = false) { migraphx::compile_options options; - migraphx::set_backend_options( - options, {{"binary_cache", dir.string()}, {"binary_cache_verify", verify}}); + migraphx::set_backend_options(options, + {{"binary_cache", path}, {"binary_cache_verify", verify}}); return options; } @@ -84,11 +92,47 @@ static migraphx::gpu::binary_cache::entry make_entry(const std::string& key) migraphx::gpu::binary_cache::entry e; e.key = key; e.op_name = "pointwise"; + e.problem = migraphx::value{{"shape", "float_type{4, 8}"}}; e.solution = migraphx::value{{"algo", "block"}}; e.code = make_code(); return e; } +// The storage backend is chosen by the extension of the cache path, so a case that has to hold +// for both is written once against a path and registered twice, once with each of these. +static std::string dir_path(const migraphx::tmp_dir& td) { return td.path.string(); } +static std::string db_path(const migraphx::tmp_dir& td) { return (td.path / "cache.db").string(); } + +/// The entry files a directory-backed cache has written. +static std::vector entry_files(const migraphx::fs::path& dir) +{ + std::vector result; + for(const auto& file : migraphx::fs::recursive_directory_iterator(dir)) + { + if(file.path().extension() == ".mxr") + result.push_back(file.path()); + } + return result; +} + +/// Rows in one table of a cache database. The count is aliased because sqlite::execute keys its +/// rows by column name, and an unaliased count(*) would be keyed by the text of the expression. +static std::size_t row_count(const std::string& path, const std::string& table) +{ + auto rows = migraphx::sqlite::read(path).execute("SELECT count(*) AS n FROM " + table + ";"); + if(rows.empty()) + return 0; + return std::stoul(rows.front().at("n")); +} + +/// How many entries a cache path holds, whichever backend wrote them. +static std::size_t stored_entry_count(const std::string& path) +{ + if(migraphx::ends_with(path, ".db")) + return row_count(path, "cache_v1"); + return entry_files(path).size(); +} + TEST_CASE(lookup_records_a_miss) { migraphx::gpu::context ctx; @@ -115,12 +159,11 @@ TEST_CASE(memory_lookup_records_reuse) EXPECT(cache.get_stats().misses == 0); } -// A second cache shares nothing in memory, so anything it finds came off disk. -TEST_CASE(disk_lookup_records_a_hit) +// A second cache shares nothing in memory, so anything it finds came out of storage. +static void disk_lookup_body(const std::string& path) { - migraphx::tmp_dir td{"binary-cache"}; migraphx::gpu::context ctx; - migraphx::gpu::binary_cache_settings settings{td.path.string(), false}; + migraphx::gpu::binary_cache_settings settings{path, false}; migraphx::gpu::binary_cache writer{settings}; writer.insert(ctx, make_entry("shared-key")); @@ -134,25 +177,30 @@ TEST_CASE(disk_lookup_records_a_hit) EXPECT(*found->fragment.get_main_module() == *make_code().fragment.get_main_module()); } -// A damaged entry must cost a recompile and nothing more. -TEST_CASE(corrupt_entry_is_ignored) +TEST_CASE(disk_lookup_records_a_hit) { migraphx::tmp_dir td{"binary-cache"}; + disk_lookup_body(dir_path(td)); +} + +TEST_CASE(sqlite_lookup_records_a_hit) +{ + migraphx::tmp_dir td{"binary-cache"}; + disk_lookup_body(db_path(td)); +} + +// A damaged entry must cost a recompile and nothing more. How an entry gets damaged is the only +// part of this that depends on the backend, so it comes in as a step. +static void corrupt_entry_body(const std::string& path, + const std::function& damage) +{ migraphx::gpu::context ctx; - migraphx::gpu::binary_cache_settings settings{td.path.string(), false}; + migraphx::gpu::binary_cache_settings settings{path, false}; migraphx::gpu::binary_cache writer{settings}; writer.insert(ctx, make_entry("damaged")); - std::size_t truncated = 0; - for(const auto& file : migraphx::fs::recursive_directory_iterator(td.path)) - { - if(file.path().extension() != ".mxr") - continue; - migraphx::write_buffer(file.path(), std::vector(8, 0)); - truncated++; - } - EXPECT(truncated > 0); + damage(path); migraphx::gpu::binary_cache reader{settings}; @@ -160,6 +208,25 @@ TEST_CASE(corrupt_entry_is_ignored) EXPECT(reader.get_stats().misses == 1); } +TEST_CASE(corrupt_entry_is_ignored) +{ + migraphx::tmp_dir td{"binary-cache"}; + corrupt_entry_body(dir_path(td), [](const std::string& dir) { + auto files = entry_files(dir); + EXPECT(files.size() == 1); + migraphx::write_buffer(files.front(), std::vector(8, 0)); + }); +} + +TEST_CASE(sqlite_corrupt_entry_is_ignored) +{ + migraphx::tmp_dir td{"binary-cache"}; + corrupt_entry_body(db_path(td), [](const std::string& db) { + EXPECT(row_count(db, "cache_v1") == 1); + migraphx::sqlite::write(db).execute("UPDATE cache_v1 SET entry = zeroblob(8);"); + }); +} + // Without a directory nothing reaches disk, though results are still shared in memory. TEST_CASE(no_directory_writes_nothing) { @@ -215,12 +282,11 @@ TEST_CASE(duplicate_kernels_compile_once_without_a_directory) EXPECT(cache->get_stats().reused == 1); } -// Compiling twice against the same directory has to leave entries behind and keep producing the +// Compiling twice against the same cache has to leave entries behind and keep producing the // same numbers as the reference, whichever half of the run they came from. -TEST_CASE(compiling_twice_populates_the_cache_and_matches_reference) +static void compiling_twice_body(const std::string& path) { - migraphx::tmp_dir td{"binary-cache"}; - auto options = cache_options(td.path); + auto options = cache_options(path); auto p_ref = pointwise_program(); p_ref.compile(migraphx::make_target("ref")); @@ -234,11 +300,7 @@ TEST_CASE(compiling_twice_populates_the_cache_and_matches_reference) auto warmup = pointwise_program(); warmup.compile(migraphx::make_target("gpu"), options); - auto entries = - std::count_if(migraphx::fs::recursive_directory_iterator{td.path}, - migraphx::fs::recursive_directory_iterator{}, - [](const auto& file) { return file.path().extension() == ".mxr"; }); - EXPECT(entries > 0); + EXPECT(stored_entry_count(path) > 0); auto t = migraphx::make_target("gpu"); auto p = pointwise_program(); @@ -260,12 +322,23 @@ TEST_CASE(compiling_twice_populates_the_cache_and_matches_reference) gpu_result.to_vector())); } +TEST_CASE(compiling_twice_populates_the_cache_and_matches_reference) +{ + migraphx::tmp_dir td{"binary-cache"}; + compiling_twice_body(dir_path(td)); +} + +TEST_CASE(sqlite_compiling_twice_populates_the_cache_and_matches_reference) +{ + migraphx::tmp_dir td{"binary-cache"}; + compiling_twice_body(db_path(td)); +} + // With verification on, every reused result is compiled again and compared, so a run that does // not throw is one where the keys really do capture what the compilers depend on. -TEST_CASE(verified_reuse_matches_fresh_compiles) +static void verified_reuse_body(const std::string& path) { - migraphx::tmp_dir td{"binary-cache"}; - auto options = cache_options(td.path, /* verify */ true); + auto options = cache_options(path, /* verify */ true); auto warmup = pointwise_program(); warmup.compile(migraphx::make_target("gpu"), options); @@ -274,6 +347,204 @@ TEST_CASE(verified_reuse_matches_fresh_compiles) p.compile(migraphx::make_target("gpu"), options); } +TEST_CASE(verified_reuse_matches_fresh_compiles) +{ + migraphx::tmp_dir td{"binary-cache"}; + verified_reuse_body(dir_path(td)); +} + +TEST_CASE(sqlite_verified_reuse_matches_fresh_compiles) +{ + migraphx::tmp_dir td{"binary-cache"}; + verified_reuse_body(db_path(td)); +} + +// The extension of the path picks the backend and nothing else does, so the only way to see the +// choice from outside is the artifact it leaves: a database file, or a tree of entry files. +TEST_CASE(extension_selects_the_backend) +{ + migraphx::gpu::context ctx; + + migraphx::tmp_dir dir_td{"binary-cache"}; + migraphx::gpu::binary_cache dir_cache{ + migraphx::gpu::binary_cache_settings{dir_path(dir_td), false}}; + dir_cache.insert(ctx, make_entry("in-a-directory")); + EXPECT(entry_files(dir_td.path).size() == 1); + + for(const char* name : {"cache.db", "cache.sqlite"}) + { + migraphx::tmp_dir db_td{"binary-cache"}; + auto path = (db_td.path / name).string(); + migraphx::gpu::binary_cache db_cache{migraphx::gpu::binary_cache_settings{path, false}}; + db_cache.insert(ctx, make_entry("in-a-database")); + + EXPECT(migraphx::fs::is_regular_file(path)); + EXPECT(row_count(path, "cache_v1") == 1); + EXPECT(entry_files(db_td.path).empty()); + } +} + +// The two backends are interchangeable only because they store the same bytes, so the file the +// one writes and the blob column the other fills have to compare equal. +TEST_CASE(backends_store_identical_bytes) +{ + migraphx::gpu::context ctx; + auto e = make_entry("interchange"); + + migraphx::tmp_dir dir_td{"binary-cache"}; + migraphx::gpu::binary_cache dir_cache{ + migraphx::gpu::binary_cache_settings{dir_path(dir_td), false}}; + dir_cache.insert(ctx, e); + auto files = entry_files(dir_td.path); + EXPECT(files.size() == 1); + auto from_file = migraphx::read_buffer(files.front()); + EXPECT(not from_file.empty()); + + migraphx::tmp_dir db_td{"binary-cache"}; + auto path = db_path(db_td); + migraphx::gpu::binary_cache db_cache{migraphx::gpu::binary_cache_settings{path, false}}; + db_cache.insert(ctx, e); + + auto select = migraphx::sqlite::read(path).prepare("SELECT entry FROM cache_v1;"); + EXPECT(select.step()); + EXPECT((select.column_blob(0) == from_file)); +} + +// A database that cannot be opened leaves a memory-only cache rather than an error. The parent +// component here is a regular file, so neither creating the directory nor opening the database +// can succeed. +TEST_CASE(unusable_database_degrades_to_memory) +{ + migraphx::tmp_dir td{"binary-cache"}; + migraphx::gpu::context ctx; + auto blocker = td.path / "not_a_dir"; + migraphx::write_buffer(blocker, std::vector(4, 0)); + migraphx::gpu::binary_cache_settings settings{(blocker / "cache.db").string(), false}; + + migraphx::gpu::binary_cache cache{settings}; + cache.insert(ctx, make_entry("nowhere")); + EXPECT(cache.get(ctx, "nowhere").has_value()); + EXPECT(cache.get_stats().reused == 1); + + // Nothing was persisted, so a second cache finds nothing. + migraphx::gpu::binary_cache reader{settings}; + EXPECT(not reader.get(ctx, "nowhere").has_value()); + EXPECT(reader.get_stats().misses == 1); +} + +// Two connections over one database, as two processes compiling against a shared cache would +// have. This cannot be two sqlite_binary_cache objects: only open() populates one, and it hands +// back the type-erased wrapper. +TEST_CASE(two_connections_share_a_database) +{ + migraphx::tmp_dir td{"binary-cache"}; + auto path = db_path(td); + + auto a = migraphx::gpu::sqlite_binary_cache::open(path); + auto b = migraphx::gpu::sqlite_binary_cache::open(path); + EXPECT(a.has_value()); + EXPECT(b.has_value()); + + const std::vector first{'f', 'i', 'r', 's', 't'}; + const std::vector second{'s', 'e', 'c', 'o', 'n', 'd'}; + + a->store("v", "dev", "k1", make_entry("k1"), first); + auto from_b = b->load("v", "dev", "k1"); + EXPECT(from_b.has_value()); + EXPECT((*from_b == first)); + + b->store("v", "dev", "k2", make_entry("k2"), second); + auto from_a = a->load("v", "dev", "k2"); + EXPECT(from_a.has_value()); + EXPECT((*from_a == second)); + + EXPECT(not a->load("v", "dev", "absent").has_value()); +} + +// A table of hashes says nothing about which build wrote it, so the database carries a +// description of that build -- written once, not once per kernel. +TEST_CASE(sqlite_records_the_version_stamp) +{ + migraphx::tmp_dir td{"binary-cache"}; + migraphx::gpu::context ctx; + auto path = db_path(td); + migraphx::gpu::binary_cache cache{migraphx::gpu::binary_cache_settings{path, false}}; + + cache.insert(ctx, make_entry("one")); + EXPECT(row_count(path, "cache_info_v1") == 1); + + cache.insert(ctx, make_entry("two")); + EXPECT(row_count(path, "cache_v1") == 2); + EXPECT(row_count(path, "cache_info_v1") == 1); + + auto rows = migraphx::sqlite::read(path).execute("SELECT version, stamp FROM cache_info_v1;"); + EXPECT(rows.size() == 1); + EXPECT(rows.front().at("version") == migraphx::gpu::binary_cache::version_dir()); + EXPECT(rows.front().at("stamp") == migraphx::gpu::binary_cache::version_stamp()); +} + +// version and device lead the primary key because they are what separates entries this build +// may use from entries it may not, so a row stored under one must not be served under another. +TEST_CASE(sqlite_scopes_entries_by_version_and_device) +{ + migraphx::tmp_dir td{"binary-cache"}; + auto backend = migraphx::gpu::sqlite_binary_cache::open(db_path(td)); + EXPECT(backend.has_value()); + + const std::vector blob{'p', 'a', 'y'}; + backend->store("v1", "dev1", "k", make_entry("k"), blob); + + EXPECT(backend->load("v1", "dev1", "k").has_value()); + EXPECT(not backend->load("v2", "dev1", "k").has_value()); + EXPECT(not backend->load("v1", "dev2", "k").has_value()); +} + +// Storing a key twice replaces the row rather than accumulating or failing, the way the file +// backend's publish-by-rename overwrites in place. Two processes compiling the same kernel is +// benign for exactly this reason. +TEST_CASE(sqlite_store_overwrites_in_place) +{ + migraphx::tmp_dir td{"binary-cache"}; + auto path = db_path(td); + auto backend = migraphx::gpu::sqlite_binary_cache::open(path); + EXPECT(backend.has_value()); + + const std::vector replacement{'n', 'e', 'w'}; + backend->store("v", "dev", "k", make_entry("k"), {'o', 'l', 'd'}); + backend->store("v", "dev", "k", make_entry("k"), replacement); + + EXPECT(row_count(path, "cache_v1") == 1); + auto got = backend->load("v", "dev", "k"); + EXPECT(got.has_value()); + EXPECT((*got == replacement)); +} + +// The backend layer moves opaque bytes and never decodes them, so a payload that is not even +// msgpack still round-trips. Both backends go through the same type-erased wrapper here, which +// is the runtime half of the static_assert in each backend's .cpp. +TEST_CASE(backends_round_trip_through_the_wrapper) +{ + migraphx::tmp_dir td{"binary-cache"}; + const std::vector blob{'\0', 'n', 'o', 't', '\0', 'm', 's', 'g', '\xff'}; + auto e = make_entry("opaque"); + + auto db = migraphx::gpu::sqlite_binary_cache::open((td.path / "cache.db").string()); + EXPECT(db.has_value()); + + std::vector backends; + backends.emplace_back(migraphx::gpu::file_binary_cache{td.path / "files"}); + backends.push_back(*db); + + for(auto& backend : backends) + { + EXPECT(not backend.load("v", "dev", "k").has_value()); + backend.store("v", "dev", "k", e, blob); + auto got = backend.load("v", "dev", "k"); + EXPECT(got.has_value()); + EXPECT((*got == blob)); + } +} + TEST_CASE(entry_round_trip) { auto e = make_entry("some-key"); diff --git a/test/sqlite.cpp b/test/sqlite.cpp index f29099f052d..b97b05862ab 100644 --- a/test/sqlite.cpp +++ b/test/sqlite.cpp @@ -1,7 +1,7 @@ /* * The MIT License (MIT) * - * Copyright (c) 2015-2025 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2015-2026 Advanced Micro Devices, Inc. All rights reserved. * * Permission is hereby granted, free of charge, to any person obtaining a copy * of this software and associated documentation files (the "Software"), to deal @@ -24,6 +24,9 @@ #include #include #include +#include +#include +#include TEST_CASE(read_write) { @@ -55,4 +58,78 @@ TEST_CASE(read_write) } } +TEST_CASE(prepared_blob_round_trip) +{ + // Bytes that raw SQL text cannot carry: an embedded NUL and a single quote. This is the + // reason binaries need parameter binding rather than string interpolation. + const std::vector blob{'\0', 'a', '\'', '\0', static_cast(0xff), 'z'}; + + migraphx::tmp_dir td{}; + auto db_path = td.path / "blob.db"; + { + auto db = migraphx::sqlite::write(db_path); + db.execute(R"__migraphx__( + CREATE TABLE IF NOT EXISTS blob_db ( + name TEXT PRIMARY KEY, + size INTEGER NOT NULL, + data BLOB NOT NULL + ); + )__migraphx__"); + + // One statement, two inserts: the reset/rebind path backends rely on. + auto insert = db.prepare("INSERT INTO blob_db (name, size, data) VALUES (?, ?, ?);"); + EXPECT(insert.valid()); + + insert.bind(1, std::string_view{"k1"}) + .bind(2, static_cast(blob.size())) + .bind(3, blob); + EXPECT(not insert.step()); + insert.reset(); + + insert.bind(1, std::string_view{"empty"}) + .bind(2, static_cast(0)) + .bind(3, std::vector{}); + EXPECT(not insert.step()); + insert.reset(); + } + { + auto db = migraphx::sqlite::read(db_path); + auto select = db.prepare("SELECT name, data FROM blob_db WHERE name = ?;"); + + { + migraphx::sqlite_stmt_reset guard{select}; + select.bind(1, std::string_view{"k1"}); + EXPECT(select.step()); + EXPECT(select.column_text(0) == "k1"); + EXPECT((select.column_blob(1) == blob)); + EXPECT(not select.step()); + } + { + // An empty blob must come back as an empty blob, not as NULL. + migraphx::sqlite_stmt_reset guard{select}; + select.bind(1, std::string_view{"empty"}); + EXPECT(select.step()); + EXPECT(select.column_blob(1).empty()); + } + { + migraphx::sqlite_stmt_reset guard{select}; + select.bind(1, std::string_view{"missing"}); + EXPECT(not select.step()); + } + } +} + +TEST_CASE(try_write_unusable_path) +{ + migraphx::tmp_dir td{}; + // A directory component that is really a file, so the database can never be created. + auto blocker = td.path / "not_a_dir"; + { + auto db = migraphx::sqlite::write(blocker); + db.execute("CREATE TABLE IF NOT EXISTS t (id INTEGER PRIMARY KEY ASC);"); + } + EXPECT(not migraphx::sqlite::try_write(blocker / "nested.db").has_value()); + EXPECT(migraphx::sqlite::try_write(td.path / "ok.db").has_value()); +} + int main(int argc, const char* argv[]) { test::run(argc, argv); } From 0919fc3139d6855913ee0c8a53d410b18d7a2e66 Mon Sep 17 00:00:00 2001 From: pnikolic-amd Date: Wed, 2 Sep 2026 16:17:40 +0200 Subject: [PATCH 02/14] Add hpp file --- tools/include/gpu/binary_cache_backend.hpp | 142 +++++++++++++++++++++ 1 file changed, 142 insertions(+) create mode 100644 tools/include/gpu/binary_cache_backend.hpp diff --git a/tools/include/gpu/binary_cache_backend.hpp b/tools/include/gpu/binary_cache_backend.hpp new file mode 100644 index 00000000000..80eac6146aa --- /dev/null +++ b/tools/include/gpu/binary_cache_backend.hpp @@ -0,0 +1,142 @@ +/* + * The MIT License (MIT) + * + * Copyright (c) 2015-2026 Advanced Micro Devices, Inc. All rights reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy + * of this software and associated documentation files (the "Software"), to deal + * in the Software without restriction, including without limitation the rights + * to use, copy, modify, merge, publish, distribute, sublicense, and/or sell + * copies of the Software, and to permit persons to whom the Software is + * furnished to do so, subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in + * all copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN + * THE SOFTWARE. + */ +// +// te.py DSL for migraphx::gpu::binary_cache_backend. +// +// The generated header lives at +// src/targets/gpu/include/migraphx/gpu/binary_cache_backend.hpp; regenerate it +// with `cd tools && python generate.py` (generate_all routes include/gpu/ inputs +// into the gpu target tree). Do not edit the generated header by hand. +// +// Any type T satisfies the binary_cache_backend concept if it provides the +// member functions listed below. The wrapper holds T by shared_ptr and forwards +// each call through a virtual dispatch, matching problem_cache_backend. +// +// Notes: +// * binary_cache_entry is defined in ; +// the include below pulls in its full definition. +// * Backends typically own non-trivial resources (a cache directory, a SQLite +// connection) and are not meaningfully copyable beyond shared ownership. +// * Both members are non-const: binary_cache::get and insert are themselves +// non-const, so nothing forces a const qualifier here, and a backend holding +// prepared statements needs the mutability. +// +#ifndef MIGRAPHX_GUARD_GPU_BINARY_CACHE_BACKEND_HPP +#define MIGRAPHX_GUARD_GPU_BINARY_CACHE_BACKEND_HPP + +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include + +namespace migraphx { +inline namespace MIGRAPHX_INLINE_NS { +namespace gpu { + +#ifdef DOXYGEN + +/// Type-erased interface for binary-cache storage backends. +/// +/// A backend persists serialized binary_cache_entry blobs to some medium (a +/// directory of files, a SQLite database, an in-memory map for tests). Entries +/// are addressed by three strings the caller has already computed: +/// +/// * `version` -- binary_cache::version_dir(), identifying the toolchain and +/// the embedded kernel sources that produced the entry. Never empty; the +/// caller skips persistence entirely when it is. +/// * `device` -- the GPU the entry was compiled for. +/// * `key_hash` -- md5 of the compile key. A hash rather than the key itself +/// because a file backend needs a short name; a collision is harmless, +/// since the caller re-checks the full key against the decoded entry. +/// +/// Together these three form the identity of an entry. A backend must keep +/// entries with different scopes distinct rather than overwriting across them. +struct binary_cache_backend +{ + /// Return the serialized entry for this key, or nullopt for a miss. + /// + /// nullopt also covers every failure: a missing file, an unreadable + /// database, a permissions problem. A cache that cannot be read is not an + /// error, it is a cache miss, and the caller recompiles. + /// + /// Must not throw. + optional> load(const std::string& version, + const std::string& device, + const std::string& key_hash); + + /// Persist `blob`, the msgpack encoding of `e`, under this key. + /// + /// `e` is passed alongside `blob` so a backend may denormalize op_name, + /// problem and solution into queryable columns. Those fields are also + /// inside `blob`, which stays the authoritative record -- a backend that + /// stores them separately must still be able to answer a load() with the + /// blob alone. + /// + /// Overwriting an existing entry is expected and safe: the content is + /// decided entirely by the key, so a writer that loses a race replaces the + /// entry with equivalent bytes. + /// + /// Must not throw. A failure to store costs a recompile next run, nothing + /// more, and the caller still keeps the result in memory. + void store(const std::string& version, + const std::string& device, + const std::string& key_hash, + const binary_cache_entry& e, + const std::vector& blob); +}; + +#else + +<% + interface( + 'binary_cache_backend', + virtual('load', + returns = 'optional>', + version = 'const std::string&', + device = 'const std::string&', + key_hash = 'const std::string&'), + virtual('store', + returns = 'void', + version = 'const std::string&', + device = 'const std::string&', + key_hash = 'const std::string&', + e = 'const binary_cache_entry&', + blob = 'const std::vector&')) +%> + +#endif + +} // namespace gpu +} // namespace MIGRAPHX_INLINE_NS +} // namespace migraphx + +#endif // MIGRAPHX_GUARD_GPU_BINARY_CACHE_BACKEND_HPP From eb7713fe7f071e9e87cdf4c67d86a294b0059f64 Mon Sep 17 00:00:00 2001 From: pnikolic-amd Date: Wed, 2 Sep 2026 16:44:13 +0200 Subject: [PATCH 03/14] Simplify skill --- src/sqlite.cpp | 37 ++++++++++--------- src/targets/gpu/binary_cache.cpp | 36 +++++++++++------- src/targets/gpu/file_binary_cache.cpp | 12 +++--- .../gpu/include/migraphx/gpu/binary_cache.hpp | 4 +- .../migraphx/gpu/file_binary_cache.hpp | 7 ++-- .../migraphx/gpu/sqlite_binary_cache.hpp | 12 +++--- src/targets/gpu/sqlite_binary_cache.cpp | 28 +++++++------- test/gpu/binary_cache.cpp | 22 +++++++---- 8 files changed, 88 insertions(+), 70 deletions(-) diff --git a/src/sqlite.cpp b/src/sqlite.cpp index 5dcdd247150..9a47df53ad2 100644 --- a/src/sqlite.cpp +++ b/src/sqlite.cpp @@ -91,11 +91,15 @@ struct sqlite_stmt_impl std::string error_message() const { return db->error_message(); } // Holding the connection keeps it alive for as long as any statement prepared on it, - // since finalizing after the connection is closed is undefined. + // since finalizing after the connection is closed is undefined. Declaration order is + // load-bearing: members destruct in reverse, so ptr is finalized before db is released. + // Do not reorder. std::shared_ptr db; sqlite3_stmt_ptr ptr; }; +constexpr int write_flags = SQLITE_OPEN_READWRITE | SQLITE_OPEN_CREATE; + sqlite sqlite::read(const fs::path& p) { sqlite r; @@ -108,8 +112,7 @@ sqlite sqlite::write(const fs::path& p) { sqlite r; r.impl = std::make_shared(); - // Using '+' instead of bitwise '|' to avoid compilation warning - r.impl->open(p, SQLITE_OPEN_READWRITE + SQLITE_OPEN_CREATE); + r.impl->open(p, write_flags); return r; } @@ -117,8 +120,7 @@ optional sqlite::try_write(const fs::path& p) { sqlite r; r.impl = std::make_shared(); - // Using '+' instead of bitwise '|' to avoid compilation warning - if(not r.impl->try_open(p, SQLITE_OPEN_READWRITE + SQLITE_OPEN_CREATE)) + if(not r.impl->try_open(p, write_flags)) return nullopt; return r; } @@ -143,10 +145,10 @@ std::vector> sqlite::execute(const sqlite_stmt sqlite::prepare(const std::string& s) { sqlite3_stmt* stmt_tmp = nullptr; - int rc = sqlite3_prepare_v2(impl->get(), s.c_str(), -1, &stmt_tmp, nullptr); + int rc = sqlite3_prepare_v2(impl->get(), s.c_str(), -1, &stmt_tmp, nullptr); sqlite_stmt result; - result.impl = std::make_shared(); - result.impl->db = impl; + result.impl = std::make_shared(); + result.impl->db = impl; result.impl->ptr = sqlite3_stmt_ptr{stmt_tmp}; if(rc != SQLITE_OK) MIGRAPHX_THROW("error preparing '" + s + "': " + impl->error_message()); @@ -157,11 +159,12 @@ void sqlite::set_busy_timeout(int ms) { sqlite3_busy_timeout(impl->get(), ms); } sqlite_stmt& sqlite_stmt::bind(int i, std::string_view s) { - // A null pointer binds SQL NULL rather than an empty string, and a default-constructed - // string_view has null data(), so empty input needs its own case. - int rc = s.empty() ? sqlite3_bind_text64(impl->get(), i, "", 0, SQLITE_STATIC, SQLITE_UTF8) - : sqlite3_bind_text64( - impl->get(), i, s.data(), s.size(), SQLITE_TRANSIENT, SQLITE_UTF8); + // A default-constructed string_view has null data(), and a null pointer binds SQL NULL + // rather than an empty string, so empty input substitutes a valid pointer. SQLITE_TRANSIENT + // makes sqlite take its own copy before returning, which is what lets callers bind + // temporaries. + const char* text = s.empty() ? "" : s.data(); + int rc = sqlite3_bind_text64(impl->get(), i, text, s.size(), SQLITE_TRANSIENT, SQLITE_UTF8); if(rc != SQLITE_OK) MIGRAPHX_THROW(impl->error_message()); return *this; @@ -178,12 +181,12 @@ sqlite_stmt& sqlite_stmt::bind(int i, std::int64_t x) sqlite_stmt& sqlite_stmt::bind(int i, const std::vector& blob) { // As with text, an empty vector's data() may be null, which would bind SQL NULL; a - // zero-length zeroblob is an empty BLOB instead. bind_blob64 is used because the - // non-64 form takes the size as an int. + // zero-length zeroblob is an empty BLOB instead. The 64-bit form is used because the + // plain one takes the size as an int, and SQLITE_TRANSIENT copies before returning so + // callers can bind temporaries. int rc = blob.empty() ? sqlite3_bind_zeroblob(impl->get(), i, 0) - : sqlite3_bind_blob64( - impl->get(), i, blob.data(), blob.size(), SQLITE_TRANSIENT); + : sqlite3_bind_blob64(impl->get(), i, blob.data(), blob.size(), SQLITE_TRANSIENT); if(rc != SQLITE_OK) MIGRAPHX_THROW(impl->error_message()); return *this; diff --git a/src/targets/gpu/binary_cache.cpp b/src/targets/gpu/binary_cache.cpp index c896c3ecb9b..47f5e9e5ca6 100644 --- a/src/targets/gpu/binary_cache.cpp +++ b/src/targets/gpu/binary_cache.cpp @@ -110,8 +110,8 @@ const std::string& binary_cache::version_stamp() /// Turn a stored blob back into an entry. Any failure is just a miss, so a damaged entry costs /// a recompile and is written over. Shared by every backend, so they all tolerate corruption /// and survive a hash collision the same way. -static optional decode_entry(const std::vector& blob, - const std::string& key) +static optional +decode_entry(const std::vector& blob, const std::string& key, const std::string& key_hash) { binary_cache::entry e; try @@ -120,28 +120,31 @@ static optional decode_entry(const std::vector& blob, } catch(const std::exception& ex) { - log::warn() << "Ignoring unreadable binary cache entry " << md5(key) << ": " << ex.what(); + log::warn() << "Ignoring unreadable binary cache entry " << key_hash << ": " << ex.what(); return nullopt; } // Entries are addressed by a hash of the key, so the full key is checked here to make a // collision a miss rather than a wrong kernel. if(e.key != key) { - log::warn() << "Ignoring binary cache entry with mismatched key: " << md5(key); + log::warn() << "Ignoring binary cache entry with mismatched key: " << key_hash; return nullopt; } return e; } -// Select the storage backend by file type, matching make_problem_cache_backend: a ".db"/".sqlite" -// path is a SQLite database, anything else is a directory of entries. +// Select the storage backend by file type, the same rule make_problem_cache_backend applies in +// problem_cache.cpp: a ".db"/".sqlite" path is a SQLite database, anything else is a directory +// of entries. The version stamp is handed to the backend here so the backends do not have to +// reach back into the cache frontend for it. static optional make_binary_cache_backend(const std::string& path) { if(path.empty()) return nullopt; + const auto& stamp = binary_cache::version_stamp(); if(ends_with(path, ".db") or ends_with(path, ".sqlite")) - return sqlite_binary_cache::open(path); // nullopt when the database is unusable - return binary_cache_backend{file_binary_cache{path}}; + return sqlite_binary_cache::open(path, stamp); // nullopt when the database is unusable + return binary_cache_backend{file_binary_cache{path, stamp}}; } // Nothing can be persisted safely when the compiler cannot be identified, since entries from @@ -165,10 +168,13 @@ optional binary_cache::get(const context& ctx, const std::string& } if(backend.has_value()) { - auto blob = backend->load(version_dir(), device_dir(ctx), md5(key)); + // Hashing the key is not free -- it is the whole compile source, which runs to + // kilobytes -- so it is done once and reused for the lookup and any diagnostics. + auto key_hash = md5(key); + auto blob = backend->load(version_dir(), device_dir(ctx), key_hash); if(blob.has_value()) { - auto e = decode_entry(*blob, key); + auto e = decode_entry(*blob, key, key_hash); if(e.has_value()) { counters.hits++; @@ -187,16 +193,18 @@ void binary_cache::insert(const context& ctx, entry e) counters.compiled++; if(backend.has_value()) { + auto key_hash = md5(e.key); try { - // Serializing inside the guard, rather than in the argument list, keeps a failure - // here a warning like any other storage failure instead of escaping insert(). + // Serializing inside the try, rather than in the call's argument list, makes a + // failure here a warning like any other storage failure instead of escaping + // insert() and failing the compile. auto blob = to_msgpack(migraphx::to_value(e)); - backend->store(version_dir(), device_dir(ctx), md5(e.key), e, blob); + backend->store(version_dir(), device_dir(ctx), key_hash, e, blob); } catch(const std::exception& ex) { - log::warn() << "Failed to store binary cache entry " << md5(e.key) << ": " << ex.what(); + log::warn() << "Failed to store binary cache entry " << key_hash << ": " << ex.what(); } } memo[std::move(e.key)] = std::move(e.code); diff --git a/src/targets/gpu/file_binary_cache.cpp b/src/targets/gpu/file_binary_cache.cpp index 92adb413c62..0832287e82e 100644 --- a/src/targets/gpu/file_binary_cache.cpp +++ b/src/targets/gpu/file_binary_cache.cpp @@ -23,7 +23,6 @@ * */ #include -#include #include #include #include @@ -58,13 +57,12 @@ static void write_atomically(const fs::path& dest, const std::vector& cont } /// Record what this build is, so a directory full of hashes can be identified later. -static void write_stamp(const fs::path& dir) +static void write_stamp(const fs::path& dir, const std::string& stamp) { - auto stamp = dir / "cache.info"; - if(fs::exists(stamp)) + auto path = dir / "cache.info"; + if(fs::exists(path)) return; - const auto& s = binary_cache::version_stamp(); - write_atomically(stamp, std::vector(s.begin(), s.end())); + write_atomically(path, std::vector(stamp.begin(), stamp.end())); } optional> file_binary_cache::load(const std::string& version, @@ -98,7 +96,7 @@ void file_binary_cache::store(const std::string& version, try { fs::create_directories(path.parent_path()); - write_stamp(root / version); + write_stamp(root / version, stamp); write_atomically(path, blob); } catch(const std::exception& ex) diff --git a/src/targets/gpu/include/migraphx/gpu/binary_cache.hpp b/src/targets/gpu/include/migraphx/gpu/binary_cache.hpp index f928d091aee..289372fd825 100644 --- a/src/targets/gpu/include/migraphx/gpu/binary_cache.hpp +++ b/src/targets/gpu/include/migraphx/gpu/binary_cache.hpp @@ -104,8 +104,8 @@ struct MIGRAPHX_GPU_EXPORT binary_cache /// since entries from different compilers could not be told apart. static const std::string& version_dir(); - /// A human-readable description of what version_dir() encodes, for backends that can store - /// a self-describing marker alongside their entries. + /// A human-readable description of what version_dir() encodes. Handed to a backend when one + /// is constructed, so it can store a self-describing marker alongside its entries. static const std::string& version_stamp(); private: diff --git a/src/targets/gpu/include/migraphx/gpu/file_binary_cache.hpp b/src/targets/gpu/include/migraphx/gpu/file_binary_cache.hpp index 8e44c807570..b1cffc60773 100644 --- a/src/targets/gpu/include/migraphx/gpu/file_binary_cache.hpp +++ b/src/targets/gpu/include/migraphx/gpu/file_binary_cache.hpp @@ -38,10 +38,10 @@ namespace gpu { // A binary_cache_backend that keeps entries as files under a root directory, laid out // ///.mxr, with a cache.info stamp beside each version -// directory describing the build that produced it. Path resolution is the caller's job. +// directory describing the build that produced it. Path resolution and the text of the stamp +// are the caller's job. struct MIGRAPHX_GPU_EXPORT file_binary_cache { - // binary_cache_backend concept members: optional> load(const std::string& version, const std::string& device, const std::string& key_hash); void store(const std::string& version, @@ -50,7 +50,8 @@ struct MIGRAPHX_GPU_EXPORT file_binary_cache const binary_cache_entry& e, const std::vector& blob); - fs::path root = {}; + fs::path root = {}; + std::string stamp = {}; }; } // namespace gpu diff --git a/src/targets/gpu/include/migraphx/gpu/sqlite_binary_cache.hpp b/src/targets/gpu/include/migraphx/gpu/sqlite_binary_cache.hpp index 9cad6f58b7b..926d9a802d9 100644 --- a/src/targets/gpu/include/migraphx/gpu/sqlite_binary_cache.hpp +++ b/src/targets/gpu/include/migraphx/gpu/sqlite_binary_cache.hpp @@ -46,12 +46,12 @@ namespace gpu { // target: migraphx/sqlite.hpp forward-declares both impl types and never includes sqlite3.h. struct MIGRAPHX_GPU_EXPORT sqlite_binary_cache { - /// Open the database, create the schema and prepare the statements. Returns nullopt when - /// any of that fails, so an unusable database leaves the cache memory-only rather than - /// raising an error. Returns the wrapper so the caller can hand the result straight back. - static optional open(const std::string& path); + /// Open the database, create the schema and prepare the statements. `stamp` is the text + /// recorded in cache_info_v1 to describe the build. Returns nullopt when any of that fails, + /// so an unusable database leaves the cache memory-only rather than raising an error. + /// Returns the wrapper so the caller can hand the result straight back. + static optional open(const std::string& path, std::string stamp); - // binary_cache_backend concept members: optional> load(const std::string& version, const std::string& device, const std::string& key_hash); void store(const std::string& version, @@ -69,6 +69,8 @@ struct MIGRAPHX_GPU_EXPORT sqlite_binary_cache sqlite_stmt get_stmt = {}; sqlite_stmt store_stmt = {}; sqlite_stmt info_stmt = {}; + /// What to write into cache_info_v1; supplied by the caller. + std::string stamp = {}; /// The last version stamped, so the stamp costs one statement per process rather than one /// per stored kernel. std::string info_written = {}; diff --git a/src/targets/gpu/sqlite_binary_cache.cpp b/src/targets/gpu/sqlite_binary_cache.cpp index 79662addfeb..62633966e25 100644 --- a/src/targets/gpu/sqlite_binary_cache.cpp +++ b/src/targets/gpu/sqlite_binary_cache.cpp @@ -23,7 +23,6 @@ * */ #include -#include #include #include #include @@ -42,8 +41,8 @@ static_assert(std::is_constructible{} namespace { -// How long to wait for a lock held by another process before giving up, matching rocFFT. This -// is the entire cross-process strategy: whatever still fails degrades to a recompile. +// How long to wait for a lock held by another process before giving up. This is the entire +// cross-process strategy: whatever still fails degrades to a recompile. constexpr int busy_timeout_ms = 5000; // The table name carries the schema version, so an incompatible change is a new table that old @@ -80,8 +79,9 @@ constexpr const char* get_sql = // INSERT OR REPLACE is the analogue of the file backend's publish-by-rename: the content is // decided entirely by the key, so two processes compiling the same kernel is benign and the -// last writer wins with equivalent bytes. The timestamp is computed by the database so it is -// consistent across writers. +// last writer wins with equivalent bytes. The timestamp is computed by the database rather +// than the process so that rows written by different machines stay comparable; nothing reads +// it yet, it is there to make pruning an old cache by age possible. constexpr const char* store_sql = "INSERT OR REPLACE INTO cache_v1" " (version, device, key_hash, op_name, problem, solution, entry, timestamp)" @@ -92,14 +92,14 @@ constexpr const char* info_sql = } // namespace -optional sqlite_binary_cache::open(const std::string& path) +optional sqlite_binary_cache::open(const std::string& path, std::string stamp) { sqlite_binary_cache r; + r.stamp = std::move(stamp); try { - // sqlite will not create missing directories, but the file backend does, so without - // this a fresh machine would silently get no cache from a path that would have worked - // had it named a directory instead of a database. + // sqlite will not create a missing parent directory, but the file backend does, so + // this keeps the two backends behaving the same on a fresh machine. auto parent = fs::path{path}.parent_path(); if(not parent.empty()) fs::create_directories(parent); @@ -138,20 +138,20 @@ void sqlite_binary_cache::stamp_version(const std::string& version) { if(info_written == version or not info_stmt.valid()) return; + // Recorded up front, and not again on success, so that a database which rejects the write + // is not retried once per stored kernel. + info_written = version; try { sqlite_stmt_reset guard{info_stmt}; - info_stmt.bind(1, version).bind(2, binary_cache::version_stamp()); + info_stmt.bind(1, version).bind(2, stamp); info_stmt.step(); - info_written = version; } catch(const std::exception& ex) { // The stamp is provenance for a human reading the database later, so failing to write - // it must not stop entries being stored. Remember the version anyway so a database - // that rejects this never retries it once per kernel. + // it must not stop entries being stored. log::warn() << "Failed to stamp the binary cache: " << ex.what(); - info_written = version; } } diff --git a/test/gpu/binary_cache.cpp b/test/gpu/binary_cache.cpp index 012c2f5a617..2ab40ceeeb6 100644 --- a/test/gpu/binary_cache.cpp +++ b/test/gpu/binary_cache.cpp @@ -125,14 +125,20 @@ static std::size_t row_count(const std::string& path, const std::string& table) return std::stoul(rows.front().at("n")); } -/// How many entries a cache path holds, whichever backend wrote them. +/// How many entries a cache path holds, whichever backend wrote them. The extensions must match +/// the ones make_binary_cache_backend routes to the SQLite backend, or this silently counts +/// files in a directory that does not exist and reports zero. static std::size_t stored_entry_count(const std::string& path) { - if(migraphx::ends_with(path, ".db")) + if(migraphx::ends_with(path, ".db") or migraphx::ends_with(path, ".sqlite")) return row_count(path, "cache_v1"); return entry_files(path).size(); } +/// Backends constructed directly take the stamp as an argument; only the test that reads +/// cache_info_v1 back cares what it says. +static const std::string test_stamp = "test-stamp\n"; + TEST_CASE(lookup_records_a_miss) { migraphx::gpu::context ctx; @@ -440,8 +446,8 @@ TEST_CASE(two_connections_share_a_database) migraphx::tmp_dir td{"binary-cache"}; auto path = db_path(td); - auto a = migraphx::gpu::sqlite_binary_cache::open(path); - auto b = migraphx::gpu::sqlite_binary_cache::open(path); + auto a = migraphx::gpu::sqlite_binary_cache::open(path, test_stamp); + auto b = migraphx::gpu::sqlite_binary_cache::open(path, test_stamp); EXPECT(a.has_value()); EXPECT(b.has_value()); @@ -488,7 +494,7 @@ TEST_CASE(sqlite_records_the_version_stamp) TEST_CASE(sqlite_scopes_entries_by_version_and_device) { migraphx::tmp_dir td{"binary-cache"}; - auto backend = migraphx::gpu::sqlite_binary_cache::open(db_path(td)); + auto backend = migraphx::gpu::sqlite_binary_cache::open(db_path(td), test_stamp); EXPECT(backend.has_value()); const std::vector blob{'p', 'a', 'y'}; @@ -506,7 +512,7 @@ TEST_CASE(sqlite_store_overwrites_in_place) { migraphx::tmp_dir td{"binary-cache"}; auto path = db_path(td); - auto backend = migraphx::gpu::sqlite_binary_cache::open(path); + auto backend = migraphx::gpu::sqlite_binary_cache::open(path, test_stamp); EXPECT(backend.has_value()); const std::vector replacement{'n', 'e', 'w'}; @@ -528,11 +534,11 @@ TEST_CASE(backends_round_trip_through_the_wrapper) const std::vector blob{'\0', 'n', 'o', 't', '\0', 'm', 's', 'g', '\xff'}; auto e = make_entry("opaque"); - auto db = migraphx::gpu::sqlite_binary_cache::open((td.path / "cache.db").string()); + auto db = migraphx::gpu::sqlite_binary_cache::open((td.path / "cache.db").string(), test_stamp); EXPECT(db.has_value()); std::vector backends; - backends.emplace_back(migraphx::gpu::file_binary_cache{td.path / "files"}); + backends.emplace_back(migraphx::gpu::file_binary_cache{td.path / "files", test_stamp}); backends.push_back(*db); for(auto& backend : backends) From 40e7437a4195c351d943e43f8317a5eaf26b7a83 Mon Sep 17 00:00:00 2001 From: pnikolic-amd Date: Wed, 2 Sep 2026 17:49:41 +0200 Subject: [PATCH 04/14] Remove cross test between file and sql --- test/gpu/binary_cache.cpp | 74 ++++++++++----------------------------- 1 file changed, 18 insertions(+), 56 deletions(-) diff --git a/test/gpu/binary_cache.cpp b/test/gpu/binary_cache.cpp index 2ab40ceeeb6..03ea4acf414 100644 --- a/test/gpu/binary_cache.cpp +++ b/test/gpu/binary_cache.cpp @@ -165,6 +165,16 @@ TEST_CASE(memory_lookup_records_reuse) EXPECT(cache.get_stats().misses == 0); } +// The cases below are registered only against a database path, not a directory. Driving the file +// backend through binary_cache puts entries under version_dir()/device_dir(), and write_atomically +// then creates a temp directory inside that, which pushes the file past Windows' MAX_PATH: +// fs::create_directories succeeds because std::filesystem uses the \\?\ prefix, but the +// std::ofstream in write_buffer does not, so every store fails and nothing is persisted. The +// bodies are still parameterized by path, so a directory case is one TEST_CASE to restore once +// write_atomically writes its temporary as a sibling instead of nesting a directory. +// backends_round_trip_through_the_wrapper still covers the file backend, where it is driven +// directly and the version and device strings are short. + // A second cache shares nothing in memory, so anything it finds came out of storage. static void disk_lookup_body(const std::string& path) { @@ -183,12 +193,6 @@ static void disk_lookup_body(const std::string& path) EXPECT(*found->fragment.get_main_module() == *make_code().fragment.get_main_module()); } -TEST_CASE(disk_lookup_records_a_hit) -{ - migraphx::tmp_dir td{"binary-cache"}; - disk_lookup_body(dir_path(td)); -} - TEST_CASE(sqlite_lookup_records_a_hit) { migraphx::tmp_dir td{"binary-cache"}; @@ -214,16 +218,6 @@ static void corrupt_entry_body(const std::string& path, EXPECT(reader.get_stats().misses == 1); } -TEST_CASE(corrupt_entry_is_ignored) -{ - migraphx::tmp_dir td{"binary-cache"}; - corrupt_entry_body(dir_path(td), [](const std::string& dir) { - auto files = entry_files(dir); - EXPECT(files.size() == 1); - migraphx::write_buffer(files.front(), std::vector(8, 0)); - }); -} - TEST_CASE(sqlite_corrupt_entry_is_ignored) { migraphx::tmp_dir td{"binary-cache"}; @@ -328,12 +322,6 @@ static void compiling_twice_body(const std::string& path) gpu_result.to_vector())); } -TEST_CASE(compiling_twice_populates_the_cache_and_matches_reference) -{ - migraphx::tmp_dir td{"binary-cache"}; - compiling_twice_body(dir_path(td)); -} - TEST_CASE(sqlite_compiling_twice_populates_the_cache_and_matches_reference) { migraphx::tmp_dir td{"binary-cache"}; @@ -353,12 +341,6 @@ static void verified_reuse_body(const std::string& path) p.compile(migraphx::make_target("gpu"), options); } -TEST_CASE(verified_reuse_matches_fresh_compiles) -{ - migraphx::tmp_dir td{"binary-cache"}; - verified_reuse_body(dir_path(td)); -} - TEST_CASE(sqlite_verified_reuse_matches_fresh_compiles) { migraphx::tmp_dir td{"binary-cache"}; @@ -366,7 +348,12 @@ TEST_CASE(sqlite_verified_reuse_matches_fresh_compiles) } // The extension of the path picks the backend and nothing else does, so the only way to see the -// choice from outside is the artifact it leaves: a database file, or a tree of entry files. +// choice from outside is the artifact it leaves: a database file, or a directory tree. +// +// The directory half checks that the version tree was laid down rather than that an entry file +// landed in it, because the entry write is what MAX_PATH defeats here. create_directories runs +// before that write and succeeds, and the SQLite backend creates no such tree, so this still +// distinguishes the two backends. TEST_CASE(extension_selects_the_backend) { migraphx::gpu::context ctx; @@ -375,7 +362,7 @@ TEST_CASE(extension_selects_the_backend) migraphx::gpu::binary_cache dir_cache{ migraphx::gpu::binary_cache_settings{dir_path(dir_td), false}}; dir_cache.insert(ctx, make_entry("in-a-directory")); - EXPECT(entry_files(dir_td.path).size() == 1); + EXPECT(migraphx::fs::is_directory(dir_td.path / migraphx::gpu::binary_cache::version_dir())); for(const char* name : {"cache.db", "cache.sqlite"}) { @@ -387,35 +374,10 @@ TEST_CASE(extension_selects_the_backend) EXPECT(migraphx::fs::is_regular_file(path)); EXPECT(row_count(path, "cache_v1") == 1); EXPECT(entry_files(db_td.path).empty()); + EXPECT(not migraphx::fs::exists(db_td.path / migraphx::gpu::binary_cache::version_dir())); } } -// The two backends are interchangeable only because they store the same bytes, so the file the -// one writes and the blob column the other fills have to compare equal. -TEST_CASE(backends_store_identical_bytes) -{ - migraphx::gpu::context ctx; - auto e = make_entry("interchange"); - - migraphx::tmp_dir dir_td{"binary-cache"}; - migraphx::gpu::binary_cache dir_cache{ - migraphx::gpu::binary_cache_settings{dir_path(dir_td), false}}; - dir_cache.insert(ctx, e); - auto files = entry_files(dir_td.path); - EXPECT(files.size() == 1); - auto from_file = migraphx::read_buffer(files.front()); - EXPECT(not from_file.empty()); - - migraphx::tmp_dir db_td{"binary-cache"}; - auto path = db_path(db_td); - migraphx::gpu::binary_cache db_cache{migraphx::gpu::binary_cache_settings{path, false}}; - db_cache.insert(ctx, e); - - auto select = migraphx::sqlite::read(path).prepare("SELECT entry FROM cache_v1;"); - EXPECT(select.step()); - EXPECT((select.column_blob(0) == from_file)); -} - // A database that cannot be opened leaves a memory-only cache rather than an error. The parent // component here is a regular file, so neither creating the directory nor opening the database // can succeed. From cdb51eaaf13c92e103b021e7ef6f423d4fd0127f Mon Sep 17 00:00:00 2001 From: pnikolic-amd Date: Fri, 25 Sep 2026 15:14:01 +0200 Subject: [PATCH 05/14] Address review feedback on SQLite binary cache backend - sqlite_stmt: make bind/step/reset private; calling it binds its args and returns a range of rows as values; drop sqlite_stmt_reset - Reset the statement when its rows go away so reads don't keep a lock - Fall back to read-only use of a database that can't be written - Batch stores in one transaction (optional begin/end_batch on backends) - Open cache storage on first use instead of at construction - Restore directory-backend tests on non-Windows; add SQLite tests - Document .db/.sqlite selection and database maintenance Co-Authored-By: Claude Opus 5.5 (1M context) --- CHANGELOG.md | 2 +- docs/dev/triage-migraphx.rst | 23 +++ src/include/migraphx/sqlite.hpp | 115 +++++++++++--- src/sqlite.cpp | 70 ++++++--- src/targets/gpu/binary_cache.cpp | 46 ++++-- src/targets/gpu/compile_ops.cpp | 26 ++-- .../gpu/include/migraphx/gpu/binary_cache.hpp | 23 +++ .../migraphx/gpu/binary_cache_backend.hpp | 86 ++++++++++- .../migraphx/gpu/sqlite_binary_cache.hpp | 30 +++- src/targets/gpu/sqlite_binary_cache.cpp | 122 +++++++++++---- test/gpu/binary_cache.cpp | 146 ++++++++++++++++-- test/sqlite.cpp | 92 ++++++----- tools/include/gpu/binary_cache_backend.hpp | 30 +++- 13 files changed, 650 insertions(+), 161 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 5972dc67bc2..2da44a782eb 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,7 +7,7 @@ Full documentation for MIGraphX is available at ### Added -* Added a binary cache for compiled GPU kernels: identical kernels within a model compile once, and setting the `MIGRAPHX_BINARY_CACHE` environment variable (or the `binary_cache` backend option) also persists them on disk so later compiles of the same kernels skip compilation entirely; a `binary_cache_verify` backend option recompiles reused kernels and fails if they differ. +* Added a binary cache for compiled GPU kernels: identical kernels within a model compile once, and setting the `MIGRAPHX_BINARY_CACHE` environment variable (or the `binary_cache` backend option) also persists them on disk so later compiles of the same kernels skip compilation entirely, in a directory or, for a path ending in `.db` or `.sqlite`, in a single SQLite database; a `binary_cache_verify` backend option recompiles reused kernels and fails if they differ. * Added a layered problem-cache priority list (searched in order, first hit wins), delivered to the GPU target through the `problem_cache_files` backend option. * Added `ArrayFeatureExtractor` ONNX operator support (#4742). * Added support for building against ROCm 7.13 and newer using TheRock (#4952) diff --git a/docs/dev/triage-migraphx.rst b/docs/dev/triage-migraphx.rst index d85d63d0dba..efdd3019081 100644 --- a/docs/dev/triage-migraphx.rst +++ b/docs/dev/triage-migraphx.rst @@ -175,6 +175,29 @@ directories for builds you no longer use: ls $HOME/.cache/migraphx # directories are named after the build that wrote them rm -r $HOME/.cache/migraphx/v1-hip22.0.* +A path ending in ``.db`` or ``.sqlite`` keeps the cache in a single SQLite database instead of a +directory, which is easier to copy between machines and can be shared by processes compiling at +the same time. Missing parent directories are created, as for a directory cache: + +.. code-block:: bash + + export MIGRAPHX_BINARY_CACHE=$HOME/.cache/migraphx/kernels.db + +Every row records the full version id of the build that wrote it in the ``version`` column, +alongside the operator name, problem and solution, so the database can be inspected and pruned +with the ``sqlite3`` shell. Deleting rows does not shrink the file until it is vacuumed: + +.. code-block:: bash + + sqlite3 $HOME/.cache/migraphx/kernels.db \ + "SELECT version, op_name, count(*) FROM cache_v1 GROUP BY version, op_name;" + sqlite3 $HOME/.cache/migraphx/kernels.db \ + "DELETE FROM cache_v1 WHERE version LIKE 'v1-hip22.0.%'; VACUUM;" + +A database that cannot be written to, such as a shared cache installed read-only, is still used +for lookups, and newly compiled kernels are kept in memory only. A database that cannot be opened +at all is skipped with a warning and the compile proceeds without a disk cache. + The same settings are available as backend options, which take precedence over the environment and are how tests configure the cache: diff --git a/src/include/migraphx/sqlite.hpp b/src/include/migraphx/sqlite.hpp index b11bb6e000b..f65217ae2e0 100644 --- a/src/include/migraphx/sqlite.hpp +++ b/src/include/migraphx/sqlite.hpp @@ -25,12 +25,18 @@ #define MIGRAPHX_GUARD_MIGRAPHX_SQLITE_HPP #include +#include #include +#include +#include #include +#include #include +#include #include #include #include +#include #include #include @@ -41,49 +47,108 @@ struct sqlite_impl; struct sqlite_stmt_impl; /// A prepared statement, holding a reference to the connection it was prepared on so it can -/// never outlive it. +/// never outlive it. Copies share the same statement. +/// +/// Calling it with arguments runs it: the arguments are bound to the parameters in order and the +/// result comes back as a range of rows. Preparing once and calling many times is the point. /// /// Not thread safe: one statement may be used by one thread at a time, even though the -/// connection itself is serialized. Reuse is the point of preparing -- prepare once, then -/// reset/bind/step per operation. +/// connection itself is serialized. struct MIGRAPHX_EXPORT sqlite_stmt { + struct rows; + sqlite_stmt() = default; - /// Bind a parameter. Indices are 1-based, matching sqlite's own convention. - sqlite_stmt& bind(int i, std::string_view s); - sqlite_stmt& bind(int i, std::int64_t x); - sqlite_stmt& bind(int i, const std::vector& blob); + /// Run the statement with xs bound to its parameters in order, and return its rows. + template + rows operator()(const Ts&... xs) const + { + if(not valid()) + MIGRAPHX_THROW("sqlite: calling a statement that was never prepared"); + // Anything left from the previous call, bindings or an unfinished result, goes first. + reset(); + sequence_c([&](auto... is) { swallow{(bind(int{is + 1}, xs), 0)...}; }); + return rows{*this}; + } + + bool valid() const { return impl != nullptr; } + + private: + // Parameter indices are 1-based, matching sqlite's own convention. + void bind(int i, std::string_view s) const; + void bind(int i, std::int64_t x) const; + void bind(int i, const std::vector& blob) const; /// Step once. True when a row is available, false when the statement is done. - bool step(); + bool step() const; /// Clear bindings and rewind, so the statement can be used again. Safe at any point, /// including after step() has thrown. - void reset() noexcept; + void reset() const noexcept; - /// Read a column of the current row. Indices here are 0-based, again matching sqlite. - std::string column_text(int i) const; - std::vector column_blob(int i) const; + /// The current row as an object keyed by column name. Blobs become value::binary and SQL + /// NULL becomes a null value. + value to_value() const; - bool valid() const { return impl != nullptr; } - - private: friend struct sqlite; std::shared_ptr impl; }; -/// Resets a statement on scope exit, so an early return or a thrown exception cannot leave -/// bindings or a half-consumed result set behind for whoever uses the statement next. -struct sqlite_stmt_reset +/// The rows produced by one call of a statement, as an input range of values. +/// +/// The first row is fetched when the call is made, so a statement that returns nothing, such as +/// an insert, has already run by the time the call returns, whether or not the range is +/// iterated. The statement is reset when the range is destroyed: an unfinished select holds a +/// read lock on the database until then, which would stall writers in other processes. +struct sqlite_stmt::rows { - explicit sqlite_stmt_reset(sqlite_stmt& s) : stmt(&s) {} - sqlite_stmt_reset(const sqlite_stmt_reset&) = delete; - sqlite_stmt_reset& operator=(const sqlite_stmt_reset&) = delete; - ~sqlite_stmt_reset() { stmt->reset(); } + explicit rows(sqlite_stmt s) : stmt(std::move(s)), first(stmt.step()) {} + // Only ever a prvalue returned from a call, so it never needs copying or moving, and a copy + // would reset the statement out from under the original. + rows(const rows&) = delete; + rows(rows&&) = delete; + rows& operator=(const rows&) = delete; + rows& operator=(rows&&) = delete; + ~rows() { stmt.reset(); } + + struct iterator : iterator_operators + { + using value_type = value; + using reference = value_type; + using difference_type = std::ptrdiff_t; + using iterator_category = std::input_iterator_tag; + using pointer = std::add_pointer_t>; + + iterator() = default; + + iterator(const rows* pparent, bool pavailable) : parent(pparent), available(pavailable) {} + + reference operator*() const { return parent->stmt.to_value(); } + + template + static void increment(U& x) + { + x.available = x.parent->stmt.step(); + } + + template + static auto equal(const U& x, const V& y) + { + return x.parent == y.parent and x.available == y.available; + } + + private: + const rows* parent = nullptr; + bool available = false; + }; + + iterator begin() const { return {this, first}; } + iterator end() const { return {this, false}; } private: - sqlite_stmt* stmt; + sqlite_stmt stmt; + bool first = false; }; struct MIGRAPHX_EXPORT sqlite @@ -103,6 +168,10 @@ struct MIGRAPHX_EXPORT sqlite /// How long to wait for a lock held by another connection before failing. void set_busy_timeout(int ms); + /// True when writes will be refused. Opening for writing still succeeds on a file the OS + /// has write-protected, in which case sqlite quietly opens it read-only; this is how to tell. + bool read_only() const; + bool valid() const { return impl != nullptr; } private: diff --git a/src/sqlite.cpp b/src/sqlite.cpp index 61bda7e5652..619089aced5 100644 --- a/src/sqlite.cpp +++ b/src/sqlite.cpp @@ -25,8 +25,10 @@ #include #include #include +#include #include #include +#include namespace migraphx { inline namespace MIGRAPHX_INLINE_NS { @@ -157,7 +159,9 @@ sqlite_stmt sqlite::prepare(const std::string& sql) void sqlite::set_busy_timeout(int ms) { sqlite3_busy_timeout(impl->get(), ms); } -sqlite_stmt& sqlite_stmt::bind(int i, std::string_view s) +bool sqlite::read_only() const { return sqlite3_db_readonly(impl->get(), "main") == 1; } + +void sqlite_stmt::bind(int i, std::string_view s) const { // A default-constructed string_view has null data(), and a null pointer binds SQL NULL // rather than an empty string, so empty input substitutes a valid pointer. SQLITE_TRANSIENT @@ -167,18 +171,16 @@ sqlite_stmt& sqlite_stmt::bind(int i, std::string_view s) int rc = sqlite3_bind_text64(impl->get(), i, text, s.size(), SQLITE_TRANSIENT, SQLITE_UTF8); if(rc != SQLITE_OK) MIGRAPHX_THROW(impl->error_message()); - return *this; } -sqlite_stmt& sqlite_stmt::bind(int i, std::int64_t x) +void sqlite_stmt::bind(int i, std::int64_t x) const { int rc = sqlite3_bind_int64(impl->get(), i, x); if(rc != SQLITE_OK) MIGRAPHX_THROW(impl->error_message()); - return *this; } -sqlite_stmt& sqlite_stmt::bind(int i, const std::vector& blob) +void sqlite_stmt::bind(int i, const std::vector& blob) const { // As with text, an empty vector's data() may be null, which would bind SQL NULL; a // zero-length zeroblob is an empty BLOB instead. The 64-bit form is used because the @@ -189,10 +191,9 @@ sqlite_stmt& sqlite_stmt::bind(int i, const std::vector& blob) : sqlite3_bind_blob64(impl->get(), i, blob.data(), blob.size(), SQLITE_TRANSIENT); if(rc != SQLITE_OK) MIGRAPHX_THROW(impl->error_message()); - return *this; } -bool sqlite_stmt::step() +bool sqlite_stmt::step() const { int rc = sqlite3_step(impl->get()); if(rc == SQLITE_ROW) @@ -202,7 +203,7 @@ bool sqlite_stmt::step() MIGRAPHX_THROW(impl->error_message()); } -void sqlite_stmt::reset() noexcept +void sqlite_stmt::reset() const noexcept { if(impl == nullptr) return; @@ -212,25 +213,48 @@ void sqlite_stmt::reset() noexcept (void)sqlite3_clear_bindings(impl->get()); } -std::string sqlite_stmt::column_text(int i) const +static value column_value(sqlite3_stmt* stmt, int i) { - const auto* text = sqlite3_column_text(impl->get(), i); - int n = sqlite3_column_bytes(impl->get(), i); - if(text == nullptr or n <= 0) - return {}; - // NOLINTNEXTLINE(cppcoreguidelines-pro-type-reinterpret-cast) - return {reinterpret_cast(text), static_cast(n)}; + // For text and blobs the data must be fetched before sqlite3_column_bytes: the other order + // can force a type conversion that invalidates the pointer. A zero-length value comes back + // as a null pointer, which still means empty rather than NULL. + switch(sqlite3_column_type(stmt, i)) + { + case SQLITE_INTEGER: return std::int64_t{sqlite3_column_int64(stmt, i)}; + case SQLITE_FLOAT: return sqlite3_column_double(stmt, i); + case SQLITE_TEXT: { + const auto* text = sqlite3_column_text(stmt, i); + auto n = sqlite3_column_bytes(stmt, i); + if(text == nullptr) + return std::string{}; + // NOLINTNEXTLINE(cppcoreguidelines-pro-type-reinterpret-cast) + return std::string(reinterpret_cast(text), n); + } + case SQLITE_BLOB: { + const auto* data = static_cast(sqlite3_column_blob(stmt, i)); + auto n = sqlite3_column_bytes(stmt, i); + if(data == nullptr) + return value::binary{}; + return value::binary{data, static_cast(n)}; + } + default: return {}; + } } -std::vector sqlite_stmt::column_blob(int i) const +value sqlite_stmt::to_value() const { - // sqlite3_column_blob must be called before sqlite3_column_bytes: the other order can - // force a type conversion that invalidates the pointer. - const auto* data = static_cast(sqlite3_column_blob(impl->get(), i)); - int n = sqlite3_column_bytes(impl->get(), i); - if(data == nullptr or n <= 0) - return {}; - return {data, data + n}; + auto* stmt = impl->get(); + value::object row; + auto columns = range(sqlite3_column_count(stmt)); + std::transform(columns.begin(), + columns.end(), + std::inserter(row, row.end()), + [&](std::ptrdiff_t i) { + auto col = static_cast(i); + return std::make_pair(std::string{sqlite3_column_name(stmt, col)}, + column_value(stmt, col)); + }); + return row; } } // namespace MIGRAPHX_INLINE_NS diff --git a/src/targets/gpu/binary_cache.cpp b/src/targets/gpu/binary_cache.cpp index f56c0a86304..55fa75fb83d 100644 --- a/src/targets/gpu/binary_cache.cpp +++ b/src/targets/gpu/binary_cache.cpp @@ -146,19 +146,39 @@ static optional make_binary_cache_backend(const std::strin return binary_cache_backend{file_binary_cache{path}}; } +binary_cache::binary_cache(binary_cache_settings s) : settings(std::move(s)) {} + // Nothing can be persisted safely when the compiler cannot be identified, since entries from // different toolchains would be indistinguishable. That is a property of the cache rather than // of the storage medium, so it is checked here instead of in each backend. -binary_cache::binary_cache(binary_cache_settings s) : settings(std::move(s)) +binary_cache_backend* binary_cache::get_backend() { - // Checked first so that a memory-only cache never compiles the version probe. - if(settings.path.empty()) - return; - // The version names a directory for the file backend, so it is kept short there. A database - // has no such limit and records the full id, which is self-describing. - version = version_id(not is_database_path(settings.path)); - if(not version.empty()) - backend = make_binary_cache_backend(settings.path); + if(not backend_opened) + { + backend_opened = true; + // Checked first so that a memory-only cache never compiles the version probe. + if(not settings.path.empty()) + { + // The version names a directory for the file backend, so it is kept short there. A + // database has no such limit and records the full id, which is self-describing. + version = version_id(not is_database_path(settings.path)); + if(not version.empty()) + backend = make_binary_cache_backend(settings.path); + } + } + return backend.has_value() ? &*backend : nullptr; +} + +binary_cache::store_batch::store_batch(binary_cache& c) : cache(&c) +{ + if(auto* b = cache->get_backend()) + b->begin_batch(); +} + +binary_cache::store_batch::~store_batch() +{ + if(auto* b = cache->get_backend()) + b->end_batch(); } optional binary_cache::get(const context& ctx, const std::string& key) @@ -171,12 +191,12 @@ optional binary_cache::get(const context& ctx, const std::string& counters.reused++; return it->second; } - if(backend.has_value()) + if(auto* b = get_backend()) { // Hashing the key is not free -- it is the whole compile source, which runs to // kilobytes -- so it is done once and reused for the lookup and any diagnostics. auto key_hash = md5(key); - auto blob = backend->load(version, device_dir(ctx), key_hash); + auto blob = b->load(version, device_dir(ctx), key_hash); if(blob.has_value()) { auto e = decode_entry(*blob, key, key_hash); @@ -196,7 +216,7 @@ void binary_cache::insert(const context& ctx, entry e) if(e.key.empty()) return; counters.compiled++; - if(backend.has_value()) + if(auto* b = get_backend()) { auto key_hash = md5(e.key); try @@ -205,7 +225,7 @@ void binary_cache::insert(const context& ctx, entry e) // failure here a warning like any other storage failure instead of escaping // insert() and failing the compile. auto blob = to_msgpack(migraphx::to_value(e)); - backend->store(version, device_dir(ctx), key_hash, e, blob); + b->store(version, device_dir(ctx), key_hash, e, blob); } catch(const std::exception& ex) { diff --git a/src/targets/gpu/compile_ops.cpp b/src/targets/gpu/compile_ops.cpp index ad692b3a3c3..8ea8289523e 100644 --- a/src/targets/gpu/compile_ops.cpp +++ b/src/targets/gpu/compile_ops.cpp @@ -804,17 +804,23 @@ struct compile_manager cell->result = cp->run_compile(cell->solution); }); - for(const auto& [cp, cell] : tasks) + if(not tasks.empty()) { - if(not cell->result.has_value()) - continue; - // When verifying, reused results are stored again, rewriting the same bytes - // harmlessly. - cp->store(cell->solution, cell->key, cell->result->code); - assert(not cell->result->code.empty()); - // Only the serializable code is used from here on; dropping the replace function - // releases what its closure holds and keeps it off other instructions. - cell->result->replace_fn = nullptr; + // Every plan compiles with the same context, so the stores all go to one cache, and + // batching them lets its storage commit them together rather than one at a time. + binary_cache::store_batch batch{tasks.front().first->ctx->get_binary_cache()}; + for(const auto& [cp, cell] : tasks) + { + if(not cell->result.has_value()) + continue; + // When verifying, reused results are stored again, rewriting the same bytes + // harmlessly. + cp->store(cell->solution, cell->key, cell->result->code); + assert(not cell->result->code.empty()); + // Only the serializable code is used from here on; dropping the replace function + // releases what its closure holds and keeps it off other instructions. + cell->result->replace_fn = nullptr; + } } static const auto mxr_path = string_value_of(MIGRAPHX_GPU_DUMP_BENCHMARK_MXR{}); diff --git a/src/targets/gpu/include/migraphx/gpu/binary_cache.hpp b/src/targets/gpu/include/migraphx/gpu/binary_cache.hpp index 770f7efee0a..887915135e2 100644 --- a/src/targets/gpu/include/migraphx/gpu/binary_cache.hpp +++ b/src/targets/gpu/include/migraphx/gpu/binary_cache.hpp @@ -85,6 +85,24 @@ struct MIGRAPHX_GPU_EXPORT binary_cache std::size_t compiled = 0; }; + /// Groups the inserts made while it lives, so the storage backend can commit them together + /// rather than one at a time. Scope it tightly around a run of inserts: a database holds a + /// write lock against other processes until it ends. + struct MIGRAPHX_GPU_EXPORT store_batch + { + explicit store_batch(binary_cache& c); + store_batch(const store_batch&) = delete; + store_batch(store_batch&&) = delete; + store_batch& operator=(const store_batch&) = delete; + store_batch& operator=(store_batch&&) = delete; + ~store_batch(); + + private: + binary_cache* cache; + }; + + /// Nothing is opened here; storage is set up by the first lookup or insert, so a context + /// that never compiles never touches the disk or probes the compiler. explicit binary_cache(binary_cache_settings s = {}); /// Look up a key, consulting memory first and then the cache directory. @@ -106,12 +124,17 @@ struct MIGRAPHX_GPU_EXPORT binary_cache static const std::string& version_id(bool use_short_digest); private: + /// The storage backend, opened on first use, or null for a memory-only cache. + binary_cache_backend* get_backend(); + std::unordered_map memo; binary_cache_settings settings; /// The version_id entries are stored under, in the form the backend uses. std::string version; /// Where entries are persisted, or empty for a memory-only cache. optional backend; + /// Whether get_backend has already tried to open the backend, successfully or not. + bool backend_opened = false; stats counters; }; diff --git a/src/targets/gpu/include/migraphx/gpu/binary_cache_backend.hpp b/src/targets/gpu/include/migraphx/gpu/binary_cache_backend.hpp index e57c5021c5c..106cf627d91 100644 --- a/src/targets/gpu/include/migraphx/gpu/binary_cache_backend.hpp +++ b/src/targets/gpu/include/migraphx/gpu/binary_cache_backend.hpp @@ -30,7 +30,8 @@ // into the gpu target tree). Do not edit the generated header by hand. // // Any type T satisfies the binary_cache_backend concept if it provides the -// member functions listed below. The wrapper holds T by shared_ptr and forwards +// member functions listed below; begin_batch and end_batch are optional and +// default to doing nothing. The wrapper holds T by shared_ptr and forwards // each call through a virtual dispatch, matching problem_cache_backend. // // Notes: @@ -38,9 +39,10 @@ // the include below pulls in its full definition. // * Backends typically own non-trivial resources (a cache directory, a SQLite // connection) and are not meaningfully copyable beyond shared ownership. -// * Both members are non-const: binary_cache::get and insert are themselves -// non-const, so nothing forces a const qualifier here, and a backend holding -// prepared statements needs the mutability. +// * The members are non-const: binary_cache::get and insert are themselves +// non-const, so nothing forces a const qualifier here, and a backend that +// tracks an open batch needs the mutability. A backend may still declare +// them const. // #ifndef MIGRAPHX_GUARD_GPU_BINARY_CACHE_BACKEND_HPP #define MIGRAPHX_GUARD_GPU_BINARY_CACHE_BACKEND_HPP @@ -54,6 +56,7 @@ #include #include +#include #include #include #include @@ -112,6 +115,21 @@ struct binary_cache_backend const std::string& key_hash, const binary_cache_entry& e, const std::vector& blob); + + /// Mark the start of a run of stores that may be committed together, such as + /// a database transaction, rather than one at a time. Every begin_batch is + /// followed by an end_batch, and the two are never nested. Optional: a + /// backend without them stores each entry as it comes. + /// + /// Must not throw. A backend that cannot start a batch stores entries one + /// at a time instead. + void begin_batch(); + + /// Commit the stores made since begin_batch. + /// + /// Must not throw. A failed commit costs those entries a recompile next + /// run, nothing more, and must not leave anything locked. + void end_batch(); }; #else @@ -130,6 +148,10 @@ struct MIGRAPHX_EXPORT binary_cache_backend const std::string& key_hash, const binary_cache_entry& e, const std::vector& blob); + // (optional) + void begin_batch(); + // (optional) + void end_batch(); }; #else @@ -137,6 +159,32 @@ struct MIGRAPHX_EXPORT binary_cache_backend struct binary_cache_backend { private: + template + static auto private_detail_te_default_begin_batch(char, T&& private_detail_te_self) + -> decltype(private_detail_te_self.begin_batch()) + { + private_detail_te_self.begin_batch(); + } + + template + static void private_detail_te_default_begin_batch(float, T&& private_detail_te_self) + { + migraphx::nop(private_detail_te_self); + } + + template + static auto private_detail_te_default_end_batch(char, T&& private_detail_te_self) + -> decltype(private_detail_te_self.end_batch()) + { + private_detail_te_self.end_batch(); + } + + template + static void private_detail_te_default_end_batch(float, T&& private_detail_te_self) + { + migraphx::nop(private_detail_te_self); + } + template struct private_te_unwrap_reference { @@ -162,6 +210,10 @@ struct binary_cache_backend std::declval(), std::declval(), std::declval&>()), + private_detail_te_default_begin_batch(char(0), + std::declval()), + private_detail_te_default_end_batch(char(0), + std::declval()), void()); template @@ -255,6 +307,18 @@ struct binary_cache_backend (*this).private_detail_te_get_handle().store(version, device, key_hash, e, blob); } + void begin_batch() + { + assert((*this).private_detail_te_handle_mem_var); + (*this).private_detail_te_get_handle().begin_batch(); + } + + void end_batch() + { + assert((*this).private_detail_te_handle_mem_var); + (*this).private_detail_te_get_handle().end_batch(); + } + friend bool is_shared(const binary_cache_backend& private_detail_x, const binary_cache_backend& private_detail_y) { @@ -277,6 +341,8 @@ struct binary_cache_backend const std::string& key_hash, const binary_cache_entry& e, const std::vector& blob) = 0; + virtual void begin_batch() = 0; + virtual void end_batch() = 0; }; template @@ -324,6 +390,18 @@ struct binary_cache_backend private_detail_te_value.store(version, device, key_hash, e, blob); } + void begin_batch() override + { + + private_detail_te_default_begin_batch(char(0), private_detail_te_value); + } + + void end_batch() override + { + + private_detail_te_default_end_batch(char(0), private_detail_te_value); + } + PrivateDetailTypeErasedT private_detail_te_value; }; diff --git a/src/targets/gpu/include/migraphx/gpu/sqlite_binary_cache.hpp b/src/targets/gpu/include/migraphx/gpu/sqlite_binary_cache.hpp index 1ab8411edd3..0efd4e289e5 100644 --- a/src/targets/gpu/include/migraphx/gpu/sqlite_binary_cache.hpp +++ b/src/targets/gpu/include/migraphx/gpu/sqlite_binary_cache.hpp @@ -46,23 +46,37 @@ namespace gpu { // target: migraphx/sqlite.hpp forward-declares both impl types and never includes sqlite3.h. struct MIGRAPHX_GPU_EXPORT sqlite_binary_cache { - /// Open the database, create the schema and prepare the statements. Returns nullopt when any - /// of that fails, so an unusable database leaves the cache memory-only rather than raising an - /// error. Returns the wrapper so the caller can hand the result straight back. + /// Open the database, create the schema and prepare the statements, and return the wrapper + /// so the caller can hand the result straight back. + /// + /// A database that can only be read gives a backend that serves lookups and ignores stores; + /// it is used as it stands, without creating the schema. Returns nullopt when the database + /// cannot be opened at all or entries cannot be looked up in it, so an unusable database + /// leaves the cache memory-only rather than raising an error. static optional open(const std::string& path); optional> - load(const std::string& version, const std::string& device, const std::string& key_hash); + load(const std::string& version, const std::string& device, const std::string& key_hash) const; void store(const std::string& version, const std::string& device, const std::string& key_hash, const binary_cache_entry& e, - const std::vector& blob); + const std::vector& blob) const; + + /// Open a transaction, so the stores that follow cost one commit rather than one each. + void begin_batch(); + /// Commit the transaction begin_batch opened, or roll it back if the commit fails. + void end_batch(); private: - sqlite db = {}; - sqlite_stmt get_stmt = {}; - sqlite_stmt store_stmt = {}; + sqlite db = {}; + sqlite_stmt get_stmt = {}; + sqlite_stmt store_stmt = {}; + sqlite_stmt begin_stmt = {}; + sqlite_stmt commit_stmt = {}; + sqlite_stmt rollback_stmt = {}; + /// Whether begin_batch opened a transaction that end_batch still has to close. + bool in_batch = false; }; } // namespace gpu diff --git a/src/targets/gpu/sqlite_binary_cache.cpp b/src/targets/gpu/sqlite_binary_cache.cpp index cd7504048f1..de0f1ae229b 100644 --- a/src/targets/gpu/sqlite_binary_cache.cpp +++ b/src/targets/gpu/sqlite_binary_cache.cpp @@ -26,6 +26,7 @@ #include #include #include +#include #include #include @@ -83,6 +84,14 @@ constexpr const char* store_sql = " (version, device, key_hash, op_name, problem, solution, entry, timestamp)" " VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7, CAST(STRFTIME('%s','now') AS INTEGER));"; +// Stores come in a burst after each round of compiles, and outside a transaction every one of +// them is its own commit, each waiting for the disk. IMMEDIATE takes the write lock up front, so +// a busy database is found out here, once, rather than partway through the stores; readers are +// not blocked until the commit itself. +constexpr const char* begin_sql = "BEGIN IMMEDIATE;"; +constexpr const char* commit_sql = "COMMIT;"; +constexpr const char* rollback_sql = "ROLLBACK;"; + } // namespace optional sqlite_binary_cache::open(const std::string& path) @@ -91,11 +100,20 @@ optional sqlite_binary_cache::open(const std::string& path try { // sqlite will not create a missing parent directory, but the file backend does, so - // this keeps the two backends behaving the same on a fresh machine. + // this keeps the two backends behaving the same on a fresh machine. A failure here is + // left to the open below, since an existing database may still be readable. auto parent = fs::path{path}.parent_path(); + std::error_code ec; if(not parent.empty()) - fs::create_directories(parent); + fs::create_directories(parent, ec); + + // A database that can be read but not written to is still worth having: reads serve + // hits and nothing is stored. Opening for writing already falls back to reading when the + // file itself is write-protected; reading is tried here for anything else that refuses + // a writer, such as a read-only mount, as long as there is a database to read. auto db = sqlite::try_write(path); + if(not db.has_value() and fs::exists(path)) + db = sqlite::read(path); if(not db.has_value()) { log::warn() << "Disabling the binary cache: cannot open " << path; @@ -103,8 +121,20 @@ optional sqlite_binary_cache::open(const std::string& path } r.db = std::move(*db); r.db.set_busy_timeout(busy_timeout_ms); - (void)r.db.execute(schema_sql); - // Without a working lookup there is no cache, so this failure disables the backend. + if(r.db.read_only()) + { + log::warn() << "Binary cache at " << path << " is read-only"; + } + else + { + (void)r.db.execute(schema_sql); + r.store_stmt = r.db.prepare(store_sql); + r.begin_stmt = r.db.prepare(begin_sql); + r.commit_stmt = r.db.prepare(commit_sql); + r.rollback_stmt = r.db.prepare(rollback_sql); + } + // Without a working lookup there is no cache, so this failure disables the backend. That + // includes a read-only database that was never given the schema. r.get_stmt = r.db.prepare(get_sql); } catch(const std::exception& ex) @@ -112,32 +142,25 @@ optional sqlite_binary_cache::open(const std::string& path log::warn() << "Disabling the binary cache at " << path << ": " << ex.what(); return nullopt; } - try - { - r.store_stmt = r.db.prepare(store_sql); - } - catch(const std::exception& ex) - { - // A database that can be read but not written to is still worth having: reads serve - // hits and stores quietly do nothing. - log::warn() << "Binary cache at " << path << " is read-only: " << ex.what(); - } return binary_cache_backend{std::move(r)}; } optional> sqlite_binary_cache::load(const std::string& version, const std::string& device, - const std::string& key_hash) + const std::string& key_hash) const { if(not get_stmt.valid()) return nullopt; try { - sqlite_stmt_reset guard{get_stmt}; - get_stmt.bind(1, version).bind(2, device).bind(3, key_hash); - if(not get_stmt.step()) + // The primary key makes this at most one row. + auto rows = get_stmt(version, device, key_hash); + auto it = rows.begin(); + if(it == rows.end()) return nullopt; - return get_stmt.column_blob(0); + auto row = *it; + const auto& entry = row.at("entry").get_binary(); + return std::vector(entry.begin(), entry.end()); } catch(const std::exception& ex) { @@ -151,22 +174,21 @@ void sqlite_binary_cache::store(const std::string& version, const std::string& device, const std::string& key_hash, const binary_cache_entry& e, - const std::vector& blob) + const std::vector& blob) const { if(not store_stmt.valid()) return; try { // The json strings are temporaries, which is safe because binding copies immediately. - sqlite_stmt_reset guard{store_stmt}; - store_stmt.bind(1, version) - .bind(2, device) - .bind(3, key_hash) - .bind(4, e.op_name) - .bind(5, to_json_string(e.problem)) - .bind(6, to_json_string(e.solution)) - .bind(7, blob); - store_stmt.step(); + // An insert returns no rows, and it has run by the time the call returns. + store_stmt(version, + device, + key_hash, + e.op_name, + to_json_string(e.problem), + to_json_string(e.solution), + blob); } catch(const std::exception& ex) { @@ -174,6 +196,48 @@ void sqlite_binary_cache::store(const std::string& version, } } +void sqlite_binary_cache::begin_batch() +{ + assert(not in_batch); + if(not begin_stmt.valid()) + return; + try + { + begin_stmt(); + in_batch = true; + } + catch(const std::exception& ex) + { + // Without a transaction each store commits on its own, which is slower but still works. + log::warn() << "Binary cache stores will be committed one at a time: " << ex.what(); + } +} + +void sqlite_binary_cache::end_batch() +{ + if(not in_batch) + return; + in_batch = false; + try + { + commit_stmt(); + } + catch(const std::exception& ex) + { + // A commit that fails leaves the transaction open, holding the write lock against every + // other process, so it is rolled back and the batch's entries are lost instead. + log::warn() << "Failed to commit binary cache entries: " << ex.what(); + try + { + rollback_stmt(); + } + catch(const std::exception& rex) + { + log::warn() << "Failed to roll back binary cache entries: " << rex.what(); + } + } +} + } // namespace gpu } // namespace MIGRAPHX_INLINE_NS } // namespace migraphx diff --git a/test/gpu/binary_cache.cpp b/test/gpu/binary_cache.cpp index 16a93c08a51..4dea3f94034 100644 --- a/test/gpu/binary_cache.cpp +++ b/test/gpu/binary_cache.cpp @@ -31,6 +31,8 @@ #include #include #include +#include +#include #include #include #include @@ -161,15 +163,13 @@ TEST_CASE(memory_lookup_records_reuse) EXPECT(cache.get_stats().misses == 0); } -// The cases below are registered only against a database path, not a directory. Driving the file -// backend through binary_cache puts entries under version_id()/device_dir(), and write_atomically -// then creates a temp directory inside that, which pushes the file past Windows' MAX_PATH: -// fs::create_directories succeeds because std::filesystem uses the \\?\ prefix, but the -// std::ofstream in write_buffer does not, so every store fails and nothing is persisted. The -// bodies are still parameterized by path, so a directory case is one TEST_CASE to restore once -// write_atomically writes its temporary as a sibling instead of nesting a directory. -// backends_round_trip_through_the_wrapper still covers the file backend, where it is driven -// directly and the version and device strings are short. +// The cases below are written once against a path and registered for both backends. The +// directory registrations are skipped on Windows for now: driving the file backend through +// binary_cache puts entries under version_id()/device_dir(), and write_atomically then creates a +// temp directory inside that, which pushes the file past Windows' MAX_PATH. fs::create_directories +// succeeds because std::filesystem uses the \\?\ prefix, but the std::ofstream in write_buffer +// does not, so every store fails and nothing is persisted. backends_round_trip_through_the_wrapper +// still covers the file backend there, where it is driven directly and the paths are short. // A second cache shares nothing in memory, so anything it finds came out of storage. static void disk_lookup_body(const std::string& path) @@ -189,6 +189,14 @@ static void disk_lookup_body(const std::string& path) EXPECT(*found->fragment.get_main_module() == *make_code().fragment.get_main_module()); } +#ifndef _WIN32 +TEST_CASE(disk_lookup_records_a_hit) +{ + migraphx::tmp_dir td{"binary-cache"}; + disk_lookup_body(dir_path(td)); +} +#endif + TEST_CASE(sqlite_lookup_records_a_hit) { migraphx::tmp_dir td{"binary-cache"}; @@ -214,6 +222,19 @@ static void corrupt_entry_body(const std::string& path, EXPECT(reader.get_stats().misses == 1); } +#ifndef _WIN32 +TEST_CASE(corrupt_entry_is_ignored) +{ + migraphx::tmp_dir td{"binary-cache"}; + corrupt_entry_body(dir_path(td), [](const std::string& dir) { + auto files = entry_files(dir); + EXPECT(files.size() == 1); + for(const auto& file : files) + migraphx::write_buffer(file, std::vector(8, 0)); + }); +} +#endif + TEST_CASE(sqlite_corrupt_entry_is_ignored) { migraphx::tmp_dir td{"binary-cache"}; @@ -318,6 +339,14 @@ static void compiling_twice_body(const std::string& path) gpu_result.to_vector())); } +#ifndef _WIN32 +TEST_CASE(compiling_twice_populates_the_cache_and_matches_reference) +{ + migraphx::tmp_dir td{"binary-cache"}; + compiling_twice_body(dir_path(td)); +} +#endif + TEST_CASE(sqlite_compiling_twice_populates_the_cache_and_matches_reference) { migraphx::tmp_dir td{"binary-cache"}; @@ -337,6 +366,14 @@ static void verified_reuse_body(const std::string& path) p.compile(migraphx::make_target("gpu"), options); } +#ifndef _WIN32 +TEST_CASE(verified_reuse_matches_fresh_compiles) +{ + migraphx::tmp_dir td{"binary-cache"}; + verified_reuse_body(dir_path(td)); +} +#endif + TEST_CASE(sqlite_verified_reuse_matches_fresh_compiles) { migraphx::tmp_dir td{"binary-cache"}; @@ -445,6 +482,63 @@ TEST_CASE(sqlite_records_the_full_version_id) EXPECT(rows.front().at("version") == migraphx::gpu::binary_cache::version_id(false)); } +// The op name, problem and solution are copied into columns only so a cache can be inspected +// with SQL; loads never read them, so nothing else would notice them being wrong. +TEST_CASE(sqlite_records_what_each_entry_was_compiled_for) +{ + migraphx::tmp_dir td{"binary-cache"}; + migraphx::gpu::context ctx; + auto path = db_path(td); + auto e = make_entry("described"); + migraphx::gpu::binary_cache cache{migraphx::gpu::binary_cache_settings{path, false}}; + cache.insert(ctx, e); + + auto rows = migraphx::sqlite::read(path).execute( + "SELECT key_hash, op_name, problem, solution FROM cache_v1;"); + EXPECT(rows.size() == 1); + const auto& row = rows.front(); + EXPECT(row.at("key_hash") == migraphx::md5(e.key)); + EXPECT(row.at("op_name") == e.op_name); + EXPECT(migraphx::from_json_string(row.at("problem")) == e.problem); + EXPECT(migraphx::from_json_string(row.at("solution")) == e.solution); +} + +// A database that cannot be written to, such as a shared cache installed read-only, still +// serves the entries already in it, and storing into it is quietly skipped. +TEST_CASE(sqlite_read_only_database_still_serves_hits) +{ + migraphx::tmp_dir td{"binary-cache"}; + migraphx::gpu::context ctx; + auto path = db_path(td); + migraphx::gpu::binary_cache_settings settings{path, false}; + { + migraphx::gpu::binary_cache writer{settings}; + writer.insert(ctx, make_entry("existing")); + } + + const auto writable = migraphx::fs::perms::owner_write | migraphx::fs::perms::group_write | + migraphx::fs::perms::others_write; + migraphx::fs::permissions(path, writable, migraphx::fs::perm_options::remove); + // Permissions do not stop root, so what can be checked about stores depends on whether the + // write protection actually took. + const bool protected_file = migraphx::sqlite::write(path).read_only(); + + migraphx::gpu::binary_cache reader{settings}; + EXPECT(reader.get(ctx, "existing").has_value()); + EXPECT(reader.get_stats().hits == 1); + + reader.insert(ctx, make_entry("new")); + EXPECT(reader.get(ctx, "new").has_value()); + if(protected_file) + { + EXPECT(row_count(path, "cache_v1") == 1); + } + + // Restored so the temporary directory can be removed, which Windows refuses otherwise. + migraphx::fs::permissions( + path, migraphx::fs::perms::owner_write, migraphx::fs::perm_options::add); +} + // version and device lead the primary key because they are what separates entries this build // may use from entries it may not, so a row stored under one must not be served under another. TEST_CASE(sqlite_scopes_entries_by_version_and_device) @@ -464,6 +558,40 @@ TEST_CASE(sqlite_scopes_entries_by_version_and_device) // Storing a key twice replaces the row rather than accumulating or failing, the way the file // backend's publish-by-rename overwrites in place. Two processes compiling the same kernel is // benign for exactly this reason. +// Storage is opened by the first lookup or insert, not by constructing the cache, since every +// context makes one whether or not it ever compiles anything. +TEST_CASE(storage_is_opened_on_first_use) +{ + migraphx::tmp_dir td{"binary-cache"}; + migraphx::gpu::context ctx; + auto path = db_path(td); + migraphx::gpu::binary_cache cache{migraphx::gpu::binary_cache_settings{path, false}}; + EXPECT(not migraphx::fs::exists(path)); + + EXPECT(not cache.get(ctx, "absent").has_value()); + EXPECT(migraphx::fs::exists(path)); +} + +// Inserts made inside a batch are committed together when it ends, and are all there after. +TEST_CASE(sqlite_batched_inserts_are_all_committed) +{ + migraphx::tmp_dir td{"binary-cache"}; + migraphx::gpu::context ctx; + auto path = db_path(td); + migraphx::gpu::binary_cache cache{migraphx::gpu::binary_cache_settings{path, false}}; + { + migraphx::gpu::binary_cache::store_batch batch{cache}; + cache.insert(ctx, make_entry("first")); + cache.insert(ctx, make_entry("second")); + } + EXPECT(row_count(path, "cache_v1") == 2); + + migraphx::gpu::binary_cache reader{migraphx::gpu::binary_cache_settings{path, false}}; + EXPECT(reader.get(ctx, "first").has_value()); + EXPECT(reader.get(ctx, "second").has_value()); + EXPECT(reader.get_stats().hits == 2); +} + TEST_CASE(sqlite_store_overwrites_in_place) { migraphx::tmp_dir td{"binary-cache"}; diff --git a/test/sqlite.cpp b/test/sqlite.cpp index b97b05862ab..71e5bdba20f 100644 --- a/test/sqlite.cpp +++ b/test/sqlite.cpp @@ -25,9 +25,14 @@ #include #include #include -#include #include +/// Every row a call produced, so a test can count and inspect them. +static std::vector collect(const migraphx::sqlite_stmt::rows& r) +{ + return std::vector(r.begin(), r.end()); +} + TEST_CASE(read_write) { const std::string create_table = R"__migraphx__( @@ -76,47 +81,62 @@ TEST_CASE(prepared_blob_round_trip) ); )__migraphx__"); - // One statement, two inserts: the reset/rebind path backends rely on. + // One statement, two inserts: calling it again rebinds, which is what backends rely on. + // An insert produces no rows, and runs whether or not they are iterated. auto insert = db.prepare("INSERT INTO blob_db (name, size, data) VALUES (?, ?, ?);"); EXPECT(insert.valid()); - - insert.bind(1, std::string_view{"k1"}) - .bind(2, static_cast(blob.size())) - .bind(3, blob); - EXPECT(not insert.step()); - insert.reset(); - - insert.bind(1, std::string_view{"empty"}) - .bind(2, static_cast(0)) - .bind(3, std::vector{}); - EXPECT(not insert.step()); - insert.reset(); + EXPECT(collect(insert("k1", static_cast(blob.size()), blob)).empty()); + insert("empty", std::int64_t{0}, std::vector{}); } { auto db = migraphx::sqlite::read(db_path); - auto select = db.prepare("SELECT name, data FROM blob_db WHERE name = ?;"); - - { - migraphx::sqlite_stmt_reset guard{select}; - select.bind(1, std::string_view{"k1"}); - EXPECT(select.step()); - EXPECT(select.column_text(0) == "k1"); - EXPECT((select.column_blob(1) == blob)); - EXPECT(not select.step()); - } - { - // An empty blob must come back as an empty blob, not as NULL. - migraphx::sqlite_stmt_reset guard{select}; - select.bind(1, std::string_view{"empty"}); - EXPECT(select.step()); - EXPECT(select.column_blob(1).empty()); - } - { - migraphx::sqlite_stmt_reset guard{select}; - select.bind(1, std::string_view{"missing"}); - EXPECT(not select.step()); - } + auto select = db.prepare("SELECT name, size, data FROM blob_db WHERE name = ?;"); + + auto found = collect(select("k1")); + EXPECT(found.size() == 1); + EXPECT(found.front().at("name").get_string() == "k1"); + EXPECT(found.front().at("size").to() == blob.size()); + EXPECT(found.front().at("data").get_binary() == migraphx::value::binary{blob}); + + // An empty blob must come back as an empty blob, not as NULL. + auto empty = collect(select("empty")); + EXPECT(empty.size() == 1); + EXPECT(empty.front().at("data").is_binary()); + EXPECT(empty.front().at("data").get_binary().empty()); + + EXPECT(collect(select("missing")).empty()); + } +} + +// A select abandoned after its first row must not keep holding the database. Until the +// statement is reset it holds a read lock, and a writer on another connection would wait out +// its busy timeout and then fail. +TEST_CASE(abandoned_rows_release_the_database) +{ + migraphx::tmp_dir td{}; + auto db_path = td.path / "lock.db"; + auto writer = migraphx::sqlite::write(db_path); + writer.execute("CREATE TABLE t (id INTEGER PRIMARY KEY);" + "INSERT INTO t (id) VALUES (1), (2);"); + + auto reader = migraphx::sqlite::read(db_path); + auto select = reader.prepare("SELECT id FROM t;"); + { + // Read one of the two rows and stop. + auto rows = select(); + EXPECT(rows.begin() != rows.end()); } + + auto insert = writer.prepare("INSERT INTO t (id) VALUES (?);"); + insert(std::int64_t{3}); + EXPECT(writer.execute("SELECT id FROM t;").size() == 3); +} + +TEST_CASE(unprepared_statement_throws) +{ + migraphx::sqlite_stmt stmt; + EXPECT(not stmt.valid()); + EXPECT(test::throws([&] { stmt(); })); } TEST_CASE(try_write_unusable_path) diff --git a/tools/include/gpu/binary_cache_backend.hpp b/tools/include/gpu/binary_cache_backend.hpp index cc6ddb49a63..cb31fb5d5c4 100644 --- a/tools/include/gpu/binary_cache_backend.hpp +++ b/tools/include/gpu/binary_cache_backend.hpp @@ -30,7 +30,8 @@ // into the gpu target tree). Do not edit the generated header by hand. // // Any type T satisfies the binary_cache_backend concept if it provides the -// member functions listed below. The wrapper holds T by shared_ptr and forwards +// member functions listed below; begin_batch and end_batch are optional and +// default to doing nothing. The wrapper holds T by shared_ptr and forwards // each call through a virtual dispatch, matching problem_cache_backend. // // Notes: @@ -38,9 +39,10 @@ // the include below pulls in its full definition. // * Backends typically own non-trivial resources (a cache directory, a SQLite // connection) and are not meaningfully copyable beyond shared ownership. -// * Both members are non-const: binary_cache::get and insert are themselves -// non-const, so nothing forces a const qualifier here, and a backend holding -// prepared statements needs the mutability. +// * The members are non-const: binary_cache::get and insert are themselves +// non-const, so nothing forces a const qualifier here, and a backend that +// tracks an open batch needs the mutability. A backend may still declare +// them const. // #ifndef MIGRAPHX_GUARD_GPU_BINARY_CACHE_BACKEND_HPP #define MIGRAPHX_GUARD_GPU_BINARY_CACHE_BACKEND_HPP @@ -54,6 +56,7 @@ #include #include +#include #include #include #include @@ -113,6 +116,21 @@ struct binary_cache_backend const std::string& key_hash, const binary_cache_entry& e, const std::vector& blob); + + /// Mark the start of a run of stores that may be committed together, such as + /// a database transaction, rather than one at a time. Every begin_batch is + /// followed by an end_batch, and the two are never nested. Optional: a + /// backend without them stores each entry as it comes. + /// + /// Must not throw. A backend that cannot start a batch stores entries one + /// at a time instead. + void begin_batch(); + + /// Commit the stores made since begin_batch. + /// + /// Must not throw. A failed commit costs those entries a recompile next + /// run, nothing more, and must not leave anything locked. + void end_batch(); }; #else @@ -131,7 +149,9 @@ struct binary_cache_backend device = 'const std::string&', key_hash = 'const std::string&', e = 'const binary_cache_entry&', - blob = 'const std::vector&')) + blob = 'const std::vector&'), + virtual('begin_batch', returns = 'void', default = 'migraphx::nop'), + virtual('end_batch', returns = 'void', default = 'migraphx::nop')) %> #endif From 58ab7b1d7520d7a8453fc8335c18e37ff8a88dca Mon Sep 17 00:00:00 2001 From: pnikolic-amd Date: Fri, 25 Sep 2026 15:42:16 +0200 Subject: [PATCH 06/14] Add tests --- src/targets/gpu/file_binary_cache.cpp | 24 +- test/gpu/binary_cache.cpp | 351 ++++++++++++++++++++++---- test/sqlite.cpp | 56 ++++ 3 files changed, 371 insertions(+), 60 deletions(-) diff --git a/src/targets/gpu/file_binary_cache.cpp b/src/targets/gpu/file_binary_cache.cpp index c5de351836e..9e121dc242c 100644 --- a/src/targets/gpu/file_binary_cache.cpp +++ b/src/targets/gpu/file_binary_cache.cpp @@ -26,7 +26,9 @@ #include #include #include +#include #include +#include #include namespace migraphx { @@ -48,12 +50,26 @@ static fs::path entry_path(const fs::path& root, /// Publish by rename so a reader never sees a half-written file. The temporary stays beside /// the destination since the rename is only atomic within one filesystem. +/// +/// It is a sibling file with a short unique suffix rather than a file inside a temporary +/// directory: entries already sit several directories deep, and a nested directory with a +/// fully unique name pushed the path past Windows' MAX_PATH, which std::ofstream cannot open. +/// The suffix only has to keep concurrent writers of the same entry apart. static void write_atomically(const fs::path& dest, const std::vector& content) { - tmp_dir td{"cache", dest.parent_path()}; - auto tmp = td.path / dest.filename(); - write_buffer(tmp, content); - fs::rename(tmp, dest); + auto suffix = md5(unique_string("cache")).substr(0, 16); + auto tmp = dest.parent_path() / (dest.stem().string() + "." + suffix + ".tmp"); + try + { + write_buffer(tmp, content); + fs::rename(tmp, dest); + } + catch(...) + { + std::error_code ec; + fs::remove(tmp, ec); + throw; + } } optional> file_binary_cache::load(const std::string& version, diff --git a/test/gpu/binary_cache.cpp b/test/gpu/binary_cache.cpp index 4dea3f94034..1967581d4d6 100644 --- a/test/gpu/binary_cache.cpp +++ b/test/gpu/binary_cache.cpp @@ -53,6 +53,8 @@ #include #include #include +#include +#include #include #include @@ -137,6 +139,52 @@ static std::size_t stored_entry_count(const std::string& path) return entry_files(path).size(); } +/// Every row a statement call produced. +static std::vector collect_rows(const migraphx::sqlite_stmt::rows& r) +{ + return std::vector(r.begin(), r.end()); +} + +using stored_entries = std::map>; + +/// Every entry a cache directory holds, keyed by key hash, which names each file. +static stored_entries dir_entries(const migraphx::fs::path& dir) +{ + stored_entries result; + auto files = entry_files(dir); + std::transform(files.begin(), files.end(), std::inserter(result, result.end()), [](auto f) { + return std::make_pair(f.stem().string(), migraphx::read_buffer(f)); + }); + return result; +} + +/// Every entry a cache database holds, keyed by key hash. +static stored_entries db_entries(const std::string& path) +{ + stored_entries result; + auto select = migraphx::sqlite::read(path).prepare("SELECT key_hash, entry FROM cache_v1;"); + auto rows = collect_rows(select()); + std::transform(rows.begin(), rows.end(), std::inserter(result, result.end()), [](auto row) { + const auto& blob = row.at("entry").get_binary(); + return std::make_pair(row.at("key_hash").get_string(), + std::vector(blob.begin(), blob.end())); + }); + return result; +} + +/// One of each backend over fresh storage in td: a directory at td/files and a database at +/// db_path(td). Driven directly, these skip binary_cache and its version and device strings. +static std::vector both_backends(const migraphx::tmp_dir& td) +{ + std::vector result; + result.emplace_back(migraphx::gpu::file_binary_cache{td.path / "files"}); + auto db = migraphx::gpu::sqlite_binary_cache::open(db_path(td)); + EXPECT(db.has_value()); + if(db.has_value()) + result.push_back(*db); + return result; +} + TEST_CASE(lookup_records_a_miss) { migraphx::gpu::context ctx; @@ -164,12 +212,9 @@ TEST_CASE(memory_lookup_records_reuse) } // The cases below are written once against a path and registered for both backends. The -// directory registrations are skipped on Windows for now: driving the file backend through -// binary_cache puts entries under version_id()/device_dir(), and write_atomically then creates a -// temp directory inside that, which pushes the file past Windows' MAX_PATH. fs::create_directories -// succeeds because std::filesystem uses the \\?\ prefix, but the std::ofstream in write_buffer -// does not, so every store fails and nothing is persisted. backends_round_trip_through_the_wrapper -// still covers the file backend there, where it is driven directly and the paths are short. +// directory registrations use real temporary paths on purpose: entries sit under +// version_id()/device_dir(), and on Windows that depth once pushed the entry write past +// MAX_PATH, so running them there is what keeps it from coming back. // A second cache shares nothing in memory, so anything it finds came out of storage. static void disk_lookup_body(const std::string& path) @@ -189,13 +234,11 @@ static void disk_lookup_body(const std::string& path) EXPECT(*found->fragment.get_main_module() == *make_code().fragment.get_main_module()); } -#ifndef _WIN32 TEST_CASE(disk_lookup_records_a_hit) { migraphx::tmp_dir td{"binary-cache"}; disk_lookup_body(dir_path(td)); } -#endif TEST_CASE(sqlite_lookup_records_a_hit) { @@ -222,7 +265,6 @@ static void corrupt_entry_body(const std::string& path, EXPECT(reader.get_stats().misses == 1); } -#ifndef _WIN32 TEST_CASE(corrupt_entry_is_ignored) { migraphx::tmp_dir td{"binary-cache"}; @@ -233,7 +275,6 @@ TEST_CASE(corrupt_entry_is_ignored) migraphx::write_buffer(file, std::vector(8, 0)); }); } -#endif TEST_CASE(sqlite_corrupt_entry_is_ignored) { @@ -339,13 +380,11 @@ static void compiling_twice_body(const std::string& path) gpu_result.to_vector())); } -#ifndef _WIN32 TEST_CASE(compiling_twice_populates_the_cache_and_matches_reference) { migraphx::tmp_dir td{"binary-cache"}; compiling_twice_body(dir_path(td)); } -#endif TEST_CASE(sqlite_compiling_twice_populates_the_cache_and_matches_reference) { @@ -366,13 +405,11 @@ static void verified_reuse_body(const std::string& path) p.compile(migraphx::make_target("gpu"), options); } -#ifndef _WIN32 TEST_CASE(verified_reuse_matches_fresh_compiles) { migraphx::tmp_dir td{"binary-cache"}; verified_reuse_body(dir_path(td)); } -#endif TEST_CASE(sqlite_verified_reuse_matches_fresh_compiles) { @@ -382,11 +419,6 @@ TEST_CASE(sqlite_verified_reuse_matches_fresh_compiles) // The extension of the path picks the backend and nothing else does, so the only way to see the // choice from outside is the artifact it leaves: a database file, or a directory tree. -// -// The directory half checks that the version tree was laid down rather than that an entry file -// landed in it, because the entry write is what MAX_PATH defeats here. create_directories runs -// before that write and succeeds, and the SQLite backend creates no such tree, so this still -// distinguishes the two backends. TEST_CASE(extension_selects_the_backend) { migraphx::gpu::context ctx; @@ -396,7 +428,12 @@ TEST_CASE(extension_selects_the_backend) migraphx::gpu::binary_cache dir_cache{ migraphx::gpu::binary_cache_settings{dir_path(dir_td), false}}; dir_cache.insert(ctx, make_entry("in-a-directory")); + auto files = entry_files(dir_td.path); + EXPECT(files.size() == 1); EXPECT(migraphx::fs::is_directory(dir_td.path / version_dir)); + EXPECT(std::all_of(files.begin(), files.end(), [&](const auto& f) { + return f.parent_path().parent_path() == dir_td.path / version_dir; + })); for(const char* name : {"cache.db", "cache.sqlite"}) { @@ -434,6 +471,155 @@ TEST_CASE(unusable_database_degrades_to_memory) EXPECT(reader.get_stats().misses == 1); } +// A cache path that already holds something other than a cache database is left alone: the +// cache runs from memory, and the file is not overwritten. +TEST_CASE(not_a_database_degrades_to_memory) +{ + migraphx::tmp_dir td{"binary-cache"}; + migraphx::gpu::context ctx; + auto path = db_path(td); + const std::vector garbage(64, 'x'); + migraphx::write_buffer(path, garbage); + + migraphx::gpu::binary_cache cache{migraphx::gpu::binary_cache_settings{path, false}}; + cache.insert(ctx, make_entry("in-memory")); + EXPECT(cache.get(ctx, "in-memory").has_value()); + EXPECT(cache.get_stats().reused == 1); + EXPECT((migraphx::read_buffer(path) == garbage)); +} + +// A database whose cache table has a different shape, as a future or foreign version might +// leave, cannot be stored into or looked up in, so it is skipped rather than half used. +TEST_CASE(incompatible_schema_degrades_to_memory) +{ + migraphx::tmp_dir td{"binary-cache"}; + migraphx::gpu::context ctx; + auto path = db_path(td); + migraphx::sqlite::write(path).execute("CREATE TABLE cache_v1 (unrelated INTEGER);"); + + EXPECT(not migraphx::gpu::sqlite_binary_cache::open(path).has_value()); + + migraphx::gpu::binary_cache cache{migraphx::gpu::binary_cache_settings{path, false}}; + cache.insert(ctx, make_entry("in-memory")); + EXPECT(cache.get(ctx, "in-memory").has_value()); + EXPECT(row_count(path, "cache_v1") == 0); +} + +// Both backends store the same serialized entry, so a cache can move between them. +TEST_CASE(backends_store_identical_bytes) +{ + migraphx::gpu::context ctx; + auto e = make_entry("interchange"); + + migraphx::tmp_dir dir_td{"binary-cache"}; + migraphx::gpu::binary_cache dir_cache{ + migraphx::gpu::binary_cache_settings{dir_path(dir_td), false}}; + dir_cache.insert(ctx, e); + auto files = entry_files(dir_td.path); + EXPECT(files.size() == 1); + auto from_file = migraphx::read_buffer(files.front()); + EXPECT(not from_file.empty()); + + migraphx::tmp_dir db_td{"binary-cache"}; + auto path = db_path(db_td); + migraphx::gpu::binary_cache db_cache{migraphx::gpu::binary_cache_settings{path, false}}; + db_cache.insert(ctx, e); + + auto select = migraphx::sqlite::read(path).prepare("SELECT entry FROM cache_v1;"); + auto rows = select(); + auto it = rows.begin(); + EXPECT(it != rows.end()); + auto row = *it; + const auto& from_db = row.at("entry").get_binary(); + EXPECT((std::vector(from_db.begin(), from_db.end()) == from_file)); +} + +// A whole compile against each backend has to leave the same kernels behind: the same key +// hashes, each with the same bytes. That makes the choice of backend purely a storage decision. +TEST_CASE(backends_hold_the_same_entries_after_a_compile) +{ + migraphx::tmp_dir dir_td{"binary-cache"}; + migraphx::tmp_dir db_td{"binary-cache"}; + auto path = db_path(db_td); + + auto p_dir = pointwise_program(); + p_dir.compile(migraphx::make_target("gpu"), cache_options(dir_path(dir_td))); + auto p_db = pointwise_program(); + p_db.compile(migraphx::make_target("gpu"), cache_options(path)); + + auto from_dir = dir_entries(dir_td.path); + auto from_db = db_entries(path); + EXPECT(not from_dir.empty()); + EXPECT(from_dir.size() == from_db.size()); + EXPECT((from_dir == from_db)); +} + +// An entry copied from one backend into the other is a hit there and decodes to the same code, +// so an existing cache can be converted rather than rebuilt. The only translation is the +// version, which a directory names with the short id and a database records in full. +TEST_CASE(entries_move_between_backends) +{ + migraphx::gpu::context ctx; + const auto& short_version = migraphx::gpu::binary_cache::version_id(true); + const auto& long_version = migraphx::gpu::binary_cache::version_id(false); + auto e = make_entry("moving"); + auto expected = make_code(); + + // Directory to database. + { + migraphx::tmp_dir dir_td{"binary-cache"}; + migraphx::tmp_dir db_td{"binary-cache"}; + migraphx::gpu::binary_cache writer{ + migraphx::gpu::binary_cache_settings{dir_path(dir_td), false}}; + writer.insert(ctx, e); + auto files = entry_files(dir_td.path); + EXPECT(files.size() == 1); + + const auto& file = files.front(); + auto device = file.parent_path().filename().string(); + auto db = migraphx::gpu::sqlite_binary_cache::open(db_path(db_td)); + EXPECT(db.has_value()); + db->store(long_version, device, file.stem().string(), e, migraphx::read_buffer(file)); + + migraphx::gpu::binary_cache reader{ + migraphx::gpu::binary_cache_settings{db_path(db_td), false}}; + auto found = reader.get(ctx, e.key); + EXPECT(found.has_value()); + EXPECT(reader.get_stats().hits == 1); + EXPECT(*found->fragment.get_main_module() == *expected.fragment.get_main_module()); + } + + // Database to directory. + { + migraphx::tmp_dir db_td{"binary-cache"}; + migraphx::tmp_dir dir_td{"binary-cache"}; + migraphx::gpu::binary_cache writer{ + migraphx::gpu::binary_cache_settings{db_path(db_td), false}}; + writer.insert(ctx, e); + + auto select = migraphx::sqlite::read(db_path(db_td)) + .prepare("SELECT version, device, key_hash, entry FROM cache_v1;"); + auto rows = collect_rows(select()); + EXPECT(rows.size() == 1); + const auto& row = rows.front(); + EXPECT(row.at("version").get_string() == long_version); + const auto& blob = row.at("entry").get_binary(); + migraphx::gpu::file_binary_cache files{dir_td.path}; + files.store(short_version, + row.at("device").get_string(), + row.at("key_hash").get_string(), + e, + std::vector(blob.begin(), blob.end())); + + migraphx::gpu::binary_cache reader{ + migraphx::gpu::binary_cache_settings{dir_path(dir_td), false}}; + auto found = reader.get(ctx, e.key); + EXPECT(found.has_value()); + EXPECT(reader.get_stats().hits == 1); + EXPECT(*found->fragment.get_main_module() == *expected.fragment.get_main_module()); + } +} + // Two connections over one database, as two processes compiling against a shared cache would // have. This cannot be two sqlite_binary_cache objects: only open() populates one, and it hands // back the type-erased wrapper. @@ -539,25 +725,72 @@ TEST_CASE(sqlite_read_only_database_still_serves_hits) path, migraphx::fs::perms::owner_write, migraphx::fs::perm_options::add); } -// version and device lead the primary key because they are what separates entries this build -// may use from entries it may not, so a row stored under one must not be served under another. -TEST_CASE(sqlite_scopes_entries_by_version_and_device) +// version and device separate entries this build may use from entries it may not, so an entry +// stored under one must not be served under another, whichever backend holds it. +TEST_CASE(backends_scope_entries_by_version_and_device) { migraphx::tmp_dir td{"binary-cache"}; - auto backend = migraphx::gpu::sqlite_binary_cache::open(db_path(td)); - EXPECT(backend.has_value()); - const std::vector blob{'p', 'a', 'y'}; - backend->store("v1", "dev1", "k", make_entry("k"), blob); + for(auto& backend : both_backends(td)) + { + backend.store("v1", "dev1", "k", make_entry("k"), blob); - EXPECT(backend->load("v1", "dev1", "k").has_value()); - EXPECT(not backend->load("v2", "dev1", "k").has_value()); - EXPECT(not backend->load("v1", "dev2", "k").has_value()); + EXPECT(backend.load("v1", "dev1", "k").has_value()); + EXPECT(not backend.load("v2", "dev1", "k").has_value()); + EXPECT(not backend.load("v1", "dev2", "k").has_value()); + } +} + +// Storing a key twice replaces the entry rather than accumulating or failing. Two processes +// compiling the same kernel is benign for exactly this reason. +TEST_CASE(backends_store_overwrites_in_place) +{ + migraphx::tmp_dir td{"binary-cache"}; + const std::vector replacement{'n', 'e', 'w'}; + for(auto& backend : both_backends(td)) + { + backend.store("v", "dev", "k", make_entry("k"), {'o', 'l', 'd'}); + backend.store("v", "dev", "k", make_entry("k"), replacement); + + auto got = backend.load("v", "dev", "k"); + EXPECT(got.has_value()); + EXPECT((*got == replacement)); + } + EXPECT(row_count(db_path(td), "cache_v1") == 1); + EXPECT(entry_files(td.path / "files").size() == 1); +} + +// Publishing an entry goes through a temporary beside it, and nothing of that may be left once +// the entry is in place, including when an existing entry is replaced. +TEST_CASE(file_store_leaves_only_entries_behind) +{ + migraphx::tmp_dir td{"binary-cache"}; + migraphx::gpu::context ctx; + migraphx::gpu::binary_cache_settings settings{dir_path(td), false}; + + migraphx::gpu::binary_cache first{settings}; + first.insert(ctx, make_entry("one")); + first.insert(ctx, make_entry("two")); + // A second cache stores the same key again, over the file the first one published. + migraphx::gpu::binary_cache second{settings}; + second.insert(ctx, make_entry("one")); + + std::vector files; + std::vector dirs; + for(const auto& item : migraphx::fs::recursive_directory_iterator(td.path)) + { + if(item.is_directory()) + dirs.push_back(item.path()); + else + files.push_back(item.path()); + } + EXPECT(files.size() == 2); + EXPECT(std::all_of( + files.begin(), files.end(), [](const auto& f) { return f.extension() == ".mxr"; })); + // Just the version directory and the device directory inside it. + EXPECT(dirs.size() == 2); } -// Storing a key twice replaces the row rather than accumulating or failing, the way the file -// backend's publish-by-rename overwrites in place. Two processes compiling the same kernel is -// benign for exactly this reason. // Storage is opened by the first lookup or insert, not by constructing the cache, since every // context makes one whether or not it ever compiles anything. TEST_CASE(storage_is_opened_on_first_use) @@ -572,19 +805,18 @@ TEST_CASE(storage_is_opened_on_first_use) EXPECT(migraphx::fs::exists(path)); } -// Inserts made inside a batch are committed together when it ends, and are all there after. -TEST_CASE(sqlite_batched_inserts_are_all_committed) +// Inserts made inside a batch are all there once it ends. The database commits them together; +// the directory backend has no batching and stores each one as it comes. +static void batched_inserts_body(const std::string& path) { - migraphx::tmp_dir td{"binary-cache"}; migraphx::gpu::context ctx; - auto path = db_path(td); migraphx::gpu::binary_cache cache{migraphx::gpu::binary_cache_settings{path, false}}; { migraphx::gpu::binary_cache::store_batch batch{cache}; cache.insert(ctx, make_entry("first")); cache.insert(ctx, make_entry("second")); } - EXPECT(row_count(path, "cache_v1") == 2); + EXPECT(stored_entry_count(path) == 2); migraphx::gpu::binary_cache reader{migraphx::gpu::binary_cache_settings{path, false}}; EXPECT(reader.get(ctx, "first").has_value()); @@ -592,21 +824,35 @@ TEST_CASE(sqlite_batched_inserts_are_all_committed) EXPECT(reader.get_stats().hits == 2); } -TEST_CASE(sqlite_store_overwrites_in_place) +TEST_CASE(batched_inserts_are_all_stored) { migraphx::tmp_dir td{"binary-cache"}; - auto path = db_path(td); - auto backend = migraphx::gpu::sqlite_binary_cache::open(path); - EXPECT(backend.has_value()); + batched_inserts_body(dir_path(td)); +} - const std::vector replacement{'n', 'e', 'w'}; - backend->store("v", "dev", "k", make_entry("k"), {'o', 'l', 'd'}); - backend->store("v", "dev", "k", make_entry("k"), replacement); +TEST_CASE(sqlite_batched_inserts_are_all_committed) +{ + migraphx::tmp_dir td{"binary-cache"}; + batched_inserts_body(db_path(td)); +} - EXPECT(row_count(path, "cache_v1") == 1); - auto got = backend->load("v", "dev", "k"); - EXPECT(got.has_value()); - EXPECT((*got == replacement)); +// A batch holds the database's write lock, so another connection must be able to write again as +// soon as it ends. +TEST_CASE(sqlite_batch_releases_the_database) +{ + migraphx::tmp_dir td{"binary-cache"}; + migraphx::gpu::context ctx; + auto path = db_path(td); + migraphx::gpu::binary_cache cache{migraphx::gpu::binary_cache_settings{path, false}}; + { + migraphx::gpu::binary_cache::store_batch batch{cache}; + cache.insert(ctx, make_entry("batched")); + } + + auto other = migraphx::gpu::sqlite_binary_cache::open(path); + EXPECT(other.has_value()); + other->store("v", "dev", "k", make_entry("k"), {'x'}); + EXPECT(row_count(path, "cache_v1") == 2); } // The backend layer moves opaque bytes and never decodes them, so a payload that is not even @@ -618,14 +864,7 @@ TEST_CASE(backends_round_trip_through_the_wrapper) const std::vector blob{'\0', 'n', 'o', 't', '\0', 'm', 's', 'g', '\xff'}; auto e = make_entry("opaque"); - auto db = migraphx::gpu::sqlite_binary_cache::open((td.path / "cache.db").string()); - EXPECT(db.has_value()); - - std::vector backends; - backends.emplace_back(migraphx::gpu::file_binary_cache{td.path / "files"}); - backends.push_back(*db); - - for(auto& backend : backends) + for(auto& backend : both_backends(td)) { EXPECT(not backend.load("v", "dev", "k").has_value()); backend.store("v", "dev", "k", e, blob); diff --git a/test/sqlite.cpp b/test/sqlite.cpp index 71e5bdba20f..8f84e111c6c 100644 --- a/test/sqlite.cpp +++ b/test/sqlite.cpp @@ -21,10 +21,13 @@ * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN * THE SOFTWARE. */ +#include #include #include #include +#include #include +#include #include /// Every row a call produced, so a test can count and inspect them. @@ -132,6 +135,59 @@ TEST_CASE(abandoned_rows_release_the_database) EXPECT(writer.execute("SELECT id FROM t;").size() == 3); } +// Each column comes back as the value type matching what sqlite stored, keyed by its name. +TEST_CASE(rows_convert_column_types) +{ + migraphx::tmp_dir td{}; + auto db = migraphx::sqlite::write(td.path / "types.db"); + auto select = + db.prepare("SELECT 42 AS i, 2.5 AS f, 'text' AS t, x'00ff' AS b, NULL AS n, ?1 AS p;"); + + auto rows = collect(select(std::int64_t{-7})); + EXPECT(rows.size() == 1); + const auto& row = rows.front(); + EXPECT(row.size() == 6); + EXPECT(row.at("i").is_int64()); + EXPECT(row.at("i").get_int64() == 42); + EXPECT(row.at("f").is_float()); + EXPECT(migraphx::float_equal(row.at("f").get_float(), 2.5)); + EXPECT(row.at("t").get_string() == "text"); + EXPECT(row.at("b").get_binary() == migraphx::value::binary{std::vector{0, 255}}); + EXPECT(row.at("n").is_null()); + EXPECT(row.at("p").get_int64() == -7); +} + +// A statement returning many rows yields each in turn, and calling it again starts over. +TEST_CASE(rows_iterate_in_order_and_restart) +{ + migraphx::tmp_dir td{}; + auto db = migraphx::sqlite::write(td.path / "many.db"); + db.execute("CREATE TABLE t (id INTEGER PRIMARY KEY);" + "INSERT INTO t (id) VALUES (1), (2), (3);"); + auto select = db.prepare("SELECT id FROM t WHERE id >= ?1 ORDER BY id;"); + + auto ids = [&](std::int64_t from) { + std::vector result; + auto rows = select(from); + std::transform(rows.begin(), rows.end(), std::back_inserter(result), [](const auto& row) { + return row.at("id").get_int64(); + }); + return result; + }; + EXPECT((ids(1) == std::vector{1, 2, 3})); + EXPECT((ids(2) == std::vector{2, 3})); + EXPECT(ids(4).empty()); +} + +TEST_CASE(read_only_matches_how_it_was_opened) +{ + migraphx::tmp_dir td{}; + auto path = td.path / "mode.db"; + migraphx::sqlite::write(path).execute("CREATE TABLE t (id INTEGER PRIMARY KEY);"); + EXPECT(not migraphx::sqlite::write(path).read_only()); + EXPECT(migraphx::sqlite::read(path).read_only()); +} + TEST_CASE(unprepared_statement_throws) { migraphx::sqlite_stmt stmt; From d708b88d2d4b0dbebc32c276977b8823df7cc121 Mon Sep 17 00:00:00 2001 From: pnikolic-amd Date: Fri, 25 Sep 2026 15:55:34 +0200 Subject: [PATCH 07/14] migraphx simplify Co-Authored-By: Claude Opus 5.5 (1M context) --- docs/dev/triage-migraphx.rst | 11 +- src/include/migraphx/sqlite.hpp | 29 ++- src/sqlite.cpp | 79 +++--- src/targets/gpu/binary_cache.cpp | 74 +++--- src/targets/gpu/compile_ops.cpp | 6 +- src/targets/gpu/file_binary_cache.cpp | 20 +- .../gpu/include/migraphx/gpu/binary_cache.hpp | 19 +- .../migraphx/gpu/binary_cache_backend.hpp | 23 +- .../migraphx/gpu/binary_cache_entry.hpp | 6 +- .../migraphx/gpu/file_binary_cache.hpp | 2 +- .../migraphx/gpu/sqlite_binary_cache.hpp | 23 +- src/targets/gpu/sqlite_binary_cache.cpp | 53 ++--- test/gpu/binary_cache.cpp | 225 +++++++----------- tools/include/gpu/binary_cache_backend.hpp | 23 +- 14 files changed, 254 insertions(+), 339 deletions(-) diff --git a/docs/dev/triage-migraphx.rst b/docs/dev/triage-migraphx.rst index efdd3019081..6f1dcac722b 100644 --- a/docs/dev/triage-migraphx.rst +++ b/docs/dev/triage-migraphx.rst @@ -176,8 +176,8 @@ directories for builds you no longer use: rm -r $HOME/.cache/migraphx/v1-hip22.0.* A path ending in ``.db`` or ``.sqlite`` keeps the cache in a single SQLite database instead of a -directory, which is easier to copy between machines and can be shared by processes compiling at -the same time. Missing parent directories are created, as for a directory cache: +directory, which is easier to copy between machines. Like a directory cache it can be used by +several processes at once, and missing parent directories are created: .. code-block:: bash @@ -192,11 +192,12 @@ with the ``sqlite3`` shell. Deleting rows does not shrink the file until it is v sqlite3 $HOME/.cache/migraphx/kernels.db \ "SELECT version, op_name, count(*) FROM cache_v1 GROUP BY version, op_name;" sqlite3 $HOME/.cache/migraphx/kernels.db \ - "DELETE FROM cache_v1 WHERE version LIKE 'v1-hip22.0.%'; VACUUM;" + "DELETE FROM cache_v1 WHERE version GLOB 'v1-hip22.0.*'; VACUUM;" A database that cannot be written to, such as a shared cache installed read-only, is still used -for lookups, and newly compiled kernels are kept in memory only. A database that cannot be opened -at all is skipped with a warning and the compile proceeds without a disk cache. +for lookups, and newly compiled kernels are kept in memory only. A database that cannot be +opened, or whose cache table has an unexpected layout, is skipped with a warning and the compile +proceeds without a disk cache. The same settings are available as backend options, which take precedence over the environment and are how tests configure the cache: diff --git a/src/include/migraphx/sqlite.hpp b/src/include/migraphx/sqlite.hpp index f65217ae2e0..1a9dc6a5355 100644 --- a/src/include/migraphx/sqlite.hpp +++ b/src/include/migraphx/sqlite.hpp @@ -31,12 +31,12 @@ #include #include #include +#include #include #include #include #include #include -#include #include #include @@ -50,10 +50,10 @@ struct sqlite_stmt_impl; /// never outlive it. Copies share the same statement. /// /// Calling it with arguments runs it: the arguments are bound to the parameters in order and the -/// result comes back as a range of rows. Preparing once and calling many times is the point. +/// result comes back as a range of rows. Since copies share one statement, only the rows of one +/// call may be alive at a time. /// -/// Not thread safe: one statement may be used by one thread at a time, even though the -/// connection itself is serialized. +/// Not thread safe: use a statement from one thread at a time. struct MIGRAPHX_EXPORT sqlite_stmt { struct rows; @@ -66,9 +66,11 @@ struct MIGRAPHX_EXPORT sqlite_stmt { if(not valid()) MIGRAPHX_THROW("sqlite: calling a statement that was never prepared"); + assert(sizeof...(Ts) == parameter_count()); // Anything left from the previous call, bindings or an unfinished result, goes first. reset(); - sequence_c([&](auto... is) { swallow{(bind(int{is + 1}, xs), 0)...}; }); + int i = 0; + each_args([&](const auto& x) { bind(++i, x); }, xs...); return rows{*this}; } @@ -80,6 +82,8 @@ struct MIGRAPHX_EXPORT sqlite_stmt void bind(int i, std::int64_t x) const; void bind(int i, const std::vector& blob) const; + std::size_t parameter_count() const; + /// Step once. True when a row is available, false when the statement is done. bool step() const; @@ -103,7 +107,6 @@ struct MIGRAPHX_EXPORT sqlite_stmt /// read lock on the database until then, which would stall writers in other processes. struct sqlite_stmt::rows { - explicit rows(sqlite_stmt s) : stmt(std::move(s)), first(stmt.step()) {} // Only ever a prvalue returned from a call, so it never needs copying or moving, and a copy // would reset the statement out from under the original. rows(const rows&) = delete; @@ -118,17 +121,22 @@ struct sqlite_stmt::rows using reference = value_type; using difference_type = std::ptrdiff_t; using iterator_category = std::input_iterator_tag; - using pointer = std::add_pointer_t>; + using pointer = value*; iterator() = default; iterator(const rows* pparent, bool pavailable) : parent(pparent), available(pavailable) {} - reference operator*() const { return parent->stmt.to_value(); } + reference operator*() const + { + assert(parent != nullptr and available); + return parent->stmt.to_value(); + } template static void increment(U& x) { + assert(x.parent != nullptr and x.available); x.available = x.parent->stmt.step(); } @@ -147,6 +155,9 @@ struct sqlite_stmt::rows iterator end() const { return {this, false}; } private: + friend struct sqlite_stmt; + explicit rows(sqlite_stmt s) : stmt(std::move(s)), first(stmt.step()) {} + sqlite_stmt stmt; bool first = false; }; @@ -172,8 +183,6 @@ struct MIGRAPHX_EXPORT sqlite /// has write-protected, in which case sqlite quietly opens it read-only; this is how to tell. bool read_only() const; - bool valid() const { return impl != nullptr; } - private: std::shared_ptr impl; }; diff --git a/src/sqlite.cpp b/src/sqlite.cpp index 619089aced5..fb15a844481 100644 --- a/src/sqlite.cpp +++ b/src/sqlite.cpp @@ -28,6 +28,7 @@ #include #include #include +#include #include namespace migraphx { @@ -39,9 +40,8 @@ struct sqlite_impl { sqlite3* get() const { return ptr.get(); } - // Returns false rather than throwing, so callers that treat an unusable database as - // "no cache" do not have to catch. sqlite3_open_v2 hands back a handle even on failure - // (that is where the error message lives), so `ptr` takes ownership either way. + // sqlite3_open_v2 returns a handle even on failure (it carries the error message), so ptr + // takes ownership either way. bool try_open(const fs::path& p, int flags) { sqlite3* ptr_tmp = nullptr; @@ -92,10 +92,8 @@ struct sqlite_stmt_impl sqlite3_stmt* get() const { return ptr.get(); } std::string error_message() const { return db->error_message(); } - // Holding the connection keeps it alive for as long as any statement prepared on it, - // since finalizing after the connection is closed is undefined. Declaration order is - // load-bearing: members destruct in reverse, so ptr is finalized before db is released. - // Do not reorder. + // Holding the connection keeps it alive while any statement on it exists. ptr is declared + // after db so it is finalized first; finalizing after the connection closes is undefined. std::shared_ptr db; sqlite3_stmt_ptr ptr; }; @@ -154,6 +152,8 @@ sqlite_stmt sqlite::prepare(const std::string& sql) result.impl->ptr = sqlite3_stmt_ptr{stmt_tmp}; if(rc != SQLITE_OK) MIGRAPHX_THROW("error preparing '" + sql + "': " + impl->error_message()); + // sqlite succeeds without a statement for text that holds none, such as only a comment. + assert(stmt_tmp != nullptr); return result; } @@ -193,6 +193,11 @@ void sqlite_stmt::bind(int i, const std::vector& blob) const MIGRAPHX_THROW(impl->error_message()); } +std::size_t sqlite_stmt::parameter_count() const +{ + return sqlite3_bind_parameter_count(impl->get()); +} + bool sqlite_stmt::step() const { int rc = sqlite3_step(impl->get()); @@ -205,56 +210,50 @@ bool sqlite_stmt::step() const void sqlite_stmt::reset() const noexcept { - if(impl == nullptr) - return; + assert(impl != nullptr); // The return of sqlite3_reset is the error from the preceding step(), which the caller // has already seen as a throw. There is nothing new to report, and this must not throw. (void)sqlite3_reset(impl->get()); (void)sqlite3_clear_bindings(impl->get()); } +/// Column i of the current row, keyed by its name. static value column_value(sqlite3_stmt* stmt, int i) { - // For text and blobs the data must be fetched before sqlite3_column_bytes: the other order - // can force a type conversion that invalidates the pointer. A zero-length value comes back - // as a null pointer, which still means empty rather than NULL. - switch(sqlite3_column_type(stmt, i)) + std::string name = sqlite3_column_name(stmt, i); + auto type = sqlite3_column_type(stmt, i); + switch(type) { - case SQLITE_INTEGER: return std::int64_t{sqlite3_column_int64(stmt, i)}; - case SQLITE_FLOAT: return sqlite3_column_double(stmt, i); - case SQLITE_TEXT: { - const auto* text = sqlite3_column_text(stmt, i); - auto n = sqlite3_column_bytes(stmt, i); - if(text == nullptr) - return std::string{}; - // NOLINTNEXTLINE(cppcoreguidelines-pro-type-reinterpret-cast) - return std::string(reinterpret_cast(text), n); - } + case SQLITE_INTEGER: return value(name, std::int64_t{sqlite3_column_int64(stmt, i)}); + case SQLITE_FLOAT: return value(name, sqlite3_column_double(stmt, i)); + case SQLITE_TEXT: case SQLITE_BLOB: { - const auto* data = static_cast(sqlite3_column_blob(stmt, i)); - auto n = sqlite3_column_bytes(stmt, i); - if(data == nullptr) - return value::binary{}; - return value::binary{data, static_cast(n)}; + // The data must be fetched before sqlite3_column_bytes: the other order can force a + // conversion that invalidates the pointer. Text comes back unchanged through + // sqlite3_column_blob, and a zero-length value as a null pointer, which means empty. + const auto* data = static_cast(sqlite3_column_blob(stmt, i)); + auto bytes = sqlite3_column_bytes(stmt, i); + assert(bytes >= 0); + auto size = data == nullptr ? 0 : static_cast(bytes); + if(type == SQLITE_TEXT) + return value(name, size == 0 ? std::string{} : std::string(data, size)); + return value(name, value::binary{data, size}); } - default: return {}; + default: return value(name, nullptr); } } value sqlite_stmt::to_value() const { auto* stmt = impl->get(); - value::object row; - auto columns = range(sqlite3_column_count(stmt)); - std::transform(columns.begin(), - columns.end(), - std::inserter(row, row.end()), - [&](std::ptrdiff_t i) { - auto col = static_cast(i); - return std::make_pair(std::string{sqlite3_column_name(stmt, col)}, - column_value(stmt, col)); - }); - return row; + // Built as keyed values rather than a map, so a blob is moved into place instead of copied. + std::vector columns; + auto indices = range(sqlite3_column_count(stmt)); + std::transform(indices.begin(), + indices.end(), + std::back_inserter(columns), + [&](std::ptrdiff_t i) { return column_value(stmt, static_cast(i)); }); + return value(columns, /* array_on_empty */ false); } } // namespace MIGRAPHX_INLINE_NS diff --git a/src/targets/gpu/binary_cache.cpp b/src/targets/gpu/binary_cache.cpp index 55fa75fb83d..f7366f48f23 100644 --- a/src/targets/gpu/binary_cache.cpp +++ b/src/targets/gpu/binary_cache.cpp @@ -105,9 +105,8 @@ static std::string device_dir(const context& ctx) "_wf" + std::to_string(device.get_wavefront_size()); } -/// Turn a stored blob back into an entry. Any failure is just a miss, so a damaged entry costs -/// a recompile and is written over. Shared by every backend, so they all tolerate corruption -/// and survive a hash collision the same way. +/// Turn a stored blob back into an entry. Any failure is a miss, so a damaged entry costs a +/// recompile. static optional decode_entry(const std::vector& blob, const std::string& key, const std::string& key_hash) { @@ -131,54 +130,43 @@ decode_entry(const std::vector& blob, const std::string& key, const std::s return e; } -// Select the storage backend by file type, the same rule make_problem_cache_backend applies in -// problem_cache.cpp: a ".db"/".sqlite" path is a SQLite database, anything else is a directory -// of entries. -static bool is_database_path(const std::string& path) -{ - return ends_with(path, ".db") or ends_with(path, ".sqlite"); -} - -static optional make_binary_cache_backend(const std::string& path) -{ - if(is_database_path(path)) - return sqlite_binary_cache::open(path); // nullopt when the database is unusable - return binary_cache_backend{file_binary_cache{path}}; -} - binary_cache::binary_cache(binary_cache_settings s) : settings(std::move(s)) {} -// Nothing can be persisted safely when the compiler cannot be identified, since entries from -// different toolchains would be indistinguishable. That is a property of the cache rather than -// of the storage medium, so it is checked here instead of in each backend. +// The storage backend is selected by file type, the same rule make_problem_cache_backend applies +// in problem_cache.cpp: a ".db"/".sqlite" path is a SQLite database, anything else is a +// directory of entries. A directory is named with the short version id to keep paths short; a +// database records the full id, which is self-describing. Nothing is persisted when the compiler +// cannot be identified, since entries from different toolchains would be indistinguishable. binary_cache_backend* binary_cache::get_backend() { - if(not backend_opened) - { - backend_opened = true; - // Checked first so that a memory-only cache never compiles the version probe. - if(not settings.path.empty()) - { - // The version names a directory for the file backend, so it is kept short there. A - // database has no such limit and records the full id, which is self-describing. - version = version_id(not is_database_path(settings.path)); - if(not version.empty()) - backend = make_binary_cache_backend(settings.path); - } - } + if(backend_opened) + return backend.has_value() ? &*backend : nullptr; + backend_opened = true; + const auto& path = settings.path; + // Checked first so that a memory-only cache never compiles the version probe. + if(path.empty()) + return nullptr; + const bool database = ends_with(path, ".db") or ends_with(path, ".sqlite"); + version = version_id(not database); + if(version.empty()) + return nullptr; + if(not database) + backend = binary_cache_backend{file_binary_cache{path}}; + else if(auto db = sqlite_binary_cache::open(path)) + backend = binary_cache_backend{std::move(*db)}; return backend.has_value() ? &*backend : nullptr; } -binary_cache::store_batch::store_batch(binary_cache& c) : cache(&c) +binary_cache::store_batch::store_batch(binary_cache& c) : backend(c.get_backend()) { - if(auto* b = cache->get_backend()) - b->begin_batch(); + if(backend != nullptr) + backend->begin_batch(); } binary_cache::store_batch::~store_batch() { - if(auto* b = cache->get_backend()) - b->end_batch(); + if(backend != nullptr) + backend->end_batch(); } optional binary_cache::get(const context& ctx, const std::string& key) @@ -193,8 +181,8 @@ optional binary_cache::get(const context& ctx, const std::string& } if(auto* b = get_backend()) { - // Hashing the key is not free -- it is the whole compile source, which runs to - // kilobytes -- so it is done once and reused for the lookup and any diagnostics. + // The key is the whole compile source, so it is hashed once for the lookup and any + // diagnostics. auto key_hash = md5(key); auto blob = b->load(version, device_dir(ctx), key_hash); if(blob.has_value()) @@ -221,9 +209,7 @@ void binary_cache::insert(const context& ctx, entry e) auto key_hash = md5(e.key); try { - // Serializing inside the try, rather than in the call's argument list, makes a - // failure here a warning like any other storage failure instead of escaping - // insert() and failing the compile. + // A failure to serialize or store is a warning, not a failed compile. auto blob = to_msgpack(migraphx::to_value(e)); b->store(version, device_dir(ctx), key_hash, e, blob); } diff --git a/src/targets/gpu/compile_ops.cpp b/src/targets/gpu/compile_ops.cpp index 8ea8289523e..b2e20598c3e 100644 --- a/src/targets/gpu/compile_ops.cpp +++ b/src/targets/gpu/compile_ops.cpp @@ -808,7 +808,11 @@ struct compile_manager { // Every plan compiles with the same context, so the stores all go to one cache, and // batching them lets its storage commit them together rather than one at a time. - binary_cache::store_batch batch{tasks.front().first->ctx->get_binary_cache()}; + auto* ctx = tasks.front().first->ctx; + assert(std::all_of(tasks.begin(), tasks.end(), [&](const auto& task) { + return task.first->ctx == ctx; + })); + binary_cache::store_batch batch{ctx->get_binary_cache()}; for(const auto& [cp, cell] : tasks) { if(not cell->result.has_value()) diff --git a/src/targets/gpu/file_binary_cache.cpp b/src/targets/gpu/file_binary_cache.cpp index 9e121dc242c..2ccd6f0d32b 100644 --- a/src/targets/gpu/file_binary_cache.cpp +++ b/src/targets/gpu/file_binary_cache.cpp @@ -49,12 +49,9 @@ static fs::path entry_path(const fs::path& root, } /// Publish by rename so a reader never sees a half-written file. The temporary stays beside -/// the destination since the rename is only atomic within one filesystem. -/// -/// It is a sibling file with a short unique suffix rather than a file inside a temporary -/// directory: entries already sit several directories deep, and a nested directory with a -/// fully unique name pushed the path past Windows' MAX_PATH, which std::ofstream cannot open. -/// The suffix only has to keep concurrent writers of the same entry apart. +/// the destination since the rename is only atomic within one filesystem. Its short random +/// suffix keeps concurrent writers of the same entry apart without lengthening an already deep +/// path, which on Windows must stay under MAX_PATH for std::ofstream to open it. static void write_atomically(const fs::path& dest, const std::vector& content) { auto suffix = md5(unique_string("cache")).substr(0, 16); @@ -100,15 +97,8 @@ void file_binary_cache::store(const std::string& version, auto path = entry_path(root, version, device, key_hash); // The content is decided entirely by the key, so a writer that loses the publish race // replaces the file with the same bytes and no locking is needed. - try - { - fs::create_directories(path.parent_path()); - write_atomically(path, blob); - } - catch(const std::exception& ex) - { - log::warn() << "Failed to write binary cache entry " << path << ": " << ex.what(); - } + fs::create_directories(path.parent_path()); + write_atomically(path, blob); } } // namespace gpu diff --git a/src/targets/gpu/include/migraphx/gpu/binary_cache.hpp b/src/targets/gpu/include/migraphx/gpu/binary_cache.hpp index 887915135e2..f136b1b5dc3 100644 --- a/src/targets/gpu/include/migraphx/gpu/binary_cache.hpp +++ b/src/targets/gpu/include/migraphx/gpu/binary_cache.hpp @@ -30,8 +30,6 @@ #include #include #include -#include -#include #include #include #include @@ -58,9 +56,9 @@ struct binary_cache_settings /** * Compiled kernels, keyed by a string describing what the compiler was given. * - * Results are held in memory for the life of the context and, when a cache directory is - * configured, written to disk so later runs can reuse them. Things outside the key, such as the - * compiler and the embedded kernel headers, are separated by the directory the entries live in. + * Results are held in memory for the life of the context and, when a cache path is configured, + * written to disk so later runs can reuse them. Things outside the key, such as the compiler and + * the embedded kernel headers, are separated by the version the entries are stored under. * * This is not thread safe, and deliberately so. Every key is known before any compile begins, so * the compile pass looks results up and stores them in serial passes on either side of its @@ -68,8 +66,7 @@ struct binary_cache_settings */ struct MIGRAPHX_GPU_EXPORT binary_cache { - /// What gets stored for one compiled kernel. Defined in binary_cache_entry.hpp so the - /// storage backends can name it; the alias keeps binary_cache::entry working. + /// What gets stored for one compiled kernel; see binary_cache_entry.hpp. using entry = binary_cache_entry; /// Counts of what the cache did. @@ -98,14 +95,14 @@ struct MIGRAPHX_GPU_EXPORT binary_cache ~store_batch(); private: - binary_cache* cache; + binary_cache_backend* backend; }; - /// Nothing is opened here; storage is set up by the first lookup or insert, so a context - /// that never compiles never touches the disk or probes the compiler. + /// Nothing is opened here; storage is set up by the first lookup, insert or store_batch, so + /// a context that never compiles never touches the disk or probes the compiler. explicit binary_cache(binary_cache_settings s = {}); - /// Look up a key, consulting memory first and then the cache directory. + /// Look up a key, consulting memory first and then the storage backend. optional get(const context& ctx, const std::string& key); /// Record a compiled result under its key. diff --git a/src/targets/gpu/include/migraphx/gpu/binary_cache_backend.hpp b/src/targets/gpu/include/migraphx/gpu/binary_cache_backend.hpp index 106cf627d91..2c9d166bde4 100644 --- a/src/targets/gpu/include/migraphx/gpu/binary_cache_backend.hpp +++ b/src/targets/gpu/include/migraphx/gpu/binary_cache_backend.hpp @@ -37,12 +37,11 @@ // Notes: // * binary_cache_entry is defined in ; // the include below pulls in its full definition. -// * Backends typically own non-trivial resources (a cache directory, a SQLite -// connection) and are not meaningfully copyable beyond shared ownership. -// * The members are non-const: binary_cache::get and insert are themselves -// non-const, so nothing forces a const qualifier here, and a backend that -// tracks an open batch needs the mutability. A backend may still declare -// them const. +// * Backends must be copyable: the wrapper shares T and clones it on a +// non-const call while the handle is shared. sqlite_binary_cache shares its +// connection across copies. +// * The members are non-const so a backend can track an open batch; a backend +// may still declare them const. // #ifndef MIGRAPHX_GUARD_GPU_BINARY_CACHE_BACKEND_HPP #define MIGRAPHX_GUARD_GPU_BINARY_CACHE_BACKEND_HPP @@ -70,13 +69,12 @@ namespace gpu { /// Type-erased interface for binary-cache storage backends. /// /// A backend persists serialized binary_cache_entry blobs to some medium (a -/// directory of files, a SQLite database, an in-memory map for tests). Entries -/// are addressed by three strings the caller has already computed: +/// directory of files or a SQLite database). Entries are addressed by three +/// strings the caller has already computed: /// /// * `version` -- binary_cache::version_id(), identifying the toolchain and /// the embedded kernel sources that produced the entry. Never empty; the -/// caller skips persistence entirely when it is. The short form is used -/// for directories and the full form for databases. +/// caller skips persistence entirely when it is. /// * `device` -- the GPU the entry was compiled for. /// * `key_hash` -- md5 of the compile key. A hash rather than the key itself /// because a file backend needs a short name; a collision is harmless, @@ -108,8 +106,9 @@ struct binary_cache_backend /// decided entirely by the key, so a writer that loses a race replaces the /// entry with equivalent bytes. /// - /// Must not throw. A failure to store costs a recompile next run, nothing - /// more, and the caller still keeps the result in memory. + /// May throw: the caller reports a failed store as a warning. It costs a + /// recompile next run, nothing more, and the caller still keeps the result + /// in memory. void store(const std::string& version, const std::string& device, const std::string& key_hash, diff --git a/src/targets/gpu/include/migraphx/gpu/binary_cache_entry.hpp b/src/targets/gpu/include/migraphx/gpu/binary_cache_entry.hpp index 9cbf68a59fa..13d54117264 100644 --- a/src/targets/gpu/include/migraphx/gpu/binary_cache_entry.hpp +++ b/src/targets/gpu/include/migraphx/gpu/binary_cache_entry.hpp @@ -37,11 +37,7 @@ namespace gpu { /// What gets stored for one compiled kernel. The op name, problem and solution are stored for /// offline inspection; only the key is checked when an entry is loaded. -/// -/// This lives in its own header, rather than nested inside binary_cache, so that the -/// binary_cache_backend interface can name it without including binary_cache.hpp -- which in -/// turn includes the backend header. Same reason cache_device_key.hpp exists for the problem -/// cache. binary_cache::entry remains an alias for it. +/// In its own header so binary_cache_backend.hpp can use it without a circular include. struct binary_cache_entry { std::string key = {}; diff --git a/src/targets/gpu/include/migraphx/gpu/file_binary_cache.hpp b/src/targets/gpu/include/migraphx/gpu/file_binary_cache.hpp index 10de1884176..ffc37708600 100644 --- a/src/targets/gpu/include/migraphx/gpu/file_binary_cache.hpp +++ b/src/targets/gpu/include/migraphx/gpu/file_binary_cache.hpp @@ -38,7 +38,7 @@ namespace gpu { // A binary_cache_backend that keeps entries as files under a root directory, laid out // ///.mxr. The version directory is named after the build that -// wrote it, so the tree is self-describing. Path resolution is the caller's job. +// wrote it, so the tree is self-describing. struct MIGRAPHX_GPU_EXPORT file_binary_cache { optional> diff --git a/src/targets/gpu/include/migraphx/gpu/sqlite_binary_cache.hpp b/src/targets/gpu/include/migraphx/gpu/sqlite_binary_cache.hpp index 0efd4e289e5..118e15c1e89 100644 --- a/src/targets/gpu/include/migraphx/gpu/sqlite_binary_cache.hpp +++ b/src/targets/gpu/include/migraphx/gpu/sqlite_binary_cache.hpp @@ -27,7 +27,6 @@ #include #include -#include #include #include #include @@ -46,14 +45,13 @@ namespace gpu { // target: migraphx/sqlite.hpp forward-declares both impl types and never includes sqlite3.h. struct MIGRAPHX_GPU_EXPORT sqlite_binary_cache { - /// Open the database, create the schema and prepare the statements, and return the wrapper - /// so the caller can hand the result straight back. + /// Open the database at path, creating the schema if it is writable. /// - /// A database that can only be read gives a backend that serves lookups and ignores stores; - /// it is used as it stands, without creating the schema. Returns nullopt when the database - /// cannot be opened at all or entries cannot be looked up in it, so an unusable database - /// leaves the cache memory-only rather than raising an error. - static optional open(const std::string& path); + /// A database that can only be read serves lookups and ignores stores; it is used as it + /// stands, without creating the schema. Returns nullopt when the database cannot be opened + /// at all or entries cannot be looked up in it, so an unusable database leaves the cache + /// memory-only rather than raising an error. + static optional open(const std::string& path); optional> load(const std::string& version, const std::string& device, const std::string& key_hash) const; @@ -69,12 +67,9 @@ struct MIGRAPHX_GPU_EXPORT sqlite_binary_cache void end_batch(); private: - sqlite db = {}; - sqlite_stmt get_stmt = {}; - sqlite_stmt store_stmt = {}; - sqlite_stmt begin_stmt = {}; - sqlite_stmt commit_stmt = {}; - sqlite_stmt rollback_stmt = {}; + sqlite db = {}; + sqlite_stmt get_stmt = {}; + sqlite_stmt store_stmt = {}; /// Whether begin_batch opened a transaction that end_batch still has to close. bool in_batch = false; }; diff --git a/src/targets/gpu/sqlite_binary_cache.cpp b/src/targets/gpu/sqlite_binary_cache.cpp index de0f1ae229b..2140d550013 100644 --- a/src/targets/gpu/sqlite_binary_cache.cpp +++ b/src/targets/gpu/sqlite_binary_cache.cpp @@ -23,6 +23,7 @@ * */ #include +#include #include #include #include @@ -34,9 +35,6 @@ namespace migraphx { inline namespace MIGRAPHX_INLINE_NS { namespace gpu { -// Compile-time confirmation that sqlite_binary_cache satisfies the backend concept. If a method -// signature drifts, this assertion fires at the definition site rather than at some far-away -// usage. static_assert(std::is_constructible{}, "sqlite_binary_cache must satisfy the binary_cache_backend concept"); @@ -77,8 +75,8 @@ constexpr const char* get_sql = // INSERT OR REPLACE is the analogue of the file backend's publish-by-rename: the content is // decided entirely by the key, so two processes compiling the same kernel is benign and the // last writer wins with equivalent bytes. The timestamp is computed by the database rather -// than the process so that rows written by different machines stay comparable; nothing reads -// it yet, it is there to make pruning an old cache by age possible. +// than the process so that rows written by different machines stay comparable. MIGraphX never +// reads it; it lets a cache be pruned by age. constexpr const char* store_sql = "INSERT OR REPLACE INTO cache_v1" " (version, device, key_hash, op_name, problem, solution, entry, timestamp)" @@ -94,7 +92,7 @@ constexpr const char* rollback_sql = "ROLLBACK;"; } // namespace -optional sqlite_binary_cache::open(const std::string& path) +optional sqlite_binary_cache::open(const std::string& path) { sqlite_binary_cache r; try @@ -127,11 +125,8 @@ optional sqlite_binary_cache::open(const std::string& path } else { - (void)r.db.execute(schema_sql); - r.store_stmt = r.db.prepare(store_sql); - r.begin_stmt = r.db.prepare(begin_sql); - r.commit_stmt = r.db.prepare(commit_sql); - r.rollback_stmt = r.db.prepare(rollback_sql); + r.db.execute(schema_sql); + r.store_stmt = r.db.prepare(store_sql); } // Without a working lookup there is no cache, so this failure disables the backend. That // includes a read-only database that was never given the schema. @@ -142,15 +137,13 @@ optional sqlite_binary_cache::open(const std::string& path log::warn() << "Disabling the binary cache at " << path << ": " << ex.what(); return nullopt; } - return binary_cache_backend{std::move(r)}; + return r; } optional> sqlite_binary_cache::load(const std::string& version, const std::string& device, const std::string& key_hash) const { - if(not get_stmt.valid()) - return nullopt; try { // The primary key makes this at most one row. @@ -176,34 +169,26 @@ void sqlite_binary_cache::store(const std::string& version, const binary_cache_entry& e, const std::vector& blob) const { + // Not prepared for a read-only database, whose stores are skipped. if(not store_stmt.valid()) return; - try - { - // The json strings are temporaries, which is safe because binding copies immediately. - // An insert returns no rows, and it has run by the time the call returns. - store_stmt(version, - device, - key_hash, - e.op_name, - to_json_string(e.problem), - to_json_string(e.solution), - blob); - } - catch(const std::exception& ex) - { - log::warn() << "Failed to write binary cache entry " << key_hash << ": " << ex.what(); - } + store_stmt(version, + device, + key_hash, + e.op_name, + to_json_string(e.problem), + to_json_string(e.solution), + blob); } void sqlite_binary_cache::begin_batch() { assert(not in_batch); - if(not begin_stmt.valid()) + if(not store_stmt.valid()) return; try { - begin_stmt(); + db.execute(begin_sql); in_batch = true; } catch(const std::exception& ex) @@ -220,7 +205,7 @@ void sqlite_binary_cache::end_batch() in_batch = false; try { - commit_stmt(); + db.execute(commit_sql); } catch(const std::exception& ex) { @@ -229,7 +214,7 @@ void sqlite_binary_cache::end_batch() log::warn() << "Failed to commit binary cache entries: " << ex.what(); try { - rollback_stmt(); + db.execute(rollback_sql); } catch(const std::exception& rex) { diff --git a/test/gpu/binary_cache.cpp b/test/gpu/binary_cache.cpp index 1967581d4d6..dcd083eefc8 100644 --- a/test/gpu/binary_cache.cpp +++ b/test/gpu/binary_cache.cpp @@ -30,11 +30,11 @@ #include #include #include +#include #include #include #include #include -#include #include #include #include @@ -52,7 +52,6 @@ #include #include #include -#include #include #include #include @@ -102,8 +101,7 @@ static migraphx::gpu::binary_cache::entry make_entry(const std::string& key) return e; } -// The storage backend is chosen by the extension of the cache path, so a case that has to hold -// for both is written once against a path and registered twice, once with each of these. +// The storage backend is chosen by the extension of the cache path. static std::string dir_path(const migraphx::tmp_dir& td) { return td.path.string(); } static std::string db_path(const migraphx::tmp_dir& td) { return (td.path / "cache.db").string(); } @@ -111,11 +109,12 @@ static std::string db_path(const migraphx::tmp_dir& td) { return (td.path / "cac static std::vector entry_files(const migraphx::fs::path& dir) { std::vector result; - for(const auto& file : migraphx::fs::recursive_directory_iterator(dir)) - { - if(file.path().extension() == ".mxr") - result.push_back(file.path()); - } + migraphx::transform_if( + migraphx::fs::recursive_directory_iterator{dir}, + migraphx::fs::recursive_directory_iterator{}, + std::back_inserter(result), + [](const auto& file) { return file.path().extension() == ".mxr"; }, + [](const auto& file) { return file.path(); }); return result; } @@ -129,22 +128,6 @@ static std::size_t row_count(const std::string& path, const std::string& table) return std::stoul(rows.front().at("n")); } -/// How many entries a cache path holds, whichever backend wrote them. The extensions must match -/// the ones make_binary_cache_backend routes to the SQLite backend, or this silently counts -/// files in a directory that does not exist and reports zero. -static std::size_t stored_entry_count(const std::string& path) -{ - if(migraphx::ends_with(path, ".db") or migraphx::ends_with(path, ".sqlite")) - return row_count(path, "cache_v1"); - return entry_files(path).size(); -} - -/// Every row a statement call produced. -static std::vector collect_rows(const migraphx::sqlite_stmt::rows& r) -{ - return std::vector(r.begin(), r.end()); -} - using stored_entries = std::map>; /// Every entry a cache directory holds, keyed by key hash, which names each file. @@ -163,7 +146,7 @@ static stored_entries db_entries(const std::string& path) { stored_entries result; auto select = migraphx::sqlite::read(path).prepare("SELECT key_hash, entry FROM cache_v1;"); - auto rows = collect_rows(select()); + auto rows = select(); std::transform(rows.begin(), rows.end(), std::inserter(result, result.end()), [](auto row) { const auto& blob = row.at("entry").get_binary(); return std::make_pair(row.at("key_hash").get_string(), @@ -172,6 +155,33 @@ static stored_entries db_entries(const std::string& path) return result; } +// What a case that must hold for both backends needs to know about each, so the case can be +// written once as a template and registered for both. +struct directory_backend +{ + static std::string path(const migraphx::tmp_dir& td) { return dir_path(td); } + static std::size_t stored(const std::string& p) { return entry_files(p).size(); } + /// Overwrite every stored entry with bytes that do not decode. + static void damage(const std::string& p) + { + auto files = entry_files(p); + std::for_each(files.begin(), files.end(), [](const auto& file) { + migraphx::write_buffer(file, std::vector(8, 0)); + }); + } +}; + +struct database_backend +{ + static std::string path(const migraphx::tmp_dir& td) { return db_path(td); } + static std::size_t stored(const std::string& p) { return row_count(p, "cache_v1"); } + /// Overwrite every stored entry with bytes that do not decode. + static void damage(const std::string& p) + { + migraphx::sqlite::write(p).execute("UPDATE cache_v1 SET entry = zeroblob(8);"); + } +}; + /// One of each backend over fresh storage in td: a directory at td/files and a database at /// db_path(td). Driven directly, these skip binary_cache and its version and device strings. static std::vector both_backends(const migraphx::tmp_dir& td) @@ -181,7 +191,7 @@ static std::vector both_backends(const migr auto db = migraphx::gpu::sqlite_binary_cache::open(db_path(td)); EXPECT(db.has_value()); if(db.has_value()) - result.push_back(*db); + result.emplace_back(std::move(*db)); return result; } @@ -211,16 +221,17 @@ TEST_CASE(memory_lookup_records_reuse) EXPECT(cache.get_stats().misses == 0); } -// The cases below are written once against a path and registered for both backends. The -// directory registrations use real temporary paths on purpose: entries sit under -// version_id()/device_dir(), and on Windows that depth once pushed the entry write past -// MAX_PATH, so running them there is what keeps it from coming back. +// The cases below are written once against a Backend and registered for each. The directory +// registrations use real temporary paths so that on Windows they exercise the full depth of an +// entry path against MAX_PATH. // A second cache shares nothing in memory, so anything it finds came out of storage. -static void disk_lookup_body(const std::string& path) +template +static void disk_lookup_records_a_hit() { + migraphx::tmp_dir td{"binary-cache"}; migraphx::gpu::context ctx; - migraphx::gpu::binary_cache_settings settings{path, false}; + migraphx::gpu::binary_cache_settings settings{Backend::path(td), false}; migraphx::gpu::binary_cache writer{settings}; writer.insert(ctx, make_entry("shared-key")); @@ -233,57 +244,30 @@ static void disk_lookup_body(const std::string& path) EXPECT(reader.get_stats().misses == 0); EXPECT(*found->fragment.get_main_module() == *make_code().fragment.get_main_module()); } +TEST_CASE_REGISTER(disk_lookup_records_a_hit); +TEST_CASE_REGISTER(disk_lookup_records_a_hit); -TEST_CASE(disk_lookup_records_a_hit) +// A damaged entry must cost a recompile and nothing more. +template +static void corrupt_entry_is_ignored() { migraphx::tmp_dir td{"binary-cache"}; - disk_lookup_body(dir_path(td)); -} - -TEST_CASE(sqlite_lookup_records_a_hit) -{ - migraphx::tmp_dir td{"binary-cache"}; - disk_lookup_body(db_path(td)); -} - -// A damaged entry must cost a recompile and nothing more. How an entry gets damaged is the only -// part of this that depends on the backend, so it comes in as a step. -static void corrupt_entry_body(const std::string& path, - const std::function& damage) -{ migraphx::gpu::context ctx; + auto path = Backend::path(td); migraphx::gpu::binary_cache_settings settings{path, false}; migraphx::gpu::binary_cache writer{settings}; writer.insert(ctx, make_entry("damaged")); - - damage(path); + EXPECT(Backend::stored(path) == 1); + Backend::damage(path); migraphx::gpu::binary_cache reader{settings}; EXPECT(not reader.get(ctx, "damaged").has_value()); EXPECT(reader.get_stats().misses == 1); } - -TEST_CASE(corrupt_entry_is_ignored) -{ - migraphx::tmp_dir td{"binary-cache"}; - corrupt_entry_body(dir_path(td), [](const std::string& dir) { - auto files = entry_files(dir); - EXPECT(files.size() == 1); - for(const auto& file : files) - migraphx::write_buffer(file, std::vector(8, 0)); - }); -} - -TEST_CASE(sqlite_corrupt_entry_is_ignored) -{ - migraphx::tmp_dir td{"binary-cache"}; - corrupt_entry_body(db_path(td), [](const std::string& db) { - EXPECT(row_count(db, "cache_v1") == 1); - migraphx::sqlite::write(db).execute("UPDATE cache_v1 SET entry = zeroblob(8);"); - }); -} +TEST_CASE_REGISTER(corrupt_entry_is_ignored); +TEST_CASE_REGISTER(corrupt_entry_is_ignored); // Without a directory nothing reaches disk, though results are still shared in memory. TEST_CASE(no_directory_writes_nothing) @@ -342,8 +326,11 @@ TEST_CASE(duplicate_kernels_compile_once_without_a_directory) // Compiling twice against the same cache has to leave entries behind and keep producing the // same numbers as the reference, whichever half of the run they came from. -static void compiling_twice_body(const std::string& path) +template +static void compiling_twice_populates_the_cache_and_matches_reference() { + migraphx::tmp_dir td{"binary-cache"}; + auto path = Backend::path(td); auto options = cache_options(path); auto p_ref = pointwise_program(); @@ -358,7 +345,7 @@ static void compiling_twice_body(const std::string& path) auto warmup = pointwise_program(); warmup.compile(migraphx::make_target("gpu"), options); - EXPECT(stored_entry_count(path) > 0); + EXPECT(Backend::stored(path) > 0); auto t = migraphx::make_target("gpu"); auto p = pointwise_program(); @@ -380,23 +367,16 @@ static void compiling_twice_body(const std::string& path) gpu_result.to_vector())); } -TEST_CASE(compiling_twice_populates_the_cache_and_matches_reference) -{ - migraphx::tmp_dir td{"binary-cache"}; - compiling_twice_body(dir_path(td)); -} - -TEST_CASE(sqlite_compiling_twice_populates_the_cache_and_matches_reference) -{ - migraphx::tmp_dir td{"binary-cache"}; - compiling_twice_body(db_path(td)); -} +TEST_CASE_REGISTER(compiling_twice_populates_the_cache_and_matches_reference); +TEST_CASE_REGISTER(compiling_twice_populates_the_cache_and_matches_reference); // With verification on, every reused result is compiled again and compared, so a run that does // not throw is one where the keys really do capture what the compilers depend on. -static void verified_reuse_body(const std::string& path) +template +static void verified_reuse_matches_fresh_compiles() { - auto options = cache_options(path, /* verify */ true); + migraphx::tmp_dir td{"binary-cache"}; + auto options = cache_options(Backend::path(td), /* verify */ true); auto warmup = pointwise_program(); warmup.compile(migraphx::make_target("gpu"), options); @@ -405,17 +385,8 @@ static void verified_reuse_body(const std::string& path) p.compile(migraphx::make_target("gpu"), options); } -TEST_CASE(verified_reuse_matches_fresh_compiles) -{ - migraphx::tmp_dir td{"binary-cache"}; - verified_reuse_body(dir_path(td)); -} - -TEST_CASE(sqlite_verified_reuse_matches_fresh_compiles) -{ - migraphx::tmp_dir td{"binary-cache"}; - verified_reuse_body(db_path(td)); -} +TEST_CASE_REGISTER(verified_reuse_matches_fresh_compiles); +TEST_CASE_REGISTER(verified_reuse_matches_fresh_compiles); // The extension of the path picks the backend and nothing else does, so the only way to see the // choice from outside is the artifact it leaves: a database file, or a directory tree. @@ -525,13 +496,9 @@ TEST_CASE(backends_store_identical_bytes) migraphx::gpu::binary_cache db_cache{migraphx::gpu::binary_cache_settings{path, false}}; db_cache.insert(ctx, e); - auto select = migraphx::sqlite::read(path).prepare("SELECT entry FROM cache_v1;"); - auto rows = select(); - auto it = rows.begin(); - EXPECT(it != rows.end()); - auto row = *it; - const auto& from_db = row.at("entry").get_binary(); - EXPECT((std::vector(from_db.begin(), from_db.end()) == from_file)); + auto from_db = db_entries(path); + EXPECT(from_db.size() == 1); + EXPECT((from_db.begin()->second == from_file)); } // A whole compile against each backend has to leave the same kernels behind: the same key @@ -599,7 +566,8 @@ TEST_CASE(entries_move_between_backends) auto select = migraphx::sqlite::read(db_path(db_td)) .prepare("SELECT version, device, key_hash, entry FROM cache_v1;"); - auto rows = collect_rows(select()); + auto result = select(); + auto rows = std::vector(result.begin(), result.end()); EXPECT(rows.size() == 1); const auto& row = rows.front(); EXPECT(row.at("version").get_string() == long_version); @@ -621,8 +589,7 @@ TEST_CASE(entries_move_between_backends) } // Two connections over one database, as two processes compiling against a shared cache would -// have. This cannot be two sqlite_binary_cache objects: only open() populates one, and it hands -// back the type-erased wrapper. +// have. TEST_CASE(two_connections_share_a_database) { migraphx::tmp_dir td{"binary-cache"}; @@ -775,20 +742,15 @@ TEST_CASE(file_store_leaves_only_entries_behind) migraphx::gpu::binary_cache second{settings}; second.insert(ctx, make_entry("one")); - std::vector files; - std::vector dirs; - for(const auto& item : migraphx::fs::recursive_directory_iterator(td.path)) - { - if(item.is_directory()) - dirs.push_back(item.path()); - else - files.push_back(item.path()); - } - EXPECT(files.size() == 2); - EXPECT(std::all_of( - files.begin(), files.end(), [](const auto& f) { return f.extension() == ".mxr"; })); - // Just the version directory and the device directory inside it. - EXPECT(dirs.size() == 2); + std::vector items{ + migraphx::fs::recursive_directory_iterator{td.path}, + migraphx::fs::recursive_directory_iterator{}}; + auto dirs = std::count_if( + items.begin(), items.end(), [](const auto& item) { return item.is_directory(); }); + // Just the version directory and the device directory inside it, and the two entries. + EXPECT(dirs == 2); + EXPECT(entry_files(td.path).size() == 2); + EXPECT(items.size() == 4); } // Storage is opened by the first lookup or insert, not by constructing the cache, since every @@ -807,16 +769,19 @@ TEST_CASE(storage_is_opened_on_first_use) // Inserts made inside a batch are all there once it ends. The database commits them together; // the directory backend has no batching and stores each one as it comes. -static void batched_inserts_body(const std::string& path) +template +static void batched_inserts_are_all_stored() { + migraphx::tmp_dir td{"binary-cache"}; migraphx::gpu::context ctx; + auto path = Backend::path(td); migraphx::gpu::binary_cache cache{migraphx::gpu::binary_cache_settings{path, false}}; { migraphx::gpu::binary_cache::store_batch batch{cache}; cache.insert(ctx, make_entry("first")); cache.insert(ctx, make_entry("second")); } - EXPECT(stored_entry_count(path) == 2); + EXPECT(Backend::stored(path) == 2); migraphx::gpu::binary_cache reader{migraphx::gpu::binary_cache_settings{path, false}}; EXPECT(reader.get(ctx, "first").has_value()); @@ -824,17 +789,8 @@ static void batched_inserts_body(const std::string& path) EXPECT(reader.get_stats().hits == 2); } -TEST_CASE(batched_inserts_are_all_stored) -{ - migraphx::tmp_dir td{"binary-cache"}; - batched_inserts_body(dir_path(td)); -} - -TEST_CASE(sqlite_batched_inserts_are_all_committed) -{ - migraphx::tmp_dir td{"binary-cache"}; - batched_inserts_body(db_path(td)); -} +TEST_CASE_REGISTER(batched_inserts_are_all_stored); +TEST_CASE_REGISTER(batched_inserts_are_all_stored); // A batch holds the database's write lock, so another connection must be able to write again as // soon as it ends. @@ -856,8 +812,7 @@ TEST_CASE(sqlite_batch_releases_the_database) } // The backend layer moves opaque bytes and never decodes them, so a payload that is not even -// msgpack still round-trips. Both backends go through the same type-erased wrapper here, which -// is the runtime half of the static_assert in each backend's .cpp. +// msgpack still round-trips. Both backends go through the same type-erased wrapper here. TEST_CASE(backends_round_trip_through_the_wrapper) { migraphx::tmp_dir td{"binary-cache"}; diff --git a/tools/include/gpu/binary_cache_backend.hpp b/tools/include/gpu/binary_cache_backend.hpp index cb31fb5d5c4..fa54e6b2843 100644 --- a/tools/include/gpu/binary_cache_backend.hpp +++ b/tools/include/gpu/binary_cache_backend.hpp @@ -37,12 +37,11 @@ // Notes: // * binary_cache_entry is defined in ; // the include below pulls in its full definition. -// * Backends typically own non-trivial resources (a cache directory, a SQLite -// connection) and are not meaningfully copyable beyond shared ownership. -// * The members are non-const: binary_cache::get and insert are themselves -// non-const, so nothing forces a const qualifier here, and a backend that -// tracks an open batch needs the mutability. A backend may still declare -// them const. +// * Backends must be copyable: the wrapper shares T and clones it on a +// non-const call while the handle is shared. sqlite_binary_cache shares its +// connection across copies. +// * The members are non-const so a backend can track an open batch; a backend +// may still declare them const. // #ifndef MIGRAPHX_GUARD_GPU_BINARY_CACHE_BACKEND_HPP #define MIGRAPHX_GUARD_GPU_BINARY_CACHE_BACKEND_HPP @@ -70,13 +69,12 @@ namespace gpu { /// Type-erased interface for binary-cache storage backends. /// /// A backend persists serialized binary_cache_entry blobs to some medium (a -/// directory of files, a SQLite database, an in-memory map for tests). Entries -/// are addressed by three strings the caller has already computed: +/// directory of files or a SQLite database). Entries are addressed by three +/// strings the caller has already computed: /// /// * `version` -- binary_cache::version_id(), identifying the toolchain and /// the embedded kernel sources that produced the entry. Never empty; the -/// caller skips persistence entirely when it is. The short form is used -/// for directories and the full form for databases. +/// caller skips persistence entirely when it is. /// * `device` -- the GPU the entry was compiled for. /// * `key_hash` -- md5 of the compile key. A hash rather than the key itself /// because a file backend needs a short name; a collision is harmless, @@ -109,8 +107,9 @@ struct binary_cache_backend /// decided entirely by the key, so a writer that loses a race replaces the /// entry with equivalent bytes. /// - /// Must not throw. A failure to store costs a recompile next run, nothing - /// more, and the caller still keeps the result in memory. + /// May throw: the caller reports a failed store as a warning. It costs a + /// recompile next run, nothing more, and the caller still keeps the result + /// in memory. void store(const std::string& version, const std::string& device, const std::string& key_hash, From e1d5f80c951dee41c18d39f5079479529259c30b Mon Sep 17 00:00:00 2001 From: pnikolic-amd Date: Fri, 25 Sep 2026 17:12:02 +0200 Subject: [PATCH 08/14] CI tidy fixes --- src/include/migraphx/sqlite.hpp | 28 +++++++++++-------- src/sqlite.cpp | 9 +++++- src/targets/gpu/compile_ops.cpp | 49 ++++++++++++++++++--------------- test/gpu/binary_cache.cpp | 7 +++-- 4 files changed, 55 insertions(+), 38 deletions(-) diff --git a/src/include/migraphx/sqlite.hpp b/src/include/migraphx/sqlite.hpp index 1a9dc6a5355..5efad771be0 100644 --- a/src/include/migraphx/sqlite.hpp +++ b/src/include/migraphx/sqlite.hpp @@ -60,19 +60,10 @@ struct MIGRAPHX_EXPORT sqlite_stmt sqlite_stmt() = default; - /// Run the statement with xs bound to its parameters in order, and return its rows. + /// Run the statement with xs bound to its parameters in order, and return its rows. Defined + /// after rows, which has to be complete for a function returning it to be defined. template - rows operator()(const Ts&... xs) const - { - if(not valid()) - MIGRAPHX_THROW("sqlite: calling a statement that was never prepared"); - assert(sizeof...(Ts) == parameter_count()); - // Anything left from the previous call, bindings or an unfinished result, goes first. - reset(); - int i = 0; - each_args([&](const auto& x) { bind(++i, x); }, xs...); - return rows{*this}; - } + rows operator()(const Ts&... xs) const; bool valid() const { return impl != nullptr; } @@ -162,6 +153,19 @@ struct sqlite_stmt::rows bool first = false; }; +template +sqlite_stmt::rows sqlite_stmt::operator()(const Ts&... xs) const +{ + if(not valid()) + MIGRAPHX_THROW("sqlite: calling a statement that was never prepared"); + assert(sizeof...(Ts) == parameter_count()); + // Anything left from the previous call, bindings or an unfinished result, goes first. + reset(); + int i = 0; + each_args([&](const auto& x) { bind(++i, x); }, xs...); + return rows{*this}; +} + struct MIGRAPHX_EXPORT sqlite { sqlite() = default; diff --git a/src/sqlite.cpp b/src/sqlite.cpp index fb15a844481..222462df3c0 100644 --- a/src/sqlite.cpp +++ b/src/sqlite.cpp @@ -217,7 +217,9 @@ void sqlite_stmt::reset() const noexcept (void)sqlite3_clear_bindings(impl->get()); } -/// Column i of the current row, keyed by its name. +/// Column i of the current row, keyed by its name. The values are built with parentheses +/// rather than the braces tidy suggests: value has an initializer_list constructor, which braces +/// would select, turning a keyed value into a two-element array. static value column_value(sqlite3_stmt* stmt, int i) { std::string name = sqlite3_column_name(stmt, i); @@ -225,6 +227,7 @@ static value column_value(sqlite3_stmt* stmt, int i) switch(type) { case SQLITE_INTEGER: return value(name, std::int64_t{sqlite3_column_int64(stmt, i)}); + // NOLINTNEXTLINE(modernize-return-braced-init-list) case SQLITE_FLOAT: return value(name, sqlite3_column_double(stmt, i)); case SQLITE_TEXT: case SQLITE_BLOB: { @@ -236,9 +239,12 @@ static value column_value(sqlite3_stmt* stmt, int i) assert(bytes >= 0); auto size = data == nullptr ? 0 : static_cast(bytes); if(type == SQLITE_TEXT) + // NOLINTNEXTLINE(modernize-return-braced-init-list) return value(name, size == 0 ? std::string{} : std::string(data, size)); + // NOLINTNEXTLINE(modernize-return-braced-init-list) return value(name, value::binary{data, size}); } + // NOLINTNEXTLINE(modernize-return-braced-init-list) default: return value(name, nullptr); } } @@ -253,6 +259,7 @@ value sqlite_stmt::to_value() const indices.end(), std::back_inserter(columns), [&](std::ptrdiff_t i) { return column_value(stmt, static_cast(i)); }); + // NOLINTNEXTLINE(modernize-return-braced-init-list) return value(columns, /* array_on_empty */ false); } diff --git a/src/targets/gpu/compile_ops.cpp b/src/targets/gpu/compile_ops.cpp index b2e20598c3e..d621b110391 100644 --- a/src/targets/gpu/compile_ops.cpp +++ b/src/targets/gpu/compile_ops.cpp @@ -733,6 +733,32 @@ struct compile_manager par_compile(cps.size(), [&](auto i) { cps[i].update_config(exhaustive); }); } + /// Store every compiled result in the binary cache. + static void + store_results(const std::vector>>& tasks) + { + if(tasks.empty()) + return; + // Every plan compiles with the same context, so the stores all go to one cache, and + // batching them lets its storage commit them together rather than one at a time. + auto* ctx = tasks.front().first->ctx; + assert(std::all_of( + tasks.begin(), tasks.end(), [&](const auto& task) { return task.first->ctx == ctx; })); + binary_cache::store_batch batch{ctx->get_binary_cache()}; + for(const auto& [cp, cell] : tasks) + { + if(not cell->result.has_value()) + continue; + // When verifying, reused results are stored again, rewriting the same bytes + // harmlessly. + cp->store(cell->solution, cell->key, cell->result->code); + assert(not cell->result->code.empty()); + // Only the serializable code is used from here on; dropping the replace function + // releases what its closure holds and keeps it off other instructions. + cell->result->replace_fn = nullptr; + } + } + void compile(module& m, bool is_root) { for(auto& cp : cps) @@ -804,28 +830,7 @@ struct compile_manager cell->result = cp->run_compile(cell->solution); }); - if(not tasks.empty()) - { - // Every plan compiles with the same context, so the stores all go to one cache, and - // batching them lets its storage commit them together rather than one at a time. - auto* ctx = tasks.front().first->ctx; - assert(std::all_of(tasks.begin(), tasks.end(), [&](const auto& task) { - return task.first->ctx == ctx; - })); - binary_cache::store_batch batch{ctx->get_binary_cache()}; - for(const auto& [cp, cell] : tasks) - { - if(not cell->result.has_value()) - continue; - // When verifying, reused results are stored again, rewriting the same bytes - // harmlessly. - cp->store(cell->solution, cell->key, cell->result->code); - assert(not cell->result->code.empty()); - // Only the serializable code is used from here on; dropping the replace function - // releases what its closure holds and keeps it off other instructions. - cell->result->replace_fn = nullptr; - } - } + store_results(tasks); static const auto mxr_path = string_value_of(MIGRAPHX_GPU_DUMP_BENCHMARK_MXR{}); bool dump_mxr = not mxr_path.empty(); diff --git a/test/gpu/binary_cache.cpp b/test/gpu/binary_cache.cpp index dcd083eefc8..bd9aa6f71c1 100644 --- a/test/gpu/binary_cache.cpp +++ b/test/gpu/binary_cache.cpp @@ -135,9 +135,10 @@ static stored_entries dir_entries(const migraphx::fs::path& dir) { stored_entries result; auto files = entry_files(dir); - std::transform(files.begin(), files.end(), std::inserter(result, result.end()), [](auto f) { - return std::make_pair(f.stem().string(), migraphx::read_buffer(f)); - }); + std::transform( + files.begin(), files.end(), std::inserter(result, result.end()), [](const auto& f) { + return std::make_pair(f.stem().string(), migraphx::read_buffer(f)); + }); return result; } From 58f53b0e00bce0792a35de90d4fa098ca5d170f1 Mon Sep 17 00:00:00 2001 From: pnikolic-amd Date: Wed, 30 Sep 2026 16:29:32 +0200 Subject: [PATCH 09/14] Update tools/include/gpu/binary_cache_backend.hpp Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> --- tools/include/gpu/binary_cache_backend.hpp | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/tools/include/gpu/binary_cache_backend.hpp b/tools/include/gpu/binary_cache_backend.hpp index fa54e6b2843..3fb2eea0466 100644 --- a/tools/include/gpu/binary_cache_backend.hpp +++ b/tools/include/gpu/binary_cache_backend.hpp @@ -91,9 +91,8 @@ struct binary_cache_backend /// error, it is a cache miss, and the caller recompiles. /// /// Must not throw. - optional> load(const std::string& version, - const std::string& device, - const std::string& key_hash); + optional> + load(const std::string& version, const std::string& device, const std::string& key_hash); /// Persist `blob`, the msgpack encoding of `e`, under this key. /// From bd7f4a68d30c3dcdc6ab892151fe7148a2e6fa4a Mon Sep 17 00:00:00 2001 From: pnikolic-amd Date: Wed, 30 Sep 2026 16:29:53 +0200 Subject: [PATCH 10/14] Update tools/include/gpu/binary_cache_backend.hpp Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> --- tools/include/gpu/binary_cache_backend.hpp | 31 +++++++++++----------- 1 file changed, 15 insertions(+), 16 deletions(-) diff --git a/tools/include/gpu/binary_cache_backend.hpp b/tools/include/gpu/binary_cache_backend.hpp index 3fb2eea0466..5a2c7008372 100644 --- a/tools/include/gpu/binary_cache_backend.hpp +++ b/tools/include/gpu/binary_cache_backend.hpp @@ -134,22 +134,21 @@ struct binary_cache_backend #else <% - interface( - 'binary_cache_backend', - virtual('load', - returns = 'optional>', - version = 'const std::string&', - device = 'const std::string&', - key_hash = 'const std::string&'), - virtual('store', - returns = 'void', - version = 'const std::string&', - device = 'const std::string&', - key_hash = 'const std::string&', - e = 'const binary_cache_entry&', - blob = 'const std::vector&'), - virtual('begin_batch', returns = 'void', default = 'migraphx::nop'), - virtual('end_batch', returns = 'void', default = 'migraphx::nop')) + interface('binary_cache_backend', + virtual('load', + returns = 'optional>', + version = 'const std::string&', + device = 'const std::string&', + key_hash = 'const std::string&'), + virtual('store', + returns = 'void', + version = 'const std::string&', + device = 'const std::string&', + key_hash = 'const std::string&', + e = 'const binary_cache_entry&', + blob = 'const std::vector&'), + virtual('begin_batch', returns = 'void', default = 'migraphx::nop'), + virtual('end_batch', returns = 'void', default = 'migraphx::nop')) %> #endif From dd2404899318fec29b02e7beefaa90ea32749875 Mon Sep 17 00:00:00 2001 From: pnikolic-amd Date: Thu, 1 Oct 2026 15:30:28 +0200 Subject: [PATCH 11/14] - Backend load/store now take the key and return or accept --- src/include/migraphx/sqlite.hpp | 170 +++++++------ src/include/migraphx/value.hpp | 10 + src/sqlite.cpp | 35 +-- src/targets/gpu/binary_cache.cpp | 75 ++---- src/targets/gpu/compile_ops.cpp | 35 +-- src/targets/gpu/file_binary_cache.cpp | 51 ++-- .../gpu/include/migraphx/gpu/binary_cache.hpp | 26 +- .../migraphx/gpu/binary_cache_backend.hpp | 172 +++---------- .../migraphx/gpu/file_binary_cache.hpp | 14 +- .../migraphx/gpu/sqlite_binary_cache.hpp | 23 +- src/targets/gpu/sqlite_binary_cache.cpp | 114 ++++----- test/gpu/binary_cache.cpp | 229 ++++++++++-------- test/sqlite.cpp | 2 +- tools/include/gpu/binary_cache_backend.hpp | 86 +++---- 14 files changed, 449 insertions(+), 593 deletions(-) diff --git a/src/include/migraphx/sqlite.hpp b/src/include/migraphx/sqlite.hpp index 5efad771be0..8f6fc235436 100644 --- a/src/include/migraphx/sqlite.hpp +++ b/src/include/migraphx/sqlite.hpp @@ -56,14 +56,100 @@ struct sqlite_stmt_impl; /// Not thread safe: use a statement from one thread at a time. struct MIGRAPHX_EXPORT sqlite_stmt { - struct rows; + /// The rows produced by one call of a statement, as an input range of values. + /// + /// The first row is fetched when the call is made, so a statement that returns nothing, such + /// as an insert, has already run by the time the call returns, whether or not the range is + /// iterated. The statement is reset when the range is destroyed: an unfinished select holds + /// a read lock on the database until then, which would stall writers in other processes. + /// + /// Refers to the statement it came from, which must outlive it. + struct rows + { + // Only ever a prvalue returned from a call, so it never needs copying or moving, and a + // copy would reset the statement out from under the original. + rows(const rows&) = delete; + rows(rows&&) = delete; + rows& operator=(const rows&) = delete; + rows& operator=(rows&&) = delete; + ~rows() { stmt->reset(); } + + struct iterator : iterator_operators + { + using value_type = value; + using reference = value_type; + using difference_type = std::ptrdiff_t; + using iterator_category = std::input_iterator_tag; + using pointer = value*; + + iterator() = default; + + iterator(const rows* pparent, bool pavailable) : parent(pparent), available(pavailable) + { + } + + reference operator*() const + { + assert(parent != nullptr and available); + return parent->stmt->to_value(); + } + + static void increment(iterator& x) + { + assert(x.parent != nullptr and x.available); + x.available = x.parent->stmt->step(); + } + + static bool equal(const iterator& x, const iterator& y) + { + return x.parent == y.parent and x.available == y.available; + } + + private: + const rows* parent = nullptr; + bool available = false; + }; + + iterator begin() const { return {this, first}; } + iterator end() const { return {this, false}; } + + private: + friend struct sqlite_stmt; + + explicit rows(const sqlite_stmt& s) : stmt(&s) + { + // The destructor does not run when the constructor throws, so a failed first step + // resets the statement here instead. + try + { + first = stmt->step(); + } + catch(...) + { + stmt->reset(); + throw; + } + } + + const sqlite_stmt* stmt = nullptr; + bool first = false; + }; sqlite_stmt() = default; - /// Run the statement with xs bound to its parameters in order, and return its rows. Defined - /// after rows, which has to be complete for a function returning it to be defined. + /// Run the statement with xs bound to its parameters in order, and return its rows. template - rows operator()(const Ts&... xs) const; + rows operator()(const Ts&... xs) const + { + if(not valid()) + MIGRAPHX_THROW("sqlite: calling a statement that was never prepared"); + assert(sizeof...(Ts) == parameter_count()); + // Anything left from the previous call, bindings or an unfinished result, goes first. + reset(); + int i = 0; + each_args([&](const auto& x) { bind(++i, x); }, xs...); + return rows{*this}; + } bool valid() const { return impl != nullptr; } @@ -90,82 +176,6 @@ struct MIGRAPHX_EXPORT sqlite_stmt std::shared_ptr impl; }; -/// The rows produced by one call of a statement, as an input range of values. -/// -/// The first row is fetched when the call is made, so a statement that returns nothing, such as -/// an insert, has already run by the time the call returns, whether or not the range is -/// iterated. The statement is reset when the range is destroyed: an unfinished select holds a -/// read lock on the database until then, which would stall writers in other processes. -struct sqlite_stmt::rows -{ - // Only ever a prvalue returned from a call, so it never needs copying or moving, and a copy - // would reset the statement out from under the original. - rows(const rows&) = delete; - rows(rows&&) = delete; - rows& operator=(const rows&) = delete; - rows& operator=(rows&&) = delete; - ~rows() { stmt.reset(); } - - struct iterator : iterator_operators - { - using value_type = value; - using reference = value_type; - using difference_type = std::ptrdiff_t; - using iterator_category = std::input_iterator_tag; - using pointer = value*; - - iterator() = default; - - iterator(const rows* pparent, bool pavailable) : parent(pparent), available(pavailable) {} - - reference operator*() const - { - assert(parent != nullptr and available); - return parent->stmt.to_value(); - } - - template - static void increment(U& x) - { - assert(x.parent != nullptr and x.available); - x.available = x.parent->stmt.step(); - } - - template - static auto equal(const U& x, const V& y) - { - return x.parent == y.parent and x.available == y.available; - } - - private: - const rows* parent = nullptr; - bool available = false; - }; - - iterator begin() const { return {this, first}; } - iterator end() const { return {this, false}; } - - private: - friend struct sqlite_stmt; - explicit rows(sqlite_stmt s) : stmt(std::move(s)), first(stmt.step()) {} - - sqlite_stmt stmt; - bool first = false; -}; - -template -sqlite_stmt::rows sqlite_stmt::operator()(const Ts&... xs) const -{ - if(not valid()) - MIGRAPHX_THROW("sqlite: calling a statement that was never prepared"); - assert(sizeof...(Ts) == parameter_count()); - // Anything left from the previous call, bindings or an unfinished result, goes first. - reset(); - int i = 0; - each_args([&](const auto& x) { bind(++i, x); }, xs...); - return rows{*this}; -} - struct MIGRAPHX_EXPORT sqlite { sqlite() = default; diff --git a/src/include/migraphx/value.hpp b/src/include/migraphx/value.hpp index 70ce5613fde..f81e694654d 100644 --- a/src/include/migraphx/value.hpp +++ b/src/include/migraphx/value.hpp @@ -297,6 +297,16 @@ struct MIGRAPHX_EXPORT value value(const std::pair& p) : value(p.first, p.second) { } + + /// A keyed value. Braces would select the initializer_list constructor and make a two-element + /// array instead, so returning this spares callers from suppressing the tidy check that + /// suggests them. + template + static value pair(const std::string& pkey, Ts&&... xs) + { + // NOLINTNEXTLINE(modernize-return-braced-init-list) + return value(pkey, static_cast(xs)...); + } template {})> value& operator=(T rhs) { diff --git a/src/sqlite.cpp b/src/sqlite.cpp index 222462df3c0..105050b0d38 100644 --- a/src/sqlite.cpp +++ b/src/sqlite.cpp @@ -217,18 +217,15 @@ void sqlite_stmt::reset() const noexcept (void)sqlite3_clear_bindings(impl->get()); } -/// Column i of the current row, keyed by its name. The values are built with parentheses -/// rather than the braces tidy suggests: value has an initializer_list constructor, which braces -/// would select, turning a keyed value into a two-element array. +/// Column i of the current row, keyed by its name. static value column_value(sqlite3_stmt* stmt, int i) { std::string name = sqlite3_column_name(stmt, i); auto type = sqlite3_column_type(stmt, i); switch(type) { - case SQLITE_INTEGER: return value(name, std::int64_t{sqlite3_column_int64(stmt, i)}); - // NOLINTNEXTLINE(modernize-return-braced-init-list) - case SQLITE_FLOAT: return value(name, sqlite3_column_double(stmt, i)); + case SQLITE_INTEGER: return value::pair(name, std::int64_t{sqlite3_column_int64(stmt, i)}); + case SQLITE_FLOAT: return value::pair(name, sqlite3_column_double(stmt, i)); case SQLITE_TEXT: case SQLITE_BLOB: { // The data must be fetched before sqlite3_column_bytes: the other order can force a @@ -237,30 +234,22 @@ static value column_value(sqlite3_stmt* stmt, int i) const auto* data = static_cast(sqlite3_column_blob(stmt, i)); auto bytes = sqlite3_column_bytes(stmt, i); assert(bytes >= 0); - auto size = data == nullptr ? 0 : static_cast(bytes); + std::size_t size = data == nullptr ? 0 : bytes; if(type == SQLITE_TEXT) - // NOLINTNEXTLINE(modernize-return-braced-init-list) - return value(name, size == 0 ? std::string{} : std::string(data, size)); - // NOLINTNEXTLINE(modernize-return-braced-init-list) - return value(name, value::binary{data, size}); + return value::pair(name, size == 0 ? std::string{} : std::string(data, size)); + return value::pair(name, value::binary{data, size}); } - // NOLINTNEXTLINE(modernize-return-braced-init-list) - default: return value(name, nullptr); + default: return value::pair(name, nullptr); } } value sqlite_stmt::to_value() const { - auto* stmt = impl->get(); - // Built as keyed values rather than a map, so a blob is moved into place instead of copied. - std::vector columns; - auto indices = range(sqlite3_column_count(stmt)); - std::transform(indices.begin(), - indices.end(), - std::back_inserter(columns), - [&](std::ptrdiff_t i) { return column_value(stmt, static_cast(i)); }); - // NOLINTNEXTLINE(modernize-return-braced-init-list) - return value(columns, /* array_on_empty */ false); + auto* stmt = impl->get(); + value columns = value::object{}; + for(auto i : range(sqlite3_column_count(stmt))) + columns.insert(column_value(stmt, static_cast(i))); + return columns; } } // namespace MIGRAPHX_INLINE_NS diff --git a/src/targets/gpu/binary_cache.cpp b/src/targets/gpu/binary_cache.cpp index 8913a9f5992..35a660a2be6 100644 --- a/src/targets/gpu/binary_cache.cpp +++ b/src/targets/gpu/binary_cache.cpp @@ -28,10 +28,9 @@ #include #include #include -#include -#include #include #include +#include #include namespace migraphx { @@ -105,31 +104,6 @@ static std::string device_dir(const context& ctx) "_wf" + std::to_string(device.get_wavefront_size()); } -/// Turn a stored blob back into an entry. Any failure is a miss, so a damaged entry costs a -/// recompile. -static optional -decode_entry(const std::vector& blob, const std::string& key, const std::string& key_hash) -{ - binary_cache::entry e; - try - { - migraphx::from_value(from_msgpack(blob), e); - } - catch(const std::exception& ex) - { - log::warn() << "Ignoring unreadable binary cache entry " << key_hash << ": " << ex.what(); - return nullopt; - } - // Entries are addressed by a hash of the key, so the full key is checked here to make a - // collision a miss rather than a wrong kernel. - if(e.key != key) - { - log::warn() << "Ignoring binary cache entry with mismatched key: " << key_hash; - return nullopt; - } - return e; -} - binary_cache::binary_cache(binary_cache_settings s) : settings(std::move(s)) {} // The storage backend is selected by file type, the same rule make_problem_cache_backend applies @@ -157,18 +131,6 @@ binary_cache_backend* binary_cache::get_backend() return backend.has_value() ? &*backend : nullptr; } -binary_cache::store_batch::store_batch(binary_cache& c) : backend(c.get_backend()) -{ - if(backend != nullptr) - backend->begin_batch(); -} - -binary_cache::store_batch::~store_batch() -{ - if(backend != nullptr) - backend->end_batch(); -} - optional binary_cache::get(const context& ctx, const std::string& key) { if(key.empty()) @@ -181,44 +143,39 @@ optional binary_cache::get(const context& ctx, const std::string& } if(auto* b = get_backend()) { - // The key is the whole compile source, so it is hashed once for the lookup and any - // diagnostics. - auto key_hash = md5(key); - auto blob = b->load(version, device_dir(ctx), key_hash); - if(blob.has_value()) + auto e = b->load(version, device_dir(ctx), key); + if(e.has_value()) { - auto e = decode_entry(*blob, key, key_hash); - if(e.has_value()) - { - counters.hits++; - return memo.emplace(key, std::move(e->code)).first->second; - } + counters.hits++; + return memo.emplace(key, std::move(e->code)).first->second; } } counters.misses++; return nullopt; } -void binary_cache::insert(const context& ctx, entry e) +void binary_cache::insert(const context& ctx, std::vector es) { - if(e.key.empty()) + es.erase(std::remove_if(es.begin(), es.end(), [](const entry& e) { return e.key.empty(); }), + es.end()); + if(es.empty()) return; - counters.compiled++; + counters.compiled += es.size(); if(auto* b = get_backend()) { - auto key_hash = md5(e.key); try { - // A failure to serialize or store is a warning, not a failed compile. - auto blob = to_msgpack(migraphx::to_value(e)); - b->store(version, device_dir(ctx), key_hash, e, blob); + // A failure to store is a warning, not a failed compile. + b->store(version, device_dir(ctx), es); } catch(const std::exception& ex) { - log::warn() << "Failed to store binary cache entry " << key_hash << ": " << ex.what(); + log::warn() << "Failed to store " << es.size() + << " binary cache entries: " << ex.what(); } } - memo[std::move(e.key)] = std::move(e.code); + for(auto& e : es) + memo[std::move(e.key)] = std::move(e.code); } } // namespace gpu diff --git a/src/targets/gpu/compile_ops.cpp b/src/targets/gpu/compile_ops.cpp index f8977d16b0c..46d1f512bc6 100644 --- a/src/targets/gpu/compile_ops.cpp +++ b/src/targets/gpu/compile_ops.cpp @@ -217,23 +217,22 @@ static optional cache_lookup(context& ctx, const std::string& return cr; } -/// Record a freshly compiled result, under the same restriction as cache_lookup. -static void cache_store(context& ctx, - const operation& preop, - const value& solution, - const std::string& key, - const value& problem, - const compiled_code& code) +/// What to record for a freshly compiled result, or nullopt when its key must not be cached. +static optional make_cache_entry(const operation& preop, + const value& solution, + const std::string& key, + const value& problem, + const compiled_code& code) { if(is_private_key(key)) - return; + return nullopt; binary_cache::entry e; e.key = key; e.op_name = preop.name(); e.problem = problem; e.solution = solution; e.code = code; - ctx.get_binary_cache().insert(ctx, std::move(e)); + return e; } /// Reuse an earlier result for this key, or compile and record one. For callers with a single @@ -252,7 +251,8 @@ static compiler_replace compile_cached(context& ctx, return *cached; } auto cr = compile_fragment(ctx, ins, preop, solution); - cache_store(ctx, preop, solution, key, problem, cr.code); + if(auto e = make_cache_entry(preop, solution, key, problem, cr.code)) + ctx.get_binary_cache().insert(ctx, {std::move(*e)}); return cr; } @@ -448,9 +448,10 @@ struct compile_plan return cache_lookup(*ctx, key); } - void store(const value& solution, const std::string& key, const compiled_code& code) const + optional + cache_entry(const value& solution, const std::string& key, const compiled_code& code) const { - cache_store(*ctx, preop, solution, key, config ? config->problem : value{}, code); + return make_cache_entry(preop, solution, key, config ? config->problem : value{}, code); } /// True when the cache was configured to check reused results against a fresh compile. @@ -789,24 +790,26 @@ struct compile_manager { if(tasks.empty()) return; - // Every plan compiles with the same context, so the stores all go to one cache, and - // batching them lets its storage commit them together rather than one at a time. + // Every plan compiles with the same context, so the entries all go to one cache in a + // single insert, letting its storage commit them together rather than one at a time. auto* ctx = tasks.front().first->ctx; assert(std::all_of( tasks.begin(), tasks.end(), [&](const auto& task) { return task.first->ctx == ctx; })); - binary_cache::store_batch batch{ctx->get_binary_cache()}; + std::vector entries; for(const auto& [cp, cell] : tasks) { if(not cell->result.has_value()) continue; // When verifying, reused results are stored again, rewriting the same bytes // harmlessly. - cp->store(cell->solution, cell->key, cell->result->code); + if(auto e = cp->cache_entry(cell->solution, cell->key, cell->result->code)) + entries.push_back(std::move(*e)); assert(not cell->result->code.empty()); // Only the serializable code is used from here on; dropping the replace function // releases what its closure holds and keeps it off other instructions. cell->result->replace_fn = nullptr; } + ctx->get_binary_cache().insert(*ctx, std::move(entries)); } /// Fill every cell's result, from the cache or by compiling, sharing one compile among diff --git a/src/targets/gpu/file_binary_cache.cpp b/src/targets/gpu/file_binary_cache.cpp index 2ccd6f0d32b..585f6a107df 100644 --- a/src/targets/gpu/file_binary_cache.cpp +++ b/src/targets/gpu/file_binary_cache.cpp @@ -27,6 +27,8 @@ #include #include #include +#include +#include #include #include #include @@ -38,14 +40,15 @@ namespace gpu { static_assert(std::is_constructible{}, "file_binary_cache must satisfy the binary_cache_backend concept"); -/// Where an entry lives. The caller guarantees a non-empty version, so entries compiled by -/// different toolchains can never land on the same path. +/// Where an entry lives. The key is the whole compile source, so it is hashed to keep the name +/// short. The caller guarantees a non-empty version, so entries compiled by different toolchains +/// can never land on the same path. static fs::path entry_path(const fs::path& root, const std::string& version, const std::string& device, - const std::string& key_hash) + const std::string& key) { - return root / version / device / (key_hash + ".mxr"); + return root / version / device / (md5(key) + ".mxr"); } /// Publish by rename so a reader never sees a half-written file. The temporary stays beside @@ -69,36 +72,46 @@ static void write_atomically(const fs::path& dest, const std::vector& cont } } -optional> file_binary_cache::load(const std::string& version, - const std::string& device, - const std::string& key_hash) const +optional file_binary_cache::load(const std::string& version, + const std::string& device, + const std::string& key) const { - auto path = entry_path(root, version, device, key_hash); + auto path = entry_path(root, version, device, key); + binary_cache_entry e; try { if(not fs::exists(path)) return nullopt; - return read_buffer(path); + migraphx::from_value(from_msgpack(read_buffer(path)), e); } catch(const std::exception& ex) { - // An unreadable entry is a miss, which costs a recompile and nothing else. - log::warn() << "Failed to read binary cache entry " << path << ": " << ex.what(); + // An unreadable or damaged entry is a miss, which costs a recompile and nothing else. + log::warn() << "Ignoring unreadable binary cache entry " << path << ": " << ex.what(); return nullopt; } + // Files are named by a hash of the key, so the full key is checked here to make a collision + // a miss rather than a wrong kernel. + if(e.key != key) + { + log::warn() << "Ignoring binary cache entry with mismatched key: " << path; + return nullopt; + } + return e; } void file_binary_cache::store(const std::string& version, const std::string& device, - const std::string& key_hash, - const binary_cache_entry&, - const std::vector& blob) const + const std::vector& entries) const { - auto path = entry_path(root, version, device, key_hash); - // The content is decided entirely by the key, so a writer that loses the publish race - // replaces the file with the same bytes and no locking is needed. - fs::create_directories(path.parent_path()); - write_atomically(path, blob); + for(const auto& e : entries) + { + auto path = entry_path(root, version, device, e.key); + // The content is decided entirely by the key, so a writer that loses the publish race + // replaces the file with the same bytes and no locking is needed. + fs::create_directories(path.parent_path()); + write_atomically(path, to_msgpack(migraphx::to_value(e))); + } } } // namespace gpu diff --git a/src/targets/gpu/include/migraphx/gpu/binary_cache.hpp b/src/targets/gpu/include/migraphx/gpu/binary_cache.hpp index f136b1b5dc3..9eac0dd1d01 100644 --- a/src/targets/gpu/include/migraphx/gpu/binary_cache.hpp +++ b/src/targets/gpu/include/migraphx/gpu/binary_cache.hpp @@ -33,6 +33,7 @@ #include #include #include +#include namespace migraphx { inline namespace MIGRAPHX_INLINE_NS { @@ -82,31 +83,16 @@ struct MIGRAPHX_GPU_EXPORT binary_cache std::size_t compiled = 0; }; - /// Groups the inserts made while it lives, so the storage backend can commit them together - /// rather than one at a time. Scope it tightly around a run of inserts: a database holds a - /// write lock against other processes until it ends. - struct MIGRAPHX_GPU_EXPORT store_batch - { - explicit store_batch(binary_cache& c); - store_batch(const store_batch&) = delete; - store_batch(store_batch&&) = delete; - store_batch& operator=(const store_batch&) = delete; - store_batch& operator=(store_batch&&) = delete; - ~store_batch(); - - private: - binary_cache_backend* backend; - }; - - /// Nothing is opened here; storage is set up by the first lookup, insert or store_batch, so - /// a context that never compiles never touches the disk or probes the compiler. + /// Nothing is opened here; storage is set up by the first lookup or insert, so a context that + /// never compiles never touches the disk or probes the compiler. explicit binary_cache(binary_cache_settings s = {}); /// Look up a key, consulting memory first and then the storage backend. optional get(const context& ctx, const std::string& key); - /// Record a compiled result under its key. - void insert(const context& ctx, entry e); + /// Record compiled results under their keys. They are handed to the storage backend + /// together, so it can commit them at once rather than one at a time. + void insert(const context& ctx, std::vector es); /// True when reused results should be checked against a fresh compile. bool verify() const { return settings.verify; } diff --git a/src/targets/gpu/include/migraphx/gpu/binary_cache_backend.hpp b/src/targets/gpu/include/migraphx/gpu/binary_cache_backend.hpp index 2c9d166bde4..0b91a7c14cd 100644 --- a/src/targets/gpu/include/migraphx/gpu/binary_cache_backend.hpp +++ b/src/targets/gpu/include/migraphx/gpu/binary_cache_backend.hpp @@ -30,8 +30,7 @@ // into the gpu target tree). Do not edit the generated header by hand. // // Any type T satisfies the binary_cache_backend concept if it provides the -// member functions listed below; begin_batch and end_batch are optional and -// default to doing nothing. The wrapper holds T by shared_ptr and forwards +// member functions listed below. The wrapper holds T by shared_ptr and forwards // each call through a virtual dispatch, matching problem_cache_backend. // // Notes: @@ -40,8 +39,6 @@ // * Backends must be copyable: the wrapper shares T and clones it on a // non-const call while the handle is shared. sqlite_binary_cache shares its // connection across copies. -// * The members are non-const so a backend can track an open batch; a backend -// may still declare them const. // #ifndef MIGRAPHX_GUARD_GPU_BINARY_CACHE_BACKEND_HPP #define MIGRAPHX_GUARD_GPU_BINARY_CACHE_BACKEND_HPP @@ -68,67 +65,46 @@ namespace gpu { /// Type-erased interface for binary-cache storage backends. /// -/// A backend persists serialized binary_cache_entry blobs to some medium (a -/// directory of files or a SQLite database). Entries are addressed by three -/// strings the caller has already computed: +/// A backend persists binary_cache_entry values to some medium (a directory of +/// files or a SQLite database), and decides for itself how to serialize them. +/// Entries are addressed by their key, scoped by two strings the caller has +/// already computed: /// /// * `version` -- binary_cache::version_id(), identifying the toolchain and /// the embedded kernel sources that produced the entry. Never empty; the /// caller skips persistence entirely when it is. /// * `device` -- the GPU the entry was compiled for. -/// * `key_hash` -- md5 of the compile key. A hash rather than the key itself -/// because a file backend needs a short name; a collision is harmless, -/// since the caller re-checks the full key against the decoded entry. /// -/// Together these three form the identity of an entry. A backend must keep -/// entries with different scopes distinct rather than overwriting across them. +/// A backend must keep entries with different scopes distinct rather than +/// overwriting across them. It may address entries by a hash of the key, for +/// instance to keep file names short, but must then check the full key when +/// loading so that a collision is a miss rather than a wrong kernel. struct binary_cache_backend { - /// Return the serialized entry for this key, or nullopt for a miss. + /// Return the entry stored for this key, or nullopt for a miss. /// /// nullopt also covers every failure: a missing file, an unreadable - /// database, a permissions problem. A cache that cannot be read is not an - /// error, it is a cache miss, and the caller recompiles. + /// database, a damaged entry, a permissions problem. A cache that cannot be + /// read is not an error, it is a cache miss, and the caller recompiles. /// /// Must not throw. - optional> - load(const std::string& version, const std::string& device, const std::string& key_hash); + optional + load(const std::string& version, const std::string& device, const std::string& key); - /// Persist `blob`, the msgpack encoding of `e`, under this key. - /// - /// `e` is passed alongside `blob` so a backend may denormalize op_name, - /// problem and solution into queryable columns. Those fields are also - /// inside `blob`, which stays the authoritative record -- a backend that - /// stores them separately must still be able to answer a load() with the - /// blob alone. + /// Persist every entry in `entries` under its key. The entries arrive + /// together so a backend can commit them at once, such as in one database + /// transaction, rather than one at a time. /// /// Overwriting an existing entry is expected and safe: the content is /// decided entirely by the key, so a writer that loses a race replaces the - /// entry with equivalent bytes. + /// entry with an equivalent one. /// /// May throw: the caller reports a failed store as a warning. It costs a - /// recompile next run, nothing more, and the caller still keeps the result - /// in memory. + /// recompile next run, nothing more, and the caller still keeps the results + /// in memory. A backend that throws must not leave anything locked. void store(const std::string& version, const std::string& device, - const std::string& key_hash, - const binary_cache_entry& e, - const std::vector& blob); - - /// Mark the start of a run of stores that may be committed together, such as - /// a database transaction, rather than one at a time. Every begin_batch is - /// followed by an end_batch, and the two are never nested. Optional: a - /// backend without them stores each entry as it comes. - /// - /// Must not throw. A backend that cannot start a batch stores entries one - /// at a time instead. - void begin_batch(); - - /// Commit the stores made since begin_batch. - /// - /// Must not throw. A failed commit costs those entries a recompile next - /// run, nothing more, and must not leave anything locked. - void end_batch(); + const std::vector& entries); }; #else @@ -139,18 +115,12 @@ struct binary_cache_backend struct MIGRAPHX_EXPORT binary_cache_backend { // - optional> - load(const std::string& version, const std::string& device, const std::string& key_hash); + optional + load(const std::string& version, const std::string& device, const std::string& key); // void store(const std::string& version, const std::string& device, - const std::string& key_hash, - const binary_cache_entry& e, - const std::vector& blob); - // (optional) - void begin_batch(); - // (optional) - void end_batch(); + const std::vector& entries); }; #else @@ -158,32 +128,6 @@ struct MIGRAPHX_EXPORT binary_cache_backend struct binary_cache_backend { private: - template - static auto private_detail_te_default_begin_batch(char, T&& private_detail_te_self) - -> decltype(private_detail_te_self.begin_batch()) - { - private_detail_te_self.begin_batch(); - } - - template - static void private_detail_te_default_begin_batch(float, T&& private_detail_te_self) - { - migraphx::nop(private_detail_te_self); - } - - template - static auto private_detail_te_default_end_batch(char, T&& private_detail_te_self) - -> decltype(private_detail_te_self.end_batch()) - { - private_detail_te_self.end_batch(); - } - - template - static void private_detail_te_default_end_batch(float, T&& private_detail_te_self) - { - migraphx::nop(private_detail_te_self); - } - template struct private_te_unwrap_reference { @@ -206,13 +150,7 @@ struct binary_cache_backend std::declval().store( std::declval(), std::declval(), - std::declval(), - std::declval(), - std::declval&>()), - private_detail_te_default_begin_batch(char(0), - std::declval()), - private_detail_te_default_end_batch(char(0), - std::declval()), + std::declval&>()), void()); template @@ -289,33 +227,19 @@ struct binary_cache_backend return private_detail_te_get_handle().type(); } - optional> - load(const std::string& version, const std::string& device, const std::string& key_hash) + optional + load(const std::string& version, const std::string& device, const std::string& key) { assert((*this).private_detail_te_handle_mem_var); - return (*this).private_detail_te_get_handle().load(version, device, key_hash); + return (*this).private_detail_te_get_handle().load(version, device, key); } void store(const std::string& version, const std::string& device, - const std::string& key_hash, - const binary_cache_entry& e, - const std::vector& blob) + const std::vector& entries) { assert((*this).private_detail_te_handle_mem_var); - (*this).private_detail_te_get_handle().store(version, device, key_hash, e, blob); - } - - void begin_batch() - { - assert((*this).private_detail_te_handle_mem_var); - (*this).private_detail_te_get_handle().begin_batch(); - } - - void end_batch() - { - assert((*this).private_detail_te_handle_mem_var); - (*this).private_detail_te_get_handle().end_batch(); + (*this).private_detail_te_get_handle().store(version, device, entries); } friend bool is_shared(const binary_cache_backend& private_detail_x, @@ -332,16 +256,11 @@ struct binary_cache_backend virtual std::shared_ptr clone() const = 0; virtual const std::type_info& type() const = 0; - virtual optional> load(const std::string& version, - const std::string& device, - const std::string& key_hash) = 0; + virtual optional + load(const std::string& version, const std::string& device, const std::string& key) = 0; virtual void store(const std::string& version, const std::string& device, - const std::string& key_hash, - const binary_cache_entry& e, - const std::vector& blob) = 0; - virtual void begin_batch() = 0; - virtual void end_batch() = 0; + const std::vector& entries) = 0; }; template @@ -371,34 +290,19 @@ struct binary_cache_backend const std::type_info& type() const override { return typeid(private_detail_te_value); } - optional> load(const std::string& version, - const std::string& device, - const std::string& key_hash) override + optional + load(const std::string& version, const std::string& device, const std::string& key) override { - return private_detail_te_value.load(version, device, key_hash); + return private_detail_te_value.load(version, device, key); } void store(const std::string& version, const std::string& device, - const std::string& key_hash, - const binary_cache_entry& e, - const std::vector& blob) override - { - - private_detail_te_value.store(version, device, key_hash, e, blob); - } - - void begin_batch() override - { - - private_detail_te_default_begin_batch(char(0), private_detail_te_value); - } - - void end_batch() override + const std::vector& entries) override { - private_detail_te_default_end_batch(char(0), private_detail_te_value); + private_detail_te_value.store(version, device, entries); } PrivateDetailTypeErasedT private_detail_te_value; diff --git a/src/targets/gpu/include/migraphx/gpu/file_binary_cache.hpp b/src/targets/gpu/include/migraphx/gpu/file_binary_cache.hpp index ffc37708600..0e39e334fc7 100644 --- a/src/targets/gpu/include/migraphx/gpu/file_binary_cache.hpp +++ b/src/targets/gpu/include/migraphx/gpu/file_binary_cache.hpp @@ -36,18 +36,16 @@ namespace migraphx { inline namespace MIGRAPHX_INLINE_NS { namespace gpu { -// A binary_cache_backend that keeps entries as files under a root directory, laid out -// ///.mxr. The version directory is named after the build that -// wrote it, so the tree is self-describing. +// A binary_cache_backend that keeps each entry as a msgpack file under a root directory, laid +// out ///.mxr. The version directory is named after the build +// that wrote it, so the tree is self-describing. struct MIGRAPHX_GPU_EXPORT file_binary_cache { - optional> - load(const std::string& version, const std::string& device, const std::string& key_hash) const; + optional + load(const std::string& version, const std::string& device, const std::string& key) const; void store(const std::string& version, const std::string& device, - const std::string& key_hash, - const binary_cache_entry& e, - const std::vector& blob) const; + const std::vector& entries) const; fs::path root = {}; }; diff --git a/src/targets/gpu/include/migraphx/gpu/sqlite_binary_cache.hpp b/src/targets/gpu/include/migraphx/gpu/sqlite_binary_cache.hpp index 118e15c1e89..7fdec07df2a 100644 --- a/src/targets/gpu/include/migraphx/gpu/sqlite_binary_cache.hpp +++ b/src/targets/gpu/include/migraphx/gpu/sqlite_binary_cache.hpp @@ -37,9 +37,8 @@ inline namespace MIGRAPHX_INLINE_NS { namespace gpu { // A binary_cache_backend that keeps entries as rows in a SQLite database, one row per -// (version, device, key_hash). The stored blob is byte-identical to what the file backend -// writes into a .mxr file, so the two are interchangeable payloads; op_name, problem and -// solution are additionally denormalized into columns so a cache can be inspected with SQL. +// (version, device, md5 of key), with each field of the entry in its own column so a cache can +// be inspected with SQL. The compiled code, a program fragment, is stored as a msgpack blob. // // Holding sqlite and sqlite_stmt by value does not leak the SQLite dependency into this // target: migraphx/sqlite.hpp forward-declares both impl types and never includes sqlite3.h. @@ -53,25 +52,19 @@ struct MIGRAPHX_GPU_EXPORT sqlite_binary_cache /// memory-only rather than raising an error. static optional open(const std::string& path); - optional> - load(const std::string& version, const std::string& device, const std::string& key_hash) const; + optional + load(const std::string& version, const std::string& device, const std::string& key) const; + + /// Store the entries in one transaction, so they cost one commit rather than one each. A + /// failure rolls the whole transaction back and rethrows. void store(const std::string& version, const std::string& device, - const std::string& key_hash, - const binary_cache_entry& e, - const std::vector& blob) const; - - /// Open a transaction, so the stores that follow cost one commit rather than one each. - void begin_batch(); - /// Commit the transaction begin_batch opened, or roll it back if the commit fails. - void end_batch(); + const std::vector& entries); private: sqlite db = {}; sqlite_stmt get_stmt = {}; sqlite_stmt store_stmt = {}; - /// Whether begin_batch opened a transaction that end_batch still has to close. - bool in_batch = false; }; } // namespace gpu diff --git a/src/targets/gpu/sqlite_binary_cache.cpp b/src/targets/gpu/sqlite_binary_cache.cpp index 2140d550013..914b9a6bf09 100644 --- a/src/targets/gpu/sqlite_binary_cache.cpp +++ b/src/targets/gpu/sqlite_binary_cache.cpp @@ -27,7 +27,9 @@ #include #include #include -#include +#include +#include +#include #include #include @@ -52,6 +54,10 @@ constexpr int busy_timeout_ms = 5000; // stores the payload inside the index B-tree, which suits short JSON but not a whole serialized // program fragment, which would spill into overflow chains hanging off the index. // +// Rows are addressed by a hash of the key rather than the key itself, which is the whole compile +// source and would otherwise be stored a second time in the index. The full key is kept in its +// own column and checked on load, so a collision is a miss rather than a wrong kernel. +// // The primary key leads with version so that dropping everything belonging to a superseded // toolchain is a range scan rather than a full table scan. Point lookups bind all three and do // not care about the order. @@ -60,27 +66,28 @@ CREATE TABLE IF NOT EXISTS cache_v1 ( version TEXT NOT NULL, device TEXT NOT NULL, key_hash TEXT NOT NULL, + key TEXT NOT NULL, op_name TEXT NOT NULL, problem TEXT NOT NULL, solution TEXT NOT NULL, - entry BLOB NOT NULL, + code BLOB NOT NULL, timestamp INTEGER NOT NULL, PRIMARY KEY (version, device, key_hash) ); )__migraphx__"; -constexpr const char* get_sql = - "SELECT entry FROM cache_v1 WHERE version = ?1 AND device = ?2 AND key_hash = ?3;"; +constexpr const char* get_sql = "SELECT key, op_name, problem, solution, code FROM cache_v1" + " WHERE version = ?1 AND device = ?2 AND key_hash = ?3;"; // INSERT OR REPLACE is the analogue of the file backend's publish-by-rename: the content is // decided entirely by the key, so two processes compiling the same kernel is benign and the -// last writer wins with equivalent bytes. The timestamp is computed by the database rather +// last writer wins with an equivalent row. The timestamp is computed by the database rather // than the process so that rows written by different machines stay comparable. MIGraphX never // reads it; it lets a cache be pruned by age. constexpr const char* store_sql = "INSERT OR REPLACE INTO cache_v1" - " (version, device, key_hash, op_name, problem, solution, entry, timestamp)" - " VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7, CAST(STRFTIME('%s','now') AS INTEGER));"; + " (version, device, key_hash, key, op_name, problem, solution, code, timestamp)" + " VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7, ?8, CAST(STRFTIME('%s','now') AS INTEGER));"; // Stores come in a burst after each round of compiles, and outside a transaction every one of // them is its own commit, each waiting for the disk. IMMEDIATE takes the write lock up front, so @@ -140,86 +147,79 @@ optional sqlite_binary_cache::open(const std::string& path) return r; } -optional> sqlite_binary_cache::load(const std::string& version, - const std::string& device, - const std::string& key_hash) const +optional sqlite_binary_cache::load(const std::string& version, + const std::string& device, + const std::string& key) const { try { // The primary key makes this at most one row. - auto rows = get_stmt(version, device, key_hash); + auto rows = get_stmt(version, device, md5(key)); auto it = rows.begin(); if(it == rows.end()) return nullopt; - auto row = *it; - const auto& entry = row.at("entry").get_binary(); - return std::vector(entry.begin(), entry.end()); + auto row = *it; + binary_cache_entry e; + e.key = row.at("key").get_string(); + // Rows are addressed by a hash of the key, so the full key is checked here to make a + // collision a miss rather than a wrong kernel. + if(e.key != key) + { + log::warn() << "Ignoring binary cache entry with mismatched key: " << md5(key); + return nullopt; + } + e.op_name = row.at("op_name").get_string(); + e.problem = from_json_string(row.at("problem").get_string()); + e.solution = from_json_string(row.at("solution").get_string()); + const auto& code = row.at("code").get_binary(); + migraphx::from_value(from_msgpack(std::vector(code.begin(), code.end())), e.code); + return e; } catch(const std::exception& ex) { - // A cache that cannot be read is a miss, which costs a recompile and nothing else. - log::warn() << "Failed to read binary cache entry " << key_hash << ": " << ex.what(); + // A cache that cannot be read, or a damaged row, is a miss, which costs a recompile and + // nothing else. + log::warn() << "Ignoring unreadable binary cache entry " << md5(key) << ": " << ex.what(); return nullopt; } } void sqlite_binary_cache::store(const std::string& version, const std::string& device, - const std::string& key_hash, - const binary_cache_entry& e, - const std::vector& blob) const + const std::vector& entries) { // Not prepared for a read-only database, whose stores are skipped. - if(not store_stmt.valid()) - return; - store_stmt(version, - device, - key_hash, - e.op_name, - to_json_string(e.problem), - to_json_string(e.solution), - blob); -} - -void sqlite_binary_cache::begin_batch() -{ - assert(not in_batch); - if(not store_stmt.valid()) + if(not store_stmt.valid() or entries.empty()) return; + db.execute(begin_sql); try { - db.execute(begin_sql); - in_batch = true; - } - catch(const std::exception& ex) - { - // Without a transaction each store commits on its own, which is slower but still works. - log::warn() << "Binary cache stores will be committed one at a time: " << ex.what(); - } -} - -void sqlite_binary_cache::end_batch() -{ - if(not in_batch) - return; - in_batch = false; - try - { + for(const auto& e : entries) + { + store_stmt(version, + device, + md5(e.key), + e.key, + e.op_name, + to_json_string(e.problem), + to_json_string(e.solution), + to_msgpack(migraphx::to_value(e.code))); + } db.execute(commit_sql); } - catch(const std::exception& ex) + catch(...) { - // A commit that fails leaves the transaction open, holding the write lock against every - // other process, so it is rolled back and the batch's entries are lost instead. - log::warn() << "Failed to commit binary cache entries: " << ex.what(); + // A transaction left open would hold the write lock against every other process, so it + // is rolled back and the entries are lost instead. The caller reports the original error. try { db.execute(rollback_sql); } - catch(const std::exception& rex) + catch(const std::exception& ex) { - log::warn() << "Failed to roll back binary cache entries: " << rex.what(); + log::warn() << "Failed to roll back binary cache entries: " << ex.what(); } + throw; } } diff --git a/test/gpu/binary_cache.cpp b/test/gpu/binary_cache.cpp index bd9aa6f71c1..0aad959d9e7 100644 --- a/test/gpu/binary_cache.cpp +++ b/test/gpu/binary_cache.cpp @@ -128,34 +128,44 @@ static std::size_t row_count(const std::string& path, const std::string& table) return std::stoul(rows.front().at("n")); } -using stored_entries = std::map>; +/// The full key of every stored entry, keyed by the key hash that addresses it. +using stored_entries = std::map; -/// Every entry a cache directory holds, keyed by key hash, which names each file. +/// Every entry a cache directory holds. Each file is named by its key hash and holds the msgpack +/// of the whole entry. static stored_entries dir_entries(const migraphx::fs::path& dir) { stored_entries result; auto files = entry_files(dir); std::transform( files.begin(), files.end(), std::inserter(result, result.end()), [](const auto& f) { - return std::make_pair(f.stem().string(), migraphx::read_buffer(f)); + auto v = migraphx::from_msgpack(migraphx::read_buffer(f)); + return std::make_pair(f.stem().string(), v.at("key").get_string()); }); return result; } -/// Every entry a cache database holds, keyed by key hash. +/// Every entry a cache database holds. static stored_entries db_entries(const std::string& path) { stored_entries result; - auto select = migraphx::sqlite::read(path).prepare("SELECT key_hash, entry FROM cache_v1;"); + auto select = migraphx::sqlite::read(path).prepare("SELECT key_hash, key FROM cache_v1;"); auto rows = select(); std::transform(rows.begin(), rows.end(), std::inserter(result, result.end()), [](auto row) { - const auto& blob = row.at("entry").get_binary(); - return std::make_pair(row.at("key_hash").get_string(), - std::vector(blob.begin(), blob.end())); + return std::make_pair(row.at("key_hash").get_string(), row.at("key").get_string()); }); return result; } +/// Whether two entries hold the same thing. +static bool same_entry(const migraphx::gpu::binary_cache::entry& x, + const migraphx::gpu::binary_cache::entry& y) +{ + return x.key == y.key and x.op_name == y.op_name and x.problem == y.problem and + x.solution == y.solution and x.code.fill_map == y.code.fill_map and + *x.code.fragment.get_main_module() == *y.code.fragment.get_main_module(); +} + // What a case that must hold for both backends needs to know about each, so the case can be // written once as a template and registered for both. struct directory_backend @@ -179,7 +189,8 @@ struct database_backend /// Overwrite every stored entry with bytes that do not decode. static void damage(const std::string& p) { - migraphx::sqlite::write(p).execute("UPDATE cache_v1 SET entry = zeroblob(8);"); + // 0xc1 is never used in msgpack, so the code cannot be decoded. + migraphx::sqlite::write(p).execute("UPDATE cache_v1 SET code = X'c1c1c1c1';"); } }; @@ -213,7 +224,7 @@ TEST_CASE(memory_lookup_records_reuse) migraphx::gpu::context ctx; migraphx::gpu::binary_cache cache{migraphx::gpu::binary_cache_settings{.path = ""}}; - cache.insert(ctx, make_entry("a-key")); + cache.insert(ctx, {make_entry("a-key")}); EXPECT(cache.get_stats().compiled == 1); auto found = cache.get(ctx, "a-key"); @@ -235,7 +246,7 @@ static void disk_lookup_records_a_hit() migraphx::gpu::binary_cache_settings settings{Backend::path(td), false}; migraphx::gpu::binary_cache writer{settings}; - writer.insert(ctx, make_entry("shared-key")); + writer.insert(ctx, {make_entry("shared-key")}); migraphx::gpu::binary_cache reader{settings}; @@ -258,7 +269,7 @@ static void corrupt_entry_is_ignored() migraphx::gpu::binary_cache_settings settings{path, false}; migraphx::gpu::binary_cache writer{settings}; - writer.insert(ctx, make_entry("damaged")); + writer.insert(ctx, {make_entry("damaged")}); EXPECT(Backend::stored(path) == 1); Backend::damage(path); @@ -277,7 +288,7 @@ TEST_CASE(no_directory_writes_nothing) migraphx::gpu::context ctx; migraphx::gpu::binary_cache cache{migraphx::gpu::binary_cache_settings{.path = ""}}; - cache.insert(ctx, make_entry("in-memory-only")); + cache.insert(ctx, {make_entry("in-memory-only")}); EXPECT(cache.get(ctx, "in-memory-only").has_value()); EXPECT(cache.get_stats().reused == 1); EXPECT(migraphx::fs::is_empty(td.path)); @@ -399,7 +410,7 @@ TEST_CASE(extension_selects_the_backend) migraphx::tmp_dir dir_td{"binary-cache"}; migraphx::gpu::binary_cache dir_cache{ migraphx::gpu::binary_cache_settings{dir_path(dir_td), false}}; - dir_cache.insert(ctx, make_entry("in-a-directory")); + dir_cache.insert(ctx, {make_entry("in-a-directory")}); auto files = entry_files(dir_td.path); EXPECT(files.size() == 1); EXPECT(migraphx::fs::is_directory(dir_td.path / version_dir)); @@ -412,7 +423,7 @@ TEST_CASE(extension_selects_the_backend) migraphx::tmp_dir db_td{"binary-cache"}; auto path = (db_td.path / name).string(); migraphx::gpu::binary_cache db_cache{migraphx::gpu::binary_cache_settings{path, false}}; - db_cache.insert(ctx, make_entry("in-a-database")); + db_cache.insert(ctx, {make_entry("in-a-database")}); EXPECT(migraphx::fs::is_regular_file(path)); EXPECT(row_count(path, "cache_v1") == 1); @@ -433,7 +444,7 @@ TEST_CASE(unusable_database_degrades_to_memory) migraphx::gpu::binary_cache_settings settings{(blocker / "cache.db").string(), false}; migraphx::gpu::binary_cache cache{settings}; - cache.insert(ctx, make_entry("nowhere")); + cache.insert(ctx, {make_entry("nowhere")}); EXPECT(cache.get(ctx, "nowhere").has_value()); EXPECT(cache.get_stats().reused == 1); @@ -454,7 +465,7 @@ TEST_CASE(not_a_database_degrades_to_memory) migraphx::write_buffer(path, garbage); migraphx::gpu::binary_cache cache{migraphx::gpu::binary_cache_settings{path, false}}; - cache.insert(ctx, make_entry("in-memory")); + cache.insert(ctx, {make_entry("in-memory")}); EXPECT(cache.get(ctx, "in-memory").has_value()); EXPECT(cache.get_stats().reused == 1); EXPECT((migraphx::read_buffer(path) == garbage)); @@ -472,13 +483,13 @@ TEST_CASE(incompatible_schema_degrades_to_memory) EXPECT(not migraphx::gpu::sqlite_binary_cache::open(path).has_value()); migraphx::gpu::binary_cache cache{migraphx::gpu::binary_cache_settings{path, false}}; - cache.insert(ctx, make_entry("in-memory")); + cache.insert(ctx, {make_entry("in-memory")}); EXPECT(cache.get(ctx, "in-memory").has_value()); EXPECT(row_count(path, "cache_v1") == 0); } -// Both backends store the same serialized entry, so a cache can move between them. -TEST_CASE(backends_store_identical_bytes) +// Both backends address an entry by the same key hash, and give back what they were given. +TEST_CASE(backends_store_the_same_entry) { migraphx::gpu::context ctx; auto e = make_entry("interchange"); @@ -486,24 +497,30 @@ TEST_CASE(backends_store_identical_bytes) migraphx::tmp_dir dir_td{"binary-cache"}; migraphx::gpu::binary_cache dir_cache{ migraphx::gpu::binary_cache_settings{dir_path(dir_td), false}}; - dir_cache.insert(ctx, e); - auto files = entry_files(dir_td.path); - EXPECT(files.size() == 1); - auto from_file = migraphx::read_buffer(files.front()); - EXPECT(not from_file.empty()); + dir_cache.insert(ctx, {e}); + auto from_dir = dir_entries(dir_td.path); + EXPECT(from_dir.size() == 1); + EXPECT(from_dir.begin()->first == migraphx::md5(e.key)); + EXPECT(from_dir.begin()->second == e.key); migraphx::tmp_dir db_td{"binary-cache"}; auto path = db_path(db_td); migraphx::gpu::binary_cache db_cache{migraphx::gpu::binary_cache_settings{path, false}}; - db_cache.insert(ctx, e); + db_cache.insert(ctx, {e}); + EXPECT((db_entries(path) == from_dir)); - auto from_db = db_entries(path); - EXPECT(from_db.size() == 1); - EXPECT((from_db.begin()->second == from_file)); + migraphx::tmp_dir td{"binary-cache"}; + for(auto& backend : both_backends(td)) + { + backend.store("v", "dev", {e}); + auto got = backend.load("v", "dev", e.key); + EXPECT(got.has_value()); + EXPECT(same_entry(*got, e)); + } } -// A whole compile against each backend has to leave the same kernels behind: the same key -// hashes, each with the same bytes. That makes the choice of backend purely a storage decision. +// A whole compile against each backend has to leave the same kernels behind, under the same key +// hashes. That makes the choice of backend purely a storage decision. TEST_CASE(backends_hold_the_same_entries_after_a_compile) { migraphx::tmp_dir dir_td{"binary-cache"}; @@ -522,9 +539,9 @@ TEST_CASE(backends_hold_the_same_entries_after_a_compile) EXPECT((from_dir == from_db)); } -// An entry copied from one backend into the other is a hit there and decodes to the same code, -// so an existing cache can be converted rather than rebuilt. The only translation is the -// version, which a directory names with the short id and a database records in full. +// An entry loaded from one backend and stored into the other is a hit there and decodes to the +// same code, so an existing cache can be converted rather than rebuilt. The only translation is +// the version, which a directory names with the short id and a database records in full. TEST_CASE(entries_move_between_backends) { migraphx::gpu::context ctx; @@ -539,15 +556,17 @@ TEST_CASE(entries_move_between_backends) migraphx::tmp_dir db_td{"binary-cache"}; migraphx::gpu::binary_cache writer{ migraphx::gpu::binary_cache_settings{dir_path(dir_td), false}}; - writer.insert(ctx, e); + writer.insert(ctx, {e}); auto files = entry_files(dir_td.path); EXPECT(files.size() == 1); - const auto& file = files.front(); - auto device = file.parent_path().filename().string(); - auto db = migraphx::gpu::sqlite_binary_cache::open(db_path(db_td)); + auto device = files.front().parent_path().filename().string(); + migraphx::gpu::file_binary_cache dir{dir_td.path}; + auto loaded = dir.load(short_version, device, e.key); + EXPECT(loaded.has_value()); + auto db = migraphx::gpu::sqlite_binary_cache::open(db_path(db_td)); EXPECT(db.has_value()); - db->store(long_version, device, file.stem().string(), e, migraphx::read_buffer(file)); + db->store(long_version, device, {*loaded}); migraphx::gpu::binary_cache reader{ migraphx::gpu::binary_cache_settings{db_path(db_td), false}}; @@ -563,22 +582,19 @@ TEST_CASE(entries_move_between_backends) migraphx::tmp_dir dir_td{"binary-cache"}; migraphx::gpu::binary_cache writer{ migraphx::gpu::binary_cache_settings{db_path(db_td), false}}; - writer.insert(ctx, e); + writer.insert(ctx, {e}); - auto select = migraphx::sqlite::read(db_path(db_td)) - .prepare("SELECT version, device, key_hash, entry FROM cache_v1;"); - auto result = select(); - auto rows = std::vector(result.begin(), result.end()); + auto rows = + migraphx::sqlite::read(db_path(db_td)).execute("SELECT version, device FROM cache_v1;"); EXPECT(rows.size() == 1); - const auto& row = rows.front(); - EXPECT(row.at("version").get_string() == long_version); - const auto& blob = row.at("entry").get_binary(); - migraphx::gpu::file_binary_cache files{dir_td.path}; - files.store(short_version, - row.at("device").get_string(), - row.at("key_hash").get_string(), - e, - std::vector(blob.begin(), blob.end())); + EXPECT(rows.front().at("version") == long_version); + const auto& device = rows.front().at("device"); + auto db = migraphx::gpu::sqlite_binary_cache::open(db_path(db_td)); + EXPECT(db.has_value()); + auto loaded = db->load(long_version, device, e.key); + EXPECT(loaded.has_value()); + migraphx::gpu::file_binary_cache dir{dir_td.path}; + dir.store(short_version, device, {*loaded}); migraphx::gpu::binary_cache reader{ migraphx::gpu::binary_cache_settings{dir_path(dir_td), false}}; @@ -601,18 +617,15 @@ TEST_CASE(two_connections_share_a_database) EXPECT(a.has_value()); EXPECT(b.has_value()); - const std::vector first{'f', 'i', 'r', 's', 't'}; - const std::vector second{'s', 'e', 'c', 'o', 'n', 'd'}; - - a->store("v", "dev", "k1", make_entry("k1"), first); + a->store("v", "dev", {make_entry("k1")}); auto from_b = b->load("v", "dev", "k1"); EXPECT(from_b.has_value()); - EXPECT((*from_b == first)); + EXPECT(from_b->key == "k1"); - b->store("v", "dev", "k2", make_entry("k2"), second); + b->store("v", "dev", {make_entry("k2")}); auto from_a = a->load("v", "dev", "k2"); EXPECT(from_a.has_value()); - EXPECT((*from_a == second)); + EXPECT(from_a->key == "k2"); EXPECT(not a->load("v", "dev", "absent").has_value()); } @@ -627,8 +640,8 @@ TEST_CASE(sqlite_records_the_full_version_id) auto path = db_path(td); migraphx::gpu::binary_cache cache{migraphx::gpu::binary_cache_settings{path, false}}; - cache.insert(ctx, make_entry("one")); - cache.insert(ctx, make_entry("two")); + cache.insert(ctx, {make_entry("one")}); + cache.insert(ctx, {make_entry("two")}); EXPECT(row_count(path, "cache_v1") == 2); auto rows = migraphx::sqlite::read(path).execute("SELECT DISTINCT version FROM cache_v1;"); @@ -636,8 +649,7 @@ TEST_CASE(sqlite_records_the_full_version_id) EXPECT(rows.front().at("version") == migraphx::gpu::binary_cache::version_id(false)); } -// The op name, problem and solution are copied into columns only so a cache can be inspected -// with SQL; loads never read them, so nothing else would notice them being wrong. +// Each field of an entry has its own column, so a cache can be inspected with SQL. TEST_CASE(sqlite_records_what_each_entry_was_compiled_for) { migraphx::tmp_dir td{"binary-cache"}; @@ -645,13 +657,14 @@ TEST_CASE(sqlite_records_what_each_entry_was_compiled_for) auto path = db_path(td); auto e = make_entry("described"); migraphx::gpu::binary_cache cache{migraphx::gpu::binary_cache_settings{path, false}}; - cache.insert(ctx, e); + cache.insert(ctx, {e}); auto rows = migraphx::sqlite::read(path).execute( - "SELECT key_hash, op_name, problem, solution FROM cache_v1;"); + "SELECT key_hash, key, op_name, problem, solution FROM cache_v1;"); EXPECT(rows.size() == 1); const auto& row = rows.front(); EXPECT(row.at("key_hash") == migraphx::md5(e.key)); + EXPECT(row.at("key") == e.key); EXPECT(row.at("op_name") == e.op_name); EXPECT(migraphx::from_json_string(row.at("problem")) == e.problem); EXPECT(migraphx::from_json_string(row.at("solution")) == e.solution); @@ -667,7 +680,7 @@ TEST_CASE(sqlite_read_only_database_still_serves_hits) migraphx::gpu::binary_cache_settings settings{path, false}; { migraphx::gpu::binary_cache writer{settings}; - writer.insert(ctx, make_entry("existing")); + writer.insert(ctx, {make_entry("existing")}); } const auto writable = migraphx::fs::perms::owner_write | migraphx::fs::perms::group_write | @@ -681,7 +694,7 @@ TEST_CASE(sqlite_read_only_database_still_serves_hits) EXPECT(reader.get(ctx, "existing").has_value()); EXPECT(reader.get_stats().hits == 1); - reader.insert(ctx, make_entry("new")); + reader.insert(ctx, {make_entry("new")}); EXPECT(reader.get(ctx, "new").has_value()); if(protected_file) { @@ -698,10 +711,9 @@ TEST_CASE(sqlite_read_only_database_still_serves_hits) TEST_CASE(backends_scope_entries_by_version_and_device) { migraphx::tmp_dir td{"binary-cache"}; - const std::vector blob{'p', 'a', 'y'}; for(auto& backend : both_backends(td)) { - backend.store("v1", "dev1", "k", make_entry("k"), blob); + backend.store("v1", "dev1", {make_entry("k")}); EXPECT(backend.load("v1", "dev1", "k").has_value()); EXPECT(not backend.load("v2", "dev1", "k").has_value()); @@ -714,15 +726,17 @@ TEST_CASE(backends_scope_entries_by_version_and_device) TEST_CASE(backends_store_overwrites_in_place) { migraphx::tmp_dir td{"binary-cache"}; - const std::vector replacement{'n', 'e', 'w'}; + auto original = make_entry("k"); + auto replacement = make_entry("k"); + replacement.solution = migraphx::value{{"algo", "replaced"}}; for(auto& backend : both_backends(td)) { - backend.store("v", "dev", "k", make_entry("k"), {'o', 'l', 'd'}); - backend.store("v", "dev", "k", make_entry("k"), replacement); + backend.store("v", "dev", {original}); + backend.store("v", "dev", {replacement}); auto got = backend.load("v", "dev", "k"); EXPECT(got.has_value()); - EXPECT((*got == replacement)); + EXPECT(got->solution == replacement.solution); } EXPECT(row_count(db_path(td), "cache_v1") == 1); EXPECT(entry_files(td.path / "files").size() == 1); @@ -737,11 +751,11 @@ TEST_CASE(file_store_leaves_only_entries_behind) migraphx::gpu::binary_cache_settings settings{dir_path(td), false}; migraphx::gpu::binary_cache first{settings}; - first.insert(ctx, make_entry("one")); - first.insert(ctx, make_entry("two")); + first.insert(ctx, {make_entry("one")}); + first.insert(ctx, {make_entry("two")}); // A second cache stores the same key again, over the file the first one published. migraphx::gpu::binary_cache second{settings}; - second.insert(ctx, make_entry("one")); + second.insert(ctx, {make_entry("one")}); std::vector items{ migraphx::fs::recursive_directory_iterator{td.path}, @@ -768,20 +782,16 @@ TEST_CASE(storage_is_opened_on_first_use) EXPECT(migraphx::fs::exists(path)); } -// Inserts made inside a batch are all there once it ends. The database commits them together; -// the directory backend has no batching and stores each one as it comes. +// Entries inserted together are all there afterwards. The database commits them in one +// transaction; the directory backend writes each file as it comes. template -static void batched_inserts_are_all_stored() +static void inserts_of_many_entries_are_all_stored() { migraphx::tmp_dir td{"binary-cache"}; migraphx::gpu::context ctx; auto path = Backend::path(td); migraphx::gpu::binary_cache cache{migraphx::gpu::binary_cache_settings{path, false}}; - { - migraphx::gpu::binary_cache::store_batch batch{cache}; - cache.insert(ctx, make_entry("first")); - cache.insert(ctx, make_entry("second")); - } + cache.insert(ctx, {make_entry("first"), make_entry("second")}); EXPECT(Backend::stored(path) == 2); migraphx::gpu::binary_cache reader{migraphx::gpu::binary_cache_settings{path, false}}; @@ -790,43 +800,54 @@ static void batched_inserts_are_all_stored() EXPECT(reader.get_stats().hits == 2); } -TEST_CASE_REGISTER(batched_inserts_are_all_stored); -TEST_CASE_REGISTER(batched_inserts_are_all_stored); +TEST_CASE_REGISTER(inserts_of_many_entries_are_all_stored); +TEST_CASE_REGISTER(inserts_of_many_entries_are_all_stored); -// A batch holds the database's write lock, so another connection must be able to write again as -// soon as it ends. -TEST_CASE(sqlite_batch_releases_the_database) +// A store holds the database's write lock for its transaction, so another connection must be +// able to write again as soon as it returns. +TEST_CASE(sqlite_store_releases_the_database) { migraphx::tmp_dir td{"binary-cache"}; migraphx::gpu::context ctx; auto path = db_path(td); migraphx::gpu::binary_cache cache{migraphx::gpu::binary_cache_settings{path, false}}; - { - migraphx::gpu::binary_cache::store_batch batch{cache}; - cache.insert(ctx, make_entry("batched")); - } + cache.insert(ctx, {make_entry("first"), make_entry("second")}); auto other = migraphx::gpu::sqlite_binary_cache::open(path); EXPECT(other.has_value()); - other->store("v", "dev", "k", make_entry("k"), {'x'}); - EXPECT(row_count(path, "cache_v1") == 2); + other->store("v", "dev", {make_entry("k")}); + EXPECT(row_count(path, "cache_v1") == 3); } -// The backend layer moves opaque bytes and never decodes them, so a payload that is not even -// msgpack still round-trips. Both backends go through the same type-erased wrapper here. +// Rows are addressed by a hash of the key, so a row whose stored key differs from the one asked +// for, as a hash collision would leave, is a miss rather than a wrong kernel. +TEST_CASE(sqlite_mismatched_key_is_a_miss) +{ + migraphx::tmp_dir td{"binary-cache"}; + auto path = db_path(td); + auto db = migraphx::gpu::sqlite_binary_cache::open(path); + EXPECT(db.has_value()); + db->store("v", "dev", {make_entry("k")}); + EXPECT(db->load("v", "dev", "k").has_value()); + + migraphx::sqlite::write(path).execute("UPDATE cache_v1 SET key = 'another';"); + EXPECT(not db->load("v", "dev", "k").has_value()); +} + +// Both backends go through the same type-erased wrapper and give back the entry they were +// given, field for field. TEST_CASE(backends_round_trip_through_the_wrapper) { migraphx::tmp_dir td{"binary-cache"}; - const std::vector blob{'\0', 'n', 'o', 't', '\0', 'm', 's', 'g', '\xff'}; - auto e = make_entry("opaque"); + auto e = make_entry("round-trip"); for(auto& backend : both_backends(td)) { - EXPECT(not backend.load("v", "dev", "k").has_value()); - backend.store("v", "dev", "k", e, blob); - auto got = backend.load("v", "dev", "k"); + EXPECT(not backend.load("v", "dev", e.key).has_value()); + backend.store("v", "dev", {e}); + auto got = backend.load("v", "dev", e.key); EXPECT(got.has_value()); - EXPECT((*got == blob)); + EXPECT(same_entry(*got, e)); } } diff --git a/test/sqlite.cpp b/test/sqlite.cpp index 8f84e111c6c..02910f109af 100644 --- a/test/sqlite.cpp +++ b/test/sqlite.cpp @@ -33,7 +33,7 @@ /// Every row a call produced, so a test can count and inspect them. static std::vector collect(const migraphx::sqlite_stmt::rows& r) { - return std::vector(r.begin(), r.end()); + return {r.begin(), r.end()}; } TEST_CASE(read_write) diff --git a/tools/include/gpu/binary_cache_backend.hpp b/tools/include/gpu/binary_cache_backend.hpp index 5a2c7008372..c2d3d5fb21b 100644 --- a/tools/include/gpu/binary_cache_backend.hpp +++ b/tools/include/gpu/binary_cache_backend.hpp @@ -30,8 +30,7 @@ // into the gpu target tree). Do not edit the generated header by hand. // // Any type T satisfies the binary_cache_backend concept if it provides the -// member functions listed below; begin_batch and end_batch are optional and -// default to doing nothing. The wrapper holds T by shared_ptr and forwards +// member functions listed below. The wrapper holds T by shared_ptr and forwards // each call through a virtual dispatch, matching problem_cache_backend. // // Notes: @@ -40,8 +39,6 @@ // * Backends must be copyable: the wrapper shares T and clones it on a // non-const call while the handle is shared. sqlite_binary_cache shares its // connection across copies. -// * The members are non-const so a backend can track an open batch; a backend -// may still declare them const. // #ifndef MIGRAPHX_GUARD_GPU_BINARY_CACHE_BACKEND_HPP #define MIGRAPHX_GUARD_GPU_BINARY_CACHE_BACKEND_HPP @@ -68,67 +65,46 @@ namespace gpu { /// Type-erased interface for binary-cache storage backends. /// -/// A backend persists serialized binary_cache_entry blobs to some medium (a -/// directory of files or a SQLite database). Entries are addressed by three -/// strings the caller has already computed: +/// A backend persists binary_cache_entry values to some medium (a directory of +/// files or a SQLite database), and decides for itself how to serialize them. +/// Entries are addressed by their key, scoped by two strings the caller has +/// already computed: /// /// * `version` -- binary_cache::version_id(), identifying the toolchain and /// the embedded kernel sources that produced the entry. Never empty; the /// caller skips persistence entirely when it is. /// * `device` -- the GPU the entry was compiled for. -/// * `key_hash` -- md5 of the compile key. A hash rather than the key itself -/// because a file backend needs a short name; a collision is harmless, -/// since the caller re-checks the full key against the decoded entry. /// -/// Together these three form the identity of an entry. A backend must keep -/// entries with different scopes distinct rather than overwriting across them. +/// A backend must keep entries with different scopes distinct rather than +/// overwriting across them. It may address entries by a hash of the key, for +/// instance to keep file names short, but must then check the full key when +/// loading so that a collision is a miss rather than a wrong kernel. struct binary_cache_backend { - /// Return the serialized entry for this key, or nullopt for a miss. + /// Return the entry stored for this key, or nullopt for a miss. /// /// nullopt also covers every failure: a missing file, an unreadable - /// database, a permissions problem. A cache that cannot be read is not an - /// error, it is a cache miss, and the caller recompiles. + /// database, a damaged entry, a permissions problem. A cache that cannot be + /// read is not an error, it is a cache miss, and the caller recompiles. /// /// Must not throw. - optional> - load(const std::string& version, const std::string& device, const std::string& key_hash); + optional + load(const std::string& version, const std::string& device, const std::string& key); - /// Persist `blob`, the msgpack encoding of `e`, under this key. - /// - /// `e` is passed alongside `blob` so a backend may denormalize op_name, - /// problem and solution into queryable columns. Those fields are also - /// inside `blob`, which stays the authoritative record -- a backend that - /// stores them separately must still be able to answer a load() with the - /// blob alone. + /// Persist every entry in `entries` under its key. The entries arrive + /// together so a backend can commit them at once, such as in one database + /// transaction, rather than one at a time. /// /// Overwriting an existing entry is expected and safe: the content is /// decided entirely by the key, so a writer that loses a race replaces the - /// entry with equivalent bytes. + /// entry with an equivalent one. /// /// May throw: the caller reports a failed store as a warning. It costs a - /// recompile next run, nothing more, and the caller still keeps the result - /// in memory. + /// recompile next run, nothing more, and the caller still keeps the results + /// in memory. A backend that throws must not leave anything locked. void store(const std::string& version, const std::string& device, - const std::string& key_hash, - const binary_cache_entry& e, - const std::vector& blob); - - /// Mark the start of a run of stores that may be committed together, such as - /// a database transaction, rather than one at a time. Every begin_batch is - /// followed by an end_batch, and the two are never nested. Optional: a - /// backend without them stores each entry as it comes. - /// - /// Must not throw. A backend that cannot start a batch stores entries one - /// at a time instead. - void begin_batch(); - - /// Commit the stores made since begin_batch. - /// - /// Must not throw. A failed commit costs those entries a recompile next - /// run, nothing more, and must not leave anything locked. - void end_batch(); + const std::vector& entries); }; #else @@ -136,19 +112,15 @@ struct binary_cache_backend <% interface('binary_cache_backend', virtual('load', - returns = 'optional>', - version = 'const std::string&', - device = 'const std::string&', - key_hash = 'const std::string&'), + returns = 'optional', + version = 'const std::string&', + device = 'const std::string&', + key = 'const std::string&'), virtual('store', - returns = 'void', - version = 'const std::string&', - device = 'const std::string&', - key_hash = 'const std::string&', - e = 'const binary_cache_entry&', - blob = 'const std::vector&'), - virtual('begin_batch', returns = 'void', default = 'migraphx::nop'), - virtual('end_batch', returns = 'void', default = 'migraphx::nop')) + returns = 'void', + version = 'const std::string&', + device = 'const std::string&', + entries = 'const std::vector&')) %> #endif From 217796699f1642ccad74858b9b64767192c4e3fe Mon Sep 17 00:00:00 2001 From: pnikolic-amd Date: Thu, 1 Oct 2026 15:43:20 +0200 Subject: [PATCH 12/14] sqlite: reject empty statements at runtime; a failed binary cache store leaves the database read-only instead of retrying. --- src/sqlite.cpp | 6 ++++-- .../gpu/include/migraphx/gpu/sqlite_binary_cache.hpp | 2 +- src/targets/gpu/sqlite_binary_cache.cpp | 7 ++++++- 3 files changed, 11 insertions(+), 4 deletions(-) diff --git a/src/sqlite.cpp b/src/sqlite.cpp index 105050b0d38..e00d230b8ac 100644 --- a/src/sqlite.cpp +++ b/src/sqlite.cpp @@ -152,8 +152,10 @@ sqlite_stmt sqlite::prepare(const std::string& sql) result.impl->ptr = sqlite3_stmt_ptr{stmt_tmp}; if(rc != SQLITE_OK) MIGRAPHX_THROW("error preparing '" + sql + "': " + impl->error_message()); - // sqlite succeeds without a statement for text that holds none, such as only a comment. - assert(stmt_tmp != nullptr); + // sqlite succeeds without a statement for text that holds none, such as only a comment, and + // a statement with no handle would pass null into sqlite on its first call. + if(stmt_tmp == nullptr) + MIGRAPHX_THROW("error preparing '" + sql + "': no statement in the text"); return result; } diff --git a/src/targets/gpu/include/migraphx/gpu/sqlite_binary_cache.hpp b/src/targets/gpu/include/migraphx/gpu/sqlite_binary_cache.hpp index 7fdec07df2a..921c9e87343 100644 --- a/src/targets/gpu/include/migraphx/gpu/sqlite_binary_cache.hpp +++ b/src/targets/gpu/include/migraphx/gpu/sqlite_binary_cache.hpp @@ -56,7 +56,7 @@ struct MIGRAPHX_GPU_EXPORT sqlite_binary_cache load(const std::string& version, const std::string& device, const std::string& key) const; /// Store the entries in one transaction, so they cost one commit rather than one each. A - /// failure rolls the whole transaction back and rethrows. + /// failure rolls the whole transaction back, leaves the cache read-only and rethrows. void store(const std::string& version, const std::string& device, const std::vector& entries); diff --git a/src/targets/gpu/sqlite_binary_cache.cpp b/src/targets/gpu/sqlite_binary_cache.cpp index 914b9a6bf09..c6b1f32ff04 100644 --- a/src/targets/gpu/sqlite_binary_cache.cpp +++ b/src/targets/gpu/sqlite_binary_cache.cpp @@ -191,9 +191,9 @@ void sqlite_binary_cache::store(const std::string& version, // Not prepared for a read-only database, whose stores are skipped. if(not store_stmt.valid() or entries.empty()) return; - db.execute(begin_sql); try { + db.execute(begin_sql); for(const auto& e : entries) { store_stmt(version, @@ -219,6 +219,11 @@ void sqlite_binary_cache::store(const std::string& version, { log::warn() << "Failed to roll back binary cache entries: " << ex.what(); } + // A database that refused one write, often after waiting out the busy timeout, would + // most likely refuse the next one as well, so later stores are skipped and the cache + // carries on read-only. Lookups keep working. + store_stmt = {}; + log::warn() << "Binary cache is read-only from now on"; throw; } } From 342d806a9d85f528debc7eac6fb42d91f1bf2982 Mon Sep 17 00:00:00 2001 From: pnikolic-amd Date: Thu, 1 Oct 2026 16:00:15 +0200 Subject: [PATCH 13/14] Simplify tool --- src/include/migraphx/sqlite.hpp | 4 +-- src/sqlite.cpp | 25 +++++++++---- src/targets/gpu/compile_ops.cpp | 12 +++++-- src/targets/gpu/file_binary_cache.cpp | 35 ++++++++++--------- .../migraphx/gpu/binary_cache_backend.hpp | 5 +-- src/targets/gpu/sqlite_binary_cache.cpp | 23 ++++++------ test/gpu/binary_cache.cpp | 31 +++++----------- tools/include/gpu/binary_cache_backend.hpp | 5 +-- 8 files changed, 76 insertions(+), 64 deletions(-) diff --git a/src/include/migraphx/sqlite.hpp b/src/include/migraphx/sqlite.hpp index 8f6fc235436..b5792a9afef 100644 --- a/src/include/migraphx/sqlite.hpp +++ b/src/include/migraphx/sqlite.hpp @@ -46,8 +46,8 @@ inline namespace MIGRAPHX_INLINE_NS { struct sqlite_impl; struct sqlite_stmt_impl; -/// A prepared statement, holding a reference to the connection it was prepared on so it can -/// never outlive it. Copies share the same statement. +/// A prepared statement. It shares ownership of the connection it was prepared on, which stays +/// open for as long as the statement exists. Copies share the same statement. /// /// Calling it with arguments runs it: the arguments are bound to the parameters in order and the /// result comes back as a range of rows. Since copies share one statement, only the rows of one diff --git a/src/sqlite.cpp b/src/sqlite.cpp index e00d230b8ac..f5cf58219ca 100644 --- a/src/sqlite.cpp +++ b/src/sqlite.cpp @@ -25,7 +25,6 @@ #include #include #include -#include #include #include #include @@ -144,6 +143,7 @@ std::vector> sqlite::execute(const sqlite_stmt sqlite::prepare(const std::string& sql) { + assert(impl != nullptr); sqlite3_stmt* stmt_tmp = nullptr; int rc = sqlite3_prepare_v2(impl->get(), sql.c_str(), -1, &stmt_tmp, nullptr); sqlite_stmt result; @@ -159,9 +159,17 @@ sqlite_stmt sqlite::prepare(const std::string& sql) return result; } -void sqlite::set_busy_timeout(int ms) { sqlite3_busy_timeout(impl->get(), ms); } +void sqlite::set_busy_timeout(int ms) +{ + assert(impl != nullptr); + sqlite3_busy_timeout(impl->get(), ms); +} -bool sqlite::read_only() const { return sqlite3_db_readonly(impl->get(), "main") == 1; } +bool sqlite::read_only() const +{ + assert(impl != nullptr); + return sqlite3_db_readonly(impl->get(), "main") == 1; +} void sqlite_stmt::bind(int i, std::string_view s) const { @@ -222,8 +230,10 @@ void sqlite_stmt::reset() const noexcept /// Column i of the current row, keyed by its name. static value column_value(sqlite3_stmt* stmt, int i) { - std::string name = sqlite3_column_name(stmt, i); - auto type = sqlite3_column_type(stmt, i); + // Null only when sqlite runs out of memory, which would make the string below undefined. + const char* name = sqlite3_column_name(stmt, i); + assert(name != nullptr); + auto type = sqlite3_column_type(stmt, i); switch(type) { case SQLITE_INTEGER: return value::pair(name, std::int64_t{sqlite3_column_int64(stmt, i)}); @@ -249,8 +259,9 @@ value sqlite_stmt::to_value() const { auto* stmt = impl->get(); value columns = value::object{}; - for(auto i : range(sqlite3_column_count(stmt))) - columns.insert(column_value(stmt, static_cast(i))); + const int n = sqlite3_column_count(stmt); + for(int i = 0; i < n; ++i) + columns.insert(column_value(stmt, i)); return columns; } diff --git a/src/targets/gpu/compile_ops.cpp b/src/targets/gpu/compile_ops.cpp index 46d1f512bc6..7a2cb2acfac 100644 --- a/src/targets/gpu/compile_ops.cpp +++ b/src/targets/gpu/compile_ops.cpp @@ -252,7 +252,12 @@ static compiler_replace compile_cached(context& ctx, } auto cr = compile_fragment(ctx, ins, preop, solution); if(auto e = make_cache_entry(preop, solution, key, problem, cr.code)) - ctx.get_binary_cache().insert(ctx, {std::move(*e)}); + { + // Built by push_back, since an initializer list would copy the entry rather than move it. + std::vector es; + es.push_back(std::move(*e)); + ctx.get_binary_cache().insert(ctx, std::move(es)); + } return cr; } @@ -796,12 +801,13 @@ struct compile_manager assert(std::all_of( tasks.begin(), tasks.end(), [&](const auto& task) { return task.first->ctx == ctx; })); std::vector entries; + entries.reserve(tasks.size()); for(const auto& [cp, cell] : tasks) { if(not cell->result.has_value()) continue; - // When verifying, reused results are stored again, rewriting the same bytes - // harmlessly. + // When verifying, reused results are stored again, harmlessly replacing each entry + // with an equivalent one. if(auto e = cp->cache_entry(cell->solution, cell->key, cell->result->code)) entries.push_back(std::move(*e)); assert(not cell->result->code.empty()); diff --git a/src/targets/gpu/file_binary_cache.cpp b/src/targets/gpu/file_binary_cache.cpp index 585f6a107df..63731546b5b 100644 --- a/src/targets/gpu/file_binary_cache.cpp +++ b/src/targets/gpu/file_binary_cache.cpp @@ -40,15 +40,19 @@ namespace gpu { static_assert(std::is_constructible{}, "file_binary_cache must satisfy the binary_cache_backend concept"); -/// Where an entry lives. The key is the whole compile source, so it is hashed to keep the name -/// short. The caller guarantees a non-empty version, so entries compiled by different toolchains -/// can never land on the same path. -static fs::path entry_path(const fs::path& root, - const std::string& version, - const std::string& device, - const std::string& key) +/// The directory holding one device's entries for one toolchain. The caller guarantees a +/// non-empty version, so entries compiled by different toolchains can never land in the same one. +static fs::path +entry_dir(const fs::path& root, const std::string& version, const std::string& device) { - return root / version / device / (md5(key) + ".mxr"); + return root / version / device; +} + +/// Where an entry lives in its directory. The key is the whole compile source, so it is hashed to +/// keep the name short. +static fs::path entry_path(const fs::path& dir, const std::string& key) +{ + return dir / (md5(key) + ".mxr"); } /// Publish by rename so a reader never sees a half-written file. The temporary stays beside @@ -76,7 +80,7 @@ optional file_binary_cache::load(const std::string& version, const std::string& device, const std::string& key) const { - auto path = entry_path(root, version, device, key); + auto path = entry_path(entry_dir(root, version, device), key); binary_cache_entry e; try { @@ -104,14 +108,13 @@ void file_binary_cache::store(const std::string& version, const std::string& device, const std::vector& entries) const { + // Every entry shares one directory, so it is created once for the whole batch. + auto dir = entry_dir(root, version, device); + fs::create_directories(dir); + // The content is decided entirely by the key, so a writer that loses the publish race + // replaces the file with the same bytes and no locking is needed. for(const auto& e : entries) - { - auto path = entry_path(root, version, device, e.key); - // The content is decided entirely by the key, so a writer that loses the publish race - // replaces the file with the same bytes and no locking is needed. - fs::create_directories(path.parent_path()); - write_atomically(path, to_msgpack(migraphx::to_value(e))); - } + write_atomically(entry_path(dir, e.key), to_msgpack(migraphx::to_value(e))); } } // namespace gpu diff --git a/src/targets/gpu/include/migraphx/gpu/binary_cache_backend.hpp b/src/targets/gpu/include/migraphx/gpu/binary_cache_backend.hpp index 0b91a7c14cd..0ded21c4c61 100644 --- a/src/targets/gpu/include/migraphx/gpu/binary_cache_backend.hpp +++ b/src/targets/gpu/include/migraphx/gpu/binary_cache_backend.hpp @@ -70,8 +70,9 @@ namespace gpu { /// Entries are addressed by their key, scoped by two strings the caller has /// already computed: /// -/// * `version` -- binary_cache::version_id(), identifying the toolchain and -/// the embedded kernel sources that produced the entry. Never empty; the +/// * `version` -- binary_cache::version_id(), short for the directory +/// backend and full for the database, identifying the toolchain and the +/// embedded kernel sources that produced the entry. Never empty; the /// caller skips persistence entirely when it is. /// * `device` -- the GPU the entry was compiled for. /// diff --git a/src/targets/gpu/sqlite_binary_cache.cpp b/src/targets/gpu/sqlite_binary_cache.cpp index c6b1f32ff04..56d8614fa1a 100644 --- a/src/targets/gpu/sqlite_binary_cache.cpp +++ b/src/targets/gpu/sqlite_binary_cache.cpp @@ -48,7 +48,7 @@ constexpr int busy_timeout_ms = 5000; // The table name carries the schema version, so an incompatible change is a new table that old // binaries ignore rather than a migration. This is orthogonal to binary_cache_format, which -// versions the entry payload and reaches the row through the version column. +// versions how entries are serialized and reaches each row through the version column. // // Deliberately not WITHOUT ROWID, unlike the sibling table in sqlite_problem_cache: that clause // stores the payload inside the index B-tree, which suits short JSON but not a whole serialized @@ -151,35 +151,37 @@ optional sqlite_binary_cache::load(const std::string& versio const std::string& device, const std::string& key) const { + auto key_hash = md5(key); try { // The primary key makes this at most one row. - auto rows = get_stmt(version, device, md5(key)); + auto rows = get_stmt(version, device, key_hash); auto it = rows.begin(); if(it == rows.end()) return nullopt; auto row = *it; - binary_cache_entry e; - e.key = row.at("key").get_string(); // Rows are addressed by a hash of the key, so the full key is checked here to make a // collision a miss rather than a wrong kernel. - if(e.key != key) + if(row.at("key").get_string() != key) { - log::warn() << "Ignoring binary cache entry with mismatched key: " << md5(key); + log::warn() << "Ignoring binary cache entry with mismatched key: " << key_hash; return nullopt; } + binary_cache_entry e; + e.key = key; e.op_name = row.at("op_name").get_string(); e.problem = from_json_string(row.at("problem").get_string()); e.solution = from_json_string(row.at("solution").get_string()); const auto& code = row.at("code").get_binary(); - migraphx::from_value(from_msgpack(std::vector(code.begin(), code.end())), e.code); + migraphx::from_value(from_msgpack(reinterpret_cast(code.data()), code.size()), + e.code); return e; } catch(const std::exception& ex) { // A cache that cannot be read, or a damaged row, is a miss, which costs a recompile and // nothing else. - log::warn() << "Ignoring unreadable binary cache entry " << md5(key) << ": " << ex.what(); + log::warn() << "Ignoring unreadable binary cache entry " << key_hash << ": " << ex.what(); return nullopt; } } @@ -188,8 +190,9 @@ void sqlite_binary_cache::store(const std::string& version, const std::string& device, const std::vector& entries) { - // Not prepared for a read-only database, whose stores are skipped. - if(not store_stmt.valid() or entries.empty()) + // Never prepared for a read-only database, and cleared after a failed store; either way + // stores are skipped. + if(not store_stmt.valid()) return; try { diff --git a/test/gpu/binary_cache.cpp b/test/gpu/binary_cache.cpp index 0aad959d9e7..a380bf26647 100644 --- a/test/gpu/binary_cache.cpp +++ b/test/gpu/binary_cache.cpp @@ -186,7 +186,7 @@ struct database_backend { static std::string path(const migraphx::tmp_dir& td) { return db_path(td); } static std::size_t stored(const std::string& p) { return row_count(p, "cache_v1"); } - /// Overwrite every stored entry with bytes that do not decode. + /// Overwrite every stored entry's code with bytes that do not decode. static void damage(const std::string& p) { // 0xc1 is never used in msgpack, so the code cannot be decoded. @@ -195,7 +195,8 @@ struct database_backend }; /// One of each backend over fresh storage in td: a directory at td/files and a database at -/// db_path(td). Driven directly, these skip binary_cache and its version and device strings. +/// db_path(td). Driven directly, bypassing binary_cache, so callers pass their own version and +/// device strings. static std::vector both_backends(const migraphx::tmp_dir& td) { std::vector result; @@ -488,7 +489,7 @@ TEST_CASE(incompatible_schema_degrades_to_memory) EXPECT(row_count(path, "cache_v1") == 0); } -// Both backends address an entry by the same key hash, and give back what they were given. +// Both backends address an entry by the same key hash. TEST_CASE(backends_store_the_same_entry) { migraphx::gpu::context ctx; @@ -508,18 +509,9 @@ TEST_CASE(backends_store_the_same_entry) migraphx::gpu::binary_cache db_cache{migraphx::gpu::binary_cache_settings{path, false}}; db_cache.insert(ctx, {e}); EXPECT((db_entries(path) == from_dir)); - - migraphx::tmp_dir td{"binary-cache"}; - for(auto& backend : both_backends(td)) - { - backend.store("v", "dev", {e}); - auto got = backend.load("v", "dev", e.key); - EXPECT(got.has_value()); - EXPECT(same_entry(*got, e)); - } } -// A whole compile against each backend has to leave the same kernels behind, under the same key +// A whole compile against each backend has to leave the same keys behind, under the same key // hashes. That makes the choice of backend purely a storage decision. TEST_CASE(backends_hold_the_same_entries_after_a_compile) { @@ -535,7 +527,6 @@ TEST_CASE(backends_hold_the_same_entries_after_a_compile) auto from_dir = dir_entries(dir_td.path); auto from_db = db_entries(path); EXPECT(not from_dir.empty()); - EXPECT(from_dir.size() == from_db.size()); EXPECT((from_dir == from_db)); } @@ -630,9 +621,9 @@ TEST_CASE(two_connections_share_a_database) EXPECT(not a->load("v", "dev", "absent").has_value()); } -// A table of hashes says nothing about which build wrote it, so each row records the full -// version id. A database has no path length to protect, unlike the directory backend, which -// names its directories with the short one. +// A database has no directory to name the build that wrote a row, so each row records the +// version id, and in full, since unlike the directory backend's directory names it has no path +// length to protect. TEST_CASE(sqlite_records_the_full_version_id) { migraphx::tmp_dir td{"binary-cache"}; @@ -859,11 +850,7 @@ TEST_CASE(entry_round_trip) migraphx::gpu::binary_cache::entry loaded; migraphx::from_value(migraphx::from_msgpack(buffer), loaded); - EXPECT(loaded.key == e.key); - EXPECT(loaded.op_name == e.op_name); - EXPECT(loaded.solution == e.solution); - EXPECT(loaded.code.fill_map == e.code.fill_map); - EXPECT(*loaded.code.fragment.get_main_module() == *e.code.fragment.get_main_module()); + EXPECT(same_entry(loaded, e)); } // The key has to cover everything handed to the compiler, not just the source text. Two diff --git a/tools/include/gpu/binary_cache_backend.hpp b/tools/include/gpu/binary_cache_backend.hpp index c2d3d5fb21b..c5d0f93bf20 100644 --- a/tools/include/gpu/binary_cache_backend.hpp +++ b/tools/include/gpu/binary_cache_backend.hpp @@ -70,8 +70,9 @@ namespace gpu { /// Entries are addressed by their key, scoped by two strings the caller has /// already computed: /// -/// * `version` -- binary_cache::version_id(), identifying the toolchain and -/// the embedded kernel sources that produced the entry. Never empty; the +/// * `version` -- binary_cache::version_id(), short for the directory +/// backend and full for the database, identifying the toolchain and the +/// embedded kernel sources that produced the entry. Never empty; the /// caller skips persistence entirely when it is. /// * `device` -- the GPU the entry was compiled for. /// From a641728bc8b279a4adddf4509445e20e36c8a140 Mon Sep 17 00:00:00 2001 From: pnikolic-amd Date: Thu, 1 Oct 2026 16:04:05 +0200 Subject: [PATCH 14/14] Misspell resolve --- src/targets/gpu/sqlite_binary_cache.cpp | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/targets/gpu/sqlite_binary_cache.cpp b/src/targets/gpu/sqlite_binary_cache.cpp index 56d8614fa1a..7bf302713d1 100644 --- a/src/targets/gpu/sqlite_binary_cache.cpp +++ b/src/targets/gpu/sqlite_binary_cache.cpp @@ -79,7 +79,7 @@ CREATE TABLE IF NOT EXISTS cache_v1 ( constexpr const char* get_sql = "SELECT key, op_name, problem, solution, code FROM cache_v1" " WHERE version = ?1 AND device = ?2 AND key_hash = ?3;"; -// INSERT OR REPLACE is the analogue of the file backend's publish-by-rename: the content is +// INSERT OR REPLACE is the analog of the file backend's publish-by-rename: the content is // decided entirely by the key, so two processes compiling the same kernel is benign and the // last writer wins with an equivalent row. The timestamp is computed by the database rather // than the process so that rows written by different machines stay comparable. MIGraphX never