diff --git a/src/gpl/src/nesterovBase.cpp b/src/gpl/src/nesterovBase.cpp index 3ddd7b4ee60..494f9445853 100644 --- a/src/gpl/src/nesterovBase.cpp +++ b/src/gpl/src/nesterovBase.cpp @@ -25,6 +25,7 @@ #include "backendContext.h" #include "boost/polygon/polygon.hpp" #include "boost/random/normal_distribution.hpp" +#include "boost/random/uniform_int_distribution.hpp" #include "densityGradientBackend.h" #include "fft.h" #include "gpl/Replace.h" @@ -3809,6 +3810,13 @@ void NesterovBase::commitCoordsToDeviceState(SlpSlot source) #endif } +void NesterovBase::updateDensityPenaltyFromRatio(float factor) +{ + densityPenalty_ = (densityGradSum_ != 0) + ? (wireLengthGradSum_ / densityGradSum_) * factor + : factor; +} + float NesterovBase::initDensity2(float wlCoeffX, float wlCoeffY) { if (wireLengthGradSum_ == 0) { @@ -3817,8 +3825,7 @@ float NesterovBase::initDensity2(float wlCoeffX, float wlCoeffY) } if (wireLengthGradSum_ != 0) { - densityPenalty_ - = (wireLengthGradSum_ / densityGradSum_) * npVars_->initDensityPenalty; + updateDensityPenaltyFromRatio(npVars_->initDensityPenalty); } sum_overflow_ = static_cast(getOverflowArea()) @@ -5373,6 +5380,168 @@ void NesterovBase::restoreRemovedFillers() rebuildNbDeviceCtx(); } +void NesterovBase::redistributeFillerCells() +{ + // Reads and overwrites the per-filler host vectors below directly; they + // must be fresh first. Mirrors the very first line of cutFillerCells(). + pullCoordsFromDevice(); + + if (fillerStor_.empty()) { + return; + } + + dbBlock* block = pb_->db()->getChip()->getBlock(); + + // Free capacity per bin (filler area is excluded) + std::vector& bins = getBins(); + std::vector free_capacity(bins.size(), 0.0); + double total_free_capacity = 0.0; + for (size_t b = 0; b < bins.size(); ++b) { + const Bin& bin = bins[b]; + const double scaled_bin_area + = static_cast(bin.getBinArea()) * bin.getTargetDensity(); + const double free = scaled_bin_area + - static_cast(bin.instPlacedArea()) + - static_cast(bin.getNonPlaceArea()); + free_capacity[b] = std::max(0.0, free); + total_free_capacity += free_capacity[b]; + } + + const size_t num_fillers = fillerStor_.size(); + const double free_capacity_um2 + = block->dbuAreaToMicrons(static_cast(total_free_capacity)); + const double filler_area_um2 = block->dbuAreaToMicrons(totalFillerArea_); + if (total_free_capacity <= 0.0 + || total_free_capacity < static_cast(totalFillerArea_)) { + log_->warn(GPL, + 332, + "Not enough free bin capacity ({:.3f} um^2) to redistribute " + "{} filler cells ({:.3f} um^2); leaving filler positions " + "unchanged.", + free_capacity_um2, + num_fillers, + filler_area_um2); + return; + } + + // Give every bin an integer filler quota proportional to its free + // capacity (largest-remainder method, so the quotas sum to exactly + // num_fillers instead of drifting from independent rounding). + std::vector quota(bins.size(), 0); + std::vector remainder(bins.size(), 0.0); + size_t assigned = 0; + for (size_t b = 0; b < bins.size(); ++b) { + const double exact = free_capacity[b] / total_free_capacity + * static_cast(num_fillers); + quota[b] = static_cast(exact); + remainder[b] = exact - static_cast(quota[b]); + assigned += quota[b]; + } + assigned = std::min(assigned, num_fillers); + const size_t leftover = num_fillers - assigned; + if (leftover > 0) { + std::vector order; + order.reserve(bins.size()); + for (size_t b = 0; b < bins.size(); ++b) { + order.push_back(b); + } + std::partial_sort( + order.begin(), + order.begin() + leftover, + order.end(), + [&](size_t a, size_t b) { return remainder[a] > remainder[b]; }); + for (size_t k = 0; k < leftover; ++k) { + quota[order[k]]++; + } + } + + // Walk the fillers once, handing them out to bins in quota order. + size_t bin_idx = 0; + size_t remaining_in_bin = quota[0]; + size_t repositioned = 0; + for (size_t i = 0; i < nb_gcells_.size(); ++i) { + if (!nb_gcells_[i]->isFiller()) { + continue; + } + while (remaining_in_bin == 0 && bin_idx + 1 < bins.size()) { + ++bin_idx; + remaining_in_bin = quota[bin_idx]; + } + if (remaining_in_bin == 0) { + // Quotas sum to num_fillers, so this should not happen; leave any + // stragglers where they are rather than crash. + break; + } + + const Bin& bin = bins[bin_idx]; + GCell* gcell = nb_gcells_[i]; + const int half_dx = gcell->dx() / 2; + const int half_dy = gcell->dy() / 2; + + // Jitter within the bin instead of always landing on its exact center + const int range_x = std::max(0, bin.dx() / 2 - half_dx); + const int range_y = std::max(0, bin.dy() / 2 - half_dy); + int jitter_x = 0; + int jitter_y = 0; + if (range_x > 0) { + jitter_x = boost::random::uniform_int_distribution( + -range_x, range_x)(generator_); + } + if (range_y > 0) { + jitter_y = boost::random::uniform_int_distribution( + -range_y, range_y)(generator_); + } + const int cx = bin.cx() + jitter_x; + const int cy = bin.cy() + jitter_y; + gcell->setCenterLocation(cx, cy); + + // Update location in NesterovBase + const FloatPoint pos(static_cast(cx), static_cast(cy)); + curCoordi_[i] = pos; + curSLPCoordi_[i] = pos; + prevSLPCoordi_[i] = pos; + nextCoordi_[i] = pos; + nextSLPCoordi_[i] = pos; + initCoordi_[i] = pos; + snapshotCoordi_[i] = pos; + snapshotSLPCoordi_[i] = pos; + + // Reset the gradients + curSLPWireLengthGrads_[i] = FloatPoint(); + curSLPDensityGrads_[i] = FloatPoint(); + curSLPSumGrads_[i] = FloatPoint(); + + --remaining_in_bin; + ++repositioned; + } + updateGCellDensityCenterLocation(curCoordi_); + updateDensityFieldBin(); + +#ifdef ENABLE_GPU + // The loop above wrote the new filler positions to the host vectors only; + // push them to the device context (built earlier, before this call, by + // Replace::initNesterovPlace) so the Nesterov loop does not start from the + // pre-redistribution filler positions. Mirrors revertToSnapshot(). + if (nb_device_ctx_) { + nb_device_ctx_->syncCoordsToDevice(curSLPCoordi_, + prevSLPCoordi_, + curCoordi_, + curSLPSumGrads_, + prevSLPSumGrads_); + commitCoordsToDeviceState(SlpSlot::Cur); + host_coords_fresh_ = true; + } +#endif + + log_->info(GPL, + 333, + "Redistributed {} filler cells into free bin capacity ({:.3f} " + "um^2 available, {:.3f} um^2 needed).", + repositioned, + free_capacity_um2, + filler_area_um2); +} + void NesterovBaseCommon::destroyCbkGNet(odb::dbNet* db_net) { debugPrint(log_, GPL, "callbacks", 3, "NBC destroyGNet"); diff --git a/src/gpl/src/nesterovBase.h b/src/gpl/src/nesterovBase.h index 927a79037b1..20caf46320e 100644 --- a/src/gpl/src/nesterovBase.h +++ b/src/gpl/src/nesterovBase.h @@ -1096,6 +1096,9 @@ class NesterovBase float getSumOverflowUnscaled() const { return sum_overflow_unscaled_; } float getBaseWireLengthCoef() const { return baseWireLengthCoef_; } float getDensityPenalty() const { return densityPenalty_; } + // Sets densityPenalty_ from the wirelength/density gradient ratio times + // factor + void updateDensityPenaltyFromRatio(float factor); float getWireLengthGradSum() const { return wireLengthGradSum_; } float getDensityGradSum() const { return densityGradSum_; } @@ -1282,6 +1285,9 @@ class NesterovBase void resetMinSumOverflow(); bool isDiverged() const { return isDiverged_; } + // Resets isDiverged_ when no snapshot exists to revert to instead (the only + // other place that clears it is revertToSnapshot()). + void clearDivergence() { isDiverged_ = false; } void createCbkGCell(odb::dbInst* db_inst, size_t stor_index); std::optional> destroyCbkGCell( @@ -1296,6 +1302,9 @@ class NesterovBase void restoreRemovedFillers(); void clearRemovedFillers() { removed_fillers_.clear(); } + // Directly redistributes the existing filler into the free space. + void redistributeFillerCells(); + void appendGCellCSVNote(const std::string& filename, int iteration, const std::string& message) const; diff --git a/src/gpl/src/nesterovPlace.cpp b/src/gpl/src/nesterovPlace.cpp index 64ecb28b952..150416bea0e 100644 --- a/src/gpl/src/nesterovPlace.cpp +++ b/src/gpl/src/nesterovPlace.cpp @@ -583,6 +583,64 @@ void NesterovPlace::runTimingDriven(int iter, } } +void NesterovPlace::enableIncrementalDensityPenaltyGuard() +{ + incremental_penalty_guard_requested_ = true; +} + +void NesterovPlace::clearDivergence() +{ + num_region_diverged_ = 0; + divergeMsg_ = ""; + divergeCode_ = 0; + for (auto& nb : nbVec_) { + nb->clearDivergence(); + } +} + +void NesterovPlace::applyDensityPenaltyFactor(float factor) +{ + for (auto& nb : nbVec_) { + nb->updateDensityPenaltyFromRatio(factor); + } +} + +void NesterovPlace::guardIncrementalDensityPenalty(float& current_factor, + float& best_overflow, + int& retries) +{ + // React on the very first regression (a guard-parameter sweep found no + // patience delay to be best) by escalating the density penalty in place. + constexpr float kOverflowTolerance = 0.005f; + constexpr float kGrowthRatio = 2.0f; + constexpr int kMaxRetries = 10; + + if (average_overflow_unscaled_ < best_overflow - kOverflowTolerance) { + best_overflow = average_overflow_unscaled_; + return; + } + + if (average_overflow_unscaled_ <= best_overflow + kOverflowTolerance) { + return; + } + + if (retries >= kMaxRetries) { + return; + } + + ++retries; + current_factor *= kGrowthRatio; + applyDensityPenaltyFactor(current_factor); + log_->info(GPL, + 193, + "Incremental density-penalty guard: overflow regressed past " + "{:.3f}; escalating the penalty factor to {:g} in place " + "(retry {}).", + best_overflow, + current_factor, + retries); +} + bool NesterovPlace::isDiverged(float& diverge_snapshot_WlCoefX, float& diverge_snapshot_WlCoefY, bool& is_diverge_snapshot_saved) @@ -1079,6 +1137,10 @@ int NesterovPlace::doNesterovPlace(int start_iter) { // if replace diverged in init() function, Nesterov must be skipped. if (num_region_diverged_ > 0) { + if (allow_divergence_recovery_) { + log_->warn(GPL, divergeCode_, divergeMsg_); + return start_iter; + } log_->error(GPL, divergeCode_, divergeMsg_); } @@ -1096,6 +1158,24 @@ int NesterovPlace::doNesterovPlace(int start_iter) // backTracking variable. float curA = 1.0; + // density penalty guard info + bool penalty_guard_active = incremental_penalty_guard_requested_ + && !npVars_.routability_driven_mode; + incremental_penalty_guard_requested_ = false; + float penalty_guard_factor = 1; + float penalty_guard_best_overflow = average_overflow_unscaled_; + int penalty_guard_retries = 0; + + if (penalty_guard_active) { + applyDensityPenaltyFactor(penalty_guard_factor); + log_->info(GPL, + 191, + "Incremental density-penalty guard enabled: penalty factor " + "{:g}, starting overflow {:.3f}.", + penalty_guard_factor, + penalty_guard_best_overflow); + } + int routability_driven_revert_count = 0; int routability_gpl_iter_count_ = 0; int timing_driven_count = 0; @@ -1166,6 +1246,12 @@ int NesterovPlace::doNesterovPlace(int start_iter) updateNextIter(nesterov_iter); + if (penalty_guard_active) { + guardIncrementalDensityPenalty(penalty_guard_factor, + penalty_guard_best_overflow, + penalty_guard_retries); + } + updateIterGraphics(nesterov_iter, reports_dir, routability_driven_dir, @@ -1236,7 +1322,11 @@ int NesterovPlace::doNesterovPlace(int start_iter) updateDb(); if (num_region_diverged_ > 0) { - log_->error(GPL, divergeCode_, divergeMsg_); + if (allow_divergence_recovery_) { + log_->warn(GPL, divergeCode_, divergeMsg_); + } else { + log_->error(GPL, divergeCode_, divergeMsg_); + } } if (graphics_ && graphics_->enabled() && npVars_.debug) { diff --git a/src/gpl/src/nesterovPlace.h b/src/gpl/src/nesterovPlace.h index a9fb6cd9042..bc71deede3f 100644 --- a/src/gpl/src/nesterovPlace.h +++ b/src/gpl/src/nesterovPlace.h @@ -62,9 +62,30 @@ class NesterovPlace float getWireLengthCoefX() const { return wireLengthCoefX_; } float getWireLengthCoefY() const { return wireLengthCoefY_; } NesterovPlaceVars& getNpVars() { return npVars_; } + float getAverageOverflow() const { return average_overflow_unscaled_; } void setTargetOverflow(float overflow) { npVars_.targetOverflow = overflow; } void setMaxIters(int limit) { npVars_.maxNesterovIter = limit; } + // Enables the incremental density-penalty guard for the next + // doNesterovPlace() call; consumed (cleared) at the start of that call. + void enableIncrementalDensityPenaltyGuard(); + // Clears a divergence left over from a previous doNesterovPlace() call + // (e.g. one the caller intends to retry from). Without this, the leftover + // state trips the divergence check at the very start of the next + // doNesterovPlace() call before it does any work. + void clearDivergence(); + // When true, a divergence is reported as a warning and doNesterovPlace() + // returns normally (check divergedLastRun()) instead of logging an ERROR + // and throwing. Defaults to false, so a plain (non-incremental) placement + // run still fails loudly and immediately on divergence, as before. The + // caller is responsible for toggling this back off once the recoverable + // window has passed (e.g. incremental placement's phase 2 has no further + // fallback, so it should not set this). + void setAllowDivergenceRecovery(bool allow) + { + allow_divergence_recovery_ = allow; + } + bool divergedLastRun() const { return num_region_diverged_ > 0; } void npUpdatePrevGradient(const std::shared_ptr& nb); void npUpdateCurGradient(const std::shared_ptr& nb); @@ -120,6 +141,20 @@ class NesterovPlace bool isPlacementSettled() const; bool isConverged(int gpl_iter_count, int routability_gpl_iter_count); + // Re-derives densityPenalty_ for every region via + // NesterovBase::updateDensityPenaltyFromRatio() - the same formula + // NesterovBase::initDensity2() uses at true init, just re-triggered + // mid-run. Leaves wireLengthCoefX_/Y_ untouched, unlike calling init() + // again. + void applyDensityPenaltyFactor(float factor); + // Checked once per outer iteration while the incremental density-penalty + // guard is enabled. On any regression past best_overflow, escalates + // current_factor in place (no revert - see the .cpp for why) and keeps + // running; otherwise a no-op. The escalation multiplier is fixed at the + // best value found by a guard-parameter sweep (see the .cpp). + void guardIncrementalDensityPenalty(float& current_factor, + float& best_overflow, + int& retries); // The top-level (unfenced/full-die) region is always nbVec_[0]. NesterovBase* getTopLevelNB() const; std::string getReportsDir() const; @@ -171,6 +206,7 @@ class NesterovPlace int num_region_diverged_ = 0; bool is_routability_need_ = true; + bool allow_divergence_recovery_ = false; std::string divergeMsg_; int divergeCode_ = 0; @@ -181,6 +217,8 @@ class NesterovPlace int placement_gif_key_ = -1; int routability_gif_key_ = -1; + bool incremental_penalty_guard_requested_ = false; + void init(); void reset(); diff --git a/src/gpl/src/replace.cpp b/src/gpl/src/replace.cpp index ba9f5ec53db..498ce387b02 100644 --- a/src/gpl/src/replace.cpp +++ b/src/gpl/src/replace.cpp @@ -177,6 +177,13 @@ void Replace::doIncrementalPlace(const int threads, const PlaceOptions& options) } } + if (total_placeable_insts_ == 0) { + // initNesterovPlace() would hit this same condition and refuse to build + // np_, so bail out here instead of leaving every later np_ use guarded. + log_->warn(GPL, 139, "No placeable instances - skipping placement."); + return; + } + log_->info(GPL, 154, "Identified {} placed instances", placed_cnt); log_->info(GPL, 155, "Identified {} not placed instances", unplaced_cnt); @@ -189,20 +196,40 @@ void Replace::doIncrementalPlace(const int threads, const PlaceOptions& options) return; } - // Roughly place the unplaced objects (allow more overflow). - // Limit iterations to prevent objects drifting too far or - // non-convergence. + // Phase 1: place the unplaced (new) objects with everything else locked, + // capped at 600 iterations so it can't run away if it fails to converge. PlaceOptions locked_options = options; - locked_options.overflow = std::max(options.overflow, 0.2f); - locked_options.nesterovPlaceMaxIter = 300; + locked_options.nesterovPlaceMaxIter = 600; + + doInitialPlace(threads, locked_options); - // Use uniform density for incremental runs to fill gaps effectively - if (!options.uniformTargetDensityMode) { - locked_options.uniformTargetDensityMode = true; + // Build NesterovBase now (instead of lazily in doNesterovPlace() below) so + // fillers can be redistributed before phase 1 sees its first iteration. + if (initNesterovPlace(locked_options, threads, true)) { + for (auto& nb : nbVec_) { + nb->redistributeFillerCells(); + } } - doInitialPlace(threads, locked_options); - const int iter = doNesterovPlace(threads, locked_options); + // Phase 1 is allowed to diverge and hand off to phase 2 instead of + // aborting the whole incremental run, so doNesterovPlace() must not throw + // (and log an ERROR) on a divergence it is expected to recover from; it + // reports one via divergedLastRun() instead. Any other error (a resizer or + // timing-driven failure, say) still throws and propagates normally - + // nothing here is set up to recover from those. + np_->setAllowDivergenceRecovery(true); + int iter = doNesterovPlace(threads, locked_options); + const bool phase1_diverged = np_->divergedLastRun(); + if (phase1_diverged) { + log_->warn(GPL, + 195, + "Phase 1 of incremental placement diverged before reaching " + "overflow {:.3f}; continuing to phase 2 anyway.", + locked_options.overflow); + np_->clearDivergence(); + } + // Phase 2 has no further fallback, so a divergence there must fail loudly. + np_->setAllowDivergenceRecovery(false); // Finish the overflow resolution from the locked placement log_->info(GPL, 133, "Unlocking all instances"); @@ -210,12 +237,17 @@ void Replace::doIncrementalPlace(const int threads, const PlaceOptions& options) pb->unlockAll(); } - if (options.overflow < locked_options.overflow) { - PlaceOptions final_options = options; - final_options.uniformTargetDensityMode = true; - final_options.initDensityPenaltyFactor = 1; - - doNesterovPlace(threads, final_options, iter + 1); + // Phase 1 may have run out of its iteration budget short of the real + // target even without diverging, so check the actual overflow reached + // rather than trusting the target alone. + const bool phase1_missed_target + = !phase1_diverged && np_->getAverageOverflow() > options.overflow; + + if (phase1_diverged || phase1_missed_target) { + // Enable phase 2's density-penalty controller to ramp the penalty up in + // place whenever overflow regresses, instead of backing off. + np_->enableIncrementalDensityPenaltyGuard(); + doNesterovPlace(threads, options, iter + 1); } } diff --git a/src/gpl/test/BUILD b/src/gpl/test/BUILD index 7debc84fe2e..b98e5ed18f7 100644 --- a/src/gpl/test/BUILD +++ b/src/gpl/test/BUILD @@ -56,6 +56,7 @@ TESTS = [ PASSFAIL_TESTS = [ "incremental02", + "incremental03", "place_ios01", "place_ios_few_cells", "place_ios_top_layer", @@ -311,6 +312,18 @@ cc_test( ], ) +cc_test( + name = "nesterov_divergence_test", + srcs = ["nesterov_divergence_test.cpp"], + deps = [ + "//src/gpl:gpl_impl", + "//src/odb/src/db", + "//src/utl", + "@googletest//:gtest", + "@googletest//:gtest_main", + ], +) + py_test( name = "gpl_man_tcl_check", srcs = ["gpl_man_tcl_check.py"], diff --git a/src/gpl/test/CMakeLists.txt b/src/gpl/test/CMakeLists.txt index fd95b6d3be6..645d3711154 100644 --- a/src/gpl/test/CMakeLists.txt +++ b/src/gpl/test/CMakeLists.txt @@ -48,6 +48,7 @@ or_integration_tests( region01 PASSFAIL_TESTS incremental02 + incremental03 place_ios01 place_ios_few_cells place_ios_top_layer @@ -218,8 +219,32 @@ if(ENABLE_GPU) COMMAND $ --gtest_list_tests) endif() +add_executable(nesterov_divergence_test + nesterov_divergence_test.cpp +) + +target_include_directories(nesterov_divergence_test + PRIVATE + ${PROJECT_SOURCE_DIR} +) + +target_link_libraries(nesterov_divergence_test PUBLIC + GTest::gtest + GTest::gtest_main + gpl_lib +) + +gtest_discover_tests(nesterov_divergence_test + WORKING_DIRECTORY ${CMAKE_CURRENT_SOURCE_DIR} + ${gpl_gpu_test_discovery} +) +if(ENABLE_GPU) + add_test(NAME nesterov_divergence_test_load_sentinel + COMMAND $ --gtest_list_tests) +endif() + add_dependencies(build_and_test fft_test mbff_test placer_base_test - estimate_target_density_test) + estimate_target_density_test nesterov_divergence_test) # GPU FFT correctness test. Built only on ENABLE_GPU=ON: it links the GPU FFT # backend (src/gpl/src/gpu/gpuFftBackend.cpp) via gpl_lib and, with the default diff --git a/src/gpl/test/incremental03.def b/src/gpl/test/incremental03.def new file mode 120000 index 00000000000..95b37cfa23e --- /dev/null +++ b/src/gpl/test/incremental03.def @@ -0,0 +1 @@ +./incremental02.def \ No newline at end of file diff --git a/src/gpl/test/incremental03.tcl b/src/gpl/test/incremental03.tcl new file mode 100644 index 00000000000..bb9e7300284 --- /dev/null +++ b/src/gpl/test/incremental03.tcl @@ -0,0 +1,45 @@ +# Phase 1 of incremental placement is expected to recover from a divergence +# instead of aborting the whole run (see Replace::doIncrementalPlace()): it +# logs a warning and hands off to phase 2 rather than letting the ERROR +# Logger::error() would normally raise propagate. -max_phi_coef pushes the +# density penalty escalation hard enough to diverge for real (GPL-0305) +# within phase 1's 600-iteration cap, instead of needing thousands of +# iterations to overflow a float on a design that would otherwise converge. +# +# This is a passfail test, not a log_compare one: the scenario drives the +# solver into a numerically chaotic regime by design (that is the whole +# point), so the exact iteration-by-iteration HPWL/overflow numbers are not +# bit-reproducible across compilers/build systems (observed to differ +# between the CMake and Bazel builds here) even though the qualitative log +# messages this test actually checks for are. with_output_to_variable +# captures the run's utl::Logger output so those messages can be checked by +# fixed substring, independent of the numbers around them. +# +# incremental03.def is a symlink to incremental02.def (same "some gates +# unplaced to simulate rmp" aes design). + +source helpers.tcl +set test_name incremental03 +read_lef ./nangate45.lef +read_def ./$test_name.def + +set_thread_count 4 +with_output_to_variable log_text { + global_placement -incremental -density 0.3 -pad_left 2 -pad_right 2 \ + -max_phi_coef 50 +} +puts $log_text + +if { [string match {*\[ERROR GPL-0305\]*} $log_text] } { + error "GPL-0305 (phase 1's divergence) was logged as an ERROR instead of\ + a WARNING: the recoverable-divergence signaling regressed" +} +if { ![string match {*\[WARNING GPL-0305\]*} $log_text] } { + error "Expected phase 1 to actually diverge (GPL-0305 as a WARNING); it\ + did not, so this test is not exercising the recovery path" +} +if { ![string match {*\[WARNING GPL-0195\]*} $log_text] } { + error "Expected GPL-0195 (phase 1 diverged, continuing to phase 2 anyway)" +} + +puts pass diff --git a/src/gpl/test/nesterov_divergence_test.cpp b/src/gpl/test/nesterov_divergence_test.cpp new file mode 100644 index 00000000000..6e8cf3029fe --- /dev/null +++ b/src/gpl/test/nesterov_divergence_test.cpp @@ -0,0 +1,164 @@ +// SPDX-License-Identifier: BSD-3-Clause +// Copyright (c) 2026, The OpenROAD Authors + +#include +#include +#include + +#include "gpl/Replace.h" +#include "gtest/gtest.h" +#include "odb/db.h" +#include "odb/dbTypes.h" +#include "odb/geom.h" +#include "src/gpl/src/graphicsNone.h" +#include "src/gpl/src/nesterovBase.h" +#include "src/gpl/src/nesterovPlace.h" +#include "src/gpl/src/placerBase.h" +#include "utl/Logger.h" +#include "utl/deleter.h" + +namespace gpl { +namespace { + +// Builds the same PlacerBase/NesterovBase/NesterovPlace object graph +// Replace::initNesterovPlace wires up, minus odb/sta/rsz, with every movable +// cell stacked on a single site and an unreachable overflow target. That +// local congestion can never be relieved, so densityPenalty_ (which the main +// loop multiplies by phiCoef every iteration, nesterovBase.cpp:4368) keeps +// escalating until it pushes the wirelength/density gradients to Inf/NaN - +// the same genuine numerical divergence (GPL-305/306) a pathological +// incremental-placement input can hit in phase 1. This mirrors +// Replace::doIncrementalPlace's own locked_options.nesterovPlaceMaxIter cap +// (replace.cpp:196), just uncapped here so the test does not depend on +// exactly how many iterations the escalation takes to overflow a float. +class NesterovDivergenceRecoveryTest : public ::testing::Test +{ + protected: + static constexpr int kDbuPerMicron = 1000; + static constexpr int kSiteWidth = 200; + static constexpr int kRowHeight = 2000; + static constexpr int kRowSites = 100; + static constexpr int kRows = 10; + static constexpr int kCellWidth = 2 * kSiteWidth; + static constexpr int kCellCount = 80; + + void SetUp() override + { + db_ = utl::UniquePtrWithDeleter(odb::dbDatabase::create(), + odb::dbDatabase::destroy); + db_->setLogger(&logger_); + db_->setDbuPerMicron(kDbuPerMicron); + + odb::dbTech* tech = odb::dbTech::create(db_.get(), "tech"); + odb::dbTechLayer::create(tech, "metal1", odb::dbTechLayerType::ROUTING); + odb::dbLib* lib = odb::dbLib::create(db_.get(), "lib", tech); + odb::dbSite* site = odb::dbSite::create(lib, "site"); + site->setWidth(kSiteWidth); + site->setHeight(kRowHeight); + site->setClass(odb::dbSiteClass::CORE); + + odb::dbChip* chip = odb::dbChip::create(db_.get(), tech); + odb::dbBlock* block = odb::dbBlock::create(chip, "top"); + block->setDieArea(odb::Rect( + 0, 0, (kRowSites + 2) * kSiteWidth, (kRows + 2) * kRowHeight)); + for (int row = 0; row < kRows; ++row) { + odb::dbRow::create(block, + ("row" + std::to_string(row)).c_str(), + site, + kSiteWidth, + (row + 1) * kRowHeight, + odb::dbOrientType::R0, + odb::dbRowDir::HORIZONTAL, + kRowSites, + kSiteWidth); + } + block->setCoreArea(block->computeCoreArea()); + + odb::dbMaster* master = odb::dbMaster::create(lib, "core_cell"); + master->setType(odb::dbMasterType(odb::dbMasterType::CORE)); + master->setWidth(kCellWidth); + master->setHeight(kRowHeight); + master->setSite(site); + master->setFrozen(); + + // Every cell stacked on the same site: the bin(s) covering it hold + // kCellCount cells' worth of area, far more than any achievable density + // can relieve, so the overflow target below can never be satisfied. + for (int i = 0; i < kCellCount; ++i) { + odb::dbInst* inst = odb::dbInst::create( + block, master, ("u_" + std::to_string(i)).c_str()); + inst->setLocation(kSiteWidth, kRowHeight); + inst->setPlacementStatus(odb::dbPlacementStatus::PLACED); + } + + PlaceOptions options; + options.density = 0.99f; + options.overflow = 1e-6f; // unreachable given the clustered layout above + options.nesterovPlaceMaxIter = 20000; + + const PlacerBaseVars pb_vars(options); + pbc_ = std::make_shared(db_.get(), pb_vars, &logger_); + pbVec_.push_back(std::make_shared( + db_.get(), pbc_, &logger_, /* check_density = */ false)); + + const NesterovBaseVars nb_vars(options); + nbc_ = std::make_shared( + nb_vars, pbc_, &logger_, /* num_threads = */ 1, Clusters{}); + nbVec_.push_back( + std::make_shared(nb_vars, pbVec_[0], nbc_, &logger_)); + + const NesterovPlaceVars np_vars(options, nbc_->getHpwl()); + np_ = std::make_unique(np_vars, + pbc_, + nbc_, + pbVec_, + nbVec_, + /* rb = */ nullptr, + /* tb = */ nullptr, + /* cb = */ nullptr, + std::make_unique(), + &logger_); + } + + utl::Logger logger_; + utl::UniquePtrWithDeleter db_; + std::shared_ptr pbc_; + std::shared_ptr nbc_; + std::vector> pbVec_; + std::vector> nbVec_; + std::unique_ptr np_; +}; + +// Reproduces the incremental-placement recovery bug with a real numerical +// divergence instead of a hand-set flag: when phase 1 diverges, +// Replace::doIncrementalPlace catches the exception and calls +// NesterovPlace::clearDivergence() so phase 2 gets a clean start. +// clearDivergence() only resets NesterovPlace's own +// num_region_diverged_/divergeMsg_/divergeCode_ (nesterovPlace.cpp:591-596); +// it never touches NesterovBase::isDiverged_ on the region that actually +// diverged. The getter nb->isDiverged() just returns that stale flag +// (nesterovBase.h), which doBackTracking() reads unconditionally on every +// call (nesterovPlace.cpp:1012/1026), so phase 2's very first iteration +// would see "diverged" again before it has run any gradient computation of +// its own, and NesterovPlace::doNesterovPlace() would re-throw immediately - +// exactly the "continuing to phase 2 anyway" recovery not actually +// recovering. +TEST_F(NesterovDivergenceRecoveryTest, ClearDivergenceLeavesPerRegionFlagSet) +{ + EXPECT_THROW(np_->doNesterovPlace(), std::runtime_error); + + NesterovBase& nb = *nbVec_[0]; + ASSERT_TRUE(nb.isDiverged()) + << "test setup must drive a real numerical divergence"; + + np_->clearDivergence(); + + EXPECT_FALSE(nb.isDiverged()) + << "clearDivergence() is supposed to give phase 2 a clean start, but " + "NesterovBase::isDiverged_ is still set on the region that " + "diverged in phase 1, so phase 2 will see \"divergence\" again " + "before it runs a single iteration of its own."; +} + +} // namespace +} // namespace gpl diff --git a/src/odb/test/replace_hier_mod1.ok b/src/odb/test/replace_hier_mod1.ok index 9d3c1676930..4a8e38225a1 100644 --- a/src/odb/test/replace_hier_mod1.ok +++ b/src/odb/test/replace_hier_mod1.ok @@ -193,6 +193,7 @@ Repair timing output passed/skipped equivalence test [INFO GPL-0028] Bin count (X, Y): 2 , 32 [INFO GPL-0029] Bin size (W * H): 19.950 * 37.494 um [INFO GPL-0030] Number of bins: 64 +[INFO GPL-0333] Redistributed 185 filler cells into free bin capacity (500.630 um^2 available, 478.720 um^2 needed). [INFO GPL-0084] ---- Execute Nesterov Global Placement. [INFO GPL-0031] HPWL: Half-Perimeter Wirelength Iteration | Overflow | HPWL (um) | HPWL(%) | Penalty | Group @@ -206,37 +207,68 @@ Iteration | Overflow | HPWL (um) | HPWL(%) | Penalty | Group 60 | 0.4016 | 5.584150e+01 | -0.00% | 1.47e-12 | 70 | 0.4016 | 5.583900e+01 | -0.00% | 2.39e-12 | 80 | 0.4016 | 5.583650e+01 | -0.00% | 3.89e-12 | - 90 | 0.4016 | 5.583450e+01 | -0.00% | 6.34e-12 | - 100 | 0.4016 | 5.585450e+01 | +0.04% | 1.03e-11 | - 110 | 0.4016 | 5.655750e+01 | +1.26% | 1.68e-11 | - 120 | 0.4016 | 5.744200e+01 | +1.56% | 2.74e-11 | - 130 | 0.4016 | 5.862450e+01 | +2.06% | 4.46e-11 | - 140 | 0.4016 | 6.089650e+01 | +3.88% | 7.27e-11 | - 150 | 0.4016 | 6.454650e+01 | +5.99% | 1.18e-10 | - 160 | 0.4016 | 6.972550e+01 | +8.02% | 1.93e-10 | - 170 | 0.4016 | 7.625950e+01 | +9.37% | 3.14e-10 | - 180 | 0.4016 | 8.237550e+01 | +8.02% | 5.12e-10 | - 190 | 0.4016 | 8.416050e+01 | +2.17% | 8.34e-10 | - 200 | 0.4016 | 7.670450e+01 | -8.86% | 1.36e-09 | - 210 | 0.4016 | 8.335050e+01 | +8.66% | 2.21e-09 | - 220 | 0.4016 | 8.611450e+01 | +3.32% | 3.60e-09 | - 230 | 0.4016 | 8.534350e+01 | -0.90% | 5.87e-09 | - 240 | 0.4016 | 8.311100e+01 | -2.62% | 9.56e-09 | - 250 | 0.4016 | 7.333650e+01 | -11.76% | 1.56e-08 | - 260 | 0.4015 | 6.638600e+01 | -9.48% | 2.54e-08 | - 270 | 0.4011 | 5.804750e+01 | -12.56% | 4.13e-08 | - 280 | 0.4007 | 6.055900e+01 | +4.33% | 6.73e-08 | - 290 | 0.3999 | 6.497400e+01 | +7.29% | 1.10e-07 | -[WARNING GPL-1010] GPL reached the maximum number of iterations for nesterov 300. Placement may have failed to converge. + 90 | 0.4016 | 5.583400e+01 | -0.00% | 6.34e-12 | + 100 | 0.4016 | 5.583150e+01 | -0.00% | 1.03e-11 | + 110 | 0.4016 | 5.582900e+01 | -0.00% | 1.68e-11 | + 120 | 0.4016 | 5.582750e+01 | -0.00% | 2.74e-11 | + 130 | 0.4016 | 5.605300e+01 | +0.40% | 4.46e-11 | + 140 | 0.4016 | 5.822100e+01 | +3.87% | 7.27e-11 | + 150 | 0.4016 | 5.991600e+01 | +2.91% | 1.18e-10 | + 160 | 0.4016 | 6.225900e+01 | +3.91% | 1.93e-10 | + 170 | 0.4016 | 6.499200e+01 | +4.39% | 3.14e-10 | + 180 | 0.4016 | 6.703500e+01 | +3.14% | 5.12e-10 | + 190 | 0.4016 | 6.639200e+01 | -0.96% | 8.34e-10 | + 200 | 0.4016 | 6.070000e+01 | -8.57% | 1.36e-09 | + 210 | 0.4016 | 6.524100e+01 | +7.48% | 2.21e-09 | + 220 | 0.4016 | 6.728800e+01 | +3.14% | 3.60e-09 | + 230 | 0.4016 | 6.705600e+01 | -0.34% | 5.87e-09 | + 240 | 0.4016 | 6.123600e+01 | -8.68% | 9.56e-09 | + 250 | 0.4016 | 5.960550e+01 | -2.66% | 1.56e-08 | + 260 | 0.4014 | 5.635450e+01 | -5.45% | 2.54e-08 | + 270 | 0.4011 | 5.736250e+01 | +1.79% | 4.13e-08 | + 280 | 0.4007 | 5.910350e+01 | +3.04% | 6.73e-08 | + 290 | 0.3999 | 6.208050e+01 | +5.04% | 1.10e-07 | + 300 | 0.3986 | 6.707050e+01 | +8.04% | 1.79e-07 | + 310 | 0.3967 | 7.507150e+01 | +11.93% | 2.91e-07 | + 320 | 0.3939 | 8.753050e+01 | +16.60% | 4.74e-07 | + 330 | 0.3939 | 9.714050e+01 | +10.98% | 7.72e-07 | + 340 | 0.3934 | 1.235405e+02 | +27.18% | 1.26e-06 | + 350 | 0.3896 | 1.768265e+02 | +43.13% | 2.05e-06 | + 360 | 0.3883 | 1.997405e+02 | +12.96% | 3.34e-06 | + 370 | 0.3886 | 1.940495e+02 | -2.85% | 5.43e-06 | + 380 | 0.3882 | 2.020225e+02 | +4.11% | 8.85e-06 | + 390 | 0.3878 | 2.233625e+02 | +10.56% | 1.44e-05 | + 400 | 0.3878 | 2.436285e+02 | +9.07% | 2.35e-05 | + 410 | 0.3878 | 2.552845e+02 | +4.78% | 3.82e-05 | + 420 | 0.3878 | 2.596015e+02 | +1.69% | 6.23e-05 | + 430 | 0.3878 | 2.597185e+02 | +0.05% | 1.01e-04 | + 440 | 0.3878 | 2.573825e+02 | -0.90% | 1.65e-04 | + 450 | 0.3878 | 2.540795e+02 | -1.28% | 2.69e-04 | + 460 | 0.3878 | 2.506015e+02 | -1.37% | 4.39e-04 | + 470 | 0.3878 | 2.480015e+02 | -1.04% | 7.14e-04 | + 480 | 0.3878 | 2.476175e+02 | -0.15% | 1.16e-03 | + 490 | 0.3878 | 2.495955e+02 | +0.80% | 1.90e-03 | + 500 | 0.3878 | 2.531295e+02 | +1.42% | 3.09e-03 | + 510 | 0.3878 | 2.571985e+02 | +1.61% | 5.03e-03 | + 520 | 0.3878 | 2.610375e+02 | +1.49% | 8.19e-03 | + 530 | 0.3878 | 2.641395e+02 | +1.19% | 1.33e-02 | + 540 | 0.3878 | 2.657625e+02 | +0.61% | 2.17e-02 | + 550 | 0.3878 | 2.654585e+02 | -0.11% | 3.54e-02 | + 560 | 0.3878 | 2.635865e+02 | -0.71% | 5.77e-02 | + 570 | 0.3878 | 2.607575e+02 | -1.07% | 9.39e-02 | + 580 | 0.3878 | 2.576205e+02 | -1.20% | 1.53e-01 | + 590 | 0.3878 | 2.549795e+02 | -1.03% | 2.49e-01 | +[WARNING GPL-1010] GPL reached the maximum number of iterations for nesterov 600. Placement may have failed to converge. [INFO GPL-1014] Final placement area: 54.55 (+0.00%) -[INFO GPL-1018] Final HPWL (um): 71.41 +[INFO GPL-1018] Final HPWL (um): 253.24 [INFO GPL-0133] Unlocking all instances [INFO GPL-0084] ---- Execute Nesterov Global Placement. - 310 | 0.2414 | 1.417410e+02 | +118.15% | 2.74e-07 | - 320 | 0.1927 | 1.395945e+02 | -1.51% | 4.04e-07 | - 329 | 0.0995 | 1.695535e+02 | | 5.95e-07 | +[INFO GPL-0191] Incremental density-penalty guard enabled: penalty factor 1, starting overflow 0.388. + 610 | 0.2955 | 1.350575e+02 | -47.03% | 1.66e-07 | + 620 | 0.1513 | 1.432355e+02 | +6.06% | 2.44e-07 | + 624 | 0.0953 | 1.481870e+02 | | 2.96e-07 | --------------------------------------------------------------- -[INFO GPL-1001] Global placement finished at iteration 329 +[INFO GPL-1001] Global placement finished at iteration 624 [INFO GPL-1002] Placed Cell Area 54.5513 [INFO GPL-1003] Available Free Area 47872.0200 [INFO GPL-1004] Minimum Feasible Density 0.0100 (cell_area / free_area) @@ -245,7 +277,7 @@ Iteration | Overflow | HPWL (um) | HPWL(%) | Penalty | Group [INFO GPL-1008] - For 80% usage of free space: 0.0014 [INFO GPL-1009] - For 50% usage of free space: 0.0023 [INFO GPL-1014] Final placement area: 54.55 (+0.00%) -[INFO GPL-1018] Final HPWL (um): 169.33 +[INFO GPL-1018] Final HPWL (um): 148.60 [INFO DPL-0006] Core area: 47872.02 um^2 [INFO DPL-0007] Movable instances area: 59.85 um^2 [INFO DPL-0008] Fixed instances area within core: 0.00 um^2 @@ -261,19 +293,19 @@ Iteration | Overflow | HPWL (um) | HPWL(%) | Penalty | Group | Total | Illegal | Illegal Iteration | Violations | Cells | Sites --------------------------------------------- - 0 | 6 | 2 | 4 + 0 | 4 | 2 | 2 1 | 0 | 0 | 0 Negotiation phase 1 converged at iteration 1. Detailed Placement Analysis --------------------------------- legalizer negotiation total moves 20 -total displacement 25.3 u -average displacement 1.2 u -max displacement 2.7 u -original HPWL 169.3 u -legalized HPWL 179.4 u -delta HPWL 6 % +total displacement 31.7 u +average displacement 1.5 u +max displacement 3.4 u +original HPWL 148.6 u +legalized HPWL 160.1 u +delta HPWL 8 % Cell type report for bc1 (inv_chain_bc1) Cell type report: Count Area @@ -288,13 +320,13 @@ Net u1z Number of pins: 5 Driver pins - u1/Z output (BUF_X1) (20, 612) + u1/Z output (BUF_X1) (19, 615) Load pins - bc1/u4/A input (INV_X1) 1.598-1.720 (16, 613) - bc2/u2/A input (BUF_X1) 0.906-0.983 (19, 610) - ic1/u4/A input (INV_X1) 1.598-1.720 (17, 611) - ic2/u4/A input (INV_X1) 1.598-1.720 (17, 614) + bc1/u4/A input (INV_X1) 1.598-1.720 (19, 628) + bc2/u2/A input (BUF_X1) 0.906-0.983 (16, 631) + ic1/u4/A input (INV_X1) 1.598-1.720 (19, 629) + ic2/u4/A input (INV_X1) 1.598-1.720 (19, 629) Net u3z Pin capacitance: 1.096-1.158 @@ -305,10 +337,10 @@ Net u3z Number of pins: 2 Driver pins - bc1/u5/ZN output (INV_X1) (21, 612) + bc1/u5/ZN output (INV_X1) (17, 632) Load pins - r2/D input (DFF_X1) 1.096-1.158 (18, 614) + r2/D input (DFF_X1) 1.096-1.158 (16, 629) Startpoint: r1 (rising edge-triggered flip-flop clocked by clk) Endpoint: r2 (rising edge-triggered flip-flop clocked by clk) @@ -321,24 +353,24 @@ Corner: slow 0.000 0.000 clock clk (rise edge) 0.000 0.000 clock network delay (ideal) 0.000 0.000 ^ r1/CK (DFF_X1) - 0.400 0.400 ^ r1/Q (DFF_X1) - 0.145 0.545 ^ u1/Z (BUF_X1) - 0.031 0.576 v bc1/u4/ZN (INV_X1) - 0.036 0.612 ^ bc1/u5/ZN (INV_X1) - 0.000 0.612 ^ r2/D (DFF_X1) - 0.612 data arrival time + 0.383 0.383 ^ r1/Q (DFF_X1) + 0.146 0.529 ^ u1/Z (BUF_X1) + 0.033 0.562 v bc1/u4/ZN (INV_X1) + 0.036 0.598 ^ bc1/u5/ZN (INV_X1) + 0.000 0.598 ^ r2/D (DFF_X1) + 0.598 data arrival time 0.300 0.300 clock clk (rise edge) 0.000 0.300 clock network delay (ideal) 0.000 0.300 clock reconvergence pessimism 0.300 ^ r2/CK (DFF_X1) - -0.073 0.227 library setup time - 0.227 data required time + -0.072 0.228 library setup time + 0.228 data required time ----------------------------------------------------------- - 0.227 data required time - -0.612 data arrival time + 0.228 data required time + -0.598 data arrival time ----------------------------------------------------------- - -0.384 slack (VIOLATED) + -0.371 slack (VIOLATED) Equivalence check - swap (buffer_chain -> inv_chain) @@ -375,11 +407,11 @@ Repair timing output passed/skipped equivalence test [INFO GPL-0005] ---- Execute Conjugate Gradient Initial Placement. [INFO GPL-0051] Source of initial instance position counters: Odb location = 0 Core center = 2 Region center = 0 -[InitialPlace] Iter: 1 conjugate gradient residual: 0.00000000 HPWL: 2914200 -[InitialPlace] Iter: 2 conjugate gradient residual: 0.00000000 HPWL: 405304 -[InitialPlace] Iter: 3 conjugate gradient residual: 0.00000012 HPWL: 404018 -[InitialPlace] Iter: 4 conjugate gradient residual: 0.00000000 HPWL: 403883 -[InitialPlace] Iter: 5 conjugate gradient residual: 0.00000005 HPWL: 403939 +[InitialPlace] Iter: 1 conjugate gradient residual: 0.00000000 HPWL: 2917810 +[InitialPlace] Iter: 2 conjugate gradient residual: 0.00000000 HPWL: 373816 +[InitialPlace] Iter: 3 conjugate gradient residual: 0.00000000 HPWL: 369271 +[InitialPlace] Iter: 4 conjugate gradient residual: 0.00000000 HPWL: 368500 +[InitialPlace] Iter: 5 conjugate gradient residual: 0.00000001 HPWL: 368684 [INFO GPL-0033] ---- Initialize Nesterov Region: Top-level [INFO GPL-0023] Placement target density: 0.0112 [INFO GPL-0024] Movable insts average area: 2.623 um^2 @@ -389,14 +421,53 @@ Repair timing output passed/skipped equivalence test [INFO GPL-0028] Bin count (X, Y): 2 , 32 [INFO GPL-0029] Bin size (W * H): 19.950 * 37.494 um [INFO GPL-0030] Number of bins: 64 +[INFO GPL-0333] Redistributed 183 filler cells into free bin capacity (484.452 um^2 available, 478.720 um^2 needed). [INFO GPL-0084] ---- Execute Nesterov Global Placement. [INFO GPL-0031] HPWL: Half-Perimeter Wirelength Iteration | Overflow | HPWL (um) | HPWL(%) | Penalty | Group --------------------------------------------------------------- - 0 | 0.1105 | 1.708630e+02 | +0.00% | 6.75e-13 | - 0 | 0.1105 | 1.708630e+02 | | 7.02e-13 | + 0 | 0.1043 | 1.525955e+02 | +0.00% | 5.31e-13 | + 10 | 0.1048 | 1.535715e+02 | +0.64% | 7.82e-13 | + 20 | 0.1048 | 1.536090e+02 | +0.02% | 1.15e-12 | + 30 | 0.1048 | 1.536175e+02 | +0.01% | 1.70e-12 | + 40 | 0.1048 | 1.536220e+02 | +0.00% | 2.50e-12 | + 50 | 0.1048 | 1.536255e+02 | +0.00% | 3.69e-12 | + 60 | 0.1048 | 1.536295e+02 | +0.00% | 5.43e-12 | + 70 | 0.1048 | 1.536325e+02 | +0.00% | 8.00e-12 | + 80 | 0.1048 | 1.536365e+02 | +0.00% | 1.18e-11 | + 90 | 0.1048 | 1.536395e+02 | +0.00% | 1.74e-11 | + 100 | 0.1048 | 1.536435e+02 | +0.00% | 2.56e-11 | + 110 | 0.1048 | 1.536465e+02 | +0.00% | 3.77e-11 | + 120 | 0.1048 | 1.536500e+02 | +0.00% | 5.55e-11 | + 130 | 0.1048 | 1.536530e+02 | +0.00% | 8.17e-11 | + 140 | 0.1048 | 1.536565e+02 | +0.00% | 1.20e-10 | + 150 | 0.1048 | 1.536595e+02 | +0.00% | 1.77e-10 | + 160 | 0.1048 | 1.536630e+02 | +0.00% | 2.61e-10 | + 170 | 0.1048 | 1.536660e+02 | +0.00% | 3.85e-10 | + 180 | 0.1048 | 1.536695e+02 | +0.00% | 5.67e-10 | + 190 | 0.1048 | 1.536590e+02 | -0.01% | 8.35e-10 | + 200 | 0.1048 | 1.547485e+02 | +0.71% | 1.23e-09 | + 210 | 0.1049 | 1.649720e+02 | +6.61% | 1.81e-09 | + 220 | 0.1049 | 1.677145e+02 | +1.66% | 2.67e-09 | + 230 | 0.1049 | 1.629785e+02 | -2.82% | 3.93e-09 | + 240 | 0.1048 | 1.539135e+02 | -5.56% | 5.80e-09 | + 250 | 0.1048 | 1.663955e+02 | +8.11% | 8.54e-09 | + 260 | 0.1049 | 1.651630e+02 | -0.74% | 1.26e-08 | + 270 | 0.1049 | 1.692115e+02 | +2.45% | 1.85e-08 | + 280 | 0.1048 | 1.611910e+02 | -4.74% | 2.73e-08 | + 290 | 0.1047 | 1.612850e+02 | +0.06% | 4.02e-08 | + 300 | 0.1046 | 1.557070e+02 | -3.46% | 5.92e-08 | + 310 | 0.1044 | 1.531520e+02 | -1.64% | 8.73e-08 | + 320 | 0.1043 | 1.527540e+02 | -0.26% | 1.29e-07 | + 330 | 0.1041 | 1.529460e+02 | +0.13% | 1.89e-07 | + 340 | 0.1037 | 1.534960e+02 | +0.36% | 2.79e-07 | + 350 | 0.1032 | 1.542745e+02 | +0.51% | 4.11e-07 | + 360 | 0.1026 | 1.565115e+02 | +1.45% | 6.05e-07 | + 370 | 0.1016 | 1.600920e+02 | +2.29% | 8.92e-07 | + 380 | 0.1002 | 1.656250e+02 | +3.46% | 1.31e-06 | + 382 | 0.0999 | 1.669570e+02 | | 1.48e-06 | --------------------------------------------------------------- -[INFO GPL-1001] Global placement finished at iteration 0 +[INFO GPL-1001] Global placement finished at iteration 382 [INFO GPL-1002] Placed Cell Area 55.0833 [INFO GPL-1003] Available Free Area 47872.0200 [INFO GPL-1004] Minimum Feasible Density 0.0100 (cell_area / free_area) @@ -405,67 +476,8 @@ Iteration | Overflow | HPWL (um) | HPWL(%) | Penalty | Group [INFO GPL-1008] - For 80% usage of free space: 0.0014 [INFO GPL-1009] - For 50% usage of free space: 0.0023 [INFO GPL-1014] Final placement area: 55.08 (+0.00%) -[INFO GPL-1018] Final HPWL (um): 170.78 +[INFO GPL-1018] Final HPWL (um): 167.05 [INFO GPL-0133] Unlocking all instances -[INFO GPL-0084] ---- Execute Nesterov Global Placement. - 10 | 0.3985 | 3.484600e+01 | -79.61% | 9.09e-13 | - 20 | 0.3986 | 2.555950e+01 | -26.65% | 1.21e-12 | - 30 | 0.3983 | 2.267350e+01 | -11.29% | 1.61e-12 | - 40 | 0.3949 | 2.221550e+01 | -2.02% | 2.15e-12 | - 50 | 0.3949 | 2.192350e+01 | -1.31% | 2.86e-12 | - 60 | 0.3949 | 2.196650e+01 | +0.20% | 3.82e-12 | - 70 | 0.3954 | 2.201450e+01 | +0.22% | 5.08e-12 | - 80 | 0.3955 | 2.226400e+01 | +1.13% | 6.77e-12 | - 90 | 0.3951 | 2.275850e+01 | +2.22% | 9.02e-12 | - 100 | 0.3948 | 2.348500e+01 | +3.19% | 1.20e-11 | - 110 | 0.3955 | 2.442400e+01 | +4.00% | 1.60e-11 | - 120 | 0.3962 | 2.561800e+01 | +4.89% | 2.13e-11 | - 130 | 0.3969 | 2.719850e+01 | +6.17% | 2.84e-11 | - 140 | 0.3979 | 2.927500e+01 | +7.63% | 3.79e-11 | - 150 | 0.3993 | 3.200750e+01 | +9.33% | 5.05e-11 | - 160 | 0.4012 | 3.550800e+01 | +10.94% | 6.72e-11 | - 170 | 0.4037 | 3.987450e+01 | +12.30% | 8.96e-11 | - 180 | 0.4067 | 4.511950e+01 | +13.15% | 1.19e-10 | - 190 | 0.4101 | 5.087050e+01 | +12.75% | 1.59e-10 | - 200 | 0.4134 | 5.657250e+01 | +11.21% | 2.12e-10 | - 210 | 0.4163 | 6.095700e+01 | +7.75% | 2.82e-10 | - 220 | 0.4191 | 6.198950e+01 | +1.69% | 3.76e-10 | - 230 | 0.4223 | 5.795050e+01 | -6.52% | 5.01e-10 | - 240 | 0.4252 | 5.324250e+01 | -8.12% | 6.67e-10 | - 250 | 0.4266 | 5.972950e+01 | +12.18% | 8.89e-10 | - 260 | 0.4284 | 6.835500e+01 | +14.44% | 1.18e-09 | - 270 | 0.4326 | 6.468400e+01 | -5.37% | 1.58e-09 | - 280 | 0.4373 | 6.659800e+01 | +2.96% | 2.10e-09 | - 290 | 0.4318 | 6.394150e+01 | -3.99% | 2.80e-09 | - 300 | 0.4280 | 6.945400e+01 | +8.62% | 3.73e-09 | - 310 | 0.4183 | 6.523400e+01 | -6.08% | 4.97e-09 | - 320 | 0.4063 | 6.538350e+01 | +0.23% | 6.62e-09 | - 330 | 0.4041 | 6.422550e+01 | -1.77% | 8.82e-09 | - 340 | 0.4034 | 6.031400e+01 | -6.09% | 1.18e-08 | - 350 | 0.3984 | 6.511350e+01 | +7.96% | 1.57e-08 | - 360 | 0.3903 | 6.812700e+01 | +4.63% | 2.09e-08 | - 370 | 0.3647 | 7.416850e+01 | +8.87% | 2.78e-08 | - 380 | 0.3279 | 7.902550e+01 | +6.55% | 3.70e-08 | - 390 | 0.3093 | 8.282700e+01 | +4.81% | 4.93e-08 | - 400 | 0.2762 | 8.767000e+01 | +5.85% | 6.57e-08 | - 410 | 0.2453 | 9.740150e+01 | +11.10% | 8.76e-08 | - 420 | 0.2201 | 1.049820e+02 | +7.78% | 1.17e-07 | - 430 | 0.1747 | 1.143050e+02 | +8.88% | 1.55e-07 | - 440 | 0.1466 | 1.202775e+02 | +5.23% | 2.07e-07 | - 450 | 0.1243 | 1.259800e+02 | +4.74% | 2.76e-07 | - 460 | 0.1138 | 1.453035e+02 | +15.34% | 3.68e-07 | - 466 | 0.0985 | 1.518525e+02 | | 4.49e-07 | ---------------------------------------------------------------- -[INFO GPL-1001] Global placement finished at iteration 466 -[INFO GPL-1002] Placed Cell Area 55.0833 -[INFO GPL-1003] Available Free Area 47872.0200 -[INFO GPL-1004] Minimum Feasible Density 0.0100 (cell_area / free_area) -[INFO GPL-1006] Suggested Target Densities: -[INFO GPL-1007] - For 90% usage of free space: 0.0013 -[INFO GPL-1008] - For 80% usage of free space: 0.0014 -[INFO GPL-1009] - For 50% usage of free space: 0.0023 -[INFO GPL-1014] Final placement area: 55.08 (+0.00%) -[INFO GPL-1018] Final HPWL (um): 151.81 [INFO DPL-0006] Core area: 47872.02 um^2 [INFO DPL-0007] Movable instances area: 60.38 um^2 [INFO DPL-0008] Fixed instances area within core: 0.00 um^2 @@ -478,22 +490,16 @@ Iteration | Overflow | HPWL (um) | HPWL(%) | Penalty | Group [INFO DPL-1104] NegotiationLegalizer DRC penalty: 5. [INFO DPL-0392] Negotiation cell height distribution (1 unique row-count(s)): [INFO DPL-0393] height 1 row(s): 21 cells - | Total | Illegal | Illegal -Iteration | Violations | Cells | Sites ---------------------------------------------- - 0 | 15 | 3 | 12 - 1 | 0 | 0 | 0 -Negotiation phase 1 converged at iteration 1. Detailed Placement Analysis --------------------------------- legalizer negotiation -total moves 22 -total displacement 34.0 u -average displacement 1.6 u -max displacement 3.7 u -original HPWL 151.8 u -legalized HPWL 166.7 u -delta HPWL 10 % +total moves 0 +total displacement 2.5 u +average displacement 0.1 u +max displacement 0.6 u +original HPWL 167.0 u +legalized HPWL 165.7 u +delta HPWL -1 % Cell type report for bc1 (buffer_chain) Cell type report: Count Area @@ -508,13 +514,13 @@ Net u1z Number of pins: 5 Driver pins - u1/Z output (BUF_X1) (19, 603) + u1/Z output (BUF_X1) (19, 615) Load pins - bc1/u2/A input (BUF_X1) 0.906-0.983 (18, 611) - bc2/u2/A input (BUF_X1) 0.906-0.983 (19, 611) - ic1/u4/A input (INV_X1) 1.598-1.720 (22, 614) - ic2/u4/A input (INV_X1) 1.598-1.720 (16, 610) + bc1/u2/A input (BUF_X1) 0.906-0.983 (19, 636) + bc2/u2/A input (BUF_X1) 0.906-0.983 (16, 631) + ic1/u4/A input (INV_X1) 1.598-1.720 (19, 629) + ic2/u4/A input (INV_X1) 1.598-1.720 (19, 629) Net u3z Pin capacitance: 1.096-1.158 @@ -525,10 +531,10 @@ Net u3z Number of pins: 2 Driver pins - bc1/u3/Z output (BUF_X1) (19, 612) + bc1/u3/Z output (BUF_X1) (18, 636) Load pins - r2/D input (DFF_X1) 1.096-1.158 (19, 614) + r2/D input (DFF_X1) 1.096-1.158 (16, 629) Startpoint: r1 (rising edge-triggered flip-flop clocked by clk) Endpoint: r2 (rising edge-triggered flip-flop clocked by clk) @@ -542,23 +548,23 @@ Corner: slow 0.000 0.000 clock network delay (ideal) 0.000 0.000 ^ r1/CK (DFF_X1) 0.383 0.383 ^ r1/Q (DFF_X1) - 0.140 0.524 ^ u1/Z (BUF_X1) - 0.075 0.599 ^ bc1/u2/Z (BUF_X1) - 0.057 0.656 ^ bc1/u3/Z (BUF_X1) - 0.000 0.656 ^ r2/D (DFF_X1) - 0.656 data arrival time + 0.144 0.526 ^ u1/Z (BUF_X1) + 0.076 0.602 ^ bc1/u2/Z (BUF_X1) + 0.062 0.664 ^ bc1/u3/Z (BUF_X1) + 0.000 0.664 ^ r2/D (DFF_X1) + 0.664 data arrival time 0.300 0.300 clock clk (rise edge) 0.000 0.300 clock network delay (ideal) 0.000 0.300 clock reconvergence pessimism 0.300 ^ r2/CK (DFF_X1) - -0.072 0.228 library setup time - 0.228 data required time + -0.074 0.226 library setup time + 0.226 data required time ----------------------------------------------------------- - 0.228 data required time - -0.656 data arrival time + 0.226 data required time + -0.664 data arrival time ----------------------------------------------------------- - -0.428 slack (VIOLATED) + -0.438 slack (VIOLATED) Equivalence check - swap for rollback (inv_chain -> buffer_chain) @@ -595,11 +601,11 @@ Repair timing output passed/skipped equivalence test [INFO GPL-0005] ---- Execute Conjugate Gradient Initial Placement. [INFO GPL-0051] Source of initial instance position counters: Odb location = 0 Core center = 2 Region center = 0 -[InitialPlace] Iter: 1 conjugate gradient residual: 0.00000000 HPWL: 2892120 -[InitialPlace] Iter: 2 conjugate gradient residual: 0.00000000 HPWL: 404466 -[InitialPlace] Iter: 3 conjugate gradient residual: 0.00000000 HPWL: 397998 -[InitialPlace] Iter: 4 conjugate gradient residual: 0.00000000 HPWL: 395613 -[InitialPlace] Iter: 5 conjugate gradient residual: 0.00000000 HPWL: 395150 +[InitialPlace] Iter: 1 conjugate gradient residual: 0.00000000 HPWL: 2917430 +[InitialPlace] Iter: 2 conjugate gradient residual: 0.00000000 HPWL: 373668 +[InitialPlace] Iter: 3 conjugate gradient residual: 0.00000000 HPWL: 369133 +[InitialPlace] Iter: 4 conjugate gradient residual: 0.00000000 HPWL: 367975 +[InitialPlace] Iter: 5 conjugate gradient residual: 0.00000001 HPWL: 367942 [INFO GPL-0033] ---- Initialize Nesterov Region: Top-level [INFO GPL-0023] Placement target density: 0.0111 [INFO GPL-0024] Movable insts average area: 2.598 um^2 @@ -609,12 +615,13 @@ Repair timing output passed/skipped equivalence test [INFO GPL-0028] Bin count (X, Y): 2 , 32 [INFO GPL-0029] Bin size (W * H): 19.950 * 37.494 um [INFO GPL-0030] Number of bins: 64 +[INFO GPL-0333] Redistributed 185 filler cells into free bin capacity (484.138 um^2 available, 478.720 um^2 needed). [INFO GPL-0084] ---- Execute Nesterov Global Placement. [INFO GPL-0031] HPWL: Half-Perimeter Wirelength Iteration | Overflow | HPWL (um) | HPWL(%) | Penalty | Group --------------------------------------------------------------- - 0 | 0.0893 | 1.652730e+02 | +0.00% | 9.91e-13 | - 0 | 0.0893 | 1.652730e+02 | | 1.03e-12 | + 0 | 0.0994 | 1.519720e+02 | +0.00% | 5.81e-13 | + 0 | 0.0994 | 1.519720e+02 | | 6.04e-13 | --------------------------------------------------------------- [INFO GPL-1001] Global placement finished at iteration 0 [INFO GPL-1002] Placed Cell Area 54.5513 @@ -625,66 +632,8 @@ Iteration | Overflow | HPWL (um) | HPWL(%) | Penalty | Group [INFO GPL-1008] - For 80% usage of free space: 0.0014 [INFO GPL-1009] - For 50% usage of free space: 0.0023 [INFO GPL-1014] Final placement area: 54.55 (+0.00%) -[INFO GPL-1018] Final HPWL (um): 165.39 +[INFO GPL-1018] Final HPWL (um): 152.13 [INFO GPL-0133] Unlocking all instances -[INFO GPL-0084] ---- Execute Nesterov Global Placement. - 10 | 0.3890 | 3.256150e+01 | -80.30% | 1.33e-12 | - 20 | 0.3914 | 2.518450e+01 | -22.66% | 1.78e-12 | - 30 | 0.3933 | 2.295800e+01 | -8.84% | 2.37e-12 | - 40 | 0.3923 | 2.182350e+01 | -4.94% | 3.15e-12 | - 50 | 0.3922 | 2.158050e+01 | -1.11% | 4.20e-12 | - 60 | 0.3922 | 2.174000e+01 | +0.74% | 5.60e-12 | - 70 | 0.3922 | 2.203500e+01 | +1.36% | 7.46e-12 | - 80 | 0.3921 | 2.249150e+01 | +2.07% | 9.94e-12 | - 90 | 0.3921 | 2.313950e+01 | +2.88% | 1.32e-11 | - 100 | 0.3920 | 2.402950e+01 | +3.85% | 1.76e-11 | - 110 | 0.3922 | 2.520850e+01 | +4.91% | 2.35e-11 | - 120 | 0.3930 | 2.679950e+01 | +6.31% | 3.13e-11 | - 130 | 0.3941 | 2.891050e+01 | +7.88% | 4.17e-11 | - 140 | 0.3955 | 3.164900e+01 | +9.47% | 5.56e-11 | - 150 | 0.3973 | 3.511650e+01 | +10.96% | 7.40e-11 | - 160 | 0.3996 | 3.942950e+01 | +12.28% | 9.86e-11 | - 170 | 0.4025 | 4.455500e+01 | +13.00% | 1.31e-10 | - 180 | 0.4059 | 5.009900e+01 | +12.44% | 1.75e-10 | - 190 | 0.4094 | 5.540350e+01 | +10.59% | 2.33e-10 | - 200 | 0.4125 | 5.918550e+01 | +6.83% | 3.11e-10 | - 210 | 0.4153 | 5.972150e+01 | +0.91% | 4.14e-10 | - 220 | 0.4191 | 5.488800e+01 | -8.09% | 5.52e-10 | - 230 | 0.4239 | 4.993000e+01 | -9.03% | 7.35e-10 | - 240 | 0.4278 | 6.007900e+01 | +20.33% | 9.79e-10 | - 250 | 0.4289 | 6.846450e+01 | +13.96% | 1.30e-09 | - 260 | 0.4291 | 6.177550e+01 | -9.77% | 1.74e-09 | - 270 | 0.4321 | 6.650450e+01 | +7.66% | 2.32e-09 | - 280 | 0.4281 | 6.147250e+01 | -7.57% | 3.08e-09 | - 290 | 0.4211 | 6.807550e+01 | +10.74% | 4.11e-09 | - 300 | 0.4122 | 6.426600e+01 | -5.60% | 5.48e-09 | - 310 | 0.4012 | 6.509350e+01 | +1.29% | 7.29e-09 | - 320 | 0.4009 | 6.473600e+01 | -0.55% | 9.72e-09 | - 330 | 0.3999 | 6.447800e+01 | -0.40% | 1.29e-08 | - 340 | 0.3951 | 6.589050e+01 | +2.19% | 1.73e-08 | - 350 | 0.3817 | 6.983150e+01 | +5.98% | 2.30e-08 | - 360 | 0.3559 | 7.611650e+01 | +9.00% | 3.06e-08 | - 370 | 0.3192 | 8.099100e+01 | +6.40% | 4.08e-08 | - 380 | 0.2981 | 8.368050e+01 | +3.32% | 5.43e-08 | - 390 | 0.2676 | 9.158850e+01 | +9.45% | 7.24e-08 | - 400 | 0.2406 | 9.808950e+01 | +7.10% | 9.65e-08 | - 410 | 0.2095 | 1.064120e+02 | +8.48% | 1.29e-07 | - 420 | 0.1687 | 1.134220e+02 | +6.59% | 1.71e-07 | - 430 | 0.1350 | 1.224815e+02 | +7.99% | 2.28e-07 | - 440 | 0.1194 | 1.356325e+02 | +10.74% | 3.04e-07 | - 450 | 0.1028 | 1.440700e+02 | +6.22% | 4.05e-07 | - 452 | 0.0983 | 1.460590e+02 | | 4.41e-07 | ---------------------------------------------------------------- -[INFO GPL-1001] Global placement finished at iteration 452 -[INFO GPL-1002] Placed Cell Area 54.5513 -[INFO GPL-1003] Available Free Area 47872.0200 -[INFO GPL-1004] Minimum Feasible Density 0.0100 (cell_area / free_area) -[INFO GPL-1006] Suggested Target Densities: -[INFO GPL-1007] - For 90% usage of free space: 0.0013 -[INFO GPL-1008] - For 80% usage of free space: 0.0014 -[INFO GPL-1009] - For 50% usage of free space: 0.0023 -[INFO GPL-1014] Final placement area: 54.55 (+0.00%) -[INFO GPL-1018] Final HPWL (um): 146.18 [INFO DPL-0006] Core area: 47872.02 um^2 [INFO DPL-0007] Movable instances area: 59.85 um^2 [INFO DPL-0008] Fixed instances area within core: 0.00 um^2 @@ -700,19 +649,18 @@ Iteration | Overflow | HPWL (um) | HPWL(%) | Penalty | Group | Total | Illegal | Illegal Iteration | Violations | Cells | Sites --------------------------------------------- - 0 | 8 | 2 | 6 - 1 | 0 | 0 | 0 -Negotiation phase 1 converged at iteration 1. + 0 | 0 | 0 | 0 +Negotiation phase 1 converged at iteration 0. Detailed Placement Analysis --------------------------------- legalizer negotiation -total moves 22 -total displacement 30.6 u -average displacement 1.5 u -max displacement 3.8 u -original HPWL 146.2 u -legalized HPWL 161.7 u -delta HPWL 11 % +total moves 12 +total displacement 4.5 u +average displacement 0.2 u +max displacement 2.5 u +original HPWL 152.1 u +legalized HPWL 154.7 u +delta HPWL 2 % Cell type report for bc1 (inv_chain_bc1_1) Cell type report: Count Area @@ -727,13 +675,13 @@ Net u1z Number of pins: 5 Driver pins - u1/Z output (BUF_X1) (19, 603) + u1/Z output (BUF_X1) (19, 615) Load pins - bc1/u4/A input (INV_X1) 1.598-1.720 (21, 611) - bc2/u2/A input (BUF_X1) 0.906-0.983 (19, 610) - ic1/u4/A input (INV_X1) 1.598-1.720 (21, 613) - ic2/u4/A input (INV_X1) 1.598-1.720 (16, 608) + bc1/u4/A input (INV_X1) 1.598-1.720 (17, 628) + bc2/u2/A input (BUF_X1) 0.906-0.983 (16, 631) + ic1/u4/A input (INV_X1) 1.598-1.720 (19, 629) + ic2/u4/A input (INV_X1) 1.598-1.720 (19, 629) Net u3z Pin capacitance: 1.096-1.158 @@ -744,10 +692,10 @@ Net u3z Number of pins: 2 Driver pins - bc1/u5/ZN output (INV_X1) (17, 611) + bc1/u5/ZN output (INV_X1) (15, 629) Load pins - r2/D input (DFF_X1) 1.096-1.158 (19, 614) + r2/D input (DFF_X1) 1.096-1.158 (16, 629) Startpoint: r1 (rising edge-triggered flip-flop clocked by clk) Endpoint: r2 (rising edge-triggered flip-flop clocked by clk) @@ -761,9 +709,9 @@ Corner: slow 0.000 0.000 clock network delay (ideal) 0.000 0.000 ^ r1/CK (DFF_X1) 0.383 0.383 ^ r1/Q (DFF_X1) - 0.145 0.527 ^ u1/Z (BUF_X1) - 0.032 0.559 v bc1/u4/ZN (INV_X1) - 0.037 0.596 ^ bc1/u5/ZN (INV_X1) + 0.147 0.530 ^ u1/Z (BUF_X1) + 0.032 0.562 v bc1/u4/ZN (INV_X1) + 0.034 0.596 ^ bc1/u5/ZN (INV_X1) 0.000 0.596 ^ r2/D (DFF_X1) 0.596 data arrival time @@ -771,13 +719,13 @@ Corner: slow 0.000 0.300 clock network delay (ideal) 0.000 0.300 clock reconvergence pessimism 0.300 ^ r2/CK (DFF_X1) - -0.073 0.227 library setup time - 0.227 data required time + -0.072 0.228 library setup time + 0.228 data required time ----------------------------------------------------------- - 0.227 data required time + 0.228 data required time -0.596 data arrival time ----------------------------------------------------------- - -0.369 slack (VIOLATED) + -0.368 slack (VIOLATED) Equivalence check - redo swap (buffer_chain -> inv_chain)